[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"portal-settings:stajic:de":3,"public-menus:all":37,"post:what-is-rag-the-simplest-explanation-of-how-it-works:de":204,"related:post:what-is-rag-the-simplest-explanation-of-how-it-works:de:1":1794},{"statusCode":4,"data":5,"message":36},200,{"tenantId":6,"lang":7,"defaultLang":7,"siteUrl":8,"contactEmail":9,"brandName":10,"logoUrl":11,"siteName":10,"siteDescription":12,"ogImage":9,"robotsIndex":13,"socialLinks":9,"reservedSlugs":9,"seoPolicy":14},"stajic","de","https:\u002F\u002Fstajic.de",null,"Stajic Platform","\u002FLogo_Planet.svg","Stajic Portal",true,{"branding":15,"relatedContent":16,"crossDomainLinks":17},{"logoUrl":11},{"enabled":13},[18,21,24,27,30,33],{"url":19,"label":20,"isActive":13,"showInFooter":13,"includeInSameAs":13},"https:\u002F\u002Ffigure.rocks","figure.rocks",{"url":22,"label":23,"isActive":13,"showInFooter":13,"includeInSameAs":13},"https:\u002F\u002Floving.rocks","loving.rocks",{"url":25,"label":26,"isActive":13,"showInFooter":13,"includeInSameAs":13},"https:\u002F\u002Fbazify.com","bazify.com",{"url":28,"label":29,"isActive":13,"showInFooter":13,"includeInSameAs":13},"https:\u002F\u002Fbazify.de","bazify.de",{"url":31,"label":32,"isActive":13,"showInFooter":13,"includeInSameAs":13},"https:\u002F\u002Fbazify.at","bazify.at",{"url":34,"label":35,"isActive":13,"showInFooter":13,"includeInSameAs":13},"https:\u002F\u002Fbazify.ba","bazify.ba","Portal settings resolved",[38,44],{"id":39,"name":40,"location":41,"isActive":13,"isDefault":42,"items":43},1,"main-navigation","header",false,[],{"id":45,"name":46,"location":47,"isActive":13,"isDefault":13,"items":48},4,"main-menu","sidebar",[49,65,78,92,102,117,132],{"id":50,"title":51,"url":59,"target":60,"icon":61,"isActive":13,"type":62,"productId":9,"categoryId":9,"shopCategoryId":9,"articleId":9,"pageId":63,"portfolioId":9,"children":64},"item-18",{"de":52,"en":53,"es":54,"fr":55,"it":53,"ru":56,"sr":57,"zh":58},"Startseite","Home","Inicio","Accueil","Главная","Почетна","首页","\u002Ffull-stack-web-developer-munich-performance-seo-and-maintainable-builds","_self","i-lucide-home","page",111,[],{"id":66,"title":67,"url":74,"target":60,"icon":75,"isActive":13,"type":62,"productId":9,"categoryId":9,"shopCategoryId":9,"articleId":9,"pageId":76,"portfolioId":9,"children":77},"item-22",{"de":68,"en":68,"es":69,"fr":68,"it":70,"ru":71,"sr":72,"zh":73},"Vision","Visión","Visione","Видение","Визија","想象","\u002Fueber-uns-webdesign-muenchen-webaplikation","i-lucide-eye",113,[],{"id":79,"title":80,"url":88,"target":60,"icon":89,"isActive":13,"type":62,"productId":9,"categoryId":9,"shopCategoryId":9,"articleId":9,"pageId":90,"portfolioId":9,"children":91},"item-19",{"de":81,"en":82,"es":83,"fr":82,"it":84,"ru":85,"sr":86,"zh":87},"Leistungen","Services","Servicios","Servizi","Услуги","Услуге","服务","\u002Fservices-dienstleistungen-muenchen","i-lucide-wrench",116,[],{"id":93,"title":94,"url":98,"target":60,"icon":99,"isActive":13,"type":62,"productId":9,"categoryId":9,"shopCategoryId":9,"articleId":9,"pageId":100,"portfolioId":9,"children":101},"item-23",{"de":95,"en":95,"es":95,"fr":95,"it":95,"ru":96,"sr":96,"zh":97},"Blog","Блог","博客","\u002Fblog","i-lucide-book-open",112,[],{"id":103,"title":104,"url":113,"target":60,"icon":114,"isActive":13,"type":62,"productId":9,"categoryId":9,"shopCategoryId":9,"articleId":9,"pageId":115,"portfolioId":9,"children":116},"item-32",{"de":105,"en":106,"es":107,"fr":108,"it":109,"ru":110,"sr":111,"zh":112},"Neue Technologien","New Technologies","Nuevas tecnologías","Nouvelles technologies","Nuove tecnologie","Новые технологии","Нове технологије","新技术！","\u002Fneue-webtechnologien","i-lucide-sparkles",122,[],{"id":118,"title":119,"url":128,"target":60,"icon":129,"isActive":13,"type":62,"productId":9,"categoryId":9,"shopCategoryId":9,"articleId":9,"pageId":130,"portfolioId":9,"children":131},"item-20",{"de":120,"en":121,"es":122,"fr":123,"it":124,"ru":125,"sr":126,"zh":127},"Kontakt","Contact us!","Contacto","Contact","Contatto","Контакт","Контактирајте нас","联系我们！","\u002Fcontact","i-lucide-mail",115,[],{"id":133,"title":134,"url":143,"target":60,"icon":144,"isActive":13,"type":62,"productId":9,"categoryId":9,"shopCategoryId":9,"articleId":9,"pageId":145,"portfolioId":9,"children":146},"item-21",{"de":135,"en":136,"es":137,"fr":138,"it":139,"ru":140,"sr":141,"zh":142},"Unsere Arbeit","Our Work","Nuestro trabajo","Nos réalisations","I nostri lavori","Наши работы","Наши радови","文件夹","\u002Fportfolio","i-lucide-briefcase",114,[147,160,174,180,192],{"id":148,"title":149,"url":143,"target":60,"icon":158,"isActive":13,"type":62,"productId":9,"categoryId":9,"shopCategoryId":9,"articleId":9,"pageId":145,"portfolioId":9,"children":159},"item-24",{"de":150,"en":151,"es":152,"fr":153,"it":154,"ru":155,"sr":156,"zh":157},"Alle Projekte","All Projects","Todos los proyectos","Tous les projets","Tutti i progetti","Все проекты","Сви пројекти","所有项目","i-lucide-grid-3x3",[],{"id":161,"title":162,"url":170,"target":60,"icon":171,"isActive":13,"type":172,"productId":9,"categoryId":9,"shopCategoryId":9,"articleId":9,"pageId":9,"portfolioId":9,"children":173},"item-29",{"de":163,"en":164,"es":165,"fr":166,"it":167,"ru":168,"sr":169,"zh":142},"Local Roots, Global Reach","Local Roots - Global Reach","Empresa local ","Entreprise locale","Azienda locale","Местная компания","Локално предузеће глобално тржиште","\u002Fportfolio\u002Flocal-roots-global-reach-communication-media-systems-for-modern-business","i-lucide-folder","custom",[],{"id":175,"title":176,"url":178,"target":60,"icon":171,"isActive":13,"type":172,"productId":9,"categoryId":9,"shopCategoryId":9,"articleId":9,"pageId":9,"portfolioId":9,"children":179},"item-28",{"de":177,"en":177,"es":177,"fr":177,"it":177,"ru":177,"sr":177,"zh":177},"Solr Suggester","\u002Fportfolio\u002Fsolr-fuzzy-suggester-und-solr-infix-suggester-abfrage-ueber-ajax-und-filterung",[],{"id":181,"title":182,"url":190,"target":60,"icon":171,"isActive":13,"type":172,"productId":9,"categoryId":9,"shopCategoryId":9,"articleId":9,"pageId":9,"portfolioId":9,"children":191},"item-27",{"de":183,"en":184,"es":185,"fr":186,"it":187,"ru":188,"sr":189,"zh":184},"Firmenwebseite SEO","Company Website SEO","Sitio web corporativo SEO","Site web d’entreprise SEO","Sito web aziendale SEO","Корпоративный сайт SEO","Пословна веб-страница SEO","\u002Fportfolio\u002Fseo-sem-branding-mobile-webseite-muenchen",[],{"id":193,"title":194,"url":202,"target":60,"icon":171,"isActive":13,"type":172,"productId":9,"categoryId":9,"shopCategoryId":9,"articleId":9,"pageId":9,"portfolioId":9,"children":203},"item-31",{"de":195,"en":196,"es":197,"fr":198,"it":199,"ru":200,"sr":201,"zh":196},"Digitalisierungsportal","Digitalization Portal","Portal de digitalización","Portail de numérisation","Portale di digitalizzazione","Портал цифровизации","Портал за дигитализацију","\u002Fportfolio\u002Fdigitalisierungsportal-archiv-museum-bibliothek-ead-lido-mets-mods",[],{"statusCode":4,"data":205,"message":1793},{"id":206,"title":207,"slug":208,"content":209,"contentJson":210,"excerpt":873,"featuredImage":874,"featuredImageAlt":875,"featuredImageCaption":9,"featuredImageTitle":9,"featuredImageCopyright":9,"featuredImageAuthor":9,"featuredImageSourceUrl":9,"featuredImageLicense":9,"featuredImageIsAiGenerated":42,"status":876,"publishedAt":877,"createdAt":878,"updatedAt":879,"seoLocalePaths":880,"categories":889,"author":914,"translations":919},"478","Was ist RAG? Die einfachste Erklärung, wie es funktioniert","what-is-rag-the-simplest-explanation-of-how-it-works","\u003Cp>RAG klingt kompliziert, weil der Name kompliziert ist. Die Idee ist es nicht. RAG bedeutet einfach: Bevor die KI antwortet, sucht sie zuerst nach relevanten Informationen aus einer Wissensquelle und gibt diese Informationen an das Sprachmodell weiter.\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--info my-6 rounded-xl border p-5 border-blue-300 bg-blue-50 dark:border-blue-900 dark:bg-blue-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">RAG in einem Satz\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">\u003Cstrong>RAG ist der Schritt, bei dem eine KI eine Wissensdatenbank nach nützlichen Informationen durchsucht, bevor das LLM die Antwort schreibt.\u003C\u002Fstrong>\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Cp>Stellen Sie sich ein LLM wie eine kluge Person vor, die an einem Schreibtisch sitzt. RAG ist der Bibliothekar, der die richtige Seite aus dem richtigen Buch bringt. Das LLM liest dann diese Seite und antwortet Ihnen.\u003C\u002Fp>\n\u003Cnav class=\"editorjs-toc\" data-editorjs-toc=\"true\" aria-label=\"Inhalt\">\u003Cstrong class=\"editorjs-toc__title\">Inhalt\u003C\u002Fstrong>\u003Col class=\"editorjs-toc__list editorjs-toc__list--depth-0\">\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-5\" class=\"editorjs-toc__link\">Zuerst: Was macht das LLM?\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-10\" class=\"editorjs-toc__link\">Dann: Was ist die Wissensdatenbank?\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-15\" class=\"editorjs-toc__link\">Was macht RAG also tatsächlich?\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-19\" class=\"editorjs-toc__link\">Ein sehr einfaches Beispiel\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-25\" class=\"editorjs-toc__link\">Nun der wichtige Teil: RAG ist nicht der aktuelle Zustand\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-30\" class=\"editorjs-toc__link\">Was ist eine Zustandsdatenbank?\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-36\" class=\"editorjs-toc__link\">Wie die drei Teile zusammenwirken\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-40\" class=\"editorjs-toc__link\">Verwendet RAG immer eine Vektordatenbank?\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-45\" class=\"editorjs-toc__link\">Was ist ein Embedding, einfach erklärt?\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-50\" class=\"editorjs-toc__link\">RAG ist auch kein Gedächtnis\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-54\" class=\"editorjs-toc__link\">Ein echtes Spielbeispiel: PUBG Ally\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-60\" class=\"editorjs-toc__link\">Ein vollständiges Beispiel\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-64\" class=\"editorjs-toc__link\">Warum überhaupt RAG verwenden?\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-68\" class=\"editorjs-toc__link\">Was RAG nicht garantiert\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-72\" class=\"editorjs-toc__link\">Das einfachste mentale Modell zum Merken\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-75\" class=\"editorjs-toc__link\">Fazit\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-79\" class=\"editorjs-toc__link\">FAQ\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-81\" class=\"editorjs-toc__link\">Glossar\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-83\" class=\"editorjs-toc__link\">Primärquellen\u003C\u002Fa>\u003C\u002Fli>\u003C\u002Fol>\u003C\u002Fnav>\n\u003Ch2 id=\"section-5\">Zuerst: Was macht das LLM?\u003C\u002Fh2>\n\u003Cp>Das LLM ist der Teil, der Sprache versteht und Sprache erzeugt. Es kann Ihre Frage lesen, Anweisungen verstehen, Informationen vergleichen, etwas erklären und eine Antwort schreiben.\u003C\u002Fp>\n\u003Cp>Aber das LLM weiß nicht automatisch, was sich gerade in Ihrer Unternehmensdatenbank, Ihrer Spielsitzung, Ihren privaten Dokumenten oder einer Datei befindet, die Sie vor fünf Minuten erstellt haben.\u003C\u002Fp>\n\u003Cp>Es weiß nur, was bereits im Modell enthalten ist, plus alle Informationen, die die Anwendung ihm in der aktuellen Anfrage gibt.\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--note my-6 rounded-xl border p-5 border-gray-300 bg-gray-50 dark:border-gray-700 dark:bg-gray-900\u002F40\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">Einfache Regel\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">Das LLM \u003Cstrong>denkt und schreibt\u003C\u002Fstrong>. Es besitzt nicht automatisch alle Ihre aktuellen Daten.\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-10\">Dann: Was ist die Wissensdatenbank?\u003C\u002Fh2>\n\u003Cp>Eine Wissensdatenbank ist einfach eine Information, die die Anwendung durchsuchen kann.\u003C\u002Fp>\n\u003Cp>Sie könnte PDFs, Handbücher, Produktdokumentationen, Support-Artikel, Verträge, Spielregeln, Waffendaten, interne Unternehmensdokumente, Datenbankeinträge oder anderen Text enthalten.\u003C\u002Fp>\n\u003Cp>Die Wissensdatenbank kann lokal auf Ihrem eigenen Rechner sein. Sie kann auf einem Server sein. Sie kann in einer Vektordatenbank sein. Sie kann auch aus normalen Dateien erstellt werden. RAG bedeutet nicht Internet.\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--success my-6 rounded-xl border p-5 border-emerald-300 bg-emerald-50 dark:border-emerald-900 dark:bg-emerald-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">Wichtig\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">\u003Cstrong>RAG erfordert kein Internet.\u003C\u002Fstrong> Die Informationen können vollständig lokal sein.\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-15\">Was macht RAG also tatsächlich?\u003C\u002Fh2>\n\u003Csection class=\"editorjs-process my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">Der gesamte RAG-Prozess\u003C\u002Fh3>\u003Cdiv class=\"flex flex-col sm:flex-row gap-3\">\u003Cdiv class=\"editorjs-process__step min-w-0 flex-1 rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">1\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">1. Sie stellen eine Frage\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Zum Beispiel: Welche Munition verwendet diese Waffe?\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__arrow shrink-0 self-center text-xl text-gray-400 rotate-90 sm:rotate-0\" aria-hidden=\"true\">→\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0 flex-1 rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">2\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">2. RAG durchsucht die Wissensdatenbank\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Das System sucht nach den kleinen Informationsstücken, die für Ihre Frage am relevantesten sind.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__arrow shrink-0 self-center text-xl text-gray-400 rotate-90 sm:rotate-0\" aria-hidden=\"true\">→\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0 flex-1 rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">3\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">3. RAG gibt diese Stücke an das LLM\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Das LLM erhält die Frage plus die abgerufenen Informationen.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__arrow shrink-0 self-center text-xl text-gray-400 rotate-90 sm:rotate-0\" aria-hidden=\"true\">→\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0 flex-1 rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">4\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">4. Das LLM schreibt die Antwort\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Es verwendet die abgerufenen Informationen als Kontext für die Antwort.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Cp>Das ist RAG.\u003C\u002Fp>\n\u003Cp>Der vollständige Name ist Retrieval-Augmented Generation. Retrieval bedeutet, die relevanten Informationen zu finden. Augmented bedeutet, diese Informationen zum Kontext des Modells hinzuzufügen. Generation bedeutet, dass das LLM die endgültige Antwort schreibt.\u003C\u002Fp>\n\u003Ch2 id=\"section-19\">Ein sehr einfaches Beispiel\u003C\u002Fh2>\n\u003Cp>Stellen Sie sich vor, Sie haben eine lokale Wissensdatenbank über ein Spiel.\u003C\u002Fp>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Wissensdatenbank enthält\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Beispiel\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Waffen\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">AKM verwendet 7,62-mm-Munition\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Heilgegenstände\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Med Kit stellt Gesundheit wieder her\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Anbauteile\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Dieses Anbauteil funktioniert mit diesen Waffen\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Kartenregeln\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Diese Zone verhält sich auf diese Weise\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Cp>Sie fragen: „Welche Munition verwendet die AKM?“\u003C\u002Fp>\n\u003Cp>RAG durchsucht die Wissensdatenbank und findet den Eintrag über die AKM. Es gibt dieses kleine Stück Information an das LLM weiter. Das LLM antwortet dann: „Die AKM verwendet 7,62-mm-Munition.“\u003C\u002Fp>\n\u003Cp>Das LLM brauchte nicht die gesamte Datenbank. RAG brachte nur den nützlichen Teil.\u003C\u002Fp>\n\u003Ch2 id=\"section-25\">Nun der wichtige Teil: RAG ist nicht der aktuelle Zustand\u003C\u002Fh2>\n\u003Cp>Hier werden viele Erklärungen verwirrend.\u003C\u002Fp>\n\u003Cp>RAG gibt der KI normalerweise Wissen. Ein Zustandssystem gibt der KI Fakten darüber, was gerade jetzt wahr ist.\u003C\u002Fp>\n\u003Csection class=\"editorjs-comparison my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">Wissen vs. aktueller Zustand\u003C\u002Fh3>\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left dark:border-gray-700 dark:bg-gray-900\">\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">RAG \u002F Wissen\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">Aktueller Zustand\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Waffe\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Munition\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Gesundheit\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Gegner\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Caside class=\"editorjs-callout editorjs-callout--warning my-6 rounded-xl border p-5 border-amber-300 bg-amber-50 dark:border-amber-900 dark:bg-amber-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">Diese beiden nicht vermischen\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">RAG antwortet: \u003Cstrong>Was ist allgemein wahr?\u003C\u002Fstrong>\u003Cbr>Zustand antwortet: \u003Cstrong>Was ist gerade jetzt wahr?\u003C\u002Fstrong>\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-30\">Was ist eine Zustandsdatenbank?\u003C\u002Fh2>\n\u003Cp>Eine Zustandsdatenbank oder ein Zustandsspeicher ist einfach ein Ort, an dem die Anwendung aktuelle Fakten aufbewahrt.\u003C\u002Fp>\n\u003Cp>In einem Spiel kennt die Engine bereits Dinge wie Ihre Gesundheit, Position, Inventar, Munition, aktuelle Mission, nahe Objekte und Gegnerstatus. Ein KI-System kann ausgewählte Teile dieses Zustands dem Modell zugänglich machen.\u003C\u002Fp>\n\u003Cp>In einer Geschäftsanwendung könnte dieselbe Idee eine Bestelldatenbank, ein Kundendatensatz, ein Projektstatus oder der aktuelle Wert eines Sensors sein.\u003C\u002Fp>\n\u003Cp>Der Zustand wird von der Anwendung selbst erstellt, während Dinge geschehen. Wenn Sie Gesundheit verlieren, aktualisiert das Spiel den Gesundheitswert. Wenn Sie Munition aufnehmen, ändert sich das Inventar. Wenn eine Bestellung bezahlt wird, ändert das Geschäftssystem den Bestellstatus.\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--info my-6 rounded-xl border p-5 border-blue-300 bg-blue-50 dark:border-blue-900 dark:bg-blue-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">Einfache Regel\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">Die Anwendung erstellt und aktualisiert \u003Cstrong>Zustand\u003C\u002Fstrong>. RAG durchsucht \u003Cstrong>Wissen\u003C\u002Fstrong>. Das LLM verwendet beides, um zu entscheiden, was es sagen oder tun soll.\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-36\">Wie die drei Teile zusammenwirken\u003C\u002Fh2>\n\u003Csection class=\"editorjs-process my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">LLM + Zustand + RAG\u003C\u002Fh3>\u003Cdiv class=\"grid grid-cols-1 md:grid-cols-2 xl:grid-cols-3 gap-4\">\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">1\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">1. Aktueller Zustand\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Die Anwendung teilt der KI mit, was jetzt gilt: Gesundheit 41 %, AKM ausgerüstet, 23 Schuss.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">2\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">2. RAG\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Das System ruft nützliches Wissen ab: wie die Waffe funktioniert, welches Heilmittel verfügbar ist oder eine relevante Regel.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">3\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">3. LLM\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Das Modell erhält die Frage, den aktuellen Zustand und das abgerufene Wissen.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">4\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">4. Schlussfolgerung\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Das LLM kombiniert diese Eingaben und entscheidet, welche Antwort oder übergeordnete Aktion sinnvoll ist.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">5\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">5. Anwendung\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Wenn eine Aktion erforderlich ist, führt die Anwendung oder die Spiel-Engine sie aus und aktualisiert den Zustand erneut.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Cp>Die grundlegende Architektur ist also:\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--note my-6 rounded-xl border p-5 border-gray-300 bg-gray-50 dark:border-gray-700 dark:bg-gray-900\u002F40\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">Die einfachste Architektur\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">\u003Cstrong>Zustand = was jetzt gilt\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>RAG = nützliches Wissen\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>LLM = versteht, schlussfolgert und schreibt\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Anwendung = führt die reale Aktion aus\u003C\u002Fstrong>\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-40\">Verwendet RAG immer eine Vektordatenbank?\u003C\u002Fh2>\n\u003Cp>Nein.\u003C\u002Fp>\n\u003Cp>Eine Vektordatenbank ist eine gängige Methode, um semantische Suche aufzubauen, aber sie ist nicht die Definition von RAG.\u003C\u002Fp>\n\u003Cp>Der wichtige Teil ist das Abrufen: Das System findet relevante externe Informationen und fügt sie dem Kontext des LLM hinzu, bevor die Antwort generiert wird.\u003C\u002Fp>\n\u003Cp>OpenAIs File Search kann beispielsweise mit Dateien arbeiten, die in Vektorspeichern abgelegt sind. Dateien werden in kleinere Stücke aufgeteilt, damit das System die Teile abrufen kann, die für eine Frage relevant sind. Das ist eine Umsetzung derselben Grundidee.\u003C\u002Fp>\n\u003Ch2 id=\"section-45\">Was ist ein Embedding, einfach erklärt?\u003C\u002Fh2>\n\u003Cp>Du musst Embeddings nicht verstehen, um RAG zu verstehen.\u003C\u002Fp>\n\u003Cp>Aber die einfache Version ist diese: Ein Embedding ist eine numerische Darstellung von Bedeutung. Es hilft einem Suchsystem, Text zu finden, der konzeptionell ähnlich ist, auch wenn die Wörter nicht genau dieselben sind.\u003C\u002Fp>\n\u003Cp>Zum Beispiel sucht eine normale Stichwortsuche möglicherweise nach den genauen Wörtern „Autoreparatur“. Die semantische Suche kann auch verstehen, dass „repariere mein Fahrzeug“ ein ähnliches Thema betrifft.\u003C\u002Fp>\n\u003Cp>Das macht Embeddings für RAG nützlich, aber RAG kann auch Stichwortsuche, Datenbankabfragen oder eine Mischung aus mehreren Methoden verwenden.\u003C\u002Fp>\n\u003Ch2 id=\"section-50\">RAG ist auch kein Gedächtnis\u003C\u002Fh2>\n\u003Cp>Gedächtnis ist ein weiteres Konzept, das oft mit RAG vermischt wird.\u003C\u002Fp>\n\u003Cp>Gedächtnis sind normalerweise Informationen, die das System über frühere Interaktionen oder frühere Ereignisse behält. RAG ist der Mechanismus, mit dem relevantes Wissen abgerufen wird, wenn es benötigt wird.\u003C\u002Fp>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Teil\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Einfache Bedeutung\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">LLM\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Der Teil, der Sprache versteht und generiert\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">RAG\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Der Teil, der vor der Antwort relevantes Wissen nachschlägt\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Wissensbasis\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Die Informationen, die RAG durchsuchen kann\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Zustand\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Was gerade jetzt in der Anwendung oder Welt gilt\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Gedächtnis\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Informationen, die aus früheren Interaktionen oder Ereignissen behalten werden\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Werkzeug \u002F Aktion\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Etwas, das die KI aufrufen oder die Anwendung bitten darf zu tun\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Kontext\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Die Informationen, die dem LLM für diese Anfrage gerade vorgelegt werden\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-54\">Ein echtes Spielbeispiel: PUBG Ally\u003C\u002Fh2>\n\u003Cp>PUBG Ally ist ein nützliches Beispiel, weil es den Unterschied sichtbar macht.\u003C\u002Fp>\n\u003Cp>KRAFTON beschreibt den Live-Match-Zustand als separate Quelle der Wahrheit. Das Spiel gibt aktuelle Fakten über Beobachtungswerkzeuge preis: aktuelle Waffe, Munition, Gesundheit, Status der Sicherheitszone, Gegenstände in der Nähe und Kampfsituation.\u003C\u002Fp>\n\u003Cp>Die Wissenssuche ist eine andere Aufgabe. Das System kann kuratiertes Wissen über Waffen, Aufsätze, Gegenstände und Regeln nutzen. Das ACE Game Agent SDK von NVIDIA bietet außerdem eine separate RAG-API zum Abrufen von Wissen aus von Entwicklern erstellten Datenbanken.\u003C\u002Fp>\n\u003Cp>Das gibt uns die klare Trennung: Die Spiel-Engine sagt, was gerade passiert, der Abruf liefert relevantes Wissen, und das Sprachmodell entscheidet, was die Informationen bedeuten.\u003C\u002Fp>\n\u003Caside class=\"editorjs-referral my-6\">\u003Ca href=\"https:\u002F\u002Ffigure.rocks\u002Fblog\u002Fpubg-ally-shows-why-ai-teammates-need-two-brains-fast-reflexes-and-slow-reasoning\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"flex flex-col sm:flex-row gap-4 rounded-xl border border-gray-200 dark:border-gray-700 p-4 transition hover:border-primary-500\">\u003Cdiv class=\"min-w-0 flex-1\">\u003Cstrong class=\"block text-lg text-gray-900 dark:text-gray-100\">PUBG Ally zeigt, warum KI-Teamkollegen zwei Gehirne brauchen: schnelle Reflexe und langsames Denken\u003C\u002Fstrong>\u003Cp class=\"mt-2 text-sm text-gray-600 dark:text-gray-300\">Ein praktisches Spielbeispiel, das zeigt, wie Live-Zustand, Sprachschlussfolgerung und deterministische spielseitige Steuerung zusammenwirken können.\u003C\u002Fp>\u003Cspan class=\"mt-3 inline-flex text-sm font-medium text-primary-600 dark:text-primary-400\">Den Artikel zur PUBG-Ally-Architektur lesen →\u003C\u002Fspan>\u003C\u002Fdiv>\u003C\u002Fa>\u003C\u002Faside>\n\u003Ch2 id=\"section-60\">Ein vollständiges Beispiel\u003C\u002Fh2>\n\u003Cp>Stellen Sie sich vor, Sie sagen einem KI-Teamkollegen: „Ich habe wenig Gesundheit. Sollten wir angreifen?“\u003C\u002Fp>\n\u003Csection class=\"editorjs-process my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">Was als Nächstes passiert\u003C\u002Fh3>\u003Cdiv class=\"grid grid-cols-1 md:grid-cols-2 xl:grid-cols-3 gap-4\">\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">1\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">Zustand\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Das Spiel meldet: Gesundheit 24 %, ein Gegner in der Nähe, zwei Heilgegenstände verfügbar.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">2\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">RAG\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Das Wissenssystem ruft die relevanten Regeln für den Heilgegenstand und möglicherweise Informationen über die aktuelle Waffe oder taktische Mechanik ab.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">3\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">LLM\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Das Modell kombiniert Ihre Anfrage, den aktuellen Zustand und das abgerufene Wissen.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">4\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">Entscheidung\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Es kommt zu dem Schluss, dass zuerst zu heilen sicherer ist, als sofort anzugreifen.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">5\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">Werkzeug \u002F Spiel-Engine\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Der Agent fordert eine legale Spielaktion an, z. B. sich in Deckung zu bewegen oder den Heilgegenstand zu verwenden.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">6\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">Neuer Zustand\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Das Spiel führt die Aktion aus und meldet die aktualisierte Situation an den Agenten zurück.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Cp>RAG hat die Figur nicht gesteuert. Die Zustandsdatenbank hat nicht geschlussfolgert. Das LLM hat das Spiel nicht direkt verändert. Jeder Teil hatte eine Aufgabe.\u003C\u002Fp>\n\u003Ch2 id=\"section-64\">Warum überhaupt RAG verwenden?\u003C\u002Fh2>\n\u003Cp>Weil es langsam, teuer und oft verwirrend wäre, jedes Dokument, jede Regel und jeden Datenbankeintrag in jeden Prompt aufzunehmen.\u003C\u002Fp>\n\u003Cp>RAG ermöglicht es dem System, nur die Informationen auszuwählen, die für die aktuelle Frage nützlich sind.\u003C\u002Fp>\n\u003Cp>Es ermöglicht außerdem, die Wissensbasis zu aktualisieren, ohne das gesamte Sprachmodell neu zu trainieren. Ändern Sie das Dokument oder die Datenbank, bauen Sie den Index bei Bedarf neu auf oder aktualisieren Sie ihn, und der nächste Abruf kann die neueren Informationen verwenden.\u003C\u002Fp>\n\u003Ch2 id=\"section-68\">Was RAG nicht garantiert\u003C\u002Fh2>\n\u003Cp>RAG kann die Fundierung verbessern, aber es macht eine Antwort nicht automatisch korrekt.\u003C\u002Fp>\n\u003Cp>Der Abrufschritt kann das falsche Dokument finden. Das richtige Dokument kann veraltet sein. Das LLM kann gute Belege missverstehen. Oder der aktuelle Zustand kann sich geändert haben.\u003C\u002Fp>\n\u003Cp>Ein zuverlässiges System muss daher den Abruf, die Aktualität des Zustands und die endgültige Schlussfolgerung des Modells getrennt validieren.\u003C\u002Fp>\n\u003Ch2 id=\"section-72\">Das einfachste mentale Modell zum Merken\u003C\u002Fh2>\n\u003Csection class=\"editorjs-comparison my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">Stellen Sie sich ein KI-System wie eine Person an einem Schreibtisch vor\u003C\u002Fh3>\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left dark:border-gray-700 dark:bg-gray-900\">\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">Analogie\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">KI-System\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Denkende Person\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Ein Nachschlagewerk finden\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Bücher im Regal\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Aktuelles Dashboard oder Instrumententafel\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Notizen von früheren Besprechungen\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Etwas in der realen Welt tun\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Caside class=\"editorjs-callout editorjs-callout--success my-6 rounded-xl border p-5 border-emerald-300 bg-emerald-50 dark:border-emerald-900 dark:bg-emerald-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">Wenn Sie sich nur das merken\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">\u003Cstrong>LLM = Gehirn.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>RAG = Bibliothekar.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Wissensbasis = Bibliothek.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Zustand = was das Dashboard gerade jetzt anzeigt.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Werkzeuge = die Hände, die tatsächlich etwas tun können.\u003C\u002Fstrong>\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-75\">Fazit\u003C\u002Fh2>\n\u003Cp>RAG ist viel weniger mysteriös, sobald die Teile getrennt sind.\u003C\u002Fp>\n\u003Cp>Das LLM versteht und erzeugt Sprache. Die Anwendung verwaltet den aktuellen Zustand. Die Wissensbasis speichert Informationen. RAG findet den nützlichen Teil dieser Informationen und fügt ihn in den Kontext des LLM ein. Werkzeuge oder die Anwendung führen reale Aktionen aus.\u003C\u002Fp>\n\u003Cp>Das ist die grundlegende Architektur hinter vielen modernen KI-Assistenten und Agenten.\u003C\u002Fp>\n\u003Ch2 id=\"section-79\">FAQ\u003C\u002Fh2>\n\u003Csection class=\"editorjs-faq my-6 rounded-xl border border-gray-200 p-5 dark:border-gray-700\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">RAG in einfachen Worten\u003C\u002Fh3>\u003Cdiv id=\"faq1\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">Was ist RAG in einfachen Worten?\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">RAG ist ein Schritt, bei dem eine KI eine Wissensquelle nach relevanten Informationen durchsucht, bevor das Sprachmodell seine Antwort schreibt.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq2\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">Braucht RAG das Internet?\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">Nein. Die Wissensbasis kann vollständig lokal auf Ihrem Computer oder Server liegen.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq3\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">Ist RAG dasselbe wie eine Datenbank?\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">Nein. Die Datenbank oder Dateien enthalten die Informationen. RAG ist der Abrufprozess, der den nützlichen Teil findet und ihn dem LLM gibt.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq4\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">Ist RAG dasselbe wie Gedächtnis?\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">Nein. Das Gedächtnis speichert normalerweise frühere Interaktionen oder Ereignisse. RAG ruft relevantes Wissen ab, wenn es benötigt wird.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq5\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">Ist der aktuelle Anwendungszustand Teil von RAG?\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">Nicht unbedingt. Der aktuelle Zustand wird normalerweise direkt von der Anwendung oder einem Zustandsspeicher bezogen. RAG wird besser als Abruf aus einer Wissensquelle verstanden.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq6\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">Macht RAG KI-Antworten korrekt?\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">Nein. Es kann bessere Belege liefern, aber der Abruf kann immer noch falsch oder veraltet sein und das LLM kann immer noch falsch schlussfolgern.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-81\">Glossar\u003C\u002Fh2>\n\u003Csection class=\"editorjs-glossary my-6 rounded-xl border border-gray-200 dark:border-gray-700 p-5\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">Die grundlegenden Begriffe\u003C\u002Fh3>\u003Cdl>\u003Cdiv id=\"llm\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">LLM\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Ein Sprachmodell, das Text versteht und erzeugt und über Informationen, die in seinen Kontext gestellt werden, schlussfolgern kann.\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"rag\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">RAG\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Retrieval-Augmented Generation: Abrufen relevanter externer Informationen und Hinzufügen zum Kontext des Modells, bevor eine Antwort generiert wird.\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"knowledge-base\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">Wissensbasis\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Die Dateien, Dokumente, Datensätze oder anderen Informationen, die der Abruf durchsuchen kann.\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"state\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">Zustand\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Die aktuellen Fakten einer Anwendung, eines Systems oder der Welt zu einem bestimmten Zeitpunkt.\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"context\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">Kontext\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Die Informationen, die dem Sprachmodell derzeit für eine Anfrage oder einen Schlussfolgerungsschritt bereitgestellt werden.\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"embedding\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">Embedding\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Eine numerische Darstellung von Bedeutung, die der semantischen Suche helfen kann, konzeptionell ähnliche Informationen zu finden.\u003C\u002Fdd>\u003C\u002Fdiv>\u003C\u002Fdl>\u003C\u002Fsection>\n\u003Ch2 id=\"section-83\">Primärquellen\u003C\u002Fh2>\n\u003Ca href=\"https:\u002F\u002Fplatform.openai.com\u002Fdocs\u002Fapi-reference\u002Fvector-stores-files\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">OpenAI — Vector Store Files\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">Offizielle Dokumentation, die zeigt, wie Dateien an Vektorspeicher angehängt, in Chunks aufgeteilt und für den Dateisuche-Abruf verfügbar gemacht werden können.\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fplatform.openai.com\u002Fdocs\u002Fquickstart\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">OpenAI — Developer Quickstart\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">Offizielle OpenAI-Dokumentation, die Werkzeuge wie die Dateisuche beschreibt, um Modellen Zugriff auf externe Informationen zu geben.\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fdeveloper.nvidia.com\u002Face-for-games\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">NVIDIA Developer — ACE for Games\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">Offizielle NVIDIA-Dokumentation, die separate Agent-, Chat- und RAG-APIs beschreibt, um Spielcharaktere mit Spielzustand, kontextuellem Wissen und modellgesteuerten Aktionen zu verbinden.\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fdeveloper.nvidia.com\u002Fblog\u002Fhow-krafton-built-pubg-ally-a-co-playable-character-powered-by-nvidia-ace\u002F\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">NVIDIA Developer — How KRAFTON Built PUBG Ally\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">Offizielle technische Erklärung, die den Live-Match-Zustand von der Wissenssuche und der Schlussfolgerung des Sprachmodells trennt.\u003C\u002Fp>\u003C\u002Fa>",{"time":211,"blocks":212,"version":872},1790377506482,[213,219,227,232,240,245,250,255,260,267,272,277,282,287,294,299,319,324,329,334,339,360,365,370,375,380,385,390,421,428,433,438,443,448,453,458,463,484,489,495,500,505,510,515,520,525,530,535,540,545,550,555,560,589,594,599,604,609,614,623,628,633,654,659,664,669,674,679,684,689,694,699,704,740,746,751,756,761,766,771,801,806,830,835,845,854,863],{"id":214,"data":215,"type":217,"tunes":218},"intro",{"text":216},"RAG klingt kompliziert, weil der Name kompliziert ist. Die Idee ist es nicht. RAG bedeutet einfach: Bevor die KI antwortet, sucht sie zuerst nach relevanten Informationen aus einer Wissensquelle und gibt diese Informationen an das Sprachmodell weiter.","paragraph",{},{"id":220,"data":221,"type":225,"tunes":226},"one-sentence",{"body":222,"title":223,"variant":224},"\u003Cstrong>RAG ist der Schritt, bei dem eine KI eine Wissensdatenbank nach nützlichen Informationen durchsucht, bevor das LLM die Antwort schreibt.\u003C\u002Fstrong>","RAG in einem Satz","info","callout",{},{"id":228,"data":229,"type":217,"tunes":231},"analogy",{"text":230},"Stellen Sie sich ein LLM wie eine kluge Person vor, die an einem Schreibtisch sitzt. RAG ist der Bibliothekar, der die richtige Seite aus dem richtigen Buch bringt. Das LLM liest dann diese Seite und antwortet Ihnen.",{},{"id":233,"data":234,"type":238,"tunes":239},"toc",{"title":235,"maxLevel":236,"minLevel":237},"Inhalt",3,2,"tableOfContents",{},{"id":241,"data":242,"type":41,"tunes":244},"h-llm",{"text":243,"level":237},"Zuerst: Was macht das LLM?",{},{"id":246,"data":247,"type":217,"tunes":249},"p-llm-1",{"text":248},"Das LLM ist der Teil, der Sprache versteht und Sprache erzeugt. Es kann Ihre Frage lesen, Anweisungen verstehen, Informationen vergleichen, etwas erklären und eine Antwort schreiben.",{},{"id":251,"data":252,"type":217,"tunes":254},"p-llm-2",{"text":253},"Aber das LLM weiß nicht automatisch, was sich gerade in Ihrer Unternehmensdatenbank, Ihrer Spielsitzung, Ihren privaten Dokumenten oder einer Datei befindet, die Sie vor fünf Minuten erstellt haben.",{},{"id":256,"data":257,"type":217,"tunes":259},"p-llm-3",{"text":258},"Es weiß nur, was bereits im Modell enthalten ist, plus alle Informationen, die die Anwendung ihm in der aktuellen Anfrage gibt.",{},{"id":261,"data":262,"type":225,"tunes":266},"llm-rule",{"body":263,"title":264,"variant":265},"Das LLM \u003Cstrong>denkt und schreibt\u003C\u002Fstrong>. Es besitzt nicht automatisch alle Ihre aktuellen Daten.","Einfache Regel","note",{},{"id":268,"data":269,"type":41,"tunes":271},"h-kb",{"text":270,"level":237},"Dann: Was ist die Wissensdatenbank?",{},{"id":273,"data":274,"type":217,"tunes":276},"p-kb-1",{"text":275},"Eine Wissensdatenbank ist einfach eine Information, die die Anwendung durchsuchen kann.",{},{"id":278,"data":279,"type":217,"tunes":281},"p-kb-2",{"text":280},"Sie könnte PDFs, Handbücher, Produktdokumentationen, Support-Artikel, Verträge, Spielregeln, Waffendaten, interne Unternehmensdokumente, Datenbankeinträge oder anderen Text enthalten.",{},{"id":283,"data":284,"type":217,"tunes":286},"p-kb-3",{"text":285},"Die Wissensdatenbank kann lokal auf Ihrem eigenen Rechner sein. Sie kann auf einem Server sein. Sie kann in einer Vektordatenbank sein. Sie kann auch aus normalen Dateien erstellt werden. RAG bedeutet nicht Internet.",{},{"id":288,"data":289,"type":225,"tunes":293},"no-internet",{"body":290,"title":291,"variant":292},"\u003Cstrong>RAG erfordert kein Internet.\u003C\u002Fstrong> Die Informationen können vollständig lokal sein.","Wichtig","success",{},{"id":295,"data":296,"type":41,"tunes":298},"h-rag",{"text":297,"level":237},"Was macht RAG also tatsächlich?",{},{"id":300,"data":301,"type":317,"tunes":318},"rag-flow",{"steps":302,"title":315,"orientation":316},[303,306,309,312],{"label":304,"description":305},"1. Sie stellen eine Frage","Zum Beispiel: Welche Munition verwendet diese Waffe?",{"label":307,"description":308},"2. RAG durchsucht die Wissensdatenbank","Das System sucht nach den kleinen Informationsstücken, die für Ihre Frage am relevantesten sind.",{"label":310,"description":311},"3. RAG gibt diese Stücke an das LLM","Das LLM erhält die Frage plus die abgerufenen Informationen.",{"label":313,"description":314},"4. Das LLM schreibt die Antwort","Es verwendet die abgerufenen Informationen als Kontext für die Antwort.","Der gesamte RAG-Prozess","auto","processFlow",{},{"id":320,"data":321,"type":217,"tunes":323},"rag-that-is-it",{"text":322},"Das ist RAG.",{},{"id":325,"data":326,"type":217,"tunes":328},"rag-name",{"text":327},"Der vollständige Name ist Retrieval-Augmented Generation. Retrieval bedeutet, die relevanten Informationen zu finden. Augmented bedeutet, diese Informationen zum Kontext des Modells hinzuzufügen. Generation bedeutet, dass das LLM die endgültige Antwort schreibt.",{},{"id":330,"data":331,"type":41,"tunes":333},"h-example",{"text":332,"level":237},"Ein sehr einfaches Beispiel",{},{"id":335,"data":336,"type":217,"tunes":338},"p-ex-1",{"text":337},"Stellen Sie sich vor, Sie haben eine lokale Wissensdatenbank über ein Spiel.",{},{"id":340,"data":341,"type":358,"tunes":359},"kb-table",{"content":342,"stretched":42,"withHeadings":13},[343,346,349,352,355],[344,345],"Wissensdatenbank enthält","Beispiel",[347,348],"Waffen","AKM verwendet 7,62-mm-Munition",[350,351],"Heilgegenstände","Med Kit stellt Gesundheit wieder her",[353,354],"Anbauteile","Dieses Anbauteil funktioniert mit diesen Waffen",[356,357],"Kartenregeln","Diese Zone verhält sich auf diese Weise","table",{},{"id":361,"data":362,"type":217,"tunes":364},"p-ex-2",{"text":363},"Sie fragen: „Welche Munition verwendet die AKM?“",{},{"id":366,"data":367,"type":217,"tunes":369},"p-ex-3",{"text":368},"RAG durchsucht die Wissensdatenbank und findet den Eintrag über die AKM. Es gibt dieses kleine Stück Information an das LLM weiter. Das LLM antwortet dann: „Die AKM verwendet 7,62-mm-Munition.“",{},{"id":371,"data":372,"type":217,"tunes":374},"p-ex-4",{"text":373},"Das LLM brauchte nicht die gesamte Datenbank. RAG brachte nur den nützlichen Teil.",{},{"id":376,"data":377,"type":41,"tunes":379},"h-state",{"text":378,"level":237},"Nun der wichtige Teil: RAG ist nicht der aktuelle Zustand",{},{"id":381,"data":382,"type":217,"tunes":384},"p-state-1",{"text":383},"Hier werden viele Erklärungen verwirrend.",{},{"id":386,"data":387,"type":217,"tunes":389},"p-state-2",{"text":388},"RAG gibt der KI normalerweise Wissen. Ein Zustandssystem gibt der KI Fakten darüber, was gerade jetzt wahr ist.",{},{"id":391,"data":392,"type":419,"tunes":420},"knowledge-state",{"rows":393,"title":411,"layout":358,"columns":412},[394,399,403,407],{"id":395,"label":396,"values":397},"weapon","Waffe",[398,398],"",{"id":400,"label":401,"values":402},"ammo","Munition",[398,398],{"id":404,"label":405,"values":406},"health","Gesundheit",[398,398],{"id":408,"label":409,"values":410},"enemy","Gegner",[398,398],"Wissen vs. aktueller Zustand",[413,416],{"id":414,"label":415},"knowledge","RAG \u002F Wissen",{"id":417,"label":418},"state","Aktueller Zustand","comparison",{},{"id":422,"data":423,"type":225,"tunes":427},"dont-mix",{"body":424,"title":425,"variant":426},"RAG antwortet: \u003Cstrong>Was ist allgemein wahr?\u003C\u002Fstrong>\u003Cbr>Zustand antwortet: \u003Cstrong>Was ist gerade jetzt wahr?\u003C\u002Fstrong>","Diese beiden nicht vermischen","warning",{},{"id":429,"data":430,"type":41,"tunes":432},"h-state-db",{"text":431,"level":237},"Was ist eine Zustandsdatenbank?",{},{"id":434,"data":435,"type":217,"tunes":437},"p-statedb-1",{"text":436},"Eine Zustandsdatenbank oder ein Zustandsspeicher ist einfach ein Ort, an dem die Anwendung aktuelle Fakten aufbewahrt.",{},{"id":439,"data":440,"type":217,"tunes":442},"p-statedb-2",{"text":441},"In einem Spiel kennt die Engine bereits Dinge wie Ihre Gesundheit, Position, Inventar, Munition, aktuelle Mission, nahe Objekte und Gegnerstatus. Ein KI-System kann ausgewählte Teile dieses Zustands dem Modell zugänglich machen.",{},{"id":444,"data":445,"type":217,"tunes":447},"p-statedb-3",{"text":446},"In einer Geschäftsanwendung könnte dieselbe Idee eine Bestelldatenbank, ein Kundendatensatz, ein Projektstatus oder der aktuelle Wert eines Sensors sein.",{},{"id":449,"data":450,"type":217,"tunes":452},"p-statedb-4",{"text":451},"Der Zustand wird von der Anwendung selbst erstellt, während Dinge geschehen. Wenn Sie Gesundheit verlieren, aktualisiert das Spiel den Gesundheitswert. Wenn Sie Munition aufnehmen, ändert sich das Inventar. Wenn eine Bestellung bezahlt wird, ändert das Geschäftssystem den Bestellstatus.",{},{"id":454,"data":455,"type":225,"tunes":457},"state-rule",{"body":456,"title":264,"variant":224},"Die Anwendung erstellt und aktualisiert \u003Cstrong>Zustand\u003C\u002Fstrong>. RAG durchsucht \u003Cstrong>Wissen\u003C\u002Fstrong>. Das LLM verwendet beides, um zu entscheiden, was es sagen oder tun soll.",{},{"id":459,"data":460,"type":41,"tunes":462},"h-together",{"text":461,"level":237},"Wie die drei Teile zusammenwirken",{},{"id":464,"data":465,"type":317,"tunes":483},"together-flow",{"steps":466,"title":482,"orientation":316},[467,470,473,476,479],{"label":468,"description":469},"1. Aktueller Zustand","Die Anwendung teilt der KI mit, was jetzt gilt: Gesundheit 41 %, AKM ausgerüstet, 23 Schuss.",{"label":471,"description":472},"2. RAG","Das System ruft nützliches Wissen ab: wie die Waffe funktioniert, welches Heilmittel verfügbar ist oder eine relevante Regel.",{"label":474,"description":475},"3. LLM","Das Modell erhält die Frage, den aktuellen Zustand und das abgerufene Wissen.",{"label":477,"description":478},"4. Schlussfolgerung","Das LLM kombiniert diese Eingaben und entscheidet, welche Antwort oder übergeordnete Aktion sinnvoll ist.",{"label":480,"description":481},"5. Anwendung","Wenn eine Aktion erforderlich ist, führt die Anwendung oder die Spiel-Engine sie aus und aktualisiert den Zustand erneut.","LLM + Zustand + RAG",{},{"id":485,"data":486,"type":217,"tunes":488},"p-arch-intro",{"text":487},"Die grundlegende Architektur ist also:",{},{"id":490,"data":491,"type":225,"tunes":494},"simple-architecture",{"body":492,"title":493,"variant":265},"\u003Cstrong>Zustand = was jetzt gilt\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>RAG = nützliches Wissen\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>LLM = versteht, schlussfolgert und schreibt\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Anwendung = führt die reale Aktion aus\u003C\u002Fstrong>","Die einfachste Architektur",{},{"id":496,"data":497,"type":41,"tunes":499},"h-vector",{"text":498,"level":237},"Verwendet RAG immer eine Vektordatenbank?",{},{"id":501,"data":502,"type":217,"tunes":504},"p-vector-1",{"text":503},"Nein.",{},{"id":506,"data":507,"type":217,"tunes":509},"p-vector-2",{"text":508},"Eine Vektordatenbank ist eine gängige Methode, um semantische Suche aufzubauen, aber sie ist nicht die Definition von RAG.",{},{"id":511,"data":512,"type":217,"tunes":514},"p-vector-3",{"text":513},"Der wichtige Teil ist das Abrufen: Das System findet relevante externe Informationen und fügt sie dem Kontext des LLM hinzu, bevor die Antwort generiert wird.",{},{"id":516,"data":517,"type":217,"tunes":519},"p-vector-4",{"text":518},"OpenAIs File Search kann beispielsweise mit Dateien arbeiten, die in Vektorspeichern abgelegt sind. Dateien werden in kleinere Stücke aufgeteilt, damit das System die Teile abrufen kann, die für eine Frage relevant sind. Das ist eine Umsetzung derselben Grundidee.",{},{"id":521,"data":522,"type":41,"tunes":524},"h-embedding",{"text":523,"level":237},"Was ist ein Embedding, einfach erklärt?",{},{"id":526,"data":527,"type":217,"tunes":529},"p-emb-1",{"text":528},"Du musst Embeddings nicht verstehen, um RAG zu verstehen.",{},{"id":531,"data":532,"type":217,"tunes":534},"p-emb-2",{"text":533},"Aber die einfache Version ist diese: Ein Embedding ist eine numerische Darstellung von Bedeutung. Es hilft einem Suchsystem, Text zu finden, der konzeptionell ähnlich ist, auch wenn die Wörter nicht genau dieselben sind.",{},{"id":536,"data":537,"type":217,"tunes":539},"p-emb-3",{"text":538},"Zum Beispiel sucht eine normale Stichwortsuche möglicherweise nach den genauen Wörtern „Autoreparatur“. Die semantische Suche kann auch verstehen, dass „repariere mein Fahrzeug“ ein ähnliches Thema betrifft.",{},{"id":541,"data":542,"type":217,"tunes":544},"p-emb-4",{"text":543},"Das macht Embeddings für RAG nützlich, aber RAG kann auch Stichwortsuche, Datenbankabfragen oder eine Mischung aus mehreren Methoden verwenden.",{},{"id":546,"data":547,"type":41,"tunes":549},"h-memory",{"text":548,"level":237},"RAG ist auch kein Gedächtnis",{},{"id":551,"data":552,"type":217,"tunes":554},"p-memory-1",{"text":553},"Gedächtnis ist ein weiteres Konzept, das oft mit RAG vermischt wird.",{},{"id":556,"data":557,"type":217,"tunes":559},"p-memory-2",{"text":558},"Gedächtnis sind normalerweise Informationen, die das System über frühere Interaktionen oder frühere Ereignisse behält. RAG ist der Mechanismus, mit dem relevantes Wissen abgerufen wird, wenn es benötigt wird.",{},{"id":561,"data":562,"type":358,"tunes":588},"parts-table",{"content":563,"stretched":42,"withHeadings":13},[564,567,570,573,576,579,582,585],[565,566],"Teil","Einfache Bedeutung",[568,569],"LLM","Der Teil, der Sprache versteht und generiert",[571,572],"RAG","Der Teil, der vor der Antwort relevantes Wissen nachschlägt",[574,575],"Wissensbasis","Die Informationen, die RAG durchsuchen kann",[577,578],"Zustand","Was gerade jetzt in der Anwendung oder Welt gilt",[580,581],"Gedächtnis","Informationen, die aus früheren Interaktionen oder Ereignissen behalten werden",[583,584],"Werkzeug \u002F Aktion","Etwas, das die KI aufrufen oder die Anwendung bitten darf zu tun",[586,587],"Kontext","Die Informationen, die dem LLM für diese Anfrage gerade vorgelegt werden",{},{"id":590,"data":591,"type":41,"tunes":593},"h-pubg",{"text":592,"level":237},"Ein echtes Spielbeispiel: PUBG Ally",{},{"id":595,"data":596,"type":217,"tunes":598},"p-pubg-1",{"text":597},"PUBG Ally ist ein nützliches Beispiel, weil es den Unterschied sichtbar macht.",{},{"id":600,"data":601,"type":217,"tunes":603},"p-pubg-2",{"text":602},"KRAFTON beschreibt den Live-Match-Zustand als separate Quelle der Wahrheit. Das Spiel gibt aktuelle Fakten über Beobachtungswerkzeuge preis: aktuelle Waffe, Munition, Gesundheit, Status der Sicherheitszone, Gegenstände in der Nähe und Kampfsituation.",{},{"id":605,"data":606,"type":217,"tunes":608},"p-pubg-3",{"text":607},"Die Wissenssuche ist eine andere Aufgabe. Das System kann kuratiertes Wissen über Waffen, Aufsätze, Gegenstände und Regeln nutzen. Das ACE Game Agent SDK von NVIDIA bietet außerdem eine separate RAG-API zum Abrufen von Wissen aus von Entwicklern erstellten Datenbanken.",{},{"id":610,"data":611,"type":217,"tunes":613},"p-pubg-4",{"text":612},"Das gibt uns die klare Trennung: Die Spiel-Engine sagt, was gerade passiert, der Abruf liefert relevantes Wissen, und das Sprachmodell entscheidet, was die Informationen bedeuten.",{},{"id":615,"data":616,"type":621,"tunes":622},"ref-pubg",{"url":617,"title":618,"excerpt":619,"ctaLabel":620},"https:\u002F\u002Ffigure.rocks\u002Fblog\u002Fpubg-ally-shows-why-ai-teammates-need-two-brains-fast-reflexes-and-slow-reasoning","PUBG Ally zeigt, warum KI-Teamkollegen zwei Gehirne brauchen: schnelle Reflexe und langsames Denken","Ein praktisches Spielbeispiel, das zeigt, wie Live-Zustand, Sprachschlussfolgerung und deterministische spielseitige Steuerung zusammenwirken können.","Den Artikel zur PUBG-Ally-Architektur lesen","referralArticle",{},{"id":624,"data":625,"type":41,"tunes":627},"h-complete",{"text":626,"level":237},"Ein vollständiges Beispiel",{},{"id":629,"data":630,"type":217,"tunes":632},"p-complete-1",{"text":631},"Stellen Sie sich vor, Sie sagen einem KI-Teamkollegen: „Ich habe wenig Gesundheit. Sollten wir angreifen?“",{},{"id":634,"data":635,"type":317,"tunes":653},"complete-flow",{"steps":636,"title":652,"orientation":316},[637,639,641,643,646,649],{"label":577,"description":638},"Das Spiel meldet: Gesundheit 24 %, ein Gegner in der Nähe, zwei Heilgegenstände verfügbar.",{"label":571,"description":640},"Das Wissenssystem ruft die relevanten Regeln für den Heilgegenstand und möglicherweise Informationen über die aktuelle Waffe oder taktische Mechanik ab.",{"label":568,"description":642},"Das Modell kombiniert Ihre Anfrage, den aktuellen Zustand und das abgerufene Wissen.",{"label":644,"description":645},"Entscheidung","Es kommt zu dem Schluss, dass zuerst zu heilen sicherer ist, als sofort anzugreifen.",{"label":647,"description":648},"Werkzeug \u002F Spiel-Engine","Der Agent fordert eine legale Spielaktion an, z. B. sich in Deckung zu bewegen oder den Heilgegenstand zu verwenden.",{"label":650,"description":651},"Neuer Zustand","Das Spiel führt die Aktion aus und meldet die aktualisierte Situation an den Agenten zurück.","Was als Nächstes passiert",{},{"id":655,"data":656,"type":217,"tunes":658},"p-complete-2",{"text":657},"RAG hat die Figur nicht gesteuert. Die Zustandsdatenbank hat nicht geschlussfolgert. Das LLM hat das Spiel nicht direkt verändert. Jeder Teil hatte eine Aufgabe.",{},{"id":660,"data":661,"type":41,"tunes":663},"h-why",{"text":662,"level":237},"Warum überhaupt RAG verwenden?",{},{"id":665,"data":666,"type":217,"tunes":668},"p-why-1",{"text":667},"Weil es langsam, teuer und oft verwirrend wäre, jedes Dokument, jede Regel und jeden Datenbankeintrag in jeden Prompt aufzunehmen.",{},{"id":670,"data":671,"type":217,"tunes":673},"p-why-2",{"text":672},"RAG ermöglicht es dem System, nur die Informationen auszuwählen, die für die aktuelle Frage nützlich sind.",{},{"id":675,"data":676,"type":217,"tunes":678},"p-why-3",{"text":677},"Es ermöglicht außerdem, die Wissensbasis zu aktualisieren, ohne das gesamte Sprachmodell neu zu trainieren. Ändern Sie das Dokument oder die Datenbank, bauen Sie den Index bei Bedarf neu auf oder aktualisieren Sie ihn, und der nächste Abruf kann die neueren Informationen verwenden.",{},{"id":680,"data":681,"type":41,"tunes":683},"h-not-guarantee",{"text":682,"level":237},"Was RAG nicht garantiert",{},{"id":685,"data":686,"type":217,"tunes":688},"p-not-1",{"text":687},"RAG kann die Fundierung verbessern, aber es macht eine Antwort nicht automatisch korrekt.",{},{"id":690,"data":691,"type":217,"tunes":693},"p-not-2",{"text":692},"Der Abrufschritt kann das falsche Dokument finden. Das richtige Dokument kann veraltet sein. Das LLM kann gute Belege missverstehen. Oder der aktuelle Zustand kann sich geändert haben.",{},{"id":695,"data":696,"type":217,"tunes":698},"p-not-3",{"text":697},"Ein zuverlässiges System muss daher den Abruf, die Aktualität des Zustands und die endgültige Schlussfolgerung des Modells getrennt validieren.",{},{"id":700,"data":701,"type":41,"tunes":703},"h-mental",{"text":702,"level":237},"Das einfachste mentale Modell zum Merken",{},{"id":705,"data":706,"type":419,"tunes":739},"mental-table",{"rows":707,"title":732,"layout":358,"columns":733},[708,712,716,720,724,728],{"id":709,"label":710,"values":711},"brain","Denkende Person",[398,398],{"id":713,"label":714,"values":715},"library","Ein Nachschlagewerk finden",[398,398],{"id":717,"label":718,"values":719},"books","Bücher im Regal",[398,398],{"id":721,"label":722,"values":723},"dashboard","Aktuelles Dashboard oder Instrumententafel",[398,398],{"id":725,"label":726,"values":727},"notes","Notizen von früheren Besprechungen",[398,398],{"id":729,"label":730,"values":731},"hands","Etwas in der realen Welt tun",[398,398],"Stellen Sie sich ein KI-System wie eine Person an einem Schreibtisch vor",[734,736],{"id":228,"label":735},"Analogie",{"id":737,"label":738},"system","KI-System",{},{"id":741,"data":742,"type":225,"tunes":745},"remember",{"body":743,"title":744,"variant":292},"\u003Cstrong>LLM = Gehirn.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>RAG = Bibliothekar.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Wissensbasis = Bibliothek.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Zustand = was das Dashboard gerade jetzt anzeigt.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Werkzeuge = die Hände, die tatsächlich etwas tun können.\u003C\u002Fstrong>","Wenn Sie sich nur das merken",{},{"id":747,"data":748,"type":41,"tunes":750},"h-conclusion",{"text":749,"level":237},"Fazit",{},{"id":752,"data":753,"type":217,"tunes":755},"p-conc-1",{"text":754},"RAG ist viel weniger mysteriös, sobald die Teile getrennt sind.",{},{"id":757,"data":758,"type":217,"tunes":760},"p-conc-2",{"text":759},"Das LLM versteht und erzeugt Sprache. Die Anwendung verwaltet den aktuellen Zustand. Die Wissensbasis speichert Informationen. RAG findet den nützlichen Teil dieser Informationen und fügt ihn in den Kontext des LLM ein. Werkzeuge oder die Anwendung führen reale Aktionen aus.",{},{"id":762,"data":763,"type":217,"tunes":765},"p-conc-3",{"text":764},"Das ist die grundlegende Architektur hinter vielen modernen KI-Assistenten und Agenten.",{},{"id":767,"data":768,"type":41,"tunes":770},"h-faq",{"text":769,"level":237},"FAQ",{},{"id":772,"data":773,"type":772,"tunes":800},"faq",{"items":774,"title":799},[775,779,783,787,791,795],{"id":776,"answer":777,"question":778},"faq1","RAG ist ein Schritt, bei dem eine KI eine Wissensquelle nach relevanten Informationen durchsucht, bevor das Sprachmodell seine Antwort schreibt.","Was ist RAG in einfachen Worten?",{"id":780,"answer":781,"question":782},"faq2","Nein. Die Wissensbasis kann vollständig lokal auf Ihrem Computer oder Server liegen.","Braucht RAG das Internet?",{"id":784,"answer":785,"question":786},"faq3","Nein. Die Datenbank oder Dateien enthalten die Informationen. RAG ist der Abrufprozess, der den nützlichen Teil findet und ihn dem LLM gibt.","Ist RAG dasselbe wie eine Datenbank?",{"id":788,"answer":789,"question":790},"faq4","Nein. Das Gedächtnis speichert normalerweise frühere Interaktionen oder Ereignisse. RAG ruft relevantes Wissen ab, wenn es benötigt wird.","Ist RAG dasselbe wie Gedächtnis?",{"id":792,"answer":793,"question":794},"faq5","Nicht unbedingt. Der aktuelle Zustand wird normalerweise direkt von der Anwendung oder einem Zustandsspeicher bezogen. RAG wird besser als Abruf aus einer Wissensquelle verstanden.","Ist der aktuelle Anwendungszustand Teil von RAG?",{"id":796,"answer":797,"question":798},"faq6","Nein. Es kann bessere Belege liefern, aber der Abruf kann immer noch falsch oder veraltet sein und das LLM kann immer noch falsch schlussfolgern.","Macht RAG KI-Antworten korrekt?","RAG in einfachen Worten",{},{"id":802,"data":803,"type":41,"tunes":805},"h-glossary",{"text":804,"level":237},"Glossar",{},{"id":807,"data":808,"type":807,"tunes":829},"glossary",{"title":809,"entries":810},"Die grundlegenden Begriffe",[811,814,817,820,822,825],{"term":568,"anchor":812,"definition":813},"llm","Ein Sprachmodell, das Text versteht und erzeugt und über Informationen, die in seinen Kontext gestellt werden, schlussfolgern kann.",{"term":571,"anchor":815,"definition":816},"rag","Retrieval-Augmented Generation: Abrufen relevanter externer Informationen und Hinzufügen zum Kontext des Modells, bevor eine Antwort generiert wird.",{"term":574,"anchor":818,"definition":819},"knowledge-base","Die Dateien, Dokumente, Datensätze oder anderen Informationen, die der Abruf durchsuchen kann.",{"term":577,"anchor":417,"definition":821},"Die aktuellen Fakten einer Anwendung, eines Systems oder der Welt zu einem bestimmten Zeitpunkt.",{"term":586,"anchor":823,"definition":824},"context","Die Informationen, die dem Sprachmodell derzeit für eine Anfrage oder einen Schlussfolgerungsschritt bereitgestellt werden.",{"term":826,"anchor":827,"definition":828},"Embedding","embedding","Eine numerische Darstellung von Bedeutung, die der semantischen Suche helfen kann, konzeptionell ähnliche Informationen zu finden.",{},{"id":831,"data":832,"type":41,"tunes":834},"h-sources",{"text":833,"level":237},"Primärquellen",{},{"id":836,"data":837,"type":843,"tunes":844},"src-openai-vector",{"link":838,"meta":839},"https:\u002F\u002Fplatform.openai.com\u002Fdocs\u002Fapi-reference\u002Fvector-stores-files",{"image":840,"title":841,"description":842},{"url":398},"OpenAI — Vector Store Files","Offizielle Dokumentation, die zeigt, wie Dateien an Vektorspeicher angehängt, in Chunks aufgeteilt und für den Dateisuche-Abruf verfügbar gemacht werden können.","linkTool",{},{"id":846,"data":847,"type":843,"tunes":853},"src-openai-quickstart",{"link":848,"meta":849},"https:\u002F\u002Fplatform.openai.com\u002Fdocs\u002Fquickstart",{"image":850,"title":851,"description":852},{"url":398},"OpenAI — Developer Quickstart","Offizielle OpenAI-Dokumentation, die Werkzeuge wie die Dateisuche beschreibt, um Modellen Zugriff auf externe Informationen zu geben.",{},{"id":855,"data":856,"type":843,"tunes":862},"src-nvidia-ace",{"link":857,"meta":858},"https:\u002F\u002Fdeveloper.nvidia.com\u002Face-for-games",{"image":859,"title":860,"description":861},{"url":398},"NVIDIA Developer — ACE for Games","Offizielle NVIDIA-Dokumentation, die separate Agent-, Chat- und RAG-APIs beschreibt, um Spielcharaktere mit Spielzustand, kontextuellem Wissen und modellgesteuerten Aktionen zu verbinden.",{},{"id":864,"data":865,"type":843,"tunes":871},"src-nvidia-pubg",{"link":866,"meta":867},"https:\u002F\u002Fdeveloper.nvidia.com\u002Fblog\u002Fhow-krafton-built-pubg-ally-a-co-playable-character-powered-by-nvidia-ace\u002F",{"image":868,"title":869,"description":870},{"url":398},"NVIDIA Developer — How KRAFTON Built PUBG Ally","Offizielle technische Erklärung, die den Live-Match-Zustand von der Wissenssuche und der Schlussfolgerung des Sprachmodells trennt.",{},"2.31","RAG klingt kompliziert, aber die Idee ist einfach: Bevor eine KI antwortet, sucht sie zunächst nützliche Informationen aus einer Wissensquelle und gibt diese Informationen an das Sprachmodell weiter. Dieser Leitfaden erklärt RAG, LLMs, Zustand, Gedächtnis und Werkzeuge anhand eines einfachen mentalen Modells.","\u002Fuploads\u002F2026\u002F09\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works-1790377492124-khjagt.webp","what-is-rag-the-simplest-explanation-of-how-it-works-1790377492124-khjagt","PUBLISHED","2026-09-25T19:03:00.000Z","2026-09-25T23:03:13.651Z","2026-09-25T23:41:17.272Z",{"en":881,"de":882,"sr":883,"es":884,"fr":885,"it":886,"ru":887,"zh":888},"\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works","\u002Fde\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works","\u002Fsr\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works","\u002Fes\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works","\u002Ffr\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works","\u002Fit\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works","\u002Fru\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works","\u002Fzh\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works",[890,894,898,902,906,910],{"id":891,"name":892,"slug":893},46,"Überblick","overview",{"id":895,"name":896,"slug":897},57,"Daten-Grenzen","data-boundaries",{"id":899,"name":900,"slug":901},51,"Anti-Pattern","anti-patterns",{"id":903,"name":904,"slug":905},58,"Evaluation & Qualitäts-Gates","evaluation",{"id":907,"name":908,"slug":909},56,"Use-Case-Portfolio","use-case-portfolio",{"id":911,"name":912,"slug":913},60,"Kosten- & Latenz-Kontrollen","cost-and-latency",{"id":915,"login":916,"email":917,"displayName":918},"20","rooth8233","aleksandar@stajic.de","Aleksandar Stajić",[920,1267],{"lang":7,"title":207,"content":209,"contentJson":921,"excerpt":873},{"time":211,"blocks":922,"version":872},[923,926,929,932,935,938,941,944,947,950,953,956,959,962,965,968,976,979,982,985,988,997,1000,1003,1006,1009,1012,1015,1030,1033,1036,1039,1042,1045,1048,1051,1054,1063,1066,1069,1072,1075,1078,1081,1084,1087,1090,1093,1096,1099,1102,1105,1108,1120,1123,1126,1129,1132,1135,1138,1141,1144,1154,1157,1160,1163,1166,1169,1172,1175,1178,1181,1184,1203,1206,1209,1212,1215,1218,1221,1231,1234,1244,1247,1252,1257,1262],{"id":214,"data":924,"type":217,"tunes":925},{"text":216},{},{"id":220,"data":927,"type":225,"tunes":928},{"body":222,"title":223,"variant":224},{},{"id":228,"data":930,"type":217,"tunes":931},{"text":230},{},{"id":233,"data":933,"type":238,"tunes":934},{"title":235,"maxLevel":236,"minLevel":237},{},{"id":241,"data":936,"type":41,"tunes":937},{"text":243,"level":237},{},{"id":246,"data":939,"type":217,"tunes":940},{"text":248},{},{"id":251,"data":942,"type":217,"tunes":943},{"text":253},{},{"id":256,"data":945,"type":217,"tunes":946},{"text":258},{},{"id":261,"data":948,"type":225,"tunes":949},{"body":263,"title":264,"variant":265},{},{"id":268,"data":951,"type":41,"tunes":952},{"text":270,"level":237},{},{"id":273,"data":954,"type":217,"tunes":955},{"text":275},{},{"id":278,"data":957,"type":217,"tunes":958},{"text":280},{},{"id":283,"data":960,"type":217,"tunes":961},{"text":285},{},{"id":288,"data":963,"type":225,"tunes":964},{"body":290,"title":291,"variant":292},{},{"id":295,"data":966,"type":41,"tunes":967},{"text":297,"level":237},{},{"id":300,"data":969,"type":317,"tunes":975},{"steps":970,"title":315,"orientation":316},[971,972,973,974],{"label":304,"description":305},{"label":307,"description":308},{"label":310,"description":311},{"label":313,"description":314},{},{"id":320,"data":977,"type":217,"tunes":978},{"text":322},{},{"id":325,"data":980,"type":217,"tunes":981},{"text":327},{},{"id":330,"data":983,"type":41,"tunes":984},{"text":332,"level":237},{},{"id":335,"data":986,"type":217,"tunes":987},{"text":337},{},{"id":340,"data":989,"type":358,"tunes":996},{"content":990,"stretched":42,"withHeadings":13},[991,992,993,994,995],[344,345],[347,348],[350,351],[353,354],[356,357],{},{"id":361,"data":998,"type":217,"tunes":999},{"text":363},{},{"id":366,"data":1001,"type":217,"tunes":1002},{"text":368},{},{"id":371,"data":1004,"type":217,"tunes":1005},{"text":373},{},{"id":376,"data":1007,"type":41,"tunes":1008},{"text":378,"level":237},{},{"id":381,"data":1010,"type":217,"tunes":1011},{"text":383},{},{"id":386,"data":1013,"type":217,"tunes":1014},{"text":388},{},{"id":391,"data":1016,"type":419,"tunes":1029},{"rows":1017,"title":411,"layout":358,"columns":1026},[1018,1020,1022,1024],{"id":395,"label":396,"values":1019},[398,398],{"id":400,"label":401,"values":1021},[398,398],{"id":404,"label":405,"values":1023},[398,398],{"id":408,"label":409,"values":1025},[398,398],[1027,1028],{"id":414,"label":415},{"id":417,"label":418},{},{"id":422,"data":1031,"type":225,"tunes":1032},{"body":424,"title":425,"variant":426},{},{"id":429,"data":1034,"type":41,"tunes":1035},{"text":431,"level":237},{},{"id":434,"data":1037,"type":217,"tunes":1038},{"text":436},{},{"id":439,"data":1040,"type":217,"tunes":1041},{"text":441},{},{"id":444,"data":1043,"type":217,"tunes":1044},{"text":446},{},{"id":449,"data":1046,"type":217,"tunes":1047},{"text":451},{},{"id":454,"data":1049,"type":225,"tunes":1050},{"body":456,"title":264,"variant":224},{},{"id":459,"data":1052,"type":41,"tunes":1053},{"text":461,"level":237},{},{"id":464,"data":1055,"type":317,"tunes":1062},{"steps":1056,"title":482,"orientation":316},[1057,1058,1059,1060,1061],{"label":468,"description":469},{"label":471,"description":472},{"label":474,"description":475},{"label":477,"description":478},{"label":480,"description":481},{},{"id":485,"data":1064,"type":217,"tunes":1065},{"text":487},{},{"id":490,"data":1067,"type":225,"tunes":1068},{"body":492,"title":493,"variant":265},{},{"id":496,"data":1070,"type":41,"tunes":1071},{"text":498,"level":237},{},{"id":501,"data":1073,"type":217,"tunes":1074},{"text":503},{},{"id":506,"data":1076,"type":217,"tunes":1077},{"text":508},{},{"id":511,"data":1079,"type":217,"tunes":1080},{"text":513},{},{"id":516,"data":1082,"type":217,"tunes":1083},{"text":518},{},{"id":521,"data":1085,"type":41,"tunes":1086},{"text":523,"level":237},{},{"id":526,"data":1088,"type":217,"tunes":1089},{"text":528},{},{"id":531,"data":1091,"type":217,"tunes":1092},{"text":533},{},{"id":536,"data":1094,"type":217,"tunes":1095},{"text":538},{},{"id":541,"data":1097,"type":217,"tunes":1098},{"text":543},{},{"id":546,"data":1100,"type":41,"tunes":1101},{"text":548,"level":237},{},{"id":551,"data":1103,"type":217,"tunes":1104},{"text":553},{},{"id":556,"data":1106,"type":217,"tunes":1107},{"text":558},{},{"id":561,"data":1109,"type":358,"tunes":1119},{"content":1110,"stretched":42,"withHeadings":13},[1111,1112,1113,1114,1115,1116,1117,1118],[565,566],[568,569],[571,572],[574,575],[577,578],[580,581],[583,584],[586,587],{},{"id":590,"data":1121,"type":41,"tunes":1122},{"text":592,"level":237},{},{"id":595,"data":1124,"type":217,"tunes":1125},{"text":597},{},{"id":600,"data":1127,"type":217,"tunes":1128},{"text":602},{},{"id":605,"data":1130,"type":217,"tunes":1131},{"text":607},{},{"id":610,"data":1133,"type":217,"tunes":1134},{"text":612},{},{"id":615,"data":1136,"type":621,"tunes":1137},{"url":617,"title":618,"excerpt":619,"ctaLabel":620},{},{"id":624,"data":1139,"type":41,"tunes":1140},{"text":626,"level":237},{},{"id":629,"data":1142,"type":217,"tunes":1143},{"text":631},{},{"id":634,"data":1145,"type":317,"tunes":1153},{"steps":1146,"title":652,"orientation":316},[1147,1148,1149,1150,1151,1152],{"label":577,"description":638},{"label":571,"description":640},{"label":568,"description":642},{"label":644,"description":645},{"label":647,"description":648},{"label":650,"description":651},{},{"id":655,"data":1155,"type":217,"tunes":1156},{"text":657},{},{"id":660,"data":1158,"type":41,"tunes":1159},{"text":662,"level":237},{},{"id":665,"data":1161,"type":217,"tunes":1162},{"text":667},{},{"id":670,"data":1164,"type":217,"tunes":1165},{"text":672},{},{"id":675,"data":1167,"type":217,"tunes":1168},{"text":677},{},{"id":680,"data":1170,"type":41,"tunes":1171},{"text":682,"level":237},{},{"id":685,"data":1173,"type":217,"tunes":1174},{"text":687},{},{"id":690,"data":1176,"type":217,"tunes":1177},{"text":692},{},{"id":695,"data":1179,"type":217,"tunes":1180},{"text":697},{},{"id":700,"data":1182,"type":41,"tunes":1183},{"text":702,"level":237},{},{"id":705,"data":1185,"type":419,"tunes":1202},{"rows":1186,"title":732,"layout":358,"columns":1199},[1187,1189,1191,1193,1195,1197],{"id":709,"label":710,"values":1188},[398,398],{"id":713,"label":714,"values":1190},[398,398],{"id":717,"label":718,"values":1192},[398,398],{"id":721,"label":722,"values":1194},[398,398],{"id":725,"label":726,"values":1196},[398,398],{"id":729,"label":730,"values":1198},[398,398],[1200,1201],{"id":228,"label":735},{"id":737,"label":738},{},{"id":741,"data":1204,"type":225,"tunes":1205},{"body":743,"title":744,"variant":292},{},{"id":747,"data":1207,"type":41,"tunes":1208},{"text":749,"level":237},{},{"id":752,"data":1210,"type":217,"tunes":1211},{"text":754},{},{"id":757,"data":1213,"type":217,"tunes":1214},{"text":759},{},{"id":762,"data":1216,"type":217,"tunes":1217},{"text":764},{},{"id":767,"data":1219,"type":41,"tunes":1220},{"text":769,"level":237},{},{"id":772,"data":1222,"type":772,"tunes":1230},{"items":1223,"title":799},[1224,1225,1226,1227,1228,1229],{"id":776,"answer":777,"question":778},{"id":780,"answer":781,"question":782},{"id":784,"answer":785,"question":786},{"id":788,"answer":789,"question":790},{"id":792,"answer":793,"question":794},{"id":796,"answer":797,"question":798},{},{"id":802,"data":1232,"type":41,"tunes":1233},{"text":804,"level":237},{},{"id":807,"data":1235,"type":807,"tunes":1243},{"title":809,"entries":1236},[1237,1238,1239,1240,1241,1242],{"term":568,"anchor":812,"definition":813},{"term":571,"anchor":815,"definition":816},{"term":574,"anchor":818,"definition":819},{"term":577,"anchor":417,"definition":821},{"term":586,"anchor":823,"definition":824},{"term":826,"anchor":827,"definition":828},{},{"id":831,"data":1245,"type":41,"tunes":1246},{"text":833,"level":237},{},{"id":836,"data":1248,"type":843,"tunes":1251},{"link":838,"meta":1249},{"image":1250,"title":841,"description":842},{"url":398},{},{"id":846,"data":1253,"type":843,"tunes":1256},{"link":848,"meta":1254},{"image":1255,"title":851,"description":852},{"url":398},{},{"id":855,"data":1258,"type":843,"tunes":1261},{"link":857,"meta":1259},{"image":1260,"title":860,"description":861},{"url":398},{},{"id":864,"data":1263,"type":843,"tunes":1266},{"link":866,"meta":1264},{"image":1265,"title":869,"description":870},{"url":398},{},{"lang":1268,"title":1269,"content":1270,"contentJson":1271,"excerpt":1792},"en","What Is RAG? The Simplest Explanation of How It Works","{\"time\":1790377494031,\"blocks\":[{\"id\":\"intro\",\"type\":\"paragraph\",\"data\":{\"text\":\"RAG sounds complicated because the name is complicated. The idea is not. RAG simply means: before the AI answers, it first looks up relevant information from a knowledge source and gives that information to the language model.\"},\"tunes\":{}},{\"id\":\"one-sentence\",\"type\":\"callout\",\"data\":{\"variant\":\"info\",\"title\":\"RAG in one sentence\",\"body\":\"\u003Cstrong>RAG is the step where an AI searches a knowledge base for useful information before the LLM writes the answer.\u003C\u002Fstrong>\"},\"tunes\":{}},{\"id\":\"analogy\",\"type\":\"paragraph\",\"data\":{\"text\":\"Think of an LLM as a smart person sitting at a desk. RAG is the librarian who brings the right page from the right book. The LLM then reads that page and answers you.\"},\"tunes\":{}},{\"id\":\"toc\",\"type\":\"tableOfContents\",\"data\":{\"title\":\"Contents\",\"minLevel\":2,\"maxLevel\":3},\"tunes\":{}},{\"id\":\"h-llm\",\"type\":\"header\",\"data\":{\"text\":\"First: what does the LLM do?\",\"level\":2},\"tunes\":{}},{\"id\":\"p-llm-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The LLM is the part that understands language and produces language. It can read your question, understand instructions, compare information, explain something and write an answer.\"},\"tunes\":{}},{\"id\":\"p-llm-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"But the LLM does not automatically know what is currently inside your company database, your game session, your private documents or a file you created five minutes ago.\"},\"tunes\":{}},{\"id\":\"p-llm-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"It only knows what is already inside the model plus whatever information the application gives it in the current request.\"},\"tunes\":{}},{\"id\":\"llm-rule\",\"type\":\"callout\",\"data\":{\"variant\":\"note\",\"title\":\"Simple rule\",\"body\":\"The LLM \u003Cstrong>thinks and writes\u003C\u002Fstrong>. It does not automatically own all of your current data.\"},\"tunes\":{}},{\"id\":\"h-kb\",\"type\":\"header\",\"data\":{\"text\":\"Then: what is the knowledge base?\",\"level\":2},\"tunes\":{}},{\"id\":\"p-kb-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A knowledge base is simply information the application can search.\"},\"tunes\":{}},{\"id\":\"p-kb-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"It could contain PDFs, manuals, product documentation, support articles, contracts, game rules, weapon data, internal company documents, database records or other text.\"},\"tunes\":{}},{\"id\":\"p-kb-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The knowledge base can be local on your own machine. It can be on a server. It can be in a vector database. It can also be built from normal files. RAG does not mean Internet.\"},\"tunes\":{}},{\"id\":\"no-internet\",\"type\":\"callout\",\"data\":{\"variant\":\"success\",\"title\":\"Important\",\"body\":\"\u003Cstrong>RAG does not require the Internet.\u003C\u002Fstrong> The information can be completely local.\"},\"tunes\":{}},{\"id\":\"h-rag\",\"type\":\"header\",\"data\":{\"text\":\"So what does RAG actually do?\",\"level\":2},\"tunes\":{}},{\"id\":\"rag-flow\",\"type\":\"processFlow\",\"data\":{\"title\":\"The whole RAG process\",\"orientation\":\"auto\",\"steps\":[{\"label\":\"1. You ask a question\",\"description\":\"For example: Which ammunition does this weapon use?\"},{\"label\":\"2. RAG searches the knowledge base\",\"description\":\"The system looks for the small pieces of information most relevant to your question.\"},{\"label\":\"3. RAG gives those pieces to the LLM\",\"description\":\"The LLM receives the question plus the retrieved information.\"},{\"label\":\"4. The LLM writes the answer\",\"description\":\"It uses the retrieved information as context for the response.\"}]},\"tunes\":{}},{\"id\":\"rag-that-is-it\",\"type\":\"paragraph\",\"data\":{\"text\":\"That is RAG.\"},\"tunes\":{}},{\"id\":\"rag-name\",\"type\":\"paragraph\",\"data\":{\"text\":\"The full name is Retrieval-Augmented Generation. Retrieval means finding the relevant information. Augmented means adding that information to the model's context. Generation means the LLM writes the final answer.\"},\"tunes\":{}},{\"id\":\"h-example\",\"type\":\"header\",\"data\":{\"text\":\"A very simple example\",\"level\":2},\"tunes\":{}},{\"id\":\"p-ex-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Imagine you have a local knowledge base about a game.\"},\"tunes\":{}},{\"id\":\"kb-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Knowledge base contains\",\"Example\"],[\"Weapons\",\"AKM uses 7.62 mm ammunition\"],[\"Healing items\",\"Med Kit restores health\"],[\"Attachments\",\"This attachment works with these weapons\"],[\"Map rules\",\"This zone behaves in this way\"]]},\"tunes\":{}},{\"id\":\"p-ex-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"You ask: “Which ammunition does the AKM use?”\"},\"tunes\":{}},{\"id\":\"p-ex-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"RAG searches the knowledge base and finds the entry about the AKM. It gives that small piece of information to the LLM. The LLM then answers: “The AKM uses 7.62 mm ammunition.”\"},\"tunes\":{}},{\"id\":\"p-ex-4\",\"type\":\"paragraph\",\"data\":{\"text\":\"The LLM did not need the entire database. RAG only brought the useful part.\"},\"tunes\":{}},{\"id\":\"h-state\",\"type\":\"header\",\"data\":{\"text\":\"Now the important part: RAG is not the current state\",\"level\":2},\"tunes\":{}},{\"id\":\"p-state-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"This is where many explanations become confusing.\"},\"tunes\":{}},{\"id\":\"p-state-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"RAG usually gives the AI knowledge. A state system gives the AI facts about what is true right now.\"},\"tunes\":{}},{\"id\":\"knowledge-state\",\"type\":\"comparison\",\"data\":{\"title\":\"Knowledge vs current state\",\"layout\":\"table\",\"columns\":[{\"id\":\"knowledge\",\"label\":\"RAG \u002F knowledge\"},{\"id\":\"state\",\"label\":\"Current state\"}],\"rows\":[{\"id\":\"weapon\",\"label\":\"Weapon\",\"values\":[\"\",\"\"]},{\"id\":\"ammo\",\"label\":\"Ammunition\",\"values\":[\"\",\"\"]},{\"id\":\"health\",\"label\":\"Health\",\"values\":[\"\",\"\"]},{\"id\":\"enemy\",\"label\":\"Enemy\",\"values\":[\"\",\"\"]}]},\"tunes\":{}},{\"id\":\"dont-mix\",\"type\":\"callout\",\"data\":{\"variant\":\"warning\",\"title\":\"Do not mix these two\",\"body\":\"RAG answers: \u003Cstrong>What is generally true?\u003C\u002Fstrong>\u003Cbr>State answers: \u003Cstrong>What is true right now?\u003C\u002Fstrong>\"},\"tunes\":{}},{\"id\":\"h-state-db\",\"type\":\"header\",\"data\":{\"text\":\"What is a state database?\",\"level\":2},\"tunes\":{}},{\"id\":\"p-statedb-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A state database or state store is simply a place where the application keeps current facts.\"},\"tunes\":{}},{\"id\":\"p-statedb-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"In a game, the engine already knows things such as your health, position, inventory, ammunition, current mission, nearby objects and enemy status. An AI system can expose selected parts of that state to the model.\"},\"tunes\":{}},{\"id\":\"p-statedb-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"In a business application, the same idea could be an order database, a customer record, a project status or the current value of a sensor.\"},\"tunes\":{}},{\"id\":\"p-statedb-4\",\"type\":\"paragraph\",\"data\":{\"text\":\"The state is created by the application itself as things happen. If you lose health, the game updates the health value. If you pick up ammunition, the inventory changes. If an order is paid, the business system changes the order status.\"},\"tunes\":{}},{\"id\":\"state-rule\",\"type\":\"callout\",\"data\":{\"variant\":\"info\",\"title\":\"Simple rule\",\"body\":\"The application creates and updates \u003Cstrong>state\u003C\u002Fstrong>. RAG searches \u003Cstrong>knowledge\u003C\u002Fstrong>. The LLM uses both to decide what to say or do.\"},\"tunes\":{}},{\"id\":\"h-together\",\"type\":\"header\",\"data\":{\"text\":\"How the three pieces work together\",\"level\":2},\"tunes\":{}},{\"id\":\"together-flow\",\"type\":\"processFlow\",\"data\":{\"title\":\"LLM + state + RAG\",\"orientation\":\"auto\",\"steps\":[{\"label\":\"1. Current state\",\"description\":\"The application tells the AI what is true now: health 41%, AKM equipped, 23 rounds.\"},{\"label\":\"2. RAG\",\"description\":\"The system retrieves useful knowledge: how the weapon works, which healing item is available, or a relevant rule.\"},{\"label\":\"3. LLM\",\"description\":\"The model receives the question, current state and retrieved knowledge.\"},{\"label\":\"4. Reasoning\",\"description\":\"The LLM combines those inputs and decides what answer or high-level action makes sense.\"},{\"label\":\"5. Application\",\"description\":\"If an action is required, the application or game engine executes it and updates the state again.\"}]},\"tunes\":{}},{\"id\":\"p-arch-intro\",\"type\":\"paragraph\",\"data\":{\"text\":\"So the basic architecture is:\"},\"tunes\":{}},{\"id\":\"simple-architecture\",\"type\":\"callout\",\"data\":{\"variant\":\"note\",\"title\":\"The simplest architecture\",\"body\":\"\u003Cstrong>State = what is true now\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>RAG = useful knowledge\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>LLM = understands, reasons and writes\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Application = performs the real action\u003C\u002Fstrong>\"},\"tunes\":{}},{\"id\":\"h-vector\",\"type\":\"header\",\"data\":{\"text\":\"Does RAG always use a vector database?\",\"level\":2},\"tunes\":{}},{\"id\":\"p-vector-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"No.\"},\"tunes\":{}},{\"id\":\"p-vector-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"A vector database is a common way to build semantic search, but it is not the definition of RAG.\"},\"tunes\":{}},{\"id\":\"p-vector-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The important part is retrieval: the system finds relevant external information and adds it to the LLM's context before the answer is generated.\"},\"tunes\":{}},{\"id\":\"p-vector-4\",\"type\":\"paragraph\",\"data\":{\"text\":\"OpenAI's File Search, for example, can work with files stored in vector stores. Files are chunked into smaller pieces so the system can retrieve the parts that are relevant to a question. That is one implementation of the same basic idea.\"},\"tunes\":{}},{\"id\":\"h-embedding\",\"type\":\"header\",\"data\":{\"text\":\"What is an embedding, in plain English?\",\"level\":2},\"tunes\":{}},{\"id\":\"p-emb-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"You do not need to understand embeddings to understand RAG.\"},\"tunes\":{}},{\"id\":\"p-emb-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"But the simple version is this: an embedding is a numerical representation of meaning. It helps a search system find text that is conceptually similar even when the words are not exactly the same.\"},\"tunes\":{}},{\"id\":\"p-emb-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"For example, a normal keyword search may look for the exact words “car repair.” Semantic search can also understand that “fix my vehicle” is about a similar topic.\"},\"tunes\":{}},{\"id\":\"p-emb-4\",\"type\":\"paragraph\",\"data\":{\"text\":\"That makes embeddings useful for RAG, but RAG can also use keyword search, database queries or a hybrid of several methods.\"},\"tunes\":{}},{\"id\":\"h-memory\",\"type\":\"header\",\"data\":{\"text\":\"RAG is not memory either\",\"level\":2},\"tunes\":{}},{\"id\":\"p-memory-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Memory is another concept that is often mixed together with RAG.\"},\"tunes\":{}},{\"id\":\"p-memory-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Memory is usually information the system keeps about previous interactions or previous events. RAG is the mechanism used to retrieve relevant knowledge when it is needed.\"},\"tunes\":{}},{\"id\":\"parts-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Part\",\"Simple meaning\"],[\"LLM\",\"The part that understands and generates language\"],[\"RAG\",\"The part that looks up relevant knowledge before the answer\"],[\"Knowledge base\",\"The information RAG can search\"],[\"State\",\"What is true right now in the application or world\"],[\"Memory\",\"Information kept from previous interactions or events\"],[\"Tool \u002F action\",\"Something the AI is allowed to call or ask the application to do\"],[\"Context\",\"The information currently placed in front of the LLM for this request\"]]},\"tunes\":{}},{\"id\":\"h-pubg\",\"type\":\"header\",\"data\":{\"text\":\"A real game example: PUBG Ally\",\"level\":2},\"tunes\":{}},{\"id\":\"p-pubg-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"PUBG Ally is a useful example because it makes the difference visible.\"},\"tunes\":{}},{\"id\":\"p-pubg-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"KRAFTON describes live match state as a separate source of truth. The game exposes current facts through observation tools: current weapon, ammunition, health, safe-zone status, nearby items and combat situation.\"},\"tunes\":{}},{\"id\":\"p-pubg-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"Knowledge lookup is a different job. The system can use curated knowledge about weapons, attachments, items and rules. NVIDIA's ACE Game Agent SDK also exposes a separate RAG API for retrieving knowledge from developer-built databases.\"},\"tunes\":{}},{\"id\":\"p-pubg-4\",\"type\":\"paragraph\",\"data\":{\"text\":\"That gives us the clean separation: the game engine says what is happening now, retrieval provides relevant knowledge, and the language model decides what the information means.\"},\"tunes\":{}},{\"id\":\"ref-pubg\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Ffigure.rocks\u002Fblog\u002Fpubg-ally-shows-why-ai-teammates-need-two-brains-fast-reflexes-and-slow-reasoning\",\"title\":\"PUBG Ally Shows Why AI Teammates Need Two Brains: Fast Reflexes and Slow Reasoning\",\"excerpt\":\"A practical game example showing how live state, language reasoning and deterministic game-side control can work together.\",\"ctaLabel\":\"Read the PUBG Ally architecture article\"},\"tunes\":{}},{\"id\":\"h-complete\",\"type\":\"header\",\"data\":{\"text\":\"One complete example\",\"level\":2},\"tunes\":{}},{\"id\":\"p-complete-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Imagine you tell an AI teammate: “I am low on health. Should we attack?”\"},\"tunes\":{}},{\"id\":\"complete-flow\",\"type\":\"processFlow\",\"data\":{\"title\":\"What happens next\",\"orientation\":\"auto\",\"steps\":[{\"label\":\"State\",\"description\":\"The game reports: health 24%, one enemy nearby, two healing items available.\"},{\"label\":\"RAG\",\"description\":\"The knowledge system retrieves the relevant rules for the healing item and perhaps information about the current weapon or tactical mechanic.\"},{\"label\":\"LLM\",\"description\":\"The model combines your request, the current state and the retrieved knowledge.\"},{\"label\":\"Decision\",\"description\":\"It concludes that healing first is safer than attacking immediately.\"},{\"label\":\"Tool \u002F game engine\",\"description\":\"The agent requests a legal game action such as moving to cover or using the healing item.\"},{\"label\":\"New state\",\"description\":\"The game executes the action and reports the updated situation back to the agent.\"}]},\"tunes\":{}},{\"id\":\"p-complete-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"RAG did not control the character. The state database did not reason. The LLM did not directly change the game. Each part had one job.\"},\"tunes\":{}},{\"id\":\"h-why\",\"type\":\"header\",\"data\":{\"text\":\"Why use RAG at all?\",\"level\":2},\"tunes\":{}},{\"id\":\"p-why-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Because putting every document, rule and database record into every prompt would be slow, expensive and often confusing.\"},\"tunes\":{}},{\"id\":\"p-why-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"RAG lets the system select only the information that is useful for the current question.\"},\"tunes\":{}},{\"id\":\"p-why-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"It also lets you update the knowledge base without retraining the entire language model. Change the document or database, rebuild or refresh the index when necessary, and the next retrieval can use the newer information.\"},\"tunes\":{}},{\"id\":\"h-not-guarantee\",\"type\":\"header\",\"data\":{\"text\":\"What RAG does not guarantee\",\"level\":2},\"tunes\":{}},{\"id\":\"p-not-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"RAG can improve grounding, but it does not make an answer automatically correct.\"},\"tunes\":{}},{\"id\":\"p-not-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The retrieval step can find the wrong document. The correct document can be outdated. The LLM can misunderstand good evidence. Or the current state can have changed.\"},\"tunes\":{}},{\"id\":\"p-not-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"A reliable system therefore has to validate retrieval, state freshness and the model's final reasoning separately.\"},\"tunes\":{}},{\"id\":\"h-mental\",\"type\":\"header\",\"data\":{\"text\":\"The easiest mental model to remember\",\"level\":2},\"tunes\":{}},{\"id\":\"mental-table\",\"type\":\"comparison\",\"data\":{\"title\":\"Think of an AI system like a person at a desk\",\"layout\":\"table\",\"columns\":[{\"id\":\"analogy\",\"label\":\"Analogy\"},{\"id\":\"system\",\"label\":\"AI system\"}],\"rows\":[{\"id\":\"brain\",\"label\":\"Person thinking\",\"values\":[\"\",\"\"]},{\"id\":\"library\",\"label\":\"Finding a reference book\",\"values\":[\"\",\"\"]},{\"id\":\"books\",\"label\":\"Books on the shelf\",\"values\":[\"\",\"\"]},{\"id\":\"dashboard\",\"label\":\"Current dashboard or instrument panel\",\"values\":[\"\",\"\"]},{\"id\":\"notes\",\"label\":\"Notes from earlier meetings\",\"values\":[\"\",\"\"]},{\"id\":\"hands\",\"label\":\"Doing something in the real world\",\"values\":[\"\",\"\"]}]},\"tunes\":{}},{\"id\":\"remember\",\"type\":\"callout\",\"data\":{\"variant\":\"success\",\"title\":\"If you remember only this\",\"body\":\"\u003Cstrong>LLM = brain.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>RAG = librarian.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Knowledge base = library.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>State = what the dashboard says right now.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Tools = the hands that can actually do something.\u003C\u002Fstrong>\"},\"tunes\":{}},{\"id\":\"h-conclusion\",\"type\":\"header\",\"data\":{\"text\":\"Conclusion\",\"level\":2},\"tunes\":{}},{\"id\":\"p-conc-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"RAG is much less mysterious once the parts are separated.\"},\"tunes\":{}},{\"id\":\"p-conc-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The LLM understands and generates language. The application maintains current state. The knowledge base stores information. RAG finds the useful part of that information and puts it into the LLM's context. Tools or the application perform real actions.\"},\"tunes\":{}},{\"id\":\"p-conc-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"That is the basic architecture behind many modern AI assistants and agents.\"},\"tunes\":{}},{\"id\":\"h-faq\",\"type\":\"header\",\"data\":{\"text\":\"FAQ\",\"level\":2},\"tunes\":{}},{\"id\":\"faq\",\"type\":\"faq\",\"data\":{\"title\":\"RAG in plain English\",\"items\":[{\"id\":\"faq1\",\"question\":\"What is RAG in simple terms?\",\"answer\":\"RAG is a step where an AI searches a knowledge source for relevant information before the language model writes its answer.\"},{\"id\":\"faq2\",\"question\":\"Does RAG need the Internet?\",\"answer\":\"No. The knowledge base can be completely local on your computer or server.\"},{\"id\":\"faq3\",\"question\":\"Is RAG the same as a database?\",\"answer\":\"No. The database or files contain the information. RAG is the retrieval process that finds the useful part and gives it to the LLM.\"},{\"id\":\"faq4\",\"question\":\"Is RAG the same as memory?\",\"answer\":\"No. Memory usually stores previous interactions or events. RAG retrieves relevant knowledge when it is needed.\"},{\"id\":\"faq5\",\"question\":\"Is current application state part of RAG?\",\"answer\":\"Not necessarily. Current state is usually obtained directly from the application or a state store. RAG is better understood as retrieval from a knowledge source.\"},{\"id\":\"faq6\",\"question\":\"Does RAG make AI answers correct?\",\"answer\":\"No. It can provide better evidence, but retrieval can still be wrong or outdated and the LLM can still reason incorrectly.\"}]},\"tunes\":{}},{\"id\":\"h-glossary\",\"type\":\"header\",\"data\":{\"text\":\"Glossary\",\"level\":2},\"tunes\":{}},{\"id\":\"glossary\",\"type\":\"glossary\",\"data\":{\"title\":\"The basic terms\",\"entries\":[{\"term\":\"LLM\",\"definition\":\"A language model that understands and generates text and can reason over information placed in its context.\",\"anchor\":\"llm\"},{\"term\":\"RAG\",\"definition\":\"Retrieval-Augmented Generation: retrieving relevant external information and adding it to the model's context before generating an answer.\",\"anchor\":\"rag\"},{\"term\":\"Knowledge base\",\"definition\":\"The files, documents, records or other information that retrieval can search.\",\"anchor\":\"knowledge-base\"},{\"term\":\"State\",\"definition\":\"The current facts of an application, system or world at a particular moment.\",\"anchor\":\"state\"},{\"term\":\"Context\",\"definition\":\"The information currently supplied to the language model for one request or reasoning step.\",\"anchor\":\"context\"},{\"term\":\"Embedding\",\"definition\":\"A numerical representation of meaning that can help semantic search find conceptually similar information.\",\"anchor\":\"embedding\"}]},\"tunes\":{}},{\"id\":\"h-sources\",\"type\":\"header\",\"data\":{\"text\":\"Primary sources\",\"level\":2},\"tunes\":{}},{\"id\":\"src-openai-vector\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fplatform.openai.com\u002Fdocs\u002Fapi-reference\u002Fvector-stores-files\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — Vector Store Files\",\"description\":\"Official documentation showing how files can be attached to vector stores, chunked and made available to file-search retrieval.\"}},\"tunes\":{}},{\"id\":\"src-openai-quickstart\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fplatform.openai.com\u002Fdocs\u002Fquickstart\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — Developer Quickstart\",\"description\":\"Official OpenAI documentation describing tools such as file search for giving models access to external information.\"}},\"tunes\":{}},{\"id\":\"src-nvidia-ace\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdeveloper.nvidia.com\u002Face-for-games\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"NVIDIA Developer — ACE for Games\",\"description\":\"Official NVIDIA documentation describing separate Agent, Chat and RAG APIs for connecting game characters to game state, contextual knowledge and model-driven actions.\"}},\"tunes\":{}},{\"id\":\"src-nvidia-pubg\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdeveloper.nvidia.com\u002Fblog\u002Fhow-krafton-built-pubg-ally-a-co-playable-character-powered-by-nvidia-ace\u002F\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"NVIDIA Developer — How KRAFTON Built PUBG Ally\",\"description\":\"Official technical explanation separating live match state from knowledge lookup and language-model reasoning.\"}},\"tunes\":{}}],\"version\":\"2.31.6\"}",{"time":1272,"blocks":1273,"version":1791},1790377494031,[1274,1278,1283,1287,1291,1295,1299,1303,1307,1312,1316,1320,1324,1328,1333,1337,1354,1358,1362,1366,1370,1389,1393,1397,1401,1405,1409,1413,1435,1440,1444,1448,1452,1456,1460,1464,1468,1486,1490,1495,1499,1503,1507,1511,1515,1519,1523,1527,1531,1535,1539,1543,1547,1573,1577,1581,1585,1589,1593,1599,1603,1607,1627,1631,1635,1639,1643,1647,1651,1655,1659,1663,1667,1695,1700,1704,1708,1712,1716,1719,1742,1746,1763,1767,1773,1779,1785],{"id":214,"data":1275,"type":217,"tunes":1277},{"text":1276},"RAG sounds complicated because the name is complicated. The idea is not. RAG simply means: before the AI answers, it first looks up relevant information from a knowledge source and gives that information to the language model.",{},{"id":220,"data":1279,"type":225,"tunes":1282},{"body":1280,"title":1281,"variant":224},"\u003Cstrong>RAG is the step where an AI searches a knowledge base for useful information before the LLM writes the answer.\u003C\u002Fstrong>","RAG in one sentence",{},{"id":228,"data":1284,"type":217,"tunes":1286},{"text":1285},"Think of an LLM as a smart person sitting at a desk. RAG is the librarian who brings the right page from the right book. The LLM then reads that page and answers you.",{},{"id":233,"data":1288,"type":238,"tunes":1290},{"title":1289,"maxLevel":236,"minLevel":237},"Contents",{},{"id":241,"data":1292,"type":41,"tunes":1294},{"text":1293,"level":237},"First: what does the LLM do?",{},{"id":246,"data":1296,"type":217,"tunes":1298},{"text":1297},"The LLM is the part that understands language and produces language. It can read your question, understand instructions, compare information, explain something and write an answer.",{},{"id":251,"data":1300,"type":217,"tunes":1302},{"text":1301},"But the LLM does not automatically know what is currently inside your company database, your game session, your private documents or a file you created five minutes ago.",{},{"id":256,"data":1304,"type":217,"tunes":1306},{"text":1305},"It only knows what is already inside the model plus whatever information the application gives it in the current request.",{},{"id":261,"data":1308,"type":225,"tunes":1311},{"body":1309,"title":1310,"variant":265},"The LLM \u003Cstrong>thinks and writes\u003C\u002Fstrong>. It does not automatically own all of your current data.","Simple rule",{},{"id":268,"data":1313,"type":41,"tunes":1315},{"text":1314,"level":237},"Then: what is the knowledge base?",{},{"id":273,"data":1317,"type":217,"tunes":1319},{"text":1318},"A knowledge base is simply information the application can search.",{},{"id":278,"data":1321,"type":217,"tunes":1323},{"text":1322},"It could contain PDFs, manuals, product documentation, support articles, contracts, game rules, weapon data, internal company documents, database records or other text.",{},{"id":283,"data":1325,"type":217,"tunes":1327},{"text":1326},"The knowledge base can be local on your own machine. It can be on a server. It can be in a vector database. It can also be built from normal files. RAG does not mean Internet.",{},{"id":288,"data":1329,"type":225,"tunes":1332},{"body":1330,"title":1331,"variant":292},"\u003Cstrong>RAG does not require the Internet.\u003C\u002Fstrong> The information can be completely local.","Important",{},{"id":295,"data":1334,"type":41,"tunes":1336},{"text":1335,"level":237},"So what does RAG actually do?",{},{"id":300,"data":1338,"type":317,"tunes":1353},{"steps":1339,"title":1352,"orientation":316},[1340,1343,1346,1349],{"label":1341,"description":1342},"1. You ask a question","For example: Which ammunition does this weapon use?",{"label":1344,"description":1345},"2. RAG searches the knowledge base","The system looks for the small pieces of information most relevant to your question.",{"label":1347,"description":1348},"3. RAG gives those pieces to the LLM","The LLM receives the question plus the retrieved information.",{"label":1350,"description":1351},"4. The LLM writes the answer","It uses the retrieved information as context for the response.","The whole RAG process",{},{"id":320,"data":1355,"type":217,"tunes":1357},{"text":1356},"That is RAG.",{},{"id":325,"data":1359,"type":217,"tunes":1361},{"text":1360},"The full name is Retrieval-Augmented Generation. Retrieval means finding the relevant information. Augmented means adding that information to the model's context. Generation means the LLM writes the final answer.",{},{"id":330,"data":1363,"type":41,"tunes":1365},{"text":1364,"level":237},"A very simple example",{},{"id":335,"data":1367,"type":217,"tunes":1369},{"text":1368},"Imagine you have a local knowledge base about a game.",{},{"id":340,"data":1371,"type":358,"tunes":1388},{"content":1372,"stretched":42,"withHeadings":13},[1373,1376,1379,1382,1385],[1374,1375],"Knowledge base contains","Example",[1377,1378],"Weapons","AKM uses 7.62 mm ammunition",[1380,1381],"Healing items","Med Kit restores health",[1383,1384],"Attachments","This attachment works with these weapons",[1386,1387],"Map rules","This zone behaves in this way",{},{"id":361,"data":1390,"type":217,"tunes":1392},{"text":1391},"You ask: “Which ammunition does the AKM use?”",{},{"id":366,"data":1394,"type":217,"tunes":1396},{"text":1395},"RAG searches the knowledge base and finds the entry about the AKM. It gives that small piece of information to the LLM. The LLM then answers: “The AKM uses 7.62 mm ammunition.”",{},{"id":371,"data":1398,"type":217,"tunes":1400},{"text":1399},"The LLM did not need the entire database. RAG only brought the useful part.",{},{"id":376,"data":1402,"type":41,"tunes":1404},{"text":1403,"level":237},"Now the important part: RAG is not the current state",{},{"id":381,"data":1406,"type":217,"tunes":1408},{"text":1407},"This is where many explanations become confusing.",{},{"id":386,"data":1410,"type":217,"tunes":1412},{"text":1411},"RAG usually gives the AI knowledge. A state system gives the AI facts about what is true right now.",{},{"id":391,"data":1414,"type":419,"tunes":1434},{"rows":1415,"title":1428,"layout":358,"columns":1429},[1416,1419,1422,1425],{"id":395,"label":1417,"values":1418},"Weapon",[398,398],{"id":400,"label":1420,"values":1421},"Ammunition",[398,398],{"id":404,"label":1423,"values":1424},"Health",[398,398],{"id":408,"label":1426,"values":1427},"Enemy",[398,398],"Knowledge vs current state",[1430,1432],{"id":414,"label":1431},"RAG \u002F knowledge",{"id":417,"label":1433},"Current state",{},{"id":422,"data":1436,"type":225,"tunes":1439},{"body":1437,"title":1438,"variant":426},"RAG answers: \u003Cstrong>What is generally true?\u003C\u002Fstrong>\u003Cbr>State answers: \u003Cstrong>What is true right now?\u003C\u002Fstrong>","Do not mix these two",{},{"id":429,"data":1441,"type":41,"tunes":1443},{"text":1442,"level":237},"What is a state database?",{},{"id":434,"data":1445,"type":217,"tunes":1447},{"text":1446},"A state database or state store is simply a place where the application keeps current facts.",{},{"id":439,"data":1449,"type":217,"tunes":1451},{"text":1450},"In a game, the engine already knows things such as your health, position, inventory, ammunition, current mission, nearby objects and enemy status. An AI system can expose selected parts of that state to the model.",{},{"id":444,"data":1453,"type":217,"tunes":1455},{"text":1454},"In a business application, the same idea could be an order database, a customer record, a project status or the current value of a sensor.",{},{"id":449,"data":1457,"type":217,"tunes":1459},{"text":1458},"The state is created by the application itself as things happen. If you lose health, the game updates the health value. If you pick up ammunition, the inventory changes. If an order is paid, the business system changes the order status.",{},{"id":454,"data":1461,"type":225,"tunes":1463},{"body":1462,"title":1310,"variant":224},"The application creates and updates \u003Cstrong>state\u003C\u002Fstrong>. RAG searches \u003Cstrong>knowledge\u003C\u002Fstrong>. The LLM uses both to decide what to say or do.",{},{"id":459,"data":1465,"type":41,"tunes":1467},{"text":1466,"level":237},"How the three pieces work together",{},{"id":464,"data":1469,"type":317,"tunes":1485},{"steps":1470,"title":1484,"orientation":316},[1471,1474,1476,1478,1481],{"label":1472,"description":1473},"1. Current state","The application tells the AI what is true now: health 41%, AKM equipped, 23 rounds.",{"label":471,"description":1475},"The system retrieves useful knowledge: how the weapon works, which healing item is available, or a relevant rule.",{"label":474,"description":1477},"The model receives the question, current state and retrieved knowledge.",{"label":1479,"description":1480},"4. Reasoning","The LLM combines those inputs and decides what answer or high-level action makes sense.",{"label":1482,"description":1483},"5. Application","If an action is required, the application or game engine executes it and updates the state again.","LLM + state + RAG",{},{"id":485,"data":1487,"type":217,"tunes":1489},{"text":1488},"So the basic architecture is:",{},{"id":490,"data":1491,"type":225,"tunes":1494},{"body":1492,"title":1493,"variant":265},"\u003Cstrong>State = what is true now\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>RAG = useful knowledge\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>LLM = understands, reasons and writes\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Application = performs the real action\u003C\u002Fstrong>","The simplest architecture",{},{"id":496,"data":1496,"type":41,"tunes":1498},{"text":1497,"level":237},"Does RAG always use a vector database?",{},{"id":501,"data":1500,"type":217,"tunes":1502},{"text":1501},"No.",{},{"id":506,"data":1504,"type":217,"tunes":1506},{"text":1505},"A vector database is a common way to build semantic search, but it is not the definition of RAG.",{},{"id":511,"data":1508,"type":217,"tunes":1510},{"text":1509},"The important part is retrieval: the system finds relevant external information and adds it to the LLM's context before the answer is generated.",{},{"id":516,"data":1512,"type":217,"tunes":1514},{"text":1513},"OpenAI's File Search, for example, can work with files stored in vector stores. Files are chunked into smaller pieces so the system can retrieve the parts that are relevant to a question. That is one implementation of the same basic idea.",{},{"id":521,"data":1516,"type":41,"tunes":1518},{"text":1517,"level":237},"What is an embedding, in plain English?",{},{"id":526,"data":1520,"type":217,"tunes":1522},{"text":1521},"You do not need to understand embeddings to understand RAG.",{},{"id":531,"data":1524,"type":217,"tunes":1526},{"text":1525},"But the simple version is this: an embedding is a numerical representation of meaning. It helps a search system find text that is conceptually similar even when the words are not exactly the same.",{},{"id":536,"data":1528,"type":217,"tunes":1530},{"text":1529},"For example, a normal keyword search may look for the exact words “car repair.” Semantic search can also understand that “fix my vehicle” is about a similar topic.",{},{"id":541,"data":1532,"type":217,"tunes":1534},{"text":1533},"That makes embeddings useful for RAG, but RAG can also use keyword search, database queries or a hybrid of several methods.",{},{"id":546,"data":1536,"type":41,"tunes":1538},{"text":1537,"level":237},"RAG is not memory either",{},{"id":551,"data":1540,"type":217,"tunes":1542},{"text":1541},"Memory is another concept that is often mixed together with RAG.",{},{"id":556,"data":1544,"type":217,"tunes":1546},{"text":1545},"Memory is usually information the system keeps about previous interactions or previous events. RAG is the mechanism used to retrieve relevant knowledge when it is needed.",{},{"id":561,"data":1548,"type":358,"tunes":1572},{"content":1549,"stretched":42,"withHeadings":13},[1550,1553,1555,1557,1560,1563,1566,1569],[1551,1552],"Part","Simple meaning",[568,1554],"The part that understands and generates language",[571,1556],"The part that looks up relevant knowledge before the answer",[1558,1559],"Knowledge base","The information RAG can search",[1561,1562],"State","What is true right now in the application or world",[1564,1565],"Memory","Information kept from previous interactions or events",[1567,1568],"Tool \u002F action","Something the AI is allowed to call or ask the application to do",[1570,1571],"Context","The information currently placed in front of the LLM for this request",{},{"id":590,"data":1574,"type":41,"tunes":1576},{"text":1575,"level":237},"A real game example: PUBG Ally",{},{"id":595,"data":1578,"type":217,"tunes":1580},{"text":1579},"PUBG Ally is a useful example because it makes the difference visible.",{},{"id":600,"data":1582,"type":217,"tunes":1584},{"text":1583},"KRAFTON describes live match state as a separate source of truth. The game exposes current facts through observation tools: current weapon, ammunition, health, safe-zone status, nearby items and combat situation.",{},{"id":605,"data":1586,"type":217,"tunes":1588},{"text":1587},"Knowledge lookup is a different job. The system can use curated knowledge about weapons, attachments, items and rules. NVIDIA's ACE Game Agent SDK also exposes a separate RAG API for retrieving knowledge from developer-built databases.",{},{"id":610,"data":1590,"type":217,"tunes":1592},{"text":1591},"That gives us the clean separation: the game engine says what is happening now, retrieval provides relevant knowledge, and the language model decides what the information means.",{},{"id":615,"data":1594,"type":621,"tunes":1598},{"url":617,"title":1595,"excerpt":1596,"ctaLabel":1597},"PUBG Ally Shows Why AI Teammates Need Two Brains: Fast Reflexes and Slow Reasoning","A practical game example showing how live state, language reasoning and deterministic game-side control can work together.","Read the PUBG Ally architecture article",{},{"id":624,"data":1600,"type":41,"tunes":1602},{"text":1601,"level":237},"One complete example",{},{"id":629,"data":1604,"type":217,"tunes":1606},{"text":1605},"Imagine you tell an AI teammate: “I am low on health. Should we attack?”",{},{"id":634,"data":1608,"type":317,"tunes":1626},{"steps":1609,"title":1625,"orientation":316},[1610,1612,1614,1616,1619,1622],{"label":1561,"description":1611},"The game reports: health 24%, one enemy nearby, two healing items available.",{"label":571,"description":1613},"The knowledge system retrieves the relevant rules for the healing item and perhaps information about the current weapon or tactical mechanic.",{"label":568,"description":1615},"The model combines your request, the current state and the retrieved knowledge.",{"label":1617,"description":1618},"Decision","It concludes that healing first is safer than attacking immediately.",{"label":1620,"description":1621},"Tool \u002F game engine","The agent requests a legal game action such as moving to cover or using the healing item.",{"label":1623,"description":1624},"New state","The game executes the action and reports the updated situation back to the agent.","What happens next",{},{"id":655,"data":1628,"type":217,"tunes":1630},{"text":1629},"RAG did not control the character. The state database did not reason. The LLM did not directly change the game. Each part had one job.",{},{"id":660,"data":1632,"type":41,"tunes":1634},{"text":1633,"level":237},"Why use RAG at all?",{},{"id":665,"data":1636,"type":217,"tunes":1638},{"text":1637},"Because putting every document, rule and database record into every prompt would be slow, expensive and often confusing.",{},{"id":670,"data":1640,"type":217,"tunes":1642},{"text":1641},"RAG lets the system select only the information that is useful for the current question.",{},{"id":675,"data":1644,"type":217,"tunes":1646},{"text":1645},"It also lets you update the knowledge base without retraining the entire language model. Change the document or database, rebuild or refresh the index when necessary, and the next retrieval can use the newer information.",{},{"id":680,"data":1648,"type":41,"tunes":1650},{"text":1649,"level":237},"What RAG does not guarantee",{},{"id":685,"data":1652,"type":217,"tunes":1654},{"text":1653},"RAG can improve grounding, but it does not make an answer automatically correct.",{},{"id":690,"data":1656,"type":217,"tunes":1658},{"text":1657},"The retrieval step can find the wrong document. The correct document can be outdated. The LLM can misunderstand good evidence. Or the current state can have changed.",{},{"id":695,"data":1660,"type":217,"tunes":1662},{"text":1661},"A reliable system therefore has to validate retrieval, state freshness and the model's final reasoning separately.",{},{"id":700,"data":1664,"type":41,"tunes":1666},{"text":1665,"level":237},"The easiest mental model to remember",{},{"id":705,"data":1668,"type":419,"tunes":1694},{"rows":1669,"title":1688,"layout":358,"columns":1689},[1670,1673,1676,1679,1682,1685],{"id":709,"label":1671,"values":1672},"Person thinking",[398,398],{"id":713,"label":1674,"values":1675},"Finding a reference book",[398,398],{"id":717,"label":1677,"values":1678},"Books on the shelf",[398,398],{"id":721,"label":1680,"values":1681},"Current dashboard or instrument panel",[398,398],{"id":725,"label":1683,"values":1684},"Notes from earlier meetings",[398,398],{"id":729,"label":1686,"values":1687},"Doing something in the real world",[398,398],"Think of an AI system like a person at a desk",[1690,1692],{"id":228,"label":1691},"Analogy",{"id":737,"label":1693},"AI system",{},{"id":741,"data":1696,"type":225,"tunes":1699},{"body":1697,"title":1698,"variant":292},"\u003Cstrong>LLM = brain.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>RAG = librarian.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Knowledge base = library.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>State = what the dashboard says right now.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Tools = the hands that can actually do something.\u003C\u002Fstrong>","If you remember only this",{},{"id":747,"data":1701,"type":41,"tunes":1703},{"text":1702,"level":237},"Conclusion",{},{"id":752,"data":1705,"type":217,"tunes":1707},{"text":1706},"RAG is much less mysterious once the parts are separated.",{},{"id":757,"data":1709,"type":217,"tunes":1711},{"text":1710},"The LLM understands and generates language. The application maintains current state. The knowledge base stores information. RAG finds the useful part of that information and puts it into the LLM's context. Tools or the application perform real actions.",{},{"id":762,"data":1713,"type":217,"tunes":1715},{"text":1714},"That is the basic architecture behind many modern AI assistants and agents.",{},{"id":767,"data":1717,"type":41,"tunes":1718},{"text":769,"level":237},{},{"id":772,"data":1720,"type":772,"tunes":1741},{"items":1721,"title":1740},[1722,1725,1728,1731,1734,1737],{"id":776,"answer":1723,"question":1724},"RAG is a step where an AI searches a knowledge source for relevant information before the language model writes its answer.","What is RAG in simple terms?",{"id":780,"answer":1726,"question":1727},"No. The knowledge base can be completely local on your computer or server.","Does RAG need the Internet?",{"id":784,"answer":1729,"question":1730},"No. The database or files contain the information. RAG is the retrieval process that finds the useful part and gives it to the LLM.","Is RAG the same as a database?",{"id":788,"answer":1732,"question":1733},"No. Memory usually stores previous interactions or events. RAG retrieves relevant knowledge when it is needed.","Is RAG the same as memory?",{"id":792,"answer":1735,"question":1736},"Not necessarily. Current state is usually obtained directly from the application or a state store. RAG is better understood as retrieval from a knowledge source.","Is current application state part of RAG?",{"id":796,"answer":1738,"question":1739},"No. It can provide better evidence, but retrieval can still be wrong or outdated and the LLM can still reason incorrectly.","Does RAG make AI answers correct?","RAG in plain English",{},{"id":802,"data":1743,"type":41,"tunes":1745},{"text":1744,"level":237},"Glossary",{},{"id":807,"data":1747,"type":807,"tunes":1762},{"title":1748,"entries":1749},"The basic terms",[1750,1752,1754,1756,1758,1760],{"term":568,"anchor":812,"definition":1751},"A language model that understands and generates text and can reason over information placed in its context.",{"term":571,"anchor":815,"definition":1753},"Retrieval-Augmented Generation: retrieving relevant external information and adding it to the model's context before generating an answer.",{"term":1558,"anchor":818,"definition":1755},"The files, documents, records or other information that retrieval can search.",{"term":1561,"anchor":417,"definition":1757},"The current facts of an application, system or world at a particular moment.",{"term":1570,"anchor":823,"definition":1759},"The information currently supplied to the language model for one request or reasoning step.",{"term":826,"anchor":827,"definition":1761},"A numerical representation of meaning that can help semantic search find conceptually similar information.",{},{"id":831,"data":1764,"type":41,"tunes":1766},{"text":1765,"level":237},"Primary sources",{},{"id":836,"data":1768,"type":843,"tunes":1772},{"link":838,"meta":1769},{"image":1770,"title":841,"description":1771},{"url":398},"Official documentation showing how files can be attached to vector stores, chunked and made available to file-search retrieval.",{},{"id":846,"data":1774,"type":843,"tunes":1778},{"link":848,"meta":1775},{"image":1776,"title":851,"description":1777},{"url":398},"Official OpenAI documentation describing tools such as file search for giving models access to external information.",{},{"id":855,"data":1780,"type":843,"tunes":1784},{"link":857,"meta":1781},{"image":1782,"title":860,"description":1783},{"url":398},"Official NVIDIA documentation describing separate Agent, Chat and RAG APIs for connecting game characters to game state, contextual knowledge and model-driven actions.",{},{"id":864,"data":1786,"type":843,"tunes":1790},{"link":866,"meta":1787},{"image":1788,"title":869,"description":1789},{"url":398},"Official technical explanation separating live match state from knowledge lookup and language-model reasoning.",{},"2.31.6","RAG sounds complicated, but the idea is simple: before an AI answers, it first looks up useful information from a knowledge source and gives that information to the language model. This guide explains RAG, LLMs, state, memory and tools using one simple mental model.","Post erfolgreich abgerufen",{"items":1795,"source":1879,"manualIds":1880,"manualMatchedIds":1881},[1796,1803,1809,1816,1823,1830,1837,1844,1851,1858,1865,1872],{"id":1797,"slug":1798,"title":1799,"excerpt":1800,"featuredImage":1801,"publishedAt":1802},"468","ai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context","KI-Agenten-Gedächtnis ist kein RAG: Wie man Gedächtnis, Retrieval, Zustand und Kontext voneinander trennt","Agentengedächtnis, RAG, Zustand und Kontext werden oft so verwendet, als wären sie austauschbar. Das sind sie nicht. Dieses praktische Architekturmodell trennt die vier Schichten, zeigt, wohin jede gehört, und erklärt, was kaputtgeht, wenn Systeme sie zu einer einzigen zusammenfassen.","\u002Fuploads\u002F2026\u002F09\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context-1790350560308-np0xy6.webp","2026-09-25T11:34:00.000Z",{"id":1804,"slug":1805,"title":1806,"excerpt":9,"featuredImage":1807,"publishedAt":1808},"369","git-with-automatic-upload-and-synchronization-to-a-production-server","Git with automatic upload and synchronization to a production server","\u002Fuploads\u002F2024\u002F05\u002Fstep-by-step-guide-illustration-showing-the-process-of-setting-up-Git-with-auto-upload-and-synchronization-to-a-production-server-large.webp","2024-05-28T22:48:00.000Z",{"id":1810,"slug":1811,"title":1812,"excerpt":1813,"featuredImage":1814,"publishedAt":1815},"459","ollama-is-not-the-product-building-production-ready-open-llm-applications","Ollama ist nicht das Produkt: Entwicklung produktionsreifer Open-LLM-Anwendungen","Das Ausführen eines lokalen Modells mit Ollama ist einfach. Das Erstellen einer produktionsreifen Open-LLM-Anwendung ist schwieriger: Es erfordert RAG, Zugriffskontrolle, Anbieterabstraktion, Evaluierung, Protokollierung, Bereitstellungsdisziplin und eine kontrollierte Anwendungsschicht um das Modell herum.","\u002Fuploads\u002F2026\u002F06\u002Follama-is-not-the-product-building-production-ready-open-llm-applications-1782679361640-h0usqf.webp","2026-06-28T16:39:00.000Z",{"id":1817,"slug":1818,"title":1819,"excerpt":1820,"featuredImage":1821,"publishedAt":1822},"469","rag-failed-but-which-layer-actually-failed-a-diagnostic-method","RAG fehlgeschlagen – aber welche Ebene ist tatsächlich fehlgeschlagen? Eine diagnostische Methode","Wenn eine RAG-Antwort falsch ist, ist es zu vage, das Retrieval oder das Modell verantwortlich zu machen. Diese Diagnosemethode isoliert Quellenabdeckung, Query-Konstruktion, Retrieval, Ranking, Kontextzusammenstellung, Generierung, Evidenzzuordnung und Aktualität – sodass der tatsächliche Fehler reproduziert und behoben werden kann.","\u002Fuploads\u002F2026\u002F09\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method-1790350847177-pior4c.webp","2026-09-24T19:39:00.000Z",{"id":1824,"slug":1825,"title":1826,"excerpt":1827,"featuredImage":1828,"publishedAt":1829},"476","mcp-vs-a2a-vs-ucp-vs-ap2-vs-a2ui-the-agent-protocol-stack-explained","MCP vs A2A vs UCP vs AP2 vs A2UI: Der Agenten-Protokoll-Stack erklärt","MCP, A2A, UCP, AP2 und A2UI werden oft als konkurrierende Agentenstandards dargestellt. Sie lösen größtenteils unterschiedliche Interoperabilitätsprobleme. Dieser Leitfaden ordnet jedes Protokoll der Grenze zu, die es tatsächlich standardisiert—und zeigt, wie sie in einem Produktionssystem zusammenarbeiten können.","\u002Fuploads\u002F2026\u002F09\u002Fmcp-vs-a2a-vs-ucp-vs-ap2-vs-a2ui-the-agent-protocol-stack-explained-1790352625869-2ezle0.webp","2026-09-25T12:09:00.000Z",{"id":1831,"slug":1832,"title":1833,"excerpt":1834,"featuredImage":1835,"publishedAt":1836},"364","tipps-fuer-die-verbesserung-der-seo-suchmaschinenoptimierung","Meistern des SEO-Workflows: Essenzielle Optimierungsstrategien für organisches Wachstum","Ein strukturierter SEO-Workflow ist entscheidend für nachhaltiges organisches Wachstum. Lerne die zehn grundlegenden Strategien, von der Keyword-Recherche und technischen Optimierung bis hin zur Content-Qualität und Performance-Analyse.","\u002Fuploads\u002F2026\u002F03\u002Ftipps-fuer-die-verbesserung-der-seo-suchmaschinenoptimierung-1774866098131-hwkzrg.webp","2024-01-26T06:35:00.000Z",{"id":1838,"slug":1839,"title":1840,"excerpt":1841,"featuredImage":1842,"publishedAt":1843},"473","openai-agents-api-vs-agents-sdk-vs-responses-api-what-should-you-build-on-in-2026","OpenAI Agents API vs. Agents SDK vs. Responses API: Worauf sollten Sie 2026 aufbauen?","Der Agent-Stack von OpenAI hat sich im September 2026 geändert. Dieser Architekturleitfaden unterscheidet die Agents API, das Agents SDK, die Responses API und das Codex SDK nach Runtime-Ownership—sodass Teams die richtige Kontrollgrenze wählen können, anstatt Produktnamen zu vergleichen.","\u002Fuploads\u002F2026\u002F09\u002Fopenai-agents-api-vs-agents-sdk-vs-responses-api-what-should-you-build-on-in-2026-1790351846714-zi7lus.webp","2026-09-25T11:56:00.000Z",{"id":1845,"slug":1846,"title":1847,"excerpt":1848,"featuredImage":1849,"publishedAt":1850},"477","computer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system","Computer-Use-Agenten: Warum eine erfolgreiche Demo dennoch ein unzuverlässiges System sein kann","Computer-Use-Agenten können mittlerweile beeindruckende Browser- und Desktop-Workflows abschließen, aber ein erfolgreicher Durchlauf beweist Fähigkeit—nicht Zuverlässigkeit. Dieser Artikel zeigt, wie man Wiederholbarkeit, Umgebungsrobustheit, Steuerung über lange Zeithorizonte, Zustandsbewusstsein, Ergebnisüberprüfung und sichere Zielhandhabung testet.","\u002Fuploads\u002F2026\u002F09\u002Fcomputer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system-1790352854690-75qnrg.webp","2026-09-25T12:13:00.000Z",{"id":1852,"slug":1853,"title":1854,"excerpt":1855,"featuredImage":1856,"publishedAt":1857},"470","what-should-an-ai-agent-remember-forget-recompute-or-retrieve-again","Was sollte ein KI-Agent behalten, vergessen, neu berechnen oder erneut abrufen?","Langlaufende Agenten sollten sich nicht alles merken. Dieser Artikel bietet ein praktisches Lebenszyklusmodell für die Entscheidung, was in den dauerhaften Speicher gehört, was erneut abgerufen werden sollte, was sicherer neu zu berechnen ist und was ablaufen oder ersetzt werden sollte.","\u002Fuploads\u002F2026\u002F09\u002Fwhat-should-an-ai-agent-remember-forget-recompute-or-retrieve-again-1790351131087-iehz28.webp","2026-09-25T09:43:00.000Z",{"id":1859,"slug":1860,"title":1861,"excerpt":1862,"featuredImage":1863,"publishedAt":1864},"472","why-more-context-can-make-ai-answers-worse","Warum mehr Kontext KI-Antworten verschlechtern kann","Ein größeres Kontextfenster garantiert keine bessere Antwort. Dieser Artikel erklärt, wie Signalverwässerung, widersprüchliche Belege, veralteter Zustand, Positionssensitivität und verlustbehaftete Kompression die KI-Zuverlässigkeit verringern können—und stellt einen praktischen Context Pressure Test vor.","\u002Fuploads\u002F2026\u002F09\u002Fwhy-more-context-can-make-ai-answers-worse-1790351615793-2ntv2v.webp","2026-09-25T11:51:00.000Z",{"id":1866,"slug":1867,"title":1868,"excerpt":1869,"featuredImage":1870,"publishedAt":1871},"460","ai-agent-reliability-why-the-final-answer-is-not-enough","Zuverlässigkeit von KI-Agenten: Warum die endgültige Antwort nicht ausreicht","Korrekte Ausgabe beweist weder korrektes Denken, sichere Ausführung noch ein vertrauenswürdiges System.","\u002Fuploads\u002F2026\u002F09\u002Fai-agent-reliability-why-the-final-answer-is-not-enough-1788955466306-pl0qhz.webp","2026-09-09T04:01:00.000Z",{"id":1873,"slug":1874,"title":1875,"excerpt":1876,"featuredImage":1877,"publishedAt":1878},"467","the-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers","Die Antwortgültigkeitsgrenze: Die fehlende Schicht zwischen Relevanz und zuverlässigen KI-Antworten","Eine Quelle kann relevant und maßgeblich sein und dennoch falsch für die gestellte Frage. Die fehlende Ebene ist die Anwendbarkeit: die Bedingungen, unter denen eine Antwort gilt, und die Veränderungen, die erzwingen, dass sie überdacht werden muss. Dieser Artikel führt die Answer Validity Boundary als ein Quellendesign-Muster für Menschen, KI-Suche und RAG-Systeme ein.","\u002Fuploads\u002F2026\u002F09\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers-1790272901306-1g5jly.webp","2026-09-24T11:59:00.000Z","fallback",[],[]]