[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"portal-settings:stajic:zh":3,"public-menus:all":38,"post:what-is-rag-the-simplest-explanation-of-how-it-works:zh":205,"related:post:what-is-rag-the-simplest-explanation-of-how-it-works:zh:1":1800},{"statusCode":4,"data":5,"message":37},200,{"tenantId":6,"lang":7,"defaultLang":8,"siteUrl":9,"contactEmail":10,"brandName":11,"logoUrl":12,"siteName":11,"siteDescription":13,"ogImage":10,"robotsIndex":14,"socialLinks":10,"reservedSlugs":10,"seoPolicy":15},"stajic","zh","de","https:\u002F\u002Fstajic.de",null,"Stajic Platform","\u002FLogo_Planet.svg","Stajic Portal",true,{"branding":16,"relatedContent":17,"crossDomainLinks":18},{"logoUrl":12},{"enabled":14},[19,22,25,28,31,34],{"url":20,"label":21,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Ffigure.rocks","figure.rocks",{"url":23,"label":24,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Floving.rocks","loving.rocks",{"url":26,"label":27,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.com","bazify.com",{"url":29,"label":30,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.de","bazify.de",{"url":32,"label":33,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.at","bazify.at",{"url":35,"label":36,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.ba","bazify.ba","Portal settings resolved",[39,45],{"id":40,"name":41,"location":42,"isActive":14,"isDefault":43,"items":44},1,"main-navigation","header",false,[],{"id":46,"name":47,"location":48,"isActive":14,"isDefault":14,"items":49},4,"main-menu","sidebar",[50,66,79,93,103,118,133],{"id":51,"title":52,"url":60,"target":61,"icon":62,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":64,"portfolioId":10,"children":65},"item-18",{"de":53,"en":54,"es":55,"fr":56,"it":54,"ru":57,"sr":58,"zh":59},"Startseite","Home","Inicio","Accueil","Главная","Почетна","首页","\u002Ffull-stack-web-developer-munich-performance-seo-and-maintainable-builds","_self","i-lucide-home","page",111,[],{"id":67,"title":68,"url":75,"target":61,"icon":76,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":77,"portfolioId":10,"children":78},"item-22",{"de":69,"en":69,"es":70,"fr":69,"it":71,"ru":72,"sr":73,"zh":74},"Vision","Visión","Visione","Видение","Визија","想象","\u002Fueber-uns-webdesign-muenchen-webaplikation","i-lucide-eye",113,[],{"id":80,"title":81,"url":89,"target":61,"icon":90,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":91,"portfolioId":10,"children":92},"item-19",{"de":82,"en":83,"es":84,"fr":83,"it":85,"ru":86,"sr":87,"zh":88},"Leistungen","Services","Servicios","Servizi","Услуги","Услуге","服务","\u002Fservices-dienstleistungen-muenchen","i-lucide-wrench",116,[],{"id":94,"title":95,"url":99,"target":61,"icon":100,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":101,"portfolioId":10,"children":102},"item-23",{"de":96,"en":96,"es":96,"fr":96,"it":96,"ru":97,"sr":97,"zh":98},"Blog","Блог","博客","\u002Fblog","i-lucide-book-open",112,[],{"id":104,"title":105,"url":114,"target":61,"icon":115,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":116,"portfolioId":10,"children":117},"item-32",{"de":106,"en":107,"es":108,"fr":109,"it":110,"ru":111,"sr":112,"zh":113},"Neue Technologien","New Technologies","Nuevas tecnologías","Nouvelles technologies","Nuove tecnologie","Новые технологии","Нове технологије","新技术！","\u002Fneue-webtechnologien","i-lucide-sparkles",122,[],{"id":119,"title":120,"url":129,"target":61,"icon":130,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":131,"portfolioId":10,"children":132},"item-20",{"de":121,"en":122,"es":123,"fr":124,"it":125,"ru":126,"sr":127,"zh":128},"Kontakt","Contact us!","Contacto","Contact","Contatto","Контакт","Контактирајте нас","联系我们！","\u002Fcontact","i-lucide-mail",115,[],{"id":134,"title":135,"url":144,"target":61,"icon":145,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":147},"item-21",{"de":136,"en":137,"es":138,"fr":139,"it":140,"ru":141,"sr":142,"zh":143},"Unsere Arbeit","Our Work","Nuestro trabajo","Nos réalisations","I nostri lavori","Наши работы","Наши радови","文件夹","\u002Fportfolio","i-lucide-briefcase",114,[148,161,175,181,193],{"id":149,"title":150,"url":144,"target":61,"icon":159,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":160},"item-24",{"de":151,"en":152,"es":153,"fr":154,"it":155,"ru":156,"sr":157,"zh":158},"Alle Projekte","All Projects","Todos los proyectos","Tous les projets","Tutti i progetti","Все проекты","Сви пројекти","所有项目","i-lucide-grid-3x3",[],{"id":162,"title":163,"url":171,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":174},"item-29",{"de":164,"en":165,"es":166,"fr":167,"it":168,"ru":169,"sr":170,"zh":143},"Local Roots, Global Reach","Local Roots - Global Reach","Empresa local ","Entreprise locale","Azienda locale","Местная компания","Локално предузеће глобално тржиште","\u002Fportfolio\u002Flocal-roots-global-reach-communication-media-systems-for-modern-business","i-lucide-folder","custom",[],{"id":176,"title":177,"url":179,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":180},"item-28",{"de":178,"en":178,"es":178,"fr":178,"it":178,"ru":178,"sr":178,"zh":178},"Solr Suggester","\u002Fportfolio\u002Fsolr-fuzzy-suggester-und-solr-infix-suggester-abfrage-ueber-ajax-und-filterung",[],{"id":182,"title":183,"url":191,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":192},"item-27",{"de":184,"en":185,"es":186,"fr":187,"it":188,"ru":189,"sr":190,"zh":185},"Firmenwebseite SEO","Company Website SEO","Sitio web corporativo SEO","Site web d’entreprise SEO","Sito web aziendale SEO","Корпоративный сайт SEO","Пословна веб-страница SEO","\u002Fportfolio\u002Fseo-sem-branding-mobile-webseite-muenchen",[],{"id":194,"title":195,"url":203,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":204},"item-31",{"de":196,"en":197,"es":198,"fr":199,"it":200,"ru":201,"sr":202,"zh":197},"Digitalisierungsportal","Digitalization Portal","Portal de digitalización","Portail de numérisation","Portale di digitalizzazione","Портал цифровизации","Портал за дигитализацију","\u002Fportfolio\u002Fdigitalisierungsportal-archiv-museum-bibliothek-ead-lido-mets-mods",[],{"statusCode":4,"data":206,"message":1799},{"id":207,"title":208,"slug":209,"content":210,"contentJson":211,"excerpt":873,"featuredImage":874,"featuredImageAlt":875,"featuredImageCaption":10,"featuredImageTitle":10,"featuredImageCopyright":10,"featuredImageAuthor":10,"featuredImageSourceUrl":10,"featuredImageLicense":10,"featuredImageIsAiGenerated":43,"status":876,"publishedAt":877,"createdAt":878,"updatedAt":879,"seoLocalePaths":880,"categories":889,"author":914,"translations":919},"478","什么是RAG？对其工作原理的最简单解释","what-is-rag-the-simplest-explanation-of-how-it-works","\u003Cp>RAG 听起来很复杂，因为这个名字很复杂。但它的概念并不复杂。RAG 简单来说就是：在 AI 回答之前，它先从知识源中查找相关信息，并将这些信息提供给语言模型。\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--info my-6 rounded-xl border p-5 border-blue-300 bg-blue-50 dark:border-blue-900 dark:bg-blue-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">一句话解释 RAG\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">\u003Cstrong>RAG 是 AI 在 LLM 写出答案之前，先在知识库中搜索有用信息的步骤。\u003C\u002Fstrong>\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Cp>把 LLM 想象成一个坐在桌前的聪明人。RAG 就是图书管理员，从正确的书中取出正确的那一页。然后 LLM 阅读那一页并回答你。\u003C\u002Fp>\n\u003Cnav class=\"editorjs-toc\" data-editorjs-toc=\"true\" aria-label=\"目录\">\u003Cstrong class=\"editorjs-toc__title\">目录\u003C\u002Fstrong>\u003Col class=\"editorjs-toc__list editorjs-toc__list--depth-0\">\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-5\" class=\"editorjs-toc__link\">首先：LLM 做什么？\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-10\" class=\"editorjs-toc__link\">然后：什么是知识库？\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-15\" class=\"editorjs-toc__link\">那么 RAG 实际上做什么？\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-19\" class=\"editorjs-toc__link\">一个非常简单的例子\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-25\" class=\"editorjs-toc__link\">现在重要部分：RAG 不是当前状态\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-30\" class=\"editorjs-toc__link\">什么是状态数据库？\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-36\" class=\"editorjs-toc__link\">这三个部分如何协同工作\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-40\" class=\"editorjs-toc__link\">RAG 总是使用向量数据库吗？\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-45\" class=\"editorjs-toc__link\">用通俗的话说，什么是嵌入？\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-50\" class=\"editorjs-toc__link\">RAG 也不是记忆\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-54\" class=\"editorjs-toc__link\">一个真实的游戏示例：PUBG Ally\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-60\" class=\"editorjs-toc__link\">一个完整的例子\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-64\" class=\"editorjs-toc__link\">为什么要使用 RAG？\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-68\" class=\"editorjs-toc__link\">RAG 不能保证什么\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-72\" class=\"editorjs-toc__link\">最容易记住的心智模型\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-75\" class=\"editorjs-toc__link\">结论\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-79\" class=\"editorjs-toc__link\">常见问题\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-81\" class=\"editorjs-toc__link\">术语表\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-83\" class=\"editorjs-toc__link\">主要来源\u003C\u002Fa>\u003C\u002Fli>\u003C\u002Fol>\u003C\u002Fnav>\n\u003Ch2 id=\"section-5\">首先：LLM 做什么？\u003C\u002Fh2>\n\u003Cp>LLM 是理解语言和生成语言的部分。它可以阅读你的问题、理解指令、比较信息、解释事物并写出答案。\u003C\u002Fp>\n\u003Cp>但 LLM 并不会自动知道你的公司数据库、游戏会话、私人文档或你五分钟前创建的文件中当前有什么内容。\u003C\u002Fp>\n\u003Cp>它只知道模型内部已有的内容，以及应用程序在当前请求中提供给它的任何信息。\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--note my-6 rounded-xl border p-5 border-gray-300 bg-gray-50 dark:border-gray-700 dark:bg-gray-900\u002F40\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">简单规则\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">LLM \u003Cstrong>思考和写作\u003C\u002Fstrong>。它并不自动拥有你所有的当前数据。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-10\">然后：什么是知识库？\u003C\u002Fh2>\n\u003Cp>知识库就是应用程序可以搜索的信息。\u003C\u002Fp>\n\u003Cp>它可以包含 PDF、手册、产品文档、支持文章、合同、游戏规则、武器数据、公司内部文件、数据库记录或其他文本。\u003C\u002Fp>\n\u003Cp>知识库可以在你自己的机器上本地运行。它可以在服务器上。它可以在向量数据库中。它也可以由普通文件构建。RAG 并不意味着互联网。\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--success my-6 rounded-xl border p-5 border-emerald-300 bg-emerald-50 dark:border-emerald-900 dark:bg-emerald-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">重要\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">\u003Cstrong>RAG 不需要互联网。\u003C\u002Fstrong>信息可以完全在本地。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-15\">那么 RAG 实际上做什么？\u003C\u002Fh2>\n\u003Csection class=\"editorjs-process my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">整个 RAG 过程\u003C\u002Fh3>\u003Cdiv class=\"flex flex-col sm:flex-row gap-3\">\u003Cdiv class=\"editorjs-process__step min-w-0 flex-1 rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">1\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">1. 你提问\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">例如：这个武器使用哪种弹药？\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__arrow shrink-0 self-center text-xl text-gray-400 rotate-90 sm:rotate-0\" aria-hidden=\"true\">→\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0 flex-1 rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">2\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">2. RAG 搜索知识库\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">系统寻找与你的问题最相关的小块信息。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__arrow shrink-0 self-center text-xl text-gray-400 rotate-90 sm:rotate-0\" aria-hidden=\"true\">→\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0 flex-1 rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">3\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">3. RAG 将这些信息提供给 LLM\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">LLM 收到问题以及检索到的信息。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__arrow shrink-0 self-center text-xl text-gray-400 rotate-90 sm:rotate-0\" aria-hidden=\"true\">→\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0 flex-1 rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">4\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">4. LLM 写出答案\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">它使用检索到的信息作为回答的上下文。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Cp>这就是 RAG。\u003C\u002Fp>\n\u003Cp>全称是检索增强生成（Retrieval-Augmented Generation）。检索意味着找到相关信息。增强意味着将这些信息添加到模型的上下文中。生成意味着 LLM 写出最终答案。\u003C\u002Fp>\n\u003Ch2 id=\"section-19\">一个非常简单的例子\u003C\u002Fh2>\n\u003Cp>假设你有一个关于某个游戏的本地知识库。\u003C\u002Fp>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">知识库包含\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">示例\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">武器\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">AKM 使用 7.62 毫米弹药\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">治疗物品\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">医疗包恢复生命值\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">配件\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">此配件适用于这些武器\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">地图规则\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">此区域以这种方式运作\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Cp>你问：“AKM 使用哪种弹药？”\u003C\u002Fp>\n\u003Cp>RAG 搜索知识库并找到关于 AKM 的条目。它把这小部分信息交给 LLM。然后 LLM 回答：“AKM 使用 7.62 毫米弹药。”\u003C\u002Fp>\n\u003Cp>LLM 不需要整个数据库。RAG 只带来了有用的部分。\u003C\u002Fp>\n\u003Ch2 id=\"section-25\">现在重要部分：RAG 不是当前状态\u003C\u002Fh2>\n\u003Cp>这就是许多解释变得令人困惑的地方。\u003C\u002Fp>\n\u003Cp>RAG 通常给 AI 知识。状态系统给 AI 关于当前真实情况的事实。\u003C\u002Fp>\n\u003Csection class=\"editorjs-comparison my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">知识与当前状态\u003C\u002Fh3>\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left dark:border-gray-700 dark:bg-gray-900\">\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">RAG \u002F 知识\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">当前状态\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">武器\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">弹药\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">生命值\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">敌人\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Caside class=\"editorjs-callout editorjs-callout--warning my-6 rounded-xl border p-5 border-amber-300 bg-amber-50 dark:border-amber-900 dark:bg-amber-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">不要混淆这两者\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">RAG 回答：\u003Cstrong>什么通常是真实的？\u003C\u002Fstrong>\u003Cbr>状态回答：\u003Cstrong>现在什么是真实的？\u003C\u002Fstrong>\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-30\">什么是状态数据库？\u003C\u002Fh2>\n\u003Cp>状态数据库或状态存储只是应用程序保存当前事实的地方。\u003C\u002Fp>\n\u003Cp>在游戏中，引擎已经知道诸如你的生命值、位置、库存、弹药、当前任务、附近物体和敌人状态等信息。AI 系统可以将该状态的选定部分暴露给模型。\u003C\u002Fp>\n\u003Cp>在商业应用程序中，同样的想法可以是订单数据库、客户记录、项目状态或传感器的当前值。\u003C\u002Fp>\n\u003Cp>状态由应用程序本身在事情发生时创建。如果你失去生命值，游戏会更新生命值。如果你捡起弹药，库存会改变。如果订单已支付，业务系统会更改订单状态。\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--info my-6 rounded-xl border p-5 border-blue-300 bg-blue-50 dark:border-blue-900 dark:bg-blue-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">简单规则\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">应用程序创建并更新\u003Cstrong>状态\u003C\u002Fstrong>。RAG 搜索\u003Cstrong>知识\u003C\u002Fstrong>。LLM 使用两者来决定说什么或做什么。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-36\">这三个部分如何协同工作\u003C\u002Fh2>\n\u003Csection class=\"editorjs-process my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">LLM + 状态 + RAG\u003C\u002Fh3>\u003Cdiv class=\"grid grid-cols-1 md:grid-cols-2 xl:grid-cols-3 gap-4\">\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">1\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">1. 当前状态\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">应用程序告诉 AI 当前的真实情况：生命值 41%，已装备 AKM，23 发子弹。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">2\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">2. RAG\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">系统检索有用的知识：武器如何工作、有哪些治疗物品可用，或相关规则。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">3\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">3. LLM\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">模型接收问题、当前状态和检索到的知识。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">4\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">4. 推理\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">LLM 结合这些输入，决定什么回答或高层级行动是合理的。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">5\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">5. 应用程序\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">如果需要执行某个行动，应用程序或游戏引擎会执行它并再次更新状态。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Cp>所以基本架构是：\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--note my-6 rounded-xl border p-5 border-gray-300 bg-gray-50 dark:border-gray-700 dark:bg-gray-900\u002F40\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">最简单的架构\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">\u003Cstrong>状态 = 当前的真实情况\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>RAG = 有用的知识\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>LLM = 理解、推理和写作\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>应用程序 = 执行真实行动\u003C\u002Fstrong>\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-40\">RAG 总是使用向量数据库吗？\u003C\u002Fh2>\n\u003Cp>不是。\u003C\u002Fp>\n\u003Cp>向量数据库是构建语义搜索的常见方式，但它并不是 RAG 的定义。\u003C\u002Fp>\n\u003Cp>重要的部分是检索：系统找到相关的外部信息，并在生成答案之前将其添加到 LLM 的上下文中。\u003C\u002Fp>\n\u003Cp>例如，OpenAI 的 File Search 可以处理存储在向量存储中的文件。文件被分割成更小的片段，以便系统能够检索与问题相关的部分。这是同一基本思想的一种实现。\u003C\u002Fp>\n\u003Ch2 id=\"section-45\">用通俗的话说，什么是嵌入？\u003C\u002Fh2>\n\u003Cp>你不需要理解嵌入就能理解 RAG。\u003C\u002Fp>\n\u003Cp>但简单的版本是这样的：嵌入是含义的数值表示。它帮助搜索系统找到概念上相似的文本，即使单词不完全相同。\u003C\u002Fp>\n\u003Cp>例如，普通的关键词搜索可能会查找确切的词语“汽车维修”。语义搜索还可以理解“修我的车”是关于类似主题的。\u003C\u002Fp>\n\u003Cp>这使得嵌入对 RAG 有用，但 RAG 也可以使用关键词搜索、数据库查询或多种方法的混合。\u003C\u002Fp>\n\u003Ch2 id=\"section-50\">RAG 也不是记忆\u003C\u002Fh2>\n\u003Cp>记忆是另一个经常与 RAG 混淆的概念。\u003C\u002Fp>\n\u003Cp>记忆通常是系统保存的关于先前交互或先前事件的信息。RAG 是用于在需要时检索相关知识的机制。\u003C\u002Fp>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">部分\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">简单含义\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">LLM\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">理解和生成语言的部分\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">RAG\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">在答案之前查找相关知识的部分\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">知识库\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">RAG 可以搜索的信息\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">状态\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">应用程序或世界中当前的真实情况\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">记忆\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">从先前交互或事件中保留的信息\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">工具 \u002F 行动\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">AI 被允许调用或要求应用程序执行的事情\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">上下文\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">当前为此请求放置在 LLM 前面的信息\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-54\">一个真实的游戏示例：PUBG Ally\u003C\u002Fh2>\n\u003Cp>PUBG Ally 是一个有用的例子，因为它让这种差异变得显而易见。\u003C\u002Fp>\n\u003Cp>KRAFTON 将实时对局状态描述为一个独立的真相来源。游戏通过观察工具暴露当前事实：当前武器、弹药、生命值、安全区状态、附近物品和战斗情况。\u003C\u002Fp>\n\u003Cp>知识查询是另一项不同的工作。系统可以使用关于武器、配件、物品和规则的精选知识。NVIDIA 的 ACE Game Agent SDK 还暴露了一个独立的 RAG API，用于从开发者构建的数据库中检索知识。\u003C\u002Fp>\n\u003Cp>这就为我们提供了清晰的分离：游戏引擎说明当前正在发生什么，检索提供相关知识，而语言模型决定这些信息的含义。\u003C\u002Fp>\n\u003Caside class=\"editorjs-referral my-6\">\u003Ca href=\"https:\u002F\u002Ffigure.rocks\u002Fblog\u002Fpubg-ally-shows-why-ai-teammates-need-two-brains-fast-reflexes-and-slow-reasoning\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"flex flex-col sm:flex-row gap-4 rounded-xl border border-gray-200 dark:border-gray-700 p-4 transition hover:border-primary-500\">\u003Cdiv class=\"min-w-0 flex-1\">\u003Cstrong class=\"block text-lg text-gray-900 dark:text-gray-100\">PUBG Ally 展示了为什么 AI 队友需要两个大脑：快速反射与慢速推理\u003C\u002Fstrong>\u003Cp class=\"mt-2 text-sm text-gray-600 dark:text-gray-300\">一个实际的游戏示例，展示了实时状态、语言推理和确定性的游戏侧控制如何协同工作。\u003C\u002Fp>\u003Cspan class=\"mt-3 inline-flex text-sm font-medium text-primary-600 dark:text-primary-400\">阅读 PUBG Ally 架构文章 →\u003C\u002Fspan>\u003C\u002Fdiv>\u003C\u002Fa>\u003C\u002Faside>\n\u003Ch2 id=\"section-60\">一个完整的例子\u003C\u002Fh2>\n\u003Cp>想象一下，你对一个 AI 队友说：“我生命值很低。我们应该进攻吗？”\u003C\u002Fp>\n\u003Csection class=\"editorjs-process my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">接下来会发生什么\u003C\u002Fh3>\u003Cdiv class=\"grid grid-cols-1 md:grid-cols-2 xl:grid-cols-3 gap-4\">\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">1\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">状态\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">游戏报告：生命值 24%，附近有一名敌人，有两个治疗物品可用。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">2\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">RAG\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">知识系统检索治疗物品的相关规则，可能还有关于当前武器或战术机制的信息。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">3\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">LLM\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">模型将你的请求、当前状态和检索到的知识结合起来。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">4\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">决策\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">它得出结论：先治疗比立即进攻更安全。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">5\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">工具 \u002F 游戏引擎\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">智能体请求一个合法的游戏动作，例如移动到掩体或使用治疗物品。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">6\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">新状态\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">游戏执行该动作，并将更新后的情况报告回智能体。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Cp>RAG 没有控制角色。状态数据库没有进行推理。LLM 没有直接改变游戏。每个部分都只负责一项工作。\u003C\u002Fp>\n\u003Ch2 id=\"section-64\">为什么要使用 RAG？\u003C\u002Fh2>\n\u003Cp>因为把每一份文档、规则和数据库记录都放进每一个提示中会缓慢、昂贵，而且常常令人困惑。\u003C\u002Fp>\n\u003Cp>RAG 让系统只选择对当前问题有用的信息。\u003C\u002Fp>\n\u003Cp>它还让你无需重新训练整个语言模型就能更新知识库。更改文档或数据库，必要时重建或刷新索引，下一次检索就可以使用更新的信息。\u003C\u002Fp>\n\u003Ch2 id=\"section-68\">RAG 不能保证什么\u003C\u002Fh2>\n\u003Cp>RAG 可以改善依据性，但它不会让答案自动变得正确。\u003C\u002Fp>\n\u003Cp>检索步骤可能找到错误的文档。正确的文档可能已经过时。LLM 可能误解良好的证据。或者当前状态可能已经改变。\u003C\u002Fp>\n\u003Cp>因此，一个可靠的系统必须分别验证检索结果、状态新鲜度和模型的最终推理。\u003C\u002Fp>\n\u003Ch2 id=\"section-72\">最容易记住的心智模型\u003C\u002Fh2>\n\u003Csection class=\"editorjs-comparison my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">把AI系统想象成坐在桌前的人\u003C\u002Fh3>\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left dark:border-gray-700 dark:bg-gray-900\">\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">类比\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">AI系统\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">人在思考\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">查找参考书\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">书架上的书\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">当前仪表盘或仪表板\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">之前会议的笔记\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">在现实世界中做事\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Caside class=\"editorjs-callout editorjs-callout--success my-6 rounded-xl border p-5 border-emerald-300 bg-emerald-50 dark:border-emerald-900 dark:bg-emerald-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">如果你只记住这一点\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">\u003Cstrong>LLM = 大脑。\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>RAG = 图书管理员。\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>知识库 = 图书馆。\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>状态 = 仪表盘现在显示的内容。\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>工具 = 真正能做事的手。\u003C\u002Fstrong>\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-75\">结论\u003C\u002Fh2>\n\u003Cp>一旦把各个部分分开，RAG就没那么神秘了。\u003C\u002Fp>\n\u003Cp>LLM理解和生成语言。应用程序维护当前状态。知识库存储信息。RAG找到其中有用的部分并将其放入LLM的上下文中。工具或应用程序执行实际操作。\u003C\u002Fp>\n\u003Cp>这就是许多现代AI助手和代理背后的基本架构。\u003C\u002Fp>\n\u003Ch2 id=\"section-79\">常见问题\u003C\u002Fh2>\n\u003Csection class=\"editorjs-faq my-6 rounded-xl border border-gray-200 p-5 dark:border-gray-700\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">用通俗英语解释RAG\u003C\u002Fh3>\u003Cdiv id=\"faq1\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">用简单的话说，什么是RAG？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">RAG是AI在语言模型写出答案之前，先在知识源中搜索相关信息的一个步骤。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq2\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">RAG需要互联网吗？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">不需要。知识库可以完全位于你的计算机或服务器本地。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq3\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">RAG和数据库一样吗？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">不一样。数据库或文件包含信息。RAG是检索过程，它找到有用的部分并将其提供给LLM。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq4\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">RAG和记忆一样吗？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">不一样。记忆通常存储以前的交互或事件。RAG在需要时检索相关知识。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq5\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">当前应用程序状态是RAG的一部分吗？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">不一定。当前状态通常直接从应用程序或状态存储中获取。RAG更好地理解为从知识源中检索。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq6\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">RAG能让AI的答案正确吗？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">不能。它可以提供更好的证据，但检索仍然可能错误或过时，LLM仍然可能推理错误。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-81\">术语表\u003C\u002Fh2>\n\u003Csection class=\"editorjs-glossary my-6 rounded-xl border border-gray-200 dark:border-gray-700 p-5\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">基本术语\u003C\u002Fh3>\u003Cdl>\u003Cdiv id=\"llm\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">LLM\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">一种语言模型，能够理解和生成文本，并能对其上下文中的信息进行推理。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"rag\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">RAG\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">检索增强生成：在生成答案之前，检索相关的外部信息并将其添加到模型的上下文中。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"knowledge-base\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">知识库\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">检索可以搜索的文件、文档、记录或其他信息。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"state\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">状态\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">应用程序、系统或世界在特定时刻的当前事实。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"context\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">上下文\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">当前为一次请求或推理步骤提供给语言模型的信息。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"embedding\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">嵌入\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">一种表示含义的数值表示，可以帮助语义搜索找到概念上相似的信息。\u003C\u002Fdd>\u003C\u002Fdiv>\u003C\u002Fdl>\u003C\u002Fsection>\n\u003Ch2 id=\"section-83\">主要来源\u003C\u002Fh2>\n\u003Ca href=\"https:\u002F\u002Fplatform.openai.com\u002Fdocs\u002Fapi-reference\u002Fvector-stores-files\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">OpenAI — 向量存储文件\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">官方文档展示了如何将文件附加到向量存储、分块并使其可用于文件搜索检索。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fplatform.openai.com\u002Fdocs\u002Fquickstart\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">OpenAI — 开发者快速入门\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">OpenAI官方文档，描述了诸如文件搜索之类的工具，用于让模型访问外部信息。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fdeveloper.nvidia.com\u002Face-for-games\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">NVIDIA开发者 — 游戏ACE\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">NVIDIA官方文档，描述了独立的Agent、Chat和RAG API，用于将游戏角色连接到游戏状态、上下文知识和模型驱动的操作。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fdeveloper.nvidia.com\u002Fblog\u002Fhow-krafton-built-pubg-ally-a-co-playable-character-powered-by-nvidia-ace\u002F\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">NVIDIA开发者 — KRAFTON如何构建PUBG Ally\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">官方技术说明，将实时比赛状态与知识查找和语言模型推理分开。\u003C\u002Fp>\u003C\u002Fa>",{"time":212,"blocks":213,"version":872},1790377667516,[214,220,228,233,241,246,251,256,261,268,273,278,283,288,295,300,320,325,330,335,340,361,366,371,376,381,386,391,421,428,433,438,443,448,453,458,463,484,489,495,500,505,510,515,520,525,530,535,540,545,550,555,560,589,594,599,604,609,614,623,628,633,654,659,664,669,674,679,684,689,694,699,704,740,746,751,756,761,766,771,801,806,830,835,845,854,863],{"id":215,"data":216,"type":218,"tunes":219},"intro",{"text":217},"RAG 听起来很复杂，因为这个名字很复杂。但它的概念并不复杂。RAG 简单来说就是：在 AI 回答之前，它先从知识源中查找相关信息，并将这些信息提供给语言模型。","paragraph",{},{"id":221,"data":222,"type":226,"tunes":227},"one-sentence",{"body":223,"title":224,"variant":225},"\u003Cstrong>RAG 是 AI 在 LLM 写出答案之前，先在知识库中搜索有用信息的步骤。\u003C\u002Fstrong>","一句话解释 RAG","info","callout",{},{"id":229,"data":230,"type":218,"tunes":232},"analogy",{"text":231},"把 LLM 想象成一个坐在桌前的聪明人。RAG 就是图书管理员，从正确的书中取出正确的那一页。然后 LLM 阅读那一页并回答你。",{},{"id":234,"data":235,"type":239,"tunes":240},"toc",{"title":236,"maxLevel":237,"minLevel":238},"目录",3,2,"tableOfContents",{},{"id":242,"data":243,"type":42,"tunes":245},"h-llm",{"text":244,"level":238},"首先：LLM 做什么？",{},{"id":247,"data":248,"type":218,"tunes":250},"p-llm-1",{"text":249},"LLM 是理解语言和生成语言的部分。它可以阅读你的问题、理解指令、比较信息、解释事物并写出答案。",{},{"id":252,"data":253,"type":218,"tunes":255},"p-llm-2",{"text":254},"但 LLM 并不会自动知道你的公司数据库、游戏会话、私人文档或你五分钟前创建的文件中当前有什么内容。",{},{"id":257,"data":258,"type":218,"tunes":260},"p-llm-3",{"text":259},"它只知道模型内部已有的内容，以及应用程序在当前请求中提供给它的任何信息。",{},{"id":262,"data":263,"type":226,"tunes":267},"llm-rule",{"body":264,"title":265,"variant":266},"LLM \u003Cstrong>思考和写作\u003C\u002Fstrong>。它并不自动拥有你所有的当前数据。","简单规则","note",{},{"id":269,"data":270,"type":42,"tunes":272},"h-kb",{"text":271,"level":238},"然后：什么是知识库？",{},{"id":274,"data":275,"type":218,"tunes":277},"p-kb-1",{"text":276},"知识库就是应用程序可以搜索的信息。",{},{"id":279,"data":280,"type":218,"tunes":282},"p-kb-2",{"text":281},"它可以包含 PDF、手册、产品文档、支持文章、合同、游戏规则、武器数据、公司内部文件、数据库记录或其他文本。",{},{"id":284,"data":285,"type":218,"tunes":287},"p-kb-3",{"text":286},"知识库可以在你自己的机器上本地运行。它可以在服务器上。它可以在向量数据库中。它也可以由普通文件构建。RAG 并不意味着互联网。",{},{"id":289,"data":290,"type":226,"tunes":294},"no-internet",{"body":291,"title":292,"variant":293},"\u003Cstrong>RAG 不需要互联网。\u003C\u002Fstrong>信息可以完全在本地。","重要","success",{},{"id":296,"data":297,"type":42,"tunes":299},"h-rag",{"text":298,"level":238},"那么 RAG 实际上做什么？",{},{"id":301,"data":302,"type":318,"tunes":319},"rag-flow",{"steps":303,"title":316,"orientation":317},[304,307,310,313],{"label":305,"description":306},"1. 你提问","例如：这个武器使用哪种弹药？",{"label":308,"description":309},"2. RAG 搜索知识库","系统寻找与你的问题最相关的小块信息。",{"label":311,"description":312},"3. RAG 将这些信息提供给 LLM","LLM 收到问题以及检索到的信息。",{"label":314,"description":315},"4. LLM 写出答案","它使用检索到的信息作为回答的上下文。","整个 RAG 过程","auto","processFlow",{},{"id":321,"data":322,"type":218,"tunes":324},"rag-that-is-it",{"text":323},"这就是 RAG。",{},{"id":326,"data":327,"type":218,"tunes":329},"rag-name",{"text":328},"全称是检索增强生成（Retrieval-Augmented Generation）。检索意味着找到相关信息。增强意味着将这些信息添加到模型的上下文中。生成意味着 LLM 写出最终答案。",{},{"id":331,"data":332,"type":42,"tunes":334},"h-example",{"text":333,"level":238},"一个非常简单的例子",{},{"id":336,"data":337,"type":218,"tunes":339},"p-ex-1",{"text":338},"假设你有一个关于某个游戏的本地知识库。",{},{"id":341,"data":342,"type":359,"tunes":360},"kb-table",{"content":343,"stretched":43,"withHeadings":14},[344,347,350,353,356],[345,346],"知识库包含","示例",[348,349],"武器","AKM 使用 7.62 毫米弹药",[351,352],"治疗物品","医疗包恢复生命值",[354,355],"配件","此配件适用于这些武器",[357,358],"地图规则","此区域以这种方式运作","table",{},{"id":362,"data":363,"type":218,"tunes":365},"p-ex-2",{"text":364},"你问：“AKM 使用哪种弹药？”",{},{"id":367,"data":368,"type":218,"tunes":370},"p-ex-3",{"text":369},"RAG 搜索知识库并找到关于 AKM 的条目。它把这小部分信息交给 LLM。然后 LLM 回答：“AKM 使用 7.62 毫米弹药。”",{},{"id":372,"data":373,"type":218,"tunes":375},"p-ex-4",{"text":374},"LLM 不需要整个数据库。RAG 只带来了有用的部分。",{},{"id":377,"data":378,"type":42,"tunes":380},"h-state",{"text":379,"level":238},"现在重要部分：RAG 不是当前状态",{},{"id":382,"data":383,"type":218,"tunes":385},"p-state-1",{"text":384},"这就是许多解释变得令人困惑的地方。",{},{"id":387,"data":388,"type":218,"tunes":390},"p-state-2",{"text":389},"RAG 通常给 AI 知识。状态系统给 AI 关于当前真实情况的事实。",{},{"id":392,"data":393,"type":419,"tunes":420},"knowledge-state",{"rows":394,"title":411,"layout":359,"columns":412},[395,399,403,407],{"id":396,"label":348,"values":397},"weapon",[398,398],"",{"id":400,"label":401,"values":402},"ammo","弹药",[398,398],{"id":404,"label":405,"values":406},"health","生命值",[398,398],{"id":408,"label":409,"values":410},"enemy","敌人",[398,398],"知识与当前状态",[413,416],{"id":414,"label":415},"knowledge","RAG \u002F 知识",{"id":417,"label":418},"state","当前状态","comparison",{},{"id":422,"data":423,"type":226,"tunes":427},"dont-mix",{"body":424,"title":425,"variant":426},"RAG 回答：\u003Cstrong>什么通常是真实的？\u003C\u002Fstrong>\u003Cbr>状态回答：\u003Cstrong>现在什么是真实的？\u003C\u002Fstrong>","不要混淆这两者","warning",{},{"id":429,"data":430,"type":42,"tunes":432},"h-state-db",{"text":431,"level":238},"什么是状态数据库？",{},{"id":434,"data":435,"type":218,"tunes":437},"p-statedb-1",{"text":436},"状态数据库或状态存储只是应用程序保存当前事实的地方。",{},{"id":439,"data":440,"type":218,"tunes":442},"p-statedb-2",{"text":441},"在游戏中，引擎已经知道诸如你的生命值、位置、库存、弹药、当前任务、附近物体和敌人状态等信息。AI 系统可以将该状态的选定部分暴露给模型。",{},{"id":444,"data":445,"type":218,"tunes":447},"p-statedb-3",{"text":446},"在商业应用程序中，同样的想法可以是订单数据库、客户记录、项目状态或传感器的当前值。",{},{"id":449,"data":450,"type":218,"tunes":452},"p-statedb-4",{"text":451},"状态由应用程序本身在事情发生时创建。如果你失去生命值，游戏会更新生命值。如果你捡起弹药，库存会改变。如果订单已支付，业务系统会更改订单状态。",{},{"id":454,"data":455,"type":226,"tunes":457},"state-rule",{"body":456,"title":265,"variant":225},"应用程序创建并更新\u003Cstrong>状态\u003C\u002Fstrong>。RAG 搜索\u003Cstrong>知识\u003C\u002Fstrong>。LLM 使用两者来决定说什么或做什么。",{},{"id":459,"data":460,"type":42,"tunes":462},"h-together",{"text":461,"level":238},"这三个部分如何协同工作",{},{"id":464,"data":465,"type":318,"tunes":483},"together-flow",{"steps":466,"title":482,"orientation":317},[467,470,473,476,479],{"label":468,"description":469},"1. 当前状态","应用程序告诉 AI 当前的真实情况：生命值 41%，已装备 AKM，23 发子弹。",{"label":471,"description":472},"2. RAG","系统检索有用的知识：武器如何工作、有哪些治疗物品可用，或相关规则。",{"label":474,"description":475},"3. LLM","模型接收问题、当前状态和检索到的知识。",{"label":477,"description":478},"4. 推理","LLM 结合这些输入，决定什么回答或高层级行动是合理的。",{"label":480,"description":481},"5. 应用程序","如果需要执行某个行动，应用程序或游戏引擎会执行它并再次更新状态。","LLM + 状态 + RAG",{},{"id":485,"data":486,"type":218,"tunes":488},"p-arch-intro",{"text":487},"所以基本架构是：",{},{"id":490,"data":491,"type":226,"tunes":494},"simple-architecture",{"body":492,"title":493,"variant":266},"\u003Cstrong>状态 = 当前的真实情况\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>RAG = 有用的知识\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>LLM = 理解、推理和写作\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>应用程序 = 执行真实行动\u003C\u002Fstrong>","最简单的架构",{},{"id":496,"data":497,"type":42,"tunes":499},"h-vector",{"text":498,"level":238},"RAG 总是使用向量数据库吗？",{},{"id":501,"data":502,"type":218,"tunes":504},"p-vector-1",{"text":503},"不是。",{},{"id":506,"data":507,"type":218,"tunes":509},"p-vector-2",{"text":508},"向量数据库是构建语义搜索的常见方式，但它并不是 RAG 的定义。",{},{"id":511,"data":512,"type":218,"tunes":514},"p-vector-3",{"text":513},"重要的部分是检索：系统找到相关的外部信息，并在生成答案之前将其添加到 LLM 的上下文中。",{},{"id":516,"data":517,"type":218,"tunes":519},"p-vector-4",{"text":518},"例如，OpenAI 的 File Search 可以处理存储在向量存储中的文件。文件被分割成更小的片段，以便系统能够检索与问题相关的部分。这是同一基本思想的一种实现。",{},{"id":521,"data":522,"type":42,"tunes":524},"h-embedding",{"text":523,"level":238},"用通俗的话说，什么是嵌入？",{},{"id":526,"data":527,"type":218,"tunes":529},"p-emb-1",{"text":528},"你不需要理解嵌入就能理解 RAG。",{},{"id":531,"data":532,"type":218,"tunes":534},"p-emb-2",{"text":533},"但简单的版本是这样的：嵌入是含义的数值表示。它帮助搜索系统找到概念上相似的文本，即使单词不完全相同。",{},{"id":536,"data":537,"type":218,"tunes":539},"p-emb-3",{"text":538},"例如，普通的关键词搜索可能会查找确切的词语“汽车维修”。语义搜索还可以理解“修我的车”是关于类似主题的。",{},{"id":541,"data":542,"type":218,"tunes":544},"p-emb-4",{"text":543},"这使得嵌入对 RAG 有用，但 RAG 也可以使用关键词搜索、数据库查询或多种方法的混合。",{},{"id":546,"data":547,"type":42,"tunes":549},"h-memory",{"text":548,"level":238},"RAG 也不是记忆",{},{"id":551,"data":552,"type":218,"tunes":554},"p-memory-1",{"text":553},"记忆是另一个经常与 RAG 混淆的概念。",{},{"id":556,"data":557,"type":218,"tunes":559},"p-memory-2",{"text":558},"记忆通常是系统保存的关于先前交互或先前事件的信息。RAG 是用于在需要时检索相关知识的机制。",{},{"id":561,"data":562,"type":359,"tunes":588},"parts-table",{"content":563,"stretched":43,"withHeadings":14},[564,567,570,573,576,579,582,585],[565,566],"部分","简单含义",[568,569],"LLM","理解和生成语言的部分",[571,572],"RAG","在答案之前查找相关知识的部分",[574,575],"知识库","RAG 可以搜索的信息",[577,578],"状态","应用程序或世界中当前的真实情况",[580,581],"记忆","从先前交互或事件中保留的信息",[583,584],"工具 \u002F 行动","AI 被允许调用或要求应用程序执行的事情",[586,587],"上下文","当前为此请求放置在 LLM 前面的信息",{},{"id":590,"data":591,"type":42,"tunes":593},"h-pubg",{"text":592,"level":238},"一个真实的游戏示例：PUBG Ally",{},{"id":595,"data":596,"type":218,"tunes":598},"p-pubg-1",{"text":597},"PUBG Ally 是一个有用的例子，因为它让这种差异变得显而易见。",{},{"id":600,"data":601,"type":218,"tunes":603},"p-pubg-2",{"text":602},"KRAFTON 将实时对局状态描述为一个独立的真相来源。游戏通过观察工具暴露当前事实：当前武器、弹药、生命值、安全区状态、附近物品和战斗情况。",{},{"id":605,"data":606,"type":218,"tunes":608},"p-pubg-3",{"text":607},"知识查询是另一项不同的工作。系统可以使用关于武器、配件、物品和规则的精选知识。NVIDIA 的 ACE Game Agent SDK 还暴露了一个独立的 RAG API，用于从开发者构建的数据库中检索知识。",{},{"id":610,"data":611,"type":218,"tunes":613},"p-pubg-4",{"text":612},"这就为我们提供了清晰的分离：游戏引擎说明当前正在发生什么，检索提供相关知识，而语言模型决定这些信息的含义。",{},{"id":615,"data":616,"type":621,"tunes":622},"ref-pubg",{"url":617,"title":618,"excerpt":619,"ctaLabel":620},"https:\u002F\u002Ffigure.rocks\u002Fblog\u002Fpubg-ally-shows-why-ai-teammates-need-two-brains-fast-reflexes-and-slow-reasoning","PUBG Ally 展示了为什么 AI 队友需要两个大脑：快速反射与慢速推理","一个实际的游戏示例，展示了实时状态、语言推理和确定性的游戏侧控制如何协同工作。","阅读 PUBG Ally 架构文章","referralArticle",{},{"id":624,"data":625,"type":42,"tunes":627},"h-complete",{"text":626,"level":238},"一个完整的例子",{},{"id":629,"data":630,"type":218,"tunes":632},"p-complete-1",{"text":631},"想象一下，你对一个 AI 队友说：“我生命值很低。我们应该进攻吗？”",{},{"id":634,"data":635,"type":318,"tunes":653},"complete-flow",{"steps":636,"title":652,"orientation":317},[637,639,641,643,646,649],{"label":577,"description":638},"游戏报告：生命值 24%，附近有一名敌人，有两个治疗物品可用。",{"label":571,"description":640},"知识系统检索治疗物品的相关规则，可能还有关于当前武器或战术机制的信息。",{"label":568,"description":642},"模型将你的请求、当前状态和检索到的知识结合起来。",{"label":644,"description":645},"决策","它得出结论：先治疗比立即进攻更安全。",{"label":647,"description":648},"工具 \u002F 游戏引擎","智能体请求一个合法的游戏动作，例如移动到掩体或使用治疗物品。",{"label":650,"description":651},"新状态","游戏执行该动作，并将更新后的情况报告回智能体。","接下来会发生什么",{},{"id":655,"data":656,"type":218,"tunes":658},"p-complete-2",{"text":657},"RAG 没有控制角色。状态数据库没有进行推理。LLM 没有直接改变游戏。每个部分都只负责一项工作。",{},{"id":660,"data":661,"type":42,"tunes":663},"h-why",{"text":662,"level":238},"为什么要使用 RAG？",{},{"id":665,"data":666,"type":218,"tunes":668},"p-why-1",{"text":667},"因为把每一份文档、规则和数据库记录都放进每一个提示中会缓慢、昂贵，而且常常令人困惑。",{},{"id":670,"data":671,"type":218,"tunes":673},"p-why-2",{"text":672},"RAG 让系统只选择对当前问题有用的信息。",{},{"id":675,"data":676,"type":218,"tunes":678},"p-why-3",{"text":677},"它还让你无需重新训练整个语言模型就能更新知识库。更改文档或数据库，必要时重建或刷新索引，下一次检索就可以使用更新的信息。",{},{"id":680,"data":681,"type":42,"tunes":683},"h-not-guarantee",{"text":682,"level":238},"RAG 不能保证什么",{},{"id":685,"data":686,"type":218,"tunes":688},"p-not-1",{"text":687},"RAG 可以改善依据性，但它不会让答案自动变得正确。",{},{"id":690,"data":691,"type":218,"tunes":693},"p-not-2",{"text":692},"检索步骤可能找到错误的文档。正确的文档可能已经过时。LLM 可能误解良好的证据。或者当前状态可能已经改变。",{},{"id":695,"data":696,"type":218,"tunes":698},"p-not-3",{"text":697},"因此，一个可靠的系统必须分别验证检索结果、状态新鲜度和模型的最终推理。",{},{"id":700,"data":701,"type":42,"tunes":703},"h-mental",{"text":702,"level":238},"最容易记住的心智模型",{},{"id":705,"data":706,"type":419,"tunes":739},"mental-table",{"rows":707,"title":732,"layout":359,"columns":733},[708,712,716,720,724,728],{"id":709,"label":710,"values":711},"brain","人在思考",[398,398],{"id":713,"label":714,"values":715},"library","查找参考书",[398,398],{"id":717,"label":718,"values":719},"books","书架上的书",[398,398],{"id":721,"label":722,"values":723},"dashboard","当前仪表盘或仪表板",[398,398],{"id":725,"label":726,"values":727},"notes","之前会议的笔记",[398,398],{"id":729,"label":730,"values":731},"hands","在现实世界中做事",[398,398],"把AI系统想象成坐在桌前的人",[734,736],{"id":229,"label":735},"类比",{"id":737,"label":738},"system","AI系统",{},{"id":741,"data":742,"type":226,"tunes":745},"remember",{"body":743,"title":744,"variant":293},"\u003Cstrong>LLM = 大脑。\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>RAG = 图书管理员。\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>知识库 = 图书馆。\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>状态 = 仪表盘现在显示的内容。\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>工具 = 真正能做事的手。\u003C\u002Fstrong>","如果你只记住这一点",{},{"id":747,"data":748,"type":42,"tunes":750},"h-conclusion",{"text":749,"level":238},"结论",{},{"id":752,"data":753,"type":218,"tunes":755},"p-conc-1",{"text":754},"一旦把各个部分分开，RAG就没那么神秘了。",{},{"id":757,"data":758,"type":218,"tunes":760},"p-conc-2",{"text":759},"LLM理解和生成语言。应用程序维护当前状态。知识库存储信息。RAG找到其中有用的部分并将其放入LLM的上下文中。工具或应用程序执行实际操作。",{},{"id":762,"data":763,"type":218,"tunes":765},"p-conc-3",{"text":764},"这就是许多现代AI助手和代理背后的基本架构。",{},{"id":767,"data":768,"type":42,"tunes":770},"h-faq",{"text":769,"level":238},"常见问题",{},{"id":772,"data":773,"type":772,"tunes":800},"faq",{"items":774,"title":799},[775,779,783,787,791,795],{"id":776,"answer":777,"question":778},"faq1","RAG是AI在语言模型写出答案之前，先在知识源中搜索相关信息的一个步骤。","用简单的话说，什么是RAG？",{"id":780,"answer":781,"question":782},"faq2","不需要。知识库可以完全位于你的计算机或服务器本地。","RAG需要互联网吗？",{"id":784,"answer":785,"question":786},"faq3","不一样。数据库或文件包含信息。RAG是检索过程，它找到有用的部分并将其提供给LLM。","RAG和数据库一样吗？",{"id":788,"answer":789,"question":790},"faq4","不一样。记忆通常存储以前的交互或事件。RAG在需要时检索相关知识。","RAG和记忆一样吗？",{"id":792,"answer":793,"question":794},"faq5","不一定。当前状态通常直接从应用程序或状态存储中获取。RAG更好地理解为从知识源中检索。","当前应用程序状态是RAG的一部分吗？",{"id":796,"answer":797,"question":798},"faq6","不能。它可以提供更好的证据，但检索仍然可能错误或过时，LLM仍然可能推理错误。","RAG能让AI的答案正确吗？","用通俗英语解释RAG",{},{"id":802,"data":803,"type":42,"tunes":805},"h-glossary",{"text":804,"level":238},"术语表",{},{"id":807,"data":808,"type":807,"tunes":829},"glossary",{"title":809,"entries":810},"基本术语",[811,814,817,820,822,825],{"term":568,"anchor":812,"definition":813},"llm","一种语言模型，能够理解和生成文本，并能对其上下文中的信息进行推理。",{"term":571,"anchor":815,"definition":816},"rag","检索增强生成：在生成答案之前，检索相关的外部信息并将其添加到模型的上下文中。",{"term":574,"anchor":818,"definition":819},"knowledge-base","检索可以搜索的文件、文档、记录或其他信息。",{"term":577,"anchor":417,"definition":821},"应用程序、系统或世界在特定时刻的当前事实。",{"term":586,"anchor":823,"definition":824},"context","当前为一次请求或推理步骤提供给语言模型的信息。",{"term":826,"anchor":827,"definition":828},"嵌入","embedding","一种表示含义的数值表示，可以帮助语义搜索找到概念上相似的信息。",{},{"id":831,"data":832,"type":42,"tunes":834},"h-sources",{"text":833,"level":238},"主要来源",{},{"id":836,"data":837,"type":843,"tunes":844},"src-openai-vector",{"link":838,"meta":839},"https:\u002F\u002Fplatform.openai.com\u002Fdocs\u002Fapi-reference\u002Fvector-stores-files",{"image":840,"title":841,"description":842},{"url":398},"OpenAI — 向量存储文件","官方文档展示了如何将文件附加到向量存储、分块并使其可用于文件搜索检索。","linkTool",{},{"id":846,"data":847,"type":843,"tunes":853},"src-openai-quickstart",{"link":848,"meta":849},"https:\u002F\u002Fplatform.openai.com\u002Fdocs\u002Fquickstart",{"image":850,"title":851,"description":852},{"url":398},"OpenAI — 开发者快速入门","OpenAI官方文档，描述了诸如文件搜索之类的工具，用于让模型访问外部信息。",{},{"id":855,"data":856,"type":843,"tunes":862},"src-nvidia-ace",{"link":857,"meta":858},"https:\u002F\u002Fdeveloper.nvidia.com\u002Face-for-games",{"image":859,"title":860,"description":861},{"url":398},"NVIDIA开发者 — 游戏ACE","NVIDIA官方文档，描述了独立的Agent、Chat和RAG API，用于将游戏角色连接到游戏状态、上下文知识和模型驱动的操作。",{},{"id":864,"data":865,"type":843,"tunes":871},"src-nvidia-pubg",{"link":866,"meta":867},"https:\u002F\u002Fdeveloper.nvidia.com\u002Fblog\u002Fhow-krafton-built-pubg-ally-a-co-playable-character-powered-by-nvidia-ace\u002F",{"image":868,"title":869,"description":870},{"url":398},"NVIDIA开发者 — KRAFTON如何构建PUBG Ally","官方技术说明，将实时比赛状态与知识查找和语言模型推理分开。",{},"2.31","RAG听起来很复杂，但想法很简单：在AI回答之前，它先从知识源查找有用的信息，并将该信息提供给语言模型。本指南使用一个简单的思维模型来解释RAG、LLM、状态、记忆和工具。","\u002Fuploads\u002F2026\u002F09\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works-1790377492124-khjagt.webp","what-is-rag-the-simplest-explanation-of-how-it-works-1790377492124-khjagt","PUBLISHED","2026-09-25T19:03:00.000Z","2026-09-25T23:03:13.651Z","2026-09-25T23:41:17.272Z",{"en":881,"de":882,"sr":883,"es":884,"fr":885,"it":886,"ru":887,"zh":888},"\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works","\u002Fde\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works","\u002Fsr\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works","\u002Fes\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works","\u002Ffr\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works","\u002Fit\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works","\u002Fru\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works","\u002Fzh\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works",[890,894,898,902,906,910],{"id":891,"name":892,"slug":893},46,"概览","overview",{"id":895,"name":896,"slug":897},57,"数据边界","data-boundaries",{"id":899,"name":900,"slug":901},51,"反模式","anti-patterns",{"id":903,"name":904,"slug":905},58,"评估与质量门槛","evaluation",{"id":907,"name":908,"slug":909},56,"用例组合","use-case-portfolio",{"id":911,"name":912,"slug":913},60,"成本与延迟控制","cost-and-latency",{"id":915,"login":916,"email":917,"displayName":918},"20","rooth8233","aleksandar@stajic.de","Aleksandar Stajić",[920,1452],{"lang":921,"title":922,"content":923,"contentJson":924,"excerpt":1451},"en","What Is RAG? The Simplest Explanation of How It Works","{\"time\":1790377494031,\"blocks\":[{\"id\":\"intro\",\"type\":\"paragraph\",\"data\":{\"text\":\"RAG sounds complicated because the name is complicated. The idea is not. RAG simply means: before the AI answers, it first looks up relevant information from a knowledge source and gives that information to the language model.\"},\"tunes\":{}},{\"id\":\"one-sentence\",\"type\":\"callout\",\"data\":{\"variant\":\"info\",\"title\":\"RAG in one sentence\",\"body\":\"\u003Cstrong>RAG is the step where an AI searches a knowledge base for useful information before the LLM writes the answer.\u003C\u002Fstrong>\"},\"tunes\":{}},{\"id\":\"analogy\",\"type\":\"paragraph\",\"data\":{\"text\":\"Think of an LLM as a smart person sitting at a desk. RAG is the librarian who brings the right page from the right book. The LLM then reads that page and answers you.\"},\"tunes\":{}},{\"id\":\"toc\",\"type\":\"tableOfContents\",\"data\":{\"title\":\"Contents\",\"minLevel\":2,\"maxLevel\":3},\"tunes\":{}},{\"id\":\"h-llm\",\"type\":\"header\",\"data\":{\"text\":\"First: what does the LLM do?\",\"level\":2},\"tunes\":{}},{\"id\":\"p-llm-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The LLM is the part that understands language and produces language. It can read your question, understand instructions, compare information, explain something and write an answer.\"},\"tunes\":{}},{\"id\":\"p-llm-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"But the LLM does not automatically know what is currently inside your company database, your game session, your private documents or a file you created five minutes ago.\"},\"tunes\":{}},{\"id\":\"p-llm-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"It only knows what is already inside the model plus whatever information the application gives it in the current request.\"},\"tunes\":{}},{\"id\":\"llm-rule\",\"type\":\"callout\",\"data\":{\"variant\":\"note\",\"title\":\"Simple rule\",\"body\":\"The LLM \u003Cstrong>thinks and writes\u003C\u002Fstrong>. It does not automatically own all of your current data.\"},\"tunes\":{}},{\"id\":\"h-kb\",\"type\":\"header\",\"data\":{\"text\":\"Then: what is the knowledge base?\",\"level\":2},\"tunes\":{}},{\"id\":\"p-kb-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A knowledge base is simply information the application can search.\"},\"tunes\":{}},{\"id\":\"p-kb-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"It could contain PDFs, manuals, product documentation, support articles, contracts, game rules, weapon data, internal company documents, database records or other text.\"},\"tunes\":{}},{\"id\":\"p-kb-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The knowledge base can be local on your own machine. It can be on a server. It can be in a vector database. It can also be built from normal files. RAG does not mean Internet.\"},\"tunes\":{}},{\"id\":\"no-internet\",\"type\":\"callout\",\"data\":{\"variant\":\"success\",\"title\":\"Important\",\"body\":\"\u003Cstrong>RAG does not require the Internet.\u003C\u002Fstrong> The information can be completely local.\"},\"tunes\":{}},{\"id\":\"h-rag\",\"type\":\"header\",\"data\":{\"text\":\"So what does RAG actually do?\",\"level\":2},\"tunes\":{}},{\"id\":\"rag-flow\",\"type\":\"processFlow\",\"data\":{\"title\":\"The whole RAG process\",\"orientation\":\"auto\",\"steps\":[{\"label\":\"1. You ask a question\",\"description\":\"For example: Which ammunition does this weapon use?\"},{\"label\":\"2. RAG searches the knowledge base\",\"description\":\"The system looks for the small pieces of information most relevant to your question.\"},{\"label\":\"3. RAG gives those pieces to the LLM\",\"description\":\"The LLM receives the question plus the retrieved information.\"},{\"label\":\"4. The LLM writes the answer\",\"description\":\"It uses the retrieved information as context for the response.\"}]},\"tunes\":{}},{\"id\":\"rag-that-is-it\",\"type\":\"paragraph\",\"data\":{\"text\":\"That is RAG.\"},\"tunes\":{}},{\"id\":\"rag-name\",\"type\":\"paragraph\",\"data\":{\"text\":\"The full name is Retrieval-Augmented Generation. Retrieval means finding the relevant information. Augmented means adding that information to the model's context. Generation means the LLM writes the final answer.\"},\"tunes\":{}},{\"id\":\"h-example\",\"type\":\"header\",\"data\":{\"text\":\"A very simple example\",\"level\":2},\"tunes\":{}},{\"id\":\"p-ex-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Imagine you have a local knowledge base about a game.\"},\"tunes\":{}},{\"id\":\"kb-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Knowledge base contains\",\"Example\"],[\"Weapons\",\"AKM uses 7.62 mm ammunition\"],[\"Healing items\",\"Med Kit restores health\"],[\"Attachments\",\"This attachment works with these weapons\"],[\"Map rules\",\"This zone behaves in this way\"]]},\"tunes\":{}},{\"id\":\"p-ex-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"You ask: “Which ammunition does the AKM use?”\"},\"tunes\":{}},{\"id\":\"p-ex-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"RAG searches the knowledge base and finds the entry about the AKM. It gives that small piece of information to the LLM. The LLM then answers: “The AKM uses 7.62 mm ammunition.”\"},\"tunes\":{}},{\"id\":\"p-ex-4\",\"type\":\"paragraph\",\"data\":{\"text\":\"The LLM did not need the entire database. RAG only brought the useful part.\"},\"tunes\":{}},{\"id\":\"h-state\",\"type\":\"header\",\"data\":{\"text\":\"Now the important part: RAG is not the current state\",\"level\":2},\"tunes\":{}},{\"id\":\"p-state-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"This is where many explanations become confusing.\"},\"tunes\":{}},{\"id\":\"p-state-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"RAG usually gives the AI knowledge. A state system gives the AI facts about what is true right now.\"},\"tunes\":{}},{\"id\":\"knowledge-state\",\"type\":\"comparison\",\"data\":{\"title\":\"Knowledge vs current state\",\"layout\":\"table\",\"columns\":[{\"id\":\"knowledge\",\"label\":\"RAG \u002F knowledge\"},{\"id\":\"state\",\"label\":\"Current state\"}],\"rows\":[{\"id\":\"weapon\",\"label\":\"Weapon\",\"values\":[\"\",\"\"]},{\"id\":\"ammo\",\"label\":\"Ammunition\",\"values\":[\"\",\"\"]},{\"id\":\"health\",\"label\":\"Health\",\"values\":[\"\",\"\"]},{\"id\":\"enemy\",\"label\":\"Enemy\",\"values\":[\"\",\"\"]}]},\"tunes\":{}},{\"id\":\"dont-mix\",\"type\":\"callout\",\"data\":{\"variant\":\"warning\",\"title\":\"Do not mix these two\",\"body\":\"RAG answers: \u003Cstrong>What is generally true?\u003C\u002Fstrong>\u003Cbr>State answers: \u003Cstrong>What is true right now?\u003C\u002Fstrong>\"},\"tunes\":{}},{\"id\":\"h-state-db\",\"type\":\"header\",\"data\":{\"text\":\"What is a state database?\",\"level\":2},\"tunes\":{}},{\"id\":\"p-statedb-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A state database or state store is simply a place where the application keeps current facts.\"},\"tunes\":{}},{\"id\":\"p-statedb-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"In a game, the engine already knows things such as your health, position, inventory, ammunition, current mission, nearby objects and enemy status. An AI system can expose selected parts of that state to the model.\"},\"tunes\":{}},{\"id\":\"p-statedb-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"In a business application, the same idea could be an order database, a customer record, a project status or the current value of a sensor.\"},\"tunes\":{}},{\"id\":\"p-statedb-4\",\"type\":\"paragraph\",\"data\":{\"text\":\"The state is created by the application itself as things happen. If you lose health, the game updates the health value. If you pick up ammunition, the inventory changes. If an order is paid, the business system changes the order status.\"},\"tunes\":{}},{\"id\":\"state-rule\",\"type\":\"callout\",\"data\":{\"variant\":\"info\",\"title\":\"Simple rule\",\"body\":\"The application creates and updates \u003Cstrong>state\u003C\u002Fstrong>. RAG searches \u003Cstrong>knowledge\u003C\u002Fstrong>. The LLM uses both to decide what to say or do.\"},\"tunes\":{}},{\"id\":\"h-together\",\"type\":\"header\",\"data\":{\"text\":\"How the three pieces work together\",\"level\":2},\"tunes\":{}},{\"id\":\"together-flow\",\"type\":\"processFlow\",\"data\":{\"title\":\"LLM + state + RAG\",\"orientation\":\"auto\",\"steps\":[{\"label\":\"1. Current state\",\"description\":\"The application tells the AI what is true now: health 41%, AKM equipped, 23 rounds.\"},{\"label\":\"2. RAG\",\"description\":\"The system retrieves useful knowledge: how the weapon works, which healing item is available, or a relevant rule.\"},{\"label\":\"3. LLM\",\"description\":\"The model receives the question, current state and retrieved knowledge.\"},{\"label\":\"4. Reasoning\",\"description\":\"The LLM combines those inputs and decides what answer or high-level action makes sense.\"},{\"label\":\"5. Application\",\"description\":\"If an action is required, the application or game engine executes it and updates the state again.\"}]},\"tunes\":{}},{\"id\":\"p-arch-intro\",\"type\":\"paragraph\",\"data\":{\"text\":\"So the basic architecture is:\"},\"tunes\":{}},{\"id\":\"simple-architecture\",\"type\":\"callout\",\"data\":{\"variant\":\"note\",\"title\":\"The simplest architecture\",\"body\":\"\u003Cstrong>State = what is true now\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>RAG = useful knowledge\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>LLM = understands, reasons and writes\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Application = performs the real action\u003C\u002Fstrong>\"},\"tunes\":{}},{\"id\":\"h-vector\",\"type\":\"header\",\"data\":{\"text\":\"Does RAG always use a vector database?\",\"level\":2},\"tunes\":{}},{\"id\":\"p-vector-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"No.\"},\"tunes\":{}},{\"id\":\"p-vector-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"A vector database is a common way to build semantic search, but it is not the definition of RAG.\"},\"tunes\":{}},{\"id\":\"p-vector-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The important part is retrieval: the system finds relevant external information and adds it to the LLM's context before the answer is generated.\"},\"tunes\":{}},{\"id\":\"p-vector-4\",\"type\":\"paragraph\",\"data\":{\"text\":\"OpenAI's File Search, for example, can work with files stored in vector stores. Files are chunked into smaller pieces so the system can retrieve the parts that are relevant to a question. That is one implementation of the same basic idea.\"},\"tunes\":{}},{\"id\":\"h-embedding\",\"type\":\"header\",\"data\":{\"text\":\"What is an embedding, in plain English?\",\"level\":2},\"tunes\":{}},{\"id\":\"p-emb-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"You do not need to understand embeddings to understand RAG.\"},\"tunes\":{}},{\"id\":\"p-emb-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"But the simple version is this: an embedding is a numerical representation of meaning. It helps a search system find text that is conceptually similar even when the words are not exactly the same.\"},\"tunes\":{}},{\"id\":\"p-emb-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"For example, a normal keyword search may look for the exact words “car repair.” Semantic search can also understand that “fix my vehicle” is about a similar topic.\"},\"tunes\":{}},{\"id\":\"p-emb-4\",\"type\":\"paragraph\",\"data\":{\"text\":\"That makes embeddings useful for RAG, but RAG can also use keyword search, database queries or a hybrid of several methods.\"},\"tunes\":{}},{\"id\":\"h-memory\",\"type\":\"header\",\"data\":{\"text\":\"RAG is not memory either\",\"level\":2},\"tunes\":{}},{\"id\":\"p-memory-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Memory is another concept that is often mixed together with RAG.\"},\"tunes\":{}},{\"id\":\"p-memory-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Memory is usually information the system keeps about previous interactions or previous events. RAG is the mechanism used to retrieve relevant knowledge when it is needed.\"},\"tunes\":{}},{\"id\":\"parts-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Part\",\"Simple meaning\"],[\"LLM\",\"The part that understands and generates language\"],[\"RAG\",\"The part that looks up relevant knowledge before the answer\"],[\"Knowledge base\",\"The information RAG can search\"],[\"State\",\"What is true right now in the application or world\"],[\"Memory\",\"Information kept from previous interactions or events\"],[\"Tool \u002F action\",\"Something the AI is allowed to call or ask the application to do\"],[\"Context\",\"The information currently placed in front of the LLM for this request\"]]},\"tunes\":{}},{\"id\":\"h-pubg\",\"type\":\"header\",\"data\":{\"text\":\"A real game example: PUBG Ally\",\"level\":2},\"tunes\":{}},{\"id\":\"p-pubg-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"PUBG Ally is a useful example because it makes the difference visible.\"},\"tunes\":{}},{\"id\":\"p-pubg-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"KRAFTON describes live match state as a separate source of truth. The game exposes current facts through observation tools: current weapon, ammunition, health, safe-zone status, nearby items and combat situation.\"},\"tunes\":{}},{\"id\":\"p-pubg-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"Knowledge lookup is a different job. The system can use curated knowledge about weapons, attachments, items and rules. NVIDIA's ACE Game Agent SDK also exposes a separate RAG API for retrieving knowledge from developer-built databases.\"},\"tunes\":{}},{\"id\":\"p-pubg-4\",\"type\":\"paragraph\",\"data\":{\"text\":\"That gives us the clean separation: the game engine says what is happening now, retrieval provides relevant knowledge, and the language model decides what the information means.\"},\"tunes\":{}},{\"id\":\"ref-pubg\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Ffigure.rocks\u002Fblog\u002Fpubg-ally-shows-why-ai-teammates-need-two-brains-fast-reflexes-and-slow-reasoning\",\"title\":\"PUBG Ally Shows Why AI Teammates Need Two Brains: Fast Reflexes and Slow Reasoning\",\"excerpt\":\"A practical game example showing how live state, language reasoning and deterministic game-side control can work together.\",\"ctaLabel\":\"Read the PUBG Ally architecture article\"},\"tunes\":{}},{\"id\":\"h-complete\",\"type\":\"header\",\"data\":{\"text\":\"One complete example\",\"level\":2},\"tunes\":{}},{\"id\":\"p-complete-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Imagine you tell an AI teammate: “I am low on health. Should we attack?”\"},\"tunes\":{}},{\"id\":\"complete-flow\",\"type\":\"processFlow\",\"data\":{\"title\":\"What happens next\",\"orientation\":\"auto\",\"steps\":[{\"label\":\"State\",\"description\":\"The game reports: health 24%, one enemy nearby, two healing items available.\"},{\"label\":\"RAG\",\"description\":\"The knowledge system retrieves the relevant rules for the healing item and perhaps information about the current weapon or tactical mechanic.\"},{\"label\":\"LLM\",\"description\":\"The model combines your request, the current state and the retrieved knowledge.\"},{\"label\":\"Decision\",\"description\":\"It concludes that healing first is safer than attacking immediately.\"},{\"label\":\"Tool \u002F game engine\",\"description\":\"The agent requests a legal game action such as moving to cover or using the healing item.\"},{\"label\":\"New state\",\"description\":\"The game executes the action and reports the updated situation back to the agent.\"}]},\"tunes\":{}},{\"id\":\"p-complete-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"RAG did not control the character. The state database did not reason. The LLM did not directly change the game. Each part had one job.\"},\"tunes\":{}},{\"id\":\"h-why\",\"type\":\"header\",\"data\":{\"text\":\"Why use RAG at all?\",\"level\":2},\"tunes\":{}},{\"id\":\"p-why-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Because putting every document, rule and database record into every prompt would be slow, expensive and often confusing.\"},\"tunes\":{}},{\"id\":\"p-why-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"RAG lets the system select only the information that is useful for the current question.\"},\"tunes\":{}},{\"id\":\"p-why-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"It also lets you update the knowledge base without retraining the entire language model. Change the document or database, rebuild or refresh the index when necessary, and the next retrieval can use the newer information.\"},\"tunes\":{}},{\"id\":\"h-not-guarantee\",\"type\":\"header\",\"data\":{\"text\":\"What RAG does not guarantee\",\"level\":2},\"tunes\":{}},{\"id\":\"p-not-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"RAG can improve grounding, but it does not make an answer automatically correct.\"},\"tunes\":{}},{\"id\":\"p-not-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The retrieval step can find the wrong document. The correct document can be outdated. The LLM can misunderstand good evidence. Or the current state can have changed.\"},\"tunes\":{}},{\"id\":\"p-not-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"A reliable system therefore has to validate retrieval, state freshness and the model's final reasoning separately.\"},\"tunes\":{}},{\"id\":\"h-mental\",\"type\":\"header\",\"data\":{\"text\":\"The easiest mental model to remember\",\"level\":2},\"tunes\":{}},{\"id\":\"mental-table\",\"type\":\"comparison\",\"data\":{\"title\":\"Think of an AI system like a person at a desk\",\"layout\":\"table\",\"columns\":[{\"id\":\"analogy\",\"label\":\"Analogy\"},{\"id\":\"system\",\"label\":\"AI system\"}],\"rows\":[{\"id\":\"brain\",\"label\":\"Person thinking\",\"values\":[\"\",\"\"]},{\"id\":\"library\",\"label\":\"Finding a reference book\",\"values\":[\"\",\"\"]},{\"id\":\"books\",\"label\":\"Books on the shelf\",\"values\":[\"\",\"\"]},{\"id\":\"dashboard\",\"label\":\"Current dashboard or instrument panel\",\"values\":[\"\",\"\"]},{\"id\":\"notes\",\"label\":\"Notes from earlier meetings\",\"values\":[\"\",\"\"]},{\"id\":\"hands\",\"label\":\"Doing something in the real world\",\"values\":[\"\",\"\"]}]},\"tunes\":{}},{\"id\":\"remember\",\"type\":\"callout\",\"data\":{\"variant\":\"success\",\"title\":\"If you remember only this\",\"body\":\"\u003Cstrong>LLM = brain.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>RAG = librarian.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Knowledge base = library.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>State = what the dashboard says right now.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Tools = the hands that can actually do something.\u003C\u002Fstrong>\"},\"tunes\":{}},{\"id\":\"h-conclusion\",\"type\":\"header\",\"data\":{\"text\":\"Conclusion\",\"level\":2},\"tunes\":{}},{\"id\":\"p-conc-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"RAG is much less mysterious once the parts are separated.\"},\"tunes\":{}},{\"id\":\"p-conc-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The LLM understands and generates language. The application maintains current state. The knowledge base stores information. RAG finds the useful part of that information and puts it into the LLM's context. Tools or the application perform real actions.\"},\"tunes\":{}},{\"id\":\"p-conc-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"That is the basic architecture behind many modern AI assistants and agents.\"},\"tunes\":{}},{\"id\":\"h-faq\",\"type\":\"header\",\"data\":{\"text\":\"FAQ\",\"level\":2},\"tunes\":{}},{\"id\":\"faq\",\"type\":\"faq\",\"data\":{\"title\":\"RAG in plain English\",\"items\":[{\"id\":\"faq1\",\"question\":\"What is RAG in simple terms?\",\"answer\":\"RAG is a step where an AI searches a knowledge source for relevant information before the language model writes its answer.\"},{\"id\":\"faq2\",\"question\":\"Does RAG need the Internet?\",\"answer\":\"No. The knowledge base can be completely local on your computer or server.\"},{\"id\":\"faq3\",\"question\":\"Is RAG the same as a database?\",\"answer\":\"No. The database or files contain the information. RAG is the retrieval process that finds the useful part and gives it to the LLM.\"},{\"id\":\"faq4\",\"question\":\"Is RAG the same as memory?\",\"answer\":\"No. Memory usually stores previous interactions or events. RAG retrieves relevant knowledge when it is needed.\"},{\"id\":\"faq5\",\"question\":\"Is current application state part of RAG?\",\"answer\":\"Not necessarily. Current state is usually obtained directly from the application or a state store. RAG is better understood as retrieval from a knowledge source.\"},{\"id\":\"faq6\",\"question\":\"Does RAG make AI answers correct?\",\"answer\":\"No. It can provide better evidence, but retrieval can still be wrong or outdated and the LLM can still reason incorrectly.\"}]},\"tunes\":{}},{\"id\":\"h-glossary\",\"type\":\"header\",\"data\":{\"text\":\"Glossary\",\"level\":2},\"tunes\":{}},{\"id\":\"glossary\",\"type\":\"glossary\",\"data\":{\"title\":\"The basic terms\",\"entries\":[{\"term\":\"LLM\",\"definition\":\"A language model that understands and generates text and can reason over information placed in its context.\",\"anchor\":\"llm\"},{\"term\":\"RAG\",\"definition\":\"Retrieval-Augmented Generation: retrieving relevant external information and adding it to the model's context before generating an answer.\",\"anchor\":\"rag\"},{\"term\":\"Knowledge base\",\"definition\":\"The files, documents, records or other information that retrieval can search.\",\"anchor\":\"knowledge-base\"},{\"term\":\"State\",\"definition\":\"The current facts of an application, system or world at a particular moment.\",\"anchor\":\"state\"},{\"term\":\"Context\",\"definition\":\"The information currently supplied to the language model for one request or reasoning step.\",\"anchor\":\"context\"},{\"term\":\"Embedding\",\"definition\":\"A numerical representation of meaning that can help semantic search find conceptually similar information.\",\"anchor\":\"embedding\"}]},\"tunes\":{}},{\"id\":\"h-sources\",\"type\":\"header\",\"data\":{\"text\":\"Primary sources\",\"level\":2},\"tunes\":{}},{\"id\":\"src-openai-vector\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fplatform.openai.com\u002Fdocs\u002Fapi-reference\u002Fvector-stores-files\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — Vector Store Files\",\"description\":\"Official documentation showing how files can be attached to vector stores, chunked and made available to file-search retrieval.\"}},\"tunes\":{}},{\"id\":\"src-openai-quickstart\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fplatform.openai.com\u002Fdocs\u002Fquickstart\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — Developer Quickstart\",\"description\":\"Official OpenAI documentation describing tools such as file search for giving models access to external information.\"}},\"tunes\":{}},{\"id\":\"src-nvidia-ace\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdeveloper.nvidia.com\u002Face-for-games\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"NVIDIA Developer — ACE for Games\",\"description\":\"Official NVIDIA documentation describing separate Agent, Chat and RAG APIs for connecting game characters to game state, contextual knowledge and model-driven actions.\"}},\"tunes\":{}},{\"id\":\"src-nvidia-pubg\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdeveloper.nvidia.com\u002Fblog\u002Fhow-krafton-built-pubg-ally-a-co-playable-character-powered-by-nvidia-ace\u002F\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"NVIDIA Developer — How KRAFTON Built PUBG Ally\",\"description\":\"Official technical explanation separating live match state from knowledge lookup and language-model reasoning.\"}},\"tunes\":{}}],\"version\":\"2.31.6\"}",{"time":925,"blocks":926,"version":1450},1790377494031,[927,931,936,940,944,948,952,956,960,965,969,973,977,981,986,990,1007,1011,1015,1019,1023,1042,1046,1050,1054,1058,1062,1066,1088,1093,1097,1101,1105,1109,1113,1117,1121,1139,1143,1148,1152,1156,1160,1164,1168,1172,1176,1180,1184,1188,1192,1196,1200,1226,1230,1234,1238,1242,1246,1252,1256,1260,1280,1284,1288,1292,1296,1300,1304,1308,1312,1316,1320,1348,1353,1357,1361,1365,1369,1373,1396,1400,1418,1422,1429,1436,1443],{"id":215,"data":928,"type":218,"tunes":930},{"text":929},"RAG sounds complicated because the name is complicated. The idea is not. RAG simply means: before the AI answers, it first looks up relevant information from a knowledge source and gives that information to the language model.",{},{"id":221,"data":932,"type":226,"tunes":935},{"body":933,"title":934,"variant":225},"\u003Cstrong>RAG is the step where an AI searches a knowledge base for useful information before the LLM writes the answer.\u003C\u002Fstrong>","RAG in one sentence",{},{"id":229,"data":937,"type":218,"tunes":939},{"text":938},"Think of an LLM as a smart person sitting at a desk. RAG is the librarian who brings the right page from the right book. The LLM then reads that page and answers you.",{},{"id":234,"data":941,"type":239,"tunes":943},{"title":942,"maxLevel":237,"minLevel":238},"Contents",{},{"id":242,"data":945,"type":42,"tunes":947},{"text":946,"level":238},"First: what does the LLM do?",{},{"id":247,"data":949,"type":218,"tunes":951},{"text":950},"The LLM is the part that understands language and produces language. It can read your question, understand instructions, compare information, explain something and write an answer.",{},{"id":252,"data":953,"type":218,"tunes":955},{"text":954},"But the LLM does not automatically know what is currently inside your company database, your game session, your private documents or a file you created five minutes ago.",{},{"id":257,"data":957,"type":218,"tunes":959},{"text":958},"It only knows what is already inside the model plus whatever information the application gives it in the current request.",{},{"id":262,"data":961,"type":226,"tunes":964},{"body":962,"title":963,"variant":266},"The LLM \u003Cstrong>thinks and writes\u003C\u002Fstrong>. It does not automatically own all of your current data.","Simple rule",{},{"id":269,"data":966,"type":42,"tunes":968},{"text":967,"level":238},"Then: what is the knowledge base?",{},{"id":274,"data":970,"type":218,"tunes":972},{"text":971},"A knowledge base is simply information the application can search.",{},{"id":279,"data":974,"type":218,"tunes":976},{"text":975},"It could contain PDFs, manuals, product documentation, support articles, contracts, game rules, weapon data, internal company documents, database records or other text.",{},{"id":284,"data":978,"type":218,"tunes":980},{"text":979},"The knowledge base can be local on your own machine. It can be on a server. It can be in a vector database. It can also be built from normal files. RAG does not mean Internet.",{},{"id":289,"data":982,"type":226,"tunes":985},{"body":983,"title":984,"variant":293},"\u003Cstrong>RAG does not require the Internet.\u003C\u002Fstrong> The information can be completely local.","Important",{},{"id":296,"data":987,"type":42,"tunes":989},{"text":988,"level":238},"So what does RAG actually do?",{},{"id":301,"data":991,"type":318,"tunes":1006},{"steps":992,"title":1005,"orientation":317},[993,996,999,1002],{"label":994,"description":995},"1. You ask a question","For example: Which ammunition does this weapon use?",{"label":997,"description":998},"2. RAG searches the knowledge base","The system looks for the small pieces of information most relevant to your question.",{"label":1000,"description":1001},"3. RAG gives those pieces to the LLM","The LLM receives the question plus the retrieved information.",{"label":1003,"description":1004},"4. The LLM writes the answer","It uses the retrieved information as context for the response.","The whole RAG process",{},{"id":321,"data":1008,"type":218,"tunes":1010},{"text":1009},"That is RAG.",{},{"id":326,"data":1012,"type":218,"tunes":1014},{"text":1013},"The full name is Retrieval-Augmented Generation. Retrieval means finding the relevant information. Augmented means adding that information to the model's context. Generation means the LLM writes the final answer.",{},{"id":331,"data":1016,"type":42,"tunes":1018},{"text":1017,"level":238},"A very simple example",{},{"id":336,"data":1020,"type":218,"tunes":1022},{"text":1021},"Imagine you have a local knowledge base about a game.",{},{"id":341,"data":1024,"type":359,"tunes":1041},{"content":1025,"stretched":43,"withHeadings":14},[1026,1029,1032,1035,1038],[1027,1028],"Knowledge base contains","Example",[1030,1031],"Weapons","AKM uses 7.62 mm ammunition",[1033,1034],"Healing items","Med Kit restores health",[1036,1037],"Attachments","This attachment works with these weapons",[1039,1040],"Map rules","This zone behaves in this way",{},{"id":362,"data":1043,"type":218,"tunes":1045},{"text":1044},"You ask: “Which ammunition does the AKM use?”",{},{"id":367,"data":1047,"type":218,"tunes":1049},{"text":1048},"RAG searches the knowledge base and finds the entry about the AKM. It gives that small piece of information to the LLM. The LLM then answers: “The AKM uses 7.62 mm ammunition.”",{},{"id":372,"data":1051,"type":218,"tunes":1053},{"text":1052},"The LLM did not need the entire database. RAG only brought the useful part.",{},{"id":377,"data":1055,"type":42,"tunes":1057},{"text":1056,"level":238},"Now the important part: RAG is not the current state",{},{"id":382,"data":1059,"type":218,"tunes":1061},{"text":1060},"This is where many explanations become confusing.",{},{"id":387,"data":1063,"type":218,"tunes":1065},{"text":1064},"RAG usually gives the AI knowledge. A state system gives the AI facts about what is true right now.",{},{"id":392,"data":1067,"type":419,"tunes":1087},{"rows":1068,"title":1081,"layout":359,"columns":1082},[1069,1072,1075,1078],{"id":396,"label":1070,"values":1071},"Weapon",[398,398],{"id":400,"label":1073,"values":1074},"Ammunition",[398,398],{"id":404,"label":1076,"values":1077},"Health",[398,398],{"id":408,"label":1079,"values":1080},"Enemy",[398,398],"Knowledge vs current state",[1083,1085],{"id":414,"label":1084},"RAG \u002F knowledge",{"id":417,"label":1086},"Current state",{},{"id":422,"data":1089,"type":226,"tunes":1092},{"body":1090,"title":1091,"variant":426},"RAG answers: \u003Cstrong>What is generally true?\u003C\u002Fstrong>\u003Cbr>State answers: \u003Cstrong>What is true right now?\u003C\u002Fstrong>","Do not mix these two",{},{"id":429,"data":1094,"type":42,"tunes":1096},{"text":1095,"level":238},"What is a state database?",{},{"id":434,"data":1098,"type":218,"tunes":1100},{"text":1099},"A state database or state store is simply a place where the application keeps current facts.",{},{"id":439,"data":1102,"type":218,"tunes":1104},{"text":1103},"In a game, the engine already knows things such as your health, position, inventory, ammunition, current mission, nearby objects and enemy status. An AI system can expose selected parts of that state to the model.",{},{"id":444,"data":1106,"type":218,"tunes":1108},{"text":1107},"In a business application, the same idea could be an order database, a customer record, a project status or the current value of a sensor.",{},{"id":449,"data":1110,"type":218,"tunes":1112},{"text":1111},"The state is created by the application itself as things happen. If you lose health, the game updates the health value. If you pick up ammunition, the inventory changes. If an order is paid, the business system changes the order status.",{},{"id":454,"data":1114,"type":226,"tunes":1116},{"body":1115,"title":963,"variant":225},"The application creates and updates \u003Cstrong>state\u003C\u002Fstrong>. RAG searches \u003Cstrong>knowledge\u003C\u002Fstrong>. The LLM uses both to decide what to say or do.",{},{"id":459,"data":1118,"type":42,"tunes":1120},{"text":1119,"level":238},"How the three pieces work together",{},{"id":464,"data":1122,"type":318,"tunes":1138},{"steps":1123,"title":1137,"orientation":317},[1124,1127,1129,1131,1134],{"label":1125,"description":1126},"1. Current state","The application tells the AI what is true now: health 41%, AKM equipped, 23 rounds.",{"label":471,"description":1128},"The system retrieves useful knowledge: how the weapon works, which healing item is available, or a relevant rule.",{"label":474,"description":1130},"The model receives the question, current state and retrieved knowledge.",{"label":1132,"description":1133},"4. Reasoning","The LLM combines those inputs and decides what answer or high-level action makes sense.",{"label":1135,"description":1136},"5. Application","If an action is required, the application or game engine executes it and updates the state again.","LLM + state + RAG",{},{"id":485,"data":1140,"type":218,"tunes":1142},{"text":1141},"So the basic architecture is:",{},{"id":490,"data":1144,"type":226,"tunes":1147},{"body":1145,"title":1146,"variant":266},"\u003Cstrong>State = what is true now\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>RAG = useful knowledge\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>LLM = understands, reasons and writes\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Application = performs the real action\u003C\u002Fstrong>","The simplest architecture",{},{"id":496,"data":1149,"type":42,"tunes":1151},{"text":1150,"level":238},"Does RAG always use a vector database?",{},{"id":501,"data":1153,"type":218,"tunes":1155},{"text":1154},"No.",{},{"id":506,"data":1157,"type":218,"tunes":1159},{"text":1158},"A vector database is a common way to build semantic search, but it is not the definition of RAG.",{},{"id":511,"data":1161,"type":218,"tunes":1163},{"text":1162},"The important part is retrieval: the system finds relevant external information and adds it to the LLM's context before the answer is generated.",{},{"id":516,"data":1165,"type":218,"tunes":1167},{"text":1166},"OpenAI's File Search, for example, can work with files stored in vector stores. Files are chunked into smaller pieces so the system can retrieve the parts that are relevant to a question. That is one implementation of the same basic idea.",{},{"id":521,"data":1169,"type":42,"tunes":1171},{"text":1170,"level":238},"What is an embedding, in plain English?",{},{"id":526,"data":1173,"type":218,"tunes":1175},{"text":1174},"You do not need to understand embeddings to understand RAG.",{},{"id":531,"data":1177,"type":218,"tunes":1179},{"text":1178},"But the simple version is this: an embedding is a numerical representation of meaning. It helps a search system find text that is conceptually similar even when the words are not exactly the same.",{},{"id":536,"data":1181,"type":218,"tunes":1183},{"text":1182},"For example, a normal keyword search may look for the exact words “car repair.” Semantic search can also understand that “fix my vehicle” is about a similar topic.",{},{"id":541,"data":1185,"type":218,"tunes":1187},{"text":1186},"That makes embeddings useful for RAG, but RAG can also use keyword search, database queries or a hybrid of several methods.",{},{"id":546,"data":1189,"type":42,"tunes":1191},{"text":1190,"level":238},"RAG is not memory either",{},{"id":551,"data":1193,"type":218,"tunes":1195},{"text":1194},"Memory is another concept that is often mixed together with RAG.",{},{"id":556,"data":1197,"type":218,"tunes":1199},{"text":1198},"Memory is usually information the system keeps about previous interactions or previous events. RAG is the mechanism used to retrieve relevant knowledge when it is needed.",{},{"id":561,"data":1201,"type":359,"tunes":1225},{"content":1202,"stretched":43,"withHeadings":14},[1203,1206,1208,1210,1213,1216,1219,1222],[1204,1205],"Part","Simple meaning",[568,1207],"The part that understands and generates language",[571,1209],"The part that looks up relevant knowledge before the answer",[1211,1212],"Knowledge base","The information RAG can search",[1214,1215],"State","What is true right now in the application or world",[1217,1218],"Memory","Information kept from previous interactions or events",[1220,1221],"Tool \u002F action","Something the AI is allowed to call or ask the application to do",[1223,1224],"Context","The information currently placed in front of the LLM for this request",{},{"id":590,"data":1227,"type":42,"tunes":1229},{"text":1228,"level":238},"A real game example: PUBG Ally",{},{"id":595,"data":1231,"type":218,"tunes":1233},{"text":1232},"PUBG Ally is a useful example because it makes the difference visible.",{},{"id":600,"data":1235,"type":218,"tunes":1237},{"text":1236},"KRAFTON describes live match state as a separate source of truth. The game exposes current facts through observation tools: current weapon, ammunition, health, safe-zone status, nearby items and combat situation.",{},{"id":605,"data":1239,"type":218,"tunes":1241},{"text":1240},"Knowledge lookup is a different job. The system can use curated knowledge about weapons, attachments, items and rules. NVIDIA's ACE Game Agent SDK also exposes a separate RAG API for retrieving knowledge from developer-built databases.",{},{"id":610,"data":1243,"type":218,"tunes":1245},{"text":1244},"That gives us the clean separation: the game engine says what is happening now, retrieval provides relevant knowledge, and the language model decides what the information means.",{},{"id":615,"data":1247,"type":621,"tunes":1251},{"url":617,"title":1248,"excerpt":1249,"ctaLabel":1250},"PUBG Ally Shows Why AI Teammates Need Two Brains: Fast Reflexes and Slow Reasoning","A practical game example showing how live state, language reasoning and deterministic game-side control can work together.","Read the PUBG Ally architecture article",{},{"id":624,"data":1253,"type":42,"tunes":1255},{"text":1254,"level":238},"One complete example",{},{"id":629,"data":1257,"type":218,"tunes":1259},{"text":1258},"Imagine you tell an AI teammate: “I am low on health. Should we attack?”",{},{"id":634,"data":1261,"type":318,"tunes":1279},{"steps":1262,"title":1278,"orientation":317},[1263,1265,1267,1269,1272,1275],{"label":1214,"description":1264},"The game reports: health 24%, one enemy nearby, two healing items available.",{"label":571,"description":1266},"The knowledge system retrieves the relevant rules for the healing item and perhaps information about the current weapon or tactical mechanic.",{"label":568,"description":1268},"The model combines your request, the current state and the retrieved knowledge.",{"label":1270,"description":1271},"Decision","It concludes that healing first is safer than attacking immediately.",{"label":1273,"description":1274},"Tool \u002F game engine","The agent requests a legal game action such as moving to cover or using the healing item.",{"label":1276,"description":1277},"New state","The game executes the action and reports the updated situation back to the agent.","What happens next",{},{"id":655,"data":1281,"type":218,"tunes":1283},{"text":1282},"RAG did not control the character. The state database did not reason. The LLM did not directly change the game. Each part had one job.",{},{"id":660,"data":1285,"type":42,"tunes":1287},{"text":1286,"level":238},"Why use RAG at all?",{},{"id":665,"data":1289,"type":218,"tunes":1291},{"text":1290},"Because putting every document, rule and database record into every prompt would be slow, expensive and often confusing.",{},{"id":670,"data":1293,"type":218,"tunes":1295},{"text":1294},"RAG lets the system select only the information that is useful for the current question.",{},{"id":675,"data":1297,"type":218,"tunes":1299},{"text":1298},"It also lets you update the knowledge base without retraining the entire language model. Change the document or database, rebuild or refresh the index when necessary, and the next retrieval can use the newer information.",{},{"id":680,"data":1301,"type":42,"tunes":1303},{"text":1302,"level":238},"What RAG does not guarantee",{},{"id":685,"data":1305,"type":218,"tunes":1307},{"text":1306},"RAG can improve grounding, but it does not make an answer automatically correct.",{},{"id":690,"data":1309,"type":218,"tunes":1311},{"text":1310},"The retrieval step can find the wrong document. The correct document can be outdated. The LLM can misunderstand good evidence. Or the current state can have changed.",{},{"id":695,"data":1313,"type":218,"tunes":1315},{"text":1314},"A reliable system therefore has to validate retrieval, state freshness and the model's final reasoning separately.",{},{"id":700,"data":1317,"type":42,"tunes":1319},{"text":1318,"level":238},"The easiest mental model to remember",{},{"id":705,"data":1321,"type":419,"tunes":1347},{"rows":1322,"title":1341,"layout":359,"columns":1342},[1323,1326,1329,1332,1335,1338],{"id":709,"label":1324,"values":1325},"Person thinking",[398,398],{"id":713,"label":1327,"values":1328},"Finding a reference book",[398,398],{"id":717,"label":1330,"values":1331},"Books on the shelf",[398,398],{"id":721,"label":1333,"values":1334},"Current dashboard or instrument panel",[398,398],{"id":725,"label":1336,"values":1337},"Notes from earlier meetings",[398,398],{"id":729,"label":1339,"values":1340},"Doing something in the real world",[398,398],"Think of an AI system like a person at a desk",[1343,1345],{"id":229,"label":1344},"Analogy",{"id":737,"label":1346},"AI system",{},{"id":741,"data":1349,"type":226,"tunes":1352},{"body":1350,"title":1351,"variant":293},"\u003Cstrong>LLM = brain.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>RAG = librarian.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Knowledge base = library.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>State = what the dashboard says right now.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Tools = the hands that can actually do something.\u003C\u002Fstrong>","If you remember only this",{},{"id":747,"data":1354,"type":42,"tunes":1356},{"text":1355,"level":238},"Conclusion",{},{"id":752,"data":1358,"type":218,"tunes":1360},{"text":1359},"RAG is much less mysterious once the parts are separated.",{},{"id":757,"data":1362,"type":218,"tunes":1364},{"text":1363},"The LLM understands and generates language. The application maintains current state. The knowledge base stores information. RAG finds the useful part of that information and puts it into the LLM's context. Tools or the application perform real actions.",{},{"id":762,"data":1366,"type":218,"tunes":1368},{"text":1367},"That is the basic architecture behind many modern AI assistants and agents.",{},{"id":767,"data":1370,"type":42,"tunes":1372},{"text":1371,"level":238},"FAQ",{},{"id":772,"data":1374,"type":772,"tunes":1395},{"items":1375,"title":1394},[1376,1379,1382,1385,1388,1391],{"id":776,"answer":1377,"question":1378},"RAG is a step where an AI searches a knowledge source for relevant information before the language model writes its answer.","What is RAG in simple terms?",{"id":780,"answer":1380,"question":1381},"No. The knowledge base can be completely local on your computer or server.","Does RAG need the Internet?",{"id":784,"answer":1383,"question":1384},"No. The database or files contain the information. RAG is the retrieval process that finds the useful part and gives it to the LLM.","Is RAG the same as a database?",{"id":788,"answer":1386,"question":1387},"No. Memory usually stores previous interactions or events. RAG retrieves relevant knowledge when it is needed.","Is RAG the same as memory?",{"id":792,"answer":1389,"question":1390},"Not necessarily. Current state is usually obtained directly from the application or a state store. RAG is better understood as retrieval from a knowledge source.","Is current application state part of RAG?",{"id":796,"answer":1392,"question":1393},"No. It can provide better evidence, but retrieval can still be wrong or outdated and the LLM can still reason incorrectly.","Does RAG make AI answers correct?","RAG in plain English",{},{"id":802,"data":1397,"type":42,"tunes":1399},{"text":1398,"level":238},"Glossary",{},{"id":807,"data":1401,"type":807,"tunes":1417},{"title":1402,"entries":1403},"The basic terms",[1404,1406,1408,1410,1412,1414],{"term":568,"anchor":812,"definition":1405},"A language model that understands and generates text and can reason over information placed in its context.",{"term":571,"anchor":815,"definition":1407},"Retrieval-Augmented Generation: retrieving relevant external information and adding it to the model's context before generating an answer.",{"term":1211,"anchor":818,"definition":1409},"The files, documents, records or other information that retrieval can search.",{"term":1214,"anchor":417,"definition":1411},"The current facts of an application, system or world at a particular moment.",{"term":1223,"anchor":823,"definition":1413},"The information currently supplied to the language model for one request or reasoning step.",{"term":1415,"anchor":827,"definition":1416},"Embedding","A numerical representation of meaning that can help semantic search find conceptually similar information.",{},{"id":831,"data":1419,"type":42,"tunes":1421},{"text":1420,"level":238},"Primary sources",{},{"id":836,"data":1423,"type":843,"tunes":1428},{"link":838,"meta":1424},{"image":1425,"title":1426,"description":1427},{"url":398},"OpenAI — Vector Store Files","Official documentation showing how files can be attached to vector stores, chunked and made available to file-search retrieval.",{},{"id":846,"data":1430,"type":843,"tunes":1435},{"link":848,"meta":1431},{"image":1432,"title":1433,"description":1434},{"url":398},"OpenAI — Developer Quickstart","Official OpenAI documentation describing tools such as file search for giving models access to external information.",{},{"id":855,"data":1437,"type":843,"tunes":1442},{"link":857,"meta":1438},{"image":1439,"title":1440,"description":1441},{"url":398},"NVIDIA Developer — ACE for Games","Official NVIDIA documentation describing separate Agent, Chat and RAG APIs for connecting game characters to game state, contextual knowledge and model-driven actions.",{},{"id":864,"data":1444,"type":843,"tunes":1449},{"link":866,"meta":1445},{"image":1446,"title":1447,"description":1448},{"url":398},"NVIDIA Developer — How KRAFTON Built PUBG Ally","Official technical explanation separating live match state from knowledge lookup and language-model reasoning.",{},"2.31.6","RAG sounds complicated, but the idea is simple: before an AI answers, it first looks up useful information from a knowledge source and gives that information to the language model. This guide explains RAG, LLMs, state, memory and tools using one simple mental model.",{"lang":7,"title":208,"content":210,"contentJson":1453,"excerpt":873},{"time":212,"blocks":1454,"version":872},[1455,1458,1461,1464,1467,1470,1473,1476,1479,1482,1485,1488,1491,1494,1497,1500,1508,1511,1514,1517,1520,1529,1532,1535,1538,1541,1544,1547,1562,1565,1568,1571,1574,1577,1580,1583,1586,1595,1598,1601,1604,1607,1610,1613,1616,1619,1622,1625,1628,1631,1634,1637,1640,1652,1655,1658,1661,1664,1667,1670,1673,1676,1686,1689,1692,1695,1698,1701,1704,1707,1710,1713,1716,1735,1738,1741,1744,1747,1750,1753,1763,1766,1776,1779,1784,1789,1794],{"id":215,"data":1456,"type":218,"tunes":1457},{"text":217},{},{"id":221,"data":1459,"type":226,"tunes":1460},{"body":223,"title":224,"variant":225},{},{"id":229,"data":1462,"type":218,"tunes":1463},{"text":231},{},{"id":234,"data":1465,"type":239,"tunes":1466},{"title":236,"maxLevel":237,"minLevel":238},{},{"id":242,"data":1468,"type":42,"tunes":1469},{"text":244,"level":238},{},{"id":247,"data":1471,"type":218,"tunes":1472},{"text":249},{},{"id":252,"data":1474,"type":218,"tunes":1475},{"text":254},{},{"id":257,"data":1477,"type":218,"tunes":1478},{"text":259},{},{"id":262,"data":1480,"type":226,"tunes":1481},{"body":264,"title":265,"variant":266},{},{"id":269,"data":1483,"type":42,"tunes":1484},{"text":271,"level":238},{},{"id":274,"data":1486,"type":218,"tunes":1487},{"text":276},{},{"id":279,"data":1489,"type":218,"tunes":1490},{"text":281},{},{"id":284,"data":1492,"type":218,"tunes":1493},{"text":286},{},{"id":289,"data":1495,"type":226,"tunes":1496},{"body":291,"title":292,"variant":293},{},{"id":296,"data":1498,"type":42,"tunes":1499},{"text":298,"level":238},{},{"id":301,"data":1501,"type":318,"tunes":1507},{"steps":1502,"title":316,"orientation":317},[1503,1504,1505,1506],{"label":305,"description":306},{"label":308,"description":309},{"label":311,"description":312},{"label":314,"description":315},{},{"id":321,"data":1509,"type":218,"tunes":1510},{"text":323},{},{"id":326,"data":1512,"type":218,"tunes":1513},{"text":328},{},{"id":331,"data":1515,"type":42,"tunes":1516},{"text":333,"level":238},{},{"id":336,"data":1518,"type":218,"tunes":1519},{"text":338},{},{"id":341,"data":1521,"type":359,"tunes":1528},{"content":1522,"stretched":43,"withHeadings":14},[1523,1524,1525,1526,1527],[345,346],[348,349],[351,352],[354,355],[357,358],{},{"id":362,"data":1530,"type":218,"tunes":1531},{"text":364},{},{"id":367,"data":1533,"type":218,"tunes":1534},{"text":369},{},{"id":372,"data":1536,"type":218,"tunes":1537},{"text":374},{},{"id":377,"data":1539,"type":42,"tunes":1540},{"text":379,"level":238},{},{"id":382,"data":1542,"type":218,"tunes":1543},{"text":384},{},{"id":387,"data":1545,"type":218,"tunes":1546},{"text":389},{},{"id":392,"data":1548,"type":419,"tunes":1561},{"rows":1549,"title":411,"layout":359,"columns":1558},[1550,1552,1554,1556],{"id":396,"label":348,"values":1551},[398,398],{"id":400,"label":401,"values":1553},[398,398],{"id":404,"label":405,"values":1555},[398,398],{"id":408,"label":409,"values":1557},[398,398],[1559,1560],{"id":414,"label":415},{"id":417,"label":418},{},{"id":422,"data":1563,"type":226,"tunes":1564},{"body":424,"title":425,"variant":426},{},{"id":429,"data":1566,"type":42,"tunes":1567},{"text":431,"level":238},{},{"id":434,"data":1569,"type":218,"tunes":1570},{"text":436},{},{"id":439,"data":1572,"type":218,"tunes":1573},{"text":441},{},{"id":444,"data":1575,"type":218,"tunes":1576},{"text":446},{},{"id":449,"data":1578,"type":218,"tunes":1579},{"text":451},{},{"id":454,"data":1581,"type":226,"tunes":1582},{"body":456,"title":265,"variant":225},{},{"id":459,"data":1584,"type":42,"tunes":1585},{"text":461,"level":238},{},{"id":464,"data":1587,"type":318,"tunes":1594},{"steps":1588,"title":482,"orientation":317},[1589,1590,1591,1592,1593],{"label":468,"description":469},{"label":471,"description":472},{"label":474,"description":475},{"label":477,"description":478},{"label":480,"description":481},{},{"id":485,"data":1596,"type":218,"tunes":1597},{"text":487},{},{"id":490,"data":1599,"type":226,"tunes":1600},{"body":492,"title":493,"variant":266},{},{"id":496,"data":1602,"type":42,"tunes":1603},{"text":498,"level":238},{},{"id":501,"data":1605,"type":218,"tunes":1606},{"text":503},{},{"id":506,"data":1608,"type":218,"tunes":1609},{"text":508},{},{"id":511,"data":1611,"type":218,"tunes":1612},{"text":513},{},{"id":516,"data":1614,"type":218,"tunes":1615},{"text":518},{},{"id":521,"data":1617,"type":42,"tunes":1618},{"text":523,"level":238},{},{"id":526,"data":1620,"type":218,"tunes":1621},{"text":528},{},{"id":531,"data":1623,"type":218,"tunes":1624},{"text":533},{},{"id":536,"data":1626,"type":218,"tunes":1627},{"text":538},{},{"id":541,"data":1629,"type":218,"tunes":1630},{"text":543},{},{"id":546,"data":1632,"type":42,"tunes":1633},{"text":548,"level":238},{},{"id":551,"data":1635,"type":218,"tunes":1636},{"text":553},{},{"id":556,"data":1638,"type":218,"tunes":1639},{"text":558},{},{"id":561,"data":1641,"type":359,"tunes":1651},{"content":1642,"stretched":43,"withHeadings":14},[1643,1644,1645,1646,1647,1648,1649,1650],[565,566],[568,569],[571,572],[574,575],[577,578],[580,581],[583,584],[586,587],{},{"id":590,"data":1653,"type":42,"tunes":1654},{"text":592,"level":238},{},{"id":595,"data":1656,"type":218,"tunes":1657},{"text":597},{},{"id":600,"data":1659,"type":218,"tunes":1660},{"text":602},{},{"id":605,"data":1662,"type":218,"tunes":1663},{"text":607},{},{"id":610,"data":1665,"type":218,"tunes":1666},{"text":612},{},{"id":615,"data":1668,"type":621,"tunes":1669},{"url":617,"title":618,"excerpt":619,"ctaLabel":620},{},{"id":624,"data":1671,"type":42,"tunes":1672},{"text":626,"level":238},{},{"id":629,"data":1674,"type":218,"tunes":1675},{"text":631},{},{"id":634,"data":1677,"type":318,"tunes":1685},{"steps":1678,"title":652,"orientation":317},[1679,1680,1681,1682,1683,1684],{"label":577,"description":638},{"label":571,"description":640},{"label":568,"description":642},{"label":644,"description":645},{"label":647,"description":648},{"label":650,"description":651},{},{"id":655,"data":1687,"type":218,"tunes":1688},{"text":657},{},{"id":660,"data":1690,"type":42,"tunes":1691},{"text":662,"level":238},{},{"id":665,"data":1693,"type":218,"tunes":1694},{"text":667},{},{"id":670,"data":1696,"type":218,"tunes":1697},{"text":672},{},{"id":675,"data":1699,"type":218,"tunes":1700},{"text":677},{},{"id":680,"data":1702,"type":42,"tunes":1703},{"text":682,"level":238},{},{"id":685,"data":1705,"type":218,"tunes":1706},{"text":687},{},{"id":690,"data":1708,"type":218,"tunes":1709},{"text":692},{},{"id":695,"data":1711,"type":218,"tunes":1712},{"text":697},{},{"id":700,"data":1714,"type":42,"tunes":1715},{"text":702,"level":238},{},{"id":705,"data":1717,"type":419,"tunes":1734},{"rows":1718,"title":732,"layout":359,"columns":1731},[1719,1721,1723,1725,1727,1729],{"id":709,"label":710,"values":1720},[398,398],{"id":713,"label":714,"values":1722},[398,398],{"id":717,"label":718,"values":1724},[398,398],{"id":721,"label":722,"values":1726},[398,398],{"id":725,"label":726,"values":1728},[398,398],{"id":729,"label":730,"values":1730},[398,398],[1732,1733],{"id":229,"label":735},{"id":737,"label":738},{},{"id":741,"data":1736,"type":226,"tunes":1737},{"body":743,"title":744,"variant":293},{},{"id":747,"data":1739,"type":42,"tunes":1740},{"text":749,"level":238},{},{"id":752,"data":1742,"type":218,"tunes":1743},{"text":754},{},{"id":757,"data":1745,"type":218,"tunes":1746},{"text":759},{},{"id":762,"data":1748,"type":218,"tunes":1749},{"text":764},{},{"id":767,"data":1751,"type":42,"tunes":1752},{"text":769,"level":238},{},{"id":772,"data":1754,"type":772,"tunes":1762},{"items":1755,"title":799},[1756,1757,1758,1759,1760,1761],{"id":776,"answer":777,"question":778},{"id":780,"answer":781,"question":782},{"id":784,"answer":785,"question":786},{"id":788,"answer":789,"question":790},{"id":792,"answer":793,"question":794},{"id":796,"answer":797,"question":798},{},{"id":802,"data":1764,"type":42,"tunes":1765},{"text":804,"level":238},{},{"id":807,"data":1767,"type":807,"tunes":1775},{"title":809,"entries":1768},[1769,1770,1771,1772,1773,1774],{"term":568,"anchor":812,"definition":813},{"term":571,"anchor":815,"definition":816},{"term":574,"anchor":818,"definition":819},{"term":577,"anchor":417,"definition":821},{"term":586,"anchor":823,"definition":824},{"term":826,"anchor":827,"definition":828},{},{"id":831,"data":1777,"type":42,"tunes":1778},{"text":833,"level":238},{},{"id":836,"data":1780,"type":843,"tunes":1783},{"link":838,"meta":1781},{"image":1782,"title":841,"description":842},{"url":398},{},{"id":846,"data":1785,"type":843,"tunes":1788},{"link":848,"meta":1786},{"image":1787,"title":851,"description":852},{"url":398},{},{"id":855,"data":1790,"type":843,"tunes":1793},{"link":857,"meta":1791},{"image":1792,"title":860,"description":861},{"url":398},{},{"id":864,"data":1795,"type":843,"tunes":1798},{"link":866,"meta":1796},{"image":1797,"title":869,"description":870},{"url":398},{},"Post erfolgreich abgerufen",{"items":1801,"source":1886,"manualIds":1887,"manualMatchedIds":1888},[1802,1809,1816,1823,1830,1837,1844,1851,1858,1865,1872,1879],{"id":1803,"slug":1804,"title":1805,"excerpt":1806,"featuredImage":1807,"publishedAt":1808},"466","the-gpu-is-not-the-product-future-proof-private-ai-architecture","GPU 不是产品：面向未来的私有 AI 架构","私有 AI 基础设施不应围绕单一 GPU 或单一模型来设计。更具韧性的做法是将快速推理 GPU、内存充裕的 AI 系统、物理 AI 节点以及可选的前沿云模型，统一置于一个具备能力感知的路由层之后。","\u002Fuploads\u002F2026\u002F09\u002Fthe-gpu-is-not-the-product-future-proof-private-ai-architecture-1790140878812-8hsl39.webp","2026-09-23T01:19:00.000Z",{"id":1810,"slug":1811,"title":1812,"excerpt":1813,"featuredImage":1814,"publishedAt":1815},"473","openai-agents-api-vs-agents-sdk-vs-responses-api-what-should-you-build-on-in-2026","OpenAI Agents API 与 Agents SDK 与 Responses API：2026 年你应该基于什么来构建？","OpenAI 的智能体技术栈在 2026 年 9 月发生了变化。本架构指南按运行时归属将 Agents API、Agents SDK、Responses API 和 Codex SDK 区分开来——以便团队能够选择正确的控制边界，而不是比较产品名称。","\u002Fuploads\u002F2026\u002F09\u002Fopenai-agents-api-vs-agents-sdk-vs-responses-api-what-should-you-build-on-in-2026-1790351846714-zi7lus.webp","2026-09-25T11:56:00.000Z",{"id":1817,"slug":1818,"title":1819,"excerpt":1820,"featuredImage":1821,"publishedAt":1822},"476","mcp-vs-a2a-vs-ucp-vs-ap2-vs-a2ui-the-agent-protocol-stack-explained","MCP vs A2A vs UCP vs AP2 vs A2UI：智能体协议栈详解","MCP、A2A、UCP、AP2 和 A2UI 常被描述为相互竞争的智能体标准。它们大多解决的是不同的互操作性问题。本指南将每个协议映射到其实际标准化的边界，并展示它们如何在同一个生产系统中协同工作。","\u002Fuploads\u002F2026\u002F09\u002Fmcp-vs-a2a-vs-ucp-vs-ap2-vs-a2ui-the-agent-protocol-stack-explained-1790352625869-2ezle0.webp","2026-09-25T12:09:00.000Z",{"id":1824,"slug":1825,"title":1826,"excerpt":1827,"featuredImage":1828,"publishedAt":1829},"472","why-more-context-can-make-ai-answers-worse","为什么更多上下文会让AI的回答更糟","更大的上下文窗口并不保证更好的答案。本文解释了信号稀释、证据冲突、状态过时、位置敏感性和有损压缩如何降低AI可靠性——并介绍了一种实用的上下文压力测试。","\u002Fuploads\u002F2026\u002F09\u002Fwhy-more-context-can-make-ai-answers-worse-1790351615793-2ntv2v.webp","2026-09-25T11:51:00.000Z",{"id":1831,"slug":1832,"title":1833,"excerpt":1834,"featuredImage":1835,"publishedAt":1836},"469","rag-failed-but-which-layer-actually-failed-a-diagnostic-method","RAG失败了——但究竟是哪一层真正失败了？一种诊断方法","当RAG答案出错时，将问题归咎于检索或模型过于笼统。这种诊断方法将来源覆盖、查询构建、检索、排序、上下文组装、生成、证据归因和时效性逐一隔离，从而使实际故障能够被复现并修复。","\u002Fuploads\u002F2026\u002F09\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method-1790350847177-pior4c.webp","2026-09-24T19:39:00.000Z",{"id":1838,"slug":1839,"title":1840,"excerpt":1841,"featuredImage":1842,"publishedAt":1843},"468","ai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context","AI代理记忆不是RAG：如何区分记忆、检索、状态和上下文","代理记忆、RAG、状态和上下文经常被当作可以互换的概念来使用。它们并不是。这个实用的架构模型将这四个层次区分开来，展示了每一层各自应处的位置，并解释了当系统将它们合并为一层时会出现什么问题。","\u002Fuploads\u002F2026\u002F09\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context-1790350560308-np0xy6.webp","2026-09-25T11:34:00.000Z",{"id":1845,"slug":1846,"title":1847,"excerpt":1848,"featuredImage":1849,"publishedAt":1850},"467","the-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers","答案有效性边界：相关性到可靠AI答案之间缺失的层级","一个来源可能相关、权威，但对于所提出的问题仍然是错误的。缺失的层次是适用性：答案成立的条件，以及迫使其被重新考虑的变化。本文介绍了“答案有效性边界”这一面向人类、AI搜索和RAG系统的来源设计模式。","\u002Fuploads\u002F2026\u002F09\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers-1790272901306-1g5jly.webp","2026-09-24T11:59:00.000Z",{"id":1852,"slug":1853,"title":1854,"excerpt":1855,"featuredImage":1856,"publishedAt":1857},"477","computer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system","计算机使用代理：为什么成功的演示仍可能是一个不可靠的系统","计算机使用代理如今能够完成令人印象深刻的浏览器和桌面工作流程，但一次成功的运行证明的是能力——而非可靠性。本文展示了如何测试可重复性、环境鲁棒性、长时程控制、状态感知、结果验证以及安全的目标处理。","\u002Fuploads\u002F2026\u002F09\u002Fcomputer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system-1790352854690-75qnrg.webp","2026-09-25T12:13:00.000Z",{"id":1859,"slug":1860,"title":1861,"excerpt":1862,"featuredImage":1863,"publishedAt":1864},"460","ai-agent-reliability-why-the-final-answer-is-not-enough","AI Agent可靠性：为什么最终答案并不足够","正确的输出并不能证明推理的正确性、执行的安全性，或系统的可信赖性。","\u002Fuploads\u002F2026\u002F09\u002Fai-agent-reliability-why-the-final-answer-is-not-enough-1788955466306-pl0qhz.webp","2026-09-09T04:01:00.000Z",{"id":1866,"slug":1867,"title":1868,"excerpt":1869,"featuredImage":1870,"publishedAt":1871},"459","ollama-is-not-the-product-building-production-ready-open-llm-applications","Ollama 并非产品：构建可投入生产的开源大语言模型应用","使用Ollama运行本地模型很简单。但构建一个可用于生产环境的开源大语言模型（Open-LLM）应用则更具挑战性：它需要RAG（检索增强生成）、访问控制、供应商抽象、评估、日志记录、部署规范，以及围绕模型构建受控的应用层。","\u002Fuploads\u002F2026\u002F06\u002Follama-is-not-the-product-building-production-ready-open-llm-applications-1782679361640-h0usqf.webp","2026-06-28T16:39:00.000Z",{"id":1873,"slug":1874,"title":1875,"excerpt":1876,"featuredImage":1877,"publishedAt":1878},"364","tipps-fuer-die-verbesserung-der-seo-suchmaschinenoptimierung","Mastering the SEO Workflow: Essential Optimization Strategies for Organic Growth","A structured SEO workflow is crucial for sustainable organic growth. Learn the ten foundational strategies, from keyword research and technical optimization to content quality and performance analysis.","\u002Fuploads\u002F2026\u002F03\u002Ftipps-fuer-die-verbesserung-der-seo-suchmaschinenoptimierung-1774866098131-hwkzrg.webp","2024-01-26T06:35:00.000Z",{"id":1880,"slug":1881,"title":1882,"excerpt":1883,"featuredImage":1884,"publishedAt":1885},"470","what-should-an-ai-agent-remember-forget-recompute-or-retrieve-again","AI代理应该记住、遗忘、重新计算还是再次检索什么？","长时间运行的代理不应记住所有内容。本文提供了一个实用的生命周期模型，用于决定哪些内容应属于持久记忆、哪些内容应重新检索、哪些内容重新计算更安全，以及哪些内容应过期或被取代。","\u002Fuploads\u002F2026\u002F09\u002Fwhat-should-an-ai-agent-remember-forget-recompute-or-retrieve-again-1790351131087-iehz28.webp","2026-09-25T09:43:00.000Z","fallback",[],[]]