[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"portal-settings:stajic:zh":3,"public-menus:all":38,"post:when-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger:zh":205,"related:post:when-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger:zh:1":2231},{"statusCode":4,"data":5,"message":37},200,{"tenantId":6,"lang":7,"defaultLang":8,"siteUrl":9,"contactEmail":10,"brandName":11,"logoUrl":12,"siteName":11,"siteDescription":13,"ogImage":10,"robotsIndex":14,"socialLinks":10,"reservedSlugs":10,"seoPolicy":15},"stajic","zh","de","https:\u002F\u002Fstajic.de",null,"Stajic Platform","\u002FLogo_Planet.svg","Stajic Portal",true,{"branding":16,"relatedContent":17,"crossDomainLinks":18},{"logoUrl":12},{"enabled":14},[19,22,25,28,31,34],{"url":20,"label":21,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Ffigure.rocks","figure.rocks",{"url":23,"label":24,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Floving.rocks","loving.rocks",{"url":26,"label":27,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.com","bazify.com",{"url":29,"label":30,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.de","bazify.de",{"url":32,"label":33,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.at","bazify.at",{"url":35,"label":36,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.ba","bazify.ba","Portal settings resolved",[39,45],{"id":40,"name":41,"location":42,"isActive":14,"isDefault":43,"items":44},1,"main-navigation","header",false,[],{"id":46,"name":47,"location":48,"isActive":14,"isDefault":14,"items":49},4,"main-menu","sidebar",[50,66,79,93,103,118,133],{"id":51,"title":52,"url":60,"target":61,"icon":62,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":64,"portfolioId":10,"children":65},"item-18",{"de":53,"en":54,"es":55,"fr":56,"it":54,"ru":57,"sr":58,"zh":59},"Startseite","Home","Inicio","Accueil","Главная","Почетна","首页","\u002Ffull-stack-web-developer-munich-performance-seo-and-maintainable-builds","_self","i-lucide-home","page",111,[],{"id":67,"title":68,"url":75,"target":61,"icon":76,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":77,"portfolioId":10,"children":78},"item-22",{"de":69,"en":69,"es":70,"fr":69,"it":71,"ru":72,"sr":73,"zh":74},"Vision","Visión","Visione","Видение","Визија","想象","\u002Fueber-uns-webdesign-muenchen-webaplikation","i-lucide-eye",113,[],{"id":80,"title":81,"url":89,"target":61,"icon":90,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":91,"portfolioId":10,"children":92},"item-19",{"de":82,"en":83,"es":84,"fr":83,"it":85,"ru":86,"sr":87,"zh":88},"Leistungen","Services","Servicios","Servizi","Услуги","Услуге","服务","\u002Fservices-dienstleistungen-muenchen","i-lucide-wrench",116,[],{"id":94,"title":95,"url":99,"target":61,"icon":100,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":101,"portfolioId":10,"children":102},"item-23",{"de":96,"en":96,"es":96,"fr":96,"it":96,"ru":97,"sr":97,"zh":98},"Blog","Блог","博客","\u002Fblog","i-lucide-book-open",112,[],{"id":104,"title":105,"url":114,"target":61,"icon":115,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":116,"portfolioId":10,"children":117},"item-32",{"de":106,"en":107,"es":108,"fr":109,"it":110,"ru":111,"sr":112,"zh":113},"Neue Technologien","New Technologies","Nuevas tecnologías","Nouvelles technologies","Nuove tecnologie","Новые технологии","Нове технологије","新技术！","\u002Fneue-webtechnologien","i-lucide-sparkles",122,[],{"id":119,"title":120,"url":129,"target":61,"icon":130,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":131,"portfolioId":10,"children":132},"item-20",{"de":121,"en":122,"es":123,"fr":124,"it":125,"ru":126,"sr":127,"zh":128},"Kontakt","Contact us!","Contacto","Contact","Contatto","Контакт","Контактирајте нас","联系我们！","\u002Fcontact","i-lucide-mail",115,[],{"id":134,"title":135,"url":144,"target":61,"icon":145,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":147},"item-21",{"de":136,"en":137,"es":138,"fr":139,"it":140,"ru":141,"sr":142,"zh":143},"Unsere Arbeit","Our Work","Nuestro trabajo","Nos réalisations","I nostri lavori","Наши работы","Наши радови","文件夹","\u002Fportfolio","i-lucide-briefcase",114,[148,161,175,181,193],{"id":149,"title":150,"url":144,"target":61,"icon":159,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":160},"item-24",{"de":151,"en":152,"es":153,"fr":154,"it":155,"ru":156,"sr":157,"zh":158},"Alle Projekte","All Projects","Todos los proyectos","Tous les projets","Tutti i progetti","Все проекты","Сви пројекти","所有项目","i-lucide-grid-3x3",[],{"id":162,"title":163,"url":171,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":174},"item-29",{"de":164,"en":165,"es":166,"fr":167,"it":168,"ru":169,"sr":170,"zh":143},"Local Roots, Global Reach","Local Roots - Global Reach","Empresa local ","Entreprise locale","Azienda locale","Местная компания","Локално предузеће глобално тржиште","\u002Fportfolio\u002Flocal-roots-global-reach-communication-media-systems-for-modern-business","i-lucide-folder","custom",[],{"id":176,"title":177,"url":179,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":180},"item-28",{"de":178,"en":178,"es":178,"fr":178,"it":178,"ru":178,"sr":178,"zh":178},"Solr Suggester","\u002Fportfolio\u002Fsolr-fuzzy-suggester-und-solr-infix-suggester-abfrage-ueber-ajax-und-filterung",[],{"id":182,"title":183,"url":191,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":192},"item-27",{"de":184,"en":185,"es":186,"fr":187,"it":188,"ru":189,"sr":190,"zh":185},"Firmenwebseite SEO","Company Website SEO","Sitio web corporativo SEO","Site web d’entreprise SEO","Sito web aziendale SEO","Корпоративный сайт SEO","Пословна веб-страница SEO","\u002Fportfolio\u002Fseo-sem-branding-mobile-webseite-muenchen",[],{"id":194,"title":195,"url":203,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":204},"item-31",{"de":196,"en":197,"es":198,"fr":199,"it":200,"ru":201,"sr":202,"zh":197},"Digitalisierungsportal","Digitalization Portal","Portal de digitalización","Portail de numérisation","Portale di digitalizzazione","Портал цифровизации","Портал за дигитализацију","\u002Fportfolio\u002Fdigitalisierungsportal-archiv-museum-bibliothek-ead-lido-mets-mods",[],{"statusCode":4,"data":206,"message":2230},{"id":207,"title":208,"slug":209,"content":210,"contentJson":211,"excerpt":1045,"featuredImage":1046,"featuredImageAlt":1047,"featuredImageCaption":10,"featuredImageTitle":10,"featuredImageCopyright":10,"featuredImageAuthor":10,"featuredImageSourceUrl":10,"featuredImageLicense":10,"featuredImageIsAiGenerated":43,"status":1048,"publishedAt":1049,"createdAt":1050,"updatedAt":1051,"seoLocalePaths":1052,"categories":1061,"author":1085,"translations":1090},"480","人工智能何时应停止信任自身知识？——检索触发机制","when-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger","\u003Ch2 id=\"section-1\">问题\u003C\u002Fh2>\n\u003Cp>AI 应该在什么时候停止依赖它已经知道的内容，并在回答之前检索外部信息？\u003C\u002Fp>\n\u003Cp>这个问题看起来很简单，但它处于现代 AI 系统最重要的设计决策之一的核心位置。\u003C\u002Fp>\n\u003Cp>大型语言模型在其参数中包含了大量知识。检索增强生成会在运行时添加外部信息。但两种极端都不理想。\u003C\u002Fp>\n\u003Cp>总是信任模型可能会产生过时或缺乏支持的答案。总是检索信息会增加延迟、成本、无关上下文，并为检索错误带来新的机会。\u003C\u002Fp>\n\u003Cp>因此，真正的问题出现在 RAG 之前：究竟应该在什么时候进行检索？\u003C\u002Fp>\n\u003Cp>本文使用“检索触发器”这一术语来指代该决策。这里并不是将“检索触发器”作为研究文献中的标准化术语提出。它是一个实用的系统概念，汇集了主动式、自适应式和自我反思式检索研究中已经可见的思想。\u003C\u002Fp>\n\u003Cblockquote class=\"border-l-4 border-gray-300 pl-4 italic\">检索触发器是一种条件，表示 AI 系统应停止仅依赖内部模型知识，并在生成或最终确定答案之前获取外部证据。\u003Ccite class=\"block mt-2 text-sm\">— 工作定义\u003C\u002Fcite>\u003C\u002Fblockquote>\n\u003Cnav class=\"editorjs-toc\" data-editorjs-toc=\"true\" aria-label=\"目录\">\u003Cstrong class=\"editorjs-toc__title\">目录\u003C\u002Fstrong>\u003Col class=\"editorjs-toc__list editorjs-toc__list--depth-0\">\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-1\" class=\"editorjs-toc__link\">问题\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-10\" class=\"editorjs-toc__link\">这真正意味着什么\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-20\" class=\"editorjs-toc__link\">最简单的例子\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-33\" class=\"editorjs-toc__link\">例子失效之处\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-43\" class=\"editorjs-toc__link\">直接回答\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-48\" class=\"editorjs-toc__link\">为何如此\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-55\" class=\"editorjs-toc__link\">背景\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-65\" class=\"editorjs-toc__link\">假设\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-71\" class=\"editorjs-toc__link\">变量\u003C\u002Fa>\u003Col class=\"editorjs-toc__list editorjs-toc__list--depth-1\">\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-73\" class=\"editorjs-toc__link\">时效性\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-75\" class=\"editorjs-toc__link\">特定性\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-77\" class=\"editorjs-toc__link\">证据要求\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-79\" class=\"editorjs-toc__link\">知识覆盖\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-81\" class=\"editorjs-toc__link\">错误的后果\u003C\u002Fa>\u003C\u002Fli>\u003C\u002Fol>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-84\" class=\"editorjs-toc__link\">诊断\u002F决策方法\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-96\" class=\"editorjs-toc__link\">证据\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-104\" class=\"editorjs-toc__link\">真实示例\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-119\" class=\"editorjs-toc__link\">常见误解和失败模式\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-125\" class=\"editorjs-toc__link\">边缘情况\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-136\" class=\"editorjs-toc__link\">局限性\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-143\" class=\"editorjs-toc__link\">什么会改变这个答案？\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-149\" class=\"editorjs-toc__link\">结论\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-157\" class=\"editorjs-toc__link\">主要来源\u003C\u002Fa>\u003C\u002Fli>\u003C\u002Fol>\u003C\u002Fnav>\n\u003Ch2 id=\"section-10\">这真正意味着什么\u003C\u002Fh2>\n\u003Cp>LLM 有两种根本不同的信息获取方式。\u003C\u002Fp>\n\u003Cp>第一种是模型知识。这是表示在模型学习参数中的信息。运行时不需要数据库查询、网络搜索或文档查找。\u003C\u002Fp>\n\u003Cp>第二种是运行时知识。这是模型运行期间提供的信息：搜索结果、数据库记录、文档、API、用户文件、工具输出或其他检索到的证据。\u003C\u002Fp>\n\u003Cp>RAG 连接这两个世界。但 RAG 本身并没有回答这种连接应在何时被激活的问题。这正是检索触发器的目的。\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>Question\n   ↓\nModel Knowledge\n   ↓\nIs internal knowledge sufficient?\n   ↓\nRetrieval Trigger\n   ↓\nExternal Retrieval, if required\n   ↓\nEvidence\n   ↓\nReasoning\n   ↓\nAnswer Validity Boundary\n   ↓\nAnswer\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>因此，检索触发器位于检索之前。答案有效性边界则位于之后。\u003C\u002Fp>\n\u003Cp>第一个问的是：我需要外部证据吗？\u003C\u002Fp>\n\u003Cp>第二个问的是：我现在是否有足够证据来支持这个答案？\u003C\u002Fp>\n\u003Cp>这些是相关的决策，但它们不是同一个决策。\u003C\u002Fp>\n\u003Ch2 id=\"section-20\">最简单的例子\u003C\u002Fh2>\n\u003Cp>考虑三个问题。\u003C\u002Fp>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">问题\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">内部知识\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">检索触发\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">法国的首都是什么？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">通常足够\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">没有强烈触发\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">英伟达当前的股价是多少？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">可能过时\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">触发检索\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">这篇新科学论文是否证明X导致Y？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">无法在不检查证据的情况下确立该主张\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">强烈检索触发\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Cp>第一个问题基于一个高度稳定的事实。\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>User\n↓\n&quot;What is the capital of France?&quot;\n\nModel knowledge\n↓\nParis\n\nFresh external evidence required?\n↓\nNo\n\nAnswer\n↓\nParis\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>在回答之前检索文档通常不会增加多少价值。\u003C\u002Fp>\n\u003Cp>现在考虑一个答案不断变化的问题。\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>User\n↓\n&quot;What is the current NVIDIA stock price?&quot;\n\nModel knowledge\n↓\nPotentially outdated\n\nCurrent information required?\n↓\nYes\n\nRETRIEVAL TRIGGER\n↓\nMarket data \u002F search \u002F API\n↓\nAnswer\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>模型可能对英伟达了解很多。这并不意味着它知道现在的价格。\u003C\u002Fp>\n\u003Cp>第三个例子更为重要。\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>User\n↓\n&quot;Does this new scientific paper prove that X causes Y?&quot;\n\nModel knowledge\n↓\nCan reason about causality,\nstatistics and scientific methodology.\n\nBut:\nthe actual evidence is not available internally.\n\nRETRIEVAL TRIGGER\n↓\nRetrieve the paper\n↓\nInspect methodology\n↓\nInspect results\n↓\nCompare claim with evidence\n↓\nAnswer Validity Boundary\n↓\nAnswer\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>模型的推理能力可能完全有用。缺失的组成部分是证据。\u003C\u002Fp>\n\u003Cp>这种区别是根本性的。\u003C\u002Fp>\n\u003Ch2 id=\"section-33\">例子失效之处\u003C\u002Fh2>\n\u003Cp>上面的例子使决策看起来是二元的：检索或不检索。\u003C\u002Fp>\n\u003Cp>现实系统更为复杂。一个问题可能包含多个主张，有些稳定，有些是当前的。检索到的文档可能不一致。检索器可能返回不相关的信息。相关信息可能存在但排名不够高。文档可能权威但过时。\u003C\u002Fp>\n\u003Cp>检索本身也可能将不正确的上下文引入原本合理的答案中。\u003C\u002Fp>\n\u003Cp>这就是为什么检索不应被视为真相的自动同义词。\u003C\u002Fp>\n\u003Cp>关于自适应检索的研究已日益摆脱这样一个假设：每个查询都应采用相同的检索策略。\u003C\u002Fp>\n\u003Cp>例如，\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2310.11511\" target=\"_blank\">Self-RAG\u003C\u002Fa> 明确探索按需检索，而不是为每个输入不加区分地检索固定数量的段落。作者讨论了不必要或不相关的检索如何降低答案质量。\u003C\u002Fp>\n\u003Cp>\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2403.14403\" target=\"_blank\">Adaptive-RAG\u003C\u002Fa> 同样根据问题复杂度在不检索、单步检索和更复杂的检索策略之间进行选择。\u003C\u002Fp>\n\u003Cp>因此，重要的问题不是：这个系统有 RAG 吗？\u003C\u002Fp>\n\u003Cp>而是：这个系统能否识别何时需要检索，以及何种检索是合适的？\u003C\u002Fp>\n\u003Ch2 id=\"section-43\">直接回答\u003C\u002Fh2>\n\u003Cp>当回答需要其内部模型知识无法安全提供的、具备所需时效性、具体性、来源或证据支持的信息时，AI 应触发检索。\u003C\u002Fp>\n\u003Cp>在实际系统中，检索触发器可能由若干条件产生：\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>Need for current information\n        OR\nNeed for exact source-specific information\n        OR\nNeed for evidence or provenance\n        OR\nNeed for private\u002Fuser-specific information\n        OR\nInsufficient knowledge coverage\n        OR\nConflicting evidence\n        OR\nHigh consequence of factual error\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>如果这些条件均未实质性出现，检索可能是不必要的。如果出现一个或多个条件，外部证据便成为答案生成过程的一部分。\u003C\u002Fp>\n\u003Ch2 id=\"section-48\">为何如此\u003C\u002Fh2>\n\u003Cp>语言模型的内部知识通常被称为参数化知识。它是在训练期间学习并编码到模型参数中的。\u003C\u002Fp>\n\u003Cp>\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2005.11401\" target=\"_blank\">Lewis 等人最初的 RAG 工作\u003C\u002Fa>将检索框定为这种参数化记忆与外部非参数化记忆的结合。外部记忆可以被搜索和更新，而无需重新训练整个语言模型。\u003C\u002Fp>\n\u003Cp>这种区分造成了一个不可避免的系统性问题。\u003C\u002Fp>\n\u003Cp>模型可以知道一些事情。但模型不能假设它所知道的一切都是最新的、完整的、足够具体的，并且有必要的证据支持。\u003C\u002Fp>\n\u003Cp>因此，模型可能生成一个语言上令人信服的答案，却仍在其内部知识已不足的边界之外运作。\u003C\u002Fp>\n\u003Cp>正是在这一点上，检索触发器变得有用。\u003C\u002Fp>\n\u003Ch2 id=\"section-55\">背景\u003C\u002Fh2>\n\u003Cp>传统 RAG 通常如下所示：\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>Question\n↓\nRetrieve documents\n↓\nAdd documents to context\n↓\nGenerate answer\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>这种架构假设先检索后生成。这对许多知识密集型应用效果良好，但也可能执行不必要的检索。\u003C\u002Fp>\n\u003Cp>更先进的方法引入了自适应步骤：\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>Question\n↓\nEvaluate information requirement\n↓\n        ┌───────────────┐\n        │               │\n   no retrieval      retrieval\n        │               │\n        ↓               ↓\n model knowledge    external evidence\n        │               │\n        └───────┬───────┘\n                ↓\n              answer\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2305.06983\" target=\"_blank\">FLARE\u003C\u002Fa> 更进一步，在生成过程中考虑检索。它利用即将生成的文本和低置信度 token 作为检索额外信息的信号。\u003C\u002Fp>\n\u003Cp>Self-RAG 同样引入了允许检索、生成和批判相互作用的机制，而不是将检索视为无条件的预处理步骤。\u003C\u002Fp>\n\u003Cp>Adaptive-RAG 从查询复杂度的角度处理相同的更广泛问题：不同的问题可能需要不同的检索策略。\u003C\u002Fp>\n\u003Cp>这些方法在技术上有所不同。但它们揭示了相同的架构洞见：检索应该是一个决策，而不仅仅是一个永久开关。\u003C\u002Fp>\n\u003Ch2 id=\"section-65\">假设\u003C\u002Fh2>\n\u003Cp>检索触发框架假设系统在需要检索时至少可以访问一个外部信息源。\u003C\u002Fp>\n\u003Cp>该信息源可以是网络搜索、文档存储、向量数据库、SQL 数据库、知识图谱、API、企业系统、用户上传的文档或工具输出。\u003C\u002Fp>\n\u003Cp>它还假设检索是有成本的。这种成本不一定是财务成本。\u003C\u002Fp>\n\u003Cp>检索会引入延迟、token 消耗、上下文使用、基础设施复杂性以及检索到误导性信息的可能性。\u003C\u002Fp>\n\u003Cp>因此，最优系统不是最大化检索，而是最大化适当的检索。\u003C\u002Fp>\n\u003Ch2 id=\"section-71\">变量\u003C\u002Fh2>\n\u003Cp>一个实用的检索触发器可以考虑五个主要变量。\u003C\u002Fp>\n\u003Ch3 id=\"section-73\">时效性\u003C\u002Fh3>\n\u003Cp>所需信息发生变化的可能性有多大？法国的首都波动性极低。股票价格波动性极高。\u003C\u002Fp>\n\u003Ch3 id=\"section-75\">特定性\u003C\u002Fh3>\n\u003Cp>问题是否需要来自特定来源、文档、组织、账户或数据集的信息？如果用户询问某份具体合同的内容，通用模型知识无关紧要。必须检索该合同。\u003C\u002Fp>\n\u003Ch3 id=\"section-77\">证据要求\u003C\u002Fh3>\n\u003Cp>答案是否需要来源出处？模型可能知道某个说法被普遍接受，但当任务需要验证时，仍然需要来源。\u003C\u002Fp>\n\u003Ch3 id=\"section-79\">知识覆盖\u003C\u002Fh3>\n\u003Cp>该主题是否可能在内部模型知识中得到充分体现？罕见、专有、高度本地化或新发布的信息会产生更强的检索压力。\u003C\u002Fp>\n\u003Ch3 id=\"section-81\">错误的后果\u003C\u002Fh3>\n\u003Cp>并非每个错误答案都有相同的影响。当事实准确性对决策产生实质性影响时，可接受的证据门槛可能更高。\u003C\u002Fp>\n\u003Cp>这些变量不必实现为字面上的数值分数。它们描述的是决策面。\u003C\u002Fp>\n\u003Ch2 id=\"section-84\">诊断\u002F决策方法\u003C\u002Fh2>\n\u003Cp>一个非常简单的检索触发器可以在没有机器学习的情况下实现。\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>def should_retrieve(\n    time_sensitive=False,\n    source_specific=False,\n    evidence_required=False,\n    private_context=False,\n    knowledge_uncertain=False,\n    conflicting_information=False\n):\n    return any([\n        time_sensitive,\n        source_specific,\n        evidence_required,\n        private_context,\n        knowledge_uncertain,\n        conflicting_information,\n    ])\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>对于稳定的事实性问题：\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>should_retrieve()\n# False\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>对于当前股票价格：\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>should_retrieve(\n    time_sensitive=True\n)\n# True\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>对于科学主张：\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>should_retrieve(\n    source_specific=True,\n    evidence_required=True\n)\n# True\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>生产系统可以使这一决策复杂得多。分类器可以预测检索需求。模型可以发出特殊的控制标记。路由器可以对查询复杂度进行分类。检索也可以在生成过程中被反复触发。\u003C\u002Fp>\n\u003Cp>实现方式可以改变。架构问题保持不变：\u003C\u002Fp>\n\u003Cblockquote class=\"border-l-4 border-gray-300 pl-4 italic\">模型当前可用的证据是否足以支持它即将生成的答案？\u003C\u002Fblockquote>\n\u003Ch2 id=\"section-96\">证据\u003C\u002Fh2>\n\u003Cp>这里提出的概念与多条检索研究路线一致。\u003C\u002Fp>\n\u003Cp>最初的\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2005.11401\" target=\"_blank\">RAG架构\u003C\u002Fa>展示了将参数化模型知识与外部非参数化知识相结合的有用性，尤其是在知识密集型任务中。\u003C\u002Fp>\n\u003Cp>\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2305.06983\" target=\"_blank\">FLARE\u003C\u002Fa>明确探索了生成过程中的主动检索，包括由低置信度的即将生成内容所触发的检索。\u003C\u002Fp>\n\u003Cp>\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2310.11511\" target=\"_blank\">Self-RAG\u003C\u002Fa>展示了一种架构，其中检索可以按需发生，随后对检索到的段落和生成的内容进行反思。\u003C\u002Fp>\n\u003Cp>\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2403.14403\" target=\"_blank\">Adaptive-RAG\u003C\u002Fa>根据问题复杂度在不同策略之间动态选择，包括不需要检索的情况。\u003C\u002Fp>\n\u003Cp>这里使用术语“检索触发器”作为对更广泛决策家族的系统级抽象。\u003C\u002Fp>\n\u003Cp>它并不声称这些论文使用了相同的术语。相反，它识别出共同的架构问题：是什么导致AI系统从内部知识转向外部证据？\u003C\u002Fp>\n\u003Ch2 id=\"section-104\">真实示例\u003C\u002Fh2>\n\u003Cp>考虑一个连接到公司文档的支持助手。\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>&quot;How do I reset my password?&quot;\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>如果该流程稳定且可靠地体现在助手当前的指令中，直接回答可能是合适的。\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>&quot;What permissions does my account currently have?&quot;\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>该信息是用户特定的且动态的。检索触发器触发。系统必须检查实际的账户或授权数据。\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>&quot;Why was my production deployment rejected yesterday?&quot;\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>模型可以理解部署系统并解释常见原因。但问题询问的是特定事件。需要日志、CI\u002FCD输出或事件记录。\u003C\u002Fp>\n\u003Cp>同样的逻辑适用于网络搜索。\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>&quot;What is RAG?&quot;\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>一般性解释可能不需要检索。\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>&quot;What did the authors of Self-RAG specifically conclude about unnecessary retrieval?&quot;\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>现在需要来源特定的证据。\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>&quot;What is the latest research on adaptive retrieval?&quot;\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>这也引入了时效性要求。底层主题没有改变。信息需求改变了。\u003C\u002Fp>\n\u003Ch2 id=\"section-119\">常见误解和失败模式\u003C\u002Fh2>\n\u003Cp>更多检索自动产生更好的答案。它不会。不相关的文档消耗上下文并可能分散生成注意力。\u003C\u002Fp>\n\u003Cp>高模型置信度意味着检索不必要。模型可以自信地产生错误答案。因此，自我报告的置信度不应被视为唯一的触发器。\u003C\u002Fp>\n\u003Cp>成功检索意味着答案已验证。检索仅提供候选证据。证据仍必须相关、足够权威且被正确解释。\u003C\u002Fp>\n\u003Cp>RAG自动解决过时知识。只有当检索语料库本身包含当前信息时，它才会这样做。检索过时文档不会产生当前答案。\u003C\u002Fp>\n\u003Cp>一个检索步骤总是足够的。复杂问题可能需要多个证据或迭代检索。\u003C\u002Fp>\n\u003Ch2 id=\"section-125\">边缘情况\u003C\u002Fh2>\n\u003Cp>有些问题同时包含稳定和不稳定的信息。\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>&quot;Who founded NVIDIA, and what is its market capitalization today?&quot;\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>第一部分可能可以依靠稳定的模型知识来回答。第二部分则需要当前信息。\u003C\u002Fp>\n\u003Cp>一个足够强大的系统不应必然将整个查询视为一次检索决策。它可以在需要的地方才触发检索。\u003C\u002Fp>\n\u003Cp>另一个边缘情况是来源之间的分歧。假设检索返回了三份提出互不兼容主张的文档。\u003C\u002Fp>\n\u003Cp>检索触发器已经成功：系统识别出需要外部证据。但任务尚未完成。\u003C\u002Fp>\n\u003Cp>系统现在遇到了证据评估问题。这正是答案有效性边界变得重要的地方。\u003C\u002Fp>\n\u003Cp>系统可能已经检索到信息，但仍然没有足够的证据来做出强有力的结论。\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>Retrieval Trigger\n≠\npermission to answer\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>触发器获取证据。有效性边界决定该证据是否充分。\u003C\u002Fp>\n\u003Ch2 id=\"section-136\">局限性\u003C\u002Fh2>\n\u003Cp>检索触发器是一个概念框架，而不是通用算法。\u003C\u002Fp>\n\u003Cp>不同的系统将需要不同的触发规则。客户支持机器人、科学研究助手、搜索引擎和自主软件代理并不具有相同的证据要求。\u003C\u002Fp>\n\u003Cp>触发阈值也可能产生自身的失败模式。阈值太低会导致过度检索。阈值太高会导致无支持的回答。\u003C\u002Fp>\n\u003Cp>检索基础设施本身也很重要。一个完美的触发器连接到质量差的来源集合，仍然会产生糟糕的证据。\u003C\u002Fp>\n\u003Cp>同样，一个出色的知识库如果在需要时触发器从不激活，也几乎没有价值。\u003C\u002Fp>\n\u003Cp>因此，检索触发器只解决了更大架构中的一部分问题。\u003C\u002Fp>\n\u003Ch2 id=\"section-143\">什么会改变这个答案？\u003C\u002Fh2>\n\u003Cp>未来的模型可能包含更好的机制来识别自身的知识局限。检索器可能变得更便宜、更快速。长上下文系统可能持续携带多得多的源材料。\u003C\u002Fp>\n\u003Cp>模型也可能日益将搜索、数据库、工具和结构化知识结合起来，而不向应用开发者暴露一个独立的 RAG 阶段。\u003C\u002Fp>\n\u003Cp>这些变化可能会改变触发器的实现方式。它们不一定消除底层的决策。\u003C\u002Fp>\n\u003Cp>只要模型已可获得的信息与必须从外部获取的信息之间存在差异，系统就仍然需要某种机制来决定何时跨越这一边界。\u003C\u002Fp>\n\u003Cp>实现方式可能会从视野中消失。架构问题依然存在。\u003C\u002Fp>\n\u003Ch2 id=\"section-149\">结论\u003C\u002Fh2>\n\u003Cp>RAG 开始得太晚，无法解释整个问题。\u003C\u002Fp>\n\u003Cp>在检索能够发生之前，AI 系统必须确定检索是否必要。这个决策就是检索触发器。\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>Stable known fact\n→ answer from model knowledge\n\nCurrent fact\n→ retrieve\n\nSource-specific or evidence-dependent claim\n→ retrieve and verify\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>但更广泛的含义更为重要。可靠的 AI 不仅需要获取知识。它需要一种方法来确定其当前知识何时不足。\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>Model Knowledge\n        ↓\nRetrieval Trigger\n        ↓\nRuntime Knowledge \u002F RAG\n        ↓\nEvidence\n        ↓\nReasoning\n        ↓\nAnswer Validity Boundary\n        ↓\nAnswer\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>检索触发器决定系统何时应寻求证据。答案有效性边界决定该证据是否充分。\u003C\u002Fp>\n\u003Cp>它们共同描述了比单独 RAG 更有用的东西：一个从 AI 看似知道的内容转向它实际能够支持的内容的决策过程。\u003C\u002Fp>\n\u003Ch2 id=\"section-157\">主要来源\u003C\u002Fh2>\n\u003Cp>Patrick Lewis 等，\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2005.11401\" target=\"_blank\">Retrieval-Augmented Generation for Knowledge-Intensive NLP Tasks\u003C\u002Fa>（2020）。描述参数模型记忆与外部非参数记忆相结合的基础性 RAG 工作。\u003C\u002Fp>\n\u003Cp>Zhengbao Jiang 等，\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2305.06983\" target=\"_blank\">Active Retrieval Augmented Generation\u003C\u002Fa>（2023）。引入 FLARE 和生成过程中的主动检索，包括基于低置信度预测内容的检索。\u003C\u002Fp>\n\u003Cp>Akari Asai 等，\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2310.11511\" target=\"_blank\">Self-RAG: Learning to Retrieve, Generate, and Critique through Self-Reflection\u003C\u002Fa>（2023）。探索按需自适应检索和自我反思，而不是无条件固定检索。\u003C\u002Fp>\n\u003Cp>Soyeong Jeong 等，\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2403.14403\" target=\"_blank\">Adaptive-RAG: Learning to Adapt Retrieval-Augmented Large Language Models through Question Complexity\u003C\u002Fa>（2024）。根据输入问题在无检索、单步检索和更复杂的检索策略之间动态选择。\u003C\u002Fp>",{"time":212,"blocks":213,"version":1044},1790575308398,[214,220,226,231,236,241,246,251,259,266,271,276,281,286,291,297,302,307,312,317,322,327,348,353,358,363,368,373,378,383,388,393,398,403,408,413,418,423,428,433,438,443,448,453,458,463,468,473,478,483,488,493,498,503,508,513,518,523,528,533,538,543,548,553,558,563,568,573,578,583,588,593,598,603,608,613,618,623,628,633,638,643,648,653,658,663,668,673,678,683,688,693,698,703,708,714,719,724,729,734,739,744,749,754,759,764,769,774,779,784,789,794,799,804,809,814,819,824,829,834,839,844,849,854,859,864,869,874,879,884,889,894,899,904,909,914,919,924,929,934,939,944,949,954,959,964,969,974,979,984,989,994,999,1004,1009,1014,1019,1024,1029,1034,1039],{"id":215,"data":216,"type":42,"tunes":219},"Wt7UfNeFlS",{"text":217,"level":218},"问题",2,{},{"id":221,"data":222,"type":224,"tunes":225},"T-ZCQblBzm",{"text":223},"AI 应该在什么时候停止依赖它已经知道的内容，并在回答之前检索外部信息？","paragraph",{},{"id":227,"data":228,"type":224,"tunes":230},"vBcd4061WS",{"text":229},"这个问题看起来很简单，但它处于现代 AI 系统最重要的设计决策之一的核心位置。",{},{"id":232,"data":233,"type":224,"tunes":235},"r9NZ-Fzw0e",{"text":234},"大型语言模型在其参数中包含了大量知识。检索增强生成会在运行时添加外部信息。但两种极端都不理想。",{},{"id":237,"data":238,"type":224,"tunes":240},"CpzlgJjAVL",{"text":239},"总是信任模型可能会产生过时或缺乏支持的答案。总是检索信息会增加延迟、成本、无关上下文，并为检索错误带来新的机会。",{},{"id":242,"data":243,"type":224,"tunes":245},"yeclJhYJ1a",{"text":244},"因此，真正的问题出现在 RAG 之前：究竟应该在什么时候进行检索？",{},{"id":247,"data":248,"type":224,"tunes":250},"FgLSWvpZMg",{"text":249},"本文使用“检索触发器”这一术语来指代该决策。这里并不是将“检索触发器”作为研究文献中的标准化术语提出。它是一个实用的系统概念，汇集了主动式、自适应式和自我反思式检索研究中已经可见的思想。",{},{"id":252,"data":253,"type":257,"tunes":258},"Muzvv-2uzU",{"text":254,"caption":255,"alignment":256},"检索触发器是一种条件，表示 AI 系统应停止仅依赖内部模型知识，并在生成或最终确定答案之前获取外部证据。","工作定义","left","quote",{},{"id":260,"data":261,"type":264,"tunes":265},"1BGt1waZ01",{"title":262,"maxLevel":263,"minLevel":218},"目录",3,"tableOfContents",{},{"id":267,"data":268,"type":42,"tunes":270},"BFKJ2htjYN",{"text":269,"level":218},"这真正意味着什么",{},{"id":272,"data":273,"type":224,"tunes":275},"yfBYqVwObv",{"text":274},"LLM 有两种根本不同的信息获取方式。",{},{"id":277,"data":278,"type":224,"tunes":280},"Y4JYebztDi",{"text":279},"第一种是模型知识。这是表示在模型学习参数中的信息。运行时不需要数据库查询、网络搜索或文档查找。",{},{"id":282,"data":283,"type":224,"tunes":285},"x2L37FSTBK",{"text":284},"第二种是运行时知识。这是模型运行期间提供的信息：搜索结果、数据库记录、文档、API、用户文件、工具输出或其他检索到的证据。",{},{"id":287,"data":288,"type":224,"tunes":290},"2szDUb7_-4",{"text":289},"RAG 连接这两个世界。但 RAG 本身并没有回答这种连接应在何时被激活的问题。这正是检索触发器的目的。",{},{"id":292,"data":293,"type":295,"tunes":296},"5_yjTthHV4",{"code":294},"Question\n   ↓\nModel Knowledge\n   ↓\nIs internal knowledge sufficient?\n   ↓\nRetrieval Trigger\n   ↓\nExternal Retrieval, if required\n   ↓\nEvidence\n   ↓\nReasoning\n   ↓\nAnswer Validity Boundary\n   ↓\nAnswer","code",{},{"id":298,"data":299,"type":224,"tunes":301},"rH2K36ambR",{"text":300},"因此，检索触发器位于检索之前。答案有效性边界则位于之后。",{},{"id":303,"data":304,"type":224,"tunes":306},"L0WlGs_dTF",{"text":305},"第一个问的是：我需要外部证据吗？",{},{"id":308,"data":309,"type":224,"tunes":311},"JyE4O9aDCW",{"text":310},"第二个问的是：我现在是否有足够证据来支持这个答案？",{},{"id":313,"data":314,"type":224,"tunes":316},"L9JP5xByy4",{"text":315},"这些是相关的决策，但它们不是同一个决策。",{},{"id":318,"data":319,"type":42,"tunes":321},"4hPbiDSHek",{"text":320,"level":218},"最简单的例子",{},{"id":323,"data":324,"type":224,"tunes":326},"cER32Me6gA",{"text":325},"考虑三个问题。",{},{"id":328,"data":329,"type":346,"tunes":347},"izi7nU9FE9",{"content":330,"stretched":43,"withHeadings":14},[331,334,338,342],[217,332,333],"内部知识","检索触发",[335,336,337],"法国的首都是什么？","通常足够","没有强烈触发",[339,340,341],"英伟达当前的股价是多少？","可能过时","触发检索",[343,344,345],"这篇新科学论文是否证明X导致Y？","无法在不检查证据的情况下确立该主张","强烈检索触发","table",{},{"id":349,"data":350,"type":224,"tunes":352},"cb-Kx0fKs4",{"text":351},"第一个问题基于一个高度稳定的事实。",{},{"id":354,"data":355,"type":295,"tunes":357},"MZJzwvZUH7",{"code":356},"User\n↓\n\"What is the capital of France?\"\n\nModel knowledge\n↓\nParis\n\nFresh external evidence required?\n↓\nNo\n\nAnswer\n↓\nParis",{},{"id":359,"data":360,"type":224,"tunes":362},"fRP7-aWTJB",{"text":361},"在回答之前检索文档通常不会增加多少价值。",{},{"id":364,"data":365,"type":224,"tunes":367},"O2TaSvLoxO",{"text":366},"现在考虑一个答案不断变化的问题。",{},{"id":369,"data":370,"type":295,"tunes":372},"cNv0Dp7Mk3",{"code":371},"User\n↓\n\"What is the current NVIDIA stock price?\"\n\nModel knowledge\n↓\nPotentially outdated\n\nCurrent information required?\n↓\nYes\n\nRETRIEVAL TRIGGER\n↓\nMarket data \u002F search \u002F API\n↓\nAnswer",{},{"id":374,"data":375,"type":224,"tunes":377},"Y3NDw8awnA",{"text":376},"模型可能对英伟达了解很多。这并不意味着它知道现在的价格。",{},{"id":379,"data":380,"type":224,"tunes":382},"FwjiaA6mdJ",{"text":381},"第三个例子更为重要。",{},{"id":384,"data":385,"type":295,"tunes":387},"G48ZGtX4XK",{"code":386},"User\n↓\n\"Does this new scientific paper prove that X causes Y?\"\n\nModel knowledge\n↓\nCan reason about causality,\nstatistics and scientific methodology.\n\nBut:\nthe actual evidence is not available internally.\n\nRETRIEVAL TRIGGER\n↓\nRetrieve the paper\n↓\nInspect methodology\n↓\nInspect results\n↓\nCompare claim with evidence\n↓\nAnswer Validity Boundary\n↓\nAnswer",{},{"id":389,"data":390,"type":224,"tunes":392},"nGu-KcQC6l",{"text":391},"模型的推理能力可能完全有用。缺失的组成部分是证据。",{},{"id":394,"data":395,"type":224,"tunes":397},"_bUxnOYvHG",{"text":396},"这种区别是根本性的。",{},{"id":399,"data":400,"type":42,"tunes":402},"etbE_esRx4",{"text":401,"level":218},"例子失效之处",{},{"id":404,"data":405,"type":224,"tunes":407},"0iSdy2Msw7",{"text":406},"上面的例子使决策看起来是二元的：检索或不检索。",{},{"id":409,"data":410,"type":224,"tunes":412},"8Go7nm2niJ",{"text":411},"现实系统更为复杂。一个问题可能包含多个主张，有些稳定，有些是当前的。检索到的文档可能不一致。检索器可能返回不相关的信息。相关信息可能存在但排名不够高。文档可能权威但过时。",{},{"id":414,"data":415,"type":224,"tunes":417},"8dVjRU5cXg",{"text":416},"检索本身也可能将不正确的上下文引入原本合理的答案中。",{},{"id":419,"data":420,"type":224,"tunes":422},"pLqSH5-OJR",{"text":421},"这就是为什么检索不应被视为真相的自动同义词。",{},{"id":424,"data":425,"type":224,"tunes":427},"ww4Od2cmTr",{"text":426},"关于自适应检索的研究已日益摆脱这样一个假设：每个查询都应采用相同的检索策略。",{},{"id":429,"data":430,"type":224,"tunes":432},"1yE2LUP7cF",{"text":431},"例如，\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2310.11511\" target=\"_blank\">Self-RAG\u003C\u002Fa> 明确探索按需检索，而不是为每个输入不加区分地检索固定数量的段落。作者讨论了不必要或不相关的检索如何降低答案质量。",{},{"id":434,"data":435,"type":224,"tunes":437},"915QBDW89m",{"text":436},"\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2403.14403\" target=\"_blank\">Adaptive-RAG\u003C\u002Fa> 同样根据问题复杂度在不检索、单步检索和更复杂的检索策略之间进行选择。",{},{"id":439,"data":440,"type":224,"tunes":442},"1FBgxY0QQp",{"text":441},"因此，重要的问题不是：这个系统有 RAG 吗？",{},{"id":444,"data":445,"type":224,"tunes":447},"4xdj86u8Qz",{"text":446},"而是：这个系统能否识别何时需要检索，以及何种检索是合适的？",{},{"id":449,"data":450,"type":42,"tunes":452},"bGPa0AsJI6",{"text":451,"level":218},"直接回答",{},{"id":454,"data":455,"type":224,"tunes":457},"fBKcyJ0IcX",{"text":456},"当回答需要其内部模型知识无法安全提供的、具备所需时效性、具体性、来源或证据支持的信息时，AI 应触发检索。",{},{"id":459,"data":460,"type":224,"tunes":462},"JbIPIJjkXK",{"text":461},"在实际系统中，检索触发器可能由若干条件产生：",{},{"id":464,"data":465,"type":295,"tunes":467},"_JbTSHlrtH",{"code":466},"Need for current information\n        OR\nNeed for exact source-specific information\n        OR\nNeed for evidence or provenance\n        OR\nNeed for private\u002Fuser-specific information\n        OR\nInsufficient knowledge coverage\n        OR\nConflicting evidence\n        OR\nHigh consequence of factual error",{},{"id":469,"data":470,"type":224,"tunes":472},"Aaem6fQ_tF",{"text":471},"如果这些条件均未实质性出现，检索可能是不必要的。如果出现一个或多个条件，外部证据便成为答案生成过程的一部分。",{},{"id":474,"data":475,"type":42,"tunes":477},"x2DDg7Ue1-",{"text":476,"level":218},"为何如此",{},{"id":479,"data":480,"type":224,"tunes":482},"1GTaWG9ViB",{"text":481},"语言模型的内部知识通常被称为参数化知识。它是在训练期间学习并编码到模型参数中的。",{},{"id":484,"data":485,"type":224,"tunes":487},"klwNY3lr1d",{"text":486},"\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2005.11401\" target=\"_blank\">Lewis 等人最初的 RAG 工作\u003C\u002Fa>将检索框定为这种参数化记忆与外部非参数化记忆的结合。外部记忆可以被搜索和更新，而无需重新训练整个语言模型。",{},{"id":489,"data":490,"type":224,"tunes":492},"Fyw2AVDbxR",{"text":491},"这种区分造成了一个不可避免的系统性问题。",{},{"id":494,"data":495,"type":224,"tunes":497},"4FvbthV3in",{"text":496},"模型可以知道一些事情。但模型不能假设它所知道的一切都是最新的、完整的、足够具体的，并且有必要的证据支持。",{},{"id":499,"data":500,"type":224,"tunes":502},"asxdihTbcB",{"text":501},"因此，模型可能生成一个语言上令人信服的答案，却仍在其内部知识已不足的边界之外运作。",{},{"id":504,"data":505,"type":224,"tunes":507},"jGgq116uAa",{"text":506},"正是在这一点上，检索触发器变得有用。",{},{"id":509,"data":510,"type":42,"tunes":512},"T6q_BUeDg3",{"text":511,"level":218},"背景",{},{"id":514,"data":515,"type":224,"tunes":517},"9H_bNlyoYs",{"text":516},"传统 RAG 通常如下所示：",{},{"id":519,"data":520,"type":295,"tunes":522},"YSZR1AInSj",{"code":521},"Question\n↓\nRetrieve documents\n↓\nAdd documents to context\n↓\nGenerate answer",{},{"id":524,"data":525,"type":224,"tunes":527},"HnzY2Q9xTs",{"text":526},"这种架构假设先检索后生成。这对许多知识密集型应用效果良好，但也可能执行不必要的检索。",{},{"id":529,"data":530,"type":224,"tunes":532},"Aho03YTGAU",{"text":531},"更先进的方法引入了自适应步骤：",{},{"id":534,"data":535,"type":295,"tunes":537},"uK0l0tYLg6",{"code":536},"Question\n↓\nEvaluate information requirement\n↓\n        ┌───────────────┐\n        │               │\n   no retrieval      retrieval\n        │               │\n        ↓               ↓\n model knowledge    external evidence\n        │               │\n        └───────┬───────┘\n                ↓\n              answer",{},{"id":539,"data":540,"type":224,"tunes":542},"Upb-15aN8T",{"text":541},"\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2305.06983\" target=\"_blank\">FLARE\u003C\u002Fa> 更进一步，在生成过程中考虑检索。它利用即将生成的文本和低置信度 token 作为检索额外信息的信号。",{},{"id":544,"data":545,"type":224,"tunes":547},"I5hs5j9IKc",{"text":546},"Self-RAG 同样引入了允许检索、生成和批判相互作用的机制，而不是将检索视为无条件的预处理步骤。",{},{"id":549,"data":550,"type":224,"tunes":552},"mw2jbuWA-g",{"text":551},"Adaptive-RAG 从查询复杂度的角度处理相同的更广泛问题：不同的问题可能需要不同的检索策略。",{},{"id":554,"data":555,"type":224,"tunes":557},"DUba0EfbWg",{"text":556},"这些方法在技术上有所不同。但它们揭示了相同的架构洞见：检索应该是一个决策，而不仅仅是一个永久开关。",{},{"id":559,"data":560,"type":42,"tunes":562},"wzX0jC8H8b",{"text":561,"level":218},"假设",{},{"id":564,"data":565,"type":224,"tunes":567},"4puAk8h-NF",{"text":566},"检索触发框架假设系统在需要检索时至少可以访问一个外部信息源。",{},{"id":569,"data":570,"type":224,"tunes":572},"Qsm42lc7aC",{"text":571},"该信息源可以是网络搜索、文档存储、向量数据库、SQL 数据库、知识图谱、API、企业系统、用户上传的文档或工具输出。",{},{"id":574,"data":575,"type":224,"tunes":577},"Rznt7yvqT2",{"text":576},"它还假设检索是有成本的。这种成本不一定是财务成本。",{},{"id":579,"data":580,"type":224,"tunes":582},"wQoEfZuFPe",{"text":581},"检索会引入延迟、token 消耗、上下文使用、基础设施复杂性以及检索到误导性信息的可能性。",{},{"id":584,"data":585,"type":224,"tunes":587},"WM1F9QkT2G",{"text":586},"因此，最优系统不是最大化检索，而是最大化适当的检索。",{},{"id":589,"data":590,"type":42,"tunes":592},"Z4gw9SX7jo",{"text":591,"level":218},"变量",{},{"id":594,"data":595,"type":224,"tunes":597},"Z_sKNO6vmp",{"text":596},"一个实用的检索触发器可以考虑五个主要变量。",{},{"id":599,"data":600,"type":42,"tunes":602},"Eti88tz1T6",{"text":601,"level":263},"时效性",{},{"id":604,"data":605,"type":224,"tunes":607},"3zKe198lls",{"text":606},"所需信息发生变化的可能性有多大？法国的首都波动性极低。股票价格波动性极高。",{},{"id":609,"data":610,"type":42,"tunes":612},"ryQRR7TzC7",{"text":611,"level":263},"特定性",{},{"id":614,"data":615,"type":224,"tunes":617},"bkXBBuCBb_",{"text":616},"问题是否需要来自特定来源、文档、组织、账户或数据集的信息？如果用户询问某份具体合同的内容，通用模型知识无关紧要。必须检索该合同。",{},{"id":619,"data":620,"type":42,"tunes":622},"LlT6c-tPU2",{"text":621,"level":263},"证据要求",{},{"id":624,"data":625,"type":224,"tunes":627},"1G-aWjGT1c",{"text":626},"答案是否需要来源出处？模型可能知道某个说法被普遍接受，但当任务需要验证时，仍然需要来源。",{},{"id":629,"data":630,"type":42,"tunes":632},"lnoOCm4KDw",{"text":631,"level":263},"知识覆盖",{},{"id":634,"data":635,"type":224,"tunes":637},"nKGrZO0Zw0",{"text":636},"该主题是否可能在内部模型知识中得到充分体现？罕见、专有、高度本地化或新发布的信息会产生更强的检索压力。",{},{"id":639,"data":640,"type":42,"tunes":642},"SYp_4G0qXz",{"text":641,"level":263},"错误的后果",{},{"id":644,"data":645,"type":224,"tunes":647},"YJeo8nKsl9",{"text":646},"并非每个错误答案都有相同的影响。当事实准确性对决策产生实质性影响时，可接受的证据门槛可能更高。",{},{"id":649,"data":650,"type":224,"tunes":652},"NTh27HJjo1",{"text":651},"这些变量不必实现为字面上的数值分数。它们描述的是决策面。",{},{"id":654,"data":655,"type":42,"tunes":657},"A25id0cm1s",{"text":656,"level":218},"诊断\u002F决策方法",{},{"id":659,"data":660,"type":224,"tunes":662},"az70f7cIIF",{"text":661},"一个非常简单的检索触发器可以在没有机器学习的情况下实现。",{},{"id":664,"data":665,"type":295,"tunes":667},"yUFgVx9VRM",{"code":666},"def should_retrieve(\n    time_sensitive=False,\n    source_specific=False,\n    evidence_required=False,\n    private_context=False,\n    knowledge_uncertain=False,\n    conflicting_information=False\n):\n    return any([\n        time_sensitive,\n        source_specific,\n        evidence_required,\n        private_context,\n        knowledge_uncertain,\n        conflicting_information,\n    ])",{},{"id":669,"data":670,"type":224,"tunes":672},"tBn6sOGnKB",{"text":671},"对于稳定的事实性问题：",{},{"id":674,"data":675,"type":295,"tunes":677},"BW2rsTbqqL",{"code":676},"should_retrieve()\n# False",{},{"id":679,"data":680,"type":224,"tunes":682},"iriE0iq97f",{"text":681},"对于当前股票价格：",{},{"id":684,"data":685,"type":295,"tunes":687},"d10aolm-TW",{"code":686},"should_retrieve(\n    time_sensitive=True\n)\n# True",{},{"id":689,"data":690,"type":224,"tunes":692},"nwL_vpUi-Y",{"text":691},"对于科学主张：",{},{"id":694,"data":695,"type":295,"tunes":697},"-DXgs4BBKH",{"code":696},"should_retrieve(\n    source_specific=True,\n    evidence_required=True\n)\n# True",{},{"id":699,"data":700,"type":224,"tunes":702},"Wj1sAbZ7l8",{"text":701},"生产系统可以使这一决策复杂得多。分类器可以预测检索需求。模型可以发出特殊的控制标记。路由器可以对查询复杂度进行分类。检索也可以在生成过程中被反复触发。",{},{"id":704,"data":705,"type":224,"tunes":707},"loLbe4TkAK",{"text":706},"实现方式可以改变。架构问题保持不变：",{},{"id":709,"data":710,"type":257,"tunes":713},"T2PJWaSYp9",{"text":711,"caption":712,"alignment":256},"模型当前可用的证据是否足以支持它即将生成的答案？","",{},{"id":715,"data":716,"type":42,"tunes":718},"9Alw1zzG4E",{"text":717,"level":218},"证据",{},{"id":720,"data":721,"type":224,"tunes":723},"OgqdUwG1J-",{"text":722},"这里提出的概念与多条检索研究路线一致。",{},{"id":725,"data":726,"type":224,"tunes":728},"QxIkEDnlBu",{"text":727},"最初的\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2005.11401\" target=\"_blank\">RAG架构\u003C\u002Fa>展示了将参数化模型知识与外部非参数化知识相结合的有用性，尤其是在知识密集型任务中。",{},{"id":730,"data":731,"type":224,"tunes":733},"nqdi_kDLE3",{"text":732},"\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2305.06983\" target=\"_blank\">FLARE\u003C\u002Fa>明确探索了生成过程中的主动检索，包括由低置信度的即将生成内容所触发的检索。",{},{"id":735,"data":736,"type":224,"tunes":738},"UThAFgyEe3",{"text":737},"\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2310.11511\" target=\"_blank\">Self-RAG\u003C\u002Fa>展示了一种架构，其中检索可以按需发生，随后对检索到的段落和生成的内容进行反思。",{},{"id":740,"data":741,"type":224,"tunes":743},"CT8n5F5KLR",{"text":742},"\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2403.14403\" target=\"_blank\">Adaptive-RAG\u003C\u002Fa>根据问题复杂度在不同策略之间动态选择，包括不需要检索的情况。",{},{"id":745,"data":746,"type":224,"tunes":748},"RDjo1UGf2s",{"text":747},"这里使用术语“检索触发器”作为对更广泛决策家族的系统级抽象。",{},{"id":750,"data":751,"type":224,"tunes":753},"nmWZ-exi8b",{"text":752},"它并不声称这些论文使用了相同的术语。相反，它识别出共同的架构问题：是什么导致AI系统从内部知识转向外部证据？",{},{"id":755,"data":756,"type":42,"tunes":758},"8a-H_OlfG1",{"text":757,"level":218},"真实示例",{},{"id":760,"data":761,"type":224,"tunes":763},"l0KONt5Buo",{"text":762},"考虑一个连接到公司文档的支持助手。",{},{"id":765,"data":766,"type":295,"tunes":768},"MOy11BOFq5",{"code":767},"\"How do I reset my password?\"",{},{"id":770,"data":771,"type":224,"tunes":773},"tsZcuMS1sT",{"text":772},"如果该流程稳定且可靠地体现在助手当前的指令中，直接回答可能是合适的。",{},{"id":775,"data":776,"type":295,"tunes":778},"l53NPB6WNV",{"code":777},"\"What permissions does my account currently have?\"",{},{"id":780,"data":781,"type":224,"tunes":783},"RKQXMbXA29",{"text":782},"该信息是用户特定的且动态的。检索触发器触发。系统必须检查实际的账户或授权数据。",{},{"id":785,"data":786,"type":295,"tunes":788},"b5tNqxBKir",{"code":787},"\"Why was my production deployment rejected yesterday?\"",{},{"id":790,"data":791,"type":224,"tunes":793},"xaP7a7lV3i",{"text":792},"模型可以理解部署系统并解释常见原因。但问题询问的是特定事件。需要日志、CI\u002FCD输出或事件记录。",{},{"id":795,"data":796,"type":224,"tunes":798},"SlBdofaCVq",{"text":797},"同样的逻辑适用于网络搜索。",{},{"id":800,"data":801,"type":295,"tunes":803},"VgFaQjUMnU",{"code":802},"\"What is RAG?\"",{},{"id":805,"data":806,"type":224,"tunes":808},"6BG7aSJQzt",{"text":807},"一般性解释可能不需要检索。",{},{"id":810,"data":811,"type":295,"tunes":813},"TRngQB41uY",{"code":812},"\"What did the authors of Self-RAG specifically conclude about unnecessary retrieval?\"",{},{"id":815,"data":816,"type":224,"tunes":818},"7UIuDjIRyG",{"text":817},"现在需要来源特定的证据。",{},{"id":820,"data":821,"type":295,"tunes":823},"CG2PbVS1yz",{"code":822},"\"What is the latest research on adaptive retrieval?\"",{},{"id":825,"data":826,"type":224,"tunes":828},"G1gyMnE_E8",{"text":827},"这也引入了时效性要求。底层主题没有改变。信息需求改变了。",{},{"id":830,"data":831,"type":42,"tunes":833},"NyJtHsPsSf",{"text":832,"level":218},"常见误解和失败模式",{},{"id":835,"data":836,"type":224,"tunes":838},"1otM6VenxR",{"text":837},"更多检索自动产生更好的答案。它不会。不相关的文档消耗上下文并可能分散生成注意力。",{},{"id":840,"data":841,"type":224,"tunes":843},"7BpMfX7lOZ",{"text":842},"高模型置信度意味着检索不必要。模型可以自信地产生错误答案。因此，自我报告的置信度不应被视为唯一的触发器。",{},{"id":845,"data":846,"type":224,"tunes":848},"THz75XkfrR",{"text":847},"成功检索意味着答案已验证。检索仅提供候选证据。证据仍必须相关、足够权威且被正确解释。",{},{"id":850,"data":851,"type":224,"tunes":853},"gOUGv2dAaq",{"text":852},"RAG自动解决过时知识。只有当检索语料库本身包含当前信息时，它才会这样做。检索过时文档不会产生当前答案。",{},{"id":855,"data":856,"type":224,"tunes":858},"Mz8i-je--k",{"text":857},"一个检索步骤总是足够的。复杂问题可能需要多个证据或迭代检索。",{},{"id":860,"data":861,"type":42,"tunes":863},"imAEotM35y",{"text":862,"level":218},"边缘情况",{},{"id":865,"data":866,"type":224,"tunes":868},"8xkcG8hc9c",{"text":867},"有些问题同时包含稳定和不稳定的信息。",{},{"id":870,"data":871,"type":295,"tunes":873},"IYDiRezoWn",{"code":872},"\"Who founded NVIDIA, and what is its market capitalization today?\"",{},{"id":875,"data":876,"type":224,"tunes":878},"2kxOM8vxYh",{"text":877},"第一部分可能可以依靠稳定的模型知识来回答。第二部分则需要当前信息。",{},{"id":880,"data":881,"type":224,"tunes":883},"6PyzlxURFS",{"text":882},"一个足够强大的系统不应必然将整个查询视为一次检索决策。它可以在需要的地方才触发检索。",{},{"id":885,"data":886,"type":224,"tunes":888},"lsbZ8aQAD6",{"text":887},"另一个边缘情况是来源之间的分歧。假设检索返回了三份提出互不兼容主张的文档。",{},{"id":890,"data":891,"type":224,"tunes":893},"-Y67JvJusX",{"text":892},"检索触发器已经成功：系统识别出需要外部证据。但任务尚未完成。",{},{"id":895,"data":896,"type":224,"tunes":898},"32VdDErqUM",{"text":897},"系统现在遇到了证据评估问题。这正是答案有效性边界变得重要的地方。",{},{"id":900,"data":901,"type":224,"tunes":903},"edCyD-PqlU",{"text":902},"系统可能已经检索到信息，但仍然没有足够的证据来做出强有力的结论。",{},{"id":905,"data":906,"type":295,"tunes":908},"rmjW0MBcFo",{"code":907},"Retrieval Trigger\n≠\npermission to answer",{},{"id":910,"data":911,"type":224,"tunes":913},"gZBq0voX0-",{"text":912},"触发器获取证据。有效性边界决定该证据是否充分。",{},{"id":915,"data":916,"type":42,"tunes":918},"DmO9cFY93l",{"text":917,"level":218},"局限性",{},{"id":920,"data":921,"type":224,"tunes":923},"gV4YT_2O1X",{"text":922},"检索触发器是一个概念框架，而不是通用算法。",{},{"id":925,"data":926,"type":224,"tunes":928},"7c6OA2X4-H",{"text":927},"不同的系统将需要不同的触发规则。客户支持机器人、科学研究助手、搜索引擎和自主软件代理并不具有相同的证据要求。",{},{"id":930,"data":931,"type":224,"tunes":933},"Xn8K4ArjdA",{"text":932},"触发阈值也可能产生自身的失败模式。阈值太低会导致过度检索。阈值太高会导致无支持的回答。",{},{"id":935,"data":936,"type":224,"tunes":938},"y0gYRFZx6m",{"text":937},"检索基础设施本身也很重要。一个完美的触发器连接到质量差的来源集合，仍然会产生糟糕的证据。",{},{"id":940,"data":941,"type":224,"tunes":943},"Kcvx1v4Z1x",{"text":942},"同样，一个出色的知识库如果在需要时触发器从不激活，也几乎没有价值。",{},{"id":945,"data":946,"type":224,"tunes":948},"wcRpKvBhkb",{"text":947},"因此，检索触发器只解决了更大架构中的一部分问题。",{},{"id":950,"data":951,"type":42,"tunes":953},"YN1_g7vs7V",{"text":952,"level":218},"什么会改变这个答案？",{},{"id":955,"data":956,"type":224,"tunes":958},"4PYX_PYK_Z",{"text":957},"未来的模型可能包含更好的机制来识别自身的知识局限。检索器可能变得更便宜、更快速。长上下文系统可能持续携带多得多的源材料。",{},{"id":960,"data":961,"type":224,"tunes":963},"vyqJ8Kp8Ye",{"text":962},"模型也可能日益将搜索、数据库、工具和结构化知识结合起来，而不向应用开发者暴露一个独立的 RAG 阶段。",{},{"id":965,"data":966,"type":224,"tunes":968},"_KN0MPs6as",{"text":967},"这些变化可能会改变触发器的实现方式。它们不一定消除底层的决策。",{},{"id":970,"data":971,"type":224,"tunes":973},"Zrlr4a0utJ",{"text":972},"只要模型已可获得的信息与必须从外部获取的信息之间存在差异，系统就仍然需要某种机制来决定何时跨越这一边界。",{},{"id":975,"data":976,"type":224,"tunes":978},"4hl7r5cF1M",{"text":977},"实现方式可能会从视野中消失。架构问题依然存在。",{},{"id":980,"data":981,"type":42,"tunes":983},"o_g5g_6eSj",{"text":982,"level":218},"结论",{},{"id":985,"data":986,"type":224,"tunes":988},"L6MWm8xdAg",{"text":987},"RAG 开始得太晚，无法解释整个问题。",{},{"id":990,"data":991,"type":224,"tunes":993},"uymgoYlFM5",{"text":992},"在检索能够发生之前，AI 系统必须确定检索是否必要。这个决策就是检索触发器。",{},{"id":995,"data":996,"type":295,"tunes":998},"_KIblg0ae_",{"code":997},"Stable known fact\n→ answer from model knowledge\n\nCurrent fact\n→ retrieve\n\nSource-specific or evidence-dependent claim\n→ retrieve and verify",{},{"id":1000,"data":1001,"type":224,"tunes":1003},"unCgfbeYI5",{"text":1002},"但更广泛的含义更为重要。可靠的 AI 不仅需要获取知识。它需要一种方法来确定其当前知识何时不足。",{},{"id":1005,"data":1006,"type":295,"tunes":1008},"ZAosU4trn9",{"code":1007},"Model Knowledge\n        ↓\nRetrieval Trigger\n        ↓\nRuntime Knowledge \u002F RAG\n        ↓\nEvidence\n        ↓\nReasoning\n        ↓\nAnswer Validity Boundary\n        ↓\nAnswer",{},{"id":1010,"data":1011,"type":224,"tunes":1013},"f0ZIysaJy1",{"text":1012},"检索触发器决定系统何时应寻求证据。答案有效性边界决定该证据是否充分。",{},{"id":1015,"data":1016,"type":224,"tunes":1018},"iCg9ojv75m",{"text":1017},"它们共同描述了比单独 RAG 更有用的东西：一个从 AI 看似知道的内容转向它实际能够支持的内容的决策过程。",{},{"id":1020,"data":1021,"type":42,"tunes":1023},"cDBiNnZJv-",{"text":1022,"level":218},"主要来源",{},{"id":1025,"data":1026,"type":224,"tunes":1028},"8gumvODB16",{"text":1027},"Patrick Lewis 等，\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2005.11401\" target=\"_blank\">Retrieval-Augmented Generation for Knowledge-Intensive NLP Tasks\u003C\u002Fa>（2020）。描述参数模型记忆与外部非参数记忆相结合的基础性 RAG 工作。",{},{"id":1030,"data":1031,"type":224,"tunes":1033},"Chz6I7zmlv",{"text":1032},"Zhengbao Jiang 等，\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2305.06983\" target=\"_blank\">Active Retrieval Augmented Generation\u003C\u002Fa>（2023）。引入 FLARE 和生成过程中的主动检索，包括基于低置信度预测内容的检索。",{},{"id":1035,"data":1036,"type":224,"tunes":1038},"MRdjivpsoW",{"text":1037},"Akari Asai 等，\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2310.11511\" target=\"_blank\">Self-RAG: Learning to Retrieve, Generate, and Critique through Self-Reflection\u003C\u002Fa>（2023）。探索按需自适应检索和自我反思，而不是无条件固定检索。",{},{"id":1040,"data":1041,"type":224,"tunes":1043},"lck29euXJP",{"text":1042},"Soyeong Jeong 等，\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2403.14403\" target=\"_blank\">Adaptive-RAG: Learning to Adapt Retrieval-Augmented Large Language Models through Question Complexity\u003C\u002Fa>（2024）。根据输入问题在无检索、单步检索和更复杂的检索策略之间动态选择。",{},"2.31","AI 模型并非每个问题都需要检索。重要的问题在于知道何时其内部知识已不再足够。检索触发器是一个实用的决策边界，它决定 AI 系统何时应停止仅依赖模型知识，并在回答前获取外部证据。","\u002Fuploads\u002F2026\u002F09\u002Fwhen-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger-1790574991244-f4rpyg.webp","when-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger-1790574991244-f4rpyg","PUBLISHED","2026-09-28T01:49:00.000Z","2026-09-28T05:49:59.593Z","2026-09-28T06:02:54.212Z",{"en":1053,"de":1054,"sr":1055,"es":1056,"fr":1057,"it":1058,"ru":1059,"zh":1060},"\u002Fblog\u002Fwhen-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger","\u002Fde\u002Fblog\u002Fwhen-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger","\u002Fsr\u002Fblog\u002Fwhen-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger","\u002Fes\u002Fblog\u002Fwhen-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger","\u002Ffr\u002Fblog\u002Fwhen-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger","\u002Fit\u002Fblog\u002Fwhen-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger","\u002Fru\u002Fblog\u002Fwhen-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger","\u002Fzh\u002Fblog\u002Fwhen-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger",[1062,1065,1069,1073,1077,1081],{"id":101,"name":1063,"slug":1064},"Overview","overview-digital-platform",{"id":1066,"name":1067,"slug":1068},57,"数据边界","data-boundaries",{"id":1070,"name":1071,"slug":1072},51,"反模式","anti-patterns",{"id":1074,"name":1075,"slug":1076},58,"评估与质量门槛","evaluation",{"id":1078,"name":1079,"slug":1080},56,"用例组合","use-case-portfolio",{"id":1082,"name":1083,"slug":1084},60,"成本与延迟控制","cost-and-latency",{"id":1086,"login":1087,"email":1088,"displayName":1089},"20","rooth8233","aleksandar@stajic.de","Aleksandar Stajić",[1091,1739],{"lang":1092,"title":1093,"content":1094,"contentJson":1095,"excerpt":1738},"en","When Should an AI Stop Trusting Its Own Knowledge? — The Retrieval Trigger","{\"time\":1790574879391,\"blocks\":[{\"id\":\"Wt7UfNeFlS\",\"data\":{\"text\":\"Question\",\"level\":2},\"type\":\"header\",\"tunes\":{}},{\"id\":\"T-ZCQblBzm\",\"data\":{\"text\":\"When should an AI stop relying on what it already knows and retrieve external information before answering?\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"vBcd4061WS\",\"data\":{\"text\":\"This question appears simple, but it sits at the center of one of the most important design decisions in modern AI systems.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"r9NZ-Fzw0e\",\"data\":{\"text\":\"Large language models contain substantial knowledge in their parameters. Retrieval-Augmented Generation adds external information at runtime. But neither extreme is ideal.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"CpzlgJjAVL\",\"data\":{\"text\":\"Always trusting the model can produce outdated or unsupported answers. Always retrieving information adds latency, cost, irrelevant context and new opportunities for retrieval errors.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"yeclJhYJ1a\",\"data\":{\"text\":\"The real problem therefore comes before RAG: When should retrieval happen at all?\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"FgLSWvpZMg\",\"data\":{\"text\":\"This article uses the term Retrieval Trigger for that decision. Retrieval Trigger is not presented here as a standardized term from the research literature. It is a practical systems concept that brings together ideas already visible in research on active, adaptive and self-reflective retrieval.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"Muzvv-2uzU\",\"data\":{\"text\":\"A Retrieval Trigger is a condition indicating that an AI system should stop relying solely on internal model knowledge and obtain external evidence before producing or finalizing an answer.\",\"caption\":\"Working definition\",\"alignment\":\"left\"},\"type\":\"quote\",\"tunes\":{}},{\"id\":\"1BGt1waZ01\",\"data\":{\"title\":\"Contents\",\"maxLevel\":3,\"minLevel\":2},\"type\":\"tableOfContents\",\"tunes\":{}},{\"id\":\"BFKJ2htjYN\",\"data\":{\"text\":\"What This Really Means\",\"level\":2},\"type\":\"header\",\"tunes\":{}},{\"id\":\"yfBYqVwObv\",\"data\":{\"text\":\"An LLM has two fundamentally different ways of obtaining information.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"Y4JYebztDi\",\"data\":{\"text\":\"The first is model knowledge. This is information represented in the model's learned parameters. No database query, web search or document lookup is required at runtime.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"x2L37FSTBK\",\"data\":{\"text\":\"The second is runtime knowledge. This is information provided while the model is operating: search results, database records, documents, APIs, user files, tool outputs or other retrieved evidence.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"2szDUb7_-4\",\"data\":{\"text\":\"RAG connects these two worlds. But RAG itself does not answer the question of when that connection should be activated. That is the purpose of the Retrieval Trigger.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"5_yjTthHV4\",\"data\":{\"code\":\"Question\\n   ↓\\nModel Knowledge\\n   ↓\\nIs internal knowledge sufficient?\\n   ↓\\nRetrieval Trigger\\n   ↓\\nExternal Retrieval, if required\\n   ↓\\nEvidence\\n   ↓\\nReasoning\\n   ↓\\nAnswer Validity Boundary\\n   ↓\\nAnswer\"},\"type\":\"code\",\"tunes\":{}},{\"id\":\"rH2K36ambR\",\"data\":{\"text\":\"The Retrieval Trigger therefore sits before retrieval. The Answer Validity Boundary sits later.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"L0WlGs_dTF\",\"data\":{\"text\":\"The first asks: Do I need external evidence?\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"JyE4O9aDCW\",\"data\":{\"text\":\"The second asks: Do I now have enough evidence to support this answer?\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"L9JP5xByy4\",\"data\":{\"text\":\"These are related decisions, but they are not the same decision.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"4hPbiDSHek\",\"data\":{\"text\":\"Simplest Example\",\"level\":2},\"type\":\"header\",\"tunes\":{}},{\"id\":\"cER32Me6gA\",\"data\":{\"text\":\"Consider three questions.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"izi7nU9FE9\",\"data\":{\"content\":[[\"Question\",\"Internal knowledge\",\"Retrieval Trigger\"],[\"What is the capital of France?\",\"Usually sufficient\",\"No strong trigger\"],[\"What is the current NVIDIA stock price?\",\"Potentially outdated\",\"Trigger retrieval\"],[\"Does this new scientific paper prove that X causes Y?\",\"Cannot establish the claim without examining the evidence\",\"Strong retrieval trigger\"]],\"stretched\":false,\"withHeadings\":true},\"type\":\"table\",\"tunes\":{}},{\"id\":\"cb-Kx0fKs4\",\"data\":{\"text\":\"The first question is based on a highly stable fact.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"MZJzwvZUH7\",\"data\":{\"code\":\"User\\n↓\\n\\\"What is the capital of France?\\\"\\n\\nModel knowledge\\n↓\\nParis\\n\\nFresh external evidence required?\\n↓\\nNo\\n\\nAnswer\\n↓\\nParis\"},\"type\":\"code\",\"tunes\":{}},{\"id\":\"fRP7-aWTJB\",\"data\":{\"text\":\"Retrieving documents before answering would usually add little value.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"O2TaSvLoxO\",\"data\":{\"text\":\"Now consider a question whose answer changes continuously.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"cNv0Dp7Mk3\",\"data\":{\"code\":\"User\\n↓\\n\\\"What is the current NVIDIA stock price?\\\"\\n\\nModel knowledge\\n↓\\nPotentially outdated\\n\\nCurrent information required?\\n↓\\nYes\\n\\nRETRIEVAL TRIGGER\\n↓\\nMarket data \u002F search \u002F API\\n↓\\nAnswer\"},\"type\":\"code\",\"tunes\":{}},{\"id\":\"Y3NDw8awnA\",\"data\":{\"text\":\"The model may know a great deal about NVIDIA. That does not mean it knows the price now.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"FwjiaA6mdJ\",\"data\":{\"text\":\"The third example is even more important.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"G48ZGtX4XK\",\"data\":{\"code\":\"User\\n↓\\n\\\"Does this new scientific paper prove that X causes Y?\\\"\\n\\nModel knowledge\\n↓\\nCan reason about causality,\\nstatistics and scientific methodology.\\n\\nBut:\\nthe actual evidence is not available internally.\\n\\nRETRIEVAL TRIGGER\\n↓\\nRetrieve the paper\\n↓\\nInspect methodology\\n↓\\nInspect results\\n↓\\nCompare claim with evidence\\n↓\\nAnswer Validity Boundary\\n↓\\nAnswer\"},\"type\":\"code\",\"tunes\":{}},{\"id\":\"nGu-KcQC6l\",\"data\":{\"text\":\"The model's reasoning capability may be perfectly useful. The missing component is evidence.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"_bUxnOYvHG\",\"data\":{\"text\":\"That distinction is fundamental.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"etbE_esRx4\",\"data\":{\"text\":\"Where the Example Stops Working\",\"level\":2},\"type\":\"header\",\"tunes\":{}},{\"id\":\"0iSdy2Msw7\",\"data\":{\"text\":\"The examples above make the decision appear binary: retrieve or do not retrieve.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"8Go7nm2niJ\",\"data\":{\"text\":\"Real systems are more complicated. A question may contain several claims, some stable and some current. Retrieved documents may disagree. A retriever may return irrelevant information. The relevant information may exist but fail to rank highly enough. A document may be authoritative but outdated.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"8dVjRU5cXg\",\"data\":{\"text\":\"Retrieval itself can also introduce incorrect context into an otherwise reasonable answer.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"pLqSH5-OJR\",\"data\":{\"text\":\"This is why retrieval should not be treated as an automatic synonym for truth.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"ww4Od2cmTr\",\"data\":{\"text\":\"Research on adaptive retrieval has increasingly moved away from the assumption that every query should receive the same retrieval strategy.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"1yE2LUP7cF\",\"data\":{\"text\":\"\u003Ca href=\\\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2310.11511\\\" target=\\\"_blank\\\">Self-RAG\u003C\u002Fa>, for example, explicitly explores retrieval on demand rather than indiscriminately retrieving a fixed number of passages for every input. The authors discuss how unnecessary or irrelevant retrieval can reduce answer quality.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"915QBDW89m\",\"data\":{\"text\":\"\u003Ca href=\\\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2403.14403\\\" target=\\\"_blank\\\">Adaptive-RAG\u003C\u002Fa> similarly selects between no retrieval, single-step retrieval and more complex retrieval strategies according to question complexity.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"1FBgxY0QQp\",\"data\":{\"text\":\"So the important question is not: Does this system have RAG?\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"4xdj86u8Qz\",\"data\":{\"text\":\"It is: Can this system recognize when retrieval is necessary and what kind of retrieval is appropriate?\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"bGPa0AsJI6\",\"data\":{\"text\":\"Direct Answer\",\"level\":2},\"type\":\"header\",\"tunes\":{}},{\"id\":\"fBKcyJ0IcX\",\"data\":{\"text\":\"An AI should trigger retrieval when answering requires information that its internal model knowledge cannot safely provide with the required freshness, specificity, provenance or evidential support.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"JbIPIJjkXK\",\"data\":{\"text\":\"In practical systems, a Retrieval Trigger can emerge from several conditions:\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"_JbTSHlrtH\",\"data\":{\"code\":\"Need for current information\\n        OR\\nNeed for exact source-specific information\\n        OR\\nNeed for evidence or provenance\\n        OR\\nNeed for private\u002Fuser-specific information\\n        OR\\nInsufficient knowledge coverage\\n        OR\\nConflicting evidence\\n        OR\\nHigh consequence of factual error\"},\"type\":\"code\",\"tunes\":{}},{\"id\":\"Aaem6fQ_tF\",\"data\":{\"text\":\"If none of these conditions is materially present, retrieval may be unnecessary. If one or more are present, external evidence becomes part of the answer-generation process.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"x2DDg7Ue1-\",\"data\":{\"text\":\"Why This Is So\",\"level\":2},\"type\":\"header\",\"tunes\":{}},{\"id\":\"1GTaWG9ViB\",\"data\":{\"text\":\"A language model's internal knowledge is often described as parametric knowledge. It was learned during training and encoded into the model's parameters.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"klwNY3lr1d\",\"data\":{\"text\":\"\u003Ca href=\\\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2005.11401\\\" target=\\\"_blank\\\">Lewis et al.'s original RAG work\u003C\u002Fa> framed retrieval as a combination of this parametric memory with external, non-parametric memory. The external memory can be searched and updated without retraining the entire language model.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"Fyw2AVDbxR\",\"data\":{\"text\":\"This distinction creates an unavoidable systems problem.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"4FvbthV3in\",\"data\":{\"text\":\"The model can know things. But the model cannot assume that everything it knows is current, complete, specific enough and supported by the required evidence.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"asxdihTbcB\",\"data\":{\"text\":\"A model can therefore produce a linguistically convincing answer while still operating beyond the point where its internal knowledge is sufficient.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"jGgq116uAa\",\"data\":{\"text\":\"That point is where a Retrieval Trigger becomes useful.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"T6q_BUeDg3\",\"data\":{\"text\":\"Context\",\"level\":2},\"type\":\"header\",\"tunes\":{}},{\"id\":\"9H_bNlyoYs\",\"data\":{\"text\":\"Traditional RAG often looks like this:\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"YSZR1AInSj\",\"data\":{\"code\":\"Question\\n↓\\nRetrieve documents\\n↓\\nAdd documents to context\\n↓\\nGenerate answer\"},\"type\":\"code\",\"tunes\":{}},{\"id\":\"HnzY2Q9xTs\",\"data\":{\"text\":\"This architecture assumes retrieval before generation. That works well for many knowledge-intensive applications, but it can also perform unnecessary retrieval.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"Aho03YTGAU\",\"data\":{\"text\":\"More advanced approaches introduce an adaptive step:\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"uK0l0tYLg6\",\"data\":{\"code\":\"Question\\n↓\\nEvaluate information requirement\\n↓\\n        ┌───────────────┐\\n        │               │\\n   no retrieval      retrieval\\n        │               │\\n        ↓               ↓\\n model knowledge    external evidence\\n        │               │\\n        └───────┬───────┘\\n                ↓\\n              answer\"},\"type\":\"code\",\"tunes\":{}},{\"id\":\"Upb-15aN8T\",\"data\":{\"text\":\"\u003Ca href=\\\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2305.06983\\\" target=\\\"_blank\\\">FLARE\u003C\u002Fa> goes further by considering retrieval during generation itself. It uses upcoming generation and low-confidence tokens as signals for retrieving additional information.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"I5hs5j9IKc\",\"data\":{\"text\":\"Self-RAG similarly introduces mechanisms allowing retrieval, generation and critique to interact instead of treating retrieval as an unconditional preprocessing step.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"mw2jbuWA-g\",\"data\":{\"text\":\"Adaptive-RAG approaches the same broader problem from query complexity: different questions may require different retrieval strategies.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"DUba0EfbWg\",\"data\":{\"text\":\"These approaches differ technically. But they expose the same architectural insight: Retrieval should be a decision, not merely a permanent switch.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"wzX0jC8H8b\",\"data\":{\"text\":\"Assumptions\",\"level\":2},\"type\":\"header\",\"tunes\":{}},{\"id\":\"4puAk8h-NF\",\"data\":{\"text\":\"The Retrieval Trigger framework assumes that a system has access to at least one external information source when retrieval is required.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"Qsm42lc7aC\",\"data\":{\"text\":\"That source could be web search, a document store, vector database, SQL database, knowledge graph, API, enterprise system, user-uploaded document or tool output.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"Rznt7yvqT2\",\"data\":{\"text\":\"It also assumes that retrieval has a cost. That cost does not have to be financial.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"wQoEfZuFPe\",\"data\":{\"text\":\"Retrieval introduces latency, token consumption, context usage, infrastructure complexity and the possibility of retrieving misleading information.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"WM1F9QkT2G\",\"data\":{\"text\":\"The optimal system therefore does not maximize retrieval. It maximizes appropriate retrieval.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"Z4gw9SX7jo\",\"data\":{\"text\":\"Variables\",\"level\":2},\"type\":\"header\",\"tunes\":{}},{\"id\":\"Z_sKNO6vmp\",\"data\":{\"text\":\"A practical Retrieval Trigger can consider five primary variables.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"Eti88tz1T6\",\"data\":{\"text\":\"Freshness\",\"level\":3},\"type\":\"header\",\"tunes\":{}},{\"id\":\"3zKe198lls\",\"data\":{\"text\":\"How likely is the required information to have changed? The capital of France has very low volatility. A stock price has extremely high volatility.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"ryQRR7TzC7\",\"data\":{\"text\":\"Specificity\",\"level\":3},\"type\":\"header\",\"tunes\":{}},{\"id\":\"bkXBBuCBb_\",\"data\":{\"text\":\"Does the question require information from a particular source, document, organization, account or dataset? If the user asks what a specific contract says, general model knowledge is irrelevant. The contract must be retrieved.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"LlT6c-tPU2\",\"data\":{\"text\":\"Evidence Requirement\",\"level\":3},\"type\":\"header\",\"tunes\":{}},{\"id\":\"1G-aWjGT1c\",\"data\":{\"text\":\"Does the answer need provenance? A model may know that a claim is generally accepted but still need a source when the task requires verification.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"lnoOCm4KDw\",\"data\":{\"text\":\"Knowledge Coverage\",\"level\":3},\"type\":\"header\",\"tunes\":{}},{\"id\":\"nKGrZO0Zw0\",\"data\":{\"text\":\"Is the subject likely to be represented adequately in internal model knowledge? Rare, proprietary, highly local or newly published information creates stronger retrieval pressure.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"SYp_4G0qXz\",\"data\":{\"text\":\"Consequence of Error\",\"level\":3},\"type\":\"header\",\"tunes\":{}},{\"id\":\"YJeo8nKsl9\",\"data\":{\"text\":\"Not every incorrect answer has the same impact. Where factual accuracy materially affects a decision, the acceptable evidence threshold may be higher.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"NTh27HJjo1\",\"data\":{\"text\":\"These variables do not have to be implemented as literal numeric scores. They describe the decision surface.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"A25id0cm1s\",\"data\":{\"text\":\"Diagnostic \u002F Decision Method\",\"level\":2},\"type\":\"header\",\"tunes\":{}},{\"id\":\"az70f7cIIF\",\"data\":{\"text\":\"A very simple Retrieval Trigger can be implemented without machine learning.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"yUFgVx9VRM\",\"data\":{\"code\":\"def should_retrieve(\\n    time_sensitive=False,\\n    source_specific=False,\\n    evidence_required=False,\\n    private_context=False,\\n    knowledge_uncertain=False,\\n    conflicting_information=False\\n):\\n    return any([\\n        time_sensitive,\\n        source_specific,\\n        evidence_required,\\n        private_context,\\n        knowledge_uncertain,\\n        conflicting_information,\\n    ])\"},\"type\":\"code\",\"tunes\":{}},{\"id\":\"tBn6sOGnKB\",\"data\":{\"text\":\"For a stable factual question:\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"BW2rsTbqqL\",\"data\":{\"code\":\"should_retrieve()\\n# False\"},\"type\":\"code\",\"tunes\":{}},{\"id\":\"iriE0iq97f\",\"data\":{\"text\":\"For a current stock price:\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"d10aolm-TW\",\"data\":{\"code\":\"should_retrieve(\\n    time_sensitive=True\\n)\\n# True\"},\"type\":\"code\",\"tunes\":{}},{\"id\":\"nwL_vpUi-Y\",\"data\":{\"text\":\"For a scientific claim:\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"-DXgs4BBKH\",\"data\":{\"code\":\"should_retrieve(\\n    source_specific=True,\\n    evidence_required=True\\n)\\n# True\"},\"type\":\"code\",\"tunes\":{}},{\"id\":\"Wj1sAbZ7l8\",\"data\":{\"text\":\"Production systems can make this decision far more sophisticated. A classifier could predict retrieval requirements. A model could emit special control tokens. A router could classify query complexity. Retrieval could also be triggered repeatedly during generation.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"loLbe4TkAK\",\"data\":{\"text\":\"The implementation can change. The architectural question remains the same:\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"T2PJWaSYp9\",\"data\":{\"text\":\"Is the evidence currently available to the model sufficient for the answer it is about to produce?\",\"caption\":\"\",\"alignment\":\"left\"},\"type\":\"quote\",\"tunes\":{}},{\"id\":\"9Alw1zzG4E\",\"data\":{\"text\":\"Evidence\",\"level\":2},\"type\":\"header\",\"tunes\":{}},{\"id\":\"OgqdUwG1J-\",\"data\":{\"text\":\"The concept proposed here is consistent with several lines of retrieval research.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"QxIkEDnlBu\",\"data\":{\"text\":\"The original \u003Ca href=\\\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2005.11401\\\" target=\\\"_blank\\\">RAG architecture\u003C\u002Fa> demonstrated the usefulness of combining parametric model knowledge with external non-parametric knowledge, particularly for knowledge-intensive tasks.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"nqdi_kDLE3\",\"data\":{\"text\":\"\u003Ca href=\\\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2305.06983\\\" target=\\\"_blank\\\">FLARE\u003C\u002Fa> explicitly explores active retrieval during generation, including retrieval prompted by low-confidence upcoming content.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"UThAFgyEe3\",\"data\":{\"text\":\"\u003Ca href=\\\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2310.11511\\\" target=\\\"_blank\\\">Self-RAG\u003C\u002Fa> demonstrates an architecture in which retrieval can occur on demand and is followed by reflection on retrieved passages and generated content.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"CT8n5F5KLR\",\"data\":{\"text\":\"\u003Ca href=\\\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2403.14403\\\" target=\\\"_blank\\\">Adaptive-RAG\u003C\u002Fa> dynamically chooses among different strategies according to question complexity, including situations where no retrieval is required.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"RDjo1UGf2s\",\"data\":{\"text\":\"The term Retrieval Trigger is used here as a system-level abstraction over this broader family of decisions.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"nmWZ-exi8b\",\"data\":{\"text\":\"It does not claim that these papers use the same terminology. Instead, it identifies the shared architectural problem: What causes an AI system to transition from internal knowledge to external evidence?\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"8a-H_OlfG1\",\"data\":{\"text\":\"Real Examples\",\"level\":2},\"type\":\"header\",\"tunes\":{}},{\"id\":\"l0KONt5Buo\",\"data\":{\"text\":\"Consider a support assistant connected to a company's documentation.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"MOy11BOFq5\",\"data\":{\"code\":\"\\\"How do I reset my password?\\\"\"},\"type\":\"code\",\"tunes\":{}},{\"id\":\"tsZcuMS1sT\",\"data\":{\"text\":\"If the procedure is stable and reliably represented in the assistant's current instructions, direct answering may be appropriate.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"l53NPB6WNV\",\"data\":{\"code\":\"\\\"What permissions does my account currently have?\\\"\"},\"type\":\"code\",\"tunes\":{}},{\"id\":\"RKQXMbXA29\",\"data\":{\"text\":\"That information is user-specific and dynamic. The Retrieval Trigger fires. The system must inspect the actual account or authorization data.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"b5tNqxBKir\",\"data\":{\"code\":\"\\\"Why was my production deployment rejected yesterday?\\\"\"},\"type\":\"code\",\"tunes\":{}},{\"id\":\"xaP7a7lV3i\",\"data\":{\"text\":\"The model can understand deployment systems and explain common reasons. But the question is asking about a particular event. Logs, CI\u002FCD output or incident records are required.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"SlBdofaCVq\",\"data\":{\"text\":\"The same logic works for web search.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"VgFaQjUMnU\",\"data\":{\"code\":\"\\\"What is RAG?\\\"\"},\"type\":\"code\",\"tunes\":{}},{\"id\":\"6BG7aSJQzt\",\"data\":{\"text\":\"A general explanation may not require retrieval.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"TRngQB41uY\",\"data\":{\"code\":\"\\\"What did the authors of Self-RAG specifically conclude about unnecessary retrieval?\\\"\"},\"type\":\"code\",\"tunes\":{}},{\"id\":\"7UIuDjIRyG\",\"data\":{\"text\":\"Now source-specific evidence is required.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"CG2PbVS1yz\",\"data\":{\"code\":\"\\\"What is the latest research on adaptive retrieval?\\\"\"},\"type\":\"code\",\"tunes\":{}},{\"id\":\"G1gyMnE_E8\",\"data\":{\"text\":\"This introduces a freshness requirement as well. The underlying subject has not changed. The information requirement has.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"NyJtHsPsSf\",\"data\":{\"text\":\"Common Misconceptions and Failure Modes\",\"level\":2},\"type\":\"header\",\"tunes\":{}},{\"id\":\"1otM6VenxR\",\"data\":{\"text\":\"More retrieval automatically produces a better answer. It does not. Irrelevant documents consume context and can distract generation.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"7BpMfX7lOZ\",\"data\":{\"text\":\"High model confidence means retrieval is unnecessary. A model can produce an incorrect answer confidently. Self-reported confidence should therefore not be treated as the only trigger.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"THz75XkfrR\",\"data\":{\"text\":\"Successful retrieval means the answer is verified. Retrieval only provides candidate evidence. The evidence must still be relevant, sufficiently authoritative and correctly interpreted.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"gOUGv2dAaq\",\"data\":{\"text\":\"RAG automatically solves outdated knowledge. It only does so if the retrieval corpus itself contains current information. Retrieving an outdated document does not create a current answer.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"Mz8i-je--k\",\"data\":{\"text\":\"One retrieval step is always enough. Complex questions may require several pieces of evidence or iterative retrieval.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"imAEotM35y\",\"data\":{\"text\":\"Edge Cases\",\"level\":2},\"type\":\"header\",\"tunes\":{}},{\"id\":\"8xkcG8hc9c\",\"data\":{\"text\":\"Some questions contain both stable and unstable information.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"IYDiRezoWn\",\"data\":{\"code\":\"\\\"Who founded NVIDIA, and what is its market capitalization today?\\\"\"},\"type\":\"code\",\"tunes\":{}},{\"id\":\"2kxOM8vxYh\",\"data\":{\"text\":\"The first part may be answerable from stable model knowledge. The second part requires current information.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"6PyzlxURFS\",\"data\":{\"text\":\"A sufficiently capable system should not necessarily treat the entire query as one retrieval decision. It can trigger retrieval only where required.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"lsbZ8aQAD6\",\"data\":{\"text\":\"Another edge case is disagreement between sources. Suppose retrieval returns three documents making incompatible claims.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"-Y67JvJusX\",\"data\":{\"text\":\"The Retrieval Trigger has already succeeded: the system recognized that external evidence was required. But the task is not finished.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"32VdDErqUM\",\"data\":{\"text\":\"The system has now reached an evidence evaluation problem. This is where the Answer Validity Boundary becomes important.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"edCyD-PqlU\",\"data\":{\"text\":\"The system may have retrieved information and still not possess enough evidence to make a strong conclusion.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"rmjW0MBcFo\",\"data\":{\"code\":\"Retrieval Trigger\\n≠\\npermission to answer\"},\"type\":\"code\",\"tunes\":{}},{\"id\":\"gZBq0voX0-\",\"data\":{\"text\":\"The trigger obtains evidence. The validity boundary determines whether that evidence is sufficient.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"DmO9cFY93l\",\"data\":{\"text\":\"Limitations\",\"level\":2},\"type\":\"header\",\"tunes\":{}},{\"id\":\"gV4YT_2O1X\",\"data\":{\"text\":\"The Retrieval Trigger is a conceptual framework, not a universal algorithm.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"7c6OA2X4-H\",\"data\":{\"text\":\"Different systems will require different trigger rules. A customer-support bot, scientific research assistant, search engine and autonomous software agent do not have identical evidence requirements.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"Xn8K4ArjdA\",\"data\":{\"text\":\"Trigger thresholds can also create their own failure modes. A threshold that is too low causes excessive retrieval. A threshold that is too high causes unsupported answering.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"y0gYRFZx6m\",\"data\":{\"text\":\"The retrieval infrastructure itself also matters. A perfect trigger connected to a poor source collection still produces poor evidence.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"Kcvx1v4Z1x\",\"data\":{\"text\":\"Similarly, an excellent knowledge base provides little value if the trigger never activates when it is needed.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"wcRpKvBhkb\",\"data\":{\"text\":\"The Retrieval Trigger therefore solves only one part of a larger architecture.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"YN1_g7vs7V\",\"data\":{\"text\":\"What Would Change This Answer?\",\"level\":2},\"type\":\"header\",\"tunes\":{}},{\"id\":\"4PYX_PYK_Z\",\"data\":{\"text\":\"Future models may contain better mechanisms for identifying their own knowledge limitations. Retrievers may become cheaper and faster. Long-context systems may carry far more source material continuously.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"vyqJ8Kp8Ye\",\"data\":{\"text\":\"Models may also increasingly combine search, databases, tools and structured knowledge without exposing a distinct RAG stage to the application developer.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"_KN0MPs6as\",\"data\":{\"text\":\"These changes could alter how the trigger is implemented. They do not necessarily remove the underlying decision.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"Zrlr4a0utJ\",\"data\":{\"text\":\"As long as there is a difference between information already available to the model and information that must be obtained externally, a system still needs some mechanism for determining when to cross that boundary.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"4hl7r5cF1M\",\"data\":{\"text\":\"The implementation may disappear from view. The architectural question remains.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"o_g5g_6eSj\",\"data\":{\"text\":\"Conclusion\",\"level\":2},\"type\":\"header\",\"tunes\":{}},{\"id\":\"L6MWm8xdAg\",\"data\":{\"text\":\"RAG begins too late to explain the whole problem.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"uymgoYlFM5\",\"data\":{\"text\":\"Before retrieval can happen, an AI system must determine whether retrieval is necessary. That decision is the Retrieval Trigger.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"_KIblg0ae_\",\"data\":{\"code\":\"Stable known fact\\n→ answer from model knowledge\\n\\nCurrent fact\\n→ retrieve\\n\\nSource-specific or evidence-dependent claim\\n→ retrieve and verify\"},\"type\":\"code\",\"tunes\":{}},{\"id\":\"unCgfbeYI5\",\"data\":{\"text\":\"But the broader implication is more important. Reliable AI does not merely need access to knowledge. It needs a method for determining when its current knowledge is insufficient.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"ZAosU4trn9\",\"data\":{\"code\":\"Model Knowledge\\n        ↓\\nRetrieval Trigger\\n        ↓\\nRuntime Knowledge \u002F RAG\\n        ↓\\nEvidence\\n        ↓\\nReasoning\\n        ↓\\nAnswer Validity Boundary\\n        ↓\\nAnswer\"},\"type\":\"code\",\"tunes\":{}},{\"id\":\"f0ZIysaJy1\",\"data\":{\"text\":\"The Retrieval Trigger determines when the system should seek evidence. The Answer Validity Boundary determines whether that evidence is sufficient.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"iCg9ojv75m\",\"data\":{\"text\":\"Together they describe something more useful than RAG alone: a decision process for moving from what an AI appears to know toward what it can actually support.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"cDBiNnZJv-\",\"data\":{\"text\":\"Primary Sources\",\"level\":2},\"type\":\"header\",\"tunes\":{}},{\"id\":\"8gumvODB16\",\"data\":{\"text\":\"Patrick Lewis et al., \u003Ca href=\\\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2005.11401\\\" target=\\\"_blank\\\">Retrieval-Augmented Generation for Knowledge-Intensive NLP Tasks\u003C\u002Fa> (2020). Foundational RAG work describing the combination of parametric model memory with external non-parametric memory.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"Chz6I7zmlv\",\"data\":{\"text\":\"Zhengbao Jiang et al., \u003Ca href=\\\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2305.06983\\\" target=\\\"_blank\\\">Active Retrieval Augmented Generation\u003C\u002Fa> (2023). Introduces FLARE and active retrieval during generation, including retrieval based on low-confidence predicted content.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"MRdjivpsoW\",\"data\":{\"text\":\"Akari Asai et al., \u003Ca href=\\\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2310.11511\\\" target=\\\"_blank\\\">Self-RAG: Learning to Retrieve, Generate, and Critique through Self-Reflection\u003C\u002Fa> (2023). Explores adaptive retrieval on demand and self-reflection instead of unconditional fixed retrieval.\"},\"type\":\"paragraph\",\"tunes\":{}},{\"id\":\"lck29euXJP\",\"data\":{\"text\":\"Soyeong Jeong et al., \u003Ca href=\\\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2403.14403\\\" target=\\\"_blank\\\">Adaptive-RAG: Learning to Adapt Retrieval-Augmented Large Language Models through Question Complexity\u003C\u002Fa> (2024). Dynamically selects among no retrieval, single-step retrieval and more complex retrieval strategies according to the incoming question.\"},\"type\":\"paragraph\",\"tunes\":{}}],\"version\":\"2.31.6\"}",{"time":1096,"blocks":1097,"version":1737},1790574879391,[1098,1102,1106,1110,1114,1118,1122,1126,1131,1135,1139,1143,1147,1151,1155,1158,1162,1166,1170,1174,1178,1182,1201,1205,1208,1212,1216,1219,1223,1227,1230,1234,1238,1242,1246,1250,1254,1258,1262,1266,1270,1274,1278,1282,1286,1290,1293,1297,1301,1305,1309,1313,1317,1321,1325,1329,1333,1336,1340,1344,1347,1351,1355,1359,1363,1367,1371,1375,1379,1383,1387,1391,1395,1399,1403,1407,1411,1415,1419,1423,1427,1431,1435,1439,1443,1447,1450,1454,1457,1461,1464,1468,1471,1475,1479,1483,1487,1491,1495,1499,1503,1507,1511,1515,1519,1523,1526,1530,1533,1537,1540,1544,1548,1551,1555,1558,1562,1565,1569,1573,1577,1581,1585,1589,1593,1597,1601,1604,1608,1612,1616,1620,1624,1628,1631,1635,1639,1643,1647,1651,1655,1659,1663,1667,1671,1675,1679,1683,1687,1691,1695,1699,1702,1706,1709,1713,1717,1721,1725,1729,1733],{"id":215,"data":1099,"type":42,"tunes":1101},{"text":1100,"level":218},"Question",{},{"id":221,"data":1103,"type":224,"tunes":1105},{"text":1104},"When should an AI stop relying on what it already knows and retrieve external information before answering?",{},{"id":227,"data":1107,"type":224,"tunes":1109},{"text":1108},"This question appears simple, but it sits at the center of one of the most important design decisions in modern AI systems.",{},{"id":232,"data":1111,"type":224,"tunes":1113},{"text":1112},"Large language models contain substantial knowledge in their parameters. Retrieval-Augmented Generation adds external information at runtime. But neither extreme is ideal.",{},{"id":237,"data":1115,"type":224,"tunes":1117},{"text":1116},"Always trusting the model can produce outdated or unsupported answers. Always retrieving information adds latency, cost, irrelevant context and new opportunities for retrieval errors.",{},{"id":242,"data":1119,"type":224,"tunes":1121},{"text":1120},"The real problem therefore comes before RAG: When should retrieval happen at all?",{},{"id":247,"data":1123,"type":224,"tunes":1125},{"text":1124},"This article uses the term Retrieval Trigger for that decision. Retrieval Trigger is not presented here as a standardized term from the research literature. It is a practical systems concept that brings together ideas already visible in research on active, adaptive and self-reflective retrieval.",{},{"id":252,"data":1127,"type":257,"tunes":1130},{"text":1128,"caption":1129,"alignment":256},"A Retrieval Trigger is a condition indicating that an AI system should stop relying solely on internal model knowledge and obtain external evidence before producing or finalizing an answer.","Working definition",{},{"id":260,"data":1132,"type":264,"tunes":1134},{"title":1133,"maxLevel":263,"minLevel":218},"Contents",{},{"id":267,"data":1136,"type":42,"tunes":1138},{"text":1137,"level":218},"What This Really Means",{},{"id":272,"data":1140,"type":224,"tunes":1142},{"text":1141},"An LLM has two fundamentally different ways of obtaining information.",{},{"id":277,"data":1144,"type":224,"tunes":1146},{"text":1145},"The first is model knowledge. This is information represented in the model's learned parameters. No database query, web search or document lookup is required at runtime.",{},{"id":282,"data":1148,"type":224,"tunes":1150},{"text":1149},"The second is runtime knowledge. This is information provided while the model is operating: search results, database records, documents, APIs, user files, tool outputs or other retrieved evidence.",{},{"id":287,"data":1152,"type":224,"tunes":1154},{"text":1153},"RAG connects these two worlds. But RAG itself does not answer the question of when that connection should be activated. That is the purpose of the Retrieval Trigger.",{},{"id":292,"data":1156,"type":295,"tunes":1157},{"code":294},{},{"id":298,"data":1159,"type":224,"tunes":1161},{"text":1160},"The Retrieval Trigger therefore sits before retrieval. The Answer Validity Boundary sits later.",{},{"id":303,"data":1163,"type":224,"tunes":1165},{"text":1164},"The first asks: Do I need external evidence?",{},{"id":308,"data":1167,"type":224,"tunes":1169},{"text":1168},"The second asks: Do I now have enough evidence to support this answer?",{},{"id":313,"data":1171,"type":224,"tunes":1173},{"text":1172},"These are related decisions, but they are not the same decision.",{},{"id":318,"data":1175,"type":42,"tunes":1177},{"text":1176,"level":218},"Simplest Example",{},{"id":323,"data":1179,"type":224,"tunes":1181},{"text":1180},"Consider three questions.",{},{"id":328,"data":1183,"type":346,"tunes":1200},{"content":1184,"stretched":43,"withHeadings":14},[1185,1188,1192,1196],[1100,1186,1187],"Internal knowledge","Retrieval Trigger",[1189,1190,1191],"What is the capital of France?","Usually sufficient","No strong trigger",[1193,1194,1195],"What is the current NVIDIA stock price?","Potentially outdated","Trigger retrieval",[1197,1198,1199],"Does this new scientific paper prove that X causes Y?","Cannot establish the claim without examining the evidence","Strong retrieval trigger",{},{"id":349,"data":1202,"type":224,"tunes":1204},{"text":1203},"The first question is based on a highly stable fact.",{},{"id":354,"data":1206,"type":295,"tunes":1207},{"code":356},{},{"id":359,"data":1209,"type":224,"tunes":1211},{"text":1210},"Retrieving documents before answering would usually add little value.",{},{"id":364,"data":1213,"type":224,"tunes":1215},{"text":1214},"Now consider a question whose answer changes continuously.",{},{"id":369,"data":1217,"type":295,"tunes":1218},{"code":371},{},{"id":374,"data":1220,"type":224,"tunes":1222},{"text":1221},"The model may know a great deal about NVIDIA. That does not mean it knows the price now.",{},{"id":379,"data":1224,"type":224,"tunes":1226},{"text":1225},"The third example is even more important.",{},{"id":384,"data":1228,"type":295,"tunes":1229},{"code":386},{},{"id":389,"data":1231,"type":224,"tunes":1233},{"text":1232},"The model's reasoning capability may be perfectly useful. The missing component is evidence.",{},{"id":394,"data":1235,"type":224,"tunes":1237},{"text":1236},"That distinction is fundamental.",{},{"id":399,"data":1239,"type":42,"tunes":1241},{"text":1240,"level":218},"Where the Example Stops Working",{},{"id":404,"data":1243,"type":224,"tunes":1245},{"text":1244},"The examples above make the decision appear binary: retrieve or do not retrieve.",{},{"id":409,"data":1247,"type":224,"tunes":1249},{"text":1248},"Real systems are more complicated. A question may contain several claims, some stable and some current. Retrieved documents may disagree. A retriever may return irrelevant information. The relevant information may exist but fail to rank highly enough. A document may be authoritative but outdated.",{},{"id":414,"data":1251,"type":224,"tunes":1253},{"text":1252},"Retrieval itself can also introduce incorrect context into an otherwise reasonable answer.",{},{"id":419,"data":1255,"type":224,"tunes":1257},{"text":1256},"This is why retrieval should not be treated as an automatic synonym for truth.",{},{"id":424,"data":1259,"type":224,"tunes":1261},{"text":1260},"Research on adaptive retrieval has increasingly moved away from the assumption that every query should receive the same retrieval strategy.",{},{"id":429,"data":1263,"type":224,"tunes":1265},{"text":1264},"\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2310.11511\" target=\"_blank\">Self-RAG\u003C\u002Fa>, for example, explicitly explores retrieval on demand rather than indiscriminately retrieving a fixed number of passages for every input. The authors discuss how unnecessary or irrelevant retrieval can reduce answer quality.",{},{"id":434,"data":1267,"type":224,"tunes":1269},{"text":1268},"\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2403.14403\" target=\"_blank\">Adaptive-RAG\u003C\u002Fa> similarly selects between no retrieval, single-step retrieval and more complex retrieval strategies according to question complexity.",{},{"id":439,"data":1271,"type":224,"tunes":1273},{"text":1272},"So the important question is not: Does this system have RAG?",{},{"id":444,"data":1275,"type":224,"tunes":1277},{"text":1276},"It is: Can this system recognize when retrieval is necessary and what kind of retrieval is appropriate?",{},{"id":449,"data":1279,"type":42,"tunes":1281},{"text":1280,"level":218},"Direct Answer",{},{"id":454,"data":1283,"type":224,"tunes":1285},{"text":1284},"An AI should trigger retrieval when answering requires information that its internal model knowledge cannot safely provide with the required freshness, specificity, provenance or evidential support.",{},{"id":459,"data":1287,"type":224,"tunes":1289},{"text":1288},"In practical systems, a Retrieval Trigger can emerge from several conditions:",{},{"id":464,"data":1291,"type":295,"tunes":1292},{"code":466},{},{"id":469,"data":1294,"type":224,"tunes":1296},{"text":1295},"If none of these conditions is materially present, retrieval may be unnecessary. If one or more are present, external evidence becomes part of the answer-generation process.",{},{"id":474,"data":1298,"type":42,"tunes":1300},{"text":1299,"level":218},"Why This Is So",{},{"id":479,"data":1302,"type":224,"tunes":1304},{"text":1303},"A language model's internal knowledge is often described as parametric knowledge. It was learned during training and encoded into the model's parameters.",{},{"id":484,"data":1306,"type":224,"tunes":1308},{"text":1307},"\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2005.11401\" target=\"_blank\">Lewis et al.'s original RAG work\u003C\u002Fa> framed retrieval as a combination of this parametric memory with external, non-parametric memory. The external memory can be searched and updated without retraining the entire language model.",{},{"id":489,"data":1310,"type":224,"tunes":1312},{"text":1311},"This distinction creates an unavoidable systems problem.",{},{"id":494,"data":1314,"type":224,"tunes":1316},{"text":1315},"The model can know things. But the model cannot assume that everything it knows is current, complete, specific enough and supported by the required evidence.",{},{"id":499,"data":1318,"type":224,"tunes":1320},{"text":1319},"A model can therefore produce a linguistically convincing answer while still operating beyond the point where its internal knowledge is sufficient.",{},{"id":504,"data":1322,"type":224,"tunes":1324},{"text":1323},"That point is where a Retrieval Trigger becomes useful.",{},{"id":509,"data":1326,"type":42,"tunes":1328},{"text":1327,"level":218},"Context",{},{"id":514,"data":1330,"type":224,"tunes":1332},{"text":1331},"Traditional RAG often looks like this:",{},{"id":519,"data":1334,"type":295,"tunes":1335},{"code":521},{},{"id":524,"data":1337,"type":224,"tunes":1339},{"text":1338},"This architecture assumes retrieval before generation. That works well for many knowledge-intensive applications, but it can also perform unnecessary retrieval.",{},{"id":529,"data":1341,"type":224,"tunes":1343},{"text":1342},"More advanced approaches introduce an adaptive step:",{},{"id":534,"data":1345,"type":295,"tunes":1346},{"code":536},{},{"id":539,"data":1348,"type":224,"tunes":1350},{"text":1349},"\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2305.06983\" target=\"_blank\">FLARE\u003C\u002Fa> goes further by considering retrieval during generation itself. It uses upcoming generation and low-confidence tokens as signals for retrieving additional information.",{},{"id":544,"data":1352,"type":224,"tunes":1354},{"text":1353},"Self-RAG similarly introduces mechanisms allowing retrieval, generation and critique to interact instead of treating retrieval as an unconditional preprocessing step.",{},{"id":549,"data":1356,"type":224,"tunes":1358},{"text":1357},"Adaptive-RAG approaches the same broader problem from query complexity: different questions may require different retrieval strategies.",{},{"id":554,"data":1360,"type":224,"tunes":1362},{"text":1361},"These approaches differ technically. But they expose the same architectural insight: Retrieval should be a decision, not merely a permanent switch.",{},{"id":559,"data":1364,"type":42,"tunes":1366},{"text":1365,"level":218},"Assumptions",{},{"id":564,"data":1368,"type":224,"tunes":1370},{"text":1369},"The Retrieval Trigger framework assumes that a system has access to at least one external information source when retrieval is required.",{},{"id":569,"data":1372,"type":224,"tunes":1374},{"text":1373},"That source could be web search, a document store, vector database, SQL database, knowledge graph, API, enterprise system, user-uploaded document or tool output.",{},{"id":574,"data":1376,"type":224,"tunes":1378},{"text":1377},"It also assumes that retrieval has a cost. That cost does not have to be financial.",{},{"id":579,"data":1380,"type":224,"tunes":1382},{"text":1381},"Retrieval introduces latency, token consumption, context usage, infrastructure complexity and the possibility of retrieving misleading information.",{},{"id":584,"data":1384,"type":224,"tunes":1386},{"text":1385},"The optimal system therefore does not maximize retrieval. It maximizes appropriate retrieval.",{},{"id":589,"data":1388,"type":42,"tunes":1390},{"text":1389,"level":218},"Variables",{},{"id":594,"data":1392,"type":224,"tunes":1394},{"text":1393},"A practical Retrieval Trigger can consider five primary variables.",{},{"id":599,"data":1396,"type":42,"tunes":1398},{"text":1397,"level":263},"Freshness",{},{"id":604,"data":1400,"type":224,"tunes":1402},{"text":1401},"How likely is the required information to have changed? The capital of France has very low volatility. A stock price has extremely high volatility.",{},{"id":609,"data":1404,"type":42,"tunes":1406},{"text":1405,"level":263},"Specificity",{},{"id":614,"data":1408,"type":224,"tunes":1410},{"text":1409},"Does the question require information from a particular source, document, organization, account or dataset? If the user asks what a specific contract says, general model knowledge is irrelevant. The contract must be retrieved.",{},{"id":619,"data":1412,"type":42,"tunes":1414},{"text":1413,"level":263},"Evidence Requirement",{},{"id":624,"data":1416,"type":224,"tunes":1418},{"text":1417},"Does the answer need provenance? A model may know that a claim is generally accepted but still need a source when the task requires verification.",{},{"id":629,"data":1420,"type":42,"tunes":1422},{"text":1421,"level":263},"Knowledge Coverage",{},{"id":634,"data":1424,"type":224,"tunes":1426},{"text":1425},"Is the subject likely to be represented adequately in internal model knowledge? Rare, proprietary, highly local or newly published information creates stronger retrieval pressure.",{},{"id":639,"data":1428,"type":42,"tunes":1430},{"text":1429,"level":263},"Consequence of Error",{},{"id":644,"data":1432,"type":224,"tunes":1434},{"text":1433},"Not every incorrect answer has the same impact. Where factual accuracy materially affects a decision, the acceptable evidence threshold may be higher.",{},{"id":649,"data":1436,"type":224,"tunes":1438},{"text":1437},"These variables do not have to be implemented as literal numeric scores. They describe the decision surface.",{},{"id":654,"data":1440,"type":42,"tunes":1442},{"text":1441,"level":218},"Diagnostic \u002F Decision Method",{},{"id":659,"data":1444,"type":224,"tunes":1446},{"text":1445},"A very simple Retrieval Trigger can be implemented without machine learning.",{},{"id":664,"data":1448,"type":295,"tunes":1449},{"code":666},{},{"id":669,"data":1451,"type":224,"tunes":1453},{"text":1452},"For a stable factual question:",{},{"id":674,"data":1455,"type":295,"tunes":1456},{"code":676},{},{"id":679,"data":1458,"type":224,"tunes":1460},{"text":1459},"For a current stock price:",{},{"id":684,"data":1462,"type":295,"tunes":1463},{"code":686},{},{"id":689,"data":1465,"type":224,"tunes":1467},{"text":1466},"For a scientific claim:",{},{"id":694,"data":1469,"type":295,"tunes":1470},{"code":696},{},{"id":699,"data":1472,"type":224,"tunes":1474},{"text":1473},"Production systems can make this decision far more sophisticated. A classifier could predict retrieval requirements. A model could emit special control tokens. A router could classify query complexity. Retrieval could also be triggered repeatedly during generation.",{},{"id":704,"data":1476,"type":224,"tunes":1478},{"text":1477},"The implementation can change. The architectural question remains the same:",{},{"id":709,"data":1480,"type":257,"tunes":1482},{"text":1481,"caption":712,"alignment":256},"Is the evidence currently available to the model sufficient for the answer it is about to produce?",{},{"id":715,"data":1484,"type":42,"tunes":1486},{"text":1485,"level":218},"Evidence",{},{"id":720,"data":1488,"type":224,"tunes":1490},{"text":1489},"The concept proposed here is consistent with several lines of retrieval research.",{},{"id":725,"data":1492,"type":224,"tunes":1494},{"text":1493},"The original \u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2005.11401\" target=\"_blank\">RAG architecture\u003C\u002Fa> demonstrated the usefulness of combining parametric model knowledge with external non-parametric knowledge, particularly for knowledge-intensive tasks.",{},{"id":730,"data":1496,"type":224,"tunes":1498},{"text":1497},"\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2305.06983\" target=\"_blank\">FLARE\u003C\u002Fa> explicitly explores active retrieval during generation, including retrieval prompted by low-confidence upcoming content.",{},{"id":735,"data":1500,"type":224,"tunes":1502},{"text":1501},"\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2310.11511\" target=\"_blank\">Self-RAG\u003C\u002Fa> demonstrates an architecture in which retrieval can occur on demand and is followed by reflection on retrieved passages and generated content.",{},{"id":740,"data":1504,"type":224,"tunes":1506},{"text":1505},"\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2403.14403\" target=\"_blank\">Adaptive-RAG\u003C\u002Fa> dynamically chooses among different strategies according to question complexity, including situations where no retrieval is required.",{},{"id":745,"data":1508,"type":224,"tunes":1510},{"text":1509},"The term Retrieval Trigger is used here as a system-level abstraction over this broader family of decisions.",{},{"id":750,"data":1512,"type":224,"tunes":1514},{"text":1513},"It does not claim that these papers use the same terminology. Instead, it identifies the shared architectural problem: What causes an AI system to transition from internal knowledge to external evidence?",{},{"id":755,"data":1516,"type":42,"tunes":1518},{"text":1517,"level":218},"Real Examples",{},{"id":760,"data":1520,"type":224,"tunes":1522},{"text":1521},"Consider a support assistant connected to a company's documentation.",{},{"id":765,"data":1524,"type":295,"tunes":1525},{"code":767},{},{"id":770,"data":1527,"type":224,"tunes":1529},{"text":1528},"If the procedure is stable and reliably represented in the assistant's current instructions, direct answering may be appropriate.",{},{"id":775,"data":1531,"type":295,"tunes":1532},{"code":777},{},{"id":780,"data":1534,"type":224,"tunes":1536},{"text":1535},"That information is user-specific and dynamic. The Retrieval Trigger fires. The system must inspect the actual account or authorization data.",{},{"id":785,"data":1538,"type":295,"tunes":1539},{"code":787},{},{"id":790,"data":1541,"type":224,"tunes":1543},{"text":1542},"The model can understand deployment systems and explain common reasons. But the question is asking about a particular event. Logs, CI\u002FCD output or incident records are required.",{},{"id":795,"data":1545,"type":224,"tunes":1547},{"text":1546},"The same logic works for web search.",{},{"id":800,"data":1549,"type":295,"tunes":1550},{"code":802},{},{"id":805,"data":1552,"type":224,"tunes":1554},{"text":1553},"A general explanation may not require retrieval.",{},{"id":810,"data":1556,"type":295,"tunes":1557},{"code":812},{},{"id":815,"data":1559,"type":224,"tunes":1561},{"text":1560},"Now source-specific evidence is required.",{},{"id":820,"data":1563,"type":295,"tunes":1564},{"code":822},{},{"id":825,"data":1566,"type":224,"tunes":1568},{"text":1567},"This introduces a freshness requirement as well. The underlying subject has not changed. The information requirement has.",{},{"id":830,"data":1570,"type":42,"tunes":1572},{"text":1571,"level":218},"Common Misconceptions and Failure Modes",{},{"id":835,"data":1574,"type":224,"tunes":1576},{"text":1575},"More retrieval automatically produces a better answer. It does not. Irrelevant documents consume context and can distract generation.",{},{"id":840,"data":1578,"type":224,"tunes":1580},{"text":1579},"High model confidence means retrieval is unnecessary. A model can produce an incorrect answer confidently. Self-reported confidence should therefore not be treated as the only trigger.",{},{"id":845,"data":1582,"type":224,"tunes":1584},{"text":1583},"Successful retrieval means the answer is verified. Retrieval only provides candidate evidence. The evidence must still be relevant, sufficiently authoritative and correctly interpreted.",{},{"id":850,"data":1586,"type":224,"tunes":1588},{"text":1587},"RAG automatically solves outdated knowledge. It only does so if the retrieval corpus itself contains current information. Retrieving an outdated document does not create a current answer.",{},{"id":855,"data":1590,"type":224,"tunes":1592},{"text":1591},"One retrieval step is always enough. Complex questions may require several pieces of evidence or iterative retrieval.",{},{"id":860,"data":1594,"type":42,"tunes":1596},{"text":1595,"level":218},"Edge Cases",{},{"id":865,"data":1598,"type":224,"tunes":1600},{"text":1599},"Some questions contain both stable and unstable information.",{},{"id":870,"data":1602,"type":295,"tunes":1603},{"code":872},{},{"id":875,"data":1605,"type":224,"tunes":1607},{"text":1606},"The first part may be answerable from stable model knowledge. The second part requires current information.",{},{"id":880,"data":1609,"type":224,"tunes":1611},{"text":1610},"A sufficiently capable system should not necessarily treat the entire query as one retrieval decision. It can trigger retrieval only where required.",{},{"id":885,"data":1613,"type":224,"tunes":1615},{"text":1614},"Another edge case is disagreement between sources. Suppose retrieval returns three documents making incompatible claims.",{},{"id":890,"data":1617,"type":224,"tunes":1619},{"text":1618},"The Retrieval Trigger has already succeeded: the system recognized that external evidence was required. But the task is not finished.",{},{"id":895,"data":1621,"type":224,"tunes":1623},{"text":1622},"The system has now reached an evidence evaluation problem. This is where the Answer Validity Boundary becomes important.",{},{"id":900,"data":1625,"type":224,"tunes":1627},{"text":1626},"The system may have retrieved information and still not possess enough evidence to make a strong conclusion.",{},{"id":905,"data":1629,"type":295,"tunes":1630},{"code":907},{},{"id":910,"data":1632,"type":224,"tunes":1634},{"text":1633},"The trigger obtains evidence. The validity boundary determines whether that evidence is sufficient.",{},{"id":915,"data":1636,"type":42,"tunes":1638},{"text":1637,"level":218},"Limitations",{},{"id":920,"data":1640,"type":224,"tunes":1642},{"text":1641},"The Retrieval Trigger is a conceptual framework, not a universal algorithm.",{},{"id":925,"data":1644,"type":224,"tunes":1646},{"text":1645},"Different systems will require different trigger rules. A customer-support bot, scientific research assistant, search engine and autonomous software agent do not have identical evidence requirements.",{},{"id":930,"data":1648,"type":224,"tunes":1650},{"text":1649},"Trigger thresholds can also create their own failure modes. A threshold that is too low causes excessive retrieval. A threshold that is too high causes unsupported answering.",{},{"id":935,"data":1652,"type":224,"tunes":1654},{"text":1653},"The retrieval infrastructure itself also matters. A perfect trigger connected to a poor source collection still produces poor evidence.",{},{"id":940,"data":1656,"type":224,"tunes":1658},{"text":1657},"Similarly, an excellent knowledge base provides little value if the trigger never activates when it is needed.",{},{"id":945,"data":1660,"type":224,"tunes":1662},{"text":1661},"The Retrieval Trigger therefore solves only one part of a larger architecture.",{},{"id":950,"data":1664,"type":42,"tunes":1666},{"text":1665,"level":218},"What Would Change This Answer?",{},{"id":955,"data":1668,"type":224,"tunes":1670},{"text":1669},"Future models may contain better mechanisms for identifying their own knowledge limitations. Retrievers may become cheaper and faster. Long-context systems may carry far more source material continuously.",{},{"id":960,"data":1672,"type":224,"tunes":1674},{"text":1673},"Models may also increasingly combine search, databases, tools and structured knowledge without exposing a distinct RAG stage to the application developer.",{},{"id":965,"data":1676,"type":224,"tunes":1678},{"text":1677},"These changes could alter how the trigger is implemented. They do not necessarily remove the underlying decision.",{},{"id":970,"data":1680,"type":224,"tunes":1682},{"text":1681},"As long as there is a difference between information already available to the model and information that must be obtained externally, a system still needs some mechanism for determining when to cross that boundary.",{},{"id":975,"data":1684,"type":224,"tunes":1686},{"text":1685},"The implementation may disappear from view. The architectural question remains.",{},{"id":980,"data":1688,"type":42,"tunes":1690},{"text":1689,"level":218},"Conclusion",{},{"id":985,"data":1692,"type":224,"tunes":1694},{"text":1693},"RAG begins too late to explain the whole problem.",{},{"id":990,"data":1696,"type":224,"tunes":1698},{"text":1697},"Before retrieval can happen, an AI system must determine whether retrieval is necessary. That decision is the Retrieval Trigger.",{},{"id":995,"data":1700,"type":295,"tunes":1701},{"code":997},{},{"id":1000,"data":1703,"type":224,"tunes":1705},{"text":1704},"But the broader implication is more important. Reliable AI does not merely need access to knowledge. It needs a method for determining when its current knowledge is insufficient.",{},{"id":1005,"data":1707,"type":295,"tunes":1708},{"code":1007},{},{"id":1010,"data":1710,"type":224,"tunes":1712},{"text":1711},"The Retrieval Trigger determines when the system should seek evidence. The Answer Validity Boundary determines whether that evidence is sufficient.",{},{"id":1015,"data":1714,"type":224,"tunes":1716},{"text":1715},"Together they describe something more useful than RAG alone: a decision process for moving from what an AI appears to know toward what it can actually support.",{},{"id":1020,"data":1718,"type":42,"tunes":1720},{"text":1719,"level":218},"Primary Sources",{},{"id":1025,"data":1722,"type":224,"tunes":1724},{"text":1723},"Patrick Lewis et al., \u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2005.11401\" target=\"_blank\">Retrieval-Augmented Generation for Knowledge-Intensive NLP Tasks\u003C\u002Fa> (2020). Foundational RAG work describing the combination of parametric model memory with external non-parametric memory.",{},{"id":1030,"data":1726,"type":224,"tunes":1728},{"text":1727},"Zhengbao Jiang et al., \u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2305.06983\" target=\"_blank\">Active Retrieval Augmented Generation\u003C\u002Fa> (2023). Introduces FLARE and active retrieval during generation, including retrieval based on low-confidence predicted content.",{},{"id":1035,"data":1730,"type":224,"tunes":1732},{"text":1731},"Akari Asai et al., \u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2310.11511\" target=\"_blank\">Self-RAG: Learning to Retrieve, Generate, and Critique through Self-Reflection\u003C\u002Fa> (2023). Explores adaptive retrieval on demand and self-reflection instead of unconditional fixed retrieval.",{},{"id":1040,"data":1734,"type":224,"tunes":1736},{"text":1735},"Soyeong Jeong et al., \u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2403.14403\" target=\"_blank\">Adaptive-RAG: Learning to Adapt Retrieval-Augmented Large Language Models through Question Complexity\u003C\u002Fa> (2024). Dynamically selects among no retrieval, single-step retrieval and more complex retrieval strategies according to the incoming question.",{},"2.31.6","An AI model does not need retrieval for every question. The important problem is knowing when its internal knowledge is no longer enough. The Retrieval Trigger is a practical decision boundary that determines when an AI system should stop relying solely on model knowledge and obtain external evidence before answering.",{"lang":7,"title":208,"content":210,"contentJson":1740,"excerpt":1045},{"time":212,"blocks":1741,"version":1044},[1742,1745,1748,1751,1754,1757,1760,1763,1766,1769,1772,1775,1778,1781,1784,1787,1790,1793,1796,1799,1802,1805,1813,1816,1819,1822,1825,1828,1831,1834,1837,1840,1843,1846,1849,1852,1855,1858,1861,1864,1867,1870,1873,1876,1879,1882,1885,1888,1891,1894,1897,1900,1903,1906,1909,1912,1915,1918,1921,1924,1927,1930,1933,1936,1939,1942,1945,1948,1951,1954,1957,1960,1963,1966,1969,1972,1975,1978,1981,1984,1987,1990,1993,1996,1999,2002,2005,2008,2011,2014,2017,2020,2023,2026,2029,2032,2035,2038,2041,2044,2047,2050,2053,2056,2059,2062,2065,2068,2071,2074,2077,2080,2083,2086,2089,2092,2095,2098,2101,2104,2107,2110,2113,2116,2119,2122,2125,2128,2131,2134,2137,2140,2143,2146,2149,2152,2155,2158,2161,2164,2167,2170,2173,2176,2179,2182,2185,2188,2191,2194,2197,2200,2203,2206,2209,2212,2215,2218,2221,2224,2227],{"id":215,"data":1743,"type":42,"tunes":1744},{"text":217,"level":218},{},{"id":221,"data":1746,"type":224,"tunes":1747},{"text":223},{},{"id":227,"data":1749,"type":224,"tunes":1750},{"text":229},{},{"id":232,"data":1752,"type":224,"tunes":1753},{"text":234},{},{"id":237,"data":1755,"type":224,"tunes":1756},{"text":239},{},{"id":242,"data":1758,"type":224,"tunes":1759},{"text":244},{},{"id":247,"data":1761,"type":224,"tunes":1762},{"text":249},{},{"id":252,"data":1764,"type":257,"tunes":1765},{"text":254,"caption":255,"alignment":256},{},{"id":260,"data":1767,"type":264,"tunes":1768},{"title":262,"maxLevel":263,"minLevel":218},{},{"id":267,"data":1770,"type":42,"tunes":1771},{"text":269,"level":218},{},{"id":272,"data":1773,"type":224,"tunes":1774},{"text":274},{},{"id":277,"data":1776,"type":224,"tunes":1777},{"text":279},{},{"id":282,"data":1779,"type":224,"tunes":1780},{"text":284},{},{"id":287,"data":1782,"type":224,"tunes":1783},{"text":289},{},{"id":292,"data":1785,"type":295,"tunes":1786},{"code":294},{},{"id":298,"data":1788,"type":224,"tunes":1789},{"text":300},{},{"id":303,"data":1791,"type":224,"tunes":1792},{"text":305},{},{"id":308,"data":1794,"type":224,"tunes":1795},{"text":310},{},{"id":313,"data":1797,"type":224,"tunes":1798},{"text":315},{},{"id":318,"data":1800,"type":42,"tunes":1801},{"text":320,"level":218},{},{"id":323,"data":1803,"type":224,"tunes":1804},{"text":325},{},{"id":328,"data":1806,"type":346,"tunes":1812},{"content":1807,"stretched":43,"withHeadings":14},[1808,1809,1810,1811],[217,332,333],[335,336,337],[339,340,341],[343,344,345],{},{"id":349,"data":1814,"type":224,"tunes":1815},{"text":351},{},{"id":354,"data":1817,"type":295,"tunes":1818},{"code":356},{},{"id":359,"data":1820,"type":224,"tunes":1821},{"text":361},{},{"id":364,"data":1823,"type":224,"tunes":1824},{"text":366},{},{"id":369,"data":1826,"type":295,"tunes":1827},{"code":371},{},{"id":374,"data":1829,"type":224,"tunes":1830},{"text":376},{},{"id":379,"data":1832,"type":224,"tunes":1833},{"text":381},{},{"id":384,"data":1835,"type":295,"tunes":1836},{"code":386},{},{"id":389,"data":1838,"type":224,"tunes":1839},{"text":391},{},{"id":394,"data":1841,"type":224,"tunes":1842},{"text":396},{},{"id":399,"data":1844,"type":42,"tunes":1845},{"text":401,"level":218},{},{"id":404,"data":1847,"type":224,"tunes":1848},{"text":406},{},{"id":409,"data":1850,"type":224,"tunes":1851},{"text":411},{},{"id":414,"data":1853,"type":224,"tunes":1854},{"text":416},{},{"id":419,"data":1856,"type":224,"tunes":1857},{"text":421},{},{"id":424,"data":1859,"type":224,"tunes":1860},{"text":426},{},{"id":429,"data":1862,"type":224,"tunes":1863},{"text":431},{},{"id":434,"data":1865,"type":224,"tunes":1866},{"text":436},{},{"id":439,"data":1868,"type":224,"tunes":1869},{"text":441},{},{"id":444,"data":1871,"type":224,"tunes":1872},{"text":446},{},{"id":449,"data":1874,"type":42,"tunes":1875},{"text":451,"level":218},{},{"id":454,"data":1877,"type":224,"tunes":1878},{"text":456},{},{"id":459,"data":1880,"type":224,"tunes":1881},{"text":461},{},{"id":464,"data":1883,"type":295,"tunes":1884},{"code":466},{},{"id":469,"data":1886,"type":224,"tunes":1887},{"text":471},{},{"id":474,"data":1889,"type":42,"tunes":1890},{"text":476,"level":218},{},{"id":479,"data":1892,"type":224,"tunes":1893},{"text":481},{},{"id":484,"data":1895,"type":224,"tunes":1896},{"text":486},{},{"id":489,"data":1898,"type":224,"tunes":1899},{"text":491},{},{"id":494,"data":1901,"type":224,"tunes":1902},{"text":496},{},{"id":499,"data":1904,"type":224,"tunes":1905},{"text":501},{},{"id":504,"data":1907,"type":224,"tunes":1908},{"text":506},{},{"id":509,"data":1910,"type":42,"tunes":1911},{"text":511,"level":218},{},{"id":514,"data":1913,"type":224,"tunes":1914},{"text":516},{},{"id":519,"data":1916,"type":295,"tunes":1917},{"code":521},{},{"id":524,"data":1919,"type":224,"tunes":1920},{"text":526},{},{"id":529,"data":1922,"type":224,"tunes":1923},{"text":531},{},{"id":534,"data":1925,"type":295,"tunes":1926},{"code":536},{},{"id":539,"data":1928,"type":224,"tunes":1929},{"text":541},{},{"id":544,"data":1931,"type":224,"tunes":1932},{"text":546},{},{"id":549,"data":1934,"type":224,"tunes":1935},{"text":551},{},{"id":554,"data":1937,"type":224,"tunes":1938},{"text":556},{},{"id":559,"data":1940,"type":42,"tunes":1941},{"text":561,"level":218},{},{"id":564,"data":1943,"type":224,"tunes":1944},{"text":566},{},{"id":569,"data":1946,"type":224,"tunes":1947},{"text":571},{},{"id":574,"data":1949,"type":224,"tunes":1950},{"text":576},{},{"id":579,"data":1952,"type":224,"tunes":1953},{"text":581},{},{"id":584,"data":1955,"type":224,"tunes":1956},{"text":586},{},{"id":589,"data":1958,"type":42,"tunes":1959},{"text":591,"level":218},{},{"id":594,"data":1961,"type":224,"tunes":1962},{"text":596},{},{"id":599,"data":1964,"type":42,"tunes":1965},{"text":601,"level":263},{},{"id":604,"data":1967,"type":224,"tunes":1968},{"text":606},{},{"id":609,"data":1970,"type":42,"tunes":1971},{"text":611,"level":263},{},{"id":614,"data":1973,"type":224,"tunes":1974},{"text":616},{},{"id":619,"data":1976,"type":42,"tunes":1977},{"text":621,"level":263},{},{"id":624,"data":1979,"type":224,"tunes":1980},{"text":626},{},{"id":629,"data":1982,"type":42,"tunes":1983},{"text":631,"level":263},{},{"id":634,"data":1985,"type":224,"tunes":1986},{"text":636},{},{"id":639,"data":1988,"type":42,"tunes":1989},{"text":641,"level":263},{},{"id":644,"data":1991,"type":224,"tunes":1992},{"text":646},{},{"id":649,"data":1994,"type":224,"tunes":1995},{"text":651},{},{"id":654,"data":1997,"type":42,"tunes":1998},{"text":656,"level":218},{},{"id":659,"data":2000,"type":224,"tunes":2001},{"text":661},{},{"id":664,"data":2003,"type":295,"tunes":2004},{"code":666},{},{"id":669,"data":2006,"type":224,"tunes":2007},{"text":671},{},{"id":674,"data":2009,"type":295,"tunes":2010},{"code":676},{},{"id":679,"data":2012,"type":224,"tunes":2013},{"text":681},{},{"id":684,"data":2015,"type":295,"tunes":2016},{"code":686},{},{"id":689,"data":2018,"type":224,"tunes":2019},{"text":691},{},{"id":694,"data":2021,"type":295,"tunes":2022},{"code":696},{},{"id":699,"data":2024,"type":224,"tunes":2025},{"text":701},{},{"id":704,"data":2027,"type":224,"tunes":2028},{"text":706},{},{"id":709,"data":2030,"type":257,"tunes":2031},{"text":711,"caption":712,"alignment":256},{},{"id":715,"data":2033,"type":42,"tunes":2034},{"text":717,"level":218},{},{"id":720,"data":2036,"type":224,"tunes":2037},{"text":722},{},{"id":725,"data":2039,"type":224,"tunes":2040},{"text":727},{},{"id":730,"data":2042,"type":224,"tunes":2043},{"text":732},{},{"id":735,"data":2045,"type":224,"tunes":2046},{"text":737},{},{"id":740,"data":2048,"type":224,"tunes":2049},{"text":742},{},{"id":745,"data":2051,"type":224,"tunes":2052},{"text":747},{},{"id":750,"data":2054,"type":224,"tunes":2055},{"text":752},{},{"id":755,"data":2057,"type":42,"tunes":2058},{"text":757,"level":218},{},{"id":760,"data":2060,"type":224,"tunes":2061},{"text":762},{},{"id":765,"data":2063,"type":295,"tunes":2064},{"code":767},{},{"id":770,"data":2066,"type":224,"tunes":2067},{"text":772},{},{"id":775,"data":2069,"type":295,"tunes":2070},{"code":777},{},{"id":780,"data":2072,"type":224,"tunes":2073},{"text":782},{},{"id":785,"data":2075,"type":295,"tunes":2076},{"code":787},{},{"id":790,"data":2078,"type":224,"tunes":2079},{"text":792},{},{"id":795,"data":2081,"type":224,"tunes":2082},{"text":797},{},{"id":800,"data":2084,"type":295,"tunes":2085},{"code":802},{},{"id":805,"data":2087,"type":224,"tunes":2088},{"text":807},{},{"id":810,"data":2090,"type":295,"tunes":2091},{"code":812},{},{"id":815,"data":2093,"type":224,"tunes":2094},{"text":817},{},{"id":820,"data":2096,"type":295,"tunes":2097},{"code":822},{},{"id":825,"data":2099,"type":224,"tunes":2100},{"text":827},{},{"id":830,"data":2102,"type":42,"tunes":2103},{"text":832,"level":218},{},{"id":835,"data":2105,"type":224,"tunes":2106},{"text":837},{},{"id":840,"data":2108,"type":224,"tunes":2109},{"text":842},{},{"id":845,"data":2111,"type":224,"tunes":2112},{"text":847},{},{"id":850,"data":2114,"type":224,"tunes":2115},{"text":852},{},{"id":855,"data":2117,"type":224,"tunes":2118},{"text":857},{},{"id":860,"data":2120,"type":42,"tunes":2121},{"text":862,"level":218},{},{"id":865,"data":2123,"type":224,"tunes":2124},{"text":867},{},{"id":870,"data":2126,"type":295,"tunes":2127},{"code":872},{},{"id":875,"data":2129,"type":224,"tunes":2130},{"text":877},{},{"id":880,"data":2132,"type":224,"tunes":2133},{"text":882},{},{"id":885,"data":2135,"type":224,"tunes":2136},{"text":887},{},{"id":890,"data":2138,"type":224,"tunes":2139},{"text":892},{},{"id":895,"data":2141,"type":224,"tunes":2142},{"text":897},{},{"id":900,"data":2144,"type":224,"tunes":2145},{"text":902},{},{"id":905,"data":2147,"type":295,"tunes":2148},{"code":907},{},{"id":910,"data":2150,"type":224,"tunes":2151},{"text":912},{},{"id":915,"data":2153,"type":42,"tunes":2154},{"text":917,"level":218},{},{"id":920,"data":2156,"type":224,"tunes":2157},{"text":922},{},{"id":925,"data":2159,"type":224,"tunes":2160},{"text":927},{},{"id":930,"data":2162,"type":224,"tunes":2163},{"text":932},{},{"id":935,"data":2165,"type":224,"tunes":2166},{"text":937},{},{"id":940,"data":2168,"type":224,"tunes":2169},{"text":942},{},{"id":945,"data":2171,"type":224,"tunes":2172},{"text":947},{},{"id":950,"data":2174,"type":42,"tunes":2175},{"text":952,"level":218},{},{"id":955,"data":2177,"type":224,"tunes":2178},{"text":957},{},{"id":960,"data":2180,"type":224,"tunes":2181},{"text":962},{},{"id":965,"data":2183,"type":224,"tunes":2184},{"text":967},{},{"id":970,"data":2186,"type":224,"tunes":2187},{"text":972},{},{"id":975,"data":2189,"type":224,"tunes":2190},{"text":977},{},{"id":980,"data":2192,"type":42,"tunes":2193},{"text":982,"level":218},{},{"id":985,"data":2195,"type":224,"tunes":2196},{"text":987},{},{"id":990,"data":2198,"type":224,"tunes":2199},{"text":992},{},{"id":995,"data":2201,"type":295,"tunes":2202},{"code":997},{},{"id":1000,"data":2204,"type":224,"tunes":2205},{"text":1002},{},{"id":1005,"data":2207,"type":295,"tunes":2208},{"code":1007},{},{"id":1010,"data":2210,"type":224,"tunes":2211},{"text":1012},{},{"id":1015,"data":2213,"type":224,"tunes":2214},{"text":1017},{},{"id":1020,"data":2216,"type":42,"tunes":2217},{"text":1022,"level":218},{},{"id":1025,"data":2219,"type":224,"tunes":2220},{"text":1027},{},{"id":1030,"data":2222,"type":224,"tunes":2223},{"text":1032},{},{"id":1035,"data":2225,"type":224,"tunes":2226},{"text":1037},{},{"id":1040,"data":2228,"type":224,"tunes":2229},{"text":1042},{},"Post erfolgreich abgerufen",{"items":2232,"source":2315,"manualIds":2316,"manualMatchedIds":2317},[2233,2240,2247,2254,2259,2266,2273,2280,2287,2294,2301,2308],{"id":2234,"slug":2235,"title":2236,"excerpt":2237,"featuredImage":2238,"publishedAt":2239},"469","rag-failed-but-which-layer-actually-failed-a-diagnostic-method","RAG失败了——但究竟是哪一层真正失败了？一种诊断方法","当RAG答案出错时，将问题归咎于检索或模型过于笼统。这种诊断方法将来源覆盖、查询构建、检索、排序、上下文组装、生成、证据归因和时效性逐一隔离，从而使实际故障能够被复现并修复。","\u002Fuploads\u002F2026\u002F09\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method-1790350847177-pior4c.webp","2026-09-24T19:39:00.000Z",{"id":2241,"slug":2242,"title":2243,"excerpt":2244,"featuredImage":2245,"publishedAt":2246},"477","computer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system","计算机使用代理：为什么成功的演示仍可能是一个不可靠的系统","计算机使用代理如今能够完成令人印象深刻的浏览器和桌面工作流程，但一次成功的运行证明的是能力——而非可靠性。本文展示了如何测试可重复性、环境鲁棒性、长时程控制、状态感知、结果验证以及安全的目标处理。","\u002Fuploads\u002F2026\u002F09\u002Fcomputer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system-1790352854690-75qnrg.webp","2026-09-25T12:13:00.000Z",{"id":2248,"slug":2249,"title":2250,"excerpt":2251,"featuredImage":2252,"publishedAt":2253},"476","mcp-vs-a2a-vs-ucp-vs-ap2-vs-a2ui-the-agent-protocol-stack-explained","MCP vs A2A vs UCP vs AP2 vs A2UI：智能体协议栈详解","MCP、A2A、UCP、AP2 和 A2UI 常被描述为相互竞争的智能体标准。它们大多解决的是不同的互操作性问题。本指南将每个协议映射到其实际标准化的边界，并展示它们如何在同一个生产系统中协同工作。","\u002Fuploads\u002F2026\u002F09\u002Fmcp-vs-a2a-vs-ucp-vs-ap2-vs-a2ui-the-agent-protocol-stack-explained-1790352625869-2ezle0.webp","2026-09-25T12:09:00.000Z",{"id":2255,"slug":2256,"title":2256,"excerpt":10,"featuredImage":2257,"publishedAt":2258},"369","git-with-automatic-upload-and-synchronization-to-a-production-server","\u002Fuploads\u002F2024\u002F05\u002Fstep-by-step-guide-illustration-showing-the-process-of-setting-up-Git-with-auto-upload-and-synchronization-to-a-production-server-large.webp","2024-05-28T22:48:00.000Z",{"id":2260,"slug":2261,"title":2262,"excerpt":2263,"featuredImage":2264,"publishedAt":2265},"470","what-should-an-ai-agent-remember-forget-recompute-or-retrieve-again","AI代理应该记住、遗忘、重新计算还是再次检索什么？","长时间运行的代理不应记住所有内容。本文提供了一个实用的生命周期模型，用于决定哪些内容应属于持久记忆、哪些内容应重新检索、哪些内容重新计算更安全，以及哪些内容应过期或被取代。","\u002Fuploads\u002F2026\u002F09\u002Fwhat-should-an-ai-agent-remember-forget-recompute-or-retrieve-again-1790351131087-iehz28.webp","2026-09-25T09:43:00.000Z",{"id":2267,"slug":2268,"title":2269,"excerpt":2270,"featuredImage":2271,"publishedAt":2272},"478","what-is-rag-the-simplest-explanation-of-how-it-works","什么是RAG？对其工作原理的最简单解释","RAG听起来很复杂，但想法很简单：在AI回答之前，它先从知识源查找有用的信息，并将该信息提供给语言模型。本指南使用一个简单的思维模型来解释RAG、LLM、状态、记忆和工具。","\u002Fuploads\u002F2026\u002F09\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works-1790377492124-khjagt.webp","2026-09-25T19:03:00.000Z",{"id":2274,"slug":2275,"title":2276,"excerpt":2277,"featuredImage":2278,"publishedAt":2279},"381","enterprise-grade-multi-tenant-architecture-for-an-international-platform","企业级多租户架构，适用于国际平台","Loving Rocks 是一款企业级婚礼平台，采用真正的多租户架构设计，实现租户间数据库隔离，并内置国际化支持，以确保全球可扩展性、安全性及长期运营稳定性。","\u002Fuploads\u002F2026\u002F01\u002Fenterprise-grade-multi-tenant-architecture-for-an-international-platform-1769789121298-b6v7ak.webp","2026-01-30T12:04:00.000Z",{"id":2281,"slug":2282,"title":2283,"excerpt":2284,"featuredImage":2285,"publishedAt":2286},"460","ai-agent-reliability-why-the-final-answer-is-not-enough","AI Agent可靠性：为什么最终答案并不足够","正确的输出并不能证明推理的正确性、执行的安全性，或系统的可信赖性。","\u002Fuploads\u002F2026\u002F09\u002Fai-agent-reliability-why-the-final-answer-is-not-enough-1788955466306-pl0qhz.webp","2026-09-09T04:01:00.000Z",{"id":2288,"slug":2289,"title":2290,"excerpt":2291,"featuredImage":2292,"publishedAt":2293},"473","openai-agents-api-vs-agents-sdk-vs-responses-api-what-should-you-build-on-in-2026","OpenAI Agents API 与 Agents SDK 与 Responses API：2026 年你应该基于什么来构建？","OpenAI 的智能体技术栈在 2026 年 9 月发生了变化。本架构指南按运行时归属将 Agents API、Agents SDK、Responses API 和 Codex SDK 区分开来——以便团队能够选择正确的控制边界，而不是比较产品名称。","\u002Fuploads\u002F2026\u002F09\u002Fopenai-agents-api-vs-agents-sdk-vs-responses-api-what-should-you-build-on-in-2026-1790351846714-zi7lus.webp","2026-09-25T11:56:00.000Z",{"id":2295,"slug":2296,"title":2297,"excerpt":2298,"featuredImage":2299,"publishedAt":2300},"364","tipps-fuer-die-verbesserung-der-seo-suchmaschinenoptimierung","Mastering the SEO Workflow: Essential Optimization Strategies for Organic Growth","A structured SEO workflow is crucial for sustainable organic growth. Learn the ten foundational strategies, from keyword research and technical optimization to content quality and performance analysis.","\u002Fuploads\u002F2026\u002F03\u002Ftipps-fuer-die-verbesserung-der-seo-suchmaschinenoptimierung-1774866098131-hwkzrg.webp","2024-01-26T06:35:00.000Z",{"id":2302,"slug":2303,"title":2304,"excerpt":2305,"featuredImage":2306,"publishedAt":2307},"468","ai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context","AI代理记忆不是RAG：如何区分记忆、检索、状态和上下文","代理记忆、RAG、状态和上下文经常被当作可以互换的概念来使用。它们并不是。这个实用的架构模型将这四个层次区分开来，展示了每一层各自应处的位置，并解释了当系统将它们合并为一层时会出现什么问题。","\u002Fuploads\u002F2026\u002F09\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context-1790350560308-np0xy6.webp","2026-09-25T11:34:00.000Z",{"id":2309,"slug":2310,"title":2311,"excerpt":2312,"featuredImage":2313,"publishedAt":2314},"472","why-more-context-can-make-ai-answers-worse","为什么更多上下文会让AI的回答更糟","更大的上下文窗口并不保证更好的答案。本文解释了信号稀释、证据冲突、状态过时、位置敏感性和有损压缩如何降低AI可靠性——并介绍了一种实用的上下文压力测试。","\u002Fuploads\u002F2026\u002F09\u002Fwhy-more-context-can-make-ai-answers-worse-1790351615793-2ntv2v.webp","2026-09-25T11:51:00.000Z","fallback",[],[]]