[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"portal-settings:stajic:zh":3,"public-menus:all":38,"post:rag-failed-but-which-layer-actually-failed-a-diagnostic-method:zh":205,"related:post:rag-failed-but-which-layer-actually-failed-a-diagnostic-method:zh:1":1687},{"statusCode":4,"data":5,"message":37},200,{"tenantId":6,"lang":7,"defaultLang":8,"siteUrl":9,"contactEmail":10,"brandName":11,"logoUrl":12,"siteName":11,"siteDescription":13,"ogImage":10,"robotsIndex":14,"socialLinks":10,"reservedSlugs":10,"seoPolicy":15},"stajic","zh","de","https:\u002F\u002Fstajic.de",null,"Stajic Platform","\u002FLogo_Planet.svg","Stajic Portal",true,{"branding":16,"relatedContent":17,"crossDomainLinks":18},{"logoUrl":12},{"enabled":14},[19,22,25,28,31,34],{"url":20,"label":21,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Ffigure.rocks","figure.rocks",{"url":23,"label":24,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Floving.rocks","loving.rocks",{"url":26,"label":27,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.com","bazify.com",{"url":29,"label":30,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.de","bazify.de",{"url":32,"label":33,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.at","bazify.at",{"url":35,"label":36,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.ba","bazify.ba","Portal settings resolved",[39,45],{"id":40,"name":41,"location":42,"isActive":14,"isDefault":43,"items":44},1,"main-navigation","header",false,[],{"id":46,"name":47,"location":48,"isActive":14,"isDefault":14,"items":49},4,"main-menu","sidebar",[50,66,79,93,103,118,133],{"id":51,"title":52,"url":60,"target":61,"icon":62,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":64,"portfolioId":10,"children":65},"item-18",{"de":53,"en":54,"es":55,"fr":56,"it":54,"ru":57,"sr":58,"zh":59},"Startseite","Home","Inicio","Accueil","Главная","Почетна","首页","\u002Ffull-stack-web-developer-munich-performance-seo-and-maintainable-builds","_self","i-lucide-home","page",111,[],{"id":67,"title":68,"url":75,"target":61,"icon":76,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":77,"portfolioId":10,"children":78},"item-22",{"de":69,"en":69,"es":70,"fr":69,"it":71,"ru":72,"sr":73,"zh":74},"Vision","Visión","Visione","Видение","Визија","想象","\u002Fueber-uns-webdesign-muenchen-webaplikation","i-lucide-eye",113,[],{"id":80,"title":81,"url":89,"target":61,"icon":90,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":91,"portfolioId":10,"children":92},"item-19",{"de":82,"en":83,"es":84,"fr":83,"it":85,"ru":86,"sr":87,"zh":88},"Leistungen","Services","Servicios","Servizi","Услуги","Услуге","服务","\u002Fservices-dienstleistungen-muenchen","i-lucide-wrench",116,[],{"id":94,"title":95,"url":99,"target":61,"icon":100,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":101,"portfolioId":10,"children":102},"item-23",{"de":96,"en":96,"es":96,"fr":96,"it":96,"ru":97,"sr":97,"zh":98},"Blog","Блог","博客","\u002Fblog","i-lucide-book-open",112,[],{"id":104,"title":105,"url":114,"target":61,"icon":115,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":116,"portfolioId":10,"children":117},"item-32",{"de":106,"en":107,"es":108,"fr":109,"it":110,"ru":111,"sr":112,"zh":113},"Neue Technologien","New Technologies","Nuevas tecnologías","Nouvelles technologies","Nuove tecnologie","Новые технологии","Нове технологије","新技术！","\u002Fneue-webtechnologien","i-lucide-sparkles",122,[],{"id":119,"title":120,"url":129,"target":61,"icon":130,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":131,"portfolioId":10,"children":132},"item-20",{"de":121,"en":122,"es":123,"fr":124,"it":125,"ru":126,"sr":127,"zh":128},"Kontakt","Contact us!","Contacto","Contact","Contatto","Контакт","Контактирајте нас","联系我们！","\u002Fcontact","i-lucide-mail",115,[],{"id":134,"title":135,"url":144,"target":61,"icon":145,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":147},"item-21",{"de":136,"en":137,"es":138,"fr":139,"it":140,"ru":141,"sr":142,"zh":143},"Unsere Arbeit","Our Work","Nuestro trabajo","Nos réalisations","I nostri lavori","Наши работы","Наши радови","文件夹","\u002Fportfolio","i-lucide-briefcase",114,[148,161,175,181,193],{"id":149,"title":150,"url":144,"target":61,"icon":159,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":160},"item-24",{"de":151,"en":152,"es":153,"fr":154,"it":155,"ru":156,"sr":157,"zh":158},"Alle Projekte","All Projects","Todos los proyectos","Tous les projets","Tutti i progetti","Все проекты","Сви пројекти","所有项目","i-lucide-grid-3x3",[],{"id":162,"title":163,"url":171,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":174},"item-29",{"de":164,"en":165,"es":166,"fr":167,"it":168,"ru":169,"sr":170,"zh":143},"Local Roots, Global Reach","Local Roots - Global Reach","Empresa local ","Entreprise locale","Azienda locale","Местная компания","Локално предузеће глобално тржиште","\u002Fportfolio\u002Flocal-roots-global-reach-communication-media-systems-for-modern-business","i-lucide-folder","custom",[],{"id":176,"title":177,"url":179,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":180},"item-28",{"de":178,"en":178,"es":178,"fr":178,"it":178,"ru":178,"sr":178,"zh":178},"Solr Suggester","\u002Fportfolio\u002Fsolr-fuzzy-suggester-und-solr-infix-suggester-abfrage-ueber-ajax-und-filterung",[],{"id":182,"title":183,"url":191,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":192},"item-27",{"de":184,"en":185,"es":186,"fr":187,"it":188,"ru":189,"sr":190,"zh":185},"Firmenwebseite SEO","Company Website SEO","Sitio web corporativo SEO","Site web d’entreprise SEO","Sito web aziendale SEO","Корпоративный сайт SEO","Пословна веб-страница SEO","\u002Fportfolio\u002Fseo-sem-branding-mobile-webseite-muenchen",[],{"id":194,"title":195,"url":203,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":204},"item-31",{"de":196,"en":197,"es":198,"fr":199,"it":200,"ru":201,"sr":202,"zh":197},"Digitalisierungsportal","Digitalization Portal","Portal de digitalización","Portail de numérisation","Portale di digitalizzazione","Портал цифровизации","Портал за дигитализацију","\u002Fportfolio\u002Fdigitalisierungsportal-archiv-museum-bibliothek-ead-lido-mets-mods",[],{"statusCode":4,"data":206,"message":1686},{"id":207,"title":208,"slug":209,"content":210,"contentJson":211,"excerpt":835,"featuredImage":836,"featuredImageAlt":837,"featuredImageCaption":10,"featuredImageTitle":10,"featuredImageCopyright":10,"featuredImageAuthor":10,"featuredImageSourceUrl":10,"featuredImageLicense":10,"featuredImageIsAiGenerated":43,"status":838,"publishedAt":839,"createdAt":840,"updatedAt":841,"seoLocalePaths":842,"categories":851,"author":864,"translations":869},"469","RAG失败了——但究竟是哪一层真正失败了？一种诊断方法","rag-failed-but-which-layer-actually-failed-a-diagnostic-method","{\"time\":1790369184068,\"blocks\":[{\"id\":\"intro\",\"type\":\"paragraph\",\"data\":{\"text\":\"RAG 系统返回了一个薄弱、错误、不完整或缺乏支持的答案。通常的诊断是“检索失败”或“模型产生了幻觉”。这两个标签都过于宽泛，无法发挥作用。生产环境中的 RAG 流水线可能在检索之前、检索期间、排序期间、组装上下文期间、生成期间，或在生成之后检查证据和有效性时失败。\"},\"tunes\":{}},{\"id\":\"direct\",\"type\":\"callout\",\"data\":{\"variant\":\"info\",\"title\":\"直接回答\",\"body\":\"\u003Cstrong>不要将 RAG 作为一个组件来调试。\u003C\u002Fstrong>应将其视为一条由可独立测试的层组成的链条来诊断。首先确定所需证据是否存在于权威来源中。然后测试查询构建、候选检索、排序、上下文组装、生成、证据归因和时效性。最快的隔离技术是\u003Cstrong>预言机上下文测试\u003C\u002Fstrong>：手动将正确的证据提供给生成器。如果答案变得正确，则主要故障发生在生成之前。如果答案仍然错误，则检索不是主要问题。\"},\"tunes\":{}},{\"id\":\"model-note\",\"type\":\"callout\",\"data\":{\"variant\":\"note\",\"title\":\"关于诊断模型\",\"body\":\"本文中的 RAG 故障栈是一个实用的诊断模型，而非正式的行业标准。现有平台已经将仅检索指标与检索并生成指标分开；该模型将这种分离扩展为逐步的生产调试方法。\"},\"tunes\":{}},{\"id\":\"toc\",\"type\":\"tableOfContents\",\"data\":{\"title\":\"目录\",\"minLevel\":2,\"maxLevel\":3},\"tunes\":{}},{\"id\":\"h-not-diagnosis\",\"type\":\"header\",\"data\":{\"text\":\"为什么“RAG 失败了”不是一种诊断\",\"level\":2},\"tunes\":{}},{\"id\":\"p-not-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"检索增强生成结合了多种机制：用户请求被解释，构建一个或多个搜索，检索候选材料，对结果进行过滤或重新排序，将选定的证据插入模型上下文，模型生成答案。生产系统可能还会添加权限、元数据过滤器、时效性规则、引用、查询重写、混合搜索、工具调用、记忆和外部状态。\"},\"tunes\":{}},{\"id\":\"p-not-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"因此，错误的最终答案并不能告诉你哪个组件失败了。模型可能收到了错误的证据。它可能收到了正确的证据，但混杂了太多噪声。证据可能是正确的，但已经过时。来源可能从未包含答案。或者模型可能忽略了完全足够的上下文。\"},\"tunes\":{}},{\"id\":\"p-not-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"OpenAI 的 RAG 指南已经对检索失败和模型失败做出了根本区分：系统可能提供了错误的上下文，也可能提供了正确的上下文但仍然生成了错误的答案。AWS 同样将仅检索评估与检索并生成评估分开。对于生产诊断，这种区分应该更进一步。\"},\"tunes\":{}},{\"id\":\"h-stack\",\"type\":\"header\",\"data\":{\"text\":\"RAG 故障栈\",\"level\":2},\"tunes\":{}},{\"id\":\"table-stack\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"层\",\"问题\",\"典型故障\"],[\"1. 来源覆盖\",\"所需证据是否存在于允许的权威来源中？\",\"语料库根本无法回答该问题\"],[\"2. 查询构建\",\"系统是否搜索了正确的内容？\",\"意图、实体、过滤器、语言或时间约束丢失\"],[\"3. 候选检索\",\"相关证据是否进入了候选集？\",\"召回率低；正确的块从未被检索到\"],[\"4. 排序与过滤\",\"正确的证据是否保留下来并排名足够高？\",\"相关证据被埋没、过滤掉，或被表面相似的文本挤到后面\"],[\"5. 上下文组装\",\"模型是否收到了可用的证据？\",\"截断、糟糕的块边界、重复、冲突段落或上下文过载\"],[\"6. 生成\",\"模型是否正确使用了提供的证据？\",\"无支持的推断、指令失败、推理错误或拒绝不匹配\"],[\"7. 证据归因\",\"答案能否追溯到其声称使用的证据？\",\"引用缺失、薄弱或不正确；主张超出检索到的支持\"],[\"8. 有效性与时效性\",\"证据对于当前这个问题是否仍然有效？\",\"正确的历史证据在其有效时间、版本、司法管辖区或状态之外被重用\"]]},\"tunes\":{}},{\"id\":\"h-source\",\"type\":\"header\",\"data\":{\"text\":\"第 1 层 — 来源覆盖：系统究竟能否回答这个问题？\",\"level\":3},\"tunes\":{}},{\"id\":\"p-source-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"在调整嵌入、重排序器或提示之前，请验证答案是否存在于系统被允许使用的知识空间中。这听起来显而易见，但许多 RAG 故障实际上是语料库故障。所请求的事实可能缺失、隐藏在未索引的附件中、仅在较新的文档中可用、存储在 RAG 语料库之外的系统中，或被权限阻止。\"},\"tunes\":{}},{\"id\":\"p-source-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"检索指标无法恢复从未被索引的信息。更大的 top-k 无法检索流水线中不包含的文档。如果来源覆盖测试失败，正确的修复方法是摄取、来源选择、权限，或明确的“无法根据可用证据回答”行为。\"},\"tunes\":{}},{\"id\":\"source-warning\",\"type\":\"callout\",\"data\":{\"variant\":\"warning\",\"title\":\"故障模式\",\"body\":\"团队经常针对语料库实际上无法回答的问题来调整检索。这可能会让检索器更擅长找到相关文本，但底层的信息缺口仍未解决。\"},\"tunes\":{}},{\"id\":\"h-query\",\"type\":\"header\",\"data\":{\"text\":\"第 2 层 — 查询构建：系统是否向语料库提出了正确的问题？\",\"level\":3},\"tunes\":{}},{\"id\":\"p-query-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"用户查询并不总是检索查询。生产系统会重写问题、解析代词、提取实体、翻译语言、添加元数据约束、拆分复杂问题或生成多个搜索。每一次转换都可以改善检索，但每一次转换也可能破坏信息。\"},\"tunes\":{}},{\"id\":\"p-query-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"诸如“在九月更新之后，该政策是否仍然适用于德国的承包商？”这样的请求至少包含一个实体、一个人群、一个司法管辖区和一个时间边界。被重写为“承包商政策”的查询可能会检索到语义相关的文本，但会丢失决定答案是否有效的变量。\"},\"tunes\":{}},{\"id\":\"h-retrieval\",\"type\":\"header\",\"data\":{\"text\":\"第3层——候选检索：相关证据是否进入了集合？\",\"level\":3},\"tunes\":{}},{\"id\":\"p-ret-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"候选检索主要是一个召回问题。诊断问题还不是最佳结果是否排名第一；而是相关证据是否出现在候选池中的任何位置。如果已知的正确来源没有出现，请调查索引、分块、嵌入、词汇匹配、元数据、混合搜索、语言处理、同义词和查询扩展。\"},\"tunes\":{}},{\"id\":\"p-ret-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"这就是仅检索评估有价值的地方。AWS为仅检索RAG评估提供了上下文相关性和上下文覆盖率。重要的生产习惯是在生成之前评估检索，这样精致的最终答案就无法掩盖薄弱的候选集。\"},\"tunes\":{}},{\"id\":\"h-ranking\",\"type\":\"header\",\"data\":{\"text\":\"第4层——排序和过滤：正确的证据是否被丢弃或埋没？\",\"level\":3},\"tunes\":{}},{\"id\":\"p-rank-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"一个系统可以有良好的召回率，但仍然失败，因为相关证据的排名低于嘈杂但语义相似的材料。重排序器、新近度提升、权威权重、语言偏好、租户过滤器、访问控制、产品状态过滤器和去重都会改变最终上下文中保留的内容。\"},\"tunes\":{}},{\"id\":\"p-rank-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"因此，调试应保留完整的候选列表，而不仅仅是最终的top-k。如果黄金证据在第18位被检索到，而重排序器将其移除，那么修复方法与检索遗漏不同。\"},\"tunes\":{}},{\"id\":\"h-context\",\"type\":\"header\",\"data\":{\"text\":\"第5层——上下文组装：有用的证据是否变成了可用的上下文？\",\"level\":3},\"tunes\":{}},{\"id\":\"p-ctx-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"检索成功并不能保证上下文成功。相关的块可能被截断、与其限定词分离、重复直到主导提示、与矛盾版本混合，或被足够的无关文本包围，以至于决定性段落失去显著性。\"},\"tunes\":{}},{\"id\":\"p-ctx-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"块边界尤其重要。一个句子可能包含规则，而下一个句子包含例外。如果它们被单独索引，并且只检索到第一个，检索器可能看起来相关，而组装的上下文变得具有误导性。\"},\"tunes\":{}},{\"id\":\"h-generation\",\"type\":\"header\",\"data\":{\"text\":\"第6层——生成：模型能否正确使用正确的证据？\",\"level\":3},\"tunes\":{}},{\"id\":\"p-gen-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"一旦系统明显提供了足够的证据，生成就可以独立测试。模型可能过度概括、组合不兼容的段落、忽略否定陈述、未能遵循请求的答案格式、在事实之间发明桥梁，或从参数记忆而不是检索到的证据中回答。\"},\"tunes\":{}},{\"id\":\"p-gen-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"这就是为什么仅端到端正确性不足以进行诊断。OpenAI建议将评估作为理解应用程序行为的一种结构化方式，而Anthropic的代理评估指南强调多次试验、评分器、跟踪和现实的失败案例。对于RAG，生成器应在正常检索和受控黄金上下文下进行测试。\"},\"tunes\":{}},{\"id\":\"h-evidence\",\"type\":\"header\",\"data\":{\"text\":\"第7层——证据归因：答案是否真正得到支持？\",\"level\":3},\"tunes\":{}},{\"id\":\"p-evidence-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"一个带有引用的看似合理的答案仍然可能基础薄弱。引用的文档可能与主题相关，但不支持具体主张。一个句子可能得到支持，而另一个是推断的。引用可能指向一个来源，一旦阅读其条件，就会与答案相矛盾。\"},\"tunes\":{}},{\"id\":\"p-evidence-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"因此，引用评估属于生成之后。AWS区分引用精确度和引用覆盖率：引用的段落是否被正确引用，以及答案是否得到引用的充分支持。在生产中，主张级别的支持比将任何引用的存在视为证据质量更有用。\"},\"tunes\":{}},{\"id\":\"h-validity\",\"type\":\"header\",\"data\":{\"text\":\"第8层——有效性和新鲜度：证据对于这个现实版本是否正确？\",\"level\":3},\"tunes\":{}},{\"id\":\"p-valid-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"RAG可以检索到一个完全真实、高度相关、忠实引用的来源，但如果该来源对当前问题不再有效，仍然会产生错误答案。政策会变化。API被弃用。价格变动。软件行为在不同版本之间变化。产品库存变化。权限变化。游戏补丁改变机制。\"},\"tunes\":{}},{\"id\":\"p-valid-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"这是一个与幻觉不同的失败类别。证据是真实的；其适用性是错误的。因此，一个稳健的系统需要时间戳、相关时的版本或司法管辖区元数据、来源权威、取代规则，以及一个明确的机制来决定何时必须限制或放弃旧证据。\"},\"tunes\":{}},{\"id\":\"h-oracle\",\"type\":\"header\",\"data\":{\"text\":\"最快的隔离方法：oracle上下文测试\",\"level\":2},\"tunes\":{}},{\"id\":\"p-oracle-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"最有用的第一步拆分很简单：手动向生成器提供一小部分你知道足以回答问题的证据。保持任务和预期答案不变。\"},\"tunes\":{}},{\"id\":\"oracle-comparison\",\"type\":\"comparison\",\"data\":{\"title\":\"Oracle上下文测试\",\"layout\":\"table\",\"columns\":[{\"id\":\"result\",\"label\":\"结果\"},{\"id\":\"meaning\",\"label\":\"可能的解释\"},{\"id\":\"next\",\"label\":\"下一步诊断\"}],\"rows\":[{\"id\":\"oracle-pass\",\"label\":\"答案变得正确\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"oracle-fail\",\"label\":\"答案仍然错误\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"oracle-partial\",\"label\":\"答案改善但仍不完整\",\"values\":[\"\",\"\",\"\"]}]},\"tunes\":{}},{\"id\":\"oracle-tip\",\"type\":\"callout\",\"data\":{\"variant\":\"tip\",\"title\":\"为什么这个测试很强大\",\"body\":\"oracle上下文测试从实验中移除了大部分检索流程。它不能证明生成是完美的，但它提供了一个快速的反事实：\u003Cstrong>如果检索已经成功，模型会怎么做？\u003C\u002Fstrong>\"},\"tunes\":{}},{\"id\":\"h-sequence\",\"type\":\"header\",\"data\":{\"text\":\"生产诊断序列\",\"level\":2},\"tunes\":{}},{\"id\":\"diag-flow\",\"type\":\"processFlow\",\"data\":{\"title\":\"从证据到答案诊断故障\",\"orientation\":\"auto\",\"steps\":[{\"label\":\"1. 定义预期声明\",\"description\":\"写下预期答案、允许的不确定性以及证明其合理的证据。\"},{\"label\":\"2. 验证来源覆盖\",\"description\":\"确认权威且允许的证据存在于已索引或可访问的来源集中。\"},{\"label\":\"3. 运行oracle上下文测试\",\"description\":\"直接向生成器提供足够的黄金证据，观察答案是否变得正确。\"},{\"label\":\"4. 检查检索查询\",\"description\":\"检查重写、实体、过滤器、语言、时间约束、分解和隐藏假设。\"},{\"label\":\"5. 在重排序前检查候选\",\"description\":\"确定相关证据是否被检索到，并记录其排名。\"},{\"label\":\"6. 检查排序和上下文组装\",\"description\":\"检查重排序、元数据过滤、截断、块边界、重复、冲突和top-k组成。\"},{\"label\":\"7. 分别评估生成和引用\",\"description\":\"衡量答案正确性、完整性、忠实性和声明级证据支持。\"},{\"label\":\"8. 测试有效性边界\",\"description\":\"检查版本、日期、状态、司法管辖区、权限或替代证据是否会改变答案。\"}]},\"tunes\":{}},{\"id\":\"h-one-change\",\"type\":\"header\",\"data\":{\"text\":\"不要同时改变三层\",\"level\":2},\"tunes\":{}},{\"id\":\"p-one-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"一个常见的调试错误是在一次迭代中改变嵌入、块大小、top-k、提示和模型。如果分数提高，你不知道原因。如果变差，你不知道哪个改变导致了回归。\"},\"tunes\":{}},{\"id\":\"p-one-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"将RAG调试视为实验诊断：尽可能保持流程不变，用一个受控输入替换一个不确定的组件。黄金文档隔离检索。黄金块隔离块选择。固定上下文隔离生成。固定模型隔离检索变化。固定语料库隔离摄取和索引变化。\"},\"tunes\":{}},{\"id\":\"h-matrix\",\"type\":\"header\",\"data\":{\"text\":\"常见RAG症状的故障矩阵\",\"level\":2},\"tunes\":{}},{\"id\":\"symptom-matrix\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"症状\",\"最可能首先测试的层\",\"区分性测试\"],[\"没有相关来源出现\",\"来源覆盖 → 查询 → 候选检索\",\"手动搜索语料库，然后检查重写查询和未过滤的候选\"],[\"相关来源出现但答案错误\",\"上下文组装 → 生成\",\"使用相同来源缩减为决定性段落的oracle上下文测试\"],[\"答案有时正确，有时错误\",\"排序 → 上下文组装 → 生成变异性\",\"重复试验，同时记录检索集、排名、提示上下文和模型输出\"],[\"答案引用了正确的文档但夸大了它\",\"生成 → 证据归因 → 有效性\",\"根据确切引用的段落评估每个声明\"],[\"旧信息总是胜出\",\"排序 → 有效性\u002F新鲜度\",\"与最近性\u002F替代规则比较并检查元数据\"],[\"答案遗漏了例外\",\"分块 → 上下文组装\",\"检查规则和例外是否被拆分或截断\"],[\"增加更多top-k使质量变差\",\"排序 → 上下文过载\",\"消减低价值块并与最小证据集比较\"],[\"更换模型修复了答案\",\"生成，但不一定是检索\",\"在模型间使用相同的检索上下文重复\"],[\"更换嵌入修复了答案\",\"检索\u002F排序\",\"保持生成器和上下文模板不变，同时比较候选召回率\"]]},\"tunes\":{}},{\"id\":\"h-metrics\",\"type\":\"header\",\"data\":{\"text\":\"用每层实际能影响的指标来衡量\",\"level\":2},\"tunes\":{}},{\"id\":\"metrics-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"层\",\"有用的测量\",\"不应推断什么\"],[\"来源覆盖\",\"可回答问题率、语料库覆盖、摄取完整性\",\"不要因为缺少源材料而责怪嵌入\"],[\"候选检索\",\"Recall@k、命中率、上下文覆盖\",\"高召回率不能证明排序质量\"],[\"排序\",\"MRR、NDCG、黄金排名、precision@k\",\"好的排序不能证明生成器使用了证据\"],[\"上下文组装\",\"证据保留、重复、矛盾率、令牌利用率\",\"大上下文并不意味着有用的上下文\"],[\"生成\",\"正确性、完整性、任务成功、忠实性\",\"仅正确性不能证明有依据\"],[\"证据归因\",\"引用精确度、引用覆盖、声明支持\",\"引用数量不是证据质量\"],[\"有效性\",\"新鲜度、替代准确性、版本\u002F司法管辖区匹配\",\"相关证据不一定是适用证据\"]]},\"tunes\":{}},{\"id\":\"h-correct-answer\",\"type\":\"header\",\"data\":{\"text\":\"正确答案仍可能隐藏RAG缺陷\",\"level\":2},\"tunes\":{}},{\"id\":\"p-correct-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"反向问题也很重要。RAG系统可以在检索损坏的情况下产生正确答案。模型可能已经从训练中知道答案，从弱证据中推断出来，或者猜对了。如果评估只看最终答案，系统可能看起来健康，直到问题涉及仅存在于私有语料库中的信息。\"},\"tunes\":{}},{\"id\":\"p-correct-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"这与更广泛的代理系统中出现的可靠性问题相同：结果正确性不足以证明执行路径是可靠的。对于RAG，跟踪应至少保留检索查询、候选集、排序、最终上下文、答案、引用、模型版本、语料库\u002F索引版本和相关过滤器。\"},\"tunes\":{}},{\"id\":\"internal-reliability\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough\",\"title\":\"AI代理可靠性：为什么最终答案不够\",\"excerpt\":\"正确的输出不能证明正确的推理、安全的执行或可信赖的系统。本文将这一原则从RAG诊断扩展到代理轨迹和运营保证。\",\"ctaLabel\":\"阅读相关文章\"},\"tunes\":{}},{\"id\":\"h-hypotheses\",\"type\":\"header\",\"data\":{\"text\":\"使用竞争假设，而不是最喜欢的解释\",\"level\":2},\"tunes\":{}},{\"id\":\"p-hyp-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"如果一个糟糕的回答立刻被归为“嵌入问题”，那么调查就已经带有偏见了。更强大的调试方法是在改变系统之前写下相互竞争的假设：缺失来源、查询重写不佳、检索召回率低、重排序不佳、上下文截断、版本冲突、生成失败、引用失败或证据过时。\"},\"tunes\":{}},{\"id\":\"p-hyp-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"然后选择一个能够区分这些假设的测试。这比收集更多支持第一种解释的例子更有效率。同样的原则也适用于一般的人工智能辅助技术推理：有用的诊断是能够通过区分性测试的诊断，而不仅仅是听起来合理的诊断。\"},\"tunes\":{}},{\"id\":\"internal-reasoning\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Ffrom-research-protocol-to-a-general-ai-reasoning-framework\",\"title\":\"从研究协议到通用人工智能推理框架\",\"excerpt\":\"一种领域无关的推理方法，用于将证据与假设分开、测试相互竞争的假设并使用领域特定的验证器。\",\"ctaLabel\":\"阅读推理框架\"},\"tunes\":{}},{\"id\":\"h-change\",\"type\":\"header\",\"data\":{\"text\":\"什么会改变这个答案？\",\"level\":2},\"tunes\":{}},{\"id\":\"p-change-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"确切的诊断层会随架构而变化。一个简单的单文档RAG应用可能没有查询重写、重排序器或引用层。一个代理式检索系统可能会增加规划、多次搜索、工具选择、记忆、权限和迭代证据收集。结构化数据库查找可能根本不使用块或嵌入。\"},\"tunes\":{}},{\"id\":\"p-change-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"核心方法仍然成立：识别能够独立改变结果的组件，构建受控测试，用已知良好的输入替换不确定的组件，并使用适合该层的证据来测量每个组件。\"},\"tunes\":{}},{\"id\":\"h-limitations\",\"type\":\"header\",\"data\":{\"text\":\"局限性\",\"level\":2},\"tunes\":{}},{\"id\":\"p-limit-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"真实的失败往往是耦合的。一个弱的查询会降低召回率，进而改变重排序，进而改变上下文，进而增加生成方差。预言机上下文测试是一种诊断捷径，并不能证明某个组件是唯一原因。评估数据集也可能不具有代表性，而基于模型的评分器可能引入自身的错误。\"},\"tunes\":{}},{\"id\":\"p-limit-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"因此，所提出的堆栈最好用作调查结构：记录管道、隔离变量、重现失败、测试相互竞争的解释，并在层级别修复后进行端到端评估。\"},\"tunes\":{}},{\"id\":\"h-conclusion\",\"type\":\"header\",\"data\":{\"text\":\"结论\",\"level\":2},\"tunes\":{}},{\"id\":\"p-conclusion-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"“RAG失败了”应该是调查的开始，而不是结论。有用的诊断能够识别系统是缺乏证据、搜索不正确、未能检索到、排序不佳、组装了不可用的上下文、生成不正确、归因不当，还是应用了超出其有效性边界的证据。\"},\"tunes\":{}},{\"id\":\"p-conclusion-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"实用规则很简单：一次一层地用受控证据替换不确定性。从预言机上下文测试开始。将仅检索评估与生成评估分开。保留完整跟踪。然后修复实际失败的组件，而不是凭直觉调整整个RAG堆栈。\"},\"tunes\":{}},{\"id\":\"h-faq\",\"type\":\"header\",\"data\":{\"text\":\"常见问题\",\"level\":2},\"tunes\":{}},{\"id\":\"faq\",\"type\":\"faq\",\"data\":{\"title\":\"RAG故障诊断\",\"items\":[{\"id\":\"faq1\",\"question\":\"如何判断是RAG检索失败还是LLM失败？\",\"answer\":\"手动给模型一小部分已知正确的证据。如果答案变得正确，则调查来源覆盖、查询构建、检索、排序和上下文组装。如果模型在证据充足的情况下仍然失败，则检索不是主要问题。\"},{\"id\":\"faq2\",\"question\":\"即使检索到了正确的文档，RAG也会失败吗？\",\"answer\":\"是的。相关段落可能排名太低、被截断、与例外分离、与冲突证据混合、被无关上下文淹没，或者被生成器错误使用。\"},{\"id\":\"faq3\",\"question\":\"答案正确性足以评估RAG系统吗？\",\"answer\":\"不。模型可能依赖先验知识或运气，在检索薄弱的情况下产生正确答案。应将检索和证据支持与最终答案正确性分开评估。\"},{\"id\":\"faq4\",\"question\":\"调试RAG时应该记录什么？\",\"answer\":\"至少记录用户请求、转换后的检索查询、过滤器、候选文档及排名、最终选择的上下文、模型和提示版本、答案、引用、语料库\u002F索引版本，以及与新鲜度相关的时间或版本元数据。\"},{\"id\":\"faq5\",\"question\":\"增加top-k通常能修复RAG吗？\",\"answer\":\"不一定可靠。更大的候选或上下文集可能会提高召回率，但也可能增加噪声、矛盾、重复和上下文过载。在增加top-k之前，先测试相关证据是否缺失。\"}]},\"tunes\":{}},{\"id\":\"h-glossary\",\"type\":\"header\",\"data\":{\"text\":\"术语表\",\"level\":2},\"tunes\":{}},{\"id\":\"glossary\",\"type\":\"glossary\",\"data\":{\"title\":\"关键诊断术语\",\"entries\":[{\"term\":\"预言机上下文测试\",\"definition\":\"一种受控测试，直接向生成器提供已知充分的证据，以确定主要失败是否发生在生成之前。\",\"anchor\":\"oracle-context-test\"},{\"term\":\"候选检索\",\"definition\":\"在最终排序或上下文组装之前，选择一组初始潜在相关文档、块、记录或段落的阶段。\",\"anchor\":\"candidate-retrieval\"},{\"term\":\"上下文组装\",\"definition\":\"将检索到的证据转换为实际模型输入的过程，包括排序、截断、去重、格式化和令牌预算决策。\",\"anchor\":\"context-assembly\"},{\"term\":\"忠实度\",\"definition\":\"生成的声明在多大程度上仍由检索到或提供的证据支持，而不是引入无支持的内容。\",\"anchor\":\"faithfulness\"},{\"term\":\"上下文覆盖\",\"definition\":\"一种面向检索的度量，衡量所选证据是否覆盖回答问题所需的信息。\",\"anchor\":\"context-coverage\"},{\"term\":\"有效性边界\",\"definition\":\"声明或答案仍然适用的条件，例如时间、版本、司法管辖区、状态、人群、权限或来源假设。\",\"anchor\":\"validity-boundary\"}]},\"tunes\":{}},{\"id\":\"h-sources\",\"type\":\"header\",\"data\":{\"text\":\"主要来源和进一步阅读\",\"level\":2},\"tunes\":{}},{\"id\":\"src-openai-rag\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Foptimizing-llm-accuracy\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — 优化LLM准确性\",\"description\":\"OpenAI指南，区分RAG应用中的检索失败和LLM失败。\"}},\"tunes\":{}},{\"id\":\"src-openai-evals\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fevaluation-best-practices\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — 评估最佳实践\",\"description\":\"关于可变AI系统结构化评估和生产导向测试设计的指导。\"}},\"tunes\":{}},{\"id\":\"src-aws-rag\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdocs.aws.amazon.com\u002Fbedrock\u002Flatest\u002Fuserguide\u002Fknowledge-base-evaluation-metrics.html\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Amazon Bedrock — RAG评估指标\",\"description\":\"文档区分了仅检索指标和检索并生成指标，包括上下文相关性、覆盖率、忠实度和引用度量。\"}},\"tunes\":{}},{\"id\":\"src-anthropic-evals\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Fdemystifying-evals-for-ai-agents\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Anthropic — 揭秘AI代理的评估\",\"description\":\"关于任务、试验、评分器、追踪、回归和生产行为的实用评估指导。\"}},\"tunes\":{}},{\"id\":\"src-google-rag\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fcloud.google.com\u002Fuse-cases\u002Fretrieval-augmented-generation\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Google Cloud — 检索增强生成\",\"description\":\"RAG架构概述以及相关检索和基于生成的重要性的概述。\"}},\"tunes\":{}}],\"version\":\"2.31.6\"}",{"time":212,"blocks":213,"version":834},1790369184068,[214,220,228,235,243,248,253,258,263,268,310,315,320,325,332,337,342,347,352,357,362,367,372,377,382,387,392,397,402,407,412,417,422,427,432,437,442,447,477,484,489,521,526,531,536,541,586,591,627,632,637,642,651,656,661,666,674,679,684,689,694,699,704,709,714,719,724,750,755,783,788,798,807,816,825],{"id":215,"data":216,"type":218,"tunes":219},"intro",{"text":217},"RAG 系统返回了一个薄弱、错误、不完整或缺乏支持的答案。通常的诊断是“检索失败”或“模型产生了幻觉”。这两个标签都过于宽泛，无法发挥作用。生产环境中的 RAG 流水线可能在检索之前、检索期间、排序期间、组装上下文期间、生成期间，或在生成之后检查证据和有效性时失败。","paragraph",{},{"id":221,"data":222,"type":226,"tunes":227},"direct",{"body":223,"title":224,"variant":225},"\u003Cstrong>不要将 RAG 作为一个组件来调试。\u003C\u002Fstrong>应将其视为一条由可独立测试的层组成的链条来诊断。首先确定所需证据是否存在于权威来源中。然后测试查询构建、候选检索、排序、上下文组装、生成、证据归因和时效性。最快的隔离技术是\u003Cstrong>预言机上下文测试\u003C\u002Fstrong>：手动将正确的证据提供给生成器。如果答案变得正确，则主要故障发生在生成之前。如果答案仍然错误，则检索不是主要问题。","直接回答","info","callout",{},{"id":229,"data":230,"type":226,"tunes":234},"model-note",{"body":231,"title":232,"variant":233},"本文中的 RAG 故障栈是一个实用的诊断模型，而非正式的行业标准。现有平台已经将仅检索指标与检索并生成指标分开；该模型将这种分离扩展为逐步的生产调试方法。","关于诊断模型","note",{},{"id":236,"data":237,"type":241,"tunes":242},"toc",{"title":238,"maxLevel":239,"minLevel":240},"目录",3,2,"tableOfContents",{},{"id":244,"data":245,"type":42,"tunes":247},"h-not-diagnosis",{"text":246,"level":240},"为什么“RAG 失败了”不是一种诊断",{},{"id":249,"data":250,"type":218,"tunes":252},"p-not-1",{"text":251},"检索增强生成结合了多种机制：用户请求被解释，构建一个或多个搜索，检索候选材料，对结果进行过滤或重新排序，将选定的证据插入模型上下文，模型生成答案。生产系统可能还会添加权限、元数据过滤器、时效性规则、引用、查询重写、混合搜索、工具调用、记忆和外部状态。",{},{"id":254,"data":255,"type":218,"tunes":257},"p-not-2",{"text":256},"因此，错误的最终答案并不能告诉你哪个组件失败了。模型可能收到了错误的证据。它可能收到了正确的证据，但混杂了太多噪声。证据可能是正确的，但已经过时。来源可能从未包含答案。或者模型可能忽略了完全足够的上下文。",{},{"id":259,"data":260,"type":218,"tunes":262},"p-not-3",{"text":261},"OpenAI 的 RAG 指南已经对检索失败和模型失败做出了根本区分：系统可能提供了错误的上下文，也可能提供了正确的上下文但仍然生成了错误的答案。AWS 同样将仅检索评估与检索并生成评估分开。对于生产诊断，这种区分应该更进一步。",{},{"id":264,"data":265,"type":42,"tunes":267},"h-stack",{"text":266,"level":240},"RAG 故障栈",{},{"id":269,"data":270,"type":308,"tunes":309},"table-stack",{"content":271,"stretched":43,"withHeadings":14},[272,276,280,284,288,292,296,300,304],[273,274,275],"层","问题","典型故障",[277,278,279],"1. 来源覆盖","所需证据是否存在于允许的权威来源中？","语料库根本无法回答该问题",[281,282,283],"2. 查询构建","系统是否搜索了正确的内容？","意图、实体、过滤器、语言或时间约束丢失",[285,286,287],"3. 候选检索","相关证据是否进入了候选集？","召回率低；正确的块从未被检索到",[289,290,291],"4. 排序与过滤","正确的证据是否保留下来并排名足够高？","相关证据被埋没、过滤掉，或被表面相似的文本挤到后面",[293,294,295],"5. 上下文组装","模型是否收到了可用的证据？","截断、糟糕的块边界、重复、冲突段落或上下文过载",[297,298,299],"6. 生成","模型是否正确使用了提供的证据？","无支持的推断、指令失败、推理错误或拒绝不匹配",[301,302,303],"7. 证据归因","答案能否追溯到其声称使用的证据？","引用缺失、薄弱或不正确；主张超出检索到的支持",[305,306,307],"8. 有效性与时效性","证据对于当前这个问题是否仍然有效？","正确的历史证据在其有效时间、版本、司法管辖区或状态之外被重用","table",{},{"id":311,"data":312,"type":42,"tunes":314},"h-source",{"text":313,"level":239},"第 1 层 — 来源覆盖：系统究竟能否回答这个问题？",{},{"id":316,"data":317,"type":218,"tunes":319},"p-source-1",{"text":318},"在调整嵌入、重排序器或提示之前，请验证答案是否存在于系统被允许使用的知识空间中。这听起来显而易见，但许多 RAG 故障实际上是语料库故障。所请求的事实可能缺失、隐藏在未索引的附件中、仅在较新的文档中可用、存储在 RAG 语料库之外的系统中，或被权限阻止。",{},{"id":321,"data":322,"type":218,"tunes":324},"p-source-2",{"text":323},"检索指标无法恢复从未被索引的信息。更大的 top-k 无法检索流水线中不包含的文档。如果来源覆盖测试失败，正确的修复方法是摄取、来源选择、权限，或明确的“无法根据可用证据回答”行为。",{},{"id":326,"data":327,"type":226,"tunes":331},"source-warning",{"body":328,"title":329,"variant":330},"团队经常针对语料库实际上无法回答的问题来调整检索。这可能会让检索器更擅长找到相关文本，但底层的信息缺口仍未解决。","故障模式","warning",{},{"id":333,"data":334,"type":42,"tunes":336},"h-query",{"text":335,"level":239},"第 2 层 — 查询构建：系统是否向语料库提出了正确的问题？",{},{"id":338,"data":339,"type":218,"tunes":341},"p-query-1",{"text":340},"用户查询并不总是检索查询。生产系统会重写问题、解析代词、提取实体、翻译语言、添加元数据约束、拆分复杂问题或生成多个搜索。每一次转换都可以改善检索，但每一次转换也可能破坏信息。",{},{"id":343,"data":344,"type":218,"tunes":346},"p-query-2",{"text":345},"诸如“在九月更新之后，该政策是否仍然适用于德国的承包商？”这样的请求至少包含一个实体、一个人群、一个司法管辖区和一个时间边界。被重写为“承包商政策”的查询可能会检索到语义相关的文本，但会丢失决定答案是否有效的变量。",{},{"id":348,"data":349,"type":42,"tunes":351},"h-retrieval",{"text":350,"level":239},"第3层——候选检索：相关证据是否进入了集合？",{},{"id":353,"data":354,"type":218,"tunes":356},"p-ret-1",{"text":355},"候选检索主要是一个召回问题。诊断问题还不是最佳结果是否排名第一；而是相关证据是否出现在候选池中的任何位置。如果已知的正确来源没有出现，请调查索引、分块、嵌入、词汇匹配、元数据、混合搜索、语言处理、同义词和查询扩展。",{},{"id":358,"data":359,"type":218,"tunes":361},"p-ret-2",{"text":360},"这就是仅检索评估有价值的地方。AWS为仅检索RAG评估提供了上下文相关性和上下文覆盖率。重要的生产习惯是在生成之前评估检索，这样精致的最终答案就无法掩盖薄弱的候选集。",{},{"id":363,"data":364,"type":42,"tunes":366},"h-ranking",{"text":365,"level":239},"第4层——排序和过滤：正确的证据是否被丢弃或埋没？",{},{"id":368,"data":369,"type":218,"tunes":371},"p-rank-1",{"text":370},"一个系统可以有良好的召回率，但仍然失败，因为相关证据的排名低于嘈杂但语义相似的材料。重排序器、新近度提升、权威权重、语言偏好、租户过滤器、访问控制、产品状态过滤器和去重都会改变最终上下文中保留的内容。",{},{"id":373,"data":374,"type":218,"tunes":376},"p-rank-2",{"text":375},"因此，调试应保留完整的候选列表，而不仅仅是最终的top-k。如果黄金证据在第18位被检索到，而重排序器将其移除，那么修复方法与检索遗漏不同。",{},{"id":378,"data":379,"type":42,"tunes":381},"h-context",{"text":380,"level":239},"第5层——上下文组装：有用的证据是否变成了可用的上下文？",{},{"id":383,"data":384,"type":218,"tunes":386},"p-ctx-1",{"text":385},"检索成功并不能保证上下文成功。相关的块可能被截断、与其限定词分离、重复直到主导提示、与矛盾版本混合，或被足够的无关文本包围，以至于决定性段落失去显著性。",{},{"id":388,"data":389,"type":218,"tunes":391},"p-ctx-2",{"text":390},"块边界尤其重要。一个句子可能包含规则，而下一个句子包含例外。如果它们被单独索引，并且只检索到第一个，检索器可能看起来相关，而组装的上下文变得具有误导性。",{},{"id":393,"data":394,"type":42,"tunes":396},"h-generation",{"text":395,"level":239},"第6层——生成：模型能否正确使用正确的证据？",{},{"id":398,"data":399,"type":218,"tunes":401},"p-gen-1",{"text":400},"一旦系统明显提供了足够的证据，生成就可以独立测试。模型可能过度概括、组合不兼容的段落、忽略否定陈述、未能遵循请求的答案格式、在事实之间发明桥梁，或从参数记忆而不是检索到的证据中回答。",{},{"id":403,"data":404,"type":218,"tunes":406},"p-gen-2",{"text":405},"这就是为什么仅端到端正确性不足以进行诊断。OpenAI建议将评估作为理解应用程序行为的一种结构化方式，而Anthropic的代理评估指南强调多次试验、评分器、跟踪和现实的失败案例。对于RAG，生成器应在正常检索和受控黄金上下文下进行测试。",{},{"id":408,"data":409,"type":42,"tunes":411},"h-evidence",{"text":410,"level":239},"第7层——证据归因：答案是否真正得到支持？",{},{"id":413,"data":414,"type":218,"tunes":416},"p-evidence-1",{"text":415},"一个带有引用的看似合理的答案仍然可能基础薄弱。引用的文档可能与主题相关，但不支持具体主张。一个句子可能得到支持，而另一个是推断的。引用可能指向一个来源，一旦阅读其条件，就会与答案相矛盾。",{},{"id":418,"data":419,"type":218,"tunes":421},"p-evidence-2",{"text":420},"因此，引用评估属于生成之后。AWS区分引用精确度和引用覆盖率：引用的段落是否被正确引用，以及答案是否得到引用的充分支持。在生产中，主张级别的支持比将任何引用的存在视为证据质量更有用。",{},{"id":423,"data":424,"type":42,"tunes":426},"h-validity",{"text":425,"level":239},"第8层——有效性和新鲜度：证据对于这个现实版本是否正确？",{},{"id":428,"data":429,"type":218,"tunes":431},"p-valid-1",{"text":430},"RAG可以检索到一个完全真实、高度相关、忠实引用的来源，但如果该来源对当前问题不再有效，仍然会产生错误答案。政策会变化。API被弃用。价格变动。软件行为在不同版本之间变化。产品库存变化。权限变化。游戏补丁改变机制。",{},{"id":433,"data":434,"type":218,"tunes":436},"p-valid-2",{"text":435},"这是一个与幻觉不同的失败类别。证据是真实的；其适用性是错误的。因此，一个稳健的系统需要时间戳、相关时的版本或司法管辖区元数据、来源权威、取代规则，以及一个明确的机制来决定何时必须限制或放弃旧证据。",{},{"id":438,"data":439,"type":42,"tunes":441},"h-oracle",{"text":440,"level":240},"最快的隔离方法：oracle上下文测试",{},{"id":443,"data":444,"type":218,"tunes":446},"p-oracle-1",{"text":445},"最有用的第一步拆分很简单：手动向生成器提供一小部分你知道足以回答问题的证据。保持任务和预期答案不变。",{},{"id":448,"data":449,"type":475,"tunes":476},"oracle-comparison",{"rows":450,"title":464,"layout":308,"columns":465},[451,456,460],{"id":452,"label":453,"values":454},"oracle-pass","答案变得正确",[455,455,455],"",{"id":457,"label":458,"values":459},"oracle-fail","答案仍然错误",[455,455,455],{"id":461,"label":462,"values":463},"oracle-partial","答案改善但仍不完整",[455,455,455],"Oracle上下文测试",[466,469,472],{"id":467,"label":468},"result","结果",{"id":470,"label":471},"meaning","可能的解释",{"id":473,"label":474},"next","下一步诊断","comparison",{},{"id":478,"data":479,"type":226,"tunes":483},"oracle-tip",{"body":480,"title":481,"variant":482},"oracle上下文测试从实验中移除了大部分检索流程。它不能证明生成是完美的，但它提供了一个快速的反事实：\u003Cstrong>如果检索已经成功，模型会怎么做？\u003C\u002Fstrong>","为什么这个测试很强大","tip",{},{"id":485,"data":486,"type":42,"tunes":488},"h-sequence",{"text":487,"level":240},"生产诊断序列",{},{"id":490,"data":491,"type":519,"tunes":520},"diag-flow",{"steps":492,"title":517,"orientation":518},[493,496,499,502,505,508,511,514],{"label":494,"description":495},"1. 定义预期声明","写下预期答案、允许的不确定性以及证明其合理的证据。",{"label":497,"description":498},"2. 验证来源覆盖","确认权威且允许的证据存在于已索引或可访问的来源集中。",{"label":500,"description":501},"3. 运行oracle上下文测试","直接向生成器提供足够的黄金证据，观察答案是否变得正确。",{"label":503,"description":504},"4. 检查检索查询","检查重写、实体、过滤器、语言、时间约束、分解和隐藏假设。",{"label":506,"description":507},"5. 在重排序前检查候选","确定相关证据是否被检索到，并记录其排名。",{"label":509,"description":510},"6. 检查排序和上下文组装","检查重排序、元数据过滤、截断、块边界、重复、冲突和top-k组成。",{"label":512,"description":513},"7. 分别评估生成和引用","衡量答案正确性、完整性、忠实性和声明级证据支持。",{"label":515,"description":516},"8. 测试有效性边界","检查版本、日期、状态、司法管辖区、权限或替代证据是否会改变答案。","从证据到答案诊断故障","auto","processFlow",{},{"id":522,"data":523,"type":42,"tunes":525},"h-one-change",{"text":524,"level":240},"不要同时改变三层",{},{"id":527,"data":528,"type":218,"tunes":530},"p-one-1",{"text":529},"一个常见的调试错误是在一次迭代中改变嵌入、块大小、top-k、提示和模型。如果分数提高，你不知道原因。如果变差，你不知道哪个改变导致了回归。",{},{"id":532,"data":533,"type":218,"tunes":535},"p-one-2",{"text":534},"将RAG调试视为实验诊断：尽可能保持流程不变，用一个受控输入替换一个不确定的组件。黄金文档隔离检索。黄金块隔离块选择。固定上下文隔离生成。固定模型隔离检索变化。固定语料库隔离摄取和索引变化。",{},{"id":537,"data":538,"type":42,"tunes":540},"h-matrix",{"text":539,"level":240},"常见RAG症状的故障矩阵",{},{"id":542,"data":543,"type":308,"tunes":585},"symptom-matrix",{"content":544,"stretched":43,"withHeadings":14},[545,549,553,557,561,565,569,573,577,581],[546,547,548],"症状","最可能首先测试的层","区分性测试",[550,551,552],"没有相关来源出现","来源覆盖 → 查询 → 候选检索","手动搜索语料库，然后检查重写查询和未过滤的候选",[554,555,556],"相关来源出现但答案错误","上下文组装 → 生成","使用相同来源缩减为决定性段落的oracle上下文测试",[558,559,560],"答案有时正确，有时错误","排序 → 上下文组装 → 生成变异性","重复试验，同时记录检索集、排名、提示上下文和模型输出",[562,563,564],"答案引用了正确的文档但夸大了它","生成 → 证据归因 → 有效性","根据确切引用的段落评估每个声明",[566,567,568],"旧信息总是胜出","排序 → 有效性\u002F新鲜度","与最近性\u002F替代规则比较并检查元数据",[570,571,572],"答案遗漏了例外","分块 → 上下文组装","检查规则和例外是否被拆分或截断",[574,575,576],"增加更多top-k使质量变差","排序 → 上下文过载","消减低价值块并与最小证据集比较",[578,579,580],"更换模型修复了答案","生成，但不一定是检索","在模型间使用相同的检索上下文重复",[582,583,584],"更换嵌入修复了答案","检索\u002F排序","保持生成器和上下文模板不变，同时比较候选召回率",{},{"id":587,"data":588,"type":42,"tunes":590},"h-metrics",{"text":589,"level":240},"用每层实际能影响的指标来衡量",{},{"id":592,"data":593,"type":308,"tunes":626},"metrics-table",{"content":594,"stretched":43,"withHeadings":14},[595,598,602,606,610,614,618,622],[273,596,597],"有用的测量","不应推断什么",[599,600,601],"来源覆盖","可回答问题率、语料库覆盖、摄取完整性","不要因为缺少源材料而责怪嵌入",[603,604,605],"候选检索","Recall@k、命中率、上下文覆盖","高召回率不能证明排序质量",[607,608,609],"排序","MRR、NDCG、黄金排名、precision@k","好的排序不能证明生成器使用了证据",[611,612,613],"上下文组装","证据保留、重复、矛盾率、令牌利用率","大上下文并不意味着有用的上下文",[615,616,617],"生成","正确性、完整性、任务成功、忠实性","仅正确性不能证明有依据",[619,620,621],"证据归因","引用精确度、引用覆盖、声明支持","引用数量不是证据质量",[623,624,625],"有效性","新鲜度、替代准确性、版本\u002F司法管辖区匹配","相关证据不一定是适用证据",{},{"id":628,"data":629,"type":42,"tunes":631},"h-correct-answer",{"text":630,"level":240},"正确答案仍可能隐藏RAG缺陷",{},{"id":633,"data":634,"type":218,"tunes":636},"p-correct-1",{"text":635},"反向问题也很重要。RAG系统可以在检索损坏的情况下产生正确答案。模型可能已经从训练中知道答案，从弱证据中推断出来，或者猜对了。如果评估只看最终答案，系统可能看起来健康，直到问题涉及仅存在于私有语料库中的信息。",{},{"id":638,"data":639,"type":218,"tunes":641},"p-correct-2",{"text":640},"这与更广泛的代理系统中出现的可靠性问题相同：结果正确性不足以证明执行路径是可靠的。对于RAG，跟踪应至少保留检索查询、候选集、排序、最终上下文、答案、引用、模型版本、语料库\u002F索引版本和相关过滤器。",{},{"id":643,"data":644,"type":649,"tunes":650},"internal-reliability",{"url":645,"title":646,"excerpt":647,"ctaLabel":648},"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough","AI代理可靠性：为什么最终答案不够","正确的输出不能证明正确的推理、安全的执行或可信赖的系统。本文将这一原则从RAG诊断扩展到代理轨迹和运营保证。","阅读相关文章","referralArticle",{},{"id":652,"data":653,"type":42,"tunes":655},"h-hypotheses",{"text":654,"level":240},"使用竞争假设，而不是最喜欢的解释",{},{"id":657,"data":658,"type":218,"tunes":660},"p-hyp-1",{"text":659},"如果一个糟糕的回答立刻被归为“嵌入问题”，那么调查就已经带有偏见了。更强大的调试方法是在改变系统之前写下相互竞争的假设：缺失来源、查询重写不佳、检索召回率低、重排序不佳、上下文截断、版本冲突、生成失败、引用失败或证据过时。",{},{"id":662,"data":663,"type":218,"tunes":665},"p-hyp-2",{"text":664},"然后选择一个能够区分这些假设的测试。这比收集更多支持第一种解释的例子更有效率。同样的原则也适用于一般的人工智能辅助技术推理：有用的诊断是能够通过区分性测试的诊断，而不仅仅是听起来合理的诊断。",{},{"id":667,"data":668,"type":649,"tunes":673},"internal-reasoning",{"url":669,"title":670,"excerpt":671,"ctaLabel":672},"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Ffrom-research-protocol-to-a-general-ai-reasoning-framework","从研究协议到通用人工智能推理框架","一种领域无关的推理方法，用于将证据与假设分开、测试相互竞争的假设并使用领域特定的验证器。","阅读推理框架",{},{"id":675,"data":676,"type":42,"tunes":678},"h-change",{"text":677,"level":240},"什么会改变这个答案？",{},{"id":680,"data":681,"type":218,"tunes":683},"p-change-1",{"text":682},"确切的诊断层会随架构而变化。一个简单的单文档RAG应用可能没有查询重写、重排序器或引用层。一个代理式检索系统可能会增加规划、多次搜索、工具选择、记忆、权限和迭代证据收集。结构化数据库查找可能根本不使用块或嵌入。",{},{"id":685,"data":686,"type":218,"tunes":688},"p-change-2",{"text":687},"核心方法仍然成立：识别能够独立改变结果的组件，构建受控测试，用已知良好的输入替换不确定的组件，并使用适合该层的证据来测量每个组件。",{},{"id":690,"data":691,"type":42,"tunes":693},"h-limitations",{"text":692,"level":240},"局限性",{},{"id":695,"data":696,"type":218,"tunes":698},"p-limit-1",{"text":697},"真实的失败往往是耦合的。一个弱的查询会降低召回率，进而改变重排序，进而改变上下文，进而增加生成方差。预言机上下文测试是一种诊断捷径，并不能证明某个组件是唯一原因。评估数据集也可能不具有代表性，而基于模型的评分器可能引入自身的错误。",{},{"id":700,"data":701,"type":218,"tunes":703},"p-limit-2",{"text":702},"因此，所提出的堆栈最好用作调查结构：记录管道、隔离变量、重现失败、测试相互竞争的解释，并在层级别修复后进行端到端评估。",{},{"id":705,"data":706,"type":42,"tunes":708},"h-conclusion",{"text":707,"level":240},"结论",{},{"id":710,"data":711,"type":218,"tunes":713},"p-conclusion-1",{"text":712},"“RAG失败了”应该是调查的开始，而不是结论。有用的诊断能够识别系统是缺乏证据、搜索不正确、未能检索到、排序不佳、组装了不可用的上下文、生成不正确、归因不当，还是应用了超出其有效性边界的证据。",{},{"id":715,"data":716,"type":218,"tunes":718},"p-conclusion-2",{"text":717},"实用规则很简单：一次一层地用受控证据替换不确定性。从预言机上下文测试开始。将仅检索评估与生成评估分开。保留完整跟踪。然后修复实际失败的组件，而不是凭直觉调整整个RAG堆栈。",{},{"id":720,"data":721,"type":42,"tunes":723},"h-faq",{"text":722,"level":240},"常见问题",{},{"id":725,"data":726,"type":725,"tunes":749},"faq",{"items":727,"title":748},[728,732,736,740,744],{"id":729,"answer":730,"question":731},"faq1","手动给模型一小部分已知正确的证据。如果答案变得正确，则调查来源覆盖、查询构建、检索、排序和上下文组装。如果模型在证据充足的情况下仍然失败，则检索不是主要问题。","如何判断是RAG检索失败还是LLM失败？",{"id":733,"answer":734,"question":735},"faq2","是的。相关段落可能排名太低、被截断、与例外分离、与冲突证据混合、被无关上下文淹没，或者被生成器错误使用。","即使检索到了正确的文档，RAG也会失败吗？",{"id":737,"answer":738,"question":739},"faq3","不。模型可能依赖先验知识或运气，在检索薄弱的情况下产生正确答案。应将检索和证据支持与最终答案正确性分开评估。","答案正确性足以评估RAG系统吗？",{"id":741,"answer":742,"question":743},"faq4","至少记录用户请求、转换后的检索查询、过滤器、候选文档及排名、最终选择的上下文、模型和提示版本、答案、引用、语料库\u002F索引版本，以及与新鲜度相关的时间或版本元数据。","调试RAG时应该记录什么？",{"id":745,"answer":746,"question":747},"faq5","不一定可靠。更大的候选或上下文集可能会提高召回率，但也可能增加噪声、矛盾、重复和上下文过载。在增加top-k之前，先测试相关证据是否缺失。","增加top-k通常能修复RAG吗？","RAG故障诊断",{},{"id":751,"data":752,"type":42,"tunes":754},"h-glossary",{"text":753,"level":240},"术语表",{},{"id":756,"data":757,"type":756,"tunes":782},"glossary",{"title":758,"entries":759},"关键诊断术语",[760,764,767,770,774,778],{"term":761,"anchor":762,"definition":763},"预言机上下文测试","oracle-context-test","一种受控测试，直接向生成器提供已知充分的证据，以确定主要失败是否发生在生成之前。",{"term":603,"anchor":765,"definition":766},"candidate-retrieval","在最终排序或上下文组装之前，选择一组初始潜在相关文档、块、记录或段落的阶段。",{"term":611,"anchor":768,"definition":769},"context-assembly","将检索到的证据转换为实际模型输入的过程，包括排序、截断、去重、格式化和令牌预算决策。",{"term":771,"anchor":772,"definition":773},"忠实度","faithfulness","生成的声明在多大程度上仍由检索到或提供的证据支持，而不是引入无支持的内容。",{"term":775,"anchor":776,"definition":777},"上下文覆盖","context-coverage","一种面向检索的度量，衡量所选证据是否覆盖回答问题所需的信息。",{"term":779,"anchor":780,"definition":781},"有效性边界","validity-boundary","声明或答案仍然适用的条件，例如时间、版本、司法管辖区、状态、人群、权限或来源假设。",{},{"id":784,"data":785,"type":42,"tunes":787},"h-sources",{"text":786,"level":240},"主要来源和进一步阅读",{},{"id":789,"data":790,"type":796,"tunes":797},"src-openai-rag",{"link":791,"meta":792},"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Foptimizing-llm-accuracy",{"image":793,"title":794,"description":795},{"url":455},"OpenAI — 优化LLM准确性","OpenAI指南，区分RAG应用中的检索失败和LLM失败。","linkTool",{},{"id":799,"data":800,"type":796,"tunes":806},"src-openai-evals",{"link":801,"meta":802},"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fevaluation-best-practices",{"image":803,"title":804,"description":805},{"url":455},"OpenAI — 评估最佳实践","关于可变AI系统结构化评估和生产导向测试设计的指导。",{},{"id":808,"data":809,"type":796,"tunes":815},"src-aws-rag",{"link":810,"meta":811},"https:\u002F\u002Fdocs.aws.amazon.com\u002Fbedrock\u002Flatest\u002Fuserguide\u002Fknowledge-base-evaluation-metrics.html",{"image":812,"title":813,"description":814},{"url":455},"Amazon Bedrock — RAG评估指标","文档区分了仅检索指标和检索并生成指标，包括上下文相关性、覆盖率、忠实度和引用度量。",{},{"id":817,"data":818,"type":796,"tunes":824},"src-anthropic-evals",{"link":819,"meta":820},"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Fdemystifying-evals-for-ai-agents",{"image":821,"title":822,"description":823},{"url":455},"Anthropic — 揭秘AI代理的评估","关于任务、试验、评分器、追踪、回归和生产行为的实用评估指导。",{},{"id":826,"data":827,"type":796,"tunes":833},"src-google-rag",{"link":828,"meta":829},"https:\u002F\u002Fcloud.google.com\u002Fuse-cases\u002Fretrieval-augmented-generation",{"image":830,"title":831,"description":832},{"url":455},"Google Cloud — 检索增强生成","RAG架构概述以及相关检索和基于生成的重要性的概述。",{},"2.31.6","当RAG答案出错时，将问题归咎于检索或模型过于笼统。这种诊断方法将来源覆盖、查询构建、检索、排序、上下文组装、生成、证据归因和时效性逐一隔离，从而使实际故障能够被复现并修复。","\u002Fuploads\u002F2026\u002F09\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method-1790350847177-pior4c.webp","rag-failed-but-which-layer-actually-failed-a-diagnostic-method-1790350847177-pior4c","PUBLISHED","2026-09-24T19:39:00.000Z","2026-09-25T15:39:19.132Z","2026-09-25T20:46:25.690Z",{"en":843,"de":844,"sr":845,"es":846,"fr":847,"it":848,"ru":849,"zh":850},"\u002Fblog\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method","\u002Fde\u002Fblog\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method","\u002Fsr\u002Fblog\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method","\u002Fes\u002Fblog\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method","\u002Ffr\u002Fblog\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method","\u002Fit\u002Fblog\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method","\u002Fru\u002Fblog\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method","\u002Fzh\u002Fblog\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method",[852,856,860],{"id":853,"name":854,"slug":855},58,"评估与质量门槛","evaluation",{"id":857,"name":858,"slug":859},89,"评估框架","evaluation-harness",{"id":861,"name":862,"slug":863},85,"质量门槛","quality-gates",{"id":865,"login":866,"email":867,"displayName":868},"20","rooth8233","aleksandar@stajic.de","Aleksandar Stajić",[870,1385],{"lang":871,"title":872,"content":873,"contentJson":874,"excerpt":1384},"en","RAG Failed — But Which Layer Actually Failed? A Diagnostic Method","{\"time\":1790369097340,\"blocks\":[{\"id\":\"8zyFXn5HD5\",\"type\":\"tableOfContents\",\"data\":{\"title\":\"Contents\",\"minLevel\":2,\"maxLevel\":3},\"tunes\":{}},{\"id\":\"intro\",\"type\":\"paragraph\",\"data\":{\"text\":\"A RAG system returns a weak, wrong, incomplete, or unsupported answer. The usual diagnosis is “retrieval failed” or “the model hallucinated.” Both labels are too broad to be useful. A production RAG pipeline can fail before retrieval, during retrieval, while ranking, while assembling context, during generation, or after generation when evidence and validity are checked.\"},\"tunes\":{}},{\"id\":\"direct\",\"type\":\"callout\",\"data\":{\"variant\":\"info\",\"title\":\"Direct answer\",\"body\":\"\u003Cstrong>Do not debug RAG as one component.\u003C\u002Fstrong> Diagnose it as a chain of independently testable layers. First determine whether the required evidence exists in an authoritative source. Then test query construction, candidate retrieval, ranking, context assembly, generation, evidence attribution, and freshness. The fastest isolation technique is an \u003Cstrong>oracle-context test\u003C\u002Fstrong>: give the generator the correct evidence manually. If the answer becomes correct, the dominant failure is upstream of generation. If it remains wrong, retrieval is not the primary problem.\"},\"tunes\":{}},{\"id\":\"model-note\",\"type\":\"callout\",\"data\":{\"variant\":\"note\",\"title\":\"About the diagnostic model\",\"body\":\"The RAG Failure Stack in this article is a practical diagnostic model, not a formal industry standard. Existing platforms already separate retrieval-only metrics from retrieve-and-generate metrics; this model extends that separation into a step-by-step production debugging method.\"},\"tunes\":{}},{\"id\":\"h-not-diagnosis\",\"type\":\"header\",\"data\":{\"text\":\"Why “RAG failed” is not a diagnosis\",\"level\":2},\"tunes\":{}},{\"id\":\"p-not-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Retrieval-augmented generation combines several mechanisms: a user request is interpreted, one or more searches are constructed, candidate material is retrieved, results are filtered or reranked, selected evidence is inserted into a model context, and a model generates an answer. Production systems may add permissions, metadata filters, freshness rules, citations, query rewriting, hybrid search, tool calls, memory, and external state.\"},\"tunes\":{}},{\"id\":\"p-not-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"A wrong final answer therefore does not tell you which component failed. The model may have received the wrong evidence. It may have received the right evidence mixed with too much noise. The evidence may be correct but stale. The source may never have contained the answer. Or the model may have ignored perfectly adequate context.\"},\"tunes\":{}},{\"id\":\"p-not-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"OpenAI's RAG guidance already makes a fundamental distinction between retrieval failure and model failure: a system can supply the wrong context, or it can supply the right context and still generate the wrong answer. AWS similarly separates retrieve-only evaluation from retrieve-and-generate evaluation. For production diagnosis, that distinction should be taken further.\"},\"tunes\":{}},{\"id\":\"h-stack\",\"type\":\"header\",\"data\":{\"text\":\"The RAG Failure Stack\",\"level\":2},\"tunes\":{}},{\"id\":\"table-stack\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Layer\",\"Question\",\"Typical failure\"],[\"1. Source coverage\",\"Does the required evidence exist in an allowed authoritative source?\",\"The corpus cannot answer the question at all\"],[\"2. Query construction\",\"Did the system search for the right thing?\",\"Intent, entities, filters, language, or time constraints are lost\"],[\"3. Candidate retrieval\",\"Did the relevant evidence enter the candidate set?\",\"Low recall; the right chunk is never retrieved\"],[\"4. Ranking &amp; filtering\",\"Did the right evidence survive and rank high enough?\",\"Relevant evidence is buried, filtered out, or outranked by superficially similar text\"],[\"5. Context assembly\",\"Did the model receive usable evidence?\",\"Truncation, bad chunk boundaries, duplicates, conflicting passages, or context overload\"],[\"6. Generation\",\"Did the model use the supplied evidence correctly?\",\"Unsupported inference, instruction failure, reasoning error, or refusal mismatch\"],[\"7. Evidence attribution\",\"Can the answer be traced to the evidence it claims to use?\",\"Missing, weak, or incorrect citations; claims exceed retrieved support\"],[\"8. Validity &amp; freshness\",\"Is the evidence still valid for this question now?\",\"Correct historical evidence is reused outside its valid time, version, jurisdiction, or state\"]]},\"tunes\":{}},{\"id\":\"h-source\",\"type\":\"header\",\"data\":{\"text\":\"Layer 1 — Source coverage: can the system answer this at all?\",\"level\":3},\"tunes\":{}},{\"id\":\"p-source-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Before tuning embeddings, rerankers, or prompts, verify that the answer exists in the knowledge space the system is allowed to use. This sounds obvious, but many RAG failures are actually corpus failures. The requested fact may be absent, hidden in an unindexed attachment, available only in a newer document, stored in a system outside the RAG corpus, or blocked by permissions.\"},\"tunes\":{}},{\"id\":\"p-source-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"A retrieval metric cannot recover information that was never indexed. A larger top-k cannot retrieve a document the pipeline does not contain. If the source coverage test fails, the correct fix is ingestion, source selection, permissions, or an explicit “not answerable from available evidence” behaviour.\"},\"tunes\":{}},{\"id\":\"source-warning\",\"type\":\"callout\",\"data\":{\"variant\":\"warning\",\"title\":\"Failure pattern\",\"body\":\"Teams often tune retrieval against questions that the corpus cannot actually answer. This can make the retriever better at finding related text while leaving the underlying information gap untouched.\"},\"tunes\":{}},{\"id\":\"h-query\",\"type\":\"header\",\"data\":{\"text\":\"Layer 2 — Query construction: did the system ask the corpus the right question?\",\"level\":3},\"tunes\":{}},{\"id\":\"p-query-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The user query is not always the retrieval query. Production systems rewrite questions, resolve pronouns, extract entities, translate languages, add metadata constraints, split complex questions, or generate multiple searches. Every transformation can improve retrieval, but every transformation can also destroy information.\"},\"tunes\":{}},{\"id\":\"p-query-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"A request such as “Does the policy still apply to contractors in Germany after the September update?” contains at least an entity, a population, a jurisdiction, and a time boundary. A rewritten query that becomes “contractor policy” may retrieve semantically related text while losing the variables that decide whether the answer is valid.\"},\"tunes\":{}},{\"id\":\"h-retrieval\",\"type\":\"header\",\"data\":{\"text\":\"Layer 3 — Candidate retrieval: did the relevant evidence enter the set?\",\"level\":3},\"tunes\":{}},{\"id\":\"p-ret-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Candidate retrieval is primarily a recall problem. The diagnostic question is not yet whether the best result ranked first; it is whether relevant evidence appeared anywhere in the candidate pool. If the known correct source does not appear, investigate indexing, chunking, embeddings, lexical matching, metadata, hybrid search, language handling, synonyms, and query expansion.\"},\"tunes\":{}},{\"id\":\"p-ret-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"This is where retrieval-only evaluation is valuable. AWS exposes context relevance and context coverage for retrieve-only RAG evaluation. The important production habit is to evaluate retrieval before generation so that a polished final answer cannot hide a weak candidate set.\"},\"tunes\":{}},{\"id\":\"h-ranking\",\"type\":\"header\",\"data\":{\"text\":\"Layer 4 — Ranking and filtering: was the right evidence discarded or buried?\",\"level\":3},\"tunes\":{}},{\"id\":\"p-rank-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A system can have good recall and still fail because the relevant evidence ranks below noisy but semantically similar material. Rerankers, recency boosts, authority weights, language preferences, tenant filters, access controls, product status filters, and deduplication all change what survives into the final context.\"},\"tunes\":{}},{\"id\":\"p-rank-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Debugging should therefore preserve the full candidate list, not only the final top-k. If the gold evidence was retrieved at rank 18 and a reranker removed it, the fix is not the same as a retrieval miss.\"},\"tunes\":{}},{\"id\":\"h-context\",\"type\":\"header\",\"data\":{\"text\":\"Layer 5 — Context assembly: did useful evidence become usable context?\",\"level\":3},\"tunes\":{}},{\"id\":\"p-ctx-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Retrieval success does not guarantee context success. Relevant chunks can be truncated, separated from their qualifiers, duplicated until they dominate the prompt, mixed with contradictory versions, or surrounded by enough irrelevant text that the decisive passage loses salience.\"},\"tunes\":{}},{\"id\":\"p-ctx-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Chunk boundaries are especially important. A sentence may contain the rule while the following sentence contains the exception. If they are indexed separately and only the first is retrieved, the retriever can appear relevant while the assembled context becomes misleading.\"},\"tunes\":{}},{\"id\":\"h-generation\",\"type\":\"header\",\"data\":{\"text\":\"Layer 6 — Generation: can the model use correct evidence correctly?\",\"level\":3},\"tunes\":{}},{\"id\":\"p-gen-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Once the system has demonstrably supplied sufficient evidence, generation becomes independently testable. The model may overgeneralize, combine incompatible passages, ignore a negative statement, fail to follow the requested answer format, invent a bridge between facts, or answer from parametric memory instead of the retrieved evidence.\"},\"tunes\":{}},{\"id\":\"p-gen-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"This is why end-to-end correctness alone is insufficient for diagnosis. OpenAI recommends evaluation as a structured way to understand application behaviour, while Anthropic's agent-evaluation guidance emphasizes multiple trials, graders, traces, and realistic failure cases. For RAG, the generator should be tested both with normal retrieval and with controlled gold context.\"},\"tunes\":{}},{\"id\":\"h-evidence\",\"type\":\"header\",\"data\":{\"text\":\"Layer 7 — Evidence attribution: is the answer actually supported?\",\"level\":3},\"tunes\":{}},{\"id\":\"p-evidence-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A plausible answer with citations can still be weakly grounded. The cited document may be relevant to the topic but not support the specific claim. One sentence may be supported while another is inferred. A citation may point to a source that contradicts the answer once its conditions are read.\"},\"tunes\":{}},{\"id\":\"p-evidence-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Citation evaluation therefore belongs after generation. AWS distinguishes citation precision from citation coverage: whether cited passages are correctly cited and whether the answer is sufficiently supported by citations. In production, claim-level support is more useful than treating the presence of any citation as evidence quality.\"},\"tunes\":{}},{\"id\":\"h-validity\",\"type\":\"header\",\"data\":{\"text\":\"Layer 8 — Validity and freshness: was the evidence correct for this version of reality?\",\"level\":3},\"tunes\":{}},{\"id\":\"p-valid-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"RAG can retrieve a perfectly authentic, highly relevant, faithfully quoted source and still produce a wrong answer if the source is no longer valid for the current question. Policies change. APIs are deprecated. prices move. software behaviour changes between versions. product inventory changes. permissions change. game patches change mechanics.\"},\"tunes\":{}},{\"id\":\"p-valid-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"This is a separate failure class from hallucination. The evidence is real; its applicability is wrong. A robust system therefore needs timestamps, version or jurisdiction metadata where relevant, source authority, supersession rules, and an explicit mechanism for deciding when older evidence must be restricted or abandoned.\"},\"tunes\":{}},{\"id\":\"h-oracle\",\"type\":\"header\",\"data\":{\"text\":\"The fastest isolation method: the oracle-context test\",\"level\":2},\"tunes\":{}},{\"id\":\"p-oracle-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The most useful first split is simple: manually provide the generator with a small set of evidence that you know is sufficient to answer the question. Keep the task and expected answer unchanged.\"},\"tunes\":{}},{\"id\":\"oracle-comparison\",\"type\":\"comparison\",\"data\":{\"title\":\"Oracle-context test\",\"layout\":\"table\",\"columns\":[{\"id\":\"result\",\"label\":\"Result\"},{\"id\":\"meaning\",\"label\":\"Likely interpretation\"},{\"id\":\"next\",\"label\":\"Next diagnostic step\"}],\"rows\":[{\"id\":\"oracle-pass\",\"label\":\"Answer becomes correct\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"oracle-fail\",\"label\":\"Answer remains wrong\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"oracle-partial\",\"label\":\"Answer improves but remains incomplete\",\"values\":[\"\",\"\",\"\"]}]},\"tunes\":{}},{\"id\":\"oracle-tip\",\"type\":\"callout\",\"data\":{\"variant\":\"tip\",\"title\":\"Why this test is powerful\",\"body\":\"The oracle-context test removes most of the retrieval pipeline from the experiment. It does not prove that generation is perfect, but it gives you a fast counterfactual: \u003Cstrong>what would the model do if retrieval had already succeeded?\u003C\u002Fstrong>\"},\"tunes\":{}},{\"id\":\"h-sequence\",\"type\":\"header\",\"data\":{\"text\":\"A production diagnostic sequence\",\"level\":2},\"tunes\":{}},{\"id\":\"diag-flow\",\"type\":\"processFlow\",\"data\":{\"title\":\"Diagnose the failure from evidence to answer\",\"orientation\":\"auto\",\"steps\":[{\"label\":\"1. Define the expected claim\",\"description\":\"Write the expected answer, allowed uncertainty, and the evidence that would justify it.\"},{\"label\":\"2. Verify source coverage\",\"description\":\"Confirm that authoritative and permitted evidence exists in the indexed or reachable source set.\"},{\"label\":\"3. Run the oracle-context test\",\"description\":\"Supply sufficient gold evidence directly to the generator and observe whether the answer becomes correct.\"},{\"label\":\"4. Inspect the retrieval query\",\"description\":\"Check rewrites, entities, filters, language, time constraints, decomposition, and hidden assumptions.\"},{\"label\":\"5. Inspect candidates before reranking\",\"description\":\"Determine whether relevant evidence was retrieved at all and record its rank.\"},{\"label\":\"6. Inspect ranking and context assembly\",\"description\":\"Check reranking, metadata filters, truncation, chunk boundaries, duplicates, conflicts, and top-k composition.\"},{\"label\":\"7. Grade generation and citations separately\",\"description\":\"Measure answer correctness, completeness, faithfulness, and claim-level evidence support.\"},{\"label\":\"8. Test validity boundaries\",\"description\":\"Check whether version, date, state, jurisdiction, permissions, or superseding evidence changes the answer.\"}]},\"tunes\":{}},{\"id\":\"h-one-change\",\"type\":\"header\",\"data\":{\"text\":\"Do not change three layers at once\",\"level\":2},\"tunes\":{}},{\"id\":\"p-one-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A common debugging mistake is to change embeddings, chunk sizes, top-k, prompts, and the model in one iteration. If the score improves, you do not know why. If it gets worse, you do not know which change caused the regression.\"},\"tunes\":{}},{\"id\":\"p-one-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Treat RAG debugging like experimental diagnosis: hold as much of the pipeline constant as possible and replace one uncertain component with a controlled input. Gold documents isolate retrieval. Gold chunks isolate chunk selection. Fixed context isolates generation. A fixed model isolates retrieval changes. A fixed corpus isolates ingestion and indexing changes.\"},\"tunes\":{}},{\"id\":\"h-matrix\",\"type\":\"header\",\"data\":{\"text\":\"A failure matrix for common RAG symptoms\",\"level\":2},\"tunes\":{}},{\"id\":\"symptom-matrix\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Symptom\",\"Most likely layers to test first\",\"Discriminating test\"],[\"No relevant source appears\",\"Source coverage → Query → Candidate retrieval\",\"Search the corpus manually, then inspect rewritten query and unfiltered candidates\"],[\"Relevant source appears but answer is wrong\",\"Context assembly → Generation\",\"Oracle-context test with the same source reduced to decisive passages\"],[\"Answer is correct sometimes, wrong other times\",\"Ranking → Context assembly → Generation variability\",\"Repeat trials while logging retrieved set, rank, prompt context, and model output\"],[\"Answer cites the right document but overstates it\",\"Generation → Evidence attribution → Validity\",\"Grade each claim against the exact cited passage\"],[\"Old information keeps winning\",\"Ranking → Validity\u002Ffreshness\",\"Compare with recency\u002Fsupersession rules and inspect metadata\"],[\"Answer misses an exception\",\"Chunking → Context assembly\",\"Check whether rule and exception were split or truncated\"],[\"Adding more top-k makes quality worse\",\"Ranking → Context overload\",\"Ablate low-value chunks and compare with a minimal evidence set\"],[\"Changing the model fixes the answer\",\"Generation, but not necessarily retrieval\",\"Repeat with identical retrieved context across models\"],[\"Changing embeddings fixes the answer\",\"Retrieval\u002Franking\",\"Keep generator and context template constant while comparing candidate recall\"]]},\"tunes\":{}},{\"id\":\"h-metrics\",\"type\":\"header\",\"data\":{\"text\":\"Measure each layer with the metric it can actually influence\",\"level\":2},\"tunes\":{}},{\"id\":\"metrics-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Layer\",\"Useful measurements\",\"What not to infer\"],[\"Source coverage\",\"Answerable-question rate, corpus coverage, ingestion completeness\",\"Do not blame embeddings for missing source material\"],[\"Candidate retrieval\",\"Recall@k, hit rate, context coverage\",\"High recall does not prove ranking quality\"],[\"Ranking\",\"MRR, NDCG, gold rank, precision@k\",\"Good ranking does not prove the generator used the evidence\"],[\"Context assembly\",\"Evidence retention, duplication, contradiction rate, token utilization\",\"Large context does not mean useful context\"],[\"Generation\",\"Correctness, completeness, task success, faithfulness\",\"Correctness alone does not prove grounding\"],[\"Evidence attribution\",\"Citation precision, citation coverage, claim support\",\"A citation count is not evidence quality\"],[\"Validity\",\"Freshness, supersession accuracy, version\u002Fjurisdiction match\",\"Relevant evidence is not automatically applicable evidence\"]]},\"tunes\":{}},{\"id\":\"h-correct-answer\",\"type\":\"header\",\"data\":{\"text\":\"A correct answer can still hide a RAG defect\",\"level\":2},\"tunes\":{}},{\"id\":\"p-correct-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The reverse problem also matters. A RAG system can produce the correct answer while retrieval is broken. The model may already know the answer from training, infer it from weak evidence, or guess correctly. If evaluation looks only at the final answer, the system can appear healthy until the question reaches information that exists only in the private corpus.\"},\"tunes\":{}},{\"id\":\"p-correct-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"This is the same reliability problem that appears in agent systems more broadly: outcome correctness is not enough to prove that the execution path was reliable. For RAG, traces should preserve at least the retrieval query, candidate set, ranking, final context, answer, citations, model version, corpus\u002Findex version, and relevant filters.\"},\"tunes\":{}},{\"id\":\"internal-reliability\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough\",\"title\":\"AI Agent Reliability: Why the Final Answer Is Not Enough\",\"excerpt\":\"Correct output does not prove correct reasoning, safe execution, or a trustworthy system. This article extends that principle from RAG diagnosis to agent trajectories and operational assurance.\",\"ctaLabel\":\"Read the related article\"},\"tunes\":{}},{\"id\":\"h-hypotheses\",\"type\":\"header\",\"data\":{\"text\":\"Use competing hypotheses, not a favourite explanation\",\"level\":2},\"tunes\":{}},{\"id\":\"p-hyp-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"If a bad answer immediately becomes “an embedding problem,” the investigation is already biased. A stronger debugging method writes down competing hypotheses before changing the system: missing source, bad query rewrite, low retrieval recall, bad reranking, context truncation, conflicting versions, generation failure, citation failure, or stale evidence.\"},\"tunes\":{}},{\"id\":\"p-hyp-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Then choose a test that would separate those hypotheses. This is more efficient than collecting more examples that support the first explanation. The same principle applies to AI-assisted technical reasoning in general: a useful diagnosis is one that survives discriminating tests, not one that merely sounds plausible.\"},\"tunes\":{}},{\"id\":\"internal-reasoning\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Ffrom-research-protocol-to-a-general-ai-reasoning-framework\",\"title\":\"From Research Protocol to a General AI Reasoning Framework\",\"excerpt\":\"A domain-independent reasoning method for separating evidence from assumptions, testing competing hypotheses and using domain-specific validators.\",\"ctaLabel\":\"Read the reasoning framework\"},\"tunes\":{}},{\"id\":\"h-change\",\"type\":\"header\",\"data\":{\"text\":\"What would change this answer?\",\"level\":2},\"tunes\":{}},{\"id\":\"p-change-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The exact diagnostic layers change with architecture. A simple single-document RAG application may have no query rewriting, reranker, or citation layer. An agentic retrieval system may add planning, multiple searches, tool selection, memory, permissions, and iterative evidence gathering. A structured database lookup may not use chunks or embeddings at all.\"},\"tunes\":{}},{\"id\":\"p-change-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The core method still holds: identify the components that can independently change the result, construct controlled tests that replace uncertain components with known-good inputs, and measure each component using evidence appropriate to that layer.\"},\"tunes\":{}},{\"id\":\"h-limitations\",\"type\":\"header\",\"data\":{\"text\":\"Limitations\",\"level\":2},\"tunes\":{}},{\"id\":\"p-limit-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Real failures are often coupled. A weak query can reduce recall, which changes reranking, which changes context, which increases generation variance. The oracle-context test is a diagnostic shortcut, not proof that one component is solely responsible. Evaluation datasets can also be unrepresentative, and model-based graders can introduce their own errors.\"},\"tunes\":{}},{\"id\":\"p-limit-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The proposed stack is therefore best used as an investigation structure: log the pipeline, isolate variables, reproduce failures, test competing explanations, and keep end-to-end evaluation after layer-level fixes.\"},\"tunes\":{}},{\"id\":\"h-conclusion\",\"type\":\"header\",\"data\":{\"text\":\"Conclusion\",\"level\":2},\"tunes\":{}},{\"id\":\"p-conclusion-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"“RAG failed” should be the beginning of the investigation, not the conclusion. A useful diagnosis identifies whether the system lacked the evidence, searched incorrectly, failed to retrieve it, ranked it badly, assembled unusable context, generated incorrectly, attributed claims poorly, or applied evidence outside its validity boundary.\"},\"tunes\":{}},{\"id\":\"p-conclusion-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The practical rule is simple: replace uncertainty with controlled evidence one layer at a time. Start with the oracle-context test. Separate retrieval-only evaluation from generation evaluation. Preserve the full trace. Then fix the component that actually failed instead of tuning the entire RAG stack by intuition.\"},\"tunes\":{}},{\"id\":\"h-faq\",\"type\":\"header\",\"data\":{\"text\":\"FAQ\",\"level\":2},\"tunes\":{}},{\"id\":\"faq\",\"type\":\"faq\",\"data\":{\"title\":\"RAG failure diagnosis\",\"items\":[{\"id\":\"faq1\",\"question\":\"How can I tell whether RAG retrieval or the LLM failed?\",\"answer\":\"Give the model a small set of known-correct evidence manually. If the answer becomes correct, investigate source coverage, query construction, retrieval, ranking, and context assembly. If the model still fails with sufficient evidence, retrieval is not the primary problem.\"},{\"id\":\"faq2\",\"question\":\"Can RAG fail even when the correct document was retrieved?\",\"answer\":\"Yes. The relevant passage can be ranked too low, truncated, separated from an exception, mixed with conflicting evidence, overwhelmed by irrelevant context, or used incorrectly by the generator.\"},{\"id\":\"faq3\",\"question\":\"Is answer correctness enough to evaluate a RAG system?\",\"answer\":\"No. A model can produce a correct answer despite weak retrieval by relying on prior model knowledge or chance. Evaluate retrieval and evidence support separately from final-answer correctness.\"},{\"id\":\"faq4\",\"question\":\"What should I log when debugging RAG?\",\"answer\":\"At minimum log the user request, transformed retrieval query, filters, candidate documents and ranks, final selected context, model and prompt version, answer, citations, corpus\u002Findex version, and timing or version metadata relevant to freshness.\"},{\"id\":\"faq5\",\"question\":\"Does increasing top-k usually fix RAG?\",\"answer\":\"Not reliably. A larger candidate or context set may improve recall, but it can also add noise, contradictions, duplicates, and context overload. Test whether the relevant evidence is missing before increasing top-k.\"}]},\"tunes\":{}},{\"id\":\"h-glossary\",\"type\":\"header\",\"data\":{\"text\":\"Glossary\",\"level\":2},\"tunes\":{}},{\"id\":\"glossary\",\"type\":\"glossary\",\"data\":{\"title\":\"Key diagnostic terms\",\"entries\":[{\"term\":\"Oracle-context test\",\"definition\":\"A controlled test in which the generator is given known-sufficient evidence directly to determine whether the dominant failure is upstream of generation.\",\"anchor\":\"oracle-context-test\"},{\"term\":\"Candidate retrieval\",\"definition\":\"The stage that selects an initial set of potentially relevant documents, chunks, records, or passages before final ranking or context assembly.\",\"anchor\":\"candidate-retrieval\"},{\"term\":\"Context assembly\",\"definition\":\"The process of converting retrieved evidence into the actual model input, including ordering, truncation, deduplication, formatting, and token-budget decisions.\",\"anchor\":\"context-assembly\"},{\"term\":\"Faithfulness\",\"definition\":\"The degree to which generated claims remain supported by the retrieved or supplied evidence rather than introducing unsupported content.\",\"anchor\":\"faithfulness\"},{\"term\":\"Context coverage\",\"definition\":\"A retrieval-oriented measure of whether selected evidence covers the information needed to answer the question.\",\"anchor\":\"context-coverage\"},{\"term\":\"Validity boundary\",\"definition\":\"The conditions under which a claim or answer remains applicable, such as time, version, jurisdiction, state, population, permissions, or source assumptions.\",\"anchor\":\"validity-boundary\"}]},\"tunes\":{}},{\"id\":\"h-sources\",\"type\":\"header\",\"data\":{\"text\":\"Primary sources and further reading\",\"level\":2},\"tunes\":{}},{\"id\":\"src-openai-rag\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Foptimizing-llm-accuracy\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — Optimizing LLM Accuracy\",\"description\":\"OpenAI guidance separating retrieval failures from LLM failures in RAG applications.\"}},\"tunes\":{}},{\"id\":\"src-openai-evals\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fevaluation-best-practices\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — Evaluation Best Practices\",\"description\":\"Guidance on structured evaluation for variable AI systems and production-oriented test design.\"}},\"tunes\":{}},{\"id\":\"src-aws-rag\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdocs.aws.amazon.com\u002Fbedrock\u002Flatest\u002Fuserguide\u002Fknowledge-base-evaluation-metrics.html\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Amazon Bedrock — RAG Evaluation Metrics\",\"description\":\"Documentation separating retrieve-only metrics from retrieve-and-generate metrics, including context relevance, coverage, faithfulness and citation measures.\"}},\"tunes\":{}},{\"id\":\"src-anthropic-evals\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Fdemystifying-evals-for-ai-agents\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Anthropic — Demystifying Evals for AI Agents\",\"description\":\"Practical evaluation guidance on tasks, trials, graders, traces, regressions and production behaviour.\"}},\"tunes\":{}},{\"id\":\"src-google-rag\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fcloud.google.com\u002Fuse-cases\u002Fretrieval-augmented-generation\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Google Cloud — Retrieval-Augmented Generation\",\"description\":\"Overview of RAG architecture and the importance of relevant retrieval and grounded generation.\"}},\"tunes\":{}}],\"version\":\"2.31.6\"}",{"time":875,"blocks":876,"version":834},1790369097340,[877,882,886,891,896,900,904,908,912,916,956,960,964,968,973,977,981,985,989,993,997,1001,1005,1009,1013,1017,1021,1025,1029,1033,1037,1041,1045,1049,1053,1057,1061,1065,1086,1091,1095,1124,1128,1132,1136,1140,1184,1188,1223,1227,1231,1235,1242,1246,1250,1254,1261,1265,1269,1273,1277,1281,1285,1289,1293,1297,1301,1321,1325,1345,1349,1356,1363,1370,1377],{"id":878,"data":879,"type":241,"tunes":881},"8zyFXn5HD5",{"title":880,"maxLevel":239,"minLevel":240},"Contents",{},{"id":215,"data":883,"type":218,"tunes":885},{"text":884},"A RAG system returns a weak, wrong, incomplete, or unsupported answer. The usual diagnosis is “retrieval failed” or “the model hallucinated.” Both labels are too broad to be useful. A production RAG pipeline can fail before retrieval, during retrieval, while ranking, while assembling context, during generation, or after generation when evidence and validity are checked.",{},{"id":221,"data":887,"type":226,"tunes":890},{"body":888,"title":889,"variant":225},"\u003Cstrong>Do not debug RAG as one component.\u003C\u002Fstrong> Diagnose it as a chain of independently testable layers. First determine whether the required evidence exists in an authoritative source. Then test query construction, candidate retrieval, ranking, context assembly, generation, evidence attribution, and freshness. The fastest isolation technique is an \u003Cstrong>oracle-context test\u003C\u002Fstrong>: give the generator the correct evidence manually. If the answer becomes correct, the dominant failure is upstream of generation. If it remains wrong, retrieval is not the primary problem.","Direct answer",{},{"id":229,"data":892,"type":226,"tunes":895},{"body":893,"title":894,"variant":233},"The RAG Failure Stack in this article is a practical diagnostic model, not a formal industry standard. Existing platforms already separate retrieval-only metrics from retrieve-and-generate metrics; this model extends that separation into a step-by-step production debugging method.","About the diagnostic model",{},{"id":244,"data":897,"type":42,"tunes":899},{"text":898,"level":240},"Why “RAG failed” is not a diagnosis",{},{"id":249,"data":901,"type":218,"tunes":903},{"text":902},"Retrieval-augmented generation combines several mechanisms: a user request is interpreted, one or more searches are constructed, candidate material is retrieved, results are filtered or reranked, selected evidence is inserted into a model context, and a model generates an answer. Production systems may add permissions, metadata filters, freshness rules, citations, query rewriting, hybrid search, tool calls, memory, and external state.",{},{"id":254,"data":905,"type":218,"tunes":907},{"text":906},"A wrong final answer therefore does not tell you which component failed. The model may have received the wrong evidence. It may have received the right evidence mixed with too much noise. The evidence may be correct but stale. The source may never have contained the answer. Or the model may have ignored perfectly adequate context.",{},{"id":259,"data":909,"type":218,"tunes":911},{"text":910},"OpenAI's RAG guidance already makes a fundamental distinction between retrieval failure and model failure: a system can supply the wrong context, or it can supply the right context and still generate the wrong answer. AWS similarly separates retrieve-only evaluation from retrieve-and-generate evaluation. For production diagnosis, that distinction should be taken further.",{},{"id":264,"data":913,"type":42,"tunes":915},{"text":914,"level":240},"The RAG Failure Stack",{},{"id":269,"data":917,"type":308,"tunes":955},{"content":918,"stretched":43,"withHeadings":14},[919,923,927,931,935,939,943,947,951],[920,921,922],"Layer","Question","Typical failure",[924,925,926],"1. Source coverage","Does the required evidence exist in an allowed authoritative source?","The corpus cannot answer the question at all",[928,929,930],"2. Query construction","Did the system search for the right thing?","Intent, entities, filters, language, or time constraints are lost",[932,933,934],"3. Candidate retrieval","Did the relevant evidence enter the candidate set?","Low recall; the right chunk is never retrieved",[936,937,938],"4. Ranking &amp; filtering","Did the right evidence survive and rank high enough?","Relevant evidence is buried, filtered out, or outranked by superficially similar text",[940,941,942],"5. Context assembly","Did the model receive usable evidence?","Truncation, bad chunk boundaries, duplicates, conflicting passages, or context overload",[944,945,946],"6. Generation","Did the model use the supplied evidence correctly?","Unsupported inference, instruction failure, reasoning error, or refusal mismatch",[948,949,950],"7. Evidence attribution","Can the answer be traced to the evidence it claims to use?","Missing, weak, or incorrect citations; claims exceed retrieved support",[952,953,954],"8. Validity &amp; freshness","Is the evidence still valid for this question now?","Correct historical evidence is reused outside its valid time, version, jurisdiction, or state",{},{"id":311,"data":957,"type":42,"tunes":959},{"text":958,"level":239},"Layer 1 — Source coverage: can the system answer this at all?",{},{"id":316,"data":961,"type":218,"tunes":963},{"text":962},"Before tuning embeddings, rerankers, or prompts, verify that the answer exists in the knowledge space the system is allowed to use. This sounds obvious, but many RAG failures are actually corpus failures. The requested fact may be absent, hidden in an unindexed attachment, available only in a newer document, stored in a system outside the RAG corpus, or blocked by permissions.",{},{"id":321,"data":965,"type":218,"tunes":967},{"text":966},"A retrieval metric cannot recover information that was never indexed. A larger top-k cannot retrieve a document the pipeline does not contain. If the source coverage test fails, the correct fix is ingestion, source selection, permissions, or an explicit “not answerable from available evidence” behaviour.",{},{"id":326,"data":969,"type":226,"tunes":972},{"body":970,"title":971,"variant":330},"Teams often tune retrieval against questions that the corpus cannot actually answer. This can make the retriever better at finding related text while leaving the underlying information gap untouched.","Failure pattern",{},{"id":333,"data":974,"type":42,"tunes":976},{"text":975,"level":239},"Layer 2 — Query construction: did the system ask the corpus the right question?",{},{"id":338,"data":978,"type":218,"tunes":980},{"text":979},"The user query is not always the retrieval query. Production systems rewrite questions, resolve pronouns, extract entities, translate languages, add metadata constraints, split complex questions, or generate multiple searches. Every transformation can improve retrieval, but every transformation can also destroy information.",{},{"id":343,"data":982,"type":218,"tunes":984},{"text":983},"A request such as “Does the policy still apply to contractors in Germany after the September update?” contains at least an entity, a population, a jurisdiction, and a time boundary. A rewritten query that becomes “contractor policy” may retrieve semantically related text while losing the variables that decide whether the answer is valid.",{},{"id":348,"data":986,"type":42,"tunes":988},{"text":987,"level":239},"Layer 3 — Candidate retrieval: did the relevant evidence enter the set?",{},{"id":353,"data":990,"type":218,"tunes":992},{"text":991},"Candidate retrieval is primarily a recall problem. The diagnostic question is not yet whether the best result ranked first; it is whether relevant evidence appeared anywhere in the candidate pool. If the known correct source does not appear, investigate indexing, chunking, embeddings, lexical matching, metadata, hybrid search, language handling, synonyms, and query expansion.",{},{"id":358,"data":994,"type":218,"tunes":996},{"text":995},"This is where retrieval-only evaluation is valuable. AWS exposes context relevance and context coverage for retrieve-only RAG evaluation. The important production habit is to evaluate retrieval before generation so that a polished final answer cannot hide a weak candidate set.",{},{"id":363,"data":998,"type":42,"tunes":1000},{"text":999,"level":239},"Layer 4 — Ranking and filtering: was the right evidence discarded or buried?",{},{"id":368,"data":1002,"type":218,"tunes":1004},{"text":1003},"A system can have good recall and still fail because the relevant evidence ranks below noisy but semantically similar material. Rerankers, recency boosts, authority weights, language preferences, tenant filters, access controls, product status filters, and deduplication all change what survives into the final context.",{},{"id":373,"data":1006,"type":218,"tunes":1008},{"text":1007},"Debugging should therefore preserve the full candidate list, not only the final top-k. If the gold evidence was retrieved at rank 18 and a reranker removed it, the fix is not the same as a retrieval miss.",{},{"id":378,"data":1010,"type":42,"tunes":1012},{"text":1011,"level":239},"Layer 5 — Context assembly: did useful evidence become usable context?",{},{"id":383,"data":1014,"type":218,"tunes":1016},{"text":1015},"Retrieval success does not guarantee context success. Relevant chunks can be truncated, separated from their qualifiers, duplicated until they dominate the prompt, mixed with contradictory versions, or surrounded by enough irrelevant text that the decisive passage loses salience.",{},{"id":388,"data":1018,"type":218,"tunes":1020},{"text":1019},"Chunk boundaries are especially important. A sentence may contain the rule while the following sentence contains the exception. If they are indexed separately and only the first is retrieved, the retriever can appear relevant while the assembled context becomes misleading.",{},{"id":393,"data":1022,"type":42,"tunes":1024},{"text":1023,"level":239},"Layer 6 — Generation: can the model use correct evidence correctly?",{},{"id":398,"data":1026,"type":218,"tunes":1028},{"text":1027},"Once the system has demonstrably supplied sufficient evidence, generation becomes independently testable. The model may overgeneralize, combine incompatible passages, ignore a negative statement, fail to follow the requested answer format, invent a bridge between facts, or answer from parametric memory instead of the retrieved evidence.",{},{"id":403,"data":1030,"type":218,"tunes":1032},{"text":1031},"This is why end-to-end correctness alone is insufficient for diagnosis. OpenAI recommends evaluation as a structured way to understand application behaviour, while Anthropic's agent-evaluation guidance emphasizes multiple trials, graders, traces, and realistic failure cases. For RAG, the generator should be tested both with normal retrieval and with controlled gold context.",{},{"id":408,"data":1034,"type":42,"tunes":1036},{"text":1035,"level":239},"Layer 7 — Evidence attribution: is the answer actually supported?",{},{"id":413,"data":1038,"type":218,"tunes":1040},{"text":1039},"A plausible answer with citations can still be weakly grounded. The cited document may be relevant to the topic but not support the specific claim. One sentence may be supported while another is inferred. A citation may point to a source that contradicts the answer once its conditions are read.",{},{"id":418,"data":1042,"type":218,"tunes":1044},{"text":1043},"Citation evaluation therefore belongs after generation. AWS distinguishes citation precision from citation coverage: whether cited passages are correctly cited and whether the answer is sufficiently supported by citations. In production, claim-level support is more useful than treating the presence of any citation as evidence quality.",{},{"id":423,"data":1046,"type":42,"tunes":1048},{"text":1047,"level":239},"Layer 8 — Validity and freshness: was the evidence correct for this version of reality?",{},{"id":428,"data":1050,"type":218,"tunes":1052},{"text":1051},"RAG can retrieve a perfectly authentic, highly relevant, faithfully quoted source and still produce a wrong answer if the source is no longer valid for the current question. Policies change. APIs are deprecated. prices move. software behaviour changes between versions. product inventory changes. permissions change. game patches change mechanics.",{},{"id":433,"data":1054,"type":218,"tunes":1056},{"text":1055},"This is a separate failure class from hallucination. The evidence is real; its applicability is wrong. A robust system therefore needs timestamps, version or jurisdiction metadata where relevant, source authority, supersession rules, and an explicit mechanism for deciding when older evidence must be restricted or abandoned.",{},{"id":438,"data":1058,"type":42,"tunes":1060},{"text":1059,"level":240},"The fastest isolation method: the oracle-context test",{},{"id":443,"data":1062,"type":218,"tunes":1064},{"text":1063},"The most useful first split is simple: manually provide the generator with a small set of evidence that you know is sufficient to answer the question. Keep the task and expected answer unchanged.",{},{"id":448,"data":1066,"type":475,"tunes":1085},{"rows":1067,"title":1077,"layout":308,"columns":1078},[1068,1071,1074],{"id":452,"label":1069,"values":1070},"Answer becomes correct",[455,455,455],{"id":457,"label":1072,"values":1073},"Answer remains wrong",[455,455,455],{"id":461,"label":1075,"values":1076},"Answer improves but remains incomplete",[455,455,455],"Oracle-context test",[1079,1081,1083],{"id":467,"label":1080},"Result",{"id":470,"label":1082},"Likely interpretation",{"id":473,"label":1084},"Next diagnostic step",{},{"id":478,"data":1087,"type":226,"tunes":1090},{"body":1088,"title":1089,"variant":482},"The oracle-context test removes most of the retrieval pipeline from the experiment. It does not prove that generation is perfect, but it gives you a fast counterfactual: \u003Cstrong>what would the model do if retrieval had already succeeded?\u003C\u002Fstrong>","Why this test is powerful",{},{"id":485,"data":1092,"type":42,"tunes":1094},{"text":1093,"level":240},"A production diagnostic sequence",{},{"id":490,"data":1096,"type":519,"tunes":1123},{"steps":1097,"title":1122,"orientation":518},[1098,1101,1104,1107,1110,1113,1116,1119],{"label":1099,"description":1100},"1. Define the expected claim","Write the expected answer, allowed uncertainty, and the evidence that would justify it.",{"label":1102,"description":1103},"2. Verify source coverage","Confirm that authoritative and permitted evidence exists in the indexed or reachable source set.",{"label":1105,"description":1106},"3. Run the oracle-context test","Supply sufficient gold evidence directly to the generator and observe whether the answer becomes correct.",{"label":1108,"description":1109},"4. Inspect the retrieval query","Check rewrites, entities, filters, language, time constraints, decomposition, and hidden assumptions.",{"label":1111,"description":1112},"5. Inspect candidates before reranking","Determine whether relevant evidence was retrieved at all and record its rank.",{"label":1114,"description":1115},"6. Inspect ranking and context assembly","Check reranking, metadata filters, truncation, chunk boundaries, duplicates, conflicts, and top-k composition.",{"label":1117,"description":1118},"7. Grade generation and citations separately","Measure answer correctness, completeness, faithfulness, and claim-level evidence support.",{"label":1120,"description":1121},"8. Test validity boundaries","Check whether version, date, state, jurisdiction, permissions, or superseding evidence changes the answer.","Diagnose the failure from evidence to answer",{},{"id":522,"data":1125,"type":42,"tunes":1127},{"text":1126,"level":240},"Do not change three layers at once",{},{"id":527,"data":1129,"type":218,"tunes":1131},{"text":1130},"A common debugging mistake is to change embeddings, chunk sizes, top-k, prompts, and the model in one iteration. If the score improves, you do not know why. If it gets worse, you do not know which change caused the regression.",{},{"id":532,"data":1133,"type":218,"tunes":1135},{"text":1134},"Treat RAG debugging like experimental diagnosis: hold as much of the pipeline constant as possible and replace one uncertain component with a controlled input. Gold documents isolate retrieval. Gold chunks isolate chunk selection. Fixed context isolates generation. A fixed model isolates retrieval changes. A fixed corpus isolates ingestion and indexing changes.",{},{"id":537,"data":1137,"type":42,"tunes":1139},{"text":1138,"level":240},"A failure matrix for common RAG symptoms",{},{"id":542,"data":1141,"type":308,"tunes":1183},{"content":1142,"stretched":43,"withHeadings":14},[1143,1147,1151,1155,1159,1163,1167,1171,1175,1179],[1144,1145,1146],"Symptom","Most likely layers to test first","Discriminating test",[1148,1149,1150],"No relevant source appears","Source coverage → Query → Candidate retrieval","Search the corpus manually, then inspect rewritten query and unfiltered candidates",[1152,1153,1154],"Relevant source appears but answer is wrong","Context assembly → Generation","Oracle-context test with the same source reduced to decisive passages",[1156,1157,1158],"Answer is correct sometimes, wrong other times","Ranking → Context assembly → Generation variability","Repeat trials while logging retrieved set, rank, prompt context, and model output",[1160,1161,1162],"Answer cites the right document but overstates it","Generation → Evidence attribution → Validity","Grade each claim against the exact cited passage",[1164,1165,1166],"Old information keeps winning","Ranking → Validity\u002Ffreshness","Compare with recency\u002Fsupersession rules and inspect metadata",[1168,1169,1170],"Answer misses an exception","Chunking → Context assembly","Check whether rule and exception were split or truncated",[1172,1173,1174],"Adding more top-k makes quality worse","Ranking → Context overload","Ablate low-value chunks and compare with a minimal evidence set",[1176,1177,1178],"Changing the model fixes the answer","Generation, but not necessarily retrieval","Repeat with identical retrieved context across models",[1180,1181,1182],"Changing embeddings fixes the answer","Retrieval\u002Franking","Keep generator and context template constant while comparing candidate recall",{},{"id":587,"data":1185,"type":42,"tunes":1187},{"text":1186,"level":240},"Measure each layer with the metric it can actually influence",{},{"id":592,"data":1189,"type":308,"tunes":1222},{"content":1190,"stretched":43,"withHeadings":14},[1191,1194,1198,1202,1206,1210,1214,1218],[920,1192,1193],"Useful measurements","What not to infer",[1195,1196,1197],"Source coverage","Answerable-question rate, corpus coverage, ingestion completeness","Do not blame embeddings for missing source material",[1199,1200,1201],"Candidate retrieval","Recall@k, hit rate, context coverage","High recall does not prove ranking quality",[1203,1204,1205],"Ranking","MRR, NDCG, gold rank, precision@k","Good ranking does not prove the generator used the evidence",[1207,1208,1209],"Context assembly","Evidence retention, duplication, contradiction rate, token utilization","Large context does not mean useful context",[1211,1212,1213],"Generation","Correctness, completeness, task success, faithfulness","Correctness alone does not prove grounding",[1215,1216,1217],"Evidence attribution","Citation precision, citation coverage, claim support","A citation count is not evidence quality",[1219,1220,1221],"Validity","Freshness, supersession accuracy, version\u002Fjurisdiction match","Relevant evidence is not automatically applicable evidence",{},{"id":628,"data":1224,"type":42,"tunes":1226},{"text":1225,"level":240},"A correct answer can still hide a RAG defect",{},{"id":633,"data":1228,"type":218,"tunes":1230},{"text":1229},"The reverse problem also matters. A RAG system can produce the correct answer while retrieval is broken. The model may already know the answer from training, infer it from weak evidence, or guess correctly. If evaluation looks only at the final answer, the system can appear healthy until the question reaches information that exists only in the private corpus.",{},{"id":638,"data":1232,"type":218,"tunes":1234},{"text":1233},"This is the same reliability problem that appears in agent systems more broadly: outcome correctness is not enough to prove that the execution path was reliable. For RAG, traces should preserve at least the retrieval query, candidate set, ranking, final context, answer, citations, model version, corpus\u002Findex version, and relevant filters.",{},{"id":643,"data":1236,"type":649,"tunes":1241},{"url":1237,"title":1238,"excerpt":1239,"ctaLabel":1240},"https:\u002F\u002Fstajic.de\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough","AI Agent Reliability: Why the Final Answer Is Not Enough","Correct output does not prove correct reasoning, safe execution, or a trustworthy system. This article extends that principle from RAG diagnosis to agent trajectories and operational assurance.","Read the related article",{},{"id":652,"data":1243,"type":42,"tunes":1245},{"text":1244,"level":240},"Use competing hypotheses, not a favourite explanation",{},{"id":657,"data":1247,"type":218,"tunes":1249},{"text":1248},"If a bad answer immediately becomes “an embedding problem,” the investigation is already biased. A stronger debugging method writes down competing hypotheses before changing the system: missing source, bad query rewrite, low retrieval recall, bad reranking, context truncation, conflicting versions, generation failure, citation failure, or stale evidence.",{},{"id":662,"data":1251,"type":218,"tunes":1253},{"text":1252},"Then choose a test that would separate those hypotheses. This is more efficient than collecting more examples that support the first explanation. The same principle applies to AI-assisted technical reasoning in general: a useful diagnosis is one that survives discriminating tests, not one that merely sounds plausible.",{},{"id":667,"data":1255,"type":649,"tunes":1260},{"url":1256,"title":1257,"excerpt":1258,"ctaLabel":1259},"https:\u002F\u002Fstajic.de\u002Fblog\u002Ffrom-research-protocol-to-a-general-ai-reasoning-framework","From Research Protocol to a General AI Reasoning Framework","A domain-independent reasoning method for separating evidence from assumptions, testing competing hypotheses and using domain-specific validators.","Read the reasoning framework",{},{"id":675,"data":1262,"type":42,"tunes":1264},{"text":1263,"level":240},"What would change this answer?",{},{"id":680,"data":1266,"type":218,"tunes":1268},{"text":1267},"The exact diagnostic layers change with architecture. A simple single-document RAG application may have no query rewriting, reranker, or citation layer. An agentic retrieval system may add planning, multiple searches, tool selection, memory, permissions, and iterative evidence gathering. A structured database lookup may not use chunks or embeddings at all.",{},{"id":685,"data":1270,"type":218,"tunes":1272},{"text":1271},"The core method still holds: identify the components that can independently change the result, construct controlled tests that replace uncertain components with known-good inputs, and measure each component using evidence appropriate to that layer.",{},{"id":690,"data":1274,"type":42,"tunes":1276},{"text":1275,"level":240},"Limitations",{},{"id":695,"data":1278,"type":218,"tunes":1280},{"text":1279},"Real failures are often coupled. A weak query can reduce recall, which changes reranking, which changes context, which increases generation variance. The oracle-context test is a diagnostic shortcut, not proof that one component is solely responsible. Evaluation datasets can also be unrepresentative, and model-based graders can introduce their own errors.",{},{"id":700,"data":1282,"type":218,"tunes":1284},{"text":1283},"The proposed stack is therefore best used as an investigation structure: log the pipeline, isolate variables, reproduce failures, test competing explanations, and keep end-to-end evaluation after layer-level fixes.",{},{"id":705,"data":1286,"type":42,"tunes":1288},{"text":1287,"level":240},"Conclusion",{},{"id":710,"data":1290,"type":218,"tunes":1292},{"text":1291},"“RAG failed” should be the beginning of the investigation, not the conclusion. A useful diagnosis identifies whether the system lacked the evidence, searched incorrectly, failed to retrieve it, ranked it badly, assembled unusable context, generated incorrectly, attributed claims poorly, or applied evidence outside its validity boundary.",{},{"id":715,"data":1294,"type":218,"tunes":1296},{"text":1295},"The practical rule is simple: replace uncertainty with controlled evidence one layer at a time. Start with the oracle-context test. Separate retrieval-only evaluation from generation evaluation. Preserve the full trace. Then fix the component that actually failed instead of tuning the entire RAG stack by intuition.",{},{"id":720,"data":1298,"type":42,"tunes":1300},{"text":1299,"level":240},"FAQ",{},{"id":725,"data":1302,"type":725,"tunes":1320},{"items":1303,"title":1319},[1304,1307,1310,1313,1316],{"id":729,"answer":1305,"question":1306},"Give the model a small set of known-correct evidence manually. If the answer becomes correct, investigate source coverage, query construction, retrieval, ranking, and context assembly. If the model still fails with sufficient evidence, retrieval is not the primary problem.","How can I tell whether RAG retrieval or the LLM failed?",{"id":733,"answer":1308,"question":1309},"Yes. The relevant passage can be ranked too low, truncated, separated from an exception, mixed with conflicting evidence, overwhelmed by irrelevant context, or used incorrectly by the generator.","Can RAG fail even when the correct document was retrieved?",{"id":737,"answer":1311,"question":1312},"No. A model can produce a correct answer despite weak retrieval by relying on prior model knowledge or chance. Evaluate retrieval and evidence support separately from final-answer correctness.","Is answer correctness enough to evaluate a RAG system?",{"id":741,"answer":1314,"question":1315},"At minimum log the user request, transformed retrieval query, filters, candidate documents and ranks, final selected context, model and prompt version, answer, citations, corpus\u002Findex version, and timing or version metadata relevant to freshness.","What should I log when debugging RAG?",{"id":745,"answer":1317,"question":1318},"Not reliably. A larger candidate or context set may improve recall, but it can also add noise, contradictions, duplicates, and context overload. Test whether the relevant evidence is missing before increasing top-k.","Does increasing top-k usually fix RAG?","RAG failure diagnosis",{},{"id":751,"data":1322,"type":42,"tunes":1324},{"text":1323,"level":240},"Glossary",{},{"id":756,"data":1326,"type":756,"tunes":1344},{"title":1327,"entries":1328},"Key diagnostic terms",[1329,1331,1333,1335,1338,1341],{"term":1077,"anchor":762,"definition":1330},"A controlled test in which the generator is given known-sufficient evidence directly to determine whether the dominant failure is upstream of generation.",{"term":1199,"anchor":765,"definition":1332},"The stage that selects an initial set of potentially relevant documents, chunks, records, or passages before final ranking or context assembly.",{"term":1207,"anchor":768,"definition":1334},"The process of converting retrieved evidence into the actual model input, including ordering, truncation, deduplication, formatting, and token-budget decisions.",{"term":1336,"anchor":772,"definition":1337},"Faithfulness","The degree to which generated claims remain supported by the retrieved or supplied evidence rather than introducing unsupported content.",{"term":1339,"anchor":776,"definition":1340},"Context coverage","A retrieval-oriented measure of whether selected evidence covers the information needed to answer the question.",{"term":1342,"anchor":780,"definition":1343},"Validity boundary","The conditions under which a claim or answer remains applicable, such as time, version, jurisdiction, state, population, permissions, or source assumptions.",{},{"id":784,"data":1346,"type":42,"tunes":1348},{"text":1347,"level":240},"Primary sources and further reading",{},{"id":789,"data":1350,"type":796,"tunes":1355},{"link":791,"meta":1351},{"image":1352,"title":1353,"description":1354},{"url":455},"OpenAI — Optimizing LLM Accuracy","OpenAI guidance separating retrieval failures from LLM failures in RAG applications.",{},{"id":799,"data":1357,"type":796,"tunes":1362},{"link":801,"meta":1358},{"image":1359,"title":1360,"description":1361},{"url":455},"OpenAI — Evaluation Best Practices","Guidance on structured evaluation for variable AI systems and production-oriented test design.",{},{"id":808,"data":1364,"type":796,"tunes":1369},{"link":810,"meta":1365},{"image":1366,"title":1367,"description":1368},{"url":455},"Amazon Bedrock — RAG Evaluation Metrics","Documentation separating retrieve-only metrics from retrieve-and-generate metrics, including context relevance, coverage, faithfulness and citation measures.",{},{"id":817,"data":1371,"type":796,"tunes":1376},{"link":819,"meta":1372},{"image":1373,"title":1374,"description":1375},{"url":455},"Anthropic — Demystifying Evals for AI Agents","Practical evaluation guidance on tasks, trials, graders, traces, regressions and production behaviour.",{},{"id":826,"data":1378,"type":796,"tunes":1383},{"link":828,"meta":1379},{"image":1380,"title":1381,"description":1382},{"url":455},"Google Cloud — Retrieval-Augmented Generation","Overview of RAG architecture and the importance of relevant retrieval and grounded generation.",{},"When a RAG answer is wrong, blaming retrieval or the model is too vague. This diagnostic method isolates source coverage, query construction, retrieval, ranking, context assembly, generation, evidence attribution, and freshness—so the actual failure can be reproduced and fixed.",{"lang":7,"title":208,"content":210,"contentJson":1386,"excerpt":835},{"time":212,"blocks":1387,"version":834},[1388,1391,1394,1397,1400,1403,1406,1409,1412,1415,1428,1431,1434,1437,1440,1443,1446,1449,1452,1455,1458,1461,1464,1467,1470,1473,1476,1479,1482,1485,1488,1491,1494,1497,1500,1503,1506,1509,1523,1526,1529,1541,1544,1547,1550,1553,1567,1570,1582,1585,1588,1591,1594,1597,1600,1603,1606,1609,1612,1615,1618,1621,1624,1627,1630,1633,1636,1645,1648,1658,1661,1666,1671,1676,1681],{"id":215,"data":1389,"type":218,"tunes":1390},{"text":217},{},{"id":221,"data":1392,"type":226,"tunes":1393},{"body":223,"title":224,"variant":225},{},{"id":229,"data":1395,"type":226,"tunes":1396},{"body":231,"title":232,"variant":233},{},{"id":236,"data":1398,"type":241,"tunes":1399},{"title":238,"maxLevel":239,"minLevel":240},{},{"id":244,"data":1401,"type":42,"tunes":1402},{"text":246,"level":240},{},{"id":249,"data":1404,"type":218,"tunes":1405},{"text":251},{},{"id":254,"data":1407,"type":218,"tunes":1408},{"text":256},{},{"id":259,"data":1410,"type":218,"tunes":1411},{"text":261},{},{"id":264,"data":1413,"type":42,"tunes":1414},{"text":266,"level":240},{},{"id":269,"data":1416,"type":308,"tunes":1427},{"content":1417,"stretched":43,"withHeadings":14},[1418,1419,1420,1421,1422,1423,1424,1425,1426],[273,274,275],[277,278,279],[281,282,283],[285,286,287],[289,290,291],[293,294,295],[297,298,299],[301,302,303],[305,306,307],{},{"id":311,"data":1429,"type":42,"tunes":1430},{"text":313,"level":239},{},{"id":316,"data":1432,"type":218,"tunes":1433},{"text":318},{},{"id":321,"data":1435,"type":218,"tunes":1436},{"text":323},{},{"id":326,"data":1438,"type":226,"tunes":1439},{"body":328,"title":329,"variant":330},{},{"id":333,"data":1441,"type":42,"tunes":1442},{"text":335,"level":239},{},{"id":338,"data":1444,"type":218,"tunes":1445},{"text":340},{},{"id":343,"data":1447,"type":218,"tunes":1448},{"text":345},{},{"id":348,"data":1450,"type":42,"tunes":1451},{"text":350,"level":239},{},{"id":353,"data":1453,"type":218,"tunes":1454},{"text":355},{},{"id":358,"data":1456,"type":218,"tunes":1457},{"text":360},{},{"id":363,"data":1459,"type":42,"tunes":1460},{"text":365,"level":239},{},{"id":368,"data":1462,"type":218,"tunes":1463},{"text":370},{},{"id":373,"data":1465,"type":218,"tunes":1466},{"text":375},{},{"id":378,"data":1468,"type":42,"tunes":1469},{"text":380,"level":239},{},{"id":383,"data":1471,"type":218,"tunes":1472},{"text":385},{},{"id":388,"data":1474,"type":218,"tunes":1475},{"text":390},{},{"id":393,"data":1477,"type":42,"tunes":1478},{"text":395,"level":239},{},{"id":398,"data":1480,"type":218,"tunes":1481},{"text":400},{},{"id":403,"data":1483,"type":218,"tunes":1484},{"text":405},{},{"id":408,"data":1486,"type":42,"tunes":1487},{"text":410,"level":239},{},{"id":413,"data":1489,"type":218,"tunes":1490},{"text":415},{},{"id":418,"data":1492,"type":218,"tunes":1493},{"text":420},{},{"id":423,"data":1495,"type":42,"tunes":1496},{"text":425,"level":239},{},{"id":428,"data":1498,"type":218,"tunes":1499},{"text":430},{},{"id":433,"data":1501,"type":218,"tunes":1502},{"text":435},{},{"id":438,"data":1504,"type":42,"tunes":1505},{"text":440,"level":240},{},{"id":443,"data":1507,"type":218,"tunes":1508},{"text":445},{},{"id":448,"data":1510,"type":475,"tunes":1522},{"rows":1511,"title":464,"layout":308,"columns":1518},[1512,1514,1516],{"id":452,"label":453,"values":1513},[455,455,455],{"id":457,"label":458,"values":1515},[455,455,455],{"id":461,"label":462,"values":1517},[455,455,455],[1519,1520,1521],{"id":467,"label":468},{"id":470,"label":471},{"id":473,"label":474},{},{"id":478,"data":1524,"type":226,"tunes":1525},{"body":480,"title":481,"variant":482},{},{"id":485,"data":1527,"type":42,"tunes":1528},{"text":487,"level":240},{},{"id":490,"data":1530,"type":519,"tunes":1540},{"steps":1531,"title":517,"orientation":518},[1532,1533,1534,1535,1536,1537,1538,1539],{"label":494,"description":495},{"label":497,"description":498},{"label":500,"description":501},{"label":503,"description":504},{"label":506,"description":507},{"label":509,"description":510},{"label":512,"description":513},{"label":515,"description":516},{},{"id":522,"data":1542,"type":42,"tunes":1543},{"text":524,"level":240},{},{"id":527,"data":1545,"type":218,"tunes":1546},{"text":529},{},{"id":532,"data":1548,"type":218,"tunes":1549},{"text":534},{},{"id":537,"data":1551,"type":42,"tunes":1552},{"text":539,"level":240},{},{"id":542,"data":1554,"type":308,"tunes":1566},{"content":1555,"stretched":43,"withHeadings":14},[1556,1557,1558,1559,1560,1561,1562,1563,1564,1565],[546,547,548],[550,551,552],[554,555,556],[558,559,560],[562,563,564],[566,567,568],[570,571,572],[574,575,576],[578,579,580],[582,583,584],{},{"id":587,"data":1568,"type":42,"tunes":1569},{"text":589,"level":240},{},{"id":592,"data":1571,"type":308,"tunes":1581},{"content":1572,"stretched":43,"withHeadings":14},[1573,1574,1575,1576,1577,1578,1579,1580],[273,596,597],[599,600,601],[603,604,605],[607,608,609],[611,612,613],[615,616,617],[619,620,621],[623,624,625],{},{"id":628,"data":1583,"type":42,"tunes":1584},{"text":630,"level":240},{},{"id":633,"data":1586,"type":218,"tunes":1587},{"text":635},{},{"id":638,"data":1589,"type":218,"tunes":1590},{"text":640},{},{"id":643,"data":1592,"type":649,"tunes":1593},{"url":645,"title":646,"excerpt":647,"ctaLabel":648},{},{"id":652,"data":1595,"type":42,"tunes":1596},{"text":654,"level":240},{},{"id":657,"data":1598,"type":218,"tunes":1599},{"text":659},{},{"id":662,"data":1601,"type":218,"tunes":1602},{"text":664},{},{"id":667,"data":1604,"type":649,"tunes":1605},{"url":669,"title":670,"excerpt":671,"ctaLabel":672},{},{"id":675,"data":1607,"type":42,"tunes":1608},{"text":677,"level":240},{},{"id":680,"data":1610,"type":218,"tunes":1611},{"text":682},{},{"id":685,"data":1613,"type":218,"tunes":1614},{"text":687},{},{"id":690,"data":1616,"type":42,"tunes":1617},{"text":692,"level":240},{},{"id":695,"data":1619,"type":218,"tunes":1620},{"text":697},{},{"id":700,"data":1622,"type":218,"tunes":1623},{"text":702},{},{"id":705,"data":1625,"type":42,"tunes":1626},{"text":707,"level":240},{},{"id":710,"data":1628,"type":218,"tunes":1629},{"text":712},{},{"id":715,"data":1631,"type":218,"tunes":1632},{"text":717},{},{"id":720,"data":1634,"type":42,"tunes":1635},{"text":722,"level":240},{},{"id":725,"data":1637,"type":725,"tunes":1644},{"items":1638,"title":748},[1639,1640,1641,1642,1643],{"id":729,"answer":730,"question":731},{"id":733,"answer":734,"question":735},{"id":737,"answer":738,"question":739},{"id":741,"answer":742,"question":743},{"id":745,"answer":746,"question":747},{},{"id":751,"data":1646,"type":42,"tunes":1647},{"text":753,"level":240},{},{"id":756,"data":1649,"type":756,"tunes":1657},{"title":758,"entries":1650},[1651,1652,1653,1654,1655,1656],{"term":761,"anchor":762,"definition":763},{"term":603,"anchor":765,"definition":766},{"term":611,"anchor":768,"definition":769},{"term":771,"anchor":772,"definition":773},{"term":775,"anchor":776,"definition":777},{"term":779,"anchor":780,"definition":781},{},{"id":784,"data":1659,"type":42,"tunes":1660},{"text":786,"level":240},{},{"id":789,"data":1662,"type":796,"tunes":1665},{"link":791,"meta":1663},{"image":1664,"title":794,"description":795},{"url":455},{},{"id":799,"data":1667,"type":796,"tunes":1670},{"link":801,"meta":1668},{"image":1669,"title":804,"description":805},{"url":455},{},{"id":808,"data":1672,"type":796,"tunes":1675},{"link":810,"meta":1673},{"image":1674,"title":813,"description":814},{"url":455},{},{"id":817,"data":1677,"type":796,"tunes":1680},{"link":819,"meta":1678},{"image":1679,"title":822,"description":823},{"url":455},{},{"id":826,"data":1682,"type":796,"tunes":1685},{"link":828,"meta":1683},{"image":1684,"title":831,"description":832},{"url":455},{},"Post erfolgreich abgerufen",{"items":1688,"source":1744,"manualIds":1745,"manualMatchedIds":1746},[1689,1696,1703,1710,1717,1724,1731,1737],{"id":1690,"slug":1691,"title":1692,"excerpt":1693,"featuredImage":1694,"publishedAt":1695},"468","ai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context","AI代理记忆不是RAG：如何区分记忆、检索、状态和上下文","代理记忆、RAG、状态和上下文经常被当作可以互换的概念来使用。它们并不是。这个实用的架构模型将这四个层次区分开来，展示了每一层各自应处的位置，并解释了当系统将它们合并为一层时会出现什么问题。","\u002Fuploads\u002F2026\u002F09\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context-1790350560308-np0xy6.webp","2026-09-25T11:34:00.000Z",{"id":1697,"slug":1698,"title":1699,"excerpt":1700,"featuredImage":1701,"publishedAt":1702},"459","ollama-is-not-the-product-building-production-ready-open-llm-applications","Ollama 并非产品：构建可投入生产的开源大语言模型应用","使用Ollama运行本地模型很简单。但构建一个可用于生产环境的开源大语言模型（Open-LLM）应用则更具挑战性：它需要RAG（检索增强生成）、访问控制、供应商抽象、评估、日志记录、部署规范，以及围绕模型构建受控的应用层。","\u002Fuploads\u002F2026\u002F06\u002Follama-is-not-the-product-building-production-ready-open-llm-applications-1782679361640-h0usqf.webp","2026-06-28T16:39:00.000Z",{"id":1704,"slug":1705,"title":1706,"excerpt":1707,"featuredImage":1708,"publishedAt":1709},"460","ai-agent-reliability-why-the-final-answer-is-not-enough","AI Agent可靠性：为什么最终答案并不足够","正确的输出并不能证明推理的正确性、执行的安全性，或系统的可信赖性。","\u002Fuploads\u002F2026\u002F09\u002Fai-agent-reliability-why-the-final-answer-is-not-enough-1788955466306-pl0qhz.webp","2026-09-09T04:01:00.000Z",{"id":1711,"slug":1712,"title":1713,"excerpt":1714,"featuredImage":1715,"publishedAt":1716},"475","managed-agent-harness-vs-self-hosted-agent-loop-what-you-gain-what-you-lose","托管代理框架与自托管代理循环：你得到什么，失去什么","“自托管代理”可能意味着截然不同的架构。本指南区分了托管式运行框架、自托管执行环境和完全自主运营的代理循环——并说明了团队实际需要哪种控制边界。","\u002Fuploads\u002F2026\u002F09\u002Fmanaged-agent-harness-vs-self-hosted-agent-loop-what-you-gain-what-you-lose-1790352403475-kj10jh.webp","2026-09-25T12:05:00.000Z",{"id":1718,"slug":1719,"title":1720,"excerpt":1721,"featuredImage":1722,"publishedAt":1723},"364","tipps-fuer-die-verbesserung-der-seo-suchmaschinenoptimierung","Mastering the SEO Workflow: Essential Optimization Strategies for Organic Growth","A structured SEO workflow is crucial for sustainable organic growth. Learn the ten foundational strategies, from keyword research and technical optimization to content quality and performance analysis.","\u002Fuploads\u002F2026\u002F03\u002Ftipps-fuer-die-verbesserung-der-seo-suchmaschinenoptimierung-1774866098131-hwkzrg.webp","2024-01-26T06:35:00.000Z",{"id":1725,"slug":1726,"title":1727,"excerpt":1728,"featuredImage":1729,"publishedAt":1730},"477","computer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system","计算机使用代理：为什么成功的演示仍可能是一个不可靠的系统","计算机使用代理如今能够完成令人印象深刻的浏览器和桌面工作流程，但一次成功的运行证明的是能力——而非可靠性。本文展示了如何测试可重复性、环境鲁棒性、长时程控制、状态感知、结果验证以及安全的目标处理。","\u002Fuploads\u002F2026\u002F09\u002Fcomputer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system-1790352854690-75qnrg.webp","2026-09-25T12:13:00.000Z",{"id":1732,"slug":859,"title":1733,"excerpt":1734,"featuredImage":1735,"publishedAt":1736},"434","全面评估指南：精通LLM性能评估","本指南详细介绍了评估工具（Evaluation Harness），这是一个在企业级LLMOps流程中严格评估大型语言模型（LLM）能力的关键框架。您将学习其设置方法、最佳实践以及高级技巧，以确保模型基准测试与优化的可靠性。","\u002Fuploads\u002F2026\u002F04\u002Fevaluation-harness-1775466944495-4s0xv2.webp","2026-03-01T17:50:00.000Z",{"id":1738,"slug":1739,"title":1740,"excerpt":1741,"featuredImage":1742,"publishedAt":1743},"478","what-is-rag-the-simplest-explanation-of-how-it-works","什么是RAG？对其工作原理的最简单解释","RAG听起来很复杂，但想法很简单：在AI回答之前，它先从知识源查找有用的信息，并将该信息提供给语言模型。本指南使用一个简单的思维模型来解释RAG、LLM、状态、记忆和工具。","\u002Fuploads\u002F2026\u002F09\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works-1790377492124-khjagt.webp","2026-09-25T19:03:00.000Z","fallback",[],[]]