[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"portal-settings:stajic:zh":3,"public-menus:all":38,"post:how-to-know-whether-an-ai-agent-actually-used-the-right-evidence:zh":205,"related:post:how-to-know-whether-an-ai-agent-actually-used-the-right-evidence:zh:1":1718},{"statusCode":4,"data":5,"message":37},200,{"tenantId":6,"lang":7,"defaultLang":8,"siteUrl":9,"contactEmail":10,"brandName":11,"logoUrl":12,"siteName":11,"siteDescription":13,"ogImage":10,"robotsIndex":14,"socialLinks":10,"reservedSlugs":10,"seoPolicy":15},"stajic","zh","de","https:\u002F\u002Fstajic.de",null,"Stajic Platform","\u002FLogo_Planet.svg","Stajic Portal",true,{"branding":16,"relatedContent":17,"crossDomainLinks":18},{"logoUrl":12},{"enabled":14},[19,22,25,28,31,34],{"url":20,"label":21,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Ffigure.rocks","figure.rocks",{"url":23,"label":24,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Floving.rocks","loving.rocks",{"url":26,"label":27,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.com","bazify.com",{"url":29,"label":30,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.de","bazify.de",{"url":32,"label":33,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.at","bazify.at",{"url":35,"label":36,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.ba","bazify.ba","Portal settings resolved",[39,45],{"id":40,"name":41,"location":42,"isActive":14,"isDefault":43,"items":44},1,"main-navigation","header",false,[],{"id":46,"name":47,"location":48,"isActive":14,"isDefault":14,"items":49},4,"main-menu","sidebar",[50,66,79,93,103,118,133],{"id":51,"title":52,"url":60,"target":61,"icon":62,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":64,"portfolioId":10,"children":65},"item-18",{"de":53,"en":54,"es":55,"fr":56,"it":54,"ru":57,"sr":58,"zh":59},"Startseite","Home","Inicio","Accueil","Главная","Почетна","首页","\u002Ffull-stack-web-developer-munich-performance-seo-and-maintainable-builds","_self","i-lucide-home","page",111,[],{"id":67,"title":68,"url":75,"target":61,"icon":76,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":77,"portfolioId":10,"children":78},"item-22",{"de":69,"en":69,"es":70,"fr":69,"it":71,"ru":72,"sr":73,"zh":74},"Vision","Visión","Visione","Видение","Визија","想象","\u002Fueber-uns-webdesign-muenchen-webaplikation","i-lucide-eye",113,[],{"id":80,"title":81,"url":89,"target":61,"icon":90,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":91,"portfolioId":10,"children":92},"item-19",{"de":82,"en":83,"es":84,"fr":83,"it":85,"ru":86,"sr":87,"zh":88},"Leistungen","Services","Servicios","Servizi","Услуги","Услуге","服务","\u002Fservices-dienstleistungen-muenchen","i-lucide-wrench",116,[],{"id":94,"title":95,"url":99,"target":61,"icon":100,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":101,"portfolioId":10,"children":102},"item-23",{"de":96,"en":96,"es":96,"fr":96,"it":96,"ru":97,"sr":97,"zh":98},"Blog","Блог","博客","\u002Fblog","i-lucide-book-open",112,[],{"id":104,"title":105,"url":114,"target":61,"icon":115,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":116,"portfolioId":10,"children":117},"item-32",{"de":106,"en":107,"es":108,"fr":109,"it":110,"ru":111,"sr":112,"zh":113},"Neue Technologien","New Technologies","Nuevas tecnologías","Nouvelles technologies","Nuove tecnologie","Новые технологии","Нове технологије","新技术！","\u002Fneue-webtechnologien","i-lucide-sparkles",122,[],{"id":119,"title":120,"url":129,"target":61,"icon":130,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":131,"portfolioId":10,"children":132},"item-20",{"de":121,"en":122,"es":123,"fr":124,"it":125,"ru":126,"sr":127,"zh":128},"Kontakt","Contact us!","Contacto","Contact","Contatto","Контакт","Контактирајте нас","联系我们！","\u002Fcontact","i-lucide-mail",115,[],{"id":134,"title":135,"url":144,"target":61,"icon":145,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":147},"item-21",{"de":136,"en":137,"es":138,"fr":139,"it":140,"ru":141,"sr":142,"zh":143},"Unsere Arbeit","Our Work","Nuestro trabajo","Nos réalisations","I nostri lavori","Наши работы","Наши радови","文件夹","\u002Fportfolio","i-lucide-briefcase",114,[148,161,175,181,193],{"id":149,"title":150,"url":144,"target":61,"icon":159,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":160},"item-24",{"de":151,"en":152,"es":153,"fr":154,"it":155,"ru":156,"sr":157,"zh":158},"Alle Projekte","All Projects","Todos los proyectos","Tous les projets","Tutti i progetti","Все проекты","Сви пројекти","所有项目","i-lucide-grid-3x3",[],{"id":162,"title":163,"url":171,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":174},"item-29",{"de":164,"en":165,"es":166,"fr":167,"it":168,"ru":169,"sr":170,"zh":143},"Local Roots, Global Reach","Local Roots - Global Reach","Empresa local ","Entreprise locale","Azienda locale","Местная компания","Локално предузеће глобално тржиште","\u002Fportfolio\u002Flocal-roots-global-reach-communication-media-systems-for-modern-business","i-lucide-folder","custom",[],{"id":176,"title":177,"url":179,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":180},"item-28",{"de":178,"en":178,"es":178,"fr":178,"it":178,"ru":178,"sr":178,"zh":178},"Solr Suggester","\u002Fportfolio\u002Fsolr-fuzzy-suggester-und-solr-infix-suggester-abfrage-ueber-ajax-und-filterung",[],{"id":182,"title":183,"url":191,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":192},"item-27",{"de":184,"en":185,"es":186,"fr":187,"it":188,"ru":189,"sr":190,"zh":185},"Firmenwebseite SEO","Company Website SEO","Sitio web corporativo SEO","Site web d’entreprise SEO","Sito web aziendale SEO","Корпоративный сайт SEO","Пословна веб-страница SEO","\u002Fportfolio\u002Fseo-sem-branding-mobile-webseite-muenchen",[],{"id":194,"title":195,"url":203,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":204},"item-31",{"de":196,"en":197,"es":198,"fr":199,"it":200,"ru":201,"sr":202,"zh":197},"Digitalisierungsportal","Digitalization Portal","Portal de digitalización","Portail de numérisation","Portale di digitalizzazione","Портал цифровизации","Портал за дигитализацију","\u002Fportfolio\u002Fdigitalisierungsportal-archiv-museum-bibliothek-ead-lido-mets-mods",[],{"statusCode":4,"data":206,"message":1717},{"id":207,"title":208,"slug":209,"content":210,"contentJson":211,"excerpt":851,"featuredImage":852,"featuredImageAlt":853,"featuredImageCaption":10,"featuredImageTitle":10,"featuredImageCopyright":10,"featuredImageAuthor":10,"featuredImageSourceUrl":10,"featuredImageLicense":10,"featuredImageIsAiGenerated":43,"status":854,"publishedAt":855,"createdAt":856,"updatedAt":857,"seoLocalePaths":858,"categories":867,"author":880,"translations":885},"471","如何判断一个AI智能体是否真正使用了正确的证据","how-to-know-whether-an-ai-agent-actually-used-the-right-evidence","\u003Cnav class=\"editorjs-toc\" data-editorjs-toc=\"true\" aria-label=\"目录\">\u003Cstrong class=\"editorjs-toc__title\">目录\u003C\u002Fstrong>\u003Col class=\"editorjs-toc__list editorjs-toc__list--depth-0\">\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-5\" class=\"editorjs-toc__link\">为什么仅有引用还不够\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-8\" class=\"editorjs-toc__link\">每一项重要主张都应通过的四个问题\u003C\u002Fa>\u003Col class=\"editorjs-toc__list editorjs-toc__list--depth-1\">\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-10\" class=\"editorjs-toc__link\">1. 主张支持：来源是否证明了智能体所说的话？\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-13\" class=\"editorjs-toc__link\">2. 证据权威性：这是正确类型的来源吗？\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-16\" class=\"editorjs-toc__link\">3. 适用性：证据正确，条件错误\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-20\" class=\"editorjs-toc__link\">4. 证据使用：智能体是否真正依赖了证据？\u003C\u002Fa>\u003C\u002Fli>\u003C\u002Fol>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-23\" class=\"editorjs-toc__link\">证据利用测试\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-27\" class=\"editorjs-toc__link\">主张–证据矩阵比来源列表更有用\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-30\" class=\"editorjs-toc__link\">将检索质量与证据质量分开\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-33\" class=\"editorjs-toc__link\">将来源质量作为评分标准来评估，而不是域名白名单\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-36\" class=\"editorjs-toc__link\">证据覆盖范围：每个重要主张都需要支持，而不是每句话\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-39\" class=\"editorjs-toc__link\">证据来源必须在摘要化和记忆化后仍然保留\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-42\" class=\"editorjs-toc__link\">实用的证据记录\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-44\" class=\"editorjs-toc__link\">看似有依据但实际上没有的失败模式\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-46\" class=\"editorjs-toc__link\">如何在生产环境中评估智能体\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-49\" class=\"editorjs-toc__link\">不要让 LLM 评判者成为唯一的证据评判者\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-52\" class=\"editorjs-toc__link\">什么会改变这个答案？\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-56\" class=\"editorjs-toc__link\">局限性\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-59\" class=\"editorjs-toc__link\">结论\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-64\" class=\"editorjs-toc__link\">常见问题\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-66\" class=\"editorjs-toc__link\">术语表\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-68\" class=\"editorjs-toc__link\">主要来源与延伸阅读\u003C\u002Fa>\u003C\u002Fli>\u003C\u002Fol>\u003C\u002Fnav>\n\u003Cp>AI 智能体可以引用来源、检索文档，却仍然使用错误的证据。一个来源可能具有权威性，但与具体主张无关。一段检索到的文本可能只支持答案的一部分。一个正确的来源可能已经过时、被取代，或者只适用于错误的司法管辖区、产品版本、用户或系统状态。这就产生了一个比简单引用检查更难的评估问题：智能体是否真的为其所提出的主张使用了正确的证据？\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--info my-6 rounded-xl border p-5 border-blue-300 bg-blue-50 dark:border-blue-900 dark:bg-blue-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">直接回答\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">要判断 AI 智能体是否使用了正确的证据，应评估这条链：&lt;strong&gt;主张 → 证据 → 适用性 → 使用&lt;\u002Fstrong&gt;。对于每一项重要主张，都要验证证据是否直接支持它、是否来自适当的权威来源、是否适用于当前条件，以及在主张产生之前智能体是否实际能够获取该证据。仅凭一条引用无法证明其中任何一点。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Caside class=\"editorjs-callout editorjs-callout--note my-6 rounded-xl border p-5 border-gray-300 bg-gray-50 dark:border-gray-700 dark:bg-gray-900\u002F40\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">关于该方法\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">本文中的“主张—证据—适用性—使用”模型和证据利用测试是实用的评估方法，并非正式的行业标准。它们建立在诸如扎根性、来源质量、引用覆盖率、轨迹评估和任务特定评估等已有理念之上。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-5\">为什么仅有引用还不够\u003C\u002Fh2>\n\u003Cp>引用只能回答一个狭窄的问题：系统将某个主张或回复与某个来源关联了起来。它并不能自动证明该来源支持这一具体主张、该来源对该任务足够权威、被引用的文本包含必要条件或例外情况，或者模型依赖了该证据，而不是根据先前的模型知识生成答案。\u003C\u002Fp>\n\u003Cp>Anthropic 关于研究型智能体评估的指南明确区分了扎根性、覆盖率和来源质量。OpenAI 的智能体评估指南同样强调轨迹，因为最终输出无法揭示智能体是否选择了正确的工具或遵循了预期的工作流程。这些观点指向一个更广泛的结论：证据质量是执行路径的属性，而不仅仅是最终文本的属性。\u003C\u002Fp>\n\u003Ch2 id=\"section-8\">每一项重要主张都应通过的四个问题\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">维度\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">问题\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">典型失败\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">主张支持\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">证据是否直接支持这一确切主张？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">来源在主题上相关，但并不能证明该陈述\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">证据权威性\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">对于这类主张，这是否是适当的来源？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">在需要一手来源或实时系统的地方使用了二手摘要\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">适用性\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">证据是否适用于该时间、版本、司法管辖区、用户、状态或人群？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">一个真实的陈述被应用到了其有效条件之外\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">证据使用\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">该证据在智能体的执行路径中是否实际可用并被使用？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">最终答案是正确的，但检索到的证据无关或未被使用\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch3 id=\"section-10\">1. 主张支持：来源是否证明了智能体所说的话？\u003C\u002Fh3>\n\u003Cp>证据应在主张层面进行评估。一份文档可能与主题相关，却仍然无法支持某个具体陈述。如果来源说某项功能仅在部分区域可用，那么“该功能在全球可用”这一答案就是没有依据的，即使引用看起来似乎合理。\u003C\u002Fp>\n\u003Cp>这正是宽泛的“有依据 \u002F 无依据”判断往往过于粗糙的地方。将回复拆分为重要主张，把每项主张映射到支持它的最小证据片段，并对关系进行分类：直接支持、部分支持、矛盾或无支持。\u003C\u002Fp>\n\u003Ch3 id=\"section-13\">2. 证据权威性：这是正确类型的来源吗？\u003C\u002Fh3>\n\u003Cp>正确的证据选择不仅仅是语义相关性。来源必须适合该决策。当前账户状态应来自账户系统，而不是一封旧邮件。关于 API 行为的主张最好应依据当前供应商文档或可复现行为来核查。法律要求可能需要适用法律、监管机构或权威指南，而不是一篇泛泛的博客文章。\u003C\u002Fp>\n\u003Cp>来源权威性是任务特定的。对于供应商文档未承认的真实世界漏洞，社区报告可能是最佳证据。供应商公告对于供应商所声称的内容可能具有权威性，但对于独立性能而言则是弱证据。因此，评估者需要为任务制定明确的来源层级，而不是使用一个通用的权威评分。\u003C\u002Fp>\n\u003Ch3 id=\"section-16\">3. 适用性：证据正确，条件错误\u003C\u002Fh3>\n\u003Cp>最危险的证据错误往往不是伪造来源，而是有效来源被用在其边界之外。建议可能会随软件版本、日期、司法管辖区、硬件修订版、用户权限、产品可用性、当前游戏状态、租户配置或其他环境变量而变化。\u003C\u002Fp>\n\u003Cp>对于每个重要来源，都要保留决定其是否仍然适用的条件。这在摘要之后尤其重要：压缩后的记忆或引用可能保留了结论，却丢掉了使该结论有效的例外、日期或前提条件。\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--warning my-6 rounded-xl border p-5 border-amber-300 bg-amber-50 dark:border-amber-900 dark:bg-amber-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">证据可以是真实的，但对答案来说仍然是错误的\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">一个真实的来源，即使被准确引用，当其时间、版本、人群、管辖范围或状态与当前问题不匹配时，仍然可能产生错误的答案。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch3 id=\"section-20\">4. 证据使用：智能体是否真正依赖了证据？\u003C\u002Fh3>\n\u003Cp>即使检索失败，答案也可能是正确的。模型可能已经知道答案，从不相关的上下文中推断出答案，或者只是碰巧猜对了。如果评估只检查最终正确性，系统可能看起来有充分依据，而证据路径却是断裂的。\u003C\u002Fp>\n\u003Cp>要评估证据使用情况，需要检查追踪记录。确认检索到了哪些来源、哪些段落进入了模型上下文、它们何时可用，以及最终主张能否由这些输入解释。OpenAI 当前的智能体评估工具强调追踪评分，正是因为仅凭最终答案无法可靠地重建工作流层面的行为。\u003C\u002Fp>\n\u003Ch2 id=\"section-23\">证据利用测试\u003C\u002Fh2>\n\u003Cp>一种实用的评估可以构建为受控的反事实测试。不要只问答案是否正确，而是改变证据并观察主张是否按预期方向变化。\u003C\u002Fp>\n\u003Csection class=\"editorjs-process my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">证据利用测试\u003C\u002Fh3>\u003Cdiv class=\"grid grid-cols-1 md:grid-cols-2 xl:grid-cols-3 gap-4\">\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">1\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">1. 选择一个重要主张\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">选择一个正确性至关重要的主张，并精确定义预期答案。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">2\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">2. 确定黄金证据\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">提供足以支持该主张的最小权威证据集。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">3\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">3. 使用黄金证据运行\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">验证当正确证据可用时，智能体是否产生受支持的答案。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">4\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">4. 移除决定性证据\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">在保持其他输入稳定的情况下，移除关键支持段落并运行相同任务。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">5\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">5. 用矛盾或取代性证据替换它\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">在安全的情况下，提供能够改变正确结论的受控证据。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">6\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">6. 比较主张\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">检查答案是否跟随证据变化，还是仍然锚定于先前的模型知识。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">7\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">7. 检查追踪记录\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">确认检索到了什么、什么进入了上下文，以及主张之前是哪个来源或工具结果。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Caside class=\"editorjs-callout editorjs-callout--tip my-6 rounded-xl border p-5 border-violet-300 bg-violet-50 dark:border-violet-900 dark:bg-violet-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">关键信号\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">如果决定性证据发生了变化，但智能体的主张没有变化，那么你有证据表明系统可能没有按预期使用检索——即使原始答案恰好是正确的。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-27\">主张–证据矩阵比来源列表更有用\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">主张\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">证据\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">支持\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">权威性\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">适用性\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">追踪中是否使用\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">功能 X 可用\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">供应商文档\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">直接\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">对可用性主张具有高权威性\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">当前版本和地区必须匹配\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">是 \u002F 否\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">配置 Y 更快\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">供应商基准测试\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">部分\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">对供应商的测试具有高权威性，但不代表独立性能\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">硬件和工作负载必须匹配\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">是 \u002F 否\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">政策适用于此用户\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">当前政策 + 账户状态\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">仅组合时直接\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">高\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">管辖范围、日期、角色和账户状态必须匹配\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">是 \u002F 否\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">某产品有库存\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">实时库存 API\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">直接\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">对当前库存具有权威性\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">很快过期\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">是 \u002F 否\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Cp>这个矩阵迫使人们提出几个传统引用检查所隐藏的问题。一个主张可能需要多个来源。一个来源可能只支持主张的一部分。一个权威来源可能只有很短的有效期。而一个完全良好的来源，如果从未进入执行路径，也可能无关紧要。\u003C\u002Fp>\n\u003Ch2 id=\"section-30\">将检索质量与证据质量分开\u003C\u002Fh2>\n\u003Cp>检索指标询问是否找到了相关材料并进行了排序。证据评估询问这些材料是否证明了由此产生的主张。两者相关但并不相同。\u003C\u002Fp>\n\u003Csection class=\"editorjs-comparison my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">检索成功不等于证据成功\u003C\u002Fh3>\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left dark:border-gray-700 dark:bg-gray-900\">\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">情况\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">检索\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">证据质量\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">正确文档，错误主张\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">正确事实，过时来源\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">弱来源，正确答案\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">需要多个来源\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-33\">将来源质量作为评分标准来评估，而不是域名白名单\u003C\u002Fh2>\n\u003Cp>硬编码的“可信域名”列表很诱人，但往往很脆弱。来源质量应改为反映主张类型。有用的维度包括一手与二手地位、时效性、直接性、可复现性、独立性、领域专业知识、数据来源、更新频率，以及来源是否有夸大主张的动机。\u003C\u002Fp>\n\u003Cp>Anthropic 的研究智能体评估指南明确将来源质量检查与有据性和覆盖范围并列提出。因此，实际实施应同时评估来源说了什么，以及该来源是否适合这类陈述。\u003C\u002Fp>\n\u003Ch2 id=\"section-36\">证据覆盖范围：每个重要主张都需要支持，而不是每句话\u003C\u002Fh2>\n\u003Cp>并非每一句话都需要引用。过渡性语言、从引用值透明推导出的算术结果，或明确标注的解释，可能不需要单独的来源。但每一个实质性的、可外部验证的断言都应有足够的支撑，使评估者能够重建为什么该智能体被允许这样说。\u003C\u002Fp>\n\u003Cp>因此，覆盖度应按断言重要性加权。装饰性细节缺少支撑，与价格、资格决定、安全说明、法律要求、技术兼容性声明或驱动推荐的事实缺少支撑，并不等同。\u003C\u002Fp>\n\u003Ch2 id=\"section-39\">证据来源必须在摘要化和记忆化后仍然保留\u003C\u002Fh2>\n\u003Cp>长期运行的智能体常常会总结先前的工作或写入持久记忆。如果证据来源在这一转换过程中被剥离，未来的智能体可能检索到一个干净的结论，却不知道它来自用户陈述、实时 API、旧文档、模型推断，还是未经核实的网络结果。\u003C\u002Fp>\n\u003Cp>对于重要事实，至少应保留来源身份、检索或观察时间、证据类型、相关版本或状态，以及所存储文本是引用、摘要、推断还是派生。来源信息使后续智能体能够判断该证据应被信任、刷新、限制还是丢弃。\u003C\u002Fp>\n\u003Ch2 id=\"section-42\">实用的证据记录\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">字段\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">用途\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">claim_id\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">标识被支撑的实质性断言\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">source_id \u002F source_url \u002F system\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">标识证据来自何处\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">evidence_span\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">保留支撑该断言的最小段落、记录或工具结果\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">retrieved_at \u002F observed_at\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">允许进行新鲜度和时间线检查\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">source_version \u002F object_version\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">允许进行取代和可复现性检查\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">authority_role\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">解释为什么该来源适合该断言\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">applicability\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">存储相关日期、司法管辖区、产品版本、用户、租户、状态或其他条件\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">transformation\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">标记证据是原始、引用、摘要、规范化还是派生\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">trace_step\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">显示证据何时对智能体可用\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">support_status\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">直接、部分、矛盾、无支撑或不确定\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-44\">看似有依据但实际上没有的失败模式\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">失败模式\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">为什么它能骗过评估者\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">应测试什么\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">引用装饰\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">答案包含来源，因此看起来经过研究\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">将每个实质性断言映射到精确的支撑片段\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">权威不匹配\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">来源声誉良好，但对具体事实并不权威\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">定义针对具体断言的来源层级\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">时间不匹配\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">来源在发布时是正确的\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">检查检索时间、来源日期和取代性证据\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">条件剥离\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">摘要保留了结论，却丢弃了例外\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">将生成的断言与完整的本地来源上下文进行比较\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">事后引用\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">在答案生成后才附上一个看似合理的来源\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">检查追踪顺序以及证据是否先于断言出现\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">参数化覆盖\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型忽略检索到的证据，依据先验知识作答\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">运行反事实证据利用测试\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">证据洗白\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型推断被摘要化，随后被当作来源事实存储\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">在记忆写入过程中保留转换类型和来源信息\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">来源多数谬误\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">多个二手页面重复同一无支撑陈述\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">将断言追溯至独立或一手证据\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-46\">如何在生产环境中评估智能体\u003C\u002Fh2>\n\u003Csection class=\"editorjs-process my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">证据评估流水线\u003C\u002Fh3>\u003Cdiv class=\"grid grid-cols-1 md:grid-cols-2 xl:grid-cols-3 gap-4\">\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">1\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">1. 定义实质性断言\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">识别对任务而言正确性重要的事实、建议或决定。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">2\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">2. 构建黄金证据\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">为代表性案例创建参考证据和来源质量预期。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">3\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">3. 捕获追踪\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">记录检索查询、工具调用、返回证据、上下文构建、模型输出和引用。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">4\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">4. 评定支撑度\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">检查每个实质性断言是直接支撑、部分支撑、矛盾支撑还是无支撑。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">5\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">5. 评定权威性和适用性\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">评估来源是否合适，以及其条件是否与当前任务匹配。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">6\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">6. 运行反事实\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">移除、替换或取代决定性证据，并测试答案是否随变化而改变。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">7\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">7. 审查高影响失败\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">在自动评分不够可靠时，使用人工或领域专家审查。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">8\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">8. 将失败转化为评估案例\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">将生产失败和边缘案例加入可重复的回归数据集。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Cp>OpenAI 当前的评估指南建议采用任务特定评估、持续评估、源自生产的数据集，以及用于调试智能体行为的追踪。Anthropic 同样建议为研究型智能体组合多种评分器类型，因为正确性、来源质量、覆盖度和有据性是不同的维度。证据评估应遵循同样的模式：多个窄范围评分器比一个不透明的“质量”分数更具诊断价值。\u003C\u002Fp>\n\u003Ch2 id=\"section-49\">不要让 LLM 评判者成为唯一的证据评判者\u003C\u002Fh2>\n\u003Cp>LLM 评分器可用于可扩展的断言分类、相关性检查和成对比较，但它们可能与其所评估的系统共享同样的盲点。评分器可能接受一个看似合理但无支撑的断言，漏掉微妙的版本边界，或高估一个经过润色的来源。\u003C\u002Fp>\n\u003Cp>OpenAI 的评估指南建议根据人工判断校准自动评分器，并使用清晰、有范围的准则。对于证据密集型系统，应尽可能使用确定性检查：时间戳、对象版本、权限范围、精确来源 ID、检索顺序、文档哈希，以及证据是否在模型生成断言之前就已存在。\u003C\u002Fp>\n\u003Ch2 id=\"section-52\">什么会改变这个答案？\u003C\u002Fh2>\n\u003Cp>当智能体在小型、不可变、权威的语料库上运行，且每个答案都严格是抽取式的时候，评估可以更简单。在这种环境中，来源权威性和适用性大多是固定的，断言到片段的支撑可能就足够了。\u003C\u002Fp>\n\u003Cp>当智能体混合使用网络搜索、长期记忆、实时工具、多个司法管辖区、快速变化的信息、用户特定状态或自主行动时，评估必须变得更严格。在这些系统中，证据有效性不仅取决于来源文本，还取决于证据是何时以及如何获得的。\u003C\u002Fp>\n\u003Cp>未来的模型可能会更擅长在内部追踪来源和不确定性，但在需要可审计性的系统中，这并不能消除对外部证据记录的需求。当轨迹和来源元数据能够提供更强证据时，系统不应依赖模型对自身受何影响的自我报告。\u003C\u002Fp>\n\u003Ch2 id=\"section-56\">局限性\u003C\u002Fh2>\n\u003Cp>仅凭轨迹并不总能证明因果性的证据使用。一个来源可能存在于上下文中，却没有影响答案，而模型也可能独立地知道同一事实。反事实测试能增强推断，但测试本身也可能改变任务分布。\u003C\u002Fp>\n\u003Cp>来源权威性也可能存在争议，或取决于领域。有些问题没有单一的权威来源，专家也可能对哪些证据应获得更大权重存在分歧。在这些情况下，评估者应保留分歧，并依据明确的评分标准来评估透明度、覆盖范围和推理，而不是假装存在一个不容置疑的真理来源。\u003C\u002Fp>\n\u003Ch2 id=\"section-59\">结论\u003C\u002Fh2>\n\u003Cp>“智能体是否引用了来源？”这个问题对于生产级 AI 来说太弱了。更强的问题是：每个重要主张是否都来自真正支持它的证据，该证据是否具有正确的权威性，是否仍适用于当前条件，并且在主张提出之前是否已存在于执行路径中？\u003C\u002Fp>\n\u003Cp>这使证据从装饰变成了可评估的系统属性。捕获轨迹。将主张映射到证据。检查权威性和适用性。运行反事实证据测试。在摘要和记忆中保留来源信息。这样，正确答案不仅看似合理——它还拥有一条你可以检查的证据路径。\u003C\u002Fp>\n\u003Caside class=\"editorjs-referral my-6\">\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough\" class=\"flex flex-col sm:flex-row gap-4 rounded-xl border border-gray-200 dark:border-gray-700 p-4 transition hover:border-primary-500\">\u003Cdiv class=\"min-w-0 flex-1\">\u003Cstrong class=\"block text-lg text-gray-900 dark:text-gray-100\">AI 智能体可靠性：为什么仅有最终答案还不够\u003C\u002Fstrong>\u003Cp class=\"mt-2 text-sm text-gray-600 dark:text-gray-300\">仅凭结果正确并不能证明智能体的执行路径是安全或可靠的。这篇相关文章解释了为什么轨迹、工具和中间决策很重要。\u003C\u002Fp>\u003Cspan class=\"mt-3 inline-flex text-sm font-medium text-primary-600 dark:text-primary-400\">阅读相关文章 →\u003C\u002Fspan>\u003C\u002Fdiv>\u003C\u002Fa>\u003C\u002Faside>\n\u003Caside class=\"editorjs-referral my-6\">\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Ffrom-research-protocol-to-a-general-ai-reasoning-framework\" class=\"flex flex-col sm:flex-row gap-4 rounded-xl border border-gray-200 dark:border-gray-700 p-4 transition hover:border-primary-500\">\u003Cdiv class=\"min-w-0 flex-1\">\u003Cstrong class=\"block text-lg text-gray-900 dark:text-gray-100\">从研究协议到通用 AI 推理框架\u003C\u002Fstrong>\u003Cp class=\"mt-2 text-sm text-gray-600 dark:text-gray-300\">一种实用的推理方法，用于区分证据、假设、竞争性假设和特定领域的验证。\u003C\u002Fp>\u003Cspan class=\"mt-3 inline-flex text-sm font-medium text-primary-600 dark:text-primary-400\">阅读推理框架 →\u003C\u002Fspan>\u003C\u002Fdiv>\u003C\u002Fa>\u003C\u002Faside>\n\u003Ch2 id=\"section-64\">常见问题\u003C\u002Fh2>\n\u003Csection class=\"editorjs-faq my-6 rounded-xl border border-gray-200 p-5 dark:border-gray-700\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">评估 AI 智能体中的证据使用\u003C\u002Fh3>\u003Cdiv id=\"faq1\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">引用是否能证明 AI 答案有依据？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">不能。引用可能与主题相关，但并不支持确切主张；可能来自错误的权威来源；可能不再适用；或者可能只是被附加上去，却并未实质影响生成的答案。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq2\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">如何测试 AI 智能体是否实际使用了检索到的证据？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">使用反事实证据利用测试：先用已知正确的证据运行任务，然后在保持其他输入稳定的情况下移除或替换决定性证据。如果答案没有对证据变化作出反应，就检查模型是否依赖先验知识或其他来源。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq3\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">有依据性和来源质量有什么区别？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">有依据性问的是主张是否由所提供的证据支持。来源质量问的是证据本身对于所提出的主张类型是否足够适当且具有权威性。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq4\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">为什么真实来源仍可能导致错误的 AI 答案？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">来源可能已过时、已被取代，或仅对另一个版本、司法管辖区、用户、人群或系统状态有效，也可能包含在检索或摘要过程中丢失的条件。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq5\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">为了评估证据，我应该记录什么？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">记录检索查询、返回的来源、确切证据片段、时间戳和版本、过滤器、最终上下文、模型输出、引用以及轨迹顺序，以便评估者能够重建在每个重要主张之前有哪些证据可用。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-66\">术语表\u003C\u002Fh2>\n\u003Csection class=\"editorjs-glossary my-6 rounded-xl border border-gray-200 dark:border-gray-700 p-5\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">关键证据评估术语\u003C\u002Fh3>\u003Cdl>\u003Cdiv id=\"claim-support\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">主张支持\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">特定证据片段直接确立生成主张的程度。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"evidence-authority\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">证据权威性\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">考虑到来源的角色、出处及其与底层事实的关系，该来源在确立某一类主张时有多适当。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"applicability\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">适用性\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">证据对某一主张仍然有效的条件，包括时间、版本、司法管辖区、用户、人群、系统状态或其他边界。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"evidence-utilization\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">证据利用\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">智能体的输出是否实际响应并依赖于其执行路径中可用的证据。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"counterfactual-evidence-test\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">反事实证据测试\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">一种评估方法，通过移除、替换或更改决定性证据，来测试智能体的主张是否会相应变化。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"provenance\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">来源信息\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">记录证据来自何处、何时获取、如何被转换以及它代表哪个版本或状态的元数据。\u003C\u002Fdd>\u003C\u002Fdiv>\u003C\u002Fdl>\u003C\u002Fsection>\n\u003Ch2 id=\"section-68\">主要来源与延伸阅读\u003C\u002Fh2>\n\u003Ca href=\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fagent-evals\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">OpenAI — 评估智能体工作流\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">关于轨迹评分、工作流级评估、数据集以及可重复的智能体评估运行的指导。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fevaluation-best-practices\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">OpenAI — 评估最佳实践\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">关于特定任务评估、源自生产的数据集、限定范围的指标、持续评估和评分器校准的指导。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Fdemystifying-evals-for-ai-agents\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">Anthropic — 揭开 AI 智能体评估的神秘面纱\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">智能体评估指导，包括针对研究智能体的有依据性、覆盖范围和来源质量检查。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fopenai.com\u002Findex\u002Ftrustworthy-third-party-evaluations-foundations\u002F\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">OpenAI — 可信第三方评估的共享手册\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">强调现代智能体性能取决于工作流和环境，而不仅仅是最终模型输出的评估指导。\u003C\u002Fp>\u003C\u002Fa>",{"time":212,"blocks":213,"version":850},1790367951029,[214,222,228,236,243,248,253,258,263,289,294,299,304,309,314,319,324,329,334,341,346,351,356,361,366,394,401,406,441,446,451,456,490,495,500,505,510,515,520,525,530,535,540,578,583,624,629,659,664,669,674,679,684,689,694,699,704,709,714,719,724,729,738,746,751,777,782,808,813,823,832,841],{"id":215,"data":216,"type":220,"tunes":221},"0hk9UtwqZf",{"title":217,"maxLevel":218,"minLevel":219},"目录",3,2,"tableOfContents",{},{"id":223,"data":224,"type":226,"tunes":227},"intro",{"text":225},"AI 智能体可以引用来源、检索文档，却仍然使用错误的证据。一个来源可能具有权威性，但与具体主张无关。一段检索到的文本可能只支持答案的一部分。一个正确的来源可能已经过时、被取代，或者只适用于错误的司法管辖区、产品版本、用户或系统状态。这就产生了一个比简单引用检查更难的评估问题：智能体是否真的为其所提出的主张使用了正确的证据？","paragraph",{},{"id":229,"data":230,"type":234,"tunes":235},"direct",{"body":231,"title":232,"variant":233},"要判断 AI 智能体是否使用了正确的证据，应评估这条链：\u003Cstrong>主张 → 证据 → 适用性 → 使用\u003C\u002Fstrong>。对于每一项重要主张，都要验证证据是否直接支持它、是否来自适当的权威来源、是否适用于当前条件，以及在主张产生之前智能体是否实际能够获取该证据。仅凭一条引用无法证明其中任何一点。","直接回答","info","callout",{},{"id":237,"data":238,"type":234,"tunes":242},"method-note",{"body":239,"title":240,"variant":241},"本文中的“主张—证据—适用性—使用”模型和证据利用测试是实用的评估方法，并非正式的行业标准。它们建立在诸如扎根性、来源质量、引用覆盖率、轨迹评估和任务特定评估等已有理念之上。","关于该方法","note",{},{"id":244,"data":245,"type":42,"tunes":247},"h-citations",{"text":246,"level":219},"为什么仅有引用还不够",{},{"id":249,"data":250,"type":226,"tunes":252},"p-citations-1",{"text":251},"引用只能回答一个狭窄的问题：系统将某个主张或回复与某个来源关联了起来。它并不能自动证明该来源支持这一具体主张、该来源对该任务足够权威、被引用的文本包含必要条件或例外情况，或者模型依赖了该证据，而不是根据先前的模型知识生成答案。",{},{"id":254,"data":255,"type":226,"tunes":257},"p-citations-2",{"text":256},"Anthropic 关于研究型智能体评估的指南明确区分了扎根性、覆盖率和来源质量。OpenAI 的智能体评估指南同样强调轨迹，因为最终输出无法揭示智能体是否选择了正确的工具或遵循了预期的工作流程。这些观点指向一个更广泛的结论：证据质量是执行路径的属性，而不仅仅是最终文本的属性。",{},{"id":259,"data":260,"type":42,"tunes":262},"h-four",{"text":261,"level":219},"每一项重要主张都应通过的四个问题",{},{"id":264,"data":265,"type":287,"tunes":288},"table-four",{"content":266,"stretched":43,"withHeadings":14},[267,271,275,279,283],[268,269,270],"维度","问题","典型失败",[272,273,274],"主张支持","证据是否直接支持这一确切主张？","来源在主题上相关，但并不能证明该陈述",[276,277,278],"证据权威性","对于这类主张，这是否是适当的来源？","在需要一手来源或实时系统的地方使用了二手摘要",[280,281,282],"适用性","证据是否适用于该时间、版本、司法管辖区、用户、状态或人群？","一个真实的陈述被应用到了其有效条件之外",[284,285,286],"证据使用","该证据在智能体的执行路径中是否实际可用并被使用？","最终答案是正确的，但检索到的证据无关或未被使用","table",{},{"id":290,"data":291,"type":42,"tunes":293},"h-support",{"text":292,"level":218},"1. 主张支持：来源是否证明了智能体所说的话？",{},{"id":295,"data":296,"type":226,"tunes":298},"p-support-1",{"text":297},"证据应在主张层面进行评估。一份文档可能与主题相关，却仍然无法支持某个具体陈述。如果来源说某项功能仅在部分区域可用，那么“该功能在全球可用”这一答案就是没有依据的，即使引用看起来似乎合理。",{},{"id":300,"data":301,"type":226,"tunes":303},"p-support-2",{"text":302},"这正是宽泛的“有依据 \u002F 无依据”判断往往过于粗糙的地方。将回复拆分为重要主张，把每项主张映射到支持它的最小证据片段，并对关系进行分类：直接支持、部分支持、矛盾或无支持。",{},{"id":305,"data":306,"type":42,"tunes":308},"h-authority",{"text":307,"level":218},"2. 证据权威性：这是正确类型的来源吗？",{},{"id":310,"data":311,"type":226,"tunes":313},"p-authority-1",{"text":312},"正确的证据选择不仅仅是语义相关性。来源必须适合该决策。当前账户状态应来自账户系统，而不是一封旧邮件。关于 API 行为的主张最好应依据当前供应商文档或可复现行为来核查。法律要求可能需要适用法律、监管机构或权威指南，而不是一篇泛泛的博客文章。",{},{"id":315,"data":316,"type":226,"tunes":318},"p-authority-2",{"text":317},"来源权威性是任务特定的。对于供应商文档未承认的真实世界漏洞，社区报告可能是最佳证据。供应商公告对于供应商所声称的内容可能具有权威性，但对于独立性能而言则是弱证据。因此，评估者需要为任务制定明确的来源层级，而不是使用一个通用的权威评分。",{},{"id":320,"data":321,"type":42,"tunes":323},"h-applicability",{"text":322,"level":218},"3. 适用性：证据正确，条件错误",{},{"id":325,"data":326,"type":226,"tunes":328},"p-apply-1",{"text":327},"最危险的证据错误往往不是伪造来源，而是有效来源被用在其边界之外。建议可能会随软件版本、日期、司法管辖区、硬件修订版、用户权限、产品可用性、当前游戏状态、租户配置或其他环境变量而变化。",{},{"id":330,"data":331,"type":226,"tunes":333},"p-apply-2",{"text":332},"对于每个重要来源，都要保留决定其是否仍然适用的条件。这在摘要之后尤其重要：压缩后的记忆或引用可能保留了结论，却丢掉了使该结论有效的例外、日期或前提条件。",{},{"id":335,"data":336,"type":234,"tunes":340},"apply-warning",{"body":337,"title":338,"variant":339},"一个真实的来源，即使被准确引用，当其时间、版本、人群、管辖范围或状态与当前问题不匹配时，仍然可能产生错误的答案。","证据可以是真实的，但对答案来说仍然是错误的","warning",{},{"id":342,"data":343,"type":42,"tunes":345},"h-use",{"text":344,"level":218},"4. 证据使用：智能体是否真正依赖了证据？",{},{"id":347,"data":348,"type":226,"tunes":350},"p-use-1",{"text":349},"即使检索失败，答案也可能是正确的。模型可能已经知道答案，从不相关的上下文中推断出答案，或者只是碰巧猜对了。如果评估只检查最终正确性，系统可能看起来有充分依据，而证据路径却是断裂的。",{},{"id":352,"data":353,"type":226,"tunes":355},"p-use-2",{"text":354},"要评估证据使用情况，需要检查追踪记录。确认检索到了哪些来源、哪些段落进入了模型上下文、它们何时可用，以及最终主张能否由这些输入解释。OpenAI 当前的智能体评估工具强调追踪评分，正是因为仅凭最终答案无法可靠地重建工作流层面的行为。",{},{"id":357,"data":358,"type":42,"tunes":360},"h-eut",{"text":359,"level":219},"证据利用测试",{},{"id":362,"data":363,"type":226,"tunes":365},"p-eut-intro",{"text":364},"一种实用的评估可以构建为受控的反事实测试。不要只问答案是否正确，而是改变证据并观察主张是否按预期方向变化。",{},{"id":367,"data":368,"type":392,"tunes":393},"eut-flow",{"steps":369,"title":359,"orientation":391},[370,373,376,379,382,385,388],{"label":371,"description":372},"1. 选择一个重要主张","选择一个正确性至关重要的主张，并精确定义预期答案。",{"label":374,"description":375},"2. 确定黄金证据","提供足以支持该主张的最小权威证据集。",{"label":377,"description":378},"3. 使用黄金证据运行","验证当正确证据可用时，智能体是否产生受支持的答案。",{"label":380,"description":381},"4. 移除决定性证据","在保持其他输入稳定的情况下，移除关键支持段落并运行相同任务。",{"label":383,"description":384},"5. 用矛盾或取代性证据替换它","在安全的情况下，提供能够改变正确结论的受控证据。",{"label":386,"description":387},"6. 比较主张","检查答案是否跟随证据变化，还是仍然锚定于先前的模型知识。",{"label":389,"description":390},"7. 检查追踪记录","确认检索到了什么、什么进入了上下文，以及主张之前是哪个来源或工具结果。","auto","processFlow",{},{"id":395,"data":396,"type":234,"tunes":400},"eut-tip",{"body":397,"title":398,"variant":399},"如果决定性证据发生了变化，但智能体的主张没有变化，那么你有证据表明系统可能没有按预期使用检索——即使原始答案恰好是正确的。","关键信号","tip",{},{"id":402,"data":403,"type":42,"tunes":405},"h-matrix",{"text":404,"level":219},"主张–证据矩阵比来源列表更有用",{},{"id":407,"data":408,"type":287,"tunes":440},"claim-matrix",{"content":409,"stretched":43,"withHeadings":14},[410,416,423,429,435],[411,412,413,414,280,415],"主张","证据","支持","权威性","追踪中是否使用",[417,418,419,420,421,422],"功能 X 可用","供应商文档","直接","对可用性主张具有高权威性","当前版本和地区必须匹配","是 \u002F 否",[424,425,426,427,428,422],"配置 Y 更快","供应商基准测试","部分","对供应商的测试具有高权威性，但不代表独立性能","硬件和工作负载必须匹配",[430,431,432,433,434,422],"政策适用于此用户","当前政策 + 账户状态","仅组合时直接","高","管辖范围、日期、角色和账户状态必须匹配",[436,437,419,438,439,422],"某产品有库存","实时库存 API","对当前库存具有权威性","很快过期",{},{"id":442,"data":443,"type":226,"tunes":445},"p-matrix-1",{"text":444},"这个矩阵迫使人们提出几个传统引用检查所隐藏的问题。一个主张可能需要多个来源。一个来源可能只支持主张的一部分。一个权威来源可能只有很短的有效期。而一个完全良好的来源，如果从未进入执行路径，也可能无关紧要。",{},{"id":447,"data":448,"type":42,"tunes":450},"h-retrieval-evidence",{"text":449,"level":219},"将检索质量与证据质量分开",{},{"id":452,"data":453,"type":226,"tunes":455},"p-re-1",{"text":454},"检索指标询问是否找到了相关材料并进行了排序。证据评估询问这些材料是否证明了由此产生的主张。两者相关但并不相同。",{},{"id":457,"data":458,"type":488,"tunes":489},"retrieval-comparison",{"rows":459,"title":477,"layout":287,"columns":478},[460,465,469,473],{"id":461,"label":462,"values":463},"a","正确文档，错误主张",[464,464,464],"",{"id":466,"label":467,"values":468},"b","正确事实，过时来源",[464,464,464],{"id":470,"label":471,"values":472},"c","弱来源，正确答案",[464,464,464],{"id":474,"label":475,"values":476},"d","需要多个来源",[464,464,464],"检索成功不等于证据成功",[479,482,485],{"id":480,"label":481},"situation","情况",{"id":483,"label":484},"retrieval","检索",{"id":486,"label":487},"evidence","证据质量","comparison",{},{"id":491,"data":492,"type":42,"tunes":494},"h-source-quality",{"text":493,"level":219},"将来源质量作为评分标准来评估，而不是域名白名单",{},{"id":496,"data":497,"type":226,"tunes":499},"p-source-quality-1",{"text":498},"硬编码的“可信域名”列表很诱人，但往往很脆弱。来源质量应改为反映主张类型。有用的维度包括一手与二手地位、时效性、直接性、可复现性、独立性、领域专业知识、数据来源、更新频率，以及来源是否有夸大主张的动机。",{},{"id":501,"data":502,"type":226,"tunes":504},"p-source-quality-2",{"text":503},"Anthropic 的研究智能体评估指南明确将来源质量检查与有据性和覆盖范围并列提出。因此，实际实施应同时评估来源说了什么，以及该来源是否适合这类陈述。",{},{"id":506,"data":507,"type":42,"tunes":509},"h-coverage",{"text":508,"level":219},"证据覆盖范围：每个重要主张都需要支持，而不是每句话",{},{"id":511,"data":512,"type":226,"tunes":514},"p-coverage-1",{"text":513},"并非每一句话都需要引用。过渡性语言、从引用值透明推导出的算术结果，或明确标注的解释，可能不需要单独的来源。但每一个实质性的、可外部验证的断言都应有足够的支撑，使评估者能够重建为什么该智能体被允许这样说。",{},{"id":516,"data":517,"type":226,"tunes":519},"p-coverage-2",{"text":518},"因此，覆盖度应按断言重要性加权。装饰性细节缺少支撑，与价格、资格决定、安全说明、法律要求、技术兼容性声明或驱动推荐的事实缺少支撑，并不等同。",{},{"id":521,"data":522,"type":42,"tunes":524},"h-provenance",{"text":523,"level":219},"证据来源必须在摘要化和记忆化后仍然保留",{},{"id":526,"data":527,"type":226,"tunes":529},"p-prov-1",{"text":528},"长期运行的智能体常常会总结先前的工作或写入持久记忆。如果证据来源在这一转换过程中被剥离，未来的智能体可能检索到一个干净的结论，却不知道它来自用户陈述、实时 API、旧文档、模型推断，还是未经核实的网络结果。",{},{"id":531,"data":532,"type":226,"tunes":534},"p-prov-2",{"text":533},"对于重要事实，至少应保留来源身份、检索或观察时间、证据类型、相关版本或状态，以及所存储文本是引用、摘要、推断还是派生。来源信息使后续智能体能够判断该证据应被信任、刷新、限制还是丢弃。",{},{"id":536,"data":537,"type":42,"tunes":539},"h-record",{"text":538,"level":219},"实用的证据记录",{},{"id":541,"data":542,"type":287,"tunes":577},"evidence-record",{"content":543,"stretched":43,"withHeadings":14},[544,547,550,553,556,559,562,565,568,571,574],[545,546],"字段","用途",[548,549],"claim_id","标识被支撑的实质性断言",[551,552],"source_id \u002F source_url \u002F system","标识证据来自何处",[554,555],"evidence_span","保留支撑该断言的最小段落、记录或工具结果",[557,558],"retrieved_at \u002F observed_at","允许进行新鲜度和时间线检查",[560,561],"source_version \u002F object_version","允许进行取代和可复现性检查",[563,564],"authority_role","解释为什么该来源适合该断言",[566,567],"applicability","存储相关日期、司法管辖区、产品版本、用户、租户、状态或其他条件",[569,570],"transformation","标记证据是原始、引用、摘要、规范化还是派生",[572,573],"trace_step","显示证据何时对智能体可用",[575,576],"support_status","直接、部分、矛盾、无支撑或不确定",{},{"id":579,"data":580,"type":42,"tunes":582},"h-failures",{"text":581,"level":219},"看似有依据但实际上没有的失败模式",{},{"id":584,"data":585,"type":287,"tunes":623},"failure-table",{"content":586,"stretched":43,"withHeadings":14},[587,591,595,599,603,607,611,615,619],[588,589,590],"失败模式","为什么它能骗过评估者","应测试什么",[592,593,594],"引用装饰","答案包含来源，因此看起来经过研究","将每个实质性断言映射到精确的支撑片段",[596,597,598],"权威不匹配","来源声誉良好，但对具体事实并不权威","定义针对具体断言的来源层级",[600,601,602],"时间不匹配","来源在发布时是正确的","检查检索时间、来源日期和取代性证据",[604,605,606],"条件剥离","摘要保留了结论，却丢弃了例外","将生成的断言与完整的本地来源上下文进行比较",[608,609,610],"事后引用","在答案生成后才附上一个看似合理的来源","检查追踪顺序以及证据是否先于断言出现",[612,613,614],"参数化覆盖","模型忽略检索到的证据，依据先验知识作答","运行反事实证据利用测试",[616,617,618],"证据洗白","模型推断被摘要化，随后被当作来源事实存储","在记忆写入过程中保留转换类型和来源信息",[620,621,622],"来源多数谬误","多个二手页面重复同一无支撑陈述","将断言追溯至独立或一手证据",{},{"id":625,"data":626,"type":42,"tunes":628},"h-production",{"text":627,"level":219},"如何在生产环境中评估智能体",{},{"id":630,"data":631,"type":392,"tunes":658},"production-flow",{"steps":632,"title":657,"orientation":391},[633,636,639,642,645,648,651,654],{"label":634,"description":635},"1. 定义实质性断言","识别对任务而言正确性重要的事实、建议或决定。",{"label":637,"description":638},"2. 构建黄金证据","为代表性案例创建参考证据和来源质量预期。",{"label":640,"description":641},"3. 捕获追踪","记录检索查询、工具调用、返回证据、上下文构建、模型输出和引用。",{"label":643,"description":644},"4. 评定支撑度","检查每个实质性断言是直接支撑、部分支撑、矛盾支撑还是无支撑。",{"label":646,"description":647},"5. 评定权威性和适用性","评估来源是否合适，以及其条件是否与当前任务匹配。",{"label":649,"description":650},"6. 运行反事实","移除、替换或取代决定性证据，并测试答案是否随变化而改变。",{"label":652,"description":653},"7. 审查高影响失败","在自动评分不够可靠时，使用人工或领域专家审查。",{"label":655,"description":656},"8. 将失败转化为评估案例","将生产失败和边缘案例加入可重复的回归数据集。","证据评估流水线",{},{"id":660,"data":661,"type":226,"tunes":663},"p-production-1",{"text":662},"OpenAI 当前的评估指南建议采用任务特定评估、持续评估、源自生产的数据集，以及用于调试智能体行为的追踪。Anthropic 同样建议为研究型智能体组合多种评分器类型，因为正确性、来源质量、覆盖度和有据性是不同的维度。证据评估应遵循同样的模式：多个窄范围评分器比一个不透明的“质量”分数更具诊断价值。",{},{"id":665,"data":666,"type":42,"tunes":668},"h-judge",{"text":667,"level":219},"不要让 LLM 评判者成为唯一的证据评判者",{},{"id":670,"data":671,"type":226,"tunes":673},"p-judge-1",{"text":672},"LLM 评分器可用于可扩展的断言分类、相关性检查和成对比较，但它们可能与其所评估的系统共享同样的盲点。评分器可能接受一个看似合理但无支撑的断言，漏掉微妙的版本边界，或高估一个经过润色的来源。",{},{"id":675,"data":676,"type":226,"tunes":678},"p-judge-2",{"text":677},"OpenAI 的评估指南建议根据人工判断校准自动评分器，并使用清晰、有范围的准则。对于证据密集型系统，应尽可能使用确定性检查：时间戳、对象版本、权限范围、精确来源 ID、检索顺序、文档哈希，以及证据是否在模型生成断言之前就已存在。",{},{"id":680,"data":681,"type":42,"tunes":683},"h-change",{"text":682,"level":219},"什么会改变这个答案？",{},{"id":685,"data":686,"type":226,"tunes":688},"p-change-1",{"text":687},"当智能体在小型、不可变、权威的语料库上运行，且每个答案都严格是抽取式的时候，评估可以更简单。在这种环境中，来源权威性和适用性大多是固定的，断言到片段的支撑可能就足够了。",{},{"id":690,"data":691,"type":226,"tunes":693},"p-change-2",{"text":692},"当智能体混合使用网络搜索、长期记忆、实时工具、多个司法管辖区、快速变化的信息、用户特定状态或自主行动时，评估必须变得更严格。在这些系统中，证据有效性不仅取决于来源文本，还取决于证据是何时以及如何获得的。",{},{"id":695,"data":696,"type":226,"tunes":698},"p-change-3",{"text":697},"未来的模型可能会更擅长在内部追踪来源和不确定性，但在需要可审计性的系统中，这并不能消除对外部证据记录的需求。当轨迹和来源元数据能够提供更强证据时，系统不应依赖模型对自身受何影响的自我报告。",{},{"id":700,"data":701,"type":42,"tunes":703},"h-limitations",{"text":702,"level":219},"局限性",{},{"id":705,"data":706,"type":226,"tunes":708},"p-limit-1",{"text":707},"仅凭轨迹并不总能证明因果性的证据使用。一个来源可能存在于上下文中，却没有影响答案，而模型也可能独立地知道同一事实。反事实测试能增强推断，但测试本身也可能改变任务分布。",{},{"id":710,"data":711,"type":226,"tunes":713},"p-limit-2",{"text":712},"来源权威性也可能存在争议，或取决于领域。有些问题没有单一的权威来源，专家也可能对哪些证据应获得更大权重存在分歧。在这些情况下，评估者应保留分歧，并依据明确的评分标准来评估透明度、覆盖范围和推理，而不是假装存在一个不容置疑的真理来源。",{},{"id":715,"data":716,"type":42,"tunes":718},"h-conclusion",{"text":717,"level":219},"结论",{},{"id":720,"data":721,"type":226,"tunes":723},"p-conclusion-1",{"text":722},"“智能体是否引用了来源？”这个问题对于生产级 AI 来说太弱了。更强的问题是：每个重要主张是否都来自真正支持它的证据，该证据是否具有正确的权威性，是否仍适用于当前条件，并且在主张提出之前是否已存在于执行路径中？",{},{"id":725,"data":726,"type":226,"tunes":728},"p-conclusion-2",{"text":727},"这使证据从装饰变成了可评估的系统属性。捕获轨迹。将主张映射到证据。检查权威性和适用性。运行反事实证据测试。在摘要和记忆中保留来源信息。这样，正确答案不仅看似合理——它还拥有一条你可以检查的证据路径。",{},{"id":730,"data":731,"type":736,"tunes":737},"internal-reliability",{"url":732,"title":733,"excerpt":734,"ctaLabel":735},"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough","AI 智能体可靠性：为什么仅有最终答案还不够","仅凭结果正确并不能证明智能体的执行路径是安全或可靠的。这篇相关文章解释了为什么轨迹、工具和中间决策很重要。","阅读相关文章","referralArticle",{},{"id":739,"data":740,"type":736,"tunes":745},"internal-reasoning",{"url":741,"title":742,"excerpt":743,"ctaLabel":744},"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Ffrom-research-protocol-to-a-general-ai-reasoning-framework","从研究协议到通用 AI 推理框架","一种实用的推理方法，用于区分证据、假设、竞争性假设和特定领域的验证。","阅读推理框架",{},{"id":747,"data":748,"type":42,"tunes":750},"h-faq",{"text":749,"level":219},"常见问题",{},{"id":752,"data":753,"type":752,"tunes":776},"faq",{"items":754,"title":775},[755,759,763,767,771],{"id":756,"answer":757,"question":758},"faq1","不能。引用可能与主题相关，但并不支持确切主张；可能来自错误的权威来源；可能不再适用；或者可能只是被附加上去，却并未实质影响生成的答案。","引用是否能证明 AI 答案有依据？",{"id":760,"answer":761,"question":762},"faq2","使用反事实证据利用测试：先用已知正确的证据运行任务，然后在保持其他输入稳定的情况下移除或替换决定性证据。如果答案没有对证据变化作出反应，就检查模型是否依赖先验知识或其他来源。","如何测试 AI 智能体是否实际使用了检索到的证据？",{"id":764,"answer":765,"question":766},"faq3","有依据性问的是主张是否由所提供的证据支持。来源质量问的是证据本身对于所提出的主张类型是否足够适当且具有权威性。","有依据性和来源质量有什么区别？",{"id":768,"answer":769,"question":770},"faq4","来源可能已过时、已被取代，或仅对另一个版本、司法管辖区、用户、人群或系统状态有效，也可能包含在检索或摘要过程中丢失的条件。","为什么真实来源仍可能导致错误的 AI 答案？",{"id":772,"answer":773,"question":774},"faq5","记录检索查询、返回的来源、确切证据片段、时间戳和版本、过滤器、最终上下文、模型输出、引用以及轨迹顺序，以便评估者能够重建在每个重要主张之前有哪些证据可用。","为了评估证据，我应该记录什么？","评估 AI 智能体中的证据使用",{},{"id":778,"data":779,"type":42,"tunes":781},"h-glossary",{"text":780,"level":219},"术语表",{},{"id":783,"data":784,"type":783,"tunes":807},"glossary",{"title":785,"entries":786},"关键证据评估术语",[787,790,793,795,799,803],{"term":272,"anchor":788,"definition":789},"claim-support","特定证据片段直接确立生成主张的程度。",{"term":276,"anchor":791,"definition":792},"evidence-authority","考虑到来源的角色、出处及其与底层事实的关系，该来源在确立某一类主张时有多适当。",{"term":280,"anchor":566,"definition":794},"证据对某一主张仍然有效的条件，包括时间、版本、司法管辖区、用户、人群、系统状态或其他边界。",{"term":796,"anchor":797,"definition":798},"证据利用","evidence-utilization","智能体的输出是否实际响应并依赖于其执行路径中可用的证据。",{"term":800,"anchor":801,"definition":802},"反事实证据测试","counterfactual-evidence-test","一种评估方法，通过移除、替换或更改决定性证据，来测试智能体的主张是否会相应变化。",{"term":804,"anchor":805,"definition":806},"来源信息","provenance","记录证据来自何处、何时获取、如何被转换以及它代表哪个版本或状态的元数据。",{},{"id":809,"data":810,"type":42,"tunes":812},"h-sources",{"text":811,"level":219},"主要来源与延伸阅读",{},{"id":814,"data":815,"type":821,"tunes":822},"src-openai-agent-evals",{"link":816,"meta":817},"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fagent-evals",{"image":818,"title":819,"description":820},{"url":464},"OpenAI — 评估智能体工作流","关于轨迹评分、工作流级评估、数据集以及可重复的智能体评估运行的指导。","linkTool",{},{"id":824,"data":825,"type":821,"tunes":831},"src-openai-eval-best",{"link":826,"meta":827},"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fevaluation-best-practices",{"image":828,"title":829,"description":830},{"url":464},"OpenAI — 评估最佳实践","关于特定任务评估、源自生产的数据集、限定范围的指标、持续评估和评分器校准的指导。",{},{"id":833,"data":834,"type":821,"tunes":840},"src-anthropic-evals",{"link":835,"meta":836},"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Fdemystifying-evals-for-ai-agents",{"image":837,"title":838,"description":839},{"url":464},"Anthropic — 揭开 AI 智能体评估的神秘面纱","智能体评估指导，包括针对研究智能体的有依据性、覆盖范围和来源质量检查。",{},{"id":842,"data":843,"type":821,"tunes":849},"src-openai-thirdparty",{"link":844,"meta":845},"https:\u002F\u002Fopenai.com\u002Findex\u002Ftrustworthy-third-party-evaluations-foundations\u002F",{"image":846,"title":847,"description":848},{"url":464},"OpenAI — 可信第三方评估的共享手册","强调现代智能体性能取决于工作流和环境，而不仅仅是最终模型输出的评估指导。",{},"2.31","AI代理可以引用来源，却仍然使用错误的证据。本文介绍一种实用方法，用于核查主张支持、来源权威性、适用性、出处，以及证据是否实际影响了答案。","\u002Fuploads\u002F2026\u002F09\u002Fhow-to-know-whether-an-ai-agent-actually-used-the-right-evidence-1790351317188-o5z9ve.webp","how-to-know-whether-an-ai-agent-actually-used-the-right-evidence-1790351317188-o5z9ve","PUBLISHED","2026-09-25T11:47:00.000Z","2026-09-25T15:47:02.186Z","2026-09-25T20:33:01.720Z",{"en":859,"de":860,"sr":861,"es":862,"fr":863,"it":864,"ru":865,"zh":866},"\u002Fblog\u002Fhow-to-know-whether-an-ai-agent-actually-used-the-right-evidence","\u002Fde\u002Fblog\u002Fhow-to-know-whether-an-ai-agent-actually-used-the-right-evidence","\u002Fsr\u002Fblog\u002Fhow-to-know-whether-an-ai-agent-actually-used-the-right-evidence","\u002Fes\u002Fblog\u002Fhow-to-know-whether-an-ai-agent-actually-used-the-right-evidence","\u002Ffr\u002Fblog\u002Fhow-to-know-whether-an-ai-agent-actually-used-the-right-evidence","\u002Fit\u002Fblog\u002Fhow-to-know-whether-an-ai-agent-actually-used-the-right-evidence","\u002Fru\u002Fblog\u002Fhow-to-know-whether-an-ai-agent-actually-used-the-right-evidence","\u002Fzh\u002Fblog\u002Fhow-to-know-whether-an-ai-agent-actually-used-the-right-evidence",[868,872,876],{"id":869,"name":870,"slug":871},84,"策略与数据边界","policy-and-data",{"id":873,"name":874,"slug":875},97,"在测试集上验证","verification",{"id":877,"name":878,"slug":879},66,"内容运营","content-ops",{"id":881,"login":882,"email":883,"displayName":884},"20","rooth8233","aleksandar@stajic.de","Aleksandar Stajić",[886,1413],{"lang":887,"title":888,"content":889,"contentJson":890,"excerpt":1412},"en","How to Know Whether an AI Agent Actually Used the Right Evidence","{\"time\":1790351390243,\"blocks\":[{\"id\":\"0hk9UtwqZf\",\"type\":\"tableOfContents\",\"data\":{\"title\":\"Contents\",\"minLevel\":2,\"maxLevel\":3},\"tunes\":{}},{\"id\":\"intro\",\"type\":\"paragraph\",\"data\":{\"text\":\"An AI agent can cite sources, retrieve documents, and still use the wrong evidence. A source may be authoritative but irrelevant to the exact claim. A retrieved passage may support only part of an answer. A correct source can be stale, superseded, or valid for the wrong jurisdiction, product version, user, or system state. This creates a harder evaluation problem than simple citation checking: did the agent actually use the right evidence for the claim it made?\"},\"tunes\":{}},{\"id\":\"direct\",\"type\":\"callout\",\"data\":{\"variant\":\"info\",\"title\":\"Direct answer\",\"body\":\"To know whether an AI agent used the right evidence, evaluate the chain \u003Cstrong>Claim → Evidence → Applicability → Use\u003C\u002Fstrong>. For each material claim, verify that the evidence directly supports it, comes from an appropriate authority, applies to the current conditions, and was actually available to the agent before the claim was produced. A citation alone proves none of those things.\"},\"tunes\":{}},{\"id\":\"method-note\",\"type\":\"callout\",\"data\":{\"variant\":\"note\",\"title\":\"About the method\",\"body\":\"The Claim–Evidence–Applicability–Use model and Evidence Utilization Test in this article are practical evaluation methods, not formal industry standards. They build on established ideas such as groundedness, source quality, citation coverage, trace evaluation, and task-specific evals.\"},\"tunes\":{}},{\"id\":\"h-citations\",\"type\":\"header\",\"data\":{\"text\":\"Why citations are not enough\",\"level\":2},\"tunes\":{}},{\"id\":\"p-citations-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A citation answers only a narrow question: the system associated a claim or response with a source. It does not automatically establish that the source supports the specific claim, that the source is authoritative enough for the task, that the cited passage contains the necessary condition or exception, or that the model relied on that evidence rather than producing the answer from prior model knowledge.\"},\"tunes\":{}},{\"id\":\"p-citations-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Anthropic's guidance for research-agent evaluation explicitly separates groundedness, coverage, and source quality. OpenAI's agent-evaluation guidance similarly emphasizes traces because a final output does not reveal whether the agent selected the right tools or followed the intended workflow. Those ideas point to a broader conclusion: evidence quality is a property of the execution path, not just the final prose.\"},\"tunes\":{}},{\"id\":\"h-four\",\"type\":\"header\",\"data\":{\"text\":\"The four questions every material claim should pass\",\"level\":2},\"tunes\":{}},{\"id\":\"table-four\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Dimension\",\"Question\",\"Typical failure\"],[\"Claim support\",\"Does the evidence directly support this exact claim?\",\"The source is topically related but does not establish the statement\"],[\"Evidence authority\",\"Is this an appropriate source for this kind of claim?\",\"A secondary summary is used where a primary source or live system is required\"],[\"Applicability\",\"Does the evidence apply to this time, version, jurisdiction, user, state, or population?\",\"A true statement is applied outside its valid conditions\"],[\"Evidence use\",\"Was this evidence actually available and used in the agent's execution path?\",\"The final answer is correct, but the retrieved evidence was irrelevant or unused\"]]},\"tunes\":{}},{\"id\":\"h-support\",\"type\":\"header\",\"data\":{\"text\":\"1. Claim support: does the source establish what the agent says?\",\"level\":3},\"tunes\":{}},{\"id\":\"p-support-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Evidence should be evaluated at claim level. A document can be relevant to the subject and still fail to support a specific statement. If a source says that a feature is available in selected regions, the answer “the feature is available globally” is unsupported even though the citation looks plausible.\"},\"tunes\":{}},{\"id\":\"p-support-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"This is where broad “grounded \u002F not grounded” judgments are often too coarse. Split the response into material claims, map each claim to the smallest evidence span that supports it, and classify the relationship: direct support, partial support, contradiction, or no support.\"},\"tunes\":{}},{\"id\":\"h-authority\",\"type\":\"header\",\"data\":{\"text\":\"2. Evidence authority: is this the right kind of source?\",\"level\":3},\"tunes\":{}},{\"id\":\"p-authority-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Correct evidence selection is not only semantic relevance. The source must be suitable for the decision. Current account status should come from the account system, not an old email. An API behaviour claim should preferably be checked against current vendor documentation or reproducible behaviour. A legal requirement may need the applicable law, regulator, or authoritative guidance rather than a generic blog post.\"},\"tunes\":{}},{\"id\":\"p-authority-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Source authority is task-specific. A community report can be the best evidence for a real-world bug that vendor documentation does not acknowledge. A vendor announcement can be authoritative for what the vendor claims but weak evidence for independent performance. The evaluator therefore needs an explicit source hierarchy for the task rather than one universal authority score.\"},\"tunes\":{}},{\"id\":\"h-applicability\",\"type\":\"header\",\"data\":{\"text\":\"3. Applicability: right evidence, wrong conditions\",\"level\":3},\"tunes\":{}},{\"id\":\"p-apply-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The most dangerous evidence errors are often not fabricated sources but valid sources used outside their boundary. A recommendation can change with software version, date, jurisdiction, hardware revision, user permissions, product availability, current game state, tenant configuration, or other environmental variables.\"},\"tunes\":{}},{\"id\":\"p-apply-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"For each material source, preserve the conditions that determine whether it still applies. This is especially important after summarization: a compressed memory or citation may preserve the conclusion while dropping the exception, date, or prerequisite that made the conclusion valid.\"},\"tunes\":{}},{\"id\":\"apply-warning\",\"type\":\"callout\",\"data\":{\"variant\":\"warning\",\"title\":\"Evidence can be authentic and still be wrong for the answer\",\"body\":\"A real source, quoted accurately, can still produce a wrong answer when its time, version, population, jurisdiction, or state does not match the current question.\"},\"tunes\":{}},{\"id\":\"h-use\",\"type\":\"header\",\"data\":{\"text\":\"4. Evidence use: did the agent actually rely on the evidence?\",\"level\":3},\"tunes\":{}},{\"id\":\"p-use-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"An answer can be correct even when retrieval failed. The model may already know the answer, infer it from unrelated context, or simply guess correctly. If the evaluation checks only final correctness, the system may appear well-grounded while the evidence path is broken.\"},\"tunes\":{}},{\"id\":\"p-use-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"To evaluate evidence use, inspect the trace. Confirm which sources were retrieved, which passages reached the model context, when they became available, and whether the final claim can be explained by those inputs. OpenAI's current agent-evaluation tooling emphasizes trace grading precisely because workflow-level behaviour cannot be reconstructed reliably from the final answer alone.\"},\"tunes\":{}},{\"id\":\"h-eut\",\"type\":\"header\",\"data\":{\"text\":\"The Evidence Utilization Test\",\"level\":2},\"tunes\":{}},{\"id\":\"p-eut-intro\",\"type\":\"paragraph\",\"data\":{\"text\":\"A practical evaluation can be built as a controlled counterfactual. Instead of asking only whether the answer is correct, change the evidence and observe whether the claim changes in the expected direction.\"},\"tunes\":{}},{\"id\":\"eut-flow\",\"type\":\"processFlow\",\"data\":{\"title\":\"Evidence Utilization Test\",\"orientation\":\"auto\",\"steps\":[{\"label\":\"1. Select one material claim\",\"description\":\"Choose a claim whose correctness matters and define the expected answer precisely.\"},{\"label\":\"2. Identify gold evidence\",\"description\":\"Provide the smallest authoritative evidence set sufficient to support the claim.\"},{\"label\":\"3. Run with gold evidence\",\"description\":\"Verify that the agent produces the supported answer when the correct evidence is available.\"},{\"label\":\"4. Remove the decisive evidence\",\"description\":\"Run the same task without the key supporting passage while keeping other inputs stable.\"},{\"label\":\"5. Replace it with contradictory or superseding evidence\",\"description\":\"Where safe, provide controlled evidence that changes the correct conclusion.\"},{\"label\":\"6. Compare the claims\",\"description\":\"Check whether the answer tracks the evidence change or remains anchored to prior model knowledge.\"},{\"label\":\"7. Inspect the trace\",\"description\":\"Confirm what was retrieved, what reached context, and what source or tool result preceded the claim.\"}]},\"tunes\":{}},{\"id\":\"eut-tip\",\"type\":\"callout\",\"data\":{\"variant\":\"tip\",\"title\":\"The key signal\",\"body\":\"If the decisive evidence changes but the agent's claim does not, you have evidence that the system may not be using retrieval as intended — even when the original answer happened to be correct.\"},\"tunes\":{}},{\"id\":\"h-matrix\",\"type\":\"header\",\"data\":{\"text\":\"A claim–evidence matrix is more useful than a source list\",\"level\":2},\"tunes\":{}},{\"id\":\"claim-matrix\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Claim\",\"Evidence\",\"Support\",\"Authority\",\"Applicability\",\"Used in trace\"],[\"Feature X is available\",\"Vendor documentation\",\"Direct\",\"High for availability claim\",\"Current version and region must match\",\"Yes \u002F No\"],[\"Configuration Y is faster\",\"Vendor benchmark\",\"Partial\",\"High for vendor's test, not independent performance\",\"Hardware and workload must match\",\"Yes \u002F No\"],[\"Policy applies to this user\",\"Current policy + account state\",\"Direct only when combined\",\"High\",\"Jurisdiction, date, role and account state must match\",\"Yes \u002F No\"],[\"A product is in stock\",\"Live inventory API\",\"Direct\",\"Authoritative for current stock\",\"Expires quickly\",\"Yes \u002F No\"]]},\"tunes\":{}},{\"id\":\"p-matrix-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"This matrix forces several questions that conventional citation checking hides. One claim can require multiple sources. One source can support only part of a claim. An authoritative source can have a short validity window. And a perfectly good source can be irrelevant if it never entered the execution path.\"},\"tunes\":{}},{\"id\":\"h-retrieval-evidence\",\"type\":\"header\",\"data\":{\"text\":\"Separate retrieval quality from evidence quality\",\"level\":2},\"tunes\":{}},{\"id\":\"p-re-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Retrieval metrics ask whether relevant material was found and ranked. Evidence evaluation asks whether that material justifies the resulting claims. The two are related but not identical.\"},\"tunes\":{}},{\"id\":\"retrieval-comparison\",\"type\":\"comparison\",\"data\":{\"title\":\"Retrieval success is not evidence success\",\"layout\":\"table\",\"columns\":[{\"id\":\"situation\",\"label\":\"Situation\"},{\"id\":\"retrieval\",\"label\":\"Retrieval\"},{\"id\":\"evidence\",\"label\":\"Evidence quality\"}],\"rows\":[{\"id\":\"a\",\"label\":\"Right document, wrong claim\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"b\",\"label\":\"Right fact, stale source\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"c\",\"label\":\"Weak source, correct answer\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"d\",\"label\":\"Multiple sources required\",\"values\":[\"\",\"\",\"\"]}]},\"tunes\":{}},{\"id\":\"h-source-quality\",\"type\":\"header\",\"data\":{\"text\":\"Evaluate source quality as a rubric, not a domain whitelist\",\"level\":2},\"tunes\":{}},{\"id\":\"p-source-quality-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Hard-coded lists of “trusted domains” are tempting but often brittle. Source quality should instead reflect the claim type. Useful dimensions include primary versus secondary status, recency, directness, reproducibility, independence, domain expertise, data provenance, update cadence, and whether the source has an incentive to overstate the claim.\"},\"tunes\":{}},{\"id\":\"p-source-quality-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Anthropic's research-agent evaluation guidance explicitly calls out source-quality checks alongside groundedness and coverage. The practical implementation should therefore grade both what the source says and whether this source is appropriate for this kind of statement.\"},\"tunes\":{}},{\"id\":\"h-coverage\",\"type\":\"header\",\"data\":{\"text\":\"Evidence coverage: every important claim needs support, not every sentence\",\"level\":2},\"tunes\":{}},{\"id\":\"p-coverage-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Not every sentence needs a citation. Transitional language, arithmetic derived transparently from cited values, or clearly marked interpretation may not require a separate source. But every material externally verifiable claim should have enough support that an evaluator can reconstruct why the agent was allowed to say it.\"},\"tunes\":{}},{\"id\":\"p-coverage-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Coverage should therefore be weighted by claim importance. Missing support for a decorative detail is not equivalent to missing support for a price, eligibility decision, safety instruction, legal requirement, technical compatibility statement, or recommendation-driving fact.\"},\"tunes\":{}},{\"id\":\"h-provenance\",\"type\":\"header\",\"data\":{\"text\":\"Evidence provenance must survive summarization and memory\",\"level\":2},\"tunes\":{}},{\"id\":\"p-prov-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Long-running agents often summarize previous work or write durable memories. If evidence provenance is stripped during that transformation, future agents may retrieve a clean conclusion without knowing whether it came from a user statement, a live API, an old document, a model inference, or an unverified web result.\"},\"tunes\":{}},{\"id\":\"p-prov-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"For important facts, preserve at least the source identity, retrieval or observation time, evidence type, relevant version or state, and whether the stored text is quoted, summarized, inferred, or derived. Provenance is what allows a later agent to decide whether the evidence should be trusted, refreshed, restricted, or discarded.\"},\"tunes\":{}},{\"id\":\"h-record\",\"type\":\"header\",\"data\":{\"text\":\"A practical evidence record\",\"level\":2},\"tunes\":{}},{\"id\":\"evidence-record\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Field\",\"Purpose\"],[\"claim_id\",\"Identifies the material claim being supported\"],[\"source_id \u002F source_url \u002F system\",\"Identifies where the evidence came from\"],[\"evidence_span\",\"Preserves the smallest passage, record, or tool result that supports the claim\"],[\"retrieved_at \u002F observed_at\",\"Allows freshness and timeline checks\"],[\"source_version \u002F object_version\",\"Allows supersession and reproducibility checks\"],[\"authority_role\",\"Explains why this source is suitable for this claim\"],[\"applicability\",\"Stores relevant date, jurisdiction, product version, user, tenant, state, or other conditions\"],[\"transformation\",\"Marks whether evidence is raw, quoted, summarized, normalized, or derived\"],[\"trace_step\",\"Shows when the evidence became available to the agent\"],[\"support_status\",\"Direct, partial, contradictory, unsupported, or uncertain\"]]},\"tunes\":{}},{\"id\":\"h-failures\",\"type\":\"header\",\"data\":{\"text\":\"Failure modes that look grounded but are not\",\"level\":2},\"tunes\":{}},{\"id\":\"failure-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Failure mode\",\"Why it fools evaluators\",\"What to test\"],[\"Citation decoration\",\"The answer contains sources, so it looks researched\",\"Map each material claim to an exact supporting span\"],[\"Authority mismatch\",\"The source is reputable but not authoritative for the specific fact\",\"Define claim-specific source hierarchy\"],[\"Temporal mismatch\",\"The source was correct when published\",\"Check retrieval time, source date, and superseding evidence\"],[\"Condition stripping\",\"A summary keeps the conclusion but drops exceptions\",\"Compare generated claim with full local source context\"],[\"Post-hoc citation\",\"A plausible source is attached after the answer is generated\",\"Inspect trace ordering and whether evidence preceded the claim\"],[\"Parametric override\",\"The model ignores retrieved evidence and answers from prior knowledge\",\"Run counterfactual evidence-utilization tests\"],[\"Evidence laundering\",\"Model inference is summarized and later stored as if it were a source fact\",\"Preserve transformation type and provenance across memory writes\"],[\"Source majority fallacy\",\"Several secondary pages repeat the same unsupported statement\",\"Trace claims back to independent or primary evidence\"]]},\"tunes\":{}},{\"id\":\"h-production\",\"type\":\"header\",\"data\":{\"text\":\"How to evaluate the agent in production\",\"level\":2},\"tunes\":{}},{\"id\":\"production-flow\",\"type\":\"processFlow\",\"data\":{\"title\":\"Evidence evaluation pipeline\",\"orientation\":\"auto\",\"steps\":[{\"label\":\"1. Define material claims\",\"description\":\"Identify the facts, recommendations, or decisions whose correctness matters to the task.\"},{\"label\":\"2. Build gold evidence\",\"description\":\"Create reference evidence and source-quality expectations for representative cases.\"},{\"label\":\"3. Capture traces\",\"description\":\"Log retrieval queries, tool calls, returned evidence, context construction, model output, and citations.\"},{\"label\":\"4. Grade support\",\"description\":\"Check whether each material claim is directly, partially, contradictorily, or not supported.\"},{\"label\":\"5. Grade authority and applicability\",\"description\":\"Evaluate whether the source is appropriate and whether its conditions match the current task.\"},{\"label\":\"6. Run counterfactuals\",\"description\":\"Remove, replace, or supersede decisive evidence and test whether the answer follows the change.\"},{\"label\":\"7. Review high-impact failures\",\"description\":\"Use human or domain-expert review where automated grading is not reliable enough.\"},{\"label\":\"8. Convert failures into eval cases\",\"description\":\"Add production failures and edge cases to a repeatable regression dataset.\"}]},\"tunes\":{}},{\"id\":\"p-production-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"OpenAI's current evaluation guidance recommends task-specific evals, continuous evaluation, production-derived datasets, and traces for debugging agent behaviour. Anthropic likewise recommends combining grader types for research agents because correctness, source quality, coverage, and groundedness are separate dimensions. Evidence evaluation should follow the same pattern: several narrow graders are more diagnostic than one opaque “quality” score.\"},\"tunes\":{}},{\"id\":\"h-judge\",\"type\":\"header\",\"data\":{\"text\":\"Do not let an LLM judge become the only evidence judge\",\"level\":2},\"tunes\":{}},{\"id\":\"p-judge-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"LLM graders are useful for scalable claim classification, relevance checks, and pairwise comparisons, but they can share the same blind spots as the system they evaluate. A grader may accept a plausible but unsupported claim, miss a subtle version boundary, or overrate a polished source.\"},\"tunes\":{}},{\"id\":\"p-judge-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"OpenAI's evaluation guidance recommends calibrating automated graders against human judgment and using clear, scoped criteria. For evidence-heavy systems, deterministic checks should be used wherever possible: timestamps, object versions, permission scope, exact source IDs, retrieval order, document hashes, and whether the evidence was present before the model generated the claim.\"},\"tunes\":{}},{\"id\":\"h-change\",\"type\":\"header\",\"data\":{\"text\":\"What would change this answer?\",\"level\":2},\"tunes\":{}},{\"id\":\"p-change-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The evaluation can be simpler when the agent operates over a small, immutable, authoritative corpus and every answer is strictly extractive. In that environment, source authority and applicability are mostly fixed, and claim-to-span support may be enough.\"},\"tunes\":{}},{\"id\":\"p-change-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The evaluation must become stricter when the agent mixes web search, long-term memory, live tools, multiple jurisdictions, rapidly changing information, user-specific state, or autonomous actions. In those systems, evidence validity depends not only on the source text but also on when and how the evidence was obtained.\"},\"tunes\":{}},{\"id\":\"p-change-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"Future models may become better at internally tracking provenance and uncertainty, but that would not remove the need for external evidence records in systems that require auditability. A system should not depend on the model's self-report of what influenced it when traces and source metadata can provide stronger evidence.\"},\"tunes\":{}},{\"id\":\"h-limitations\",\"type\":\"header\",\"data\":{\"text\":\"Limitations\",\"level\":2},\"tunes\":{}},{\"id\":\"p-limit-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"It is not always possible to prove causal evidence use from traces alone. A source can be present in context without influencing the answer, and a model may independently know the same fact. Counterfactual tests strengthen the inference but can themselves change the task distribution.\"},\"tunes\":{}},{\"id\":\"p-limit-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Source authority can also be contested or domain-dependent. Some questions have no single authoritative source, and experts may disagree about which evidence deserves more weight. In those cases the evaluator should preserve disagreement and score transparency, coverage, and reasoning against an explicit rubric rather than pretending there is one unquestioned source of truth.\"},\"tunes\":{}},{\"id\":\"h-conclusion\",\"type\":\"header\",\"data\":{\"text\":\"Conclusion\",\"level\":2},\"tunes\":{}},{\"id\":\"p-conclusion-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The question “Did the agent cite a source?” is too weak for production AI. The stronger question is: Did each important claim come from evidence that actually supports it, has the right authority, still applies to the current conditions, and was available in the execution path before the claim was made?\"},\"tunes\":{}},{\"id\":\"p-conclusion-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"That turns evidence from decoration into an evaluable system property. Capture the trace. Map claims to evidence. Check authority and applicability. Run counterfactual evidence tests. Preserve provenance through summaries and memory. Then a correct answer is not only plausible — it has an evidence path you can inspect.\"},\"tunes\":{}},{\"id\":\"internal-reliability\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough\",\"title\":\"AI Agent Reliability: Why the Final Answer Is Not Enough\",\"excerpt\":\"Outcome correctness alone cannot prove that an agent's execution path was safe or reliable. This related article explains why trajectories, tools and intermediate decisions matter.\",\"ctaLabel\":\"Read the related article\"},\"tunes\":{}},{\"id\":\"internal-reasoning\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Ffrom-research-protocol-to-a-general-ai-reasoning-framework\",\"title\":\"From Research Protocol to a General AI Reasoning Framework\",\"excerpt\":\"A practical reasoning method for separating evidence, assumptions, competing hypotheses and domain-specific validation.\",\"ctaLabel\":\"Read the reasoning framework\"},\"tunes\":{}},{\"id\":\"h-faq\",\"type\":\"header\",\"data\":{\"text\":\"FAQ\",\"level\":2},\"tunes\":{}},{\"id\":\"faq\",\"type\":\"faq\",\"data\":{\"title\":\"Evaluating evidence use in AI agents\",\"items\":[{\"id\":\"faq1\",\"question\":\"Does a citation prove that an AI answer is grounded?\",\"answer\":\"No. A citation may be relevant to the topic without supporting the exact claim, may come from the wrong authority, may no longer apply, or may have been attached without materially influencing the generated answer.\"},{\"id\":\"faq2\",\"question\":\"How can I test whether an AI agent actually used retrieved evidence?\",\"answer\":\"Use a counterfactual evidence-utilization test: run the task with known-correct evidence, then remove or replace the decisive evidence while keeping other inputs stable. If the answer does not respond to the evidence change, inspect whether the model is relying on prior knowledge or another source.\"},{\"id\":\"faq3\",\"question\":\"What is the difference between groundedness and source quality?\",\"answer\":\"Groundedness asks whether claims are supported by the supplied evidence. Source quality asks whether the evidence itself is appropriate and authoritative enough for the type of claim being made.\"},{\"id\":\"faq4\",\"question\":\"Why can a real source still produce a wrong AI answer?\",\"answer\":\"The source may be stale, superseded, valid for another version, jurisdiction, user, population, or system state, or may contain conditions that were lost during retrieval or summarization.\"},{\"id\":\"faq5\",\"question\":\"What should I log for evidence evaluation?\",\"answer\":\"Log the retrieval query, returned sources, exact evidence spans, timestamps and versions, filters, final context, model output, citations, and trace ordering so evaluators can reconstruct what evidence was available before each material claim.\"}]},\"tunes\":{}},{\"id\":\"h-glossary\",\"type\":\"header\",\"data\":{\"text\":\"Glossary\",\"level\":2},\"tunes\":{}},{\"id\":\"glossary\",\"type\":\"glossary\",\"data\":{\"title\":\"Key evidence-evaluation terms\",\"entries\":[{\"term\":\"Claim support\",\"definition\":\"The degree to which a specific evidence span directly establishes a generated claim.\",\"anchor\":\"claim-support\"},{\"term\":\"Evidence authority\",\"definition\":\"How appropriate a source is for establishing a particular type of claim, given its role, provenance and relationship to the underlying fact.\",\"anchor\":\"evidence-authority\"},{\"term\":\"Applicability\",\"definition\":\"The conditions under which evidence remains valid for a claim, including time, version, jurisdiction, user, population, system state or other boundaries.\",\"anchor\":\"applicability\"},{\"term\":\"Evidence utilization\",\"definition\":\"Whether the agent's output actually responds to and depends on the evidence made available in its execution path.\",\"anchor\":\"evidence-utilization\"},{\"term\":\"Counterfactual evidence test\",\"definition\":\"An evaluation that removes, replaces or changes decisive evidence to test whether the agent's claim changes appropriately.\",\"anchor\":\"counterfactual-evidence-test\"},{\"term\":\"Provenance\",\"definition\":\"Metadata that records where evidence came from, when it was obtained, how it was transformed and what version or state it represented.\",\"anchor\":\"provenance\"}]},\"tunes\":{}},{\"id\":\"h-sources\",\"type\":\"header\",\"data\":{\"text\":\"Primary sources and further reading\",\"level\":2},\"tunes\":{}},{\"id\":\"src-openai-agent-evals\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fagent-evals\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — Evaluate Agent Workflows\",\"description\":\"Guidance on trace grading, workflow-level evaluation, datasets and repeatable eval runs for agents.\"}},\"tunes\":{}},{\"id\":\"src-openai-eval-best\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fevaluation-best-practices\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — Evaluation Best Practices\",\"description\":\"Guidance on task-specific evals, production-derived datasets, scoped metrics, continuous evaluation and grader calibration.\"}},\"tunes\":{}},{\"id\":\"src-anthropic-evals\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Fdemystifying-evals-for-ai-agents\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Anthropic — Demystifying Evals for AI Agents\",\"description\":\"Agent-evaluation guidance including groundedness, coverage and source-quality checks for research agents.\"}},\"tunes\":{}},{\"id\":\"src-openai-thirdparty\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fopenai.com\u002Findex\u002Ftrustworthy-third-party-evaluations-foundations\u002F\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — A Shared Playbook for Trustworthy Third-Party Evaluations\",\"description\":\"Evaluation guidance emphasizing that modern agent performance depends on workflow and environment, not only final model output.\"}},\"tunes\":{}}],\"version\":\"2.31.6\"}",{"time":891,"blocks":892,"version":1411},1790351390243,[893,897,901,906,911,915,919,923,927,951,955,959,963,967,971,975,979,983,987,992,996,1000,1004,1008,1012,1038,1043,1047,1081,1085,1089,1093,1117,1121,1125,1129,1133,1137,1141,1145,1149,1153,1157,1184,1188,1228,1232,1261,1265,1269,1273,1277,1281,1285,1289,1293,1297,1301,1305,1309,1313,1317,1324,1331,1335,1355,1359,1379,1383,1390,1397,1404],{"id":215,"data":894,"type":220,"tunes":896},{"title":895,"maxLevel":218,"minLevel":219},"Contents",{},{"id":223,"data":898,"type":226,"tunes":900},{"text":899},"An AI agent can cite sources, retrieve documents, and still use the wrong evidence. A source may be authoritative but irrelevant to the exact claim. A retrieved passage may support only part of an answer. A correct source can be stale, superseded, or valid for the wrong jurisdiction, product version, user, or system state. This creates a harder evaluation problem than simple citation checking: did the agent actually use the right evidence for the claim it made?",{},{"id":229,"data":902,"type":234,"tunes":905},{"body":903,"title":904,"variant":233},"To know whether an AI agent used the right evidence, evaluate the chain \u003Cstrong>Claim → Evidence → Applicability → Use\u003C\u002Fstrong>. For each material claim, verify that the evidence directly supports it, comes from an appropriate authority, applies to the current conditions, and was actually available to the agent before the claim was produced. A citation alone proves none of those things.","Direct answer",{},{"id":237,"data":907,"type":234,"tunes":910},{"body":908,"title":909,"variant":241},"The Claim–Evidence–Applicability–Use model and Evidence Utilization Test in this article are practical evaluation methods, not formal industry standards. They build on established ideas such as groundedness, source quality, citation coverage, trace evaluation, and task-specific evals.","About the method",{},{"id":244,"data":912,"type":42,"tunes":914},{"text":913,"level":219},"Why citations are not enough",{},{"id":249,"data":916,"type":226,"tunes":918},{"text":917},"A citation answers only a narrow question: the system associated a claim or response with a source. It does not automatically establish that the source supports the specific claim, that the source is authoritative enough for the task, that the cited passage contains the necessary condition or exception, or that the model relied on that evidence rather than producing the answer from prior model knowledge.",{},{"id":254,"data":920,"type":226,"tunes":922},{"text":921},"Anthropic's guidance for research-agent evaluation explicitly separates groundedness, coverage, and source quality. OpenAI's agent-evaluation guidance similarly emphasizes traces because a final output does not reveal whether the agent selected the right tools or followed the intended workflow. Those ideas point to a broader conclusion: evidence quality is a property of the execution path, not just the final prose.",{},{"id":259,"data":924,"type":42,"tunes":926},{"text":925,"level":219},"The four questions every material claim should pass",{},{"id":264,"data":928,"type":287,"tunes":950},{"content":929,"stretched":43,"withHeadings":14},[930,934,938,942,946],[931,932,933],"Dimension","Question","Typical failure",[935,936,937],"Claim support","Does the evidence directly support this exact claim?","The source is topically related but does not establish the statement",[939,940,941],"Evidence authority","Is this an appropriate source for this kind of claim?","A secondary summary is used where a primary source or live system is required",[943,944,945],"Applicability","Does the evidence apply to this time, version, jurisdiction, user, state, or population?","A true statement is applied outside its valid conditions",[947,948,949],"Evidence use","Was this evidence actually available and used in the agent's execution path?","The final answer is correct, but the retrieved evidence was irrelevant or unused",{},{"id":290,"data":952,"type":42,"tunes":954},{"text":953,"level":218},"1. Claim support: does the source establish what the agent says?",{},{"id":295,"data":956,"type":226,"tunes":958},{"text":957},"Evidence should be evaluated at claim level. A document can be relevant to the subject and still fail to support a specific statement. If a source says that a feature is available in selected regions, the answer “the feature is available globally” is unsupported even though the citation looks plausible.",{},{"id":300,"data":960,"type":226,"tunes":962},{"text":961},"This is where broad “grounded \u002F not grounded” judgments are often too coarse. Split the response into material claims, map each claim to the smallest evidence span that supports it, and classify the relationship: direct support, partial support, contradiction, or no support.",{},{"id":305,"data":964,"type":42,"tunes":966},{"text":965,"level":218},"2. Evidence authority: is this the right kind of source?",{},{"id":310,"data":968,"type":226,"tunes":970},{"text":969},"Correct evidence selection is not only semantic relevance. The source must be suitable for the decision. Current account status should come from the account system, not an old email. An API behaviour claim should preferably be checked against current vendor documentation or reproducible behaviour. A legal requirement may need the applicable law, regulator, or authoritative guidance rather than a generic blog post.",{},{"id":315,"data":972,"type":226,"tunes":974},{"text":973},"Source authority is task-specific. A community report can be the best evidence for a real-world bug that vendor documentation does not acknowledge. A vendor announcement can be authoritative for what the vendor claims but weak evidence for independent performance. The evaluator therefore needs an explicit source hierarchy for the task rather than one universal authority score.",{},{"id":320,"data":976,"type":42,"tunes":978},{"text":977,"level":218},"3. Applicability: right evidence, wrong conditions",{},{"id":325,"data":980,"type":226,"tunes":982},{"text":981},"The most dangerous evidence errors are often not fabricated sources but valid sources used outside their boundary. A recommendation can change with software version, date, jurisdiction, hardware revision, user permissions, product availability, current game state, tenant configuration, or other environmental variables.",{},{"id":330,"data":984,"type":226,"tunes":986},{"text":985},"For each material source, preserve the conditions that determine whether it still applies. This is especially important after summarization: a compressed memory or citation may preserve the conclusion while dropping the exception, date, or prerequisite that made the conclusion valid.",{},{"id":335,"data":988,"type":234,"tunes":991},{"body":989,"title":990,"variant":339},"A real source, quoted accurately, can still produce a wrong answer when its time, version, population, jurisdiction, or state does not match the current question.","Evidence can be authentic and still be wrong for the answer",{},{"id":342,"data":993,"type":42,"tunes":995},{"text":994,"level":218},"4. Evidence use: did the agent actually rely on the evidence?",{},{"id":347,"data":997,"type":226,"tunes":999},{"text":998},"An answer can be correct even when retrieval failed. The model may already know the answer, infer it from unrelated context, or simply guess correctly. If the evaluation checks only final correctness, the system may appear well-grounded while the evidence path is broken.",{},{"id":352,"data":1001,"type":226,"tunes":1003},{"text":1002},"To evaluate evidence use, inspect the trace. Confirm which sources were retrieved, which passages reached the model context, when they became available, and whether the final claim can be explained by those inputs. OpenAI's current agent-evaluation tooling emphasizes trace grading precisely because workflow-level behaviour cannot be reconstructed reliably from the final answer alone.",{},{"id":357,"data":1005,"type":42,"tunes":1007},{"text":1006,"level":219},"The Evidence Utilization Test",{},{"id":362,"data":1009,"type":226,"tunes":1011},{"text":1010},"A practical evaluation can be built as a controlled counterfactual. Instead of asking only whether the answer is correct, change the evidence and observe whether the claim changes in the expected direction.",{},{"id":367,"data":1013,"type":392,"tunes":1037},{"steps":1014,"title":1036,"orientation":391},[1015,1018,1021,1024,1027,1030,1033],{"label":1016,"description":1017},"1. Select one material claim","Choose a claim whose correctness matters and define the expected answer precisely.",{"label":1019,"description":1020},"2. Identify gold evidence","Provide the smallest authoritative evidence set sufficient to support the claim.",{"label":1022,"description":1023},"3. Run with gold evidence","Verify that the agent produces the supported answer when the correct evidence is available.",{"label":1025,"description":1026},"4. Remove the decisive evidence","Run the same task without the key supporting passage while keeping other inputs stable.",{"label":1028,"description":1029},"5. Replace it with contradictory or superseding evidence","Where safe, provide controlled evidence that changes the correct conclusion.",{"label":1031,"description":1032},"6. Compare the claims","Check whether the answer tracks the evidence change or remains anchored to prior model knowledge.",{"label":1034,"description":1035},"7. Inspect the trace","Confirm what was retrieved, what reached context, and what source or tool result preceded the claim.","Evidence Utilization Test",{},{"id":395,"data":1039,"type":234,"tunes":1042},{"body":1040,"title":1041,"variant":399},"If the decisive evidence changes but the agent's claim does not, you have evidence that the system may not be using retrieval as intended — even when the original answer happened to be correct.","The key signal",{},{"id":402,"data":1044,"type":42,"tunes":1046},{"text":1045,"level":219},"A claim–evidence matrix is more useful than a source list",{},{"id":407,"data":1048,"type":287,"tunes":1080},{"content":1049,"stretched":43,"withHeadings":14},[1050,1056,1063,1069,1075],[1051,1052,1053,1054,943,1055],"Claim","Evidence","Support","Authority","Used in trace",[1057,1058,1059,1060,1061,1062],"Feature X is available","Vendor documentation","Direct","High for availability claim","Current version and region must match","Yes \u002F No",[1064,1065,1066,1067,1068,1062],"Configuration Y is faster","Vendor benchmark","Partial","High for vendor's test, not independent performance","Hardware and workload must match",[1070,1071,1072,1073,1074,1062],"Policy applies to this user","Current policy + account state","Direct only when combined","High","Jurisdiction, date, role and account state must match",[1076,1077,1059,1078,1079,1062],"A product is in stock","Live inventory API","Authoritative for current stock","Expires quickly",{},{"id":442,"data":1082,"type":226,"tunes":1084},{"text":1083},"This matrix forces several questions that conventional citation checking hides. One claim can require multiple sources. One source can support only part of a claim. An authoritative source can have a short validity window. And a perfectly good source can be irrelevant if it never entered the execution path.",{},{"id":447,"data":1086,"type":42,"tunes":1088},{"text":1087,"level":219},"Separate retrieval quality from evidence quality",{},{"id":452,"data":1090,"type":226,"tunes":1092},{"text":1091},"Retrieval metrics ask whether relevant material was found and ranked. Evidence evaluation asks whether that material justifies the resulting claims. The two are related but not identical.",{},{"id":457,"data":1094,"type":488,"tunes":1116},{"rows":1095,"title":1108,"layout":287,"columns":1109},[1096,1099,1102,1105],{"id":461,"label":1097,"values":1098},"Right document, wrong claim",[464,464,464],{"id":466,"label":1100,"values":1101},"Right fact, stale source",[464,464,464],{"id":470,"label":1103,"values":1104},"Weak source, correct answer",[464,464,464],{"id":474,"label":1106,"values":1107},"Multiple sources required",[464,464,464],"Retrieval success is not evidence success",[1110,1112,1114],{"id":480,"label":1111},"Situation",{"id":483,"label":1113},"Retrieval",{"id":486,"label":1115},"Evidence quality",{},{"id":491,"data":1118,"type":42,"tunes":1120},{"text":1119,"level":219},"Evaluate source quality as a rubric, not a domain whitelist",{},{"id":496,"data":1122,"type":226,"tunes":1124},{"text":1123},"Hard-coded lists of “trusted domains” are tempting but often brittle. Source quality should instead reflect the claim type. Useful dimensions include primary versus secondary status, recency, directness, reproducibility, independence, domain expertise, data provenance, update cadence, and whether the source has an incentive to overstate the claim.",{},{"id":501,"data":1126,"type":226,"tunes":1128},{"text":1127},"Anthropic's research-agent evaluation guidance explicitly calls out source-quality checks alongside groundedness and coverage. The practical implementation should therefore grade both what the source says and whether this source is appropriate for this kind of statement.",{},{"id":506,"data":1130,"type":42,"tunes":1132},{"text":1131,"level":219},"Evidence coverage: every important claim needs support, not every sentence",{},{"id":511,"data":1134,"type":226,"tunes":1136},{"text":1135},"Not every sentence needs a citation. Transitional language, arithmetic derived transparently from cited values, or clearly marked interpretation may not require a separate source. But every material externally verifiable claim should have enough support that an evaluator can reconstruct why the agent was allowed to say it.",{},{"id":516,"data":1138,"type":226,"tunes":1140},{"text":1139},"Coverage should therefore be weighted by claim importance. Missing support for a decorative detail is not equivalent to missing support for a price, eligibility decision, safety instruction, legal requirement, technical compatibility statement, or recommendation-driving fact.",{},{"id":521,"data":1142,"type":42,"tunes":1144},{"text":1143,"level":219},"Evidence provenance must survive summarization and memory",{},{"id":526,"data":1146,"type":226,"tunes":1148},{"text":1147},"Long-running agents often summarize previous work or write durable memories. If evidence provenance is stripped during that transformation, future agents may retrieve a clean conclusion without knowing whether it came from a user statement, a live API, an old document, a model inference, or an unverified web result.",{},{"id":531,"data":1150,"type":226,"tunes":1152},{"text":1151},"For important facts, preserve at least the source identity, retrieval or observation time, evidence type, relevant version or state, and whether the stored text is quoted, summarized, inferred, or derived. Provenance is what allows a later agent to decide whether the evidence should be trusted, refreshed, restricted, or discarded.",{},{"id":536,"data":1154,"type":42,"tunes":1156},{"text":1155,"level":219},"A practical evidence record",{},{"id":541,"data":1158,"type":287,"tunes":1183},{"content":1159,"stretched":43,"withHeadings":14},[1160,1163,1165,1167,1169,1171,1173,1175,1177,1179,1181],[1161,1162],"Field","Purpose",[548,1164],"Identifies the material claim being supported",[551,1166],"Identifies where the evidence came from",[554,1168],"Preserves the smallest passage, record, or tool result that supports the claim",[557,1170],"Allows freshness and timeline checks",[560,1172],"Allows supersession and reproducibility checks",[563,1174],"Explains why this source is suitable for this claim",[566,1176],"Stores relevant date, jurisdiction, product version, user, tenant, state, or other conditions",[569,1178],"Marks whether evidence is raw, quoted, summarized, normalized, or derived",[572,1180],"Shows when the evidence became available to the agent",[575,1182],"Direct, partial, contradictory, unsupported, or uncertain",{},{"id":579,"data":1185,"type":42,"tunes":1187},{"text":1186,"level":219},"Failure modes that look grounded but are not",{},{"id":584,"data":1189,"type":287,"tunes":1227},{"content":1190,"stretched":43,"withHeadings":14},[1191,1195,1199,1203,1207,1211,1215,1219,1223],[1192,1193,1194],"Failure mode","Why it fools evaluators","What to test",[1196,1197,1198],"Citation decoration","The answer contains sources, so it looks researched","Map each material claim to an exact supporting span",[1200,1201,1202],"Authority mismatch","The source is reputable but not authoritative for the specific fact","Define claim-specific source hierarchy",[1204,1205,1206],"Temporal mismatch","The source was correct when published","Check retrieval time, source date, and superseding evidence",[1208,1209,1210],"Condition stripping","A summary keeps the conclusion but drops exceptions","Compare generated claim with full local source context",[1212,1213,1214],"Post-hoc citation","A plausible source is attached after the answer is generated","Inspect trace ordering and whether evidence preceded the claim",[1216,1217,1218],"Parametric override","The model ignores retrieved evidence and answers from prior knowledge","Run counterfactual evidence-utilization tests",[1220,1221,1222],"Evidence laundering","Model inference is summarized and later stored as if it were a source fact","Preserve transformation type and provenance across memory writes",[1224,1225,1226],"Source majority fallacy","Several secondary pages repeat the same unsupported statement","Trace claims back to independent or primary evidence",{},{"id":625,"data":1229,"type":42,"tunes":1231},{"text":1230,"level":219},"How to evaluate the agent in production",{},{"id":630,"data":1233,"type":392,"tunes":1260},{"steps":1234,"title":1259,"orientation":391},[1235,1238,1241,1244,1247,1250,1253,1256],{"label":1236,"description":1237},"1. Define material claims","Identify the facts, recommendations, or decisions whose correctness matters to the task.",{"label":1239,"description":1240},"2. Build gold evidence","Create reference evidence and source-quality expectations for representative cases.",{"label":1242,"description":1243},"3. Capture traces","Log retrieval queries, tool calls, returned evidence, context construction, model output, and citations.",{"label":1245,"description":1246},"4. Grade support","Check whether each material claim is directly, partially, contradictorily, or not supported.",{"label":1248,"description":1249},"5. Grade authority and applicability","Evaluate whether the source is appropriate and whether its conditions match the current task.",{"label":1251,"description":1252},"6. Run counterfactuals","Remove, replace, or supersede decisive evidence and test whether the answer follows the change.",{"label":1254,"description":1255},"7. Review high-impact failures","Use human or domain-expert review where automated grading is not reliable enough.",{"label":1257,"description":1258},"8. Convert failures into eval cases","Add production failures and edge cases to a repeatable regression dataset.","Evidence evaluation pipeline",{},{"id":660,"data":1262,"type":226,"tunes":1264},{"text":1263},"OpenAI's current evaluation guidance recommends task-specific evals, continuous evaluation, production-derived datasets, and traces for debugging agent behaviour. Anthropic likewise recommends combining grader types for research agents because correctness, source quality, coverage, and groundedness are separate dimensions. Evidence evaluation should follow the same pattern: several narrow graders are more diagnostic than one opaque “quality” score.",{},{"id":665,"data":1266,"type":42,"tunes":1268},{"text":1267,"level":219},"Do not let an LLM judge become the only evidence judge",{},{"id":670,"data":1270,"type":226,"tunes":1272},{"text":1271},"LLM graders are useful for scalable claim classification, relevance checks, and pairwise comparisons, but they can share the same blind spots as the system they evaluate. A grader may accept a plausible but unsupported claim, miss a subtle version boundary, or overrate a polished source.",{},{"id":675,"data":1274,"type":226,"tunes":1276},{"text":1275},"OpenAI's evaluation guidance recommends calibrating automated graders against human judgment and using clear, scoped criteria. For evidence-heavy systems, deterministic checks should be used wherever possible: timestamps, object versions, permission scope, exact source IDs, retrieval order, document hashes, and whether the evidence was present before the model generated the claim.",{},{"id":680,"data":1278,"type":42,"tunes":1280},{"text":1279,"level":219},"What would change this answer?",{},{"id":685,"data":1282,"type":226,"tunes":1284},{"text":1283},"The evaluation can be simpler when the agent operates over a small, immutable, authoritative corpus and every answer is strictly extractive. In that environment, source authority and applicability are mostly fixed, and claim-to-span support may be enough.",{},{"id":690,"data":1286,"type":226,"tunes":1288},{"text":1287},"The evaluation must become stricter when the agent mixes web search, long-term memory, live tools, multiple jurisdictions, rapidly changing information, user-specific state, or autonomous actions. In those systems, evidence validity depends not only on the source text but also on when and how the evidence was obtained.",{},{"id":695,"data":1290,"type":226,"tunes":1292},{"text":1291},"Future models may become better at internally tracking provenance and uncertainty, but that would not remove the need for external evidence records in systems that require auditability. A system should not depend on the model's self-report of what influenced it when traces and source metadata can provide stronger evidence.",{},{"id":700,"data":1294,"type":42,"tunes":1296},{"text":1295,"level":219},"Limitations",{},{"id":705,"data":1298,"type":226,"tunes":1300},{"text":1299},"It is not always possible to prove causal evidence use from traces alone. A source can be present in context without influencing the answer, and a model may independently know the same fact. Counterfactual tests strengthen the inference but can themselves change the task distribution.",{},{"id":710,"data":1302,"type":226,"tunes":1304},{"text":1303},"Source authority can also be contested or domain-dependent. Some questions have no single authoritative source, and experts may disagree about which evidence deserves more weight. In those cases the evaluator should preserve disagreement and score transparency, coverage, and reasoning against an explicit rubric rather than pretending there is one unquestioned source of truth.",{},{"id":715,"data":1306,"type":42,"tunes":1308},{"text":1307,"level":219},"Conclusion",{},{"id":720,"data":1310,"type":226,"tunes":1312},{"text":1311},"The question “Did the agent cite a source?” is too weak for production AI. The stronger question is: Did each important claim come from evidence that actually supports it, has the right authority, still applies to the current conditions, and was available in the execution path before the claim was made?",{},{"id":725,"data":1314,"type":226,"tunes":1316},{"text":1315},"That turns evidence from decoration into an evaluable system property. Capture the trace. Map claims to evidence. Check authority and applicability. Run counterfactual evidence tests. Preserve provenance through summaries and memory. Then a correct answer is not only plausible — it has an evidence path you can inspect.",{},{"id":730,"data":1318,"type":736,"tunes":1323},{"url":1319,"title":1320,"excerpt":1321,"ctaLabel":1322},"https:\u002F\u002Fstajic.de\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough","AI Agent Reliability: Why the Final Answer Is Not Enough","Outcome correctness alone cannot prove that an agent's execution path was safe or reliable. This related article explains why trajectories, tools and intermediate decisions matter.","Read the related article",{},{"id":739,"data":1325,"type":736,"tunes":1330},{"url":1326,"title":1327,"excerpt":1328,"ctaLabel":1329},"https:\u002F\u002Fstajic.de\u002Fblog\u002Ffrom-research-protocol-to-a-general-ai-reasoning-framework","From Research Protocol to a General AI Reasoning Framework","A practical reasoning method for separating evidence, assumptions, competing hypotheses and domain-specific validation.","Read the reasoning framework",{},{"id":747,"data":1332,"type":42,"tunes":1334},{"text":1333,"level":219},"FAQ",{},{"id":752,"data":1336,"type":752,"tunes":1354},{"items":1337,"title":1353},[1338,1341,1344,1347,1350],{"id":756,"answer":1339,"question":1340},"No. A citation may be relevant to the topic without supporting the exact claim, may come from the wrong authority, may no longer apply, or may have been attached without materially influencing the generated answer.","Does a citation prove that an AI answer is grounded?",{"id":760,"answer":1342,"question":1343},"Use a counterfactual evidence-utilization test: run the task with known-correct evidence, then remove or replace the decisive evidence while keeping other inputs stable. If the answer does not respond to the evidence change, inspect whether the model is relying on prior knowledge or another source.","How can I test whether an AI agent actually used retrieved evidence?",{"id":764,"answer":1345,"question":1346},"Groundedness asks whether claims are supported by the supplied evidence. Source quality asks whether the evidence itself is appropriate and authoritative enough for the type of claim being made.","What is the difference between groundedness and source quality?",{"id":768,"answer":1348,"question":1349},"The source may be stale, superseded, valid for another version, jurisdiction, user, population, or system state, or may contain conditions that were lost during retrieval or summarization.","Why can a real source still produce a wrong AI answer?",{"id":772,"answer":1351,"question":1352},"Log the retrieval query, returned sources, exact evidence spans, timestamps and versions, filters, final context, model output, citations, and trace ordering so evaluators can reconstruct what evidence was available before each material claim.","What should I log for evidence evaluation?","Evaluating evidence use in AI agents",{},{"id":778,"data":1356,"type":42,"tunes":1358},{"text":1357,"level":219},"Glossary",{},{"id":783,"data":1360,"type":783,"tunes":1378},{"title":1361,"entries":1362},"Key evidence-evaluation terms",[1363,1365,1367,1369,1372,1375],{"term":935,"anchor":788,"definition":1364},"The degree to which a specific evidence span directly establishes a generated claim.",{"term":939,"anchor":791,"definition":1366},"How appropriate a source is for establishing a particular type of claim, given its role, provenance and relationship to the underlying fact.",{"term":943,"anchor":566,"definition":1368},"The conditions under which evidence remains valid for a claim, including time, version, jurisdiction, user, population, system state or other boundaries.",{"term":1370,"anchor":797,"definition":1371},"Evidence utilization","Whether the agent's output actually responds to and depends on the evidence made available in its execution path.",{"term":1373,"anchor":801,"definition":1374},"Counterfactual evidence test","An evaluation that removes, replaces or changes decisive evidence to test whether the agent's claim changes appropriately.",{"term":1376,"anchor":805,"definition":1377},"Provenance","Metadata that records where evidence came from, when it was obtained, how it was transformed and what version or state it represented.",{},{"id":809,"data":1380,"type":42,"tunes":1382},{"text":1381,"level":219},"Primary sources and further reading",{},{"id":814,"data":1384,"type":821,"tunes":1389},{"link":816,"meta":1385},{"image":1386,"title":1387,"description":1388},{"url":464},"OpenAI — Evaluate Agent Workflows","Guidance on trace grading, workflow-level evaluation, datasets and repeatable eval runs for agents.",{},{"id":824,"data":1391,"type":821,"tunes":1396},{"link":826,"meta":1392},{"image":1393,"title":1394,"description":1395},{"url":464},"OpenAI — Evaluation Best Practices","Guidance on task-specific evals, production-derived datasets, scoped metrics, continuous evaluation and grader calibration.",{},{"id":833,"data":1398,"type":821,"tunes":1403},{"link":835,"meta":1399},{"image":1400,"title":1401,"description":1402},{"url":464},"Anthropic — Demystifying Evals for AI Agents","Agent-evaluation guidance including groundedness, coverage and source-quality checks for research agents.",{},{"id":842,"data":1405,"type":821,"tunes":1410},{"link":844,"meta":1406},{"image":1407,"title":1408,"description":1409},{"url":464},"OpenAI — A Shared Playbook for Trustworthy Third-Party Evaluations","Evaluation guidance emphasizing that modern agent performance depends on workflow and environment, not only final model output.",{},"2.31.6","An AI agent can cite sources and still use the wrong evidence. This article introduces a practical method for checking claim support, source authority, applicability, provenance, and whether the evidence actually influenced the answer.",{"lang":7,"title":208,"content":210,"contentJson":1414,"excerpt":851},{"time":212,"blocks":1415,"version":850},[1416,1419,1422,1425,1428,1431,1434,1437,1440,1449,1452,1455,1458,1461,1464,1467,1470,1473,1476,1479,1482,1485,1488,1491,1494,1505,1508,1511,1520,1523,1526,1529,1545,1548,1551,1554,1557,1560,1563,1566,1569,1572,1575,1590,1593,1606,1609,1621,1624,1627,1630,1633,1636,1639,1642,1645,1648,1651,1654,1657,1660,1663,1666,1669,1672,1681,1684,1694,1697,1702,1707,1712],{"id":215,"data":1417,"type":220,"tunes":1418},{"title":217,"maxLevel":218,"minLevel":219},{},{"id":223,"data":1420,"type":226,"tunes":1421},{"text":225},{},{"id":229,"data":1423,"type":234,"tunes":1424},{"body":231,"title":232,"variant":233},{},{"id":237,"data":1426,"type":234,"tunes":1427},{"body":239,"title":240,"variant":241},{},{"id":244,"data":1429,"type":42,"tunes":1430},{"text":246,"level":219},{},{"id":249,"data":1432,"type":226,"tunes":1433},{"text":251},{},{"id":254,"data":1435,"type":226,"tunes":1436},{"text":256},{},{"id":259,"data":1438,"type":42,"tunes":1439},{"text":261,"level":219},{},{"id":264,"data":1441,"type":287,"tunes":1448},{"content":1442,"stretched":43,"withHeadings":14},[1443,1444,1445,1446,1447],[268,269,270],[272,273,274],[276,277,278],[280,281,282],[284,285,286],{},{"id":290,"data":1450,"type":42,"tunes":1451},{"text":292,"level":218},{},{"id":295,"data":1453,"type":226,"tunes":1454},{"text":297},{},{"id":300,"data":1456,"type":226,"tunes":1457},{"text":302},{},{"id":305,"data":1459,"type":42,"tunes":1460},{"text":307,"level":218},{},{"id":310,"data":1462,"type":226,"tunes":1463},{"text":312},{},{"id":315,"data":1465,"type":226,"tunes":1466},{"text":317},{},{"id":320,"data":1468,"type":42,"tunes":1469},{"text":322,"level":218},{},{"id":325,"data":1471,"type":226,"tunes":1472},{"text":327},{},{"id":330,"data":1474,"type":226,"tunes":1475},{"text":332},{},{"id":335,"data":1477,"type":234,"tunes":1478},{"body":337,"title":338,"variant":339},{},{"id":342,"data":1480,"type":42,"tunes":1481},{"text":344,"level":218},{},{"id":347,"data":1483,"type":226,"tunes":1484},{"text":349},{},{"id":352,"data":1486,"type":226,"tunes":1487},{"text":354},{},{"id":357,"data":1489,"type":42,"tunes":1490},{"text":359,"level":219},{},{"id":362,"data":1492,"type":226,"tunes":1493},{"text":364},{},{"id":367,"data":1495,"type":392,"tunes":1504},{"steps":1496,"title":359,"orientation":391},[1497,1498,1499,1500,1501,1502,1503],{"label":371,"description":372},{"label":374,"description":375},{"label":377,"description":378},{"label":380,"description":381},{"label":383,"description":384},{"label":386,"description":387},{"label":389,"description":390},{},{"id":395,"data":1506,"type":234,"tunes":1507},{"body":397,"title":398,"variant":399},{},{"id":402,"data":1509,"type":42,"tunes":1510},{"text":404,"level":219},{},{"id":407,"data":1512,"type":287,"tunes":1519},{"content":1513,"stretched":43,"withHeadings":14},[1514,1515,1516,1517,1518],[411,412,413,414,280,415],[417,418,419,420,421,422],[424,425,426,427,428,422],[430,431,432,433,434,422],[436,437,419,438,439,422],{},{"id":442,"data":1521,"type":226,"tunes":1522},{"text":444},{},{"id":447,"data":1524,"type":42,"tunes":1525},{"text":449,"level":219},{},{"id":452,"data":1527,"type":226,"tunes":1528},{"text":454},{},{"id":457,"data":1530,"type":488,"tunes":1544},{"rows":1531,"title":477,"layout":287,"columns":1540},[1532,1534,1536,1538],{"id":461,"label":462,"values":1533},[464,464,464],{"id":466,"label":467,"values":1535},[464,464,464],{"id":470,"label":471,"values":1537},[464,464,464],{"id":474,"label":475,"values":1539},[464,464,464],[1541,1542,1543],{"id":480,"label":481},{"id":483,"label":484},{"id":486,"label":487},{},{"id":491,"data":1546,"type":42,"tunes":1547},{"text":493,"level":219},{},{"id":496,"data":1549,"type":226,"tunes":1550},{"text":498},{},{"id":501,"data":1552,"type":226,"tunes":1553},{"text":503},{},{"id":506,"data":1555,"type":42,"tunes":1556},{"text":508,"level":219},{},{"id":511,"data":1558,"type":226,"tunes":1559},{"text":513},{},{"id":516,"data":1561,"type":226,"tunes":1562},{"text":518},{},{"id":521,"data":1564,"type":42,"tunes":1565},{"text":523,"level":219},{},{"id":526,"data":1567,"type":226,"tunes":1568},{"text":528},{},{"id":531,"data":1570,"type":226,"tunes":1571},{"text":533},{},{"id":536,"data":1573,"type":42,"tunes":1574},{"text":538,"level":219},{},{"id":541,"data":1576,"type":287,"tunes":1589},{"content":1577,"stretched":43,"withHeadings":14},[1578,1579,1580,1581,1582,1583,1584,1585,1586,1587,1588],[545,546],[548,549],[551,552],[554,555],[557,558],[560,561],[563,564],[566,567],[569,570],[572,573],[575,576],{},{"id":579,"data":1591,"type":42,"tunes":1592},{"text":581,"level":219},{},{"id":584,"data":1594,"type":287,"tunes":1605},{"content":1595,"stretched":43,"withHeadings":14},[1596,1597,1598,1599,1600,1601,1602,1603,1604],[588,589,590],[592,593,594],[596,597,598],[600,601,602],[604,605,606],[608,609,610],[612,613,614],[616,617,618],[620,621,622],{},{"id":625,"data":1607,"type":42,"tunes":1608},{"text":627,"level":219},{},{"id":630,"data":1610,"type":392,"tunes":1620},{"steps":1611,"title":657,"orientation":391},[1612,1613,1614,1615,1616,1617,1618,1619],{"label":634,"description":635},{"label":637,"description":638},{"label":640,"description":641},{"label":643,"description":644},{"label":646,"description":647},{"label":649,"description":650},{"label":652,"description":653},{"label":655,"description":656},{},{"id":660,"data":1622,"type":226,"tunes":1623},{"text":662},{},{"id":665,"data":1625,"type":42,"tunes":1626},{"text":667,"level":219},{},{"id":670,"data":1628,"type":226,"tunes":1629},{"text":672},{},{"id":675,"data":1631,"type":226,"tunes":1632},{"text":677},{},{"id":680,"data":1634,"type":42,"tunes":1635},{"text":682,"level":219},{},{"id":685,"data":1637,"type":226,"tunes":1638},{"text":687},{},{"id":690,"data":1640,"type":226,"tunes":1641},{"text":692},{},{"id":695,"data":1643,"type":226,"tunes":1644},{"text":697},{},{"id":700,"data":1646,"type":42,"tunes":1647},{"text":702,"level":219},{},{"id":705,"data":1649,"type":226,"tunes":1650},{"text":707},{},{"id":710,"data":1652,"type":226,"tunes":1653},{"text":712},{},{"id":715,"data":1655,"type":42,"tunes":1656},{"text":717,"level":219},{},{"id":720,"data":1658,"type":226,"tunes":1659},{"text":722},{},{"id":725,"data":1661,"type":226,"tunes":1662},{"text":727},{},{"id":730,"data":1664,"type":736,"tunes":1665},{"url":732,"title":733,"excerpt":734,"ctaLabel":735},{},{"id":739,"data":1667,"type":736,"tunes":1668},{"url":741,"title":742,"excerpt":743,"ctaLabel":744},{},{"id":747,"data":1670,"type":42,"tunes":1671},{"text":749,"level":219},{},{"id":752,"data":1673,"type":752,"tunes":1680},{"items":1674,"title":775},[1675,1676,1677,1678,1679],{"id":756,"answer":757,"question":758},{"id":760,"answer":761,"question":762},{"id":764,"answer":765,"question":766},{"id":768,"answer":769,"question":770},{"id":772,"answer":773,"question":774},{},{"id":778,"data":1682,"type":42,"tunes":1683},{"text":780,"level":219},{},{"id":783,"data":1685,"type":783,"tunes":1693},{"title":785,"entries":1686},[1687,1688,1689,1690,1691,1692],{"term":272,"anchor":788,"definition":789},{"term":276,"anchor":791,"definition":792},{"term":280,"anchor":566,"definition":794},{"term":796,"anchor":797,"definition":798},{"term":800,"anchor":801,"definition":802},{"term":804,"anchor":805,"definition":806},{},{"id":809,"data":1695,"type":42,"tunes":1696},{"text":811,"level":219},{},{"id":814,"data":1698,"type":821,"tunes":1701},{"link":816,"meta":1699},{"image":1700,"title":819,"description":820},{"url":464},{},{"id":824,"data":1703,"type":821,"tunes":1706},{"link":826,"meta":1704},{"image":1705,"title":829,"description":830},{"url":464},{},{"id":833,"data":1708,"type":821,"tunes":1711},{"link":835,"meta":1709},{"image":1710,"title":838,"description":839},{"url":464},{},{"id":842,"data":1713,"type":821,"tunes":1716},{"link":844,"meta":1714},{"image":1715,"title":847,"description":848},{"url":464},{},"Post erfolgreich abgerufen",{"items":1719,"source":1783,"manualIds":1784,"manualMatchedIds":1785},[1720,1727,1734,1741,1748,1755,1762,1769,1776],{"id":1721,"slug":1722,"title":1723,"excerpt":1724,"featuredImage":1725,"publishedAt":1726},"383","canonical-architecture-url-design-resolver-logic-api-scalability-specification","规范化架构、URL 设计、解析器逻辑、API 与可扩展性规范","面向多租户门户的地理发现架构。定义了规范化 URL、解析器逻辑、缓存策略以及不依赖 CMS 耦合或数据库重构的地理读模型。该设计旨在确保 SEO 稳定性、高可扩展性，并支持未来的功能扩展，例如预订和地图。","\u002Fuploads\u002F2026\u002F01\u002Fcanonical-architecture-url-design-resolver-logic-api-scalability-specification-1769890763607-7rghbp.webp","2026-01-31T06:12:00.000Z",{"id":1728,"slug":1729,"title":1730,"excerpt":1731,"featuredImage":1732,"publishedAt":1733},"477","computer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system","计算机使用代理：为什么成功的演示仍可能是一个不可靠的系统","计算机使用代理如今能够完成令人印象深刻的浏览器和桌面工作流程，但一次成功的运行证明的是能力——而非可靠性。本文展示了如何测试可重复性、环境鲁棒性、长时程控制、状态感知、结果验证以及安全的目标处理。","\u002Fuploads\u002F2026\u002F09\u002Fcomputer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system-1790352854690-75qnrg.webp","2026-09-25T12:13:00.000Z",{"id":1735,"slug":1736,"title":1737,"excerpt":1738,"featuredImage":1739,"publishedAt":1740},"454","zbt-z8102ax-rm500u-ea-5g-modem-test","Quectel RM500U-EA在ZBT Z8102AX中：5G频段、o2德国及实际信号表现","ZBT Z8102AX 使用移远 RM500U-EA 调制解调器实现 4G 和 5G 连接。在首次实际测试中，该路由器成功连接至德国 o2 网络，使用 LTE Band 3 和 NR n28 频段。调制解调器工作正常，但更深层次的诊断功能如 RSRP、RSRQ、SINR、频段锁定及小区行为仍需进一步测试。","\u002Fuploads\u002F2026\u002F06\u002Fopenwrt-router-review-dual-sim-06-1781620597879-qay2sx.webp","2026-06-16T08:39:00.000Z",{"id":1742,"slug":1743,"title":1744,"excerpt":1745,"featuredImage":1746,"publishedAt":1747},"456","zbt-z8102ax-hardware-packaging-review","ZBT Z8102AX 硬件与包装评测：强劲路由器，薄弱包装","ZBT Z8102AX 作为一款纤薄黑色金属5G OpenWrt路由器，配备多个天线接口、双SIM卡槽、USB、LAN\u002FWAN端口及实用配件套装，给人留下扎实的第一印象。硬件设计实用且专业，但包装显然是薄弱环节。","\u002Fuploads\u002F2026\u002F06\u002Fopenwrt-router-review-dual-sim-02-1781620590938-y33j4b.webp","2026-06-16T04:40:00.000Z",{"id":1749,"slug":1750,"title":1751,"excerpt":1752,"featuredImage":1753,"publishedAt":1754},"476","mcp-vs-a2a-vs-ucp-vs-ap2-vs-a2ui-the-agent-protocol-stack-explained","MCP vs A2A vs UCP vs AP2 vs A2UI：智能体协议栈详解","MCP、A2A、UCP、AP2 和 A2UI 常被描述为相互竞争的智能体标准。它们大多解决的是不同的互操作性问题。本指南将每个协议映射到其实际标准化的边界，并展示它们如何在同一个生产系统中协同工作。","\u002Fuploads\u002F2026\u002F09\u002Fmcp-vs-a2a-vs-ucp-vs-ap2-vs-a2ui-the-agent-protocol-stack-explained-1790352625869-2ezle0.webp","2026-09-25T12:09:00.000Z",{"id":1756,"slug":1757,"title":1758,"excerpt":1759,"featuredImage":1760,"publishedAt":1761},"467","the-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers","答案有效性边界：相关性到可靠AI答案之间缺失的层级","一个来源可能相关、权威，但对于所提出的问题仍然是错误的。缺失的层次是适用性：答案成立的条件，以及迫使其被重新考虑的变化。本文介绍了“答案有效性边界”这一面向人类、AI搜索和RAG系统的来源设计模式。","\u002Fuploads\u002F2026\u002F09\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers-1790272901306-1g5jly.webp","2026-09-24T11:59:00.000Z",{"id":1763,"slug":1764,"title":1765,"excerpt":1766,"featuredImage":1767,"publishedAt":1768},"472","why-more-context-can-make-ai-answers-worse","为什么更多上下文会让AI的回答更糟","更大的上下文窗口并不保证更好的答案。本文解释了信号稀释、证据冲突、状态过时、位置敏感性和有损压缩如何降低AI可靠性——并介绍了一种实用的上下文压力测试。","\u002Fuploads\u002F2026\u002F09\u002Fwhy-more-context-can-make-ai-answers-worse-1790351615793-2ntv2v.webp","2026-09-25T11:51:00.000Z",{"id":1770,"slug":1771,"title":1772,"excerpt":1773,"featuredImage":1774,"publishedAt":1775},"470","what-should-an-ai-agent-remember-forget-recompute-or-retrieve-again","AI代理应该记住、遗忘、重新计算还是再次检索什么？","长时间运行的代理不应记住所有内容。本文提供了一个实用的生命周期模型，用于决定哪些内容应属于持久记忆、哪些内容应重新检索、哪些内容重新计算更安全，以及哪些内容应过期或被取代。","\u002Fuploads\u002F2026\u002F09\u002Fwhat-should-an-ai-agent-remember-forget-recompute-or-retrieve-again-1790351131087-iehz28.webp","2026-09-25T09:43:00.000Z",{"id":1777,"slug":1778,"title":1779,"excerpt":1780,"featuredImage":1781,"publishedAt":1782},"457","should-you-buy-5g-openwrt-router-old-firmware","你应该购买带有旧固件的5G OpenWrt路由器吗？以ZBT Z8102AX为例","购买搭载旧版固件的5G OpenWrt路由器在特定条件下是合理的。ZBT Z8102AX型号清晰展现了利弊两面：硬件实用、调制解调器工作正常，测试中路由器保持稳定，但OpenWrt 21.02版本、简陋的包装以及不明确的升级路径，要求消费者在购买时需审慎决策。","\u002Fuploads\u002F2026\u002F06\u002Fopenwrt-router-review-dual-sim-05-1781620596218-5ldld4.webp","2026-06-16T10:41:00.000Z","fallback",[],[]]