[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"portal-settings:stajic:zh":3,"public-menus:all":38,"post:ai-agent-reliability-why-the-final-answer-is-not-enough:zh":205,"related:post:ai-agent-reliability-why-the-final-answer-is-not-enough:zh:1":1130},{"statusCode":4,"data":5,"message":37},200,{"tenantId":6,"lang":7,"defaultLang":8,"siteUrl":9,"contactEmail":10,"brandName":11,"logoUrl":12,"siteName":11,"siteDescription":13,"ogImage":10,"robotsIndex":14,"socialLinks":10,"reservedSlugs":10,"seoPolicy":15},"stajic","zh","de","https:\u002F\u002Fstajic.de",null,"Stajic Platform","\u002FLogo_Planet.svg","Stajic Portal",true,{"branding":16,"relatedContent":17,"crossDomainLinks":18},{"logoUrl":12},{"enabled":14},[19,22,25,28,31,34],{"url":20,"label":21,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Ffigure.rocks","figure.rocks",{"url":23,"label":24,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Floving.rocks","loving.rocks",{"url":26,"label":27,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.com","bazify.com",{"url":29,"label":30,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.de","bazify.de",{"url":32,"label":33,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.at","bazify.at",{"url":35,"label":36,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.ba","bazify.ba","Portal settings resolved",[39,45],{"id":40,"name":41,"location":42,"isActive":14,"isDefault":43,"items":44},1,"main-navigation","header",false,[],{"id":46,"name":47,"location":48,"isActive":14,"isDefault":14,"items":49},4,"main-menu","sidebar",[50,66,79,93,103,118,133],{"id":51,"title":52,"url":60,"target":61,"icon":62,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":64,"portfolioId":10,"children":65},"item-18",{"de":53,"en":54,"es":55,"fr":56,"it":54,"ru":57,"sr":58,"zh":59},"Startseite","Home","Inicio","Accueil","Главная","Почетна","首页","\u002Ffull-stack-web-developer-munich-performance-seo-and-maintainable-builds","_self","i-lucide-home","page",111,[],{"id":67,"title":68,"url":75,"target":61,"icon":76,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":77,"portfolioId":10,"children":78},"item-22",{"de":69,"en":69,"es":70,"fr":69,"it":71,"ru":72,"sr":73,"zh":74},"Vision","Visión","Visione","Видение","Визија","想象","\u002Fueber-uns-webdesign-muenchen-webaplikation","i-lucide-eye",113,[],{"id":80,"title":81,"url":89,"target":61,"icon":90,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":91,"portfolioId":10,"children":92},"item-19",{"de":82,"en":83,"es":84,"fr":83,"it":85,"ru":86,"sr":87,"zh":88},"Leistungen","Services","Servicios","Servizi","Услуги","Услуге","服务","\u002Fservices-dienstleistungen-muenchen","i-lucide-wrench",116,[],{"id":94,"title":95,"url":99,"target":61,"icon":100,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":101,"portfolioId":10,"children":102},"item-23",{"de":96,"en":96,"es":96,"fr":96,"it":96,"ru":97,"sr":97,"zh":98},"Blog","Блог","博客","\u002Fblog","i-lucide-book-open",112,[],{"id":104,"title":105,"url":114,"target":61,"icon":115,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":116,"portfolioId":10,"children":117},"item-32",{"de":106,"en":107,"es":108,"fr":109,"it":110,"ru":111,"sr":112,"zh":113},"Neue Technologien","New Technologies","Nuevas tecnologías","Nouvelles technologies","Nuove tecnologie","Новые технологии","Нове технологије","新技术！","\u002Fneue-webtechnologien","i-lucide-sparkles",122,[],{"id":119,"title":120,"url":129,"target":61,"icon":130,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":131,"portfolioId":10,"children":132},"item-20",{"de":121,"en":122,"es":123,"fr":124,"it":125,"ru":126,"sr":127,"zh":128},"Kontakt","Contact us!","Contacto","Contact","Contatto","Контакт","Контактирајте нас","联系我们！","\u002Fcontact","i-lucide-mail",115,[],{"id":134,"title":135,"url":144,"target":61,"icon":145,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":147},"item-21",{"de":136,"en":137,"es":138,"fr":139,"it":140,"ru":141,"sr":142,"zh":143},"Unsere Arbeit","Our Work","Nuestro trabajo","Nos réalisations","I nostri lavori","Наши работы","Наши радови","文件夹","\u002Fportfolio","i-lucide-briefcase",114,[148,161,175,181,193],{"id":149,"title":150,"url":144,"target":61,"icon":159,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":160},"item-24",{"de":151,"en":152,"es":153,"fr":154,"it":155,"ru":156,"sr":157,"zh":158},"Alle Projekte","All Projects","Todos los proyectos","Tous les projets","Tutti i progetti","Все проекты","Сви пројекти","所有项目","i-lucide-grid-3x3",[],{"id":162,"title":163,"url":171,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":174},"item-29",{"de":164,"en":165,"es":166,"fr":167,"it":168,"ru":169,"sr":170,"zh":143},"Local Roots, Global Reach","Local Roots - Global Reach","Empresa local ","Entreprise locale","Azienda locale","Местная компания","Локално предузеће глобално тржиште","\u002Fportfolio\u002Flocal-roots-global-reach-communication-media-systems-for-modern-business","i-lucide-folder","custom",[],{"id":176,"title":177,"url":179,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":180},"item-28",{"de":178,"en":178,"es":178,"fr":178,"it":178,"ru":178,"sr":178,"zh":178},"Solr Suggester","\u002Fportfolio\u002Fsolr-fuzzy-suggester-und-solr-infix-suggester-abfrage-ueber-ajax-und-filterung",[],{"id":182,"title":183,"url":191,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":192},"item-27",{"de":184,"en":185,"es":186,"fr":187,"it":188,"ru":189,"sr":190,"zh":185},"Firmenwebseite SEO","Company Website SEO","Sitio web corporativo SEO","Site web d’entreprise SEO","Sito web aziendale SEO","Корпоративный сайт SEO","Пословна веб-страница SEO","\u002Fportfolio\u002Fseo-sem-branding-mobile-webseite-muenchen",[],{"id":194,"title":195,"url":203,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":204},"item-31",{"de":196,"en":197,"es":198,"fr":199,"it":200,"ru":201,"sr":202,"zh":197},"Digitalisierungsportal","Digitalization Portal","Portal de digitalización","Portail de numérisation","Portale di digitalizzazione","Портал цифровизации","Портал за дигитализацију","\u002Fportfolio\u002Fdigitalisierungsportal-archiv-museum-bibliothek-ead-lido-mets-mods",[],{"statusCode":4,"data":206,"message":1129},{"id":207,"title":208,"slug":209,"content":210,"contentJson":211,"excerpt":561,"featuredImage":562,"featuredImageAlt":563,"featuredImageCaption":10,"featuredImageTitle":10,"featuredImageCopyright":10,"featuredImageAuthor":10,"featuredImageSourceUrl":10,"featuredImageLicense":10,"featuredImageIsAiGenerated":43,"status":564,"publishedAt":565,"createdAt":566,"updatedAt":567,"seoLocalePaths":568,"categories":577,"author":586,"translations":591},"460","AI Agent可靠性：为什么最终答案并不足够","ai-agent-reliability-why-the-final-answer-is-not-enough","\u003Cp>\u003Cb>正确的输出并不能证明推理正确、执行安全或系统可信。\u003C\u002Fb>\u003C\u002Fp>\n\u003Cp>多年来，人工智能评估一直被一个看似简单的问题所主导：\u003Cb>答案正确吗？\u003C\u002Fb>对于聊天机器人来说，这有时可能就足够了。但对于一个能够搜索系统、读取数据、调用工具、修改状态、执行工作流、写入文件、与API交互或做出决策的智能体来说，这还不够。\u003C\u002Fp>\n\u003Cp>一个智能体可能在产生正确的最终答案的同时，在过程中做了几件错误的事情。它可能使用了错误的来源，误解了指令并随后弥补了错误，访问了不必要的信息，执行了未经授权的中间操作，从本应触发升级的错误中悄然恢复，或者留下了没人注意到的副作用。\u003C\u002Fp>\n\u003Cp>这造成了智能体AI的核心问题之一：\u003Cb>正确的结果并不能证明正确的轨迹。\u003C\u002Fb>\u003C\u002Fp>\n\u003Ch2>结果错觉\u003C\u002Fh2>\n\u003Cp>传统软件给了我们一个直观的正确性模型。输入进入一个确定性或基本确定性的系统，执行逻辑，产生输出，测试验证预期行为。基于LLM的系统削弱了这一假设。智能体系统则更进一步。\u003C\u002Fp>\n\u003Cul>\u003Cli>模型解释\u003C\u002Fli>\u003Cli>检索到的上下文\u003C\u002Fli>\u003Cli>工具选择\u003C\u002Fli>\u003Cli>中间观察\u003C\u002Fli>\u003Cli>外部状态\u003C\u002Fli>\u003Cli>先前的动作\u003C\u002Fli>\u003Cli>模型生成的计划\u003C\u002Fli>\u003Cli>权限边界\u003C\u002Fli>\u003Cli>重试和回退行为\u003C\u002Fli>\u003Cli>人工交互\u003C\u002Fli>\u003C\u002Ful>\n\u003Cp>从几乎相同的输入开始的两个执行可能通过非常不同的路径达到相同的结果。如果评估只观察最终输出，那么系统的大部分内容仍然是不可见的。\u003C\u002Fp>\n\u003Cp>想象一个AI智能体收到指令：\u003Ci>更新客户的账单地址。\u003C\u002Fi>地址最终被正确更新。传统的评估可能会将任务归类为成功。\u003C\u002Fp>\n\u003Col>\u003Cli>智能体搜索了几个不相关的客户记录。\u003C\u002Fli>\u003Cli>它检索了超出要求的更多个人信息。\u003C\u002Fli>\u003Cli>它最初修改了错误的账户。\u003C\u002Fli>\u003Cli>它注意到了错误。\u003C\u002Fli>\u003Cli>它撤销了更改。\u003C\u002Fli>\u003Cli>它更新了正确的账户。\u003C\u002Fli>\u003Cli>它报告成功。\u003C\u002Fli>\u003C\u002Fol>\n\u003Cp>\u003Cb>最终状态：正确。系统行为：不可接受。\u003C\u002Fb>仅基于结果的基准测试会给这次执行一个通过。生产保障系统不应该这样做。\u003C\u002Fp>\n\u003Ch2>轨迹是产品的一部分\u003C\u002Fh2>\n\u003Cp>这就是为什么AI智能体的\u003Cb>轨迹\u003C\u002Fb>必须成为一流的工程对象。轨迹是原始请求和最终结果之间的相关状态和动作的序列。\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>意图 → 上下文 → 决策 → 工具 → 动作 → 观察 → 决策 → 状态变化 → 结果\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>Zachary J. Stevens在\u003Ci>《轨迹即系统》\u003C\u002Fi>中发展了这一思想，认为智能体评估必须超越最终答案，检查在变化环境中行动的完整路径。\u003C\u002Fp>\n\u003Cblockquote class=\"border-l-4 border-gray-300 pl-4 italic\">正确的结果并不能为不可接受的轨迹开脱。\u003Ccite class=\"block mt-2 text-sm\">— Zachary J. Stevens，《轨迹即系统》\u003C\u002Fcite>\u003C\u002Fblockquote>\n\u003Ca href=\"https:\u002F\u002Fzacharyjstevens.com\u002Fdispatches\u002Fvanguard-signal\u002F009-the-trajectory-is-the-system\u002F\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">轨迹即系统\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">Zachary J. Stevens — DFEI.009 关于通过完整轨迹而非仅最终结果来评估智能体系统。\u003C\u002Fp>\u003C\u002Fa>\n\u003Cp>这种区别非常重要。因此，可靠性不仅仅是\u003Cb>正确的输出\u003C\u002Fb>。它更接近于\u003Cb>可接受的结果 + 可接受的轨迹 + 可恢复性 + 证据\u003C\u002Fb>。\u003C\u002Fp>\n\u003Ch2>正确答案可能掩盖系统缺陷\u003C\u002Fh2>\n\u003Ctable class=\"w-full border-collapse\">\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">智能体\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">最终结果\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">执行过程\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">A\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">正确\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">路径安全\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">B\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">正确\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">路径不安全\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">C\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">错误\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">安全失败\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">D\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">错误\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">不安全失败\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftable>\n\u003Cp>大多数基于基准的评估强烈倾向于A和B，而惩罚C和D。然而，在操作层面，\u003Cb>B可能比C更危险\u003C\u002Fb>。智能体C可能识别不确定性，停止执行并请求人工审查。智能体B可能自信地产生正确结果，同时违反无人监控的假设。\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>成功输出 → 信任增加 → 权限扩大 → 自动化程度提高 → 影响范围扩大\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Ch2>我们需要证据，而非信心\u003C\u002Fh2>\n\u003Cp>在采用人工智能时，最大的错误之一是将模型信心、用户满意度或历史成功率视为系统可靠性的证据。它们并不等同。\u003C\u002Fp>\n\u003Cul>\u003Cli>智能体接收了什么？\u003C\u002Fli>\u003Cli>它检索了哪些上下文？\u003C\u002Fli>\u003Cli>它调用了哪些工具？\u003C\u002Fli>\u003Cli>为什么允许该操作？\u003C\u002Fli>\u003Cli>操作前存在什么状态？\u003C\u002Fli>\u003Cli>发生了什么变化？\u003C\u002Fli>\u003Cli>发生了哪些中间失败？\u003C\u002Fli>\u003Cli>是否有重试？\u003C\u002Fli>\u003Cli>是否需要人工批准？\u003C\u002Fli>\u003Cli>执行能否被停止？\u003C\u002Fli>\u003Cli>操作能否被撤销？\u003C\u002Fli>\u003Cli>涉及哪些模型、提示词和工具版本？\u003C\u002Fli>\u003C\u002Ful>\n\u003Cp>没有这些答案，就没有严肃的操作保证。只有输出。因此，可观测性和证据必须设计到智能体架构中，而不是在部署后添加。\u003C\u002Fp>\n\u003Ch2>日志记录不等于控制\u003C\u002Fh2>\n\u003Cp>组织通常回应：\u003Ci>一切都已记录。\u003C\u002Fi>很好。但仅记录并不能控制任何事情。日志告诉你发生了什么。控制决定某事\u003Cb>是否可能发生\u003C\u002Fb>。\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>智能体请求删除 \u002Fcustomer\u002F123 ↓\n操作已记录 ↓\n删除已执行\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>这提供了可观测性。将其与以下内容进行比较：\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>智能体请求删除 \u002Fcustomer\u002F123 ↓\n策略评估 ↓\n当前身份验证 ↓\n当前操作参数检查 ↓\n风险阈值评估 ↓\n如果需要，人工批准 ↓\n操作已执行 ↓\n结果验证 ↓\n证据存储\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>现在我们接近一个控制系统。差异是架构性的，而非表面性的。\u003C\u002Fp>\n\u003Ch2>权限是必要的——但它不是保证\u003C\u002Fh2>\n\u003Cp>假设一个智能体有发送电子邮件的权限。访问控制回答：\u003Cb>这个智能体可以发送电子邮件吗？\u003C\u002Fb>它不回答：\u003Cb>这封特定的电子邮件是否应该发送给这个特定的人，并附带这个特定的附件，现在？\u003C\u002Fb>\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>能力控制\n智能体在技术上被允许做什么？ + 操作保证\n在当前状态下，这个特定操作是否合适？\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>RBAC、OAuth范围、API权限和智能体身份定义了可能操作的空间。它们不能证明该空间内的操作是合适的。强大的智能体架构需要两个层次。\u003C\u002Fp>\n\u003Ch2>第一步走错至关重要\u003C\u002Fh2>\n\u003Cp>当代理失败时，最终的错误操作往往不是失败开始的地方。真正的失败可能发生得更早。\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>错误的检索 ↓\n错误的假设 ↓\n看似合理的推理 ↓\n有效的工具调用 ↓\n错误的操作\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>如果我们只调查最终操作，我们修复的是症状。如果我们检查轨迹，我们可以识别\u003Cb>第一步错误\u003C\u002Fb>。这将无法归因的失败转化为具体的工程问题。\u003C\u002Fp>\n\u003Ch2>代理测试必须超越提示测试\u003C\u002Fh2>\n\u003Cp>提示很重要，但生产环境中的代理行为源于整个系统。\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>模型\n+\n系统提示\n+\n上下文\n+\n记忆\n+\n检索\n+\n工具\n+\n权限\n+\n工作流\n+\n外部状态\n+\n控制逻辑\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>改变其中任何一个都可能改变轨迹。因此，仅对提示进行版本控制是不够的。\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>模型版本\n提示版本\n工具版本\n策略版本\n检索版本\n工作流版本\n环境状态\n执行ID\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Ch2>代理的验收标准必须包含行为\u003C\u002Fh2>\n\u003Cp>传统的验收标准通常如下：\u003Ci>给定X，系统产生Y。\u003C\u002Fi>对于代理系统，这是不完整的。验收标准还应定义轨迹上的约束。\u003C\u002Fp>\n\u003Ch3>结果\u003C\u002Fh3>\n\u003Cp>客户的地址已正确更新。\u003C\u002Fp>\n\u003Ch3>授权\u003C\u002Fh3>\n\u003Cp>代理仅修改明确选择的客户。\u003C\u002Fp>\n\u003Ch3>数据访问\u003C\u002Fh3>\n\u003Cp>不访问无关的客户记录。\u003C\u002Fp>\n\u003Ch3>工具\u003C\u002Fh3>\n\u003Cp>仅使用经批准的CRM操作。\u003C\u002Fp>\n\u003Ch3>验证\u003C\u002Fh3>\n\u003Cp>新地址被读回并与请求的值进行比较。\u003C\u002Fp>\n\u003Ch3>失败\u003C\u002Fh3>\n\u003Cp>模糊的身份解析会停止执行。\u003C\u002Fp>\n\u003Ch3>人类权威\u003C\u002Fh3>\n\u003Cp>当风险阈值需要批准时，人类可以在执行前拒绝修改。\u003C\u002Fp>\n\u003Ch3>证据\u003C\u002Fh3>\n\u003Cp>执行留下足够的痕迹，以重建决策和状态转换。\u003C\u002Fp>\n\u003Ch3>恢复\u003C\u002Fh3>\n\u003Cp>先前的值仍然可以恢复。\u003C\u002Fp>\n\u003Ch2>人在环中还不够\u003C\u002Fh2>\n\u003Cp>添加一个人工审批框并不能自动解决问题。只有当人具备可见性、权威、时间、上下文和恢复能力时，才能控制代理。\u003C\u002Fp>\n\u003Cul>\u003Cli>\u003Cb>可见性：\u003C\u002Fb>足够的信息以了解正在发生的事情。\u003C\u002Fli>\u003Cli>\u003Cb>权威：\u003C\u002Fb>实际停止或修改操作的能力。\u003C\u002Fli>\u003Cli>\u003Cb>时间：\u003C\u002Fb>在后果发生之前进行干预。\u003C\u002Fli>\u003Cli>\u003Cb>上下文：\u003C\u002Fb>足够的证据以做出决策。\u003C\u002Fli>\u003Cli>\u003Cb>恢复能力：\u003C\u002Fb>撤销或修复操作的能力。\u003C\u002Fli>\u003C\u002Ful>\n\u003Cp>用户点击\u003Cb>批准\u003C\u002Fb>，但他们无法有意义地检查的内容，并不是强有力的治理。这是批准剧场。\u003C\u002Fp>\n\u003Ch2>回滚必须成为AI的原生能力\u003C\u002Fh2>\n\u003Cp>传统软件部署教会了我们一些有价值的东西：\u003Cb>永远不要部署无法回滚的内容。\u003C\u002Fb>我们应该将同样的原则应用于代理操作。\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>可逆\n可以自动撤销。可补偿\n不能直接撤销，但可以执行补偿操作。不可逆\n无法可靠地恢复先前的状态。\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>不可逆性越高，控制要求就应该越强。\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>读取公开文档 → 低后果\n创建草稿 → 可逆\n修改CRM记录 → 可逆但有后果\n发送外部邮件 → 实际上不可逆\n转账 → 高后果\n删除生产数据 → 可能造成灾难性后果\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Ch2>智能体需要控制平面\u003C\u002Fh2>\n\u003Cpre class=\"code-block\">\u003Ccode>用户\u002F系统意图 │ ▼ AI智能体 │ 提议的行动 │ ▼ ┌───────────────────┐ │ 控制平面 │ ├───────────────────┤ │ 身份 │ │ 授权 │ │ 策略 │ │ 风险 │ │ 状态 │ │ 证据 │ │ 人类权威 │ │ 回滚 │ └───────────────────┘ │ 批准？ \u002F \\ 否 是 │ │ 停止 ▼ 工具 │ ▼ 状态变更 │ ▼ 验证\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>\u003Cb>LLM应该提出建议。控制平面应该进行治理。\u003C\u002Fb>这种分离至关重要。模型不应成为决定其自身提议的高影响行动是否安全的最终权威。\u003C\u002Fp>\n\u003Ch2>从基准测试到运营信任\u003C\u002Fh2>\n\u003Cp>基准测试仍然有用。它们告诉我们能力、比较模型、检测回归，并帮助估计预期性能。但能力评估和运营信任回答的是不同的问题。\u003C\u002Fp>\n\u003Cp>基准测试问：\u003Cb>系统能做到吗？\u003C\u002Fb>运营保障问：\u003Cb>我们能否允许系统在此处、在这些条件下、以这些权限和后果执行此操作？\u003C\u002Fb>\u003C\u002Fp>\n\u003Ch2>可靠性应作为系统属性来衡量\u003C\u002Fh2>\n\u003Col>\u003Cli>\u003Cb>结果正确性：\u003C\u002Fb>系统是否产生了预期结果？\u003C\u002Fli>\u003Cli>\u003Cb>轨迹正确性：\u003C\u002Fb>它是否遵循了可接受的路径？\u003C\u002Fli>\u003Cli>\u003Cb>控制完整性：\u003C\u002Fb>授权、策略和干预边界是否得到尊重？\u003C\u002Fli>\u003Cli>\u003Cb>可恢复性：\u003C\u002Fb>故障能否被遏制、逆转或修复？\u003C\u002Fli>\u003Cli>\u003Cb>证据完整性：\u003C\u002Fb>执行过程能否被重建和审计？\u003C\u002Fli>\u003C\u002Fol>\n\u003Cpre class=\"code-block\">\u003Ccode>运营可靠性\n=\n结果 × 轨迹 × 控制 × 可恢复性 × 证据\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>乘法是有意为之。如果某个关键维度趋近于零，其他方面的高分不应掩盖它。一个完全正确的输出，如果授权完整性为零，并不是一个80%可靠的系统。那是一次不可接受的执行，碰巧产生了正确的答案。\u003C\u002Fp>\n\u003Ch2>成功有时是最危险的失败\u003C\u002Fh2>\n\u003Cp>失败会引起注意。成功往往不会。这使得成功但不受控的智能体轨迹特别危险。明显的失败会引发事件。隐藏的轨迹缺陷会创造\u003Cb>信心\u003C\u002Fb>。而信心会扩大自主权。\u003C\u002Fp>\n\u003Cp>因此，组织不仅应该调查\u003Ci>为什么智能体失败了？\u003C\u002Fi>他们还应该定期问：\u003Cb>为什么智能体成功了？\u003C\u002Fb>是因为架构可靠地约束和验证了执行，还是因为这次没有出错？\u003C\u002Fp>\n\u003Ch2>结论\u003C\u002Fh2>\n\u003Cp>行业正迅速从能够\u003Cb>回答\u003C\u002Fb>的AI转向能够\u003Cb>行动\u003C\u002Fb>的AI。这种转变改变了可靠性的含义。对于回答系统，评估答案往往就足够了。对于行动系统，我们必须评估路径。\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>提示 ↓\n响应变为意图 ↓\n轨迹 ↓\n行动 ↓\n状态变更 ↓\n证据 ↓\n结果\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>最终答案仍然重要，但它只是一个更大系统中可见的末端。一旦人工智能被允许影响现实世界，\u003Cb>通往答案的路径就成为答案的一部分。\u003C\u002Fb>\u003C\u002Fp>",{"time":212,"blocks":213,"version":560},1788955874503,[214,218,221,224,227,231,234,249,252,255,266,269,272,275,279,282,288,296,299,302,324,327,330,333,336,351,354,357,360,363,366,369,372,375,378,381,384,387,390,393,396,399,402,405,408,411,414,417,421,424,427,430,433,436,439,442,445,448,451,454,457,460,463,466,469,472,475,478,486,489,492,495,498,501,504,507,510,513,516,519,522,525,533,536,539,542,545,548,551,554,557],{"data":215,"type":217},{"text":216},"\u003Cb>正确的输出并不能证明推理正确、执行安全或系统可信。\u003C\u002Fb>","paragraph",{"data":219,"type":217},{"text":220},"多年来，人工智能评估一直被一个看似简单的问题所主导：\u003Cb>答案正确吗？\u003C\u002Fb>对于聊天机器人来说，这有时可能就足够了。但对于一个能够搜索系统、读取数据、调用工具、修改状态、执行工作流、写入文件、与API交互或做出决策的智能体来说，这还不够。",{"data":222,"type":217},{"text":223},"一个智能体可能在产生正确的最终答案的同时，在过程中做了几件错误的事情。它可能使用了错误的来源，误解了指令并随后弥补了错误，访问了不必要的信息，执行了未经授权的中间操作，从本应触发升级的错误中悄然恢复，或者留下了没人注意到的副作用。",{"data":225,"type":217},{"text":226},"这造成了智能体AI的核心问题之一：\u003Cb>正确的结果并不能证明正确的轨迹。\u003C\u002Fb>",{"data":228,"type":42},{"text":229,"level":230},"结果错觉",2,{"data":232,"type":217},{"text":233},"传统软件给了我们一个直观的正确性模型。输入进入一个确定性或基本确定性的系统，执行逻辑，产生输出，测试验证预期行为。基于LLM的系统削弱了这一假设。智能体系统则更进一步。",{"data":235,"type":248},{"items":236,"style":247},[237,238,239,240,241,242,243,244,245,246],"模型解释","检索到的上下文","工具选择","中间观察","外部状态","先前的动作","模型生成的计划","权限边界","重试和回退行为","人工交互","unordered","list",{"data":250,"type":217},{"text":251},"从几乎相同的输入开始的两个执行可能通过非常不同的路径达到相同的结果。如果评估只观察最终输出，那么系统的大部分内容仍然是不可见的。",{"data":253,"type":217},{"text":254},"想象一个AI智能体收到指令：\u003Ci>更新客户的账单地址。\u003C\u002Fi>地址最终被正确更新。传统的评估可能会将任务归类为成功。",{"data":256,"type":248},{"items":257,"style":265},[258,259,260,261,262,263,264],"智能体搜索了几个不相关的客户记录。","它检索了超出要求的更多个人信息。","它最初修改了错误的账户。","它注意到了错误。","它撤销了更改。","它更新了正确的账户。","它报告成功。","ordered",{"data":267,"type":217},{"text":268},"\u003Cb>最终状态：正确。系统行为：不可接受。\u003C\u002Fb>仅基于结果的基准测试会给这次执行一个通过。生产保障系统不应该这样做。",{"data":270,"type":42},{"text":271,"level":230},"轨迹是产品的一部分",{"data":273,"type":217},{"text":274},"这就是为什么AI智能体的\u003Cb>轨迹\u003C\u002Fb>必须成为一流的工程对象。轨迹是原始请求和最终结果之间的相关状态和动作的序列。",{"data":276,"type":278},{"code":277},"意图 → 上下文 → 决策 → 工具 → 动作 → 观察 → 决策 → 状态变化 → 结果","code",{"data":280,"type":217},{"text":281},"Zachary J. Stevens在\u003Ci>《轨迹即系统》\u003C\u002Fi>中发展了这一思想，认为智能体评估必须超越最终答案，检查在变化环境中行动的完整路径。",{"data":283,"type":287},{"text":284,"caption":285,"alignment":286},"正确的结果并不能为不可接受的轨迹开脱。","Zachary J. Stevens，《轨迹即系统》","left","quote",{"data":289,"type":295},{"link":290,"meta":291},"https:\u002F\u002Fzacharyjstevens.com\u002Fdispatches\u002Fvanguard-signal\u002F009-the-trajectory-is-the-system\u002F",{"image":292,"title":293,"description":294},{},"轨迹即系统","Zachary J. Stevens — DFEI.009 关于通过完整轨迹而非仅最终结果来评估智能体系统。","linkTool",{"data":297,"type":217},{"text":298},"这种区别非常重要。因此，可靠性不仅仅是\u003Cb>正确的输出\u003C\u002Fb>。它更接近于\u003Cb>可接受的结果 + 可接受的轨迹 + 可恢复性 + 证据\u003C\u002Fb>。",{"data":300,"type":42},{"text":301,"level":230},"正确答案可能掩盖系统缺陷",{"data":303,"type":323},{"content":304,"withHeadings":14},[305,309,313,316,320],[306,307,308],"智能体","最终结果","执行过程",[310,311,312],"A","正确","路径安全",[314,311,315],"B","路径不安全",[317,318,319],"C","错误","安全失败",[321,318,322],"D","不安全失败","table",{"data":325,"type":217},{"text":326},"大多数基于基准的评估强烈倾向于A和B，而惩罚C和D。然而，在操作层面，\u003Cb>B可能比C更危险\u003C\u002Fb>。智能体C可能识别不确定性，停止执行并请求人工审查。智能体B可能自信地产生正确结果，同时违反无人监控的假设。",{"data":328,"type":278},{"code":329},"成功输出 → 信任增加 → 权限扩大 → 自动化程度提高 → 影响范围扩大",{"data":331,"type":42},{"text":332,"level":230},"我们需要证据，而非信心",{"data":334,"type":217},{"text":335},"在采用人工智能时，最大的错误之一是将模型信心、用户满意度或历史成功率视为系统可靠性的证据。它们并不等同。",{"data":337,"type":248},{"items":338,"style":247},[339,340,341,342,343,344,345,346,347,348,349,350],"智能体接收了什么？","它检索了哪些上下文？","它调用了哪些工具？","为什么允许该操作？","操作前存在什么状态？","发生了什么变化？","发生了哪些中间失败？","是否有重试？","是否需要人工批准？","执行能否被停止？","操作能否被撤销？","涉及哪些模型、提示词和工具版本？",{"data":352,"type":217},{"text":353},"没有这些答案，就没有严肃的操作保证。只有输出。因此，可观测性和证据必须设计到智能体架构中，而不是在部署后添加。",{"data":355,"type":42},{"text":356,"level":230},"日志记录不等于控制",{"data":358,"type":217},{"text":359},"组织通常回应：\u003Ci>一切都已记录。\u003C\u002Fi>很好。但仅记录并不能控制任何事情。日志告诉你发生了什么。控制决定某事\u003Cb>是否可能发生\u003C\u002Fb>。",{"data":361,"type":278},{"code":362},"智能体请求删除 \u002Fcustomer\u002F123 ↓\n操作已记录 ↓\n删除已执行",{"data":364,"type":217},{"text":365},"这提供了可观测性。将其与以下内容进行比较：",{"data":367,"type":278},{"code":368},"智能体请求删除 \u002Fcustomer\u002F123 ↓\n策略评估 ↓\n当前身份验证 ↓\n当前操作参数检查 ↓\n风险阈值评估 ↓\n如果需要，人工批准 ↓\n操作已执行 ↓\n结果验证 ↓\n证据存储",{"data":370,"type":217},{"text":371},"现在我们接近一个控制系统。差异是架构性的，而非表面性的。",{"data":373,"type":42},{"text":374,"level":230},"权限是必要的——但它不是保证",{"data":376,"type":217},{"text":377},"假设一个智能体有发送电子邮件的权限。访问控制回答：\u003Cb>这个智能体可以发送电子邮件吗？\u003C\u002Fb>它不回答：\u003Cb>这封特定的电子邮件是否应该发送给这个特定的人，并附带这个特定的附件，现在？\u003C\u002Fb>",{"data":379,"type":278},{"code":380},"能力控制\n智能体在技术上被允许做什么？ + 操作保证\n在当前状态下，这个特定操作是否合适？",{"data":382,"type":217},{"text":383},"RBAC、OAuth范围、API权限和智能体身份定义了可能操作的空间。它们不能证明该空间内的操作是合适的。强大的智能体架构需要两个层次。",{"data":385,"type":42},{"text":386,"level":230},"第一步走错至关重要",{"data":388,"type":217},{"text":389},"当代理失败时，最终的错误操作往往不是失败开始的地方。真正的失败可能发生得更早。",{"data":391,"type":278},{"code":392},"错误的检索 ↓\n错误的假设 ↓\n看似合理的推理 ↓\n有效的工具调用 ↓\n错误的操作",{"data":394,"type":217},{"text":395},"如果我们只调查最终操作，我们修复的是症状。如果我们检查轨迹，我们可以识别\u003Cb>第一步错误\u003C\u002Fb>。这将无法归因的失败转化为具体的工程问题。",{"data":397,"type":42},{"text":398,"level":230},"代理测试必须超越提示测试",{"data":400,"type":217},{"text":401},"提示很重要，但生产环境中的代理行为源于整个系统。",{"data":403,"type":278},{"code":404},"模型\n+\n系统提示\n+\n上下文\n+\n记忆\n+\n检索\n+\n工具\n+\n权限\n+\n工作流\n+\n外部状态\n+\n控制逻辑",{"data":406,"type":217},{"text":407},"改变其中任何一个都可能改变轨迹。因此，仅对提示进行版本控制是不够的。",{"data":409,"type":278},{"code":410},"模型版本\n提示版本\n工具版本\n策略版本\n检索版本\n工作流版本\n环境状态\n执行ID",{"data":412,"type":42},{"text":413,"level":230},"代理的验收标准必须包含行为",{"data":415,"type":217},{"text":416},"传统的验收标准通常如下：\u003Ci>给定X，系统产生Y。\u003C\u002Fi>对于代理系统，这是不完整的。验收标准还应定义轨迹上的约束。",{"data":418,"type":42},{"text":419,"level":420},"结果",3,{"data":422,"type":217},{"text":423},"客户的地址已正确更新。",{"data":425,"type":42},{"text":426,"level":420},"授权",{"data":428,"type":217},{"text":429},"代理仅修改明确选择的客户。",{"data":431,"type":42},{"text":432,"level":420},"数据访问",{"data":434,"type":217},{"text":435},"不访问无关的客户记录。",{"data":437,"type":42},{"text":438,"level":420},"工具",{"data":440,"type":217},{"text":441},"仅使用经批准的CRM操作。",{"data":443,"type":42},{"text":444,"level":420},"验证",{"data":446,"type":217},{"text":447},"新地址被读回并与请求的值进行比较。",{"data":449,"type":42},{"text":450,"level":420},"失败",{"data":452,"type":217},{"text":453},"模糊的身份解析会停止执行。",{"data":455,"type":42},{"text":456,"level":420},"人类权威",{"data":458,"type":217},{"text":459},"当风险阈值需要批准时，人类可以在执行前拒绝修改。",{"data":461,"type":42},{"text":462,"level":420},"证据",{"data":464,"type":217},{"text":465},"执行留下足够的痕迹，以重建决策和状态转换。",{"data":467,"type":42},{"text":468,"level":420},"恢复",{"data":470,"type":217},{"text":471},"先前的值仍然可以恢复。",{"data":473,"type":42},{"text":474,"level":230},"人在环中还不够",{"data":476,"type":217},{"text":477},"添加一个人工审批框并不能自动解决问题。只有当人具备可见性、权威、时间、上下文和恢复能力时，才能控制代理。",{"data":479,"type":248},{"items":480,"style":247},[481,482,483,484,485],"\u003Cb>可见性：\u003C\u002Fb>足够的信息以了解正在发生的事情。","\u003Cb>权威：\u003C\u002Fb>实际停止或修改操作的能力。","\u003Cb>时间：\u003C\u002Fb>在后果发生之前进行干预。","\u003Cb>上下文：\u003C\u002Fb>足够的证据以做出决策。","\u003Cb>恢复能力：\u003C\u002Fb>撤销或修复操作的能力。",{"data":487,"type":217},{"text":488},"用户点击\u003Cb>批准\u003C\u002Fb>，但他们无法有意义地检查的内容，并不是强有力的治理。这是批准剧场。",{"data":490,"type":42},{"text":491,"level":230},"回滚必须成为AI的原生能力",{"data":493,"type":217},{"text":494},"传统软件部署教会了我们一些有价值的东西：\u003Cb>永远不要部署无法回滚的内容。\u003C\u002Fb>我们应该将同样的原则应用于代理操作。",{"data":496,"type":278},{"code":497},"可逆\n可以自动撤销。可补偿\n不能直接撤销，但可以执行补偿操作。不可逆\n无法可靠地恢复先前的状态。",{"data":499,"type":217},{"text":500},"不可逆性越高，控制要求就应该越强。",{"data":502,"type":278},{"code":503},"读取公开文档 → 低后果\n创建草稿 → 可逆\n修改CRM记录 → 可逆但有后果\n发送外部邮件 → 实际上不可逆\n转账 → 高后果\n删除生产数据 → 可能造成灾难性后果",{"data":505,"type":42},{"text":506,"level":230},"智能体需要控制平面",{"data":508,"type":278},{"code":509},"用户\u002F系统意图 │ ▼ AI智能体 │ 提议的行动 │ ▼ ┌───────────────────┐ │ 控制平面 │ ├───────────────────┤ │ 身份 │ │ 授权 │ │ 策略 │ │ 风险 │ │ 状态 │ │ 证据 │ │ 人类权威 │ │ 回滚 │ └───────────────────┘ │ 批准？ \u002F \\ 否 是 │ │ 停止 ▼ 工具 │ ▼ 状态变更 │ ▼ 验证",{"data":511,"type":217},{"text":512},"\u003Cb>LLM应该提出建议。控制平面应该进行治理。\u003C\u002Fb>这种分离至关重要。模型不应成为决定其自身提议的高影响行动是否安全的最终权威。",{"data":514,"type":42},{"text":515,"level":230},"从基准测试到运营信任",{"data":517,"type":217},{"text":518},"基准测试仍然有用。它们告诉我们能力、比较模型、检测回归，并帮助估计预期性能。但能力评估和运营信任回答的是不同的问题。",{"data":520,"type":217},{"text":521},"基准测试问：\u003Cb>系统能做到吗？\u003C\u002Fb>运营保障问：\u003Cb>我们能否允许系统在此处、在这些条件下、以这些权限和后果执行此操作？\u003C\u002Fb>",{"data":523,"type":42},{"text":524,"level":230},"可靠性应作为系统属性来衡量",{"data":526,"type":248},{"items":527,"style":265},[528,529,530,531,532],"\u003Cb>结果正确性：\u003C\u002Fb>系统是否产生了预期结果？","\u003Cb>轨迹正确性：\u003C\u002Fb>它是否遵循了可接受的路径？","\u003Cb>控制完整性：\u003C\u002Fb>授权、策略和干预边界是否得到尊重？","\u003Cb>可恢复性：\u003C\u002Fb>故障能否被遏制、逆转或修复？","\u003Cb>证据完整性：\u003C\u002Fb>执行过程能否被重建和审计？",{"data":534,"type":278},{"code":535},"运营可靠性\n=\n结果 × 轨迹 × 控制 × 可恢复性 × 证据",{"data":537,"type":217},{"text":538},"乘法是有意为之。如果某个关键维度趋近于零，其他方面的高分不应掩盖它。一个完全正确的输出，如果授权完整性为零，并不是一个80%可靠的系统。那是一次不可接受的执行，碰巧产生了正确的答案。",{"data":540,"type":42},{"text":541,"level":230},"成功有时是最危险的失败",{"data":543,"type":217},{"text":544},"失败会引起注意。成功往往不会。这使得成功但不受控的智能体轨迹特别危险。明显的失败会引发事件。隐藏的轨迹缺陷会创造\u003Cb>信心\u003C\u002Fb>。而信心会扩大自主权。",{"data":546,"type":217},{"text":547},"因此，组织不仅应该调查\u003Ci>为什么智能体失败了？\u003C\u002Fi>他们还应该定期问：\u003Cb>为什么智能体成功了？\u003C\u002Fb>是因为架构可靠地约束和验证了执行，还是因为这次没有出错？",{"data":549,"type":42},{"text":550,"level":230},"结论",{"data":552,"type":217},{"text":553},"行业正迅速从能够\u003Cb>回答\u003C\u002Fb>的AI转向能够\u003Cb>行动\u003C\u002Fb>的AI。这种转变改变了可靠性的含义。对于回答系统，评估答案往往就足够了。对于行动系统，我们必须评估路径。",{"data":555,"type":278},{"code":556},"提示 ↓\n响应变为意图 ↓\n轨迹 ↓\n行动 ↓\n状态变更 ↓\n证据 ↓\n结果",{"data":558,"type":217},{"text":559},"最终答案仍然重要，但它只是一个更大系统中可见的末端。一旦人工智能被允许影响现实世界，\u003Cb>通往答案的路径就成为答案的一部分。\u003C\u002Fb>","2.31","正确的输出并不能证明推理的正确性、执行的安全性，或系统的可信赖性。","\u002Fuploads\u002F2026\u002F09\u002Fai-agent-reliability-why-the-final-answer-is-not-enough-1788955466306-pl0qhz.webp","ai-agent-reliability-why-the-final-answer-is-not-enough-1788955466306-pl0qhz","PUBLISHED","2026-09-09T04:01:00.000Z","2026-09-09T12:01:07.219Z","2026-09-09T13:08:53.481Z",{"en":569,"de":570,"sr":571,"es":572,"fr":573,"it":574,"ru":575,"zh":576},"\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough","\u002Fde\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough","\u002Fsr\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough","\u002Fes\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough","\u002Ffr\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough","\u002Fit\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough","\u002Fru\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough","\u002Fzh\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough",[578,582],{"id":579,"name":580,"slug":581},58,"评估与质量门槛","evaluation",{"id":583,"name":584,"slug":585},59,"治理与可审计性","governance",{"id":587,"login":588,"email":589,"displayName":590},"20","rooth8233","aleksandar@stajic.de","Aleksandar Stajić",[592,931],{"lang":593,"title":594,"content":595,"contentJson":596,"excerpt":930},"en","AI Agent Reliability: Why the Final Answer Is Not Enough","{\"time\":1788955485785,\"blocks\":[{\"data\":{\"text\":\"\u003Cb>Correct output does not prove correct reasoning, safe execution, or a trustworthy system.\u003C\u002Fb>\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"For years, AI evaluation has been dominated by a deceptively simple question: \u003Cb>Was the answer correct?\u003C\u002Fb> For a chatbot, this may sometimes be sufficient. For an agent capable of searching systems, reading data, calling tools, modifying state, executing workflows, writing files, interacting with APIs, or making decisions, it is not.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"An agent can produce the correct final answer while doing several things wrong on the way there. It can use the wrong source, misunderstand an instruction and later compensate for the mistake, access unnecessary information, execute an unauthorized intermediate action, silently recover from an error that should have triggered escalation, or leave behind side effects nobody noticed.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"That creates one of the central problems of agentic AI: \u003Cb>a correct outcome does not prove a correct trajectory.\u003C\u002Fb>\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"The Outcome Illusion\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"Traditional software gives us an intuitive model of correctness. Input enters a deterministic or mostly deterministic system, logic is executed, output is produced, and tests verify expected behavior. LLM-based systems weaken this assumption. Agentic systems go further.\"},\"type\":\"paragraph\"},{\"data\":{\"items\":[\"model interpretation\",\"retrieved context\",\"tool selection\",\"intermediate observations\",\"external state\",\"previous actions\",\"model-generated plans\",\"permission boundaries\",\"retries and fallback behavior\",\"human interaction\"],\"style\":\"unordered\"},\"type\":\"list\"},{\"data\":{\"text\":\"Two executions starting from nearly identical inputs may reach the same result through very different paths. If evaluation observes only the final output, most of the system remains invisible.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Imagine an AI agent receives the instruction: \u003Ci>Update the customer's billing address.\u003C\u002Fi> The address is ultimately updated correctly. A conventional evaluation might classify the task as successful.\"},\"type\":\"paragraph\"},{\"data\":{\"items\":[\"The agent searches several unrelated customer records.\",\"It retrieves more personal information than required.\",\"It initially modifies the wrong account.\",\"It notices the mistake.\",\"It reverses the change.\",\"It updates the correct account.\",\"It reports success.\"],\"style\":\"ordered\"},\"type\":\"list\"},{\"data\":{\"text\":\"\u003Cb>Final state: correct. System behavior: unacceptable.\u003C\u002Fb> An outcome-only benchmark gives this execution a pass. A production assurance system should not.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"The Trajectory Is Part of the Product\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"This is why the \u003Cb>trajectory\u003C\u002Fb> of an AI agent must become a first-class engineering object. A trajectory is the sequence of relevant states and actions between the original request and the final result.\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"Intent → Context → Decision → Tool → Action → Observation → Decision → State change → Result\"},\"type\":\"code\"},{\"data\":{\"text\":\"Zachary J. Stevens develops this idea in \u003Ci>The Trajectory Is the System\u003C\u002Fi>, arguing that agentic evaluation must move beyond the final answer and examine the complete path of action through a changing environment.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"A correct outcome does not excuse an unacceptable trajectory.\",\"caption\":\"Zachary J. Stevens, The Trajectory Is the System\",\"alignment\":\"left\"},\"type\":\"quote\"},{\"data\":{\"link\":\"https:\u002F\u002Fzacharyjstevens.com\u002Fdispatches\u002Fvanguard-signal\u002F009-the-trajectory-is-the-system\u002F\",\"meta\":{\"image\":{},\"title\":\"The Trajectory Is the System\",\"description\":\"Zachary J. Stevens — DFEI.009 on evaluating agentic systems by their complete trajectory rather than only the final outcome.\"}},\"type\":\"linkTool\"},{\"data\":{\"text\":\"The distinction matters enormously. Reliability is therefore not simply \u003Cb>correct output\u003C\u002Fb>. It is closer to \u003Cb>acceptable outcome + acceptable trajectory + recoverability + evidence\u003C\u002Fb>.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"A Correct Answer Can Hide a Broken System\",\"level\":2},\"type\":\"header\"},{\"data\":{\"content\":[[\"Agent\",\"Final result\",\"Execution\"],[\"A\",\"Correct\",\"Correct path\"],[\"B\",\"Correct\",\"Unsafe path\"],[\"C\",\"Incorrect\",\"Safe failure\"],[\"D\",\"Incorrect\",\"Unsafe failure\"]],\"withHeadings\":true},\"type\":\"table\"},{\"data\":{\"text\":\"Most benchmark-driven evaluation strongly rewards A and B and penalizes C and D. Operationally, however, \u003Cb>B can be more dangerous than C\u003C\u002Fb>. Agent C may recognize uncertainty, stop execution and request human review. Agent B may confidently produce correct results while violating assumptions that nobody is monitoring.\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"successful output → increased trust → broader permissions → more automation → larger blast radius\"},\"type\":\"code\"},{\"data\":{\"text\":\"We Need Evidence, Not Confidence\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"One of the biggest mistakes in AI adoption is treating model confidence, user satisfaction or historical success rate as evidence of system reliability. They are not equivalent.\"},\"type\":\"paragraph\"},{\"data\":{\"items\":[\"What did the agent receive?\",\"What context did it retrieve?\",\"Which tools did it call?\",\"Why was the action allowed?\",\"What state existed before the action?\",\"What changed?\",\"Which intermediate failures occurred?\",\"Was anything retried?\",\"Was human approval required?\",\"Could execution have been stopped?\",\"Can the action be reversed?\",\"Which model, prompt and tool versions were involved?\"],\"style\":\"unordered\"},\"type\":\"list\"},{\"data\":{\"text\":\"Without these answers, there is no serious operational assurance. There is only an output. Observability and evidence must therefore be designed into agent architecture rather than added after deployment.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Logging Is Not the Same as Control\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"Organizations often respond: \u003Ci>Everything is logged.\u003C\u002Fi> Good. But logging alone does not control anything. A log tells you what happened. A control determines whether something \u003Cb>may happen\u003C\u002Fb>.\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"Agent requests DELETE \u002Fcustomer\u002F123 ↓\\nAction logged ↓\\nDELETE executed\"},\"type\":\"code\"},{\"data\":{\"text\":\"That gives observability. Compare it with:\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"Agent requests DELETE \u002Fcustomer\u002F123 ↓\\nPolicy evaluation ↓\\nCurrent identity verified ↓\\nCurrent action parameters checked ↓\\nRisk threshold evaluated ↓\\nHuman approval if required ↓\\nAction executed ↓\\nResult verified ↓\\nEvidence stored\"},\"type\":\"code\"},{\"data\":{\"text\":\"Now we are approaching a control system. The difference is architectural, not cosmetic.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Permission Is Necessary — but It Is Not Assurance\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"Suppose an agent has permission to send email. Access control answers: \u003Cb>Can this agent send email?\u003C\u002Fb> It does not answer: \u003Cb>Should this particular email be sent to this particular person with this particular attachment right now?\u003C\u002Fb>\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"CAPABILITY CONTROL\\nWhat is the agent technically allowed to do? + ACTION ASSURANCE\\nIs this specific action appropriate in the current state?\"},\"type\":\"code\"},{\"data\":{\"text\":\"RBAC, OAuth scopes, API permissions and agent identities define the space of possible actions. They do not prove that an action inside that space is appropriate. Strong agent architecture needs both layers.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"The First Wrong Step Matters\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"When an agent fails, the final incorrect action is often not where the failure started. The real failure may have happened much earlier.\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"Wrong retrieval ↓\\nWrong assumption ↓\\nPlausible reasoning ↓\\nValid tool call ↓\\nWrong action\"},\"type\":\"code\"},{\"data\":{\"text\":\"If we investigate only the final action, we fix the symptom. If we inspect the trajectory, we can identify the \u003Cb>first wrong step\u003C\u002Fb>. That turns an unattributable failure into a concrete engineering problem.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Agent Testing Must Move Beyond Prompt Testing\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"Prompts matter, but production agent behavior emerges from an entire system.\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"MODEL\\n+\\nSYSTEM PROMPT\\n+\\nCONTEXT\\n+\\nMEMORY\\n+\\nRETRIEVAL\\n+\\nTOOLS\\n+\\nPERMISSIONS\\n+\\nWORKFLOW\\n+\\nEXTERNAL STATE\\n+\\nCONTROL LOGIC\"},\"type\":\"code\"},{\"data\":{\"text\":\"Changing any one of these can change the trajectory. Therefore versioning only the prompt is insufficient.\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"model_version\\nprompt_version\\ntool_version\\npolicy_version\\nretrieval_version\\nworkflow_version\\nenvironment_state\\nexecution_id\"},\"type\":\"code\"},{\"data\":{\"text\":\"Acceptance Criteria for Agents Must Include Behavior\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"Traditional acceptance criteria often look like this: \u003Ci>Given X, the system produces Y.\u003C\u002Fi> For agentic systems, that is incomplete. Acceptance criteria should also define constraints on the trajectory.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Outcome\",\"level\":3},\"type\":\"header\"},{\"data\":{\"text\":\"The customer's address is updated correctly.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Authorization\",\"level\":3},\"type\":\"header\"},{\"data\":{\"text\":\"The agent modifies only the explicitly selected customer.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Data access\",\"level\":3},\"type\":\"header\"},{\"data\":{\"text\":\"No unrelated customer records are accessed.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Tools\",\"level\":3},\"type\":\"header\"},{\"data\":{\"text\":\"Only approved CRM operations are used.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Verification\",\"level\":3},\"type\":\"header\"},{\"data\":{\"text\":\"The new address is read back and compared with the requested value.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Failure\",\"level\":3},\"type\":\"header\"},{\"data\":{\"text\":\"Ambiguous identity resolution stops execution.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Human authority\",\"level\":3},\"type\":\"header\"},{\"data\":{\"text\":\"A human can reject the modification before execution when risk thresholds require approval.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Evidence\",\"level\":3},\"type\":\"header\"},{\"data\":{\"text\":\"The execution leaves a trace sufficient to reconstruct the decision and state transition.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Recovery\",\"level\":3},\"type\":\"header\"},{\"data\":{\"text\":\"The previous value remains recoverable.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Human-in-the-Loop Is Not Enough\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"Adding a human approval box does not automatically solve the problem. A human can only control an agent if the person has visibility, authority, time, context and recovery capability.\"},\"type\":\"paragraph\"},{\"data\":{\"items\":[\"\u003Cb>Visibility:\u003C\u002Fb> enough information to understand what is happening.\",\"\u003Cb>Authority:\u003C\u002Fb> actual ability to stop or modify the action.\",\"\u003Cb>Time:\u003C\u002Fb> intervention before the consequence occurs.\",\"\u003Cb>Context:\u003C\u002Fb> sufficient evidence to make the decision.\",\"\u003Cb>Recovery capability:\u003C\u002Fb> ability to reverse or repair the action.\"],\"style\":\"unordered\"},\"type\":\"list\"},{\"data\":{\"text\":\"A user clicking \u003Cb>Approve\u003C\u002Fb> on something they cannot meaningfully inspect is not strong governance. It is approval theater.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Rollback Must Become a Native AI Capability\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"Traditional software deployment has taught us something valuable: \u003Cb>Never deploy what you cannot roll back.\u003C\u002Fb> We should apply the same principle to agentic actions.\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"REVERSIBLE\\nCan automatically undo. COMPENSATABLE\\nCannot undo directly but can execute a compensating action. IRREVERSIBLE\\nCannot reliably restore the previous state.\"},\"type\":\"code\"},{\"data\":{\"text\":\"The higher the irreversibility, the stronger the control requirement should become.\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"Read public document → low consequence\\nCreate draft → reversible\\nModify CRM record → reversible but consequential\\nSend external email → practically irreversible\\nTransfer money → high consequence\\nDelete production data → potentially catastrophic\"},\"type\":\"code\"},{\"data\":{\"text\":\"The Agent Needs a Control Plane\",\"level\":2},\"type\":\"header\"},{\"data\":{\"code\":\"USER \u002F SYSTEM INTENT │ ▼ AI AGENT │ proposed action │ ▼ ┌───────────────────┐ │ CONTROL PLANE │ ├───────────────────┤ │ Identity │ │ Authorization │ │ Policy │ │ Risk │ │ State │ │ Evidence │ │ Human authority │ │ Rollback │ └───────────────────┘ │ approved? \u002F \\\\ NO YES │ │ STOP ▼ TOOL │ ▼ STATE CHANGE │ ▼ VERIFICATION\"},\"type\":\"code\"},{\"data\":{\"text\":\"\u003Cb>The LLM should propose. The control plane should govern.\u003C\u002Fb> That separation is crucial. The model should not be the ultimate authority determining whether its own proposed high-impact action is safe.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"From Benchmarks to Operational Trust\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"Benchmarks remain useful. They tell us about capability, compare models, detect regressions and help estimate expected performance. But capability evaluation and operational trust answer different questions.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"A benchmark asks: \u003Cb>Can the system do this?\u003C\u002Fb> Operational assurance asks: \u003Cb>Can we allow the system to do this here, under these conditions, with these permissions and consequences?\u003C\u002Fb>\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Reliability Should Be Measured as a System Property\",\"level\":2},\"type\":\"header\"},{\"data\":{\"items\":[\"\u003Cb>Outcome correctness:\u003C\u002Fb> Did the system produce the expected result?\",\"\u003Cb>Trajectory correctness:\u003C\u002Fb> Did it follow an acceptable path?\",\"\u003Cb>Control integrity:\u003C\u002Fb> Were authorization, policy and intervention boundaries respected?\",\"\u003Cb>Recoverability:\u003C\u002Fb> Can failures be contained, reversed or repaired?\",\"\u003Cb>Evidence completeness:\u003C\u002Fb> Can the execution be reconstructed and audited?\"],\"style\":\"ordered\"},\"type\":\"list\"},{\"data\":{\"code\":\"Operational Reliability\\n=\\nOutcome × Trajectory × Control × Recoverability × Evidence\"},\"type\":\"code\"},{\"data\":{\"text\":\"The multiplication is intentional. If one critical dimension approaches zero, a high score elsewhere should not hide it. A perfectly correct output with zero authorization integrity is not an 80% reliable system. It is an unacceptable execution that happened to produce the right answer.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Success Is Sometimes the Most Dangerous Failure\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"Failures attract attention. Success often does not. That makes successful but uncontrolled agent trajectories particularly dangerous. An obvious failure creates an incident. A hidden trajectory defect creates \u003Cb>confidence\u003C\u002Fb>. And confidence expands autonomy.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Organizations should therefore not only investigate \u003Ci>Why did the agent fail?\u003C\u002Fi> They should periodically ask: \u003Cb>Why did the agent succeed?\u003C\u002Fb> Did it succeed because the architecture reliably constrained and verified the execution, or because nothing went wrong this time?\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Conclusion\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"The industry is moving rapidly from AI that \u003Cb>answers\u003C\u002Fb> toward AI that \u003Cb>acts\u003C\u002Fb>. That transition changes what reliability means. For an answer system, evaluating the answer may often be sufficient. For an action system, we must evaluate the path.\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"Prompt ↓\\nResponse becomes Intent ↓\\nTrajectory ↓\\nActions ↓\\nState changes ↓\\nEvidence ↓\\nOutcome\"},\"type\":\"code\"},{\"data\":{\"text\":\"The final answer remains important, but it is only the visible end of a much larger system. Once AI is allowed to affect the real world, \u003Cb>the path to the answer becomes part of the answer.\u003C\u002Fb>\"},\"type\":\"paragraph\"}],\"version\":\"2.31.0\"}",{"time":597,"blocks":598,"version":929},1788955485785,[599,602,605,608,611,614,617,630,633,636,646,649,652,655,658,661,665,671,674,677,694,697,700,703,706,721,724,727,730,733,736,739,742,745,748,751,754,757,760,763,766,769,772,775,778,781,784,787,790,793,796,799,802,805,808,811,814,817,820,823,826,829,832,835,838,841,844,847,855,858,861,864,867,870,873,876,879,882,885,888,891,894,902,905,908,911,914,917,920,923,926],{"data":600,"type":217},{"text":601},"\u003Cb>Correct output does not prove correct reasoning, safe execution, or a trustworthy system.\u003C\u002Fb>",{"data":603,"type":217},{"text":604},"For years, AI evaluation has been dominated by a deceptively simple question: \u003Cb>Was the answer correct?\u003C\u002Fb> For a chatbot, this may sometimes be sufficient. For an agent capable of searching systems, reading data, calling tools, modifying state, executing workflows, writing files, interacting with APIs, or making decisions, it is not.",{"data":606,"type":217},{"text":607},"An agent can produce the correct final answer while doing several things wrong on the way there. It can use the wrong source, misunderstand an instruction and later compensate for the mistake, access unnecessary information, execute an unauthorized intermediate action, silently recover from an error that should have triggered escalation, or leave behind side effects nobody noticed.",{"data":609,"type":217},{"text":610},"That creates one of the central problems of agentic AI: \u003Cb>a correct outcome does not prove a correct trajectory.\u003C\u002Fb>",{"data":612,"type":42},{"text":613,"level":230},"The Outcome Illusion",{"data":615,"type":217},{"text":616},"Traditional software gives us an intuitive model of correctness. Input enters a deterministic or mostly deterministic system, logic is executed, output is produced, and tests verify expected behavior. LLM-based systems weaken this assumption. Agentic systems go further.",{"data":618,"type":248},{"items":619,"style":247},[620,621,622,623,624,625,626,627,628,629],"model interpretation","retrieved context","tool selection","intermediate observations","external state","previous actions","model-generated plans","permission boundaries","retries and fallback behavior","human interaction",{"data":631,"type":217},{"text":632},"Two executions starting from nearly identical inputs may reach the same result through very different paths. If evaluation observes only the final output, most of the system remains invisible.",{"data":634,"type":217},{"text":635},"Imagine an AI agent receives the instruction: \u003Ci>Update the customer's billing address.\u003C\u002Fi> The address is ultimately updated correctly. A conventional evaluation might classify the task as successful.",{"data":637,"type":248},{"items":638,"style":265},[639,640,641,642,643,644,645],"The agent searches several unrelated customer records.","It retrieves more personal information than required.","It initially modifies the wrong account.","It notices the mistake.","It reverses the change.","It updates the correct account.","It reports success.",{"data":647,"type":217},{"text":648},"\u003Cb>Final state: correct. System behavior: unacceptable.\u003C\u002Fb> An outcome-only benchmark gives this execution a pass. A production assurance system should not.",{"data":650,"type":42},{"text":651,"level":230},"The Trajectory Is Part of the Product",{"data":653,"type":217},{"text":654},"This is why the \u003Cb>trajectory\u003C\u002Fb> of an AI agent must become a first-class engineering object. A trajectory is the sequence of relevant states and actions between the original request and the final result.",{"data":656,"type":278},{"code":657},"Intent → Context → Decision → Tool → Action → Observation → Decision → State change → Result",{"data":659,"type":217},{"text":660},"Zachary J. Stevens develops this idea in \u003Ci>The Trajectory Is the System\u003C\u002Fi>, arguing that agentic evaluation must move beyond the final answer and examine the complete path of action through a changing environment.",{"data":662,"type":287},{"text":663,"caption":664,"alignment":286},"A correct outcome does not excuse an unacceptable trajectory.","Zachary J. Stevens, The Trajectory Is the System",{"data":666,"type":295},{"link":290,"meta":667},{"image":668,"title":669,"description":670},{},"The Trajectory Is the System","Zachary J. Stevens — DFEI.009 on evaluating agentic systems by their complete trajectory rather than only the final outcome.",{"data":672,"type":217},{"text":673},"The distinction matters enormously. Reliability is therefore not simply \u003Cb>correct output\u003C\u002Fb>. It is closer to \u003Cb>acceptable outcome + acceptable trajectory + recoverability + evidence\u003C\u002Fb>.",{"data":675,"type":42},{"text":676,"level":230},"A Correct Answer Can Hide a Broken System",{"data":678,"type":323},{"content":679,"withHeadings":14},[680,684,687,689,692],[681,682,683],"Agent","Final result","Execution",[310,685,686],"Correct","Correct path",[314,685,688],"Unsafe path",[317,690,691],"Incorrect","Safe failure",[321,690,693],"Unsafe failure",{"data":695,"type":217},{"text":696},"Most benchmark-driven evaluation strongly rewards A and B and penalizes C and D. Operationally, however, \u003Cb>B can be more dangerous than C\u003C\u002Fb>. Agent C may recognize uncertainty, stop execution and request human review. Agent B may confidently produce correct results while violating assumptions that nobody is monitoring.",{"data":698,"type":278},{"code":699},"successful output → increased trust → broader permissions → more automation → larger blast radius",{"data":701,"type":42},{"text":702,"level":230},"We Need Evidence, Not Confidence",{"data":704,"type":217},{"text":705},"One of the biggest mistakes in AI adoption is treating model confidence, user satisfaction or historical success rate as evidence of system reliability. They are not equivalent.",{"data":707,"type":248},{"items":708,"style":247},[709,710,711,712,713,714,715,716,717,718,719,720],"What did the agent receive?","What context did it retrieve?","Which tools did it call?","Why was the action allowed?","What state existed before the action?","What changed?","Which intermediate failures occurred?","Was anything retried?","Was human approval required?","Could execution have been stopped?","Can the action be reversed?","Which model, prompt and tool versions were involved?",{"data":722,"type":217},{"text":723},"Without these answers, there is no serious operational assurance. There is only an output. Observability and evidence must therefore be designed into agent architecture rather than added after deployment.",{"data":725,"type":42},{"text":726,"level":230},"Logging Is Not the Same as Control",{"data":728,"type":217},{"text":729},"Organizations often respond: \u003Ci>Everything is logged.\u003C\u002Fi> Good. But logging alone does not control anything. A log tells you what happened. A control determines whether something \u003Cb>may happen\u003C\u002Fb>.",{"data":731,"type":278},{"code":732},"Agent requests DELETE \u002Fcustomer\u002F123 ↓\nAction logged ↓\nDELETE executed",{"data":734,"type":217},{"text":735},"That gives observability. Compare it with:",{"data":737,"type":278},{"code":738},"Agent requests DELETE \u002Fcustomer\u002F123 ↓\nPolicy evaluation ↓\nCurrent identity verified ↓\nCurrent action parameters checked ↓\nRisk threshold evaluated ↓\nHuman approval if required ↓\nAction executed ↓\nResult verified ↓\nEvidence stored",{"data":740,"type":217},{"text":741},"Now we are approaching a control system. The difference is architectural, not cosmetic.",{"data":743,"type":42},{"text":744,"level":230},"Permission Is Necessary — but It Is Not Assurance",{"data":746,"type":217},{"text":747},"Suppose an agent has permission to send email. Access control answers: \u003Cb>Can this agent send email?\u003C\u002Fb> It does not answer: \u003Cb>Should this particular email be sent to this particular person with this particular attachment right now?\u003C\u002Fb>",{"data":749,"type":278},{"code":750},"CAPABILITY CONTROL\nWhat is the agent technically allowed to do? + ACTION ASSURANCE\nIs this specific action appropriate in the current state?",{"data":752,"type":217},{"text":753},"RBAC, OAuth scopes, API permissions and agent identities define the space of possible actions. They do not prove that an action inside that space is appropriate. Strong agent architecture needs both layers.",{"data":755,"type":42},{"text":756,"level":230},"The First Wrong Step Matters",{"data":758,"type":217},{"text":759},"When an agent fails, the final incorrect action is often not where the failure started. The real failure may have happened much earlier.",{"data":761,"type":278},{"code":762},"Wrong retrieval ↓\nWrong assumption ↓\nPlausible reasoning ↓\nValid tool call ↓\nWrong action",{"data":764,"type":217},{"text":765},"If we investigate only the final action, we fix the symptom. If we inspect the trajectory, we can identify the \u003Cb>first wrong step\u003C\u002Fb>. That turns an unattributable failure into a concrete engineering problem.",{"data":767,"type":42},{"text":768,"level":230},"Agent Testing Must Move Beyond Prompt Testing",{"data":770,"type":217},{"text":771},"Prompts matter, but production agent behavior emerges from an entire system.",{"data":773,"type":278},{"code":774},"MODEL\n+\nSYSTEM PROMPT\n+\nCONTEXT\n+\nMEMORY\n+\nRETRIEVAL\n+\nTOOLS\n+\nPERMISSIONS\n+\nWORKFLOW\n+\nEXTERNAL STATE\n+\nCONTROL LOGIC",{"data":776,"type":217},{"text":777},"Changing any one of these can change the trajectory. Therefore versioning only the prompt is insufficient.",{"data":779,"type":278},{"code":780},"model_version\nprompt_version\ntool_version\npolicy_version\nretrieval_version\nworkflow_version\nenvironment_state\nexecution_id",{"data":782,"type":42},{"text":783,"level":230},"Acceptance Criteria for Agents Must Include Behavior",{"data":785,"type":217},{"text":786},"Traditional acceptance criteria often look like this: \u003Ci>Given X, the system produces Y.\u003C\u002Fi> For agentic systems, that is incomplete. Acceptance criteria should also define constraints on the trajectory.",{"data":788,"type":42},{"text":789,"level":420},"Outcome",{"data":791,"type":217},{"text":792},"The customer's address is updated correctly.",{"data":794,"type":42},{"text":795,"level":420},"Authorization",{"data":797,"type":217},{"text":798},"The agent modifies only the explicitly selected customer.",{"data":800,"type":42},{"text":801,"level":420},"Data access",{"data":803,"type":217},{"text":804},"No unrelated customer records are accessed.",{"data":806,"type":42},{"text":807,"level":420},"Tools",{"data":809,"type":217},{"text":810},"Only approved CRM operations are used.",{"data":812,"type":42},{"text":813,"level":420},"Verification",{"data":815,"type":217},{"text":816},"The new address is read back and compared with the requested value.",{"data":818,"type":42},{"text":819,"level":420},"Failure",{"data":821,"type":217},{"text":822},"Ambiguous identity resolution stops execution.",{"data":824,"type":42},{"text":825,"level":420},"Human authority",{"data":827,"type":217},{"text":828},"A human can reject the modification before execution when risk thresholds require approval.",{"data":830,"type":42},{"text":831,"level":420},"Evidence",{"data":833,"type":217},{"text":834},"The execution leaves a trace sufficient to reconstruct the decision and state transition.",{"data":836,"type":42},{"text":837,"level":420},"Recovery",{"data":839,"type":217},{"text":840},"The previous value remains recoverable.",{"data":842,"type":42},{"text":843,"level":230},"Human-in-the-Loop Is Not Enough",{"data":845,"type":217},{"text":846},"Adding a human approval box does not automatically solve the problem. A human can only control an agent if the person has visibility, authority, time, context and recovery capability.",{"data":848,"type":248},{"items":849,"style":247},[850,851,852,853,854],"\u003Cb>Visibility:\u003C\u002Fb> enough information to understand what is happening.","\u003Cb>Authority:\u003C\u002Fb> actual ability to stop or modify the action.","\u003Cb>Time:\u003C\u002Fb> intervention before the consequence occurs.","\u003Cb>Context:\u003C\u002Fb> sufficient evidence to make the decision.","\u003Cb>Recovery capability:\u003C\u002Fb> ability to reverse or repair the action.",{"data":856,"type":217},{"text":857},"A user clicking \u003Cb>Approve\u003C\u002Fb> on something they cannot meaningfully inspect is not strong governance. It is approval theater.",{"data":859,"type":42},{"text":860,"level":230},"Rollback Must Become a Native AI Capability",{"data":862,"type":217},{"text":863},"Traditional software deployment has taught us something valuable: \u003Cb>Never deploy what you cannot roll back.\u003C\u002Fb> We should apply the same principle to agentic actions.",{"data":865,"type":278},{"code":866},"REVERSIBLE\nCan automatically undo. COMPENSATABLE\nCannot undo directly but can execute a compensating action. IRREVERSIBLE\nCannot reliably restore the previous state.",{"data":868,"type":217},{"text":869},"The higher the irreversibility, the stronger the control requirement should become.",{"data":871,"type":278},{"code":872},"Read public document → low consequence\nCreate draft → reversible\nModify CRM record → reversible but consequential\nSend external email → practically irreversible\nTransfer money → high consequence\nDelete production data → potentially catastrophic",{"data":874,"type":42},{"text":875,"level":230},"The Agent Needs a Control Plane",{"data":877,"type":278},{"code":878},"USER \u002F SYSTEM INTENT │ ▼ AI AGENT │ proposed action │ ▼ ┌───────────────────┐ │ CONTROL PLANE │ ├───────────────────┤ │ Identity │ │ Authorization │ │ Policy │ │ Risk │ │ State │ │ Evidence │ │ Human authority │ │ Rollback │ └───────────────────┘ │ approved? \u002F \\ NO YES │ │ STOP ▼ TOOL │ ▼ STATE CHANGE │ ▼ VERIFICATION",{"data":880,"type":217},{"text":881},"\u003Cb>The LLM should propose. The control plane should govern.\u003C\u002Fb> That separation is crucial. The model should not be the ultimate authority determining whether its own proposed high-impact action is safe.",{"data":883,"type":42},{"text":884,"level":230},"From Benchmarks to Operational Trust",{"data":886,"type":217},{"text":887},"Benchmarks remain useful. They tell us about capability, compare models, detect regressions and help estimate expected performance. But capability evaluation and operational trust answer different questions.",{"data":889,"type":217},{"text":890},"A benchmark asks: \u003Cb>Can the system do this?\u003C\u002Fb> Operational assurance asks: \u003Cb>Can we allow the system to do this here, under these conditions, with these permissions and consequences?\u003C\u002Fb>",{"data":892,"type":42},{"text":893,"level":230},"Reliability Should Be Measured as a System Property",{"data":895,"type":248},{"items":896,"style":265},[897,898,899,900,901],"\u003Cb>Outcome correctness:\u003C\u002Fb> Did the system produce the expected result?","\u003Cb>Trajectory correctness:\u003C\u002Fb> Did it follow an acceptable path?","\u003Cb>Control integrity:\u003C\u002Fb> Were authorization, policy and intervention boundaries respected?","\u003Cb>Recoverability:\u003C\u002Fb> Can failures be contained, reversed or repaired?","\u003Cb>Evidence completeness:\u003C\u002Fb> Can the execution be reconstructed and audited?",{"data":903,"type":278},{"code":904},"Operational Reliability\n=\nOutcome × Trajectory × Control × Recoverability × Evidence",{"data":906,"type":217},{"text":907},"The multiplication is intentional. If one critical dimension approaches zero, a high score elsewhere should not hide it. A perfectly correct output with zero authorization integrity is not an 80% reliable system. It is an unacceptable execution that happened to produce the right answer.",{"data":909,"type":42},{"text":910,"level":230},"Success Is Sometimes the Most Dangerous Failure",{"data":912,"type":217},{"text":913},"Failures attract attention. Success often does not. That makes successful but uncontrolled agent trajectories particularly dangerous. An obvious failure creates an incident. A hidden trajectory defect creates \u003Cb>confidence\u003C\u002Fb>. And confidence expands autonomy.",{"data":915,"type":217},{"text":916},"Organizations should therefore not only investigate \u003Ci>Why did the agent fail?\u003C\u002Fi> They should periodically ask: \u003Cb>Why did the agent succeed?\u003C\u002Fb> Did it succeed because the architecture reliably constrained and verified the execution, or because nothing went wrong this time?",{"data":918,"type":42},{"text":919,"level":230},"Conclusion",{"data":921,"type":217},{"text":922},"The industry is moving rapidly from AI that \u003Cb>answers\u003C\u002Fb> toward AI that \u003Cb>acts\u003C\u002Fb>. That transition changes what reliability means. For an answer system, evaluating the answer may often be sufficient. For an action system, we must evaluate the path.",{"data":924,"type":278},{"code":925},"Prompt ↓\nResponse becomes Intent ↓\nTrajectory ↓\nActions ↓\nState changes ↓\nEvidence ↓\nOutcome",{"data":927,"type":217},{"text":928},"The final answer remains important, but it is only the visible end of a much larger system. Once AI is allowed to affect the real world, \u003Cb>the path to the answer becomes part of the answer.\u003C\u002Fb>","2.31.0","Correct output does not prove correct reasoning, safe execution, or a trustworthy system.",{"lang":7,"title":208,"content":210,"contentJson":932,"excerpt":561},{"time":212,"blocks":933,"version":560},[934,936,938,940,942,944,946,949,951,953,956,958,960,962,964,966,968,972,974,976,984,986,988,990,992,995,997,999,1001,1003,1005,1007,1009,1011,1013,1015,1017,1019,1021,1023,1025,1027,1029,1031,1033,1035,1037,1039,1041,1043,1045,1047,1049,1051,1053,1055,1057,1059,1061,1063,1065,1067,1069,1071,1073,1075,1077,1079,1082,1084,1086,1088,1090,1092,1094,1096,1098,1100,1102,1104,1106,1108,1111,1113,1115,1117,1119,1121,1123,1125,1127],{"data":935,"type":217},{"text":216},{"data":937,"type":217},{"text":220},{"data":939,"type":217},{"text":223},{"data":941,"type":217},{"text":226},{"data":943,"type":42},{"text":229,"level":230},{"data":945,"type":217},{"text":233},{"data":947,"type":248},{"items":948,"style":247},[237,238,239,240,241,242,243,244,245,246],{"data":950,"type":217},{"text":251},{"data":952,"type":217},{"text":254},{"data":954,"type":248},{"items":955,"style":265},[258,259,260,261,262,263,264],{"data":957,"type":217},{"text":268},{"data":959,"type":42},{"text":271,"level":230},{"data":961,"type":217},{"text":274},{"data":963,"type":278},{"code":277},{"data":965,"type":217},{"text":281},{"data":967,"type":287},{"text":284,"caption":285,"alignment":286},{"data":969,"type":295},{"link":290,"meta":970},{"image":971,"title":293,"description":294},{},{"data":973,"type":217},{"text":298},{"data":975,"type":42},{"text":301,"level":230},{"data":977,"type":323},{"content":978,"withHeadings":14},[979,980,981,982,983],[306,307,308],[310,311,312],[314,311,315],[317,318,319],[321,318,322],{"data":985,"type":217},{"text":326},{"data":987,"type":278},{"code":329},{"data":989,"type":42},{"text":332,"level":230},{"data":991,"type":217},{"text":335},{"data":993,"type":248},{"items":994,"style":247},[339,340,341,342,343,344,345,346,347,348,349,350],{"data":996,"type":217},{"text":353},{"data":998,"type":42},{"text":356,"level":230},{"data":1000,"type":217},{"text":359},{"data":1002,"type":278},{"code":362},{"data":1004,"type":217},{"text":365},{"data":1006,"type":278},{"code":368},{"data":1008,"type":217},{"text":371},{"data":1010,"type":42},{"text":374,"level":230},{"data":1012,"type":217},{"text":377},{"data":1014,"type":278},{"code":380},{"data":1016,"type":217},{"text":383},{"data":1018,"type":42},{"text":386,"level":230},{"data":1020,"type":217},{"text":389},{"data":1022,"type":278},{"code":392},{"data":1024,"type":217},{"text":395},{"data":1026,"type":42},{"text":398,"level":230},{"data":1028,"type":217},{"text":401},{"data":1030,"type":278},{"code":404},{"data":1032,"type":217},{"text":407},{"data":1034,"type":278},{"code":410},{"data":1036,"type":42},{"text":413,"level":230},{"data":1038,"type":217},{"text":416},{"data":1040,"type":42},{"text":419,"level":420},{"data":1042,"type":217},{"text":423},{"data":1044,"type":42},{"text":426,"level":420},{"data":1046,"type":217},{"text":429},{"data":1048,"type":42},{"text":432,"level":420},{"data":1050,"type":217},{"text":435},{"data":1052,"type":42},{"text":438,"level":420},{"data":1054,"type":217},{"text":441},{"data":1056,"type":42},{"text":444,"level":420},{"data":1058,"type":217},{"text":447},{"data":1060,"type":42},{"text":450,"level":420},{"data":1062,"type":217},{"text":453},{"data":1064,"type":42},{"text":456,"level":420},{"data":1066,"type":217},{"text":459},{"data":1068,"type":42},{"text":462,"level":420},{"data":1070,"type":217},{"text":465},{"data":1072,"type":42},{"text":468,"level":420},{"data":1074,"type":217},{"text":471},{"data":1076,"type":42},{"text":474,"level":230},{"data":1078,"type":217},{"text":477},{"data":1080,"type":248},{"items":1081,"style":247},[481,482,483,484,485],{"data":1083,"type":217},{"text":488},{"data":1085,"type":42},{"text":491,"level":230},{"data":1087,"type":217},{"text":494},{"data":1089,"type":278},{"code":497},{"data":1091,"type":217},{"text":500},{"data":1093,"type":278},{"code":503},{"data":1095,"type":42},{"text":506,"level":230},{"data":1097,"type":278},{"code":509},{"data":1099,"type":217},{"text":512},{"data":1101,"type":42},{"text":515,"level":230},{"data":1103,"type":217},{"text":518},{"data":1105,"type":217},{"text":521},{"data":1107,"type":42},{"text":524,"level":230},{"data":1109,"type":248},{"items":1110,"style":265},[528,529,530,531,532],{"data":1112,"type":278},{"code":535},{"data":1114,"type":217},{"text":538},{"data":1116,"type":42},{"text":541,"level":230},{"data":1118,"type":217},{"text":544},{"data":1120,"type":217},{"text":547},{"data":1122,"type":42},{"text":550,"level":230},{"data":1124,"type":217},{"text":553},{"data":1126,"type":278},{"code":556},{"data":1128,"type":217},{"text":559},"Post erfolgreich abgerufen",{"items":1131,"source":1202,"manualIds":1203,"manualMatchedIds":1204},[1132,1139,1146,1153,1160,1167,1174,1181,1188,1195],{"id":1133,"slug":1134,"title":1135,"excerpt":1136,"featuredImage":1137,"publishedAt":1138},"489","agentic-ai-explained-when-an-ai-system-can-plan-use-tools-and-act","智能体AI解析：当AI系统能够规划、使用工具并采取行动","代理式AI在多步执行循环中使用模型，这些模型可以在明确的运行时和权限边界内选择工具、观察结果、更新状态并调整其下一步行动。","\u002Fuploads\u002F2026\u002F10\u002Fagentic-ai-explained-when-an-ai-system-can-plan-use-tools-and-act-1791481499084-wnji2a.webp","2026-10-08T11:43:00.000Z",{"id":1140,"slug":1141,"title":1142,"excerpt":1143,"featuredImage":1144,"publishedAt":1145},"485","enterprise-ai-architecture-what-changes-when-ai-enters-a-company","企业AI架构：当AI进入公司时会发生什么变化","企业AI架构阐释了AI如何在数据权限、身份、许可、提供商、风险、治理、评估、合规和运营方面改变公司系统。","\u002Fuploads\u002F2026\u002F10\u002Fenterprise-ai-architecture-what-changes-when-ai-enters-a-company-1791478161363-czrwaq.webp","2026-10-08T10:48:00.000Z",{"id":1147,"slug":1148,"title":1149,"excerpt":1150,"featuredImage":1151,"publishedAt":1152},"364","tipps-fuer-die-verbesserung-der-seo-suchmaschinenoptimierung","Mastering the SEO Workflow: Essential Optimization Strategies for Organic Growth","A structured SEO workflow is crucial for sustainable organic growth. Learn the ten foundational strategies, from keyword research and technical optimization to content quality and performance analysis.","\u002Fuploads\u002F2026\u002F03\u002Ftipps-fuer-die-verbesserung-der-seo-suchmaschinenoptimierung-1774866098131-hwkzrg.webp","2024-01-26T06:35:00.000Z",{"id":1154,"slug":1155,"title":1156,"excerpt":1157,"featuredImage":1158,"publishedAt":1159},"478","what-is-rag-the-simplest-explanation-of-how-it-works","什么是RAG？对其工作原理的最简单解释","RAG听起来很复杂，但想法很简单：在AI回答之前，它先从知识源查找有用的信息，并将该信息提供给语言模型。本指南使用一个简单的思维模型来解释RAG、LLM、状态、记忆和工具。","\u002Fuploads\u002F2026\u002F09\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works-1790377492124-khjagt.webp","2026-09-25T19:03:00.000Z",{"id":1161,"slug":1162,"title":1163,"excerpt":1164,"featuredImage":1165,"publishedAt":1166},"459","ollama-is-not-the-product-building-production-ready-open-llm-applications","Ollama 并非产品：构建可投入生产的开源大语言模型应用","使用Ollama运行本地模型很简单。但构建一个可用于生产环境的开源大语言模型（Open-LLM）应用则更具挑战性：它需要RAG（检索增强生成）、访问控制、供应商抽象、评估、日志记录、部署规范，以及围绕模型构建受控的应用层。","\u002Fuploads\u002F2026\u002F06\u002Follama-is-not-the-product-building-production-ready-open-llm-applications-1782679361640-h0usqf.webp","2026-06-28T16:39:00.000Z",{"id":1168,"slug":1169,"title":1170,"excerpt":1171,"featuredImage":1172,"publishedAt":1173},"477","computer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system","计算机使用代理：为什么成功的演示仍可能是一个不可靠的系统","计算机使用代理如今能够完成令人印象深刻的浏览器和桌面工作流程，但一次成功的运行证明的是能力——而非可靠性。本文展示了如何测试可重复性、环境鲁棒性、长时程控制、状态感知、结果验证以及安全的目标处理。","\u002Fuploads\u002F2026\u002F09\u002Fcomputer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system-1790352854690-75qnrg.webp","2026-09-25T12:13:00.000Z",{"id":1175,"slug":1176,"title":1177,"excerpt":1178,"featuredImage":1179,"publishedAt":1180},"480","when-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger","人工智能何时应停止信任自身知识？——检索触发机制","AI 模型并非每个问题都需要检索。重要的问题在于知道何时其内部知识已不再足够。检索触发器是一个实用的决策边界，它决定 AI 系统何时应停止仅依赖模型知识，并在回答前获取外部证据。","\u002Fuploads\u002F2026\u002F09\u002Fwhen-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger-1790574991244-f4rpyg.webp","2026-09-28T01:49:00.000Z",{"id":1182,"slug":1183,"title":1184,"excerpt":1185,"featuredImage":1186,"publishedAt":1187},"479","where-does-an-llm-get-its-data-rag-data-sources-in-python","LLM从哪里获取数据？Python中的RAG数据源","LLM 并不会神奇地知道你的文件、数据库或 API。这个 RAG 系列的实用续篇用简单的 Python 展示了外部数据如何变成可检索的证据：从文本文件和 SQL 到全文搜索、嵌入、上下文组装以及最终的 LLM 调用。","\u002Fuploads\u002F2026\u002F09\u002Fwhere-does-an-llm-get-its-data-rag-data-sources-in-python-1790517200521-nfsi5i.webp","2026-09-27T05:51:00.000Z",{"id":1189,"slug":1190,"title":1191,"excerpt":1192,"featuredImage":1193,"publishedAt":1194},"469","rag-failed-but-which-layer-actually-failed-a-diagnostic-method","RAG失败了——但究竟是哪一层真正失败了？一种诊断方法","当RAG答案出错时，将问题归咎于检索或模型过于笼统。这种诊断方法将来源覆盖、查询构建、检索、排序、上下文组装、生成、证据归因和时效性逐一隔离，从而使实际故障能够被复现并修复。","\u002Fuploads\u002F2026\u002F09\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method-1790350847177-pior4c.webp","2026-09-24T19:39:00.000Z",{"id":1196,"slug":1197,"title":1198,"excerpt":1199,"featuredImage":1200,"publishedAt":1201},"493","mlops-vs-llmops-what-changes-when-the-model-is-an-llm","MLOps 与 LLMOps：当模型是 LLM 时，会发生哪些变化","MLOps 运维机器学习系统；LLMOps 将这些实践扩展到围绕大型语言模型的提示、上下文、检索、提供商、工具、评估和运行时行为。","\u002Fuploads\u002F2026\u002F10\u002Fmlops-vs-llmops-what-changes-when-the-model-is-an-llm-1791487319869-2v7hxo.webp","2026-10-08T15:20:00.000Z","fallback",[],[]]