[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"portal-settings:stajic:zh":3,"public-menus:all":38,"post:agentic-ai-explained-when-an-ai-system-can-plan-use-tools-and-act:zh":205,"related:post:agentic-ai-explained-when-an-ai-system-can-plan-use-tools-and-act:zh:1":3344},{"statusCode":4,"data":5,"message":37},200,{"tenantId":6,"lang":7,"defaultLang":8,"siteUrl":9,"contactEmail":10,"brandName":11,"logoUrl":12,"siteName":11,"siteDescription":13,"ogImage":10,"robotsIndex":14,"socialLinks":10,"reservedSlugs":10,"seoPolicy":15},"stajic","zh","de","https:\u002F\u002Fstajic.de",null,"Stajic Platform","\u002FLogo_Planet.svg","Stajic Portal",true,{"branding":16,"relatedContent":17,"crossDomainLinks":18},{"logoUrl":12},{"enabled":14},[19,22,25,28,31,34],{"url":20,"label":21,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Ffigure.rocks","figure.rocks",{"url":23,"label":24,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Floving.rocks","loving.rocks",{"url":26,"label":27,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.com","bazify.com",{"url":29,"label":30,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.de","bazify.de",{"url":32,"label":33,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.at","bazify.at",{"url":35,"label":36,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.ba","bazify.ba","Portal settings resolved",[39,45],{"id":40,"name":41,"location":42,"isActive":14,"isDefault":43,"items":44},1,"main-navigation","header",false,[],{"id":46,"name":47,"location":48,"isActive":14,"isDefault":14,"items":49},4,"main-menu","sidebar",[50,66,79,93,103,118,133],{"id":51,"title":52,"url":60,"target":61,"icon":62,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":64,"portfolioId":10,"children":65},"item-18",{"de":53,"en":54,"es":55,"fr":56,"it":54,"ru":57,"sr":58,"zh":59},"Startseite","Home","Inicio","Accueil","Главная","Почетна","首页","\u002Ffull-stack-web-developer-munich-performance-seo-and-maintainable-builds","_self","i-lucide-home","page",111,[],{"id":67,"title":68,"url":75,"target":61,"icon":76,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":77,"portfolioId":10,"children":78},"item-22",{"de":69,"en":69,"es":70,"fr":69,"it":71,"ru":72,"sr":73,"zh":74},"Vision","Visión","Visione","Видение","Визија","想象","\u002Fueber-uns-webdesign-muenchen-webaplikation","i-lucide-eye",113,[],{"id":80,"title":81,"url":89,"target":61,"icon":90,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":91,"portfolioId":10,"children":92},"item-19",{"de":82,"en":83,"es":84,"fr":83,"it":85,"ru":86,"sr":87,"zh":88},"Leistungen","Services","Servicios","Servizi","Услуги","Услуге","服务","\u002Fservices-dienstleistungen-muenchen","i-lucide-wrench",116,[],{"id":94,"title":95,"url":99,"target":61,"icon":100,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":101,"portfolioId":10,"children":102},"item-23",{"de":96,"en":96,"es":96,"fr":96,"it":96,"ru":97,"sr":97,"zh":98},"Blog","Блог","博客","\u002Fblog","i-lucide-book-open",112,[],{"id":104,"title":105,"url":114,"target":61,"icon":115,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":116,"portfolioId":10,"children":117},"item-32",{"de":106,"en":107,"es":108,"fr":109,"it":110,"ru":111,"sr":112,"zh":113},"Neue Technologien","New Technologies","Nuevas tecnologías","Nouvelles technologies","Nuove tecnologie","Новые технологии","Нове технологије","新技术！","\u002Fneue-webtechnologien","i-lucide-sparkles",122,[],{"id":119,"title":120,"url":129,"target":61,"icon":130,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":131,"portfolioId":10,"children":132},"item-20",{"de":121,"en":122,"es":123,"fr":124,"it":125,"ru":126,"sr":127,"zh":128},"Kontakt","Contact us!","Contacto","Contact","Contatto","Контакт","Контактирајте нас","联系我们！","\u002Fcontact","i-lucide-mail",115,[],{"id":134,"title":135,"url":144,"target":61,"icon":145,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":147},"item-21",{"de":136,"en":137,"es":138,"fr":139,"it":140,"ru":141,"sr":142,"zh":143},"Unsere Arbeit","Our Work","Nuestro trabajo","Nos réalisations","I nostri lavori","Наши работы","Наши радови","文件夹","\u002Fportfolio","i-lucide-briefcase",114,[148,161,175,181,193],{"id":149,"title":150,"url":144,"target":61,"icon":159,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":160},"item-24",{"de":151,"en":152,"es":153,"fr":154,"it":155,"ru":156,"sr":157,"zh":158},"Alle Projekte","All Projects","Todos los proyectos","Tous les projets","Tutti i progetti","Все проекты","Сви пројекти","所有项目","i-lucide-grid-3x3",[],{"id":162,"title":163,"url":171,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":174},"item-29",{"de":164,"en":165,"es":166,"fr":167,"it":168,"ru":169,"sr":170,"zh":143},"Local Roots, Global Reach","Local Roots - Global Reach","Empresa local ","Entreprise locale","Azienda locale","Местная компания","Локално предузеће глобално тржиште","\u002Fportfolio\u002Flocal-roots-global-reach-communication-media-systems-for-modern-business","i-lucide-folder","custom",[],{"id":176,"title":177,"url":179,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":180},"item-28",{"de":178,"en":178,"es":178,"fr":178,"it":178,"ru":178,"sr":178,"zh":178},"Solr Suggester","\u002Fportfolio\u002Fsolr-fuzzy-suggester-und-solr-infix-suggester-abfrage-ueber-ajax-und-filterung",[],{"id":182,"title":183,"url":191,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":192},"item-27",{"de":184,"en":185,"es":186,"fr":187,"it":188,"ru":189,"sr":190,"zh":185},"Firmenwebseite SEO","Company Website SEO","Sitio web corporativo SEO","Site web d’entreprise SEO","Sito web aziendale SEO","Корпоративный сайт SEO","Пословна веб-страница SEO","\u002Fportfolio\u002Fseo-sem-branding-mobile-webseite-muenchen",[],{"id":194,"title":195,"url":203,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":204},"item-31",{"de":196,"en":197,"es":198,"fr":199,"it":200,"ru":201,"sr":202,"zh":197},"Digitalisierungsportal","Digitalization Portal","Portal de digitalización","Portail de numérisation","Portale di digitalizzazione","Портал цифровизации","Портал за дигитализацију","\u002Fportfolio\u002Fdigitalisierungsportal-archiv-museum-bibliothek-ead-lido-mets-mods",[],{"statusCode":4,"data":206,"message":3343},{"id":207,"title":208,"slug":209,"content":210,"contentJson":211,"excerpt":1538,"featuredImage":1539,"featuredImageAlt":1540,"featuredImageCaption":10,"featuredImageTitle":10,"featuredImageCopyright":10,"featuredImageAuthor":10,"featuredImageSourceUrl":10,"featuredImageLicense":10,"featuredImageIsAiGenerated":43,"status":1541,"publishedAt":1542,"createdAt":1543,"updatedAt":1544,"seoLocalePaths":1545,"categories":1554,"author":1567,"translations":1572},"489","智能体AI解析：当AI系统能够规划、使用工具并采取行动","agentic-ai-explained-when-an-ai-system-can-plan-use-tools-and-act","\u003Cp>智能体AI是一种AI系统，其中模型可以通过决定下一步做什么、使用工具或其他能力、观察结果、更新其工作状态并持续进行，直到达到停止条件，从而跨多个步骤追求一个目标。模型本身并不是智能体。一个可用的智能体还需要一个运行时或执行框架，用于管理上下文、工具执行、状态、权限、审批、错误以及决策与观察之间的循环。\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--info my-6 rounded-xl border p-5 border-blue-300 bg-blue-50 dark:border-blue-900 dark:bg-blue-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">直接回答\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">普通的模型调用通常是\u003Cstrong>输入 → 模型 → 输出\u003C\u002Fstrong>。智能体系统更接近\u003Cstrong>目标 → 决策 → 工具\u002F行动 → 观察 → 更新决策 → … → 结果\u003C\u002Fstrong>。\u003Cbr>\u003Cbr>关键区别不在于应用是否使用LLM或函数调用，而在于系统是否让模型对多步骤流程的下一步拥有有意义的控制权，同时由运行时约束模型实际被允许做什么。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Caside class=\"editorjs-callout editorjs-callout--warning my-6 rounded-xl border p-5 border-amber-300 bg-amber-50 dark:border-amber-900 dark:bg-amber-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">能力不等于权限\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">模型可能知道如何调用某个工具。运行时可能暴露该工具。但这两点都不意味着当前用户或智能体有权执行底层的业务操作。\u003Cstrong>工具能力、工具权限和业务权限是相互独立的层次。\u003C\u002Fstrong>\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Caside class=\"editorjs-callout editorjs-callout--note my-6 rounded-xl border p-5 border-gray-300 bg-gray-50 dark:border-gray-700 dark:bg-gray-900\u002F40\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">当前来源说明 — 2026年10月8日\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">智能体术语在不同供应商和研究社区之间仍存在差异。OpenAI目前围绕多步骤工作、工具、状态和编排来定义智能体运行时。Anthropic的实用区分仍然有用：工作流遵循预定义的代码路径，而智能体动态地指导自己的流程和工具使用。因此，本文将“智能体AI”视为一种架构谱系，而非一个标准化的产品类别。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Cnav class=\"editorjs-toc\" data-editorjs-toc=\"true\" aria-label=\"目录\">\u003Cstrong class=\"editorjs-toc__title\">目录\u003C\u002Fstrong>\u003Col class=\"editorjs-toc__list editorjs-toc__list--depth-0\">\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-6\" class=\"editorjs-toc__link\">智能体AI的真正含义\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-10\" class=\"editorjs-toc__link\">最简单的例子\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-15\" class=\"editorjs-toc__link\">简单例子止步之处\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-18\" class=\"editorjs-toc__link\">智能体与工作流\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-21\" class=\"editorjs-toc__link\">代理行为是一个谱系，而非二元标签\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-24\" class=\"editorjs-toc__link\">代理系统的最小架构\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-26\" class=\"editorjs-toc__link\">模型不是代理\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-30\" class=\"editorjs-toc__link\">工具使用是核心——但仅靠工具使用并不能构成代理\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-34\" class=\"editorjs-toc__link\">工具能力、权限和授权是不同的\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-37\" class=\"editorjs-toc__link\">运行时或执行框架是实际的执行系统\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-41\" class=\"editorjs-toc__link\">规划有用，但不需要显式计划\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-44\" class=\"editorjs-toc__link\">环境反馈使循环变得有用\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-48\" class=\"editorjs-toc__link\">智能体状态与模型上下文不同\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-51\" class=\"editorjs-toc__link\">记忆是可选的，不是智能体的定义\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-55\" class=\"editorjs-toc__link\">上下文工程在智能体中变得动态化\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-58\" class=\"editorjs-toc__link\">读取工具和副作用工具具有不同的风险\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-60\" class=\"editorjs-toc__link\">人在回路是一种控制机制，而不是智能体AI的对立面\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-64\" class=\"editorjs-toc__link\">智能体需要明确的停止条件\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-66\" class=\"editorjs-toc__link\">恢复是智能体行为的一部分\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-70\" class=\"editorjs-toc__link\">智能体AI不需要多个智能体\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-74\" class=\"editorjs-toc__link\">智能体协议是互操作性层，而非智能体本身\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-78\" class=\"editorjs-toc__link\">轨迹是智能体可靠性的一部分\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-83\" class=\"editorjs-toc__link\">智能体系统扩大了安全面\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-85\" class=\"editorjs-toc__link\">智能体可观测性必须跟随循环\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-88\" class=\"editorjs-toc__link\">如何评估智能体系统\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-90\" class=\"editorjs-toc__link\">何时适合使用智能体\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-93\" class=\"editorjs-toc__link\">原始实现证据\u003C\u002Fa>\u003Col class=\"editorjs-toc__list editorjs-toc__list--depth-1\">\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-94\" class=\"editorjs-toc__link\">Aaasaasa AI Client：模型、运行时和权限是分离的\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-99\" class=\"editorjs-toc__link\">Source of Truth Research Engine：有边界的智能体研究阶段\u003C\u002Fa>\u003C\u002Fli>\u003C\u002Fol>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-105\" class=\"editorjs-toc__link\">常见的智能体式 AI 失败模式\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-107\" class=\"editorjs-toc__link\">常见误解\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-109\" class=\"editorjs-toc__link\">实用的智能体设计顺序\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-111\" class=\"editorjs-toc__link\">智能体 AI 架构检查清单\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-113\" class=\"editorjs-toc__link\">边缘情况和限制\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-119\" class=\"editorjs-toc__link\">什么会改变这个答案？\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-123\" class=\"editorjs-toc__link\">相关规范知识\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-127\" class=\"editorjs-toc__link\">常见问题\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-129\" class=\"editorjs-toc__link\">术语表\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-131\" class=\"editorjs-toc__link\">结论\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-135\" class=\"editorjs-toc__link\">主要来源和当前指南\u003C\u002Fa>\u003C\u002Fli>\u003C\u002Fol>\u003C\u002Fnav>\n\u003Ch2 id=\"section-6\">智能体AI的真正含义\u003C\u002Fh2>\n\u003Cp>从普通生成式AI到智能体AI的重要转变在于对流程的控制。普通助手可以使用接收到的上下文来回答问题。而智能体可以决定回答问题需要额外的步骤：检查文件、搜索代码库、查询API、请求澄清、运行测试、更新工单、委派子任务或在操作失败后重试。\u003C\u002Fp>\n\u003Cp>这并不要求无限的自主权。智能体可以在狭窄的沙箱中运行，受到严格的权限限制，每个有后果的操作都需要审批。如果模型在允许的下一步中动态选择，该系统仍然是智能体的。\u003C\u002Fp>\n\u003Cp>因此，架构比标签更重要。“智能体”应该描述一种系统行为：在工具、状态和反馈之上进行迭代的模型驱动决策——而不仅仅是一个拥有更大提示词的聊天机器人。\u003C\u002Fp>\n\u003Ch2 id=\"section-10\">最简单的例子\u003C\u002Fh2>\n\u003Cp>假设一位开发者向AI系统提出：“找出测试套件失败的原因并修复这个bug。”单次模型调用只能根据给定的文本建议可能的原因。\u003C\u002Fp>\n\u003Cp>一个智能体编码系统可以检查代码库、搜索失败的测试、阅读相关文件、提出修改、编辑代码、运行测试、观察失败、修改实现并再次运行测试。\u003C\u002Fp>\n\u003Cp>智能体的部分不仅仅在于存在shell和文件工具，而在于模型可以利用环境反馈来选择下一步，而不是遵循一个完全预定义的序列。\u003C\u002Fp>\n\u003Csection class=\"editorjs-process my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">基本的智能体循环\u003C\u002Fh3>\u003Cdiv class=\"grid grid-cols-1 md:grid-cols-2 xl:grid-cols-3 gap-4\">\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">1\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">1. 接收目标\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">用户或上游系统定义目标和相关约束。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">2\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">2. 构建当前上下文\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">运行时提供指令、状态、历史、记忆、工具和当前证据。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">3\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">3. 模型决定下一步\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">模型可以回答、调用工具、请求信息、委派或停止。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">4\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">4. 运行时验证请求\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">权限、模式、审批和策略决定所提议的操作是否可以执行。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">5\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">5. 执行工具或操作\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">外部环境发生变化或返回新信息。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">6\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">6. 观察结果\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">运行时将结构化的工具输出、错误或状态变化反馈到下一个模型步骤中。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">7\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">7. 继续或停止\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">循环重复，直到成功、拒绝、升级、预算限制、超时或其他停止条件。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-15\">简单例子止步之处\u003C\u002Fh2>\n\u003Cp>并非每个多步骤AI系统都具有同等的智能体程度。一个工作流可能使用多个LLM调用和工具，但每一步都在代码中预先确定。另一个系统可能让模型决定调用哪个工具、以什么顺序、调用多少次以及何时停止。\u003C\u002Fp>\n\u003Cp>两者都可能有用。区别在于控制权在哪里。预定义工作流将更多控制权放在应用代码中。智能体则将更多战术性流程决策移入模型\u002F运行时循环中。\u003C\u002Fp>\n\u003Ch2 id=\"section-18\">智能体与工作流\u003C\u002Fh2>\n\u003Csection class=\"editorjs-comparison my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">预定义工作流与代理式控制\u003C\u002Fh3>\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left dark:border-gray-700 dark:bg-gray-900\">\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">LLM 工作流\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">代理\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">流程路径\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">工具序列\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">优势\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">风险\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Cp>Anthropic 明确区分了这两种模式：工作流通过预定义的代码路径编排模型和工具，而代理则让模型动态地指导自己的流程和工具使用。这不是唯一可能的术语，但它是一个有用的架构边界。\u003C\u002Fp>\n\u003Ch2 id=\"section-21\">代理行为是一个谱系，而非二元标签\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">层级\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">示例\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">谁决定下一步？\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">单次模型调用\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">总结这份文档\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">应用程序调用模型一次\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">工具辅助响应\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型在回答前可能使用网络搜索\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型在一次响应中从有限工具中选择\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">结构化工作流\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">分类 → 检索 → 生成 → 验证\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">应用程序工作流决定阶段\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">自适应工作流\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型可以在多个分支中选择并重试\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">应用程序与模型共享控制权\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">代理循环\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型根据观察反复选择工具\u002F动作\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型在运行时约束内指导战术执行\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">长期运行的代理\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">代理暂停、恢复、管理产物并继续\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型 + 持久化运行时管理不断演进的执行\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Cp>将上述所有系统都称为“代理”可能会掩盖重要的操作差异。模型对顺序、持续时间和动作的控制越强，运行时隔离、权限、追踪、停止条件和轨迹评估就越重要。\u003C\u002Fp>\n\u003Ch2 id=\"section-24\">代理系统的最小架构\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">组件\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">职责\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">目标 \u002F 任务\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">定义系统试图完成什么。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">解释上下文并决定下一个动作或输出。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">指令\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">定义角色、约束、优先级和任务特定策略。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">上下文组装器\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">构建每一步对模型可见的信息。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">工具目录\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">定义模型可以请求的能力。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">运行时 \u002F 执行框架\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">运行循环、执行工具、管理状态并处理停止条件。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">授权层\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">确定当前主体是否被允许执行提议的动作。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">状态 \u002F 会话\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">在轮次或执行步骤之间保留任务进度。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">观察通道\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">将工具结果和环境变化返回到下一个模型步骤。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">审批 \u002F 人工控制\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">在需要审查时暂停有后果的动作。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">追踪 \u002F 审计\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">记录模型调用、工具、转换、审批和失败。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">评估\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">根据验收标准衡量结果和执行轨迹。\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-26\">模型不是代理\u003C\u002Fh2>\n\u003Cp>语言模型从输入产生输出。它本身并不拥有文件系统、执行 shell 命令、维护持久任务状态、强制执行权限或自动再次调用自身。\u003C\u002Fp>\n\u003Cp>这些能力来自周围的运行时。同一个模型在一个应用程序中可以表现为简单的聊天模型，在另一个应用程序中可以表现为代理循环内的决策引擎。\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--success my-6 rounded-xl border p-5 border-emerald-300 bg-emerald-50 dark:border-emerald-900 dark:bg-emerald-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">架构规则\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">\u003Cstrong>模型能力决定可以提出哪些决策。运行时架构决定实际可以发生什么。\u003C\u002Fstrong>\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-30\">工具使用是核心——但仅靠工具使用并不能构成代理\u003C\u002Fh2>\n\u003Cp>工具让模型获取信息并影响外部系统。示例包括数据库读取、文件操作、shell 执行、网络搜索、浏览器控制、API 调用、工单更新或委托专业代理。\u003C\u002Fp>\n\u003Cp>单次模型调用可以使用一个工具，但仍然是有界的工具辅助响应，而不是长期运行的代理。当工具观察结果馈入一个自适应循环，模型在其中选择下一步做什么时，代理行为就出现了。\u003C\u002Fp>\n\u003Cp>工具设计很重要，因为工具是模型推理与外部现实之间的契约。模糊或重叠的工具会造成路由错误；大型非结构化输出会污染上下文；广泛的副作用工具会扩大影响范围。\u003C\u002Fp>\n\u003Ch2 id=\"section-34\">工具能力、权限和授权是不同的\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">层级\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">问题\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">能力\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">此运行时在技术上能否执行该操作？\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">工具暴露\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">该能力是否对此代理可用？\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">权限\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">在当前策略下，此代理\u002F会话是否可以使用它？\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">用户授权\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">请求主体是否被允许导致此操作？\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">业务授权\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">该操作在领域规则、审批和限制下是否有效？\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">执行\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">该操作是否实际发生？\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">审计\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">系统能否证明谁请求、批准并执行了它？\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Cp>这些层级在原型中经常被合并。模型看到一个退款工具，因此似乎能够发放退款。在生产环境中，工具仍应独立于模型的请求，验证账户、用户、交易、金额、策略和审批条件。\u003C\u002Fp>\n\u003Ch2 id=\"section-37\">运行时或执行框架是实际的执行系统\u003C\u002Fh2>\n\u003Cp>OpenAI 当前的智能体文档明确区分了运行时。不同的运行时可以在不同位置管理编排、状态、工具、沙箱和执行，而模型只是系统的一部分。\u003C\u002Fp>\n\u003Cp>Agents SDK 描述了一个循环：反复调用当前模型，检查输出，执行请求的工具或交接，并持续进行，直到模型返回最终答案或其他真正的停止点。\u003C\u002Fp>\n\u003Cp>这意味着智能体架构决策包括：编排在哪里运行、状态存储在哪里、谁执行工具、哪个沙箱包含副作用，以及谁负责重试、超时和可恢复性。\u003C\u002Fp>\n\u003Ch2 id=\"section-41\">规划有用，但不需要显式计划\u003C\u002Fh2>\n\u003Cp>智能体常被描述为“规划”的系统。在实践中，规划可以是显式的或隐式的。智能体可以先产生一个可见的多步骤计划，也可以一次选择一个下一步行动，并在每次观察后修正。\u003C\u002Fp>\n\u003Cp>对于高度不确定的任务，短视界规划可能更安全，因为环境可能使长期计划失效。架构要求是能够根据目标、当前状态和新证据选择并修正行动。\u003C\u002Fp>\n\u003Ch2 id=\"section-44\">环境反馈使循环变得有用\u003C\u002Fh2>\n\u003Cp>当智能体能够观察其行动是否有效时，它才具有操作意义。工具输出、测试结果、API 响应、文件系统状态、浏览器状态和应用程序记录提供了外部证据，系统可以用这些证据来修正下一个决策。\u003C\u002Fp>\n\u003Cp>Anthropic 的智能体指南强调了这种反馈循环：智能体使用工具，从环境中获取真实情况，评估进展，并继续或请求人工输入。\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--warning my-6 rounded-xl border p-5 border-amber-300 bg-amber-50 dark:border-amber-900 dark:bg-amber-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">自我报告不是环境证明\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">智能体说“任务已完成”并不能证明完成。在可能的情况下，通过外部系统、测试、文件、交易记录或其他可观察的结果来验证最终状态。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-48\">智能体状态与模型上下文不同\u003C\u002Fh2>\n\u003Cp>长时间运行的任务可能需要无法或不应保留在模型上下文中的状态：任务 ID、检查点、工件、批准、外部对象标识符、重试计数器和工作流状态。\u003C\u002Fp>\n\u003Cp>运行时可以将这种持久状态保存在模型窗口之外，并为下一步重建所需的上下文。这使模型可见的上下文保持专注，同时保持连续性和可恢复性。\u003C\u002Fp>\n\u003Ch2 id=\"section-51\">记忆是可选的，不是智能体的定义\u003C\u002Fh2>\n\u003Cp>如果完整任务适合一次有界运行，智能体可以在没有长期记忆的情况下成功运行。当信息必须在会话、任务或长执行周期中持续存在时，记忆才变得有用。\u003C\u002Fp>\n\u003Cp>RAG、记忆、状态和上下文解决不同的问题。将向量数据库视为“智能体记忆”或将对话历史视为“状态机”通常会隐藏重要的生命周期和权限边界。\u003C\u002Fp>\n\u003Caside class=\"editorjs-referral my-6\">\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context\" class=\"flex flex-col sm:flex-row gap-4 rounded-xl border border-gray-200 dark:border-gray-700 p-4 transition hover:border-primary-500\">\u003Cdiv class=\"min-w-0 flex-1\">\u003Cstrong class=\"block text-lg text-gray-900 dark:text-gray-100\">AI 智能体记忆不是 RAG：如何区分记忆、检索、状态和上下文\u003C\u002Fstrong>\u003Cp class=\"mt-2 text-sm text-gray-600 dark:text-gray-300\">一种实用的架构，将持久记忆、权威应用状态、检索和提供给模型的上下文分开。\u003C\u002Fp>\u003Cspan class=\"mt-3 inline-flex text-sm font-medium text-primary-600 dark:text-primary-400\">阅读记忆架构文章 →\u003C\u002Fspan>\u003C\u002Fdiv>\u003C\u002Fa>\u003C\u002Faside>\n\u003Ch2 id=\"section-55\">上下文工程在智能体中变得动态化\u003C\u002Fh2>\n\u003Cp>每次工具调用都可能产生新的上下文。每一步也可能使之前的信息过时。因此，强大的智能体运行时会在执行过程中重建或整理上下文，而不是无限期地重放所有内容。\u003C\u002Fp>\n\u003Cp>工具定义、任务状态、检索到的证据、观察结果和记忆都在争夺模型的注意力。长时间运行的智能体需要裁剪、压缩或即时加载，以使上下文与当前决策保持相关。\u003C\u002Fp>\n\u003Ch2 id=\"section-58\">读取工具和副作用工具具有不同的风险\u003C\u002Fh2>\n\u003Csection class=\"editorjs-comparison my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">信息访问与外部操作\u003C\u002Fh3>\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left dark:border-gray-700 dark:bg-gray-900\">\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">读取 \u002F 观察\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">写入 \u002F 操作\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">示例\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">主要风险\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">典型控制\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-60\">人在回路是一种控制机制，而不是智能体AI的对立面\u003C\u002Fh2>\n\u003Cp>智能体不会因为人类批准后续步骤而不再是智能体。模型仍然可以自主检查、推理、搜索和准备操作，而运行时在执行前要求人工确认。\u003C\u002Fp>\n\u003Cp>OpenAI当前的智能体安全指南明确建议在较高风险的工作流中对工具操作进行审批。Anthropic同样强调在智能体遇到阻碍或重大决策时设置检查点和人工判断。\u003C\u002Fp>\n\u003Cp>有用的架构问题不是“人类还是自主？”，而是哪些决策可以委托、哪些需要审查、哪些必须保持确定性？\u003C\u002Fp>\n\u003Ch2 id=\"section-64\">智能体需要明确的停止条件\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">停止条件\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">目的\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">成功验证的结果\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">当外部目标状态得到确认时结束。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">最大步数\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">防止失控循环。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">时间预算\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">限制实际执行时间。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">成本\u002F令牌预算\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">限制资源消耗。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">重复动作检测器\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">停止不再有进展的循环。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">权限边界\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">当下一个所需操作不被允许时暂停或停止。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">人工审批检查点\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">在后续执行前等待。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">不可恢复的工具故障\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">升级处理而不是无限重试。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">不确定性阈值\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">当任务无法安全推断时请求澄清。\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-66\">恢复是智能体行为的一部分\u003C\u002Fh2>\n\u003Cp>智能体在会失败的环境中运行：API超时、文件更改、凭证过期、网页移动以及工具返回格式错误的输出。因此，有用的智能体系统需要恢复行为，而不仅仅是顺利路径的工具循环。\u003C\u002Fp>\n\u003Cp>恢复可以包括有限重试、选择另一个工具、重新读取当前状态、询问用户、回滚部分操作或升级给人工。\u003C\u002Fp>\n\u003Cp>重试还需要幂等性意识。重复读取通常风险较低；重复支付或发送消息可能会产生重复的副作用。\u003C\u002Fp>\n\u003Ch2 id=\"section-70\">智能体AI不需要多个智能体\u003C\u002Fh2>\n\u003Cp>具有清晰工具集的单个智能体通常比多智能体架构更简单且更易于评估。当专业化能实质性改善工具隔离、策略隔离、提示清晰度、所有权或追踪可读性时，多个智能体才有用。\u003C\u002Fp>\n\u003Cp>OpenAI当前的编排指南明确建议尽可能从一个智能体开始，只有当契约或所有权边界发生实质性变化时才添加专家。\u003C\u002Fp>\n\u003Cp>多智能体系统带来了新的问题：委托质量、上下文重复、状态冲突、交接语义、身份、成本以及分布式故障处理。\u003C\u002Fp>\n\u003Ch2 id=\"section-74\">智能体协议是互操作性层，而非智能体本身\u003C\u002Fh2>\n\u003Cp>诸如 MCP 和 A2A 之类的协议可以使智能体架构具备互操作性，但它们本身并不会创建智能体循环。MCP 可以暴露工具和资源。A2A 可以连接独立实现的智能体。应用程序仍然需要运行时、授权、状态、评估和领域逻辑。\u003C\u002Fp>\n\u003Cp>这就是为什么协议能力必须与业务权限保持分离。通过 MCP 发现某个工具并不能证明当前主体有权使用它。通过 A2A 接收任务并不能证明远程智能体可以执行所有请求的操作。\u003C\u002Fp>\n\u003Caside class=\"editorjs-referral my-6\">\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fde\u002Fblog\u002Fmcp-vs-a2a-vs-ucp-vs-ap2-vs-a2ui-the-agent-protocol-stack-explained\" class=\"flex flex-col sm:flex-row gap-4 rounded-xl border border-gray-200 dark:border-gray-700 p-4 transition hover:border-primary-500\">\u003Cdiv class=\"min-w-0 flex-1\">\u003Cstrong class=\"block text-lg text-gray-900 dark:text-gray-100\">MCP vs A2A vs UCP vs AP2 vs A2UI：智能体协议栈详解\u003C\u002Fstrong>\u003Cp class=\"mt-2 text-sm text-gray-600 dark:text-gray-300\">一份协议责任映射图，说明为什么工具访问、智能体协作、商务、支付权限和智能体驱动的 UI 属于不同的互操作性边界。\u003C\u002Fp>\u003Cspan class=\"mt-3 inline-flex text-sm font-medium text-primary-600 dark:text-primary-400\">阅读智能体协议栈 →\u003C\u002Fspan>\u003C\u002Fdiv>\u003C\u002Fa>\u003C\u002Faside>\n\u003Ch2 id=\"section-78\">轨迹是智能体可靠性的一部分\u003C\u002Fh2>\n\u003Cp>对于智能体系统而言，仅凭最终答案不足以作为证据，因为智能体可能通过不安全或无效的路径达到正确结果。它可能使用未经授权的工具、跳过必需的检查、重试副作用、依赖过时状态或意外成功。\u003C\u002Fp>\n\u003Cp>因此，评估需要执行轨迹：决策、工具调用、审批、观察、状态变化和最终结果。当前 OpenAI 安全指南建议使用轨迹评分器和评估；Anthropic 的 2026 年智能体评估指南同样将多轮工具轨迹视为一等评估对象。\u003C\u002Fp>\n\u003Cp>更严格的可靠性问题是：智能体是否通过可接受、可恢复且可审计的轨迹达到了可接受的结果？\u003C\u002Fp>\n\u003Caside class=\"editorjs-referral my-6\">\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough\" class=\"flex flex-col sm:flex-row gap-4 rounded-xl border border-gray-200 dark:border-gray-700 p-4 transition hover:border-primary-500\">\u003Cdiv class=\"min-w-0 flex-1\">\u003Cstrong class=\"block text-lg text-gray-900 dark:text-gray-100\">AI 智能体可靠性：为什么最终答案还不够\u003C\u002Fstrong>\u003Cp class=\"mt-2 text-sm text-gray-600 dark:text-gray-300\">为什么生产评估必须检查轨迹、工具使用、状态转换和可恢复性，而不仅仅是最终答案。\u003C\u002Fp>\u003Cspan class=\"mt-3 inline-flex text-sm font-medium text-primary-600 dark:text-primary-400\">阅读可靠性文章 →\u003C\u002Fspan>\u003C\u002Fdiv>\u003C\u002Fa>\u003C\u002Faside>\n\u003Ch2 id=\"section-83\">智能体系统扩大了安全面\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">风险\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">智能体为何会放大它\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">架构应对\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">提示注入\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">不可信内容可能影响未来的工具决策\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">将指令与数据分离；约束工具；尽可能清理或结构化外部输入\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">权限过大\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">推理错误可能变成真实的副作用\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">最小权限、限定范围的凭证、按工具的策略和审批\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">凭证暴露\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">工具可能需要强大的密钥\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">将密钥保留在模型上下文之外；通过可信运行时代理访问\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">混淆代理\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">智能体可能以比请求用户更广泛的权限行事\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">将执行绑定到用户\u002F服务身份，并对重大操作重新授权\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">失控循环\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型反复调用工具而没有进展\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">步数、时间和成本预算以及循环检测\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">状态漂移\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">智能体制定计划后环境发生变化\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">在重大操作前重新读取权威状态\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">间接注入\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">工具\u002F网页\u002F文档内容包含针对模型的指令\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">将外部内容视为不可信数据，而非指令权威\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">审计缺口\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">最终结果无法显示执行了什么\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">追踪工具调用、审批、身份和状态变化\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-85\">智能体可观测性必须跟随循环\u003C\u002Fh2>\n\u003Cp>传统服务可观测性记录请求、延迟和错误。智能体可观测性需要额外的执行模型：哪个智能体处于活动状态、哪个模型版本做出了决策、有哪些上下文可用、选择了哪个工具、发送了什么参数、返回了什么结果以及执行为何停止。\u003C\u002Fp>\n\u003Cp>对于敏感系统，轨迹本身需要访问控制和保留策略，因为提示、工具输出和工件可能包含机密数据。\u003C\u002Fp>\n\u003Ch2 id=\"section-88\">如何评估智能体系统\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">维度\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">问题\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">示例证据\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">任务成功\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">请求的结果是否发生？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">外部状态、测试、业务结果\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">轨迹质量\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">步骤是否可接受？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">工具\u002F操作轨迹\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">工具选择\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">智能体是否选择了适当的能力？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">预期与实际工具调用\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">权限遵守\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">是否保持在允许的权限范围内？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">授权日志和拒绝操作测试\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">状态处理\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">是否使用了当前权威状态？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">新鲜度检查和状态变化测试\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">恢复\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">是否对故障做出了正确响应？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">注入超时\u002F错误场景\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">停止行为\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">是否在正确的点停止？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">步数、循环检测、最终状态证明\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">人工升级\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">在需要审查时是否提出？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">审批\u002F升级轨迹\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">成本\u002F延迟\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">自主性是否值得运营成本？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">令牌、工具调用、持续时间\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">鲁棒性\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">是否能在现实环境变化中存活？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">重复和对抗性试验\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-90\">何时适合使用智能体\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">使用智能体的情况\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">优先使用工作流或简单调用的情况\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">步骤的数量或顺序无法提前可靠确定\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">序列稳定且确定\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">系统必须检查环境并做出适应\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">单次检索加生成步骤就足够\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">根据中间结果，多个工具可能有用\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">一个已知的 API 调用即可解决任务\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">任务受益于迭代验证或修复\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">答案可以直接从提供的上下文中生成\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">失败需要灵活恢复行为\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">失败分支简单，可以显式编码\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">可以在有意义的检查点插入人工审核\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">每一步都是高风险的，无论如何都必须手动控制\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">预期价值足以证明额外的延迟、成本和复杂性是合理的\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">可预测性和低成本比灵活性更重要\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Cp>一个强有力的默认做法是：从能起作用的最简单方案开始，只有当灵活性带来可衡量的价值时，才增加智能体式复杂性。智能体用可预测性、延迟和成本来换取自适应执行。\u003C\u002Fp>\n\u003Ch2 id=\"section-93\">原始实现证据\u003C\u002Fh2>\n\u003Ch3 id=\"section-94\">Aaasaasa AI Client：模型、运行时和权限是分离的\u003C\u002Fh3>\n\u003Cp>Aaasaasa AI Client 明确区分了智能体\u002F客户端、提供者、模型、运行时位置和权限。其架构文档将权限视为核心的工具\u002F工作区策略，而不是模型属性。\u003C\u002Fp>\n\u003Cp>同一个应用可以在没有文件系统或 shell 工具的情况下提供 Direct Chat，而 Codex 运行时则在选定的工作区和权限配置文件下运行。这展示了一个核心的智能体架构边界：即使模型访问仍然可用，改变运行时\u002F工具表面也会改变系统能做什么。\u003C\u002Fp>\n\u003Cp>该仓库还区分了本地 Codex 运行时和模型位置：本地运行时可以调用云端模型。这避免了将“智能体在本地运行”等同于“推理在本地进行”的常见错误。\u003C\u002Fp>\n\u003Cp>该实现禁用了那些审批语义不满足所需权限模型的嵌入式执行路径。这支持了这样一个原则：智能体能力不应仅仅因为底层框架可以执行工具，就绕过运行时授权。\u003C\u002Fp>\n\u003Ch3 id=\"section-99\">Source of Truth Research Engine：有边界的智能体研究阶段\u003C\u002Fh3>\n\u003Cp>Source of Truth Research Engine 使用一个有边界的研究流水线：发现 → 获取 → 提取 → 验证 → 反驳 → 综合。研究任务可以通过 AI 运行时执行，而证据、来源、主张和矛盾则保留在外部持久化存储中。\u003C\u002Fp>\n\u003Cp>这有意比不受约束的自主研究智能体更受控制。这些阶段为下一步应该发生什么样的工作提供了护栏，同时仍然允许在每个有边界的任务内部进行模型驱动的研究。\u003C\u002Fp>\n\u003Cp>这种区分对智能体设计是有用的证据：自主性可以放在结构化的交付包络内，而不是统一应用到整个流程。\u003C\u002Fp>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">已实现的模式\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">智能体架构经验\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Direct Chat 没有操作系统工具\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型可以存在，但不具备智能体执行能力。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Codex 运行时具有工作区权限配置文件\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">工具权限属于运行时策略，而不是模型能力。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">提供者\u002F模型\u002F运行时是分离的概念\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">智能体框架位置和推理位置是相互独立的决策。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">为具备工具能力的运行时设置权限代理\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">能力暴露可以集中管理并受到治理。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">有边界的研究阶段\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">自主性可以在明确的流程边界内运行。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">持久化的主张\u002F证据位于模型上下文之外\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">智能体状态和证据不必只存在于对话历史中。\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Caside class=\"editorjs-callout editorjs-callout--note my-6 rounded-xl border p-5 border-gray-300 bg-gray-50 dark:border-gray-700 dark:bg-gray-900\u002F40\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">证据边界\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">这些项目展示了具体的智能体\u002F运行时、权限和有边界研究模式。它们并不是作为大规模商业自主智能体部署的证据来呈现的。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-105\">常见的智能体式 AI 失败模式\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">失败模式\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">实际失败的是什么\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“智能体”只是一个在提示中列出工具的聊天机器人\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">不存在可靠的运行时循环或工具执行架构\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">将工具支持视为权限\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">能力边界和授权边界被混为一谈\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">智能体信任自己的完成声明\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">结果没有根据外部状态进行验证\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">每个任务都变成多智能体\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">在没有真正的所有权或专业化边界的情况下增加了复杂性\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">将对话历史用作持久状态\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">可恢复性和权威状态变得脆弱\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">智能体盲目重试副作用\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">可能出现重复消息、付款或状态变更\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">没有步骤\u002F成本限制\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">智能体可能无限循环或消耗不受控制的资源\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">将工具输出信任为指令\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">间接提示注入可以重定向行为\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">只评估最终答案是否正确\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">不安全或无效的轨迹仍然不可见\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">将模型升级视为透明\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">工具选择、规划和停止行为可能发生变化\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">一个宽泛工具暴露许多特权操作\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">影响范围扩大，意图变得更难验证\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">存在人工审批，但审核者缺乏上下文\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">审批变成形式化，而不是有效\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-107\">常见误解\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">误解\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">纠正\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“LLM 就是智能体。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型是决策组件；智能体是管理工具、状态和迭代的周围系统。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“工具调用自动意味着智能体式 AI。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">单次有边界的工具调用可能不涉及自适应的多步骤智能体循环。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“智能体必须完全自主。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">智能体式系统可以要求审批，并在狭窄的权限边界内运行。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“智能体需要长期记忆。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">记忆是可选的；许多有用的智能体在没有跨会话记忆的情况下完成有边界的任务。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“智能体必须先创建书面计划。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">规划可以是显式的或隐式的，并且可以一次一步地进行。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“多智能体比单智能体更先进。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">它更复杂；只有当专业化或所有权边界证明其合理时才使用它。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“MCP 创建了一个智能体。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">MCP 暴露工具\u002F资源；运行时仍然需要智能体循环和授权模型。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“本地运行时意味着模型是本地的。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">运行时位置和推理\u002F提供者位置是分离的。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“如果最终结果正确，智能体就正确工作了。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">不安全或未经授权的轨迹仍然可能产生正确结果。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“人工审批消除了自主性。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">审批可以约束选定的操作，而流程的其余部分仍然由模型指导。\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-109\">实用的智能体设计顺序\u003C\u002Fh2>\n\u003Csection class=\"editorjs-process my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">从权限出发向外设计智能体\u003C\u002Fh3>\u003Cdiv class=\"grid grid-cols-1 md:grid-cols-2 xl:grid-cols-3 gap-4\">\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">1\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">1. 定义结果\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">说明什么外部结果或产物能证明任务成功。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">2\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">2. 判断是否真的需要智能体\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">当路径可预测时，优先使用简单调用或确定性工作流。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">3\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">3. 识别状态和事实来源\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">定义哪些系统拥有当前事实、任务进度和业务状态。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">4\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">4. 定义工具面\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">仅暴露任务所需的最小、清晰的能力集合。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">5\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">5. 绑定身份和权限\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">区分用户权限、智能体\u002F运行时权限和工具能力。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">6\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">6. 选择自主性边界\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">明确模型可以动态决定什么，以及什么保持确定性。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">7\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">7. 添加审批检查点\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">在适当情况下，要求对重大或不可逆操作进行审查。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">8\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">8. 定义停止和恢复\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">设置成功证明、预算、超时、重试、升级和循环控制。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">9\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">9. 设计上下文\u002F状态管理\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">将当前状态、记忆、工具观察和持久产物保留在正确的层级中。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">10\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">10. 追踪轨迹\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">记录足够的执行结构，以便调试和审计模型\u002F工具决策。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">11\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">11. 评估现实故障\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">测试过期状态、工具错误、提示注入、模糊请求和环境变化。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">12\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">12. 仅依据证据扩大自主性\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">当评估显示收益足以证明风险合理时，再增加权限或执行范围。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-111\">智能体 AI 架构检查清单\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">问题\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">预期证据\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">什么能证明成功？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">外部结果、产物、测试或权威状态。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">为什么需要智能体？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">路径确实依赖中间观察结果。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">哪些决策由模型驱动？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">明确的自主性边界。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">存在哪些工具？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">小型、有文档、无歧义的能力集合。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">谁可以使用每个工具？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">感知身份和上下文的授权策略。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">哪些操作需要审批？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">基于后果的审查规则。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">任务状态存放在哪里？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">由应用拥有的状态，与瞬态模型上下文分离。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">智能体如何恢复？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">重试、重新读取、回滚、澄清和升级行为。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">它如何停止？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">经验证的完成，加上步骤\u002F时间\u002F成本限制。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">如何保护副作用？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">验证、幂等性、最小权限和确认。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">执行过程能否被重建？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">工具、审批和状态转换轨迹。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">如何评估它？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">结果 + 轨迹 + 鲁棒性测试。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型\u002F运行时更新后会发生什么变化？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">针对工具选择、权限、停止和恢复的回归测试套件。\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-113\">边缘情况和限制\u003C\u002Fh2>\n\u003Cp>有些系统仅在狭义路由意义上具有“智能体”特征：模型选择一个专家或工具，然后工作流的其余部分是确定性的。这仍然可能有用，但不应被描述为等同于长时间运行的自主智能体。\u003C\u002Fp>\n\u003Cp>高度重大的领域可能会刻意限制智能体的自主性。AI 系统可以检查证据、准备建议并填写结构化表单，而人类仍然是唯一被允许提交最终交易的行为者。\u003C\u002Fp>\n\u003Cp>有些环境非常适合智能体，因为反馈是客观的。编码智能体可以运行测试；基础设施智能体可以检查指标；数据智能体可以验证查询结果。反馈较弱的开放式领域需要更谨慎的评估。\u003C\u002Fp>\n\u003Cp>智能体可以完全在本地运行、完全通过托管云服务运行，或采用混合架构。智能体行为描述的是控制流，而不是托管位置。\u003C\u002Fp>\n\u003Cp>“推理”一词不应用作智能体内部过程正确的证明。生产保障应依赖可观察的输入、动作、输出、状态和评估，而不是关于隐藏推理的不可验证主张。\u003C\u002Fp>\n\u003Ch2 id=\"section-119\">什么会改变这个答案？\u003C\u002Fh2>\n\u003Cp>供应商 API 和智能体框架将继续演进，但架构边界是稳定的：模型提出决策，运行时管理循环，工具连接环境，权限约束动作，外部观察决定实际发生了什么。\u003C\u002Fp>\n\u003Cp>随着模型变得更可靠，系统可能安全地委托更长的时间范围或更复杂的恢复行为。随着运行时验证和授权改进，一些审批步骤可能实现自动化。这些是自主性级别的变化，而不是基本责任层的变化。\u003C\u002Fp>\n\u003Cp>推荐架构也会因后果而变化。一个只读取公开来源的研究智能体，与一个写入生产配置或转移资金的智能体，可以容忍不同的控制措施。\u003C\u002Fp>\n\u003Ch2 id=\"section-123\">相关规范知识\u003C\u002Fh2>\n\u003Cp>智能体 AI 位于若干前置层之上：上下文工程决定模型看到什么；事实来源架构决定哪些信息是权威的；检索提供外部证据；运行时架构决定什么可以执行。\u003C\u002Fp>\n\u003Cp>下游节点包括工具调用、MCP、A2A、智能体身份、权限、可审计性、人在回路、编排、记忆和多智能体系统。\u003C\u002Fp>\n\u003Cp>因此，协议栈文章应在基本智能体概念之后阅读：协议标准化智能体周围的边界；它们并不定义智能体行为本身。\u003C\u002Fp>\n\u003Ch2 id=\"section-127\">常见问题\u003C\u002Fh2>\n\u003Csection class=\"editorjs-faq my-6 rounded-xl border border-gray-200 p-5 dark:border-gray-700\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">智能体AI常见问题\u003C\u002Fh3>\u003Cdiv id=\"faq1\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">什么是智能体AI？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">智能体AI是一种AI系统，其中模型可以通过选择行动或工具、观察结果、更新状态并持续进行，直到达到停止条件，从而在多个步骤中追求一个目标。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq2\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">LLM和AI智能体有什么区别？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">LLM从输入产生输出。智能体将模型与运行时、工具、状态、权限、上下文管理和迭代执行循环相结合。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq3\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">工具调用是否使系统成为智能体？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">不一定。单个工具辅助的模型响应可能是有界的且非智能体的。当工具观察驱动自适应多步循环时，智能体行为才会出现。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq4\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">智能体和AI工作流有什么区别？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">工作流通常遵循应用程序代码中定义的流程路径。智能体根据中间观察，对使用哪些步骤和工具拥有更多由模型驱动的控制权。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq5\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">智能体需要记忆吗？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">不需要。长期记忆对于跨会话的持久信息很有用，但许多智能体仅使用当前任务状态和上下文即可完成有界任务。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq6\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">AI智能体需要多个智能体吗？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">不需要。单个智能体通常更简单。当专业化、工具隔离、策略隔离或所有权边界能实质性改善系统时，多智能体系统才是合理的。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq7\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">智能体可以有人工参与吗？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">可以。智能体可以自主执行低风险分析和准备，而运行时在后果性行动之前暂停以等待人工批准。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq8\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">MCP是智能体框架吗？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">不是。MCP是一种用于暴露工具、资源和提示的互操作性协议。智能体运行时可以使用MCP，但仍需要自己的循环、状态、授权和评估。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq9\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">如何知道智能体实际完成了任务？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">在可能的情况下，通过外部状态、测试、工件或权威系统记录来验证成功，而不是信任模型自己的完成声明。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-129\">术语表\u003C\u002Fh2>\n\u003Csection class=\"editorjs-glossary my-6 rounded-xl border border-gray-200 dark:border-gray-700 p-5\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">关键智能体AI术语\u003C\u002Fh3>\u003Cdl>\u003Cdiv id=\"agentic-ai\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">智能体AI\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">AI系统行为，其中模型使用工具、观察和状态动态指导多步执行以实现目标。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"ai-agent\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">AI智能体\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">以模型为中心的系统，具有运行时、工具、状态和执行循环，可以在多个步骤中追求任务。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"agent-loop\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">智能体循环\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">模型决策、工具\u002F行动执行、观察和更新模型决策的重复循环，直到停止。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"runtime-harness\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">运行时\u002F框架\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">管理模型循环、工具、状态、批准、上下文、错误和停止条件的执行层。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"tool\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">工具\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">暴露给模型的能力，用于读取信息、计算、委托或改变外部状态。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"observation\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">观察\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">从工具或环境返回并供应给后续智能体步骤的信息。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"agent-state\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">智能体状态\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">存在于单个模型输出之外的持久任务或执行信息，可能跨步骤或暂停而存活。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"autonomy-boundary\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">自主边界\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">明确定义模型可以动态控制哪些决策和行动的显式限制。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"human-in-the-loop\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">人工参与\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">一种控制模式，其中在AI驱动过程的选定点需要人工审查、输入或批准。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"trajectory\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">轨迹\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">任务请求和最终结果之间相关状态、决策、工具调用、行动和观察的序列。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"idempotency\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">幂等性\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">允许操作重复执行而不会无意中多次应用相同副作用的属性。\u003C\u002Fdd>\u003C\u002Fdiv>\u003C\u002Fdl>\u003C\u002Fsection>\n\u003Ch2 id=\"section-131\">结论\u003C\u002Fh2>\n\u003Cp>智能体AI不仅仅是更聪明的模型或拥有更多工具的聊天机器人。它是一种系统架构，其中模型参与迭代控制循环：决策、行动、观察、更新并继续。\u003C\u002Fp>\n\u003Cp>模型提供灵活的决策制定，但周围的运行时必须拥有执行现实：权限、工具访问、状态、批准、重试、预算、停止条件、追踪和验证。\u003C\u002Fp>\n\u003Cp>因此，最有用的设计原则是：仅在明确的技术和业务边界内将战术选择委托给模型。只有当自主性、权威和证据保持可分离时，智能体能力才能成为生产能力。\u003C\u002Fp>\n\u003Ch2 id=\"section-135\">主要来源和当前指南\u003C\u002Fh2>\n\u003Cp>以下来源支持当前围绕智能体、工作流、循环、工具、编排、安全和评估的架构区分。项目部分是原始实现证据，并明确限定于仓库所展示的内容。\u003C\u002Fp>\n\u003Ca href=\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fagents\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">OpenAI — 智能体\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">当前开发者指南，定义多步工作、工具、状态、编排和智能体执行的运行时选择。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fagents\u002Fdefine-agents\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">OpenAI — 智能体定义\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">当前文档将智能体描述为模型加上指令和可选运行时行为，包括工具、护栏、MCP服务器和交接。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fagents\u002Frunning-agents\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">OpenAI — 运行智能体\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">当前关于智能体循环的文档：模型调用、工具执行或交接、继续和最终停止点。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fagents\u002Forchestration\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">OpenAI — 编排和交接\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">当前关于交接、智能体即工具以及专家智能体何时增加有用所有权或能力边界的指南。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fagent-builder-safety\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">OpenAI — 构建智能体中的安全\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">当前安全指南，涵盖工具批准、提示注入、护栏和基于追踪的评估。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Fbuilding-effective-agents\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">Anthropic — 构建有效智能体\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">工程指南，区分预定义工作流和模型指导的智能体，并描述基于工具的环境反馈循环。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Feffective-context-engineering-for-ai-agents\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">Anthropic — AI智能体的有效上下文工程\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">将智能体实际框架为LLM在循环中自主使用工具，并具有动态即时上下文管理。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Fdemystifying-evals-for-ai-agents\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">Anthropic — 揭秘AI智能体的评估\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">2026年关于评估多轮智能体的指南，这些智能体调用工具、修改状态并适应中间结果。\u003C\u002Fp>\u003C\u002Fa>",{"time":212,"blocks":213,"version":1537},1791481936358,[214,220,228,235,242,250,255,260,265,270,275,280,285,290,319,324,329,334,339,371,376,381,414,419,424,468,473,478,483,490,495,500,505,510,515,543,548,553,558,563,568,573,578,583,588,593,598,604,609,614,619,624,629,634,643,648,653,658,663,687,692,697,702,707,712,747,752,757,762,767,772,777,782,787,792,797,802,810,815,820,825,830,838,843,883,888,893,898,903,951,956,985,990,995,1000,1005,1010,1015,1020,1025,1030,1035,1040,1066,1072,1077,1121,1126,1164,1169,1211,1216,1262,1267,1272,1277,1282,1287,1292,1297,1302,1307,1312,1317,1322,1327,1332,1337,1379,1384,1434,1439,1444,1449,1454,1459,1464,1474,1483,1492,1501,1510,1519,1528],{"id":215,"data":216,"type":218,"tunes":219},"intro",{"text":217},"智能体AI是一种AI系统，其中模型可以通过决定下一步做什么、使用工具或其他能力、观察结果、更新其工作状态并持续进行，直到达到停止条件，从而跨多个步骤追求一个目标。模型本身并不是智能体。一个可用的智能体还需要一个运行时或执行框架，用于管理上下文、工具执行、状态、权限、审批、错误以及决策与观察之间的循环。","paragraph",{},{"id":221,"data":222,"type":226,"tunes":227},"direct",{"body":223,"title":224,"variant":225},"普通的模型调用通常是\u003Cstrong>输入 → 模型 → 输出\u003C\u002Fstrong>。智能体系统更接近\u003Cstrong>目标 → 决策 → 工具\u002F行动 → 观察 → 更新决策 → … → 结果\u003C\u002Fstrong>。\u003Cbr>\u003Cbr>关键区别不在于应用是否使用LLM或函数调用，而在于系统是否让模型对多步骤流程的下一步拥有有意义的控制权，同时由运行时约束模型实际被允许做什么。","直接回答","info","callout",{},{"id":229,"data":230,"type":226,"tunes":234},"boundary",{"body":231,"title":232,"variant":233},"模型可能知道如何调用某个工具。运行时可能暴露该工具。但这两点都不意味着当前用户或智能体有权执行底层的业务操作。\u003Cstrong>工具能力、工具权限和业务权限是相互独立的层次。\u003C\u002Fstrong>","能力不等于权限","warning",{},{"id":236,"data":237,"type":226,"tunes":241},"current",{"body":238,"title":239,"variant":240},"智能体术语在不同供应商和研究社区之间仍存在差异。OpenAI目前围绕多步骤工作、工具、状态和编排来定义智能体运行时。Anthropic的实用区分仍然有用：工作流遵循预定义的代码路径，而智能体动态地指导自己的流程和工具使用。因此，本文将“智能体AI”视为一种架构谱系，而非一个标准化的产品类别。","当前来源说明 — 2026年10月8日","note",{},{"id":243,"data":244,"type":248,"tunes":249},"toc",{"title":245,"maxLevel":246,"minLevel":247},"目录",3,2,"tableOfContents",{},{"id":251,"data":252,"type":42,"tunes":254},"h-meaning",{"text":253,"level":247},"智能体AI的真正含义",{},{"id":256,"data":257,"type":218,"tunes":259},"p-meaning-1",{"text":258},"从普通生成式AI到智能体AI的重要转变在于对流程的控制。普通助手可以使用接收到的上下文来回答问题。而智能体可以决定回答问题需要额外的步骤：检查文件、搜索代码库、查询API、请求澄清、运行测试、更新工单、委派子任务或在操作失败后重试。",{},{"id":261,"data":262,"type":218,"tunes":264},"p-meaning-2",{"text":263},"这并不要求无限的自主权。智能体可以在狭窄的沙箱中运行，受到严格的权限限制，每个有后果的操作都需要审批。如果模型在允许的下一步中动态选择，该系统仍然是智能体的。",{},{"id":266,"data":267,"type":218,"tunes":269},"p-meaning-3",{"text":268},"因此，架构比标签更重要。“智能体”应该描述一种系统行为：在工具、状态和反馈之上进行迭代的模型驱动决策——而不仅仅是一个拥有更大提示词的聊天机器人。",{},{"id":271,"data":272,"type":42,"tunes":274},"h-simple",{"text":273,"level":247},"最简单的例子",{},{"id":276,"data":277,"type":218,"tunes":279},"p-simple-1",{"text":278},"假设一位开发者向AI系统提出：“找出测试套件失败的原因并修复这个bug。”单次模型调用只能根据给定的文本建议可能的原因。",{},{"id":281,"data":282,"type":218,"tunes":284},"p-simple-2",{"text":283},"一个智能体编码系统可以检查代码库、搜索失败的测试、阅读相关文件、提出修改、编辑代码、运行测试、观察失败、修改实现并再次运行测试。",{},{"id":286,"data":287,"type":218,"tunes":289},"p-simple-3",{"text":288},"智能体的部分不仅仅在于存在shell和文件工具，而在于模型可以利用环境反馈来选择下一步，而不是遵循一个完全预定义的序列。",{},{"id":291,"data":292,"type":317,"tunes":318},"simple-loop",{"steps":293,"title":315,"orientation":316},[294,297,300,303,306,309,312],{"label":295,"description":296},"1. 接收目标","用户或上游系统定义目标和相关约束。",{"label":298,"description":299},"2. 构建当前上下文","运行时提供指令、状态、历史、记忆、工具和当前证据。",{"label":301,"description":302},"3. 模型决定下一步","模型可以回答、调用工具、请求信息、委派或停止。",{"label":304,"description":305},"4. 运行时验证请求","权限、模式、审批和策略决定所提议的操作是否可以执行。",{"label":307,"description":308},"5. 执行工具或操作","外部环境发生变化或返回新信息。",{"label":310,"description":311},"6. 观察结果","运行时将结构化的工具输出、错误或状态变化反馈到下一个模型步骤中。",{"label":313,"description":314},"7. 继续或停止","循环重复，直到成功、拒绝、升级、预算限制、超时或其他停止条件。","基本的智能体循环","auto","processFlow",{},{"id":320,"data":321,"type":42,"tunes":323},"h-stops",{"text":322,"level":247},"简单例子止步之处",{},{"id":325,"data":326,"type":218,"tunes":328},"p-stops-1",{"text":327},"并非每个多步骤AI系统都具有同等的智能体程度。一个工作流可能使用多个LLM调用和工具，但每一步都在代码中预先确定。另一个系统可能让模型决定调用哪个工具、以什么顺序、调用多少次以及何时停止。",{},{"id":330,"data":331,"type":218,"tunes":333},"p-stops-2",{"text":332},"两者都可能有用。区别在于控制权在哪里。预定义工作流将更多控制权放在应用代码中。智能体则将更多战术性流程决策移入模型\u002F运行时循环中。",{},{"id":335,"data":336,"type":42,"tunes":338},"h-workflow",{"text":337,"level":247},"智能体与工作流",{},{"id":340,"data":341,"type":369,"tunes":370},"workflow-comparison",{"rows":342,"title":360,"layout":361,"columns":362},[343,348,352,356],{"id":344,"label":345,"values":346},"path","流程路径",[347,347],"",{"id":349,"label":350,"values":351},"tools","工具序列",[347,347],{"id":353,"label":354,"values":355},"strength","优势",[347,347],{"id":357,"label":358,"values":359},"risk","风险",[347,347],"预定义工作流与代理式控制","table",[363,366],{"id":364,"label":365},"workflow","LLM 工作流",{"id":367,"label":368},"agent","代理","comparison",{},{"id":372,"data":373,"type":218,"tunes":375},"p-workflow-1",{"text":374},"Anthropic 明确区分了这两种模式：工作流通过预定义的代码路径编排模型和工具，而代理则让模型动态地指导自己的流程和工具使用。这不是唯一可能的术语，但它是一个有用的架构边界。",{},{"id":377,"data":378,"type":42,"tunes":380},"h-spectrum",{"text":379,"level":247},"代理行为是一个谱系，而非二元标签",{},{"id":382,"data":383,"type":361,"tunes":413},"spectrum-table",{"content":384,"stretched":43,"withHeadings":14},[385,389,393,397,401,405,409],[386,387,388],"层级","示例","谁决定下一步？",[390,391,392],"单次模型调用","总结这份文档","应用程序调用模型一次",[394,395,396],"工具辅助响应","模型在回答前可能使用网络搜索","模型在一次响应中从有限工具中选择",[398,399,400],"结构化工作流","分类 → 检索 → 生成 → 验证","应用程序工作流决定阶段",[402,403,404],"自适应工作流","模型可以在多个分支中选择并重试","应用程序与模型共享控制权",[406,407,408],"代理循环","模型根据观察反复选择工具\u002F动作","模型在运行时约束内指导战术执行",[410,411,412],"长期运行的代理","代理暂停、恢复、管理产物并继续","模型 + 持久化运行时管理不断演进的执行",{},{"id":415,"data":416,"type":218,"tunes":418},"p-spectrum-1",{"text":417},"将上述所有系统都称为“代理”可能会掩盖重要的操作差异。模型对顺序、持续时间和动作的控制越强，运行时隔离、权限、追踪、停止条件和轨迹评估就越重要。",{},{"id":420,"data":421,"type":42,"tunes":423},"h-anatomy",{"text":422,"level":247},"代理系统的最小架构",{},{"id":425,"data":426,"type":361,"tunes":467},"anatomy-table",{"content":427,"stretched":43,"withHeadings":14},[428,431,434,437,440,443,446,449,452,455,458,461,464],[429,430],"组件","职责",[432,433],"目标 \u002F 任务","定义系统试图完成什么。",[435,436],"模型","解释上下文并决定下一个动作或输出。",[438,439],"指令","定义角色、约束、优先级和任务特定策略。",[441,442],"上下文组装器","构建每一步对模型可见的信息。",[444,445],"工具目录","定义模型可以请求的能力。",[447,448],"运行时 \u002F 执行框架","运行循环、执行工具、管理状态并处理停止条件。",[450,451],"授权层","确定当前主体是否被允许执行提议的动作。",[453,454],"状态 \u002F 会话","在轮次或执行步骤之间保留任务进度。",[456,457],"观察通道","将工具结果和环境变化返回到下一个模型步骤。",[459,460],"审批 \u002F 人工控制","在需要审查时暂停有后果的动作。",[462,463],"追踪 \u002F 审计","记录模型调用、工具、转换、审批和失败。",[465,466],"评估","根据验收标准衡量结果和执行轨迹。",{},{"id":469,"data":470,"type":42,"tunes":472},"h-model-agent",{"text":471,"level":247},"模型不是代理",{},{"id":474,"data":475,"type":218,"tunes":477},"p-model-agent-1",{"text":476},"语言模型从输入产生输出。它本身并不拥有文件系统、执行 shell 命令、维护持久任务状态、强制执行权限或自动再次调用自身。",{},{"id":479,"data":480,"type":218,"tunes":482},"p-model-agent-2",{"text":481},"这些能力来自周围的运行时。同一个模型在一个应用程序中可以表现为简单的聊天模型，在另一个应用程序中可以表现为代理循环内的决策引擎。",{},{"id":484,"data":485,"type":226,"tunes":489},"model-agent-rule",{"body":486,"title":487,"variant":488},"\u003Cstrong>模型能力决定可以提出哪些决策。运行时架构决定实际可以发生什么。\u003C\u002Fstrong>","架构规则","success",{},{"id":491,"data":492,"type":42,"tunes":494},"h-tools",{"text":493,"level":247},"工具使用是核心——但仅靠工具使用并不能构成代理",{},{"id":496,"data":497,"type":218,"tunes":499},"p-tools-1",{"text":498},"工具让模型获取信息并影响外部系统。示例包括数据库读取、文件操作、shell 执行、网络搜索、浏览器控制、API 调用、工单更新或委托专业代理。",{},{"id":501,"data":502,"type":218,"tunes":504},"p-tools-2",{"text":503},"单次模型调用可以使用一个工具，但仍然是有界的工具辅助响应，而不是长期运行的代理。当工具观察结果馈入一个自适应循环，模型在其中选择下一步做什么时，代理行为就出现了。",{},{"id":506,"data":507,"type":218,"tunes":509},"p-tools-3",{"text":508},"工具设计很重要，因为工具是模型推理与外部现实之间的契约。模糊或重叠的工具会造成路由错误；大型非结构化输出会污染上下文；广泛的副作用工具会扩大影响范围。",{},{"id":511,"data":512,"type":42,"tunes":514},"h-capability-permission",{"text":513,"level":247},"工具能力、权限和授权是不同的",{},{"id":516,"data":517,"type":361,"tunes":542},"permission-table",{"content":518,"stretched":43,"withHeadings":14},[519,521,524,527,530,533,536,539],[386,520],"问题",[522,523],"能力","此运行时在技术上能否执行该操作？",[525,526],"工具暴露","该能力是否对此代理可用？",[528,529],"权限","在当前策略下，此代理\u002F会话是否可以使用它？",[531,532],"用户授权","请求主体是否被允许导致此操作？",[534,535],"业务授权","该操作在领域规则、审批和限制下是否有效？",[537,538],"执行","该操作是否实际发生？",[540,541],"审计","系统能否证明谁请求、批准并执行了它？",{},{"id":544,"data":545,"type":218,"tunes":547},"p-permission-1",{"text":546},"这些层级在原型中经常被合并。模型看到一个退款工具，因此似乎能够发放退款。在生产环境中，工具仍应独立于模型的请求，验证账户、用户、交易、金额、策略和审批条件。",{},{"id":549,"data":550,"type":42,"tunes":552},"h-runtime",{"text":551,"level":247},"运行时或执行框架是实际的执行系统",{},{"id":554,"data":555,"type":218,"tunes":557},"p-runtime-1",{"text":556},"OpenAI 当前的智能体文档明确区分了运行时。不同的运行时可以在不同位置管理编排、状态、工具、沙箱和执行，而模型只是系统的一部分。",{},{"id":559,"data":560,"type":218,"tunes":562},"p-runtime-2",{"text":561},"Agents SDK 描述了一个循环：反复调用当前模型，检查输出，执行请求的工具或交接，并持续进行，直到模型返回最终答案或其他真正的停止点。",{},{"id":564,"data":565,"type":218,"tunes":567},"p-runtime-3",{"text":566},"这意味着智能体架构决策包括：编排在哪里运行、状态存储在哪里、谁执行工具、哪个沙箱包含副作用，以及谁负责重试、超时和可恢复性。",{},{"id":569,"data":570,"type":42,"tunes":572},"h-planning",{"text":571,"level":247},"规划有用，但不需要显式计划",{},{"id":574,"data":575,"type":218,"tunes":577},"p-planning-1",{"text":576},"智能体常被描述为“规划”的系统。在实践中，规划可以是显式的或隐式的。智能体可以先产生一个可见的多步骤计划，也可以一次选择一个下一步行动，并在每次观察后修正。",{},{"id":579,"data":580,"type":218,"tunes":582},"p-planning-2",{"text":581},"对于高度不确定的任务，短视界规划可能更安全，因为环境可能使长期计划失效。架构要求是能够根据目标、当前状态和新证据选择并修正行动。",{},{"id":584,"data":585,"type":42,"tunes":587},"h-feedback",{"text":586,"level":247},"环境反馈使循环变得有用",{},{"id":589,"data":590,"type":218,"tunes":592},"p-feedback-1",{"text":591},"当智能体能够观察其行动是否有效时，它才具有操作意义。工具输出、测试结果、API 响应、文件系统状态、浏览器状态和应用程序记录提供了外部证据，系统可以用这些证据来修正下一个决策。",{},{"id":594,"data":595,"type":218,"tunes":597},"p-feedback-2",{"text":596},"Anthropic 的智能体指南强调了这种反馈循环：智能体使用工具，从环境中获取真实情况，评估进展，并继续或请求人工输入。",{},{"id":599,"data":600,"type":226,"tunes":603},"feedback-rule",{"body":601,"title":602,"variant":233},"智能体说“任务已完成”并不能证明完成。在可能的情况下，通过外部系统、测试、文件、交易记录或其他可观察的结果来验证最终状态。","自我报告不是环境证明",{},{"id":605,"data":606,"type":42,"tunes":608},"h-state",{"text":607,"level":247},"智能体状态与模型上下文不同",{},{"id":610,"data":611,"type":218,"tunes":613},"p-state-1",{"text":612},"长时间运行的任务可能需要无法或不应保留在模型上下文中的状态：任务 ID、检查点、工件、批准、外部对象标识符、重试计数器和工作流状态。",{},{"id":615,"data":616,"type":218,"tunes":618},"p-state-2",{"text":617},"运行时可以将这种持久状态保存在模型窗口之外，并为下一步重建所需的上下文。这使模型可见的上下文保持专注，同时保持连续性和可恢复性。",{},{"id":620,"data":621,"type":42,"tunes":623},"h-memory",{"text":622,"level":247},"记忆是可选的，不是智能体的定义",{},{"id":625,"data":626,"type":218,"tunes":628},"p-memory-1",{"text":627},"如果完整任务适合一次有界运行，智能体可以在没有长期记忆的情况下成功运行。当信息必须在会话、任务或长执行周期中持续存在时，记忆才变得有用。",{},{"id":630,"data":631,"type":218,"tunes":633},"p-memory-2",{"text":632},"RAG、记忆、状态和上下文解决不同的问题。将向量数据库视为“智能体记忆”或将对话历史视为“状态机”通常会隐藏重要的生命周期和权限边界。",{},{"id":635,"data":636,"type":641,"tunes":642},"ref-memory",{"url":637,"title":638,"excerpt":639,"ctaLabel":640},"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context","AI 智能体记忆不是 RAG：如何区分记忆、检索、状态和上下文","一种实用的架构，将持久记忆、权威应用状态、检索和提供给模型的上下文分开。","阅读记忆架构文章","referralArticle",{},{"id":644,"data":645,"type":42,"tunes":647},"h-context",{"text":646,"level":247},"上下文工程在智能体中变得动态化",{},{"id":649,"data":650,"type":218,"tunes":652},"p-context-1",{"text":651},"每次工具调用都可能产生新的上下文。每一步也可能使之前的信息过时。因此，强大的智能体运行时会在执行过程中重建或整理上下文，而不是无限期地重放所有内容。",{},{"id":654,"data":655,"type":218,"tunes":657},"p-context-2",{"text":656},"工具定义、任务状态、检索到的证据、观察结果和记忆都在争夺模型的注意力。长时间运行的智能体需要裁剪、压缩或即时加载，以使上下文与当前决策保持相关。",{},{"id":659,"data":660,"type":42,"tunes":662},"h-sideeffects",{"text":661,"level":247},"读取工具和副作用工具具有不同的风险",{},{"id":664,"data":665,"type":369,"tunes":686},"sideeffect-comparison",{"rows":666,"title":678,"layout":361,"columns":679},[667,670,674],{"id":668,"label":387,"values":669},"examples",[347,347],{"id":671,"label":672,"values":673},"mainrisk","主要风险",[347,347],{"id":675,"label":676,"values":677},"control","典型控制",[347,347],"信息访问与外部操作",[680,683],{"id":681,"label":682},"read","读取 \u002F 观察",{"id":684,"label":685},"write","写入 \u002F 操作",{},{"id":688,"data":689,"type":42,"tunes":691},"h-approval",{"text":690,"level":247},"人在回路是一种控制机制，而不是智能体AI的对立面",{},{"id":693,"data":694,"type":218,"tunes":696},"p-approval-1",{"text":695},"智能体不会因为人类批准后续步骤而不再是智能体。模型仍然可以自主检查、推理、搜索和准备操作，而运行时在执行前要求人工确认。",{},{"id":698,"data":699,"type":218,"tunes":701},"p-approval-2",{"text":700},"OpenAI当前的智能体安全指南明确建议在较高风险的工作流中对工具操作进行审批。Anthropic同样强调在智能体遇到阻碍或重大决策时设置检查点和人工判断。",{},{"id":703,"data":704,"type":218,"tunes":706},"p-approval-3",{"text":705},"有用的架构问题不是“人类还是自主？”，而是哪些决策可以委托、哪些需要审查、哪些必须保持确定性？",{},{"id":708,"data":709,"type":42,"tunes":711},"h-stopping",{"text":710,"level":247},"智能体需要明确的停止条件",{},{"id":713,"data":714,"type":361,"tunes":746},"stop-table",{"content":715,"stretched":43,"withHeadings":14},[716,719,722,725,728,731,734,737,740,743],[717,718],"停止条件","目的",[720,721],"成功验证的结果","当外部目标状态得到确认时结束。",[723,724],"最大步数","防止失控循环。",[726,727],"时间预算","限制实际执行时间。",[729,730],"成本\u002F令牌预算","限制资源消耗。",[732,733],"重复动作检测器","停止不再有进展的循环。",[735,736],"权限边界","当下一个所需操作不被允许时暂停或停止。",[738,739],"人工审批检查点","在后续执行前等待。",[741,742],"不可恢复的工具故障","升级处理而不是无限重试。",[744,745],"不确定性阈值","当任务无法安全推断时请求澄清。",{},{"id":748,"data":749,"type":42,"tunes":751},"h-errors",{"text":750,"level":247},"恢复是智能体行为的一部分",{},{"id":753,"data":754,"type":218,"tunes":756},"p-errors-1",{"text":755},"智能体在会失败的环境中运行：API超时、文件更改、凭证过期、网页移动以及工具返回格式错误的输出。因此，有用的智能体系统需要恢复行为，而不仅仅是顺利路径的工具循环。",{},{"id":758,"data":759,"type":218,"tunes":761},"p-errors-2",{"text":760},"恢复可以包括有限重试、选择另一个工具、重新读取当前状态、询问用户、回滚部分操作或升级给人工。",{},{"id":763,"data":764,"type":218,"tunes":766},"p-errors-3",{"text":765},"重试还需要幂等性意识。重复读取通常风险较低；重复支付或发送消息可能会产生重复的副作用。",{},{"id":768,"data":769,"type":42,"tunes":771},"h-single-multi",{"text":770,"level":247},"智能体AI不需要多个智能体",{},{"id":773,"data":774,"type":218,"tunes":776},"p-multi-1",{"text":775},"具有清晰工具集的单个智能体通常比多智能体架构更简单且更易于评估。当专业化能实质性改善工具隔离、策略隔离、提示清晰度、所有权或追踪可读性时，多个智能体才有用。",{},{"id":778,"data":779,"type":218,"tunes":781},"p-multi-2",{"text":780},"OpenAI当前的编排指南明确建议尽可能从一个智能体开始，只有当契约或所有权边界发生实质性变化时才添加专家。",{},{"id":783,"data":784,"type":218,"tunes":786},"p-multi-3",{"text":785},"多智能体系统带来了新的问题：委托质量、上下文重复、状态冲突、交接语义、身份、成本以及分布式故障处理。",{},{"id":788,"data":789,"type":42,"tunes":791},"h-protocols",{"text":790,"level":247},"智能体协议是互操作性层，而非智能体本身",{},{"id":793,"data":794,"type":218,"tunes":796},"p-protocols-1",{"text":795},"诸如 MCP 和 A2A 之类的协议可以使智能体架构具备互操作性，但它们本身并不会创建智能体循环。MCP 可以暴露工具和资源。A2A 可以连接独立实现的智能体。应用程序仍然需要运行时、授权、状态、评估和领域逻辑。",{},{"id":798,"data":799,"type":218,"tunes":801},"p-protocols-2",{"text":800},"这就是为什么协议能力必须与业务权限保持分离。通过 MCP 发现某个工具并不能证明当前主体有权使用它。通过 A2A 接收任务并不能证明远程智能体可以执行所有请求的操作。",{},{"id":803,"data":804,"type":641,"tunes":809},"ref-protocols",{"url":805,"title":806,"excerpt":807,"ctaLabel":808},"https:\u002F\u002Fstajic.de\u002Fde\u002Fblog\u002Fmcp-vs-a2a-vs-ucp-vs-ap2-vs-a2ui-the-agent-protocol-stack-explained","MCP vs A2A vs UCP vs AP2 vs A2UI：智能体协议栈详解","一份协议责任映射图，说明为什么工具访问、智能体协作、商务、支付权限和智能体驱动的 UI 属于不同的互操作性边界。","阅读智能体协议栈",{},{"id":811,"data":812,"type":42,"tunes":814},"h-reliability",{"text":813,"level":247},"轨迹是智能体可靠性的一部分",{},{"id":816,"data":817,"type":218,"tunes":819},"p-rel-1",{"text":818},"对于智能体系统而言，仅凭最终答案不足以作为证据，因为智能体可能通过不安全或无效的路径达到正确结果。它可能使用未经授权的工具、跳过必需的检查、重试副作用、依赖过时状态或意外成功。",{},{"id":821,"data":822,"type":218,"tunes":824},"p-rel-2",{"text":823},"因此，评估需要执行轨迹：决策、工具调用、审批、观察、状态变化和最终结果。当前 OpenAI 安全指南建议使用轨迹评分器和评估；Anthropic 的 2026 年智能体评估指南同样将多轮工具轨迹视为一等评估对象。",{},{"id":826,"data":827,"type":218,"tunes":829},"p-rel-3",{"text":828},"更严格的可靠性问题是：智能体是否通过可接受、可恢复且可审计的轨迹达到了可接受的结果？",{},{"id":831,"data":832,"type":641,"tunes":837},"ref-reliability",{"url":833,"title":834,"excerpt":835,"ctaLabel":836},"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough","AI 智能体可靠性：为什么最终答案还不够","为什么生产评估必须检查轨迹、工具使用、状态转换和可恢复性，而不仅仅是最终答案。","阅读可靠性文章",{},{"id":839,"data":840,"type":42,"tunes":842},"h-security",{"text":841,"level":247},"智能体系统扩大了安全面",{},{"id":844,"data":845,"type":361,"tunes":882},"security-table",{"content":846,"stretched":43,"withHeadings":14},[847,850,854,858,862,866,870,874,878],[358,848,849],"智能体为何会放大它","架构应对",[851,852,853],"提示注入","不可信内容可能影响未来的工具决策","将指令与数据分离；约束工具；尽可能清理或结构化外部输入",[855,856,857],"权限过大","推理错误可能变成真实的副作用","最小权限、限定范围的凭证、按工具的策略和审批",[859,860,861],"凭证暴露","工具可能需要强大的密钥","将密钥保留在模型上下文之外；通过可信运行时代理访问",[863,864,865],"混淆代理","智能体可能以比请求用户更广泛的权限行事","将执行绑定到用户\u002F服务身份，并对重大操作重新授权",[867,868,869],"失控循环","模型反复调用工具而没有进展","步数、时间和成本预算以及循环检测",[871,872,873],"状态漂移","智能体制定计划后环境发生变化","在重大操作前重新读取权威状态",[875,876,877],"间接注入","工具\u002F网页\u002F文档内容包含针对模型的指令","将外部内容视为不可信数据，而非指令权威",[879,880,881],"审计缺口","最终结果无法显示执行了什么","追踪工具调用、审批、身份和状态变化",{},{"id":884,"data":885,"type":42,"tunes":887},"h-observability",{"text":886,"level":247},"智能体可观测性必须跟随循环",{},{"id":889,"data":890,"type":218,"tunes":892},"p-obs-1",{"text":891},"传统服务可观测性记录请求、延迟和错误。智能体可观测性需要额外的执行模型：哪个智能体处于活动状态、哪个模型版本做出了决策、有哪些上下文可用、选择了哪个工具、发送了什么参数、返回了什么结果以及执行为何停止。",{},{"id":894,"data":895,"type":218,"tunes":897},"p-obs-2",{"text":896},"对于敏感系统，轨迹本身需要访问控制和保留策略，因为提示、工具输出和工件可能包含机密数据。",{},{"id":899,"data":900,"type":42,"tunes":902},"h-eval",{"text":901,"level":247},"如何评估智能体系统",{},{"id":904,"data":905,"type":361,"tunes":950},"eval-table",{"content":906,"stretched":43,"withHeadings":14},[907,910,914,918,922,926,930,934,938,942,946],[908,520,909],"维度","示例证据",[911,912,913],"任务成功","请求的结果是否发生？","外部状态、测试、业务结果",[915,916,917],"轨迹质量","步骤是否可接受？","工具\u002F操作轨迹",[919,920,921],"工具选择","智能体是否选择了适当的能力？","预期与实际工具调用",[923,924,925],"权限遵守","是否保持在允许的权限范围内？","授权日志和拒绝操作测试",[927,928,929],"状态处理","是否使用了当前权威状态？","新鲜度检查和状态变化测试",[931,932,933],"恢复","是否对故障做出了正确响应？","注入超时\u002F错误场景",[935,936,937],"停止行为","是否在正确的点停止？","步数、循环检测、最终状态证明",[939,940,941],"人工升级","在需要审查时是否提出？","审批\u002F升级轨迹",[943,944,945],"成本\u002F延迟","自主性是否值得运营成本？","令牌、工具调用、持续时间",[947,948,949],"鲁棒性","是否能在现实环境变化中存活？","重复和对抗性试验",{},{"id":952,"data":953,"type":42,"tunes":955},"h-use",{"text":954,"level":247},"何时适合使用智能体",{},{"id":957,"data":958,"type":361,"tunes":984},"use-table",{"content":959,"stretched":43,"withHeadings":14},[960,963,966,969,972,975,978,981],[961,962],"使用智能体的情况","优先使用工作流或简单调用的情况",[964,965],"步骤的数量或顺序无法提前可靠确定","序列稳定且确定",[967,968],"系统必须检查环境并做出适应","单次检索加生成步骤就足够",[970,971],"根据中间结果，多个工具可能有用","一个已知的 API 调用即可解决任务",[973,974],"任务受益于迭代验证或修复","答案可以直接从提供的上下文中生成",[976,977],"失败需要灵活恢复行为","失败分支简单，可以显式编码",[979,980],"可以在有意义的检查点插入人工审核","每一步都是高风险的，无论如何都必须手动控制",[982,983],"预期价值足以证明额外的延迟、成本和复杂性是合理的","可预测性和低成本比灵活性更重要",{},{"id":986,"data":987,"type":218,"tunes":989},"p-use-1",{"text":988},"一个强有力的默认做法是：从能起作用的最简单方案开始，只有当灵活性带来可衡量的价值时，才增加智能体式复杂性。智能体用可预测性、延迟和成本来换取自适应执行。",{},{"id":991,"data":992,"type":42,"tunes":994},"h-implementation",{"text":993,"level":247},"原始实现证据",{},{"id":996,"data":997,"type":42,"tunes":999},"h-client",{"text":998,"level":246},"Aaasaasa AI Client：模型、运行时和权限是分离的",{},{"id":1001,"data":1002,"type":218,"tunes":1004},"p-client-1",{"text":1003},"Aaasaasa AI Client 明确区分了智能体\u002F客户端、提供者、模型、运行时位置和权限。其架构文档将权限视为核心的工具\u002F工作区策略，而不是模型属性。",{},{"id":1006,"data":1007,"type":218,"tunes":1009},"p-client-2",{"text":1008},"同一个应用可以在没有文件系统或 shell 工具的情况下提供 Direct Chat，而 Codex 运行时则在选定的工作区和权限配置文件下运行。这展示了一个核心的智能体架构边界：即使模型访问仍然可用，改变运行时\u002F工具表面也会改变系统能做什么。",{},{"id":1011,"data":1012,"type":218,"tunes":1014},"p-client-3",{"text":1013},"该仓库还区分了本地 Codex 运行时和模型位置：本地运行时可以调用云端模型。这避免了将“智能体在本地运行”等同于“推理在本地进行”的常见错误。",{},{"id":1016,"data":1017,"type":218,"tunes":1019},"p-client-4",{"text":1018},"该实现禁用了那些审批语义不满足所需权限模型的嵌入式执行路径。这支持了这样一个原则：智能体能力不应仅仅因为底层框架可以执行工具，就绕过运行时授权。",{},{"id":1021,"data":1022,"type":42,"tunes":1024},"h-sot-agent",{"text":1023,"level":246},"Source of Truth Research Engine：有边界的智能体研究阶段",{},{"id":1026,"data":1027,"type":218,"tunes":1029},"p-sot-1",{"text":1028},"Source of Truth Research Engine 使用一个有边界的研究流水线：发现 → 获取 → 提取 → 验证 → 反驳 → 综合。研究任务可以通过 AI 运行时执行，而证据、来源、主张和矛盾则保留在外部持久化存储中。",{},{"id":1031,"data":1032,"type":218,"tunes":1034},"p-sot-2",{"text":1033},"这有意比不受约束的自主研究智能体更受控制。这些阶段为下一步应该发生什么样的工作提供了护栏，同时仍然允许在每个有边界的任务内部进行模型驱动的研究。",{},{"id":1036,"data":1037,"type":218,"tunes":1039},"p-sot-3",{"text":1038},"这种区分对智能体设计是有用的证据：自主性可以放在结构化的交付包络内，而不是统一应用到整个流程。",{},{"id":1041,"data":1042,"type":361,"tunes":1065},"impl-table",{"content":1043,"stretched":43,"withHeadings":14},[1044,1047,1050,1053,1056,1059,1062],[1045,1046],"已实现的模式","智能体架构经验",[1048,1049],"Direct Chat 没有操作系统工具","模型可以存在，但不具备智能体执行能力。",[1051,1052],"Codex 运行时具有工作区权限配置文件","工具权限属于运行时策略，而不是模型能力。",[1054,1055],"提供者\u002F模型\u002F运行时是分离的概念","智能体框架位置和推理位置是相互独立的决策。",[1057,1058],"为具备工具能力的运行时设置权限代理","能力暴露可以集中管理并受到治理。",[1060,1061],"有边界的研究阶段","自主性可以在明确的流程边界内运行。",[1063,1064],"持久化的主张\u002F证据位于模型上下文之外","智能体状态和证据不必只存在于对话历史中。",{},{"id":1067,"data":1068,"type":226,"tunes":1071},"impl-boundary",{"body":1069,"title":1070,"variant":240},"这些项目展示了具体的智能体\u002F运行时、权限和有边界研究模式。它们并不是作为大规模商业自主智能体部署的证据来呈现的。","证据边界",{},{"id":1073,"data":1074,"type":42,"tunes":1076},"h-failures",{"text":1075,"level":247},"常见的智能体式 AI 失败模式",{},{"id":1078,"data":1079,"type":361,"tunes":1120},"failure-table",{"content":1080,"stretched":43,"withHeadings":14},[1081,1084,1087,1090,1093,1096,1099,1102,1105,1108,1111,1114,1117],[1082,1083],"失败模式","实际失败的是什么",[1085,1086],"“智能体”只是一个在提示中列出工具的聊天机器人","不存在可靠的运行时循环或工具执行架构",[1088,1089],"将工具支持视为权限","能力边界和授权边界被混为一谈",[1091,1092],"智能体信任自己的完成声明","结果没有根据外部状态进行验证",[1094,1095],"每个任务都变成多智能体","在没有真正的所有权或专业化边界的情况下增加了复杂性",[1097,1098],"将对话历史用作持久状态","可恢复性和权威状态变得脆弱",[1100,1101],"智能体盲目重试副作用","可能出现重复消息、付款或状态变更",[1103,1104],"没有步骤\u002F成本限制","智能体可能无限循环或消耗不受控制的资源",[1106,1107],"将工具输出信任为指令","间接提示注入可以重定向行为",[1109,1110],"只评估最终答案是否正确","不安全或无效的轨迹仍然不可见",[1112,1113],"将模型升级视为透明","工具选择、规划和停止行为可能发生变化",[1115,1116],"一个宽泛工具暴露许多特权操作","影响范围扩大，意图变得更难验证",[1118,1119],"存在人工审批，但审核者缺乏上下文","审批变成形式化，而不是有效",{},{"id":1122,"data":1123,"type":42,"tunes":1125},"h-misconceptions",{"text":1124,"level":247},"常见误解",{},{"id":1127,"data":1128,"type":361,"tunes":1163},"misconceptions-table",{"content":1129,"stretched":43,"withHeadings":14},[1130,1133,1136,1139,1142,1145,1148,1151,1154,1157,1160],[1131,1132],"误解","纠正",[1134,1135],"“LLM 就是智能体。”","模型是决策组件；智能体是管理工具、状态和迭代的周围系统。",[1137,1138],"“工具调用自动意味着智能体式 AI。”","单次有边界的工具调用可能不涉及自适应的多步骤智能体循环。",[1140,1141],"“智能体必须完全自主。”","智能体式系统可以要求审批，并在狭窄的权限边界内运行。",[1143,1144],"“智能体需要长期记忆。”","记忆是可选的；许多有用的智能体在没有跨会话记忆的情况下完成有边界的任务。",[1146,1147],"“智能体必须先创建书面计划。”","规划可以是显式的或隐式的，并且可以一次一步地进行。",[1149,1150],"“多智能体比单智能体更先进。”","它更复杂；只有当专业化或所有权边界证明其合理时才使用它。",[1152,1153],"“MCP 创建了一个智能体。”","MCP 暴露工具\u002F资源；运行时仍然需要智能体循环和授权模型。",[1155,1156],"“本地运行时意味着模型是本地的。”","运行时位置和推理\u002F提供者位置是分离的。",[1158,1159],"“如果最终结果正确，智能体就正确工作了。”","不安全或未经授权的轨迹仍然可能产生正确结果。",[1161,1162],"“人工审批消除了自主性。”","审批可以约束选定的操作，而流程的其余部分仍然由模型指导。",{},{"id":1165,"data":1166,"type":42,"tunes":1168},"h-design",{"text":1167,"level":247},"实用的智能体设计顺序",{},{"id":1170,"data":1171,"type":317,"tunes":1210},"design-flow",{"steps":1172,"title":1209,"orientation":316},[1173,1176,1179,1182,1185,1188,1191,1194,1197,1200,1203,1206],{"label":1174,"description":1175},"1. 定义结果","说明什么外部结果或产物能证明任务成功。",{"label":1177,"description":1178},"2. 判断是否真的需要智能体","当路径可预测时，优先使用简单调用或确定性工作流。",{"label":1180,"description":1181},"3. 识别状态和事实来源","定义哪些系统拥有当前事实、任务进度和业务状态。",{"label":1183,"description":1184},"4. 定义工具面","仅暴露任务所需的最小、清晰的能力集合。",{"label":1186,"description":1187},"5. 绑定身份和权限","区分用户权限、智能体\u002F运行时权限和工具能力。",{"label":1189,"description":1190},"6. 选择自主性边界","明确模型可以动态决定什么，以及什么保持确定性。",{"label":1192,"description":1193},"7. 添加审批检查点","在适当情况下，要求对重大或不可逆操作进行审查。",{"label":1195,"description":1196},"8. 定义停止和恢复","设置成功证明、预算、超时、重试、升级和循环控制。",{"label":1198,"description":1199},"9. 设计上下文\u002F状态管理","将当前状态、记忆、工具观察和持久产物保留在正确的层级中。",{"label":1201,"description":1202},"10. 追踪轨迹","记录足够的执行结构，以便调试和审计模型\u002F工具决策。",{"label":1204,"description":1205},"11. 评估现实故障","测试过期状态、工具错误、提示注入、模糊请求和环境变化。",{"label":1207,"description":1208},"12. 仅依据证据扩大自主性","当评估显示收益足以证明风险合理时，再增加权限或执行范围。","从权限出发向外设计智能体",{},{"id":1212,"data":1213,"type":42,"tunes":1215},"h-checklist",{"text":1214,"level":247},"智能体 AI 架构检查清单",{},{"id":1217,"data":1218,"type":361,"tunes":1261},"checklist-table",{"content":1219,"stretched":43,"withHeadings":14},[1220,1222,1225,1228,1231,1234,1237,1240,1243,1246,1249,1252,1255,1258],[520,1221],"预期证据",[1223,1224],"什么能证明成功？","外部结果、产物、测试或权威状态。",[1226,1227],"为什么需要智能体？","路径确实依赖中间观察结果。",[1229,1230],"哪些决策由模型驱动？","明确的自主性边界。",[1232,1233],"存在哪些工具？","小型、有文档、无歧义的能力集合。",[1235,1236],"谁可以使用每个工具？","感知身份和上下文的授权策略。",[1238,1239],"哪些操作需要审批？","基于后果的审查规则。",[1241,1242],"任务状态存放在哪里？","由应用拥有的状态，与瞬态模型上下文分离。",[1244,1245],"智能体如何恢复？","重试、重新读取、回滚、澄清和升级行为。",[1247,1248],"它如何停止？","经验证的完成，加上步骤\u002F时间\u002F成本限制。",[1250,1251],"如何保护副作用？","验证、幂等性、最小权限和确认。",[1253,1254],"执行过程能否被重建？","工具、审批和状态转换轨迹。",[1256,1257],"如何评估它？","结果 + 轨迹 + 鲁棒性测试。",[1259,1260],"模型\u002F运行时更新后会发生什么变化？","针对工具选择、权限、停止和恢复的回归测试套件。",{},{"id":1263,"data":1264,"type":42,"tunes":1266},"h-edge",{"text":1265,"level":247},"边缘情况和限制",{},{"id":1268,"data":1269,"type":218,"tunes":1271},"p-edge-1",{"text":1270},"有些系统仅在狭义路由意义上具有“智能体”特征：模型选择一个专家或工具，然后工作流的其余部分是确定性的。这仍然可能有用，但不应被描述为等同于长时间运行的自主智能体。",{},{"id":1273,"data":1274,"type":218,"tunes":1276},"p-edge-2",{"text":1275},"高度重大的领域可能会刻意限制智能体的自主性。AI 系统可以检查证据、准备建议并填写结构化表单，而人类仍然是唯一被允许提交最终交易的行为者。",{},{"id":1278,"data":1279,"type":218,"tunes":1281},"p-edge-3",{"text":1280},"有些环境非常适合智能体，因为反馈是客观的。编码智能体可以运行测试；基础设施智能体可以检查指标；数据智能体可以验证查询结果。反馈较弱的开放式领域需要更谨慎的评估。",{},{"id":1283,"data":1284,"type":218,"tunes":1286},"p-edge-4",{"text":1285},"智能体可以完全在本地运行、完全通过托管云服务运行，或采用混合架构。智能体行为描述的是控制流，而不是托管位置。",{},{"id":1288,"data":1289,"type":218,"tunes":1291},"p-edge-5",{"text":1290},"“推理”一词不应用作智能体内部过程正确的证明。生产保障应依赖可观察的输入、动作、输出、状态和评估，而不是关于隐藏推理的不可验证主张。",{},{"id":1293,"data":1294,"type":42,"tunes":1296},"h-change",{"text":1295,"level":247},"什么会改变这个答案？",{},{"id":1298,"data":1299,"type":218,"tunes":1301},"p-change-1",{"text":1300},"供应商 API 和智能体框架将继续演进，但架构边界是稳定的：模型提出决策，运行时管理循环，工具连接环境，权限约束动作，外部观察决定实际发生了什么。",{},{"id":1303,"data":1304,"type":218,"tunes":1306},"p-change-2",{"text":1305},"随着模型变得更可靠，系统可能安全地委托更长的时间范围或更复杂的恢复行为。随着运行时验证和授权改进，一些审批步骤可能实现自动化。这些是自主性级别的变化，而不是基本责任层的变化。",{},{"id":1308,"data":1309,"type":218,"tunes":1311},"p-change-3",{"text":1310},"推荐架构也会因后果而变化。一个只读取公开来源的研究智能体，与一个写入生产配置或转移资金的智能体，可以容忍不同的控制措施。",{},{"id":1313,"data":1314,"type":42,"tunes":1316},"h-related",{"text":1315,"level":247},"相关规范知识",{},{"id":1318,"data":1319,"type":218,"tunes":1321},"p-related-1",{"text":1320},"智能体 AI 位于若干前置层之上：上下文工程决定模型看到什么；事实来源架构决定哪些信息是权威的；检索提供外部证据；运行时架构决定什么可以执行。",{},{"id":1323,"data":1324,"type":218,"tunes":1326},"p-related-2",{"text":1325},"下游节点包括工具调用、MCP、A2A、智能体身份、权限、可审计性、人在回路、编排、记忆和多智能体系统。",{},{"id":1328,"data":1329,"type":218,"tunes":1331},"p-related-3",{"text":1330},"因此，协议栈文章应在基本智能体概念之后阅读：协议标准化智能体周围的边界；它们并不定义智能体行为本身。",{},{"id":1333,"data":1334,"type":42,"tunes":1336},"h-faq",{"text":1335,"level":247},"常见问题",{},{"id":1338,"data":1339,"type":1338,"tunes":1378},"faq",{"items":1340,"title":1377},[1341,1345,1349,1353,1357,1361,1365,1369,1373],{"id":1342,"answer":1343,"question":1344},"faq1","智能体AI是一种AI系统，其中模型可以通过选择行动或工具、观察结果、更新状态并持续进行，直到达到停止条件，从而在多个步骤中追求一个目标。","什么是智能体AI？",{"id":1346,"answer":1347,"question":1348},"faq2","LLM从输入产生输出。智能体将模型与运行时、工具、状态、权限、上下文管理和迭代执行循环相结合。","LLM和AI智能体有什么区别？",{"id":1350,"answer":1351,"question":1352},"faq3","不一定。单个工具辅助的模型响应可能是有界的且非智能体的。当工具观察驱动自适应多步循环时，智能体行为才会出现。","工具调用是否使系统成为智能体？",{"id":1354,"answer":1355,"question":1356},"faq4","工作流通常遵循应用程序代码中定义的流程路径。智能体根据中间观察，对使用哪些步骤和工具拥有更多由模型驱动的控制权。","智能体和AI工作流有什么区别？",{"id":1358,"answer":1359,"question":1360},"faq5","不需要。长期记忆对于跨会话的持久信息很有用，但许多智能体仅使用当前任务状态和上下文即可完成有界任务。","智能体需要记忆吗？",{"id":1362,"answer":1363,"question":1364},"faq6","不需要。单个智能体通常更简单。当专业化、工具隔离、策略隔离或所有权边界能实质性改善系统时，多智能体系统才是合理的。","AI智能体需要多个智能体吗？",{"id":1366,"answer":1367,"question":1368},"faq7","可以。智能体可以自主执行低风险分析和准备，而运行时在后果性行动之前暂停以等待人工批准。","智能体可以有人工参与吗？",{"id":1370,"answer":1371,"question":1372},"faq8","不是。MCP是一种用于暴露工具、资源和提示的互操作性协议。智能体运行时可以使用MCP，但仍需要自己的循环、状态、授权和评估。","MCP是智能体框架吗？",{"id":1374,"answer":1375,"question":1376},"faq9","在可能的情况下，通过外部状态、测试、工件或权威系统记录来验证成功，而不是信任模型自己的完成声明。","如何知道智能体实际完成了任务？","智能体AI常见问题",{},{"id":1380,"data":1381,"type":42,"tunes":1383},"h-glossary",{"text":1382,"level":247},"术语表",{},{"id":1385,"data":1386,"type":1385,"tunes":1433},"glossary",{"title":1387,"entries":1388},"关键智能体AI术语",[1389,1393,1397,1401,1405,1409,1413,1417,1421,1425,1429],{"term":1390,"anchor":1391,"definition":1392},"智能体AI","agentic-ai","AI系统行为，其中模型使用工具、观察和状态动态指导多步执行以实现目标。",{"term":1394,"anchor":1395,"definition":1396},"AI智能体","ai-agent","以模型为中心的系统，具有运行时、工具、状态和执行循环，可以在多个步骤中追求任务。",{"term":1398,"anchor":1399,"definition":1400},"智能体循环","agent-loop","模型决策、工具\u002F行动执行、观察和更新模型决策的重复循环，直到停止。",{"term":1402,"anchor":1403,"definition":1404},"运行时\u002F框架","runtime-harness","管理模型循环、工具、状态、批准、上下文、错误和停止条件的执行层。",{"term":1406,"anchor":1407,"definition":1408},"工具","tool","暴露给模型的能力，用于读取信息、计算、委托或改变外部状态。",{"term":1410,"anchor":1411,"definition":1412},"观察","observation","从工具或环境返回并供应给后续智能体步骤的信息。",{"term":1414,"anchor":1415,"definition":1416},"智能体状态","agent-state","存在于单个模型输出之外的持久任务或执行信息，可能跨步骤或暂停而存活。",{"term":1418,"anchor":1419,"definition":1420},"自主边界","autonomy-boundary","明确定义模型可以动态控制哪些决策和行动的显式限制。",{"term":1422,"anchor":1423,"definition":1424},"人工参与","human-in-the-loop","一种控制模式，其中在AI驱动过程的选定点需要人工审查、输入或批准。",{"term":1426,"anchor":1427,"definition":1428},"轨迹","trajectory","任务请求和最终结果之间相关状态、决策、工具调用、行动和观察的序列。",{"term":1430,"anchor":1431,"definition":1432},"幂等性","idempotency","允许操作重复执行而不会无意中多次应用相同副作用的属性。",{},{"id":1435,"data":1436,"type":42,"tunes":1438},"h-conclusion",{"text":1437,"level":247},"结论",{},{"id":1440,"data":1441,"type":218,"tunes":1443},"p-conclusion-1",{"text":1442},"智能体AI不仅仅是更聪明的模型或拥有更多工具的聊天机器人。它是一种系统架构，其中模型参与迭代控制循环：决策、行动、观察、更新并继续。",{},{"id":1445,"data":1446,"type":218,"tunes":1448},"p-conclusion-2",{"text":1447},"模型提供灵活的决策制定，但周围的运行时必须拥有执行现实：权限、工具访问、状态、批准、重试、预算、停止条件、追踪和验证。",{},{"id":1450,"data":1451,"type":218,"tunes":1453},"p-conclusion-3",{"text":1452},"因此，最有用的设计原则是：仅在明确的技术和业务边界内将战术选择委托给模型。只有当自主性、权威和证据保持可分离时，智能体能力才能成为生产能力。",{},{"id":1455,"data":1456,"type":42,"tunes":1458},"h-sources",{"text":1457,"level":247},"主要来源和当前指南",{},{"id":1460,"data":1461,"type":218,"tunes":1463},"p-sources-note",{"text":1462},"以下来源支持当前围绕智能体、工作流、循环、工具、编排、安全和评估的架构区分。项目部分是原始实现证据，并明确限定于仓库所展示的内容。",{},{"id":1465,"data":1466,"type":1472,"tunes":1473},"src-openai-agents",{"link":1467,"meta":1468},"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fagents",{"image":1469,"title":1470,"description":1471},{"url":347},"OpenAI — 智能体","当前开发者指南，定义多步工作、工具、状态、编排和智能体执行的运行时选择。","linkTool",{},{"id":1475,"data":1476,"type":1472,"tunes":1482},"src-openai-definitions",{"link":1477,"meta":1478},"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fagents\u002Fdefine-agents",{"image":1479,"title":1480,"description":1481},{"url":347},"OpenAI — 智能体定义","当前文档将智能体描述为模型加上指令和可选运行时行为，包括工具、护栏、MCP服务器和交接。",{},{"id":1484,"data":1485,"type":1472,"tunes":1491},"src-openai-running",{"link":1486,"meta":1487},"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fagents\u002Frunning-agents",{"image":1488,"title":1489,"description":1490},{"url":347},"OpenAI — 运行智能体","当前关于智能体循环的文档：模型调用、工具执行或交接、继续和最终停止点。",{},{"id":1493,"data":1494,"type":1472,"tunes":1500},"src-openai-orchestration",{"link":1495,"meta":1496},"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fagents\u002Forchestration",{"image":1497,"title":1498,"description":1499},{"url":347},"OpenAI — 编排和交接","当前关于交接、智能体即工具以及专家智能体何时增加有用所有权或能力边界的指南。",{},{"id":1502,"data":1503,"type":1472,"tunes":1509},"src-openai-safety",{"link":1504,"meta":1505},"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fagent-builder-safety",{"image":1506,"title":1507,"description":1508},{"url":347},"OpenAI — 构建智能体中的安全","当前安全指南，涵盖工具批准、提示注入、护栏和基于追踪的评估。",{},{"id":1511,"data":1512,"type":1472,"tunes":1518},"src-anthropic-agents",{"link":1513,"meta":1514},"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Fbuilding-effective-agents",{"image":1515,"title":1516,"description":1517},{"url":347},"Anthropic — 构建有效智能体","工程指南，区分预定义工作流和模型指导的智能体，并描述基于工具的环境反馈循环。",{},{"id":1520,"data":1521,"type":1472,"tunes":1527},"src-anthropic-context",{"link":1522,"meta":1523},"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Feffective-context-engineering-for-ai-agents",{"image":1524,"title":1525,"description":1526},{"url":347},"Anthropic — AI智能体的有效上下文工程","将智能体实际框架为LLM在循环中自主使用工具，并具有动态即时上下文管理。",{},{"id":1529,"data":1530,"type":1472,"tunes":1536},"src-anthropic-evals",{"link":1531,"meta":1532},"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Fdemystifying-evals-for-ai-agents",{"image":1533,"title":1534,"description":1535},{"url":347},"Anthropic — 揭秘AI智能体的评估","2026年关于评估多轮智能体的指南，这些智能体调用工具、修改状态并适应中间结果。",{},"2.31","代理式AI在多步执行循环中使用模型，这些模型可以在明确的运行时和权限边界内选择工具、观察结果、更新状态并调整其下一步行动。","\u002Fuploads\u002F2026\u002F10\u002Fagentic-ai-explained-when-an-ai-system-can-plan-use-tools-and-act-1791481499084-wnji2a.webp","agentic-ai-explained-when-an-ai-system-can-plan-use-tools-and-act-1791481499084-wnji2a","PUBLISHED","2026-10-08T11:43:00.000Z","2026-10-08T17:43:28.373Z","2026-10-08T19:19:47.140Z",{"en":1546,"de":1547,"sr":1548,"es":1549,"fr":1550,"it":1551,"ru":1552,"zh":1553},"\u002Fblog\u002Fagentic-ai-explained-when-an-ai-system-can-plan-use-tools-and-act","\u002Fde\u002Fblog\u002Fagentic-ai-explained-when-an-ai-system-can-plan-use-tools-and-act","\u002Fsr\u002Fblog\u002Fagentic-ai-explained-when-an-ai-system-can-plan-use-tools-and-act","\u002Fes\u002Fblog\u002Fagentic-ai-explained-when-an-ai-system-can-plan-use-tools-and-act","\u002Ffr\u002Fblog\u002Fagentic-ai-explained-when-an-ai-system-can-plan-use-tools-and-act","\u002Fit\u002Fblog\u002Fagentic-ai-explained-when-an-ai-system-can-plan-use-tools-and-act","\u002Fru\u002Fblog\u002Fagentic-ai-explained-when-an-ai-system-can-plan-use-tools-and-act","\u002Fzh\u002Fblog\u002Fagentic-ai-explained-when-an-ai-system-can-plan-use-tools-and-act",[1555,1559,1563],{"id":1556,"name":1557,"slug":1558},84,"策略与数据边界","policy-and-data",{"id":1560,"name":1561,"slug":1562},57,"数据边界","data-boundaries",{"id":1564,"name":1565,"slug":1566},59,"治理与可审计性","governance",{"id":1568,"login":1569,"email":1570,"displayName":1571},"20","rooth8233","aleksandar@stajic.de","Aleksandar Stajić",[1573,2705],{"lang":1574,"title":1575,"content":1576,"contentJson":1577,"excerpt":2704},"en","Agentic AI Explained: When an AI System Can Plan, Use Tools and Act","{\"time\":1791487185746,\"blocks\":[{\"id\":\"intro\",\"type\":\"paragraph\",\"data\":{\"text\":\"Agentic AI is an AI system in which a model can pursue a goal across multiple steps by deciding what to do next, using tools or other capabilities, observing the results, updating its working state and continuing until it reaches a stopping condition. The model alone is not the agent. A usable agent also needs a runtime or harness that manages context, tool execution, state, permissions, approvals, errors and the loop between decisions and observations.\"},\"tunes\":{}},{\"id\":\"direct\",\"type\":\"callout\",\"data\":{\"variant\":\"info\",\"title\":\"Direct answer\",\"body\":\"A normal model call is usually \u003Cstrong>input → model → output\u003C\u002Fstrong>. An agentic system is closer to \u003Cstrong>goal → decision → tool\u002Faction → observation → updated decision → … → result\u003C\u002Fstrong>.\u003Cbr>\u003Cbr>The critical distinction is not whether an application uses an LLM or function calling. It is whether the system gives the model meaningful control over the next step of a multi-step process while a runtime constrains what the model is actually allowed to do.\"},\"tunes\":{}},{\"id\":\"boundary\",\"type\":\"callout\",\"data\":{\"variant\":\"warning\",\"title\":\"Capability is not authority\",\"body\":\"A model may know how to call a tool. A runtime may expose that tool. Neither fact means the current user or agent is authorized to execute the underlying business action. \u003Cstrong>Tool capability, tool permission and business authority are separate layers.\u003C\u002Fstrong>\"},\"tunes\":{}},{\"id\":\"current\",\"type\":\"callout\",\"data\":{\"variant\":\"note\",\"title\":\"Current-source note — 8 October 2026\",\"body\":\"Agent terminology still varies across vendors and research communities. OpenAI currently defines agent runtimes around multi-step work, tools, state and orchestration. Anthropic's practical distinction remains useful: workflows follow predefined code paths, while agents dynamically direct their own process and tool use. This article therefore treats “agentic AI” as an architectural spectrum rather than one standardized product category.\"},\"tunes\":{}},{\"id\":\"toc\",\"type\":\"tableOfContents\",\"data\":{\"title\":\"Contents\",\"minLevel\":2,\"maxLevel\":3},\"tunes\":{}},{\"id\":\"h-meaning\",\"type\":\"header\",\"data\":{\"text\":\"What agentic AI really means\",\"level\":2},\"tunes\":{}},{\"id\":\"p-meaning-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The important shift from ordinary generative AI to agentic AI is control over process. A normal assistant can answer a question using the context it receives. An agent can decide that answering requires additional steps: inspect a file, search a repository, query an API, ask for clarification, run a test, update a ticket, delegate a subtask or retry after a failed action.\"},\"tunes\":{}},{\"id\":\"p-meaning-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"This does not require unlimited autonomy. An agent can operate inside a narrow sandbox, under strict permissions, with approval required before every consequential action. The system is still agentic if the model dynamically chooses among permitted next steps.\"},\"tunes\":{}},{\"id\":\"p-meaning-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The architecture therefore matters more than the label. “Agent” should describe a system behavior: iterative model-driven decision making over tools, state and feedback — not merely a chatbot with a larger prompt.\"},\"tunes\":{}},{\"id\":\"h-simple\",\"type\":\"header\",\"data\":{\"text\":\"The simplest example\",\"level\":2},\"tunes\":{}},{\"id\":\"p-simple-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Suppose a developer asks an AI system: “Find why the test suite fails and fix the bug.” A single model call could only suggest likely causes from the text it was given.\"},\"tunes\":{}},{\"id\":\"p-simple-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"An agentic coding system can inspect the repository, search for the failing test, read relevant files, propose a change, edit the code, run the test, observe the failure, revise the implementation and run the test again.\"},\"tunes\":{}},{\"id\":\"p-simple-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The agentic part is not simply that shell and file tools exist. It is that the model can use environmental feedback to choose the next step instead of following one completely predefined sequence.\"},\"tunes\":{}},{\"id\":\"simple-loop\",\"type\":\"processFlow\",\"data\":{\"title\":\"The basic agent loop\",\"orientation\":\"auto\",\"steps\":[{\"label\":\"1. Receive a goal\",\"description\":\"The user or upstream system defines the objective and relevant constraints.\"},{\"label\":\"2. Build current context\",\"description\":\"The runtime supplies instructions, state, history, memory, tools and current evidence.\"},{\"label\":\"3. Model decides next step\",\"description\":\"The model may answer, call a tool, request information, delegate or stop.\"},{\"label\":\"4. Runtime validates the request\",\"description\":\"Permissions, schemas, approvals and policy determine whether the proposed action may execute.\"},{\"label\":\"5. Execute tool or action\",\"description\":\"The external environment changes or returns new information.\"},{\"label\":\"6. Observe the result\",\"description\":\"The runtime feeds structured tool output, errors or state changes back into the next model step.\"},{\"label\":\"7. Continue or stop\",\"description\":\"The loop repeats until success, refusal, escalation, budget limit, timeout or another stopping condition.\"}]},\"tunes\":{}},{\"id\":\"h-stops\",\"type\":\"header\",\"data\":{\"text\":\"Where the simple example stops\",\"level\":2},\"tunes\":{}},{\"id\":\"p-stops-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Not every multi-step AI system is equally agentic. A workflow may use several LLM calls and tools while every step is predetermined in code. Another system may let the model decide which tool to call, in which order, how many times and when to stop.\"},\"tunes\":{}},{\"id\":\"p-stops-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Both can be useful. The difference is where control lives. Predefined workflows put more control in application code. Agents move more tactical process decisions into the model\u002Fruntime loop.\"},\"tunes\":{}},{\"id\":\"h-workflow\",\"type\":\"header\",\"data\":{\"text\":\"Agent vs workflow\",\"level\":2},\"tunes\":{}},{\"id\":\"workflow-comparison\",\"type\":\"comparison\",\"data\":{\"title\":\"Predefined workflow and agentic control\",\"layout\":\"table\",\"columns\":[{\"id\":\"workflow\",\"label\":\"LLM workflow\"},{\"id\":\"agent\",\"label\":\"Agent\"}],\"rows\":[{\"id\":\"path\",\"label\":\"Process path\",\"values\":[\"\",\"\"]},{\"id\":\"tools\",\"label\":\"Tool sequence\",\"values\":[\"\",\"\"]},{\"id\":\"strength\",\"label\":\"Strength\",\"values\":[\"\",\"\"]},{\"id\":\"risk\",\"label\":\"Risk\",\"values\":[\"\",\"\"]}]},\"tunes\":{}},{\"id\":\"p-workflow-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Anthropic explicitly separates these two patterns: workflows orchestrate models and tools through predefined code paths, while agents let models dynamically direct their own processes and tool usage. This is not the only possible terminology, but it is a useful architecture boundary.\"},\"tunes\":{}},{\"id\":\"h-spectrum\",\"type\":\"header\",\"data\":{\"text\":\"Agentic behavior is a spectrum, not a binary label\",\"level\":2},\"tunes\":{}},{\"id\":\"spectrum-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Level\",\"Example\",\"Who decides the next step?\"],[\"Single model call\",\"Summarize this document\",\"Application calls model once\"],[\"Tool-assisted response\",\"Model may use web search before answering\",\"Model selects from bounded tools for one response\"],[\"Structured workflow\",\"Classify → retrieve → generate → validate\",\"Application workflow determines stages\"],[\"Adaptive workflow\",\"Model can choose among several branches and retry\",\"Shared control between application and model\"],[\"Agent loop\",\"Model repeatedly chooses tools\u002Factions based on observations\",\"Model directs tactical execution inside runtime constraints\"],[\"Long-running agent\",\"Agent pauses, resumes, manages artifacts and continues\",\"Model + persistent runtime manage evolving execution\"]]},\"tunes\":{}},{\"id\":\"p-spectrum-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Calling every system above an “agent” can obscure important operational differences. The stronger the model's control over sequence, duration and actions, the more important runtime isolation, permissions, tracing, stopping conditions and trajectory evaluation become.\"},\"tunes\":{}},{\"id\":\"h-anatomy\",\"type\":\"header\",\"data\":{\"text\":\"The minimum architecture of an agentic system\",\"level\":2},\"tunes\":{}},{\"id\":\"anatomy-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Component\",\"Responsibility\"],[\"Goal \u002F task\",\"Defines what the system is trying to accomplish.\"],[\"Model\",\"Interprets context and decides the next action or output.\"],[\"Instructions\",\"Define role, constraints, priorities and task-specific policy.\"],[\"Context assembler\",\"Builds the information visible to the model on each step.\"],[\"Tool catalog\",\"Defines capabilities the model may request.\"],[\"Runtime \u002F harness\",\"Runs the loop, executes tools, manages state and handles stopping conditions.\"],[\"Authorization layer\",\"Determines whether a proposed action is permitted for the current principal.\"],[\"State \u002F session\",\"Preserves task progress across turns or execution steps.\"],[\"Observation channel\",\"Returns tool results and environment changes to the next model step.\"],[\"Approvals \u002F human control\",\"Pauses consequential actions where review is required.\"],[\"Tracing \u002F audit\",\"Records model calls, tools, transitions, approvals and failures.\"],[\"Evaluation\",\"Measures outcomes and execution trajectories against acceptance criteria.\"]]},\"tunes\":{}},{\"id\":\"h-model-agent\",\"type\":\"header\",\"data\":{\"text\":\"A model is not an agent\",\"level\":2},\"tunes\":{}},{\"id\":\"p-model-agent-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A language model produces outputs from inputs. It does not by itself own a filesystem, execute a shell command, maintain durable task state, enforce permissions or automatically call itself again.\"},\"tunes\":{}},{\"id\":\"p-model-agent-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Those capabilities come from the surrounding runtime. The same model can behave as a simple chat model in one application and as the decision engine inside an agent loop in another.\"},\"tunes\":{}},{\"id\":\"model-agent-rule\",\"type\":\"callout\",\"data\":{\"variant\":\"success\",\"title\":\"Architecture rule\",\"body\":\"\u003Cstrong>Model capability determines what decisions can be proposed. Runtime architecture determines what can actually happen.\u003C\u002Fstrong>\"},\"tunes\":{}},{\"id\":\"h-tools\",\"type\":\"header\",\"data\":{\"text\":\"Tool use is central — but tool use alone does not make an agent\",\"level\":2},\"tunes\":{}},{\"id\":\"p-tools-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Tools let the model acquire information and affect external systems. Examples include database reads, file operations, shell execution, web search, browser control, API calls, ticket updates or delegated specialist agents.\"},\"tunes\":{}},{\"id\":\"p-tools-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"A single model call can use one tool and still remain a bounded tool-assisted response rather than a long-running agent. Agentic behavior appears when tool observations feed an adaptive loop in which the model chooses what to do next.\"},\"tunes\":{}},{\"id\":\"p-tools-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"Tool design matters because tools are the contract between model reasoning and external reality. Ambiguous or overlapping tools create routing errors; large unstructured outputs pollute context; broad side-effect tools increase blast radius.\"},\"tunes\":{}},{\"id\":\"h-capability-permission\",\"type\":\"header\",\"data\":{\"text\":\"Tool capability, permission and authority are different\",\"level\":2},\"tunes\":{}},{\"id\":\"permission-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Layer\",\"Question\"],[\"Capability\",\"Can this runtime technically perform the operation?\"],[\"Tool exposure\",\"Is that capability available to this agent?\"],[\"Permission\",\"May this agent\u002Fsession use it under the current policy?\"],[\"User authorization\",\"Is the requesting principal allowed to cause this operation?\"],[\"Business authority\",\"Is the operation valid under domain rules, approvals and limits?\"],[\"Execution\",\"Did the operation actually occur?\"],[\"Audit\",\"Can the system prove who requested, approved and executed it?\"]]},\"tunes\":{}},{\"id\":\"p-permission-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"These layers are frequently collapsed in prototypes. A model sees a refund tool and therefore appears able to issue refunds. In production, the tool should still validate account, user, transaction, amount, policy and approval conditions independently of the model's request.\"},\"tunes\":{}},{\"id\":\"h-runtime\",\"type\":\"header\",\"data\":{\"text\":\"The runtime or harness is the actual execution system\",\"level\":2},\"tunes\":{}},{\"id\":\"p-runtime-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"OpenAI's current agent documentation makes the runtime distinction explicit. Different runtimes can manage orchestration, state, tools, sandboxes and execution in different places, while the model remains only one part of the system.\"},\"tunes\":{}},{\"id\":\"p-runtime-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The Agents SDK describes a loop that repeatedly calls the current model, inspects the output, executes requested tools or handoffs, and continues until the model returns a final answer or another real stopping point.\"},\"tunes\":{}},{\"id\":\"p-runtime-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"This means agent architecture decisions include where orchestration runs, where state lives, who executes tools, which sandbox contains side effects, and who owns retries, timeouts and resumability.\"},\"tunes\":{}},{\"id\":\"h-planning\",\"type\":\"header\",\"data\":{\"text\":\"Planning is useful, but an explicit plan is not required\",\"level\":2},\"tunes\":{}},{\"id\":\"p-planning-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Agents are often described as systems that “plan.” In practice, planning can be explicit or implicit. An agent may first produce a visible multi-step plan, or it may choose one next action at a time and revise after every observation.\"},\"tunes\":{}},{\"id\":\"p-planning-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"For highly uncertain tasks, short-horizon planning can be safer because the environment can invalidate a long plan. The architectural requirement is the ability to choose and revise actions based on the goal, current state and new evidence.\"},\"tunes\":{}},{\"id\":\"h-feedback\",\"type\":\"header\",\"data\":{\"text\":\"Environmental feedback is what makes the loop useful\",\"level\":2},\"tunes\":{}},{\"id\":\"p-feedback-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"An agent becomes operationally meaningful when it can observe whether its action worked. Tool output, test results, API responses, filesystem state, browser state and application records provide external evidence that the system can use to revise its next decision.\"},\"tunes\":{}},{\"id\":\"p-feedback-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Anthropic's agent guidance emphasizes this feedback loop: agents use tools, obtain ground truth from the environment, assess progress and continue or request human input.\"},\"tunes\":{}},{\"id\":\"feedback-rule\",\"type\":\"callout\",\"data\":{\"variant\":\"warning\",\"title\":\"Self-report is not environmental proof\",\"body\":\"An agent saying “the task is complete” does not prove completion. Where possible, verify the final state through an external system, test, file, transaction record or other observable outcome.\"},\"tunes\":{}},{\"id\":\"h-state\",\"type\":\"header\",\"data\":{\"text\":\"Agent state is not the same as model context\",\"level\":2},\"tunes\":{}},{\"id\":\"p-state-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A long-running task may need state that cannot or should not remain in the model context: task IDs, checkpoints, artifacts, approvals, external object identifiers, retry counters and workflow status.\"},\"tunes\":{}},{\"id\":\"p-state-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The runtime can preserve this durable state outside the model window and reconstruct the context required for the next step. This keeps model-visible context focused while maintaining continuity and resumability.\"},\"tunes\":{}},{\"id\":\"h-memory\",\"type\":\"header\",\"data\":{\"text\":\"Memory is optional, not the definition of an agent\",\"level\":2},\"tunes\":{}},{\"id\":\"p-memory-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"An agent can operate successfully without long-term memory if the complete task fits inside one bounded run. Memory becomes useful when information must persist across sessions, tasks or long execution horizons.\"},\"tunes\":{}},{\"id\":\"p-memory-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"RAG, memory, state and context solve different problems. Treating a vector database as “the agent memory” or conversation history as “the state machine” usually hides important lifecycle and authority boundaries.\"},\"tunes\":{}},{\"id\":\"ref-memory\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context\",\"title\":\"AI Agent Memory Is Not RAG: How to Separate Memory, Retrieval, State and Context\",\"excerpt\":\"A practical architecture separating persistent memory, authoritative application state, retrieval and the context supplied to the model.\",\"ctaLabel\":\"Read the memory architecture article\"},\"tunes\":{}},{\"id\":\"h-context\",\"type\":\"header\",\"data\":{\"text\":\"Context engineering becomes dynamic in agents\",\"level\":2},\"tunes\":{}},{\"id\":\"p-context-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Every tool call can produce new context. Every step can also make earlier information obsolete. A strong agent runtime therefore rebuilds or curates context as execution progresses rather than replaying everything indefinitely.\"},\"tunes\":{}},{\"id\":\"p-context-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Tool definitions, task state, retrieved evidence, observations and memory all compete for the model's attention. Long-running agents need trimming, compaction or just-in-time loading so context remains relevant to the current decision.\"},\"tunes\":{}},{\"id\":\"h-sideeffects\",\"type\":\"header\",\"data\":{\"text\":\"Read tools and side-effect tools have different risk\",\"level\":2},\"tunes\":{}},{\"id\":\"sideeffect-comparison\",\"type\":\"comparison\",\"data\":{\"title\":\"Information access versus external action\",\"layout\":\"table\",\"columns\":[{\"id\":\"read\",\"label\":\"Read \u002F observe\"},{\"id\":\"write\",\"label\":\"Write \u002F act\"}],\"rows\":[{\"id\":\"examples\",\"label\":\"Examples\",\"values\":[\"\",\"\"]},{\"id\":\"mainrisk\",\"label\":\"Main risk\",\"values\":[\"\",\"\"]},{\"id\":\"control\",\"label\":\"Typical control\",\"values\":[\"\",\"\"]}]},\"tunes\":{}},{\"id\":\"h-approval\",\"type\":\"header\",\"data\":{\"text\":\"Human-in-the-loop is a control mechanism, not the opposite of agentic AI\",\"level\":2},\"tunes\":{}},{\"id\":\"p-approval-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"An agent does not stop being agentic because a human approves consequential steps. The model can still autonomously inspect, reason, search and prepare an action while the runtime requires human confirmation before execution.\"},\"tunes\":{}},{\"id\":\"p-approval-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"OpenAI's current agent safety guidance explicitly recommends approvals for tool operations in higher-risk workflows. Anthropic likewise emphasizes checkpoints and human judgment where agents encounter blockers or consequential decisions.\"},\"tunes\":{}},{\"id\":\"p-approval-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The useful architecture question is not “human or autonomous?” but which decisions can be delegated, which require review and which must remain deterministic?\"},\"tunes\":{}},{\"id\":\"h-stopping\",\"type\":\"header\",\"data\":{\"text\":\"Agents need explicit stopping conditions\",\"level\":2},\"tunes\":{}},{\"id\":\"stop-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Stopping condition\",\"Purpose\"],[\"Successful verified outcome\",\"End when the external target state is confirmed.\"],[\"Maximum steps\",\"Prevent runaway loops.\"],[\"Time budget\",\"Bound wall-clock execution.\"],[\"Cost\u002Ftoken budget\",\"Limit resource consumption.\"],[\"Repeated-action detector\",\"Stop loops that are no longer making progress.\"],[\"Permission boundary\",\"Pause or stop when the next required action is not permitted.\"],[\"Human approval checkpoint\",\"Wait before consequential execution.\"],[\"Unrecoverable tool failure\",\"Escalate instead of retrying indefinitely.\"],[\"Uncertainty threshold\",\"Ask for clarification when the task cannot be safely inferred.\"]]},\"tunes\":{}},{\"id\":\"h-errors\",\"type\":\"header\",\"data\":{\"text\":\"Recovery is part of agent behavior\",\"level\":2},\"tunes\":{}},{\"id\":\"p-errors-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Agents operate in environments that fail: APIs time out, files change, credentials expire, webpages move and tools return malformed output. A useful agentic system therefore needs recovery behavior, not just a happy-path tool loop.\"},\"tunes\":{}},{\"id\":\"p-errors-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Recovery can include retry with limits, choosing another tool, re-reading current state, asking the user, rolling back a partial action or escalating to a human.\"},\"tunes\":{}},{\"id\":\"p-errors-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"Retries also need idempotency awareness. Repeating a read is usually low risk; repeating a payment or message send can create duplicate side effects.\"},\"tunes\":{}},{\"id\":\"h-single-multi\",\"type\":\"header\",\"data\":{\"text\":\"Agentic AI does not require multiple agents\",\"level\":2},\"tunes\":{}},{\"id\":\"p-multi-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A single agent with a clear tool set is often simpler and easier to evaluate than a multi-agent architecture. Multiple agents are useful when specialization materially improves tool isolation, policy isolation, prompt clarity, ownership or trace legibility.\"},\"tunes\":{}},{\"id\":\"p-multi-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"OpenAI's current orchestration guidance explicitly recommends starting with one agent where possible and adding specialists only when the contract or ownership boundary materially changes.\"},\"tunes\":{}},{\"id\":\"p-multi-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"Multi-agent systems add new problems: delegation quality, duplicated context, conflicting state, handoff semantics, identity, cost and distributed failure handling.\"},\"tunes\":{}},{\"id\":\"h-protocols\",\"type\":\"header\",\"data\":{\"text\":\"Agent protocols are interoperability layers, not the agent itself\",\"level\":2},\"tunes\":{}},{\"id\":\"p-protocols-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Protocols such as MCP and A2A can make an agent architecture interoperable, but they do not create the agent loop by themselves. MCP can expose tools and resources. A2A can connect independently implemented agents. The application still needs runtime, authorization, state, evaluation and domain logic.\"},\"tunes\":{}},{\"id\":\"p-protocols-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"This is why protocol capability must remain separate from business authority. Discovering a tool through MCP does not prove the current principal is allowed to use it. Receiving a task through A2A does not prove the remote agent may perform every requested action.\"},\"tunes\":{}},{\"id\":\"ref-protocols\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fde\u002Fblog\u002Fmcp-vs-a2a-vs-ucp-vs-ap2-vs-a2ui-the-agent-protocol-stack-explained\",\"title\":\"MCP vs A2A vs UCP vs AP2 vs A2UI: The Agent Protocol Stack Explained\",\"excerpt\":\"A protocol-responsibility map showing why tool access, agent collaboration, commerce, payment authority and agent-driven UI belong to different interoperability boundaries.\",\"ctaLabel\":\"Read the agent protocol stack\"},\"tunes\":{}},{\"id\":\"h-reliability\",\"type\":\"header\",\"data\":{\"text\":\"The trajectory is part of agent reliability\",\"level\":2},\"tunes\":{}},{\"id\":\"p-rel-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A final answer is insufficient evidence for an agentic system because an agent can reach the right result through an unsafe or invalid path. It may use an unauthorized tool, skip a required check, retry a side effect, rely on stale state or accidentally succeed.\"},\"tunes\":{}},{\"id\":\"p-rel-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Evaluation therefore needs execution traces: decisions, tool calls, approvals, observations, state changes and final outcome. Current OpenAI safety guidance recommends trace graders and evals; Anthropic's 2026 agent-evaluation guidance similarly treats multi-turn tool trajectories as first-class evaluation objects.\"},\"tunes\":{}},{\"id\":\"p-rel-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The stronger reliability question is: Did the agent reach an acceptable outcome through an acceptable, recoverable and auditable trajectory?\"},\"tunes\":{}},{\"id\":\"ref-reliability\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough\",\"title\":\"AI Agent Reliability: Why the Final Answer Is Not Enough\",\"excerpt\":\"Why production evaluation must inspect trajectories, tool use, state transitions and recoverability rather than only final answers.\",\"ctaLabel\":\"Read the reliability article\"},\"tunes\":{}},{\"id\":\"h-security\",\"type\":\"header\",\"data\":{\"text\":\"Agentic systems increase the security surface\",\"level\":2},\"tunes\":{}},{\"id\":\"security-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Risk\",\"Why agents amplify it\",\"Architecture response\"],[\"Prompt injection\",\"Untrusted content can influence future tool decisions\",\"Separate instructions from data; constrain tools; sanitize or structure external input where possible\"],[\"Excessive permissions\",\"Reasoning errors can become real side effects\",\"Least privilege, scoped credentials, per-tool policy and approvals\"],[\"Credential exposure\",\"Tools may need powerful secrets\",\"Keep secrets outside model context; broker access through trusted runtime\"],[\"Confused deputy\",\"Agent may act with authority broader than the requesting user\",\"Bind execution to user\u002Fservice identity and re-authorize consequential actions\"],[\"Runaway loops\",\"Model repeatedly calls tools without progress\",\"Step, time and cost budgets plus loop detection\"],[\"State drift\",\"Environment changes after the agent formed a plan\",\"Re-read authoritative state before consequential actions\"],[\"Indirect injection\",\"Tool\u002Fweb\u002Fdocument content contains instructions aimed at the model\",\"Treat external content as untrusted data, not instruction authority\"],[\"Audit gap\",\"Final result cannot show what was executed\",\"Trace tool calls, approvals, identities and state changes\"]]},\"tunes\":{}},{\"id\":\"h-observability\",\"type\":\"header\",\"data\":{\"text\":\"Agent observability must follow the loop\",\"level\":2},\"tunes\":{}},{\"id\":\"p-obs-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Traditional service observability records requests, latency and errors. Agent observability needs an additional execution model: which agent was active, which model version made the decision, what context was available, which tool was selected, what arguments were sent, what result came back and why execution stopped.\"},\"tunes\":{}},{\"id\":\"p-obs-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"For sensitive systems, traces themselves require access control and retention policy because prompts, tool outputs and artifacts can contain confidential data.\"},\"tunes\":{}},{\"id\":\"h-eval\",\"type\":\"header\",\"data\":{\"text\":\"How to evaluate an agentic system\",\"level\":2},\"tunes\":{}},{\"id\":\"eval-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Dimension\",\"Question\",\"Example evidence\"],[\"Task success\",\"Did the requested outcome occur?\",\"External state, tests, business outcome\"],[\"Trajectory quality\",\"Were the steps acceptable?\",\"Tool\u002Faction trace\"],[\"Tool selection\",\"Did the agent choose appropriate capabilities?\",\"Expected vs actual tool calls\"],[\"Permission adherence\",\"Did it stay inside allowed authority?\",\"Authorization logs and denied-action tests\"],[\"State handling\",\"Did it use current authoritative state?\",\"Freshness checks and state-change tests\"],[\"Recovery\",\"Did it respond correctly to failures?\",\"Injected timeout\u002Ferror scenarios\"],[\"Stopping behavior\",\"Did it stop at the right point?\",\"Step counts, loop detection, final-state proof\"],[\"Human escalation\",\"Did it ask when review was required?\",\"Approval\u002Fescalation traces\"],[\"Cost\u002Flatency\",\"Was autonomy worth the operational cost?\",\"Tokens, tool calls, duration\"],[\"Robustness\",\"Does it survive realistic environment variation?\",\"Repeated and adversarial trials\"]]},\"tunes\":{}},{\"id\":\"h-use\",\"type\":\"header\",\"data\":{\"text\":\"When an agent is appropriate\",\"level\":2},\"tunes\":{}},{\"id\":\"use-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Use an agent when\",\"Prefer a workflow or simple call when\"],[\"The number or order of steps cannot be known reliably in advance\",\"The sequence is stable and deterministic\"],[\"The system must inspect the environment and adapt\",\"A single retrieval + generation step is sufficient\"],[\"Several tools may be useful depending on intermediate results\",\"One known API call solves the task\"],[\"The task benefits from iterative verification or repair\",\"The answer can be produced directly from supplied context\"],[\"Failures require flexible recovery behavior\",\"Failure branches are simple and can be encoded explicitly\"],[\"Human review can be inserted at meaningful checkpoints\",\"Every step is high-risk and must be manually controlled anyway\"],[\"Expected value justifies extra latency, cost and complexity\",\"Predictability and low cost matter more than flexibility\"]]},\"tunes\":{}},{\"id\":\"p-use-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A strong default is to start with the simplest solution that works and increase agentic complexity only when flexibility produces measurable value. Agents trade predictability, latency and cost for adaptive execution.\"},\"tunes\":{}},{\"id\":\"h-implementation\",\"type\":\"header\",\"data\":{\"text\":\"Original implementation evidence\",\"level\":2},\"tunes\":{}},{\"id\":\"h-client\",\"type\":\"header\",\"data\":{\"text\":\"Aaasaasa AI Client: model, runtime and permission are separate\",\"level\":3},\"tunes\":{}},{\"id\":\"p-client-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Aaasaasa AI Client explicitly separates agent\u002Fclient, provider, model, runtime location and permissions. Its architecture documentation treats permissions as central tool\u002Fworkspace policy rather than a model property.\"},\"tunes\":{}},{\"id\":\"p-client-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The same application can expose Direct Chat with no filesystem or shell tools while a Codex runtime operates under a selected workspace and permission profile. This demonstrates a core agentic architecture boundary: changing the runtime\u002Ftool surface changes what the system can do even when model access remains available.\"},\"tunes\":{}},{\"id\":\"p-client-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The repository also distinguishes a local Codex runtime from model location: a local runtime can call a cloud model. This prevents the common mistake of equating “agent runs locally” with “inference is local.”\"},\"tunes\":{}},{\"id\":\"p-client-4\",\"type\":\"paragraph\",\"data\":{\"text\":\"The implementation disables embedded execution paths whose approval semantics do not satisfy the required permission model. This supports the principle that agent capability should not bypass runtime authorization simply because an underlying framework can execute tools.\"},\"tunes\":{}},{\"id\":\"h-sot-agent\",\"type\":\"header\",\"data\":{\"text\":\"Source of Truth Research Engine: bounded agentic research stages\",\"level\":3},\"tunes\":{}},{\"id\":\"p-sot-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The Source of Truth Research Engine uses a bounded research pipeline: discover → acquire → extract → verify → contradict → synthesize. Research jobs can execute through an AI runtime while evidence, sources, claims and contradictions remain in an external persistent store.\"},\"tunes\":{}},{\"id\":\"p-sot-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"This is intentionally more controlled than an unconstrained autonomous research agent. The stages provide guardrails around what kind of work should happen next while still allowing model-driven research inside each bounded task.\"},\"tunes\":{}},{\"id\":\"p-sot-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"That distinction is useful evidence for agent design: autonomy can be placed inside a structured delivery envelope rather than applied uniformly to the entire process.\"},\"tunes\":{}},{\"id\":\"impl-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Implemented pattern\",\"Agentic architecture lesson\"],[\"Direct Chat has no OS tools\",\"A model can exist without agentic execution capability.\"],[\"Codex runtime has workspace permission profile\",\"Tool authority belongs to runtime policy, not model capability.\"],[\"Provider\u002Fmodel\u002Fruntime are separate concepts\",\"Agent harness location and inference location are independent decisions.\"],[\"Permission broker for tool-capable runtimes\",\"Capability exposure can be centralized and governed.\"],[\"Bounded research stages\",\"Autonomy can operate inside explicit process boundaries.\"],[\"Persistent claims\u002Fevidence outside model context\",\"Agent state and evidence do not need to live only in conversation history.\"]]},\"tunes\":{}},{\"id\":\"impl-boundary\",\"type\":\"callout\",\"data\":{\"variant\":\"note\",\"title\":\"Evidence boundary\",\"body\":\"These projects demonstrate concrete agent\u002Fruntime, permission and bounded-research patterns. They are not presented as proof of large-scale commercial autonomous-agent deployment.\"},\"tunes\":{}},{\"id\":\"h-failures\",\"type\":\"header\",\"data\":{\"text\":\"Common agentic AI failure modes\",\"level\":2},\"tunes\":{}},{\"id\":\"failure-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Failure mode\",\"What actually failed\"],[\"“Agent” is only a chatbot with tools listed in the prompt\",\"No reliable runtime loop or tool execution architecture exists\"],[\"Tool support is treated as permission\",\"Capability and authorization boundaries are collapsed\"],[\"Agent trusts its own completion statement\",\"Outcome is not verified against external state\"],[\"Every task becomes multi-agent\",\"Complexity increases without a real ownership or specialization boundary\"],[\"Conversation history is used as durable state\",\"Resumability and authoritative state become fragile\"],[\"Agent retries side effects blindly\",\"Duplicate messages, payments or state changes become possible\"],[\"No step\u002Fcost limits\",\"Agent can loop indefinitely or consume uncontrolled resources\"],[\"Tool output is trusted as instruction\",\"Indirect prompt injection can redirect behavior\"],[\"Correct final answer is the only evaluation\",\"Unsafe or invalid trajectories remain invisible\"],[\"Model upgrade is treated as transparent\",\"Tool selection, planning and stopping behavior can change\"],[\"One broad tool exposes many privileged operations\",\"Blast radius increases and intent becomes harder to validate\"],[\"Human approval exists but reviewer lacks context\",\"Approval becomes ceremonial rather than effective\"]]},\"tunes\":{}},{\"id\":\"h-misconceptions\",\"type\":\"header\",\"data\":{\"text\":\"Common misconceptions\",\"level\":2},\"tunes\":{}},{\"id\":\"misconceptions-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Misconception\",\"Correction\"],[\"“An LLM is an agent.”\",\"The model is the decision component; the agent is the surrounding system that manages tools, state and iteration.\"],[\"“Tool calling automatically means agentic AI.”\",\"A single bounded tool call may not involve an adaptive multi-step agent loop.\"],[\"“Agents must be fully autonomous.”\",\"Agentic systems can require approvals and operate under narrow permission boundaries.\"],[\"“Agents need long-term memory.”\",\"Memory is optional; many useful agents complete bounded tasks without cross-session memory.\"],[\"“Agents must create a written plan first.”\",\"Planning can be explicit or implicit and can occur one step at a time.\"],[\"“Multi-agent is more advanced than single-agent.”\",\"It is more complex; use it only when specialization or ownership boundaries justify it.\"],[\"“MCP creates an agent.”\",\"MCP exposes tools\u002Fresources; the runtime still needs an agent loop and authorization model.\"],[\"“A local runtime means the model is local.”\",\"Runtime location and inference\u002Fprovider location are separate.\"],[\"“If the final result is correct, the agent worked correctly.”\",\"An unsafe or unauthorized trajectory can still produce a correct result.\"],[\"“Human approval removes autonomy.”\",\"Approval can constrain selected actions while the rest of the process remains model-directed.\"]]},\"tunes\":{}},{\"id\":\"h-design\",\"type\":\"header\",\"data\":{\"text\":\"A practical agent design sequence\",\"level\":2},\"tunes\":{}},{\"id\":\"design-flow\",\"type\":\"processFlow\",\"data\":{\"title\":\"Design the agent from authority outward\",\"orientation\":\"auto\",\"steps\":[{\"label\":\"1. Define the outcome\",\"description\":\"State what external result or artifact proves task success.\"},{\"label\":\"2. Decide whether an agent is actually needed\",\"description\":\"Prefer a simple call or deterministic workflow when the path is predictable.\"},{\"label\":\"3. Identify state and Source of Truth\",\"description\":\"Define which systems own current facts, task progress and business state.\"},{\"label\":\"4. Define the tool surface\",\"description\":\"Expose the smallest set of clear capabilities required for the task.\"},{\"label\":\"5. Bind identity and permissions\",\"description\":\"Separate user authority, agent\u002Fruntime permissions and tool capabilities.\"},{\"label\":\"6. Choose autonomy boundaries\",\"description\":\"Specify what the model may decide dynamically and what remains deterministic.\"},{\"label\":\"7. Add approval checkpoints\",\"description\":\"Require review before consequential or irreversible actions where appropriate.\"},{\"label\":\"8. Define stopping and recovery\",\"description\":\"Set success proof, budgets, timeouts, retries, escalation and loop controls.\"},{\"label\":\"9. Design context\u002Fstate management\",\"description\":\"Keep current state, memory, tool observations and durable artifacts in the correct layers.\"},{\"label\":\"10. Trace the trajectory\",\"description\":\"Record enough execution structure to debug and audit model\u002Ftool decisions.\"},{\"label\":\"11. Evaluate realistic failures\",\"description\":\"Test stale state, tool errors, prompt injection, ambiguous requests and changed environments.\"},{\"label\":\"12. Expand autonomy only from evidence\",\"description\":\"Increase permissions or execution horizon when evaluation shows the benefit justifies the risk.\"}]},\"tunes\":{}},{\"id\":\"h-checklist\",\"type\":\"header\",\"data\":{\"text\":\"Agentic AI architecture checklist\",\"level\":2},\"tunes\":{}},{\"id\":\"checklist-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Question\",\"Expected evidence\"],[\"What proves success?\",\"External outcome, artifact, test or authoritative state.\"],[\"Why is an agent needed?\",\"The path genuinely depends on intermediate observations.\"],[\"Which decisions are model-driven?\",\"Explicit autonomy boundary.\"],[\"Which tools exist?\",\"Small, documented, unambiguous capability set.\"],[\"Who may use each tool?\",\"Identity- and context-aware authorization policy.\"],[\"Which actions need approval?\",\"Consequence-based review rules.\"],[\"Where does task state live?\",\"Application-owned state separate from transient model context.\"],[\"How does the agent recover?\",\"Retry, re-read, rollback, clarification and escalation behavior.\"],[\"How does it stop?\",\"Verified completion plus step\u002Ftime\u002Fcost limits.\"],[\"How are side effects protected?\",\"Validation, idempotency, least privilege and confirmation.\"],[\"Can execution be reconstructed?\",\"Tool, approval and state-transition traces.\"],[\"How is it evaluated?\",\"Outcome + trajectory + robustness tests.\"],[\"What changes after a model\u002Fruntime update?\",\"Regression suite for tool selection, permissions, stopping and recovery.\"]]},\"tunes\":{}},{\"id\":\"h-edge\",\"type\":\"header\",\"data\":{\"text\":\"Edge cases and limitations\",\"level\":2},\"tunes\":{}},{\"id\":\"p-edge-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Some systems are “agentic” only in a narrow routing sense: the model selects one specialist or tool and then the rest of the workflow is deterministic. That can still be useful, but it should not be described as equivalent to a long-running autonomous agent.\"},\"tunes\":{}},{\"id\":\"p-edge-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Highly consequential domains may intentionally restrict agent autonomy. An AI system can inspect evidence, prepare recommendations and fill structured forms while a human remains the only actor allowed to commit the final transaction.\"},\"tunes\":{}},{\"id\":\"p-edge-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"Some environments are well suited to agents because feedback is objective. Coding agents can run tests; infrastructure agents can inspect metrics; data agents can validate query results. Open-ended domains with weak feedback require more cautious evaluation.\"},\"tunes\":{}},{\"id\":\"p-edge-4\",\"type\":\"paragraph\",\"data\":{\"text\":\"An agent can operate entirely locally, entirely through managed cloud services or in a hybrid architecture. Agentic behavior describes control flow, not hosting location.\"},\"tunes\":{}},{\"id\":\"p-edge-5\",\"type\":\"paragraph\",\"data\":{\"text\":\"The term “reasoning” should not be used as proof that the agent's internal process is correct. Production assurance should rely on observable inputs, actions, outputs, state and evaluation rather than unverifiable claims about hidden reasoning.\"},\"tunes\":{}},{\"id\":\"h-change\",\"type\":\"header\",\"data\":{\"text\":\"What would change this answer?\",\"level\":2},\"tunes\":{}},{\"id\":\"p-change-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Vendor APIs and agent frameworks will continue to evolve, but the architecture boundary is stable: a model proposes decisions, a runtime manages the loop, tools connect to the environment, permissions constrain actions and external observations determine what actually happened.\"},\"tunes\":{}},{\"id\":\"p-change-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"As models become more reliable, systems may safely delegate longer horizons or more complex recovery behavior. As runtime verification and authorization improve, some approval steps may become automated. Those are changes in autonomy level, not changes to the fundamental responsibility layers.\"},\"tunes\":{}},{\"id\":\"p-change-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The recommended architecture also changes by consequence. A research agent that only reads public sources can tolerate different controls from an agent that writes production configuration or moves money.\"},\"tunes\":{}},{\"id\":\"h-related\",\"type\":\"header\",\"data\":{\"text\":\"Related canonical knowledge\",\"level\":2},\"tunes\":{}},{\"id\":\"p-related-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Agentic AI sits above several prerequisite layers: context engineering determines what the model sees; Source-of-Truth architecture determines which information is authoritative; retrieval supplies external evidence; runtime architecture determines what can execute.\"},\"tunes\":{}},{\"id\":\"p-related-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Downstream nodes include tool calling, MCP, A2A, agent identity, permissions, auditability, human-in-the-loop, orchestration, memory and multi-agent systems.\"},\"tunes\":{}},{\"id\":\"p-related-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The protocol stack article should therefore be read after the basic agent concept: protocols standardize boundaries around agents; they do not define agentic behavior itself.\"},\"tunes\":{}},{\"id\":\"h-faq\",\"type\":\"header\",\"data\":{\"text\":\"Frequently asked questions\",\"level\":2},\"tunes\":{}},{\"id\":\"faq\",\"type\":\"faq\",\"data\":{\"title\":\"Agentic AI FAQ\",\"items\":[{\"id\":\"faq1\",\"question\":\"What is agentic AI?\",\"answer\":\"Agentic AI is an AI system in which a model can pursue a goal over multiple steps by choosing actions or tools, observing results, updating its state and continuing until a stopping condition is reached.\"},{\"id\":\"faq2\",\"question\":\"What is the difference between an LLM and an AI agent?\",\"answer\":\"An LLM produces outputs from inputs. An agent combines a model with a runtime, tools, state, permissions, context management and an iterative execution loop.\"},{\"id\":\"faq3\",\"question\":\"Does tool calling make a system an agent?\",\"answer\":\"Not necessarily. A single tool-assisted model response can be bounded and non-agentic. Agentic behavior appears when tool observations drive an adaptive multi-step loop.\"},{\"id\":\"faq4\",\"question\":\"What is the difference between an agent and an AI workflow?\",\"answer\":\"A workflow usually follows a process path defined in application code. An agent has more model-driven control over which steps and tools to use based on intermediate observations.\"},{\"id\":\"faq5\",\"question\":\"Do agents need memory?\",\"answer\":\"No. Long-term memory is useful for persistent information across sessions, but many agents complete bounded tasks using only current task state and context.\"},{\"id\":\"faq6\",\"question\":\"Do AI agents need multiple agents?\",\"answer\":\"No. A single agent is often simpler. Multi-agent systems are justified when specialization, tool isolation, policy isolation or ownership boundaries materially improve the system.\"},{\"id\":\"faq7\",\"question\":\"Can an agent be human-in-the-loop?\",\"answer\":\"Yes. The agent can autonomously perform low-risk analysis and preparation while the runtime pauses for human approval before consequential actions.\"},{\"id\":\"faq8\",\"question\":\"Is MCP an agent framework?\",\"answer\":\"No. MCP is an interoperability protocol for exposing tools, resources and prompts. An agent runtime can use MCP, but still needs its own loop, state, authorization and evaluation.\"},{\"id\":\"faq9\",\"question\":\"How do you know an agent actually completed a task?\",\"answer\":\"Where possible, verify success through external state, tests, artifacts or authoritative system records rather than trusting the model's own completion statement.\"}]},\"tunes\":{}},{\"id\":\"h-glossary\",\"type\":\"header\",\"data\":{\"text\":\"Glossary\",\"level\":2},\"tunes\":{}},{\"id\":\"glossary\",\"type\":\"glossary\",\"data\":{\"title\":\"Key agentic AI terms\",\"entries\":[{\"term\":\"Agentic AI\",\"definition\":\"AI system behavior in which a model dynamically directs multi-step execution using tools, observations and state toward a goal.\",\"anchor\":\"agentic-ai\"},{\"term\":\"AI agent\",\"definition\":\"A model-centered system with runtime, tools, state and an execution loop that can pursue a task over multiple steps.\",\"anchor\":\"ai-agent\"},{\"term\":\"Agent loop\",\"definition\":\"Repeated cycle of model decision, tool\u002Faction execution, observation and updated model decision until stopping.\",\"anchor\":\"agent-loop\"},{\"term\":\"Runtime \u002F harness\",\"definition\":\"The execution layer that manages the model loop, tools, state, approvals, context, errors and stopping conditions.\",\"anchor\":\"runtime-harness\"},{\"term\":\"Tool\",\"definition\":\"A capability exposed to the model for reading information, computing, delegating or changing external state.\",\"anchor\":\"tool\"},{\"term\":\"Observation\",\"definition\":\"Information returned from a tool or environment and supplied to a later agent step.\",\"anchor\":\"observation\"},{\"term\":\"Agent state\",\"definition\":\"Persistent task or execution information that exists outside a single model output and may survive across steps or pauses.\",\"anchor\":\"agent-state\"},{\"term\":\"Autonomy boundary\",\"definition\":\"The explicit limit defining which decisions and actions the model may control dynamically.\",\"anchor\":\"autonomy-boundary\"},{\"term\":\"Human-in-the-loop\",\"definition\":\"A control pattern in which human review, input or approval is required at selected points in an AI-driven process.\",\"anchor\":\"human-in-the-loop\"},{\"term\":\"Trajectory\",\"definition\":\"The sequence of relevant states, decisions, tool calls, actions and observations between task request and final outcome.\",\"anchor\":\"trajectory\"},{\"term\":\"Idempotency\",\"definition\":\"Property that allows an operation to be repeated without unintentionally applying the same side effect multiple times.\",\"anchor\":\"idempotency\"}]},\"tunes\":{}},{\"id\":\"h-conclusion\",\"type\":\"header\",\"data\":{\"text\":\"Conclusion\",\"level\":2},\"tunes\":{}},{\"id\":\"p-conclusion-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Agentic AI is not simply a smarter model or a chatbot with more tools. It is a system architecture in which a model participates in an iterative control loop: decide, act, observe, update and continue.\"},\"tunes\":{}},{\"id\":\"p-conclusion-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The model provides flexible decision making, but the surrounding runtime must own execution reality: permissions, tool access, state, approvals, retries, budgets, stopping conditions, tracing and verification.\"},\"tunes\":{}},{\"id\":\"p-conclusion-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The most useful design principle is therefore: delegate tactical choice to the model only inside explicit technical and business boundaries. Agentic capability becomes production capability only when autonomy, authority and evidence remain separable.\"},\"tunes\":{}},{\"id\":\"h-sources\",\"type\":\"header\",\"data\":{\"text\":\"Primary sources and current guidance\",\"level\":2},\"tunes\":{}},{\"id\":\"p-sources-note\",\"type\":\"paragraph\",\"data\":{\"text\":\"The sources below support the current architectural distinctions around agents, workflows, loops, tools, orchestration, safety and evaluation. Project sections are original implementation evidence and are explicitly bounded to what the repositories demonstrate.\"},\"tunes\":{}},{\"id\":\"src-openai-agents\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fagents\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — Agents\",\"description\":\"Current developer guidance defining runtime choices for multi-step work, tools, state, orchestration and agent execution.\"}},\"tunes\":{}},{\"id\":\"src-openai-definitions\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fagents\u002Fdefine-agents\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — Agent definitions\",\"description\":\"Current documentation describing an agent as a model plus instructions and optional runtime behavior including tools, guardrails, MCP servers and handoffs.\"}},\"tunes\":{}},{\"id\":\"src-openai-running\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fagents\u002Frunning-agents\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — Running agents\",\"description\":\"Current documentation of the agent loop: model call, tool execution or handoff, continuation and final stopping point.\"}},\"tunes\":{}},{\"id\":\"src-openai-orchestration\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fagents\u002Forchestration\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — Orchestration and handoffs\",\"description\":\"Current guidance on handoffs, agents-as-tools and when specialist agents add useful ownership or capability boundaries.\"}},\"tunes\":{}},{\"id\":\"src-openai-safety\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fagent-builder-safety\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — Safety in building agents\",\"description\":\"Current safety guidance covering tool approvals, prompt injection, guardrails and trace-based evaluation.\"}},\"tunes\":{}},{\"id\":\"src-anthropic-agents\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Fbuilding-effective-agents\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Anthropic — Building effective agents\",\"description\":\"Engineering guidance distinguishing predefined workflows from model-directed agents and describing tool-based environmental feedback loops.\"}},\"tunes\":{}},{\"id\":\"src-anthropic-context\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Feffective-context-engineering-for-ai-agents\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Anthropic — Effective context engineering for AI agents\",\"description\":\"Practical framing of agents as LLMs autonomously using tools in a loop, with dynamic just-in-time context management.\"}},\"tunes\":{}},{\"id\":\"src-anthropic-evals\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Fdemystifying-evals-for-ai-agents\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Anthropic — Demystifying evals for AI agents\",\"description\":\"2026 guidance on evaluating multi-turn agents that call tools, modify state and adapt to intermediate results.\"}},\"tunes\":{}}],\"version\":\"2.31.6\"}",{"time":1578,"blocks":1579,"version":2703},1791487185746,[1580,1584,1589,1594,1599,1603,1607,1611,1615,1619,1623,1627,1631,1635,1661,1665,1669,1673,1677,1699,1703,1707,1739,1743,1747,1790,1794,1798,1802,1807,1811,1815,1819,1823,1827,1855,1859,1863,1867,1871,1875,1879,1883,1887,1891,1895,1899,1904,1908,1912,1916,1920,1924,1928,1935,1939,1943,1947,1951,1970,1974,1978,1982,1986,1990,2024,2028,2032,2036,2040,2044,2048,2052,2056,2060,2064,2068,2074,2078,2082,2086,2090,2097,2101,2140,2144,2148,2152,2156,2203,2207,2235,2239,2243,2247,2251,2255,2259,2263,2267,2271,2275,2279,2304,2309,2313,2356,2360,2397,2401,2442,2446,2491,2495,2499,2503,2507,2511,2515,2519,2523,2527,2531,2535,2539,2543,2547,2551,2583,2587,2623,2627,2631,2635,2639,2643,2647,2654,2661,2668,2675,2682,2689,2696],{"id":215,"data":1581,"type":218,"tunes":1583},{"text":1582},"Agentic AI is an AI system in which a model can pursue a goal across multiple steps by deciding what to do next, using tools or other capabilities, observing the results, updating its working state and continuing until it reaches a stopping condition. The model alone is not the agent. A usable agent also needs a runtime or harness that manages context, tool execution, state, permissions, approvals, errors and the loop between decisions and observations.",{},{"id":221,"data":1585,"type":226,"tunes":1588},{"body":1586,"title":1587,"variant":225},"A normal model call is usually \u003Cstrong>input → model → output\u003C\u002Fstrong>. An agentic system is closer to \u003Cstrong>goal → decision → tool\u002Faction → observation → updated decision → … → result\u003C\u002Fstrong>.\u003Cbr>\u003Cbr>The critical distinction is not whether an application uses an LLM or function calling. It is whether the system gives the model meaningful control over the next step of a multi-step process while a runtime constrains what the model is actually allowed to do.","Direct answer",{},{"id":229,"data":1590,"type":226,"tunes":1593},{"body":1591,"title":1592,"variant":233},"A model may know how to call a tool. A runtime may expose that tool. Neither fact means the current user or agent is authorized to execute the underlying business action. \u003Cstrong>Tool capability, tool permission and business authority are separate layers.\u003C\u002Fstrong>","Capability is not authority",{},{"id":236,"data":1595,"type":226,"tunes":1598},{"body":1596,"title":1597,"variant":240},"Agent terminology still varies across vendors and research communities. OpenAI currently defines agent runtimes around multi-step work, tools, state and orchestration. Anthropic's practical distinction remains useful: workflows follow predefined code paths, while agents dynamically direct their own process and tool use. This article therefore treats “agentic AI” as an architectural spectrum rather than one standardized product category.","Current-source note — 8 October 2026",{},{"id":243,"data":1600,"type":248,"tunes":1602},{"title":1601,"maxLevel":246,"minLevel":247},"Contents",{},{"id":251,"data":1604,"type":42,"tunes":1606},{"text":1605,"level":247},"What agentic AI really means",{},{"id":256,"data":1608,"type":218,"tunes":1610},{"text":1609},"The important shift from ordinary generative AI to agentic AI is control over process. A normal assistant can answer a question using the context it receives. An agent can decide that answering requires additional steps: inspect a file, search a repository, query an API, ask for clarification, run a test, update a ticket, delegate a subtask or retry after a failed action.",{},{"id":261,"data":1612,"type":218,"tunes":1614},{"text":1613},"This does not require unlimited autonomy. An agent can operate inside a narrow sandbox, under strict permissions, with approval required before every consequential action. The system is still agentic if the model dynamically chooses among permitted next steps.",{},{"id":266,"data":1616,"type":218,"tunes":1618},{"text":1617},"The architecture therefore matters more than the label. “Agent” should describe a system behavior: iterative model-driven decision making over tools, state and feedback — not merely a chatbot with a larger prompt.",{},{"id":271,"data":1620,"type":42,"tunes":1622},{"text":1621,"level":247},"The simplest example",{},{"id":276,"data":1624,"type":218,"tunes":1626},{"text":1625},"Suppose a developer asks an AI system: “Find why the test suite fails and fix the bug.” A single model call could only suggest likely causes from the text it was given.",{},{"id":281,"data":1628,"type":218,"tunes":1630},{"text":1629},"An agentic coding system can inspect the repository, search for the failing test, read relevant files, propose a change, edit the code, run the test, observe the failure, revise the implementation and run the test again.",{},{"id":286,"data":1632,"type":218,"tunes":1634},{"text":1633},"The agentic part is not simply that shell and file tools exist. It is that the model can use environmental feedback to choose the next step instead of following one completely predefined sequence.",{},{"id":291,"data":1636,"type":317,"tunes":1660},{"steps":1637,"title":1659,"orientation":316},[1638,1641,1644,1647,1650,1653,1656],{"label":1639,"description":1640},"1. Receive a goal","The user or upstream system defines the objective and relevant constraints.",{"label":1642,"description":1643},"2. Build current context","The runtime supplies instructions, state, history, memory, tools and current evidence.",{"label":1645,"description":1646},"3. Model decides next step","The model may answer, call a tool, request information, delegate or stop.",{"label":1648,"description":1649},"4. Runtime validates the request","Permissions, schemas, approvals and policy determine whether the proposed action may execute.",{"label":1651,"description":1652},"5. Execute tool or action","The external environment changes or returns new information.",{"label":1654,"description":1655},"6. Observe the result","The runtime feeds structured tool output, errors or state changes back into the next model step.",{"label":1657,"description":1658},"7. Continue or stop","The loop repeats until success, refusal, escalation, budget limit, timeout or another stopping condition.","The basic agent loop",{},{"id":320,"data":1662,"type":42,"tunes":1664},{"text":1663,"level":247},"Where the simple example stops",{},{"id":325,"data":1666,"type":218,"tunes":1668},{"text":1667},"Not every multi-step AI system is equally agentic. A workflow may use several LLM calls and tools while every step is predetermined in code. Another system may let the model decide which tool to call, in which order, how many times and when to stop.",{},{"id":330,"data":1670,"type":218,"tunes":1672},{"text":1671},"Both can be useful. The difference is where control lives. Predefined workflows put more control in application code. Agents move more tactical process decisions into the model\u002Fruntime loop.",{},{"id":335,"data":1674,"type":42,"tunes":1676},{"text":1675,"level":247},"Agent vs workflow",{},{"id":340,"data":1678,"type":369,"tunes":1698},{"rows":1679,"title":1692,"layout":361,"columns":1693},[1680,1683,1686,1689],{"id":344,"label":1681,"values":1682},"Process path",[347,347],{"id":349,"label":1684,"values":1685},"Tool sequence",[347,347],{"id":353,"label":1687,"values":1688},"Strength",[347,347],{"id":357,"label":1690,"values":1691},"Risk",[347,347],"Predefined workflow and agentic control",[1694,1696],{"id":364,"label":1695},"LLM workflow",{"id":367,"label":1697},"Agent",{},{"id":372,"data":1700,"type":218,"tunes":1702},{"text":1701},"Anthropic explicitly separates these two patterns: workflows orchestrate models and tools through predefined code paths, while agents let models dynamically direct their own processes and tool usage. This is not the only possible terminology, but it is a useful architecture boundary.",{},{"id":377,"data":1704,"type":42,"tunes":1706},{"text":1705,"level":247},"Agentic behavior is a spectrum, not a binary label",{},{"id":382,"data":1708,"type":361,"tunes":1738},{"content":1709,"stretched":43,"withHeadings":14},[1710,1714,1718,1722,1726,1730,1734],[1711,1712,1713],"Level","Example","Who decides the next step?",[1715,1716,1717],"Single model call","Summarize this document","Application calls model once",[1719,1720,1721],"Tool-assisted response","Model may use web search before answering","Model selects from bounded tools for one response",[1723,1724,1725],"Structured workflow","Classify → retrieve → generate → validate","Application workflow determines stages",[1727,1728,1729],"Adaptive workflow","Model can choose among several branches and retry","Shared control between application and model",[1731,1732,1733],"Agent loop","Model repeatedly chooses tools\u002Factions based on observations","Model directs tactical execution inside runtime constraints",[1735,1736,1737],"Long-running agent","Agent pauses, resumes, manages artifacts and continues","Model + persistent runtime manage evolving execution",{},{"id":415,"data":1740,"type":218,"tunes":1742},{"text":1741},"Calling every system above an “agent” can obscure important operational differences. The stronger the model's control over sequence, duration and actions, the more important runtime isolation, permissions, tracing, stopping conditions and trajectory evaluation become.",{},{"id":420,"data":1744,"type":42,"tunes":1746},{"text":1745,"level":247},"The minimum architecture of an agentic system",{},{"id":425,"data":1748,"type":361,"tunes":1789},{"content":1749,"stretched":43,"withHeadings":14},[1750,1753,1756,1759,1762,1765,1768,1771,1774,1777,1780,1783,1786],[1751,1752],"Component","Responsibility",[1754,1755],"Goal \u002F task","Defines what the system is trying to accomplish.",[1757,1758],"Model","Interprets context and decides the next action or output.",[1760,1761],"Instructions","Define role, constraints, priorities and task-specific policy.",[1763,1764],"Context assembler","Builds the information visible to the model on each step.",[1766,1767],"Tool catalog","Defines capabilities the model may request.",[1769,1770],"Runtime \u002F harness","Runs the loop, executes tools, manages state and handles stopping conditions.",[1772,1773],"Authorization layer","Determines whether a proposed action is permitted for the current principal.",[1775,1776],"State \u002F session","Preserves task progress across turns or execution steps.",[1778,1779],"Observation channel","Returns tool results and environment changes to the next model step.",[1781,1782],"Approvals \u002F human control","Pauses consequential actions where review is required.",[1784,1785],"Tracing \u002F audit","Records model calls, tools, transitions, approvals and failures.",[1787,1788],"Evaluation","Measures outcomes and execution trajectories against acceptance criteria.",{},{"id":469,"data":1791,"type":42,"tunes":1793},{"text":1792,"level":247},"A model is not an agent",{},{"id":474,"data":1795,"type":218,"tunes":1797},{"text":1796},"A language model produces outputs from inputs. It does not by itself own a filesystem, execute a shell command, maintain durable task state, enforce permissions or automatically call itself again.",{},{"id":479,"data":1799,"type":218,"tunes":1801},{"text":1800},"Those capabilities come from the surrounding runtime. The same model can behave as a simple chat model in one application and as the decision engine inside an agent loop in another.",{},{"id":484,"data":1803,"type":226,"tunes":1806},{"body":1804,"title":1805,"variant":488},"\u003Cstrong>Model capability determines what decisions can be proposed. Runtime architecture determines what can actually happen.\u003C\u002Fstrong>","Architecture rule",{},{"id":491,"data":1808,"type":42,"tunes":1810},{"text":1809,"level":247},"Tool use is central — but tool use alone does not make an agent",{},{"id":496,"data":1812,"type":218,"tunes":1814},{"text":1813},"Tools let the model acquire information and affect external systems. Examples include database reads, file operations, shell execution, web search, browser control, API calls, ticket updates or delegated specialist agents.",{},{"id":501,"data":1816,"type":218,"tunes":1818},{"text":1817},"A single model call can use one tool and still remain a bounded tool-assisted response rather than a long-running agent. Agentic behavior appears when tool observations feed an adaptive loop in which the model chooses what to do next.",{},{"id":506,"data":1820,"type":218,"tunes":1822},{"text":1821},"Tool design matters because tools are the contract between model reasoning and external reality. Ambiguous or overlapping tools create routing errors; large unstructured outputs pollute context; broad side-effect tools increase blast radius.",{},{"id":511,"data":1824,"type":42,"tunes":1826},{"text":1825,"level":247},"Tool capability, permission and authority are different",{},{"id":516,"data":1828,"type":361,"tunes":1854},{"content":1829,"stretched":43,"withHeadings":14},[1830,1833,1836,1839,1842,1845,1848,1851],[1831,1832],"Layer","Question",[1834,1835],"Capability","Can this runtime technically perform the operation?",[1837,1838],"Tool exposure","Is that capability available to this agent?",[1840,1841],"Permission","May this agent\u002Fsession use it under the current policy?",[1843,1844],"User authorization","Is the requesting principal allowed to cause this operation?",[1846,1847],"Business authority","Is the operation valid under domain rules, approvals and limits?",[1849,1850],"Execution","Did the operation actually occur?",[1852,1853],"Audit","Can the system prove who requested, approved and executed it?",{},{"id":544,"data":1856,"type":218,"tunes":1858},{"text":1857},"These layers are frequently collapsed in prototypes. A model sees a refund tool and therefore appears able to issue refunds. In production, the tool should still validate account, user, transaction, amount, policy and approval conditions independently of the model's request.",{},{"id":549,"data":1860,"type":42,"tunes":1862},{"text":1861,"level":247},"The runtime or harness is the actual execution system",{},{"id":554,"data":1864,"type":218,"tunes":1866},{"text":1865},"OpenAI's current agent documentation makes the runtime distinction explicit. Different runtimes can manage orchestration, state, tools, sandboxes and execution in different places, while the model remains only one part of the system.",{},{"id":559,"data":1868,"type":218,"tunes":1870},{"text":1869},"The Agents SDK describes a loop that repeatedly calls the current model, inspects the output, executes requested tools or handoffs, and continues until the model returns a final answer or another real stopping point.",{},{"id":564,"data":1872,"type":218,"tunes":1874},{"text":1873},"This means agent architecture decisions include where orchestration runs, where state lives, who executes tools, which sandbox contains side effects, and who owns retries, timeouts and resumability.",{},{"id":569,"data":1876,"type":42,"tunes":1878},{"text":1877,"level":247},"Planning is useful, but an explicit plan is not required",{},{"id":574,"data":1880,"type":218,"tunes":1882},{"text":1881},"Agents are often described as systems that “plan.” In practice, planning can be explicit or implicit. An agent may first produce a visible multi-step plan, or it may choose one next action at a time and revise after every observation.",{},{"id":579,"data":1884,"type":218,"tunes":1886},{"text":1885},"For highly uncertain tasks, short-horizon planning can be safer because the environment can invalidate a long plan. The architectural requirement is the ability to choose and revise actions based on the goal, current state and new evidence.",{},{"id":584,"data":1888,"type":42,"tunes":1890},{"text":1889,"level":247},"Environmental feedback is what makes the loop useful",{},{"id":589,"data":1892,"type":218,"tunes":1894},{"text":1893},"An agent becomes operationally meaningful when it can observe whether its action worked. Tool output, test results, API responses, filesystem state, browser state and application records provide external evidence that the system can use to revise its next decision.",{},{"id":594,"data":1896,"type":218,"tunes":1898},{"text":1897},"Anthropic's agent guidance emphasizes this feedback loop: agents use tools, obtain ground truth from the environment, assess progress and continue or request human input.",{},{"id":599,"data":1900,"type":226,"tunes":1903},{"body":1901,"title":1902,"variant":233},"An agent saying “the task is complete” does not prove completion. Where possible, verify the final state through an external system, test, file, transaction record or other observable outcome.","Self-report is not environmental proof",{},{"id":605,"data":1905,"type":42,"tunes":1907},{"text":1906,"level":247},"Agent state is not the same as model context",{},{"id":610,"data":1909,"type":218,"tunes":1911},{"text":1910},"A long-running task may need state that cannot or should not remain in the model context: task IDs, checkpoints, artifacts, approvals, external object identifiers, retry counters and workflow status.",{},{"id":615,"data":1913,"type":218,"tunes":1915},{"text":1914},"The runtime can preserve this durable state outside the model window and reconstruct the context required for the next step. This keeps model-visible context focused while maintaining continuity and resumability.",{},{"id":620,"data":1917,"type":42,"tunes":1919},{"text":1918,"level":247},"Memory is optional, not the definition of an agent",{},{"id":625,"data":1921,"type":218,"tunes":1923},{"text":1922},"An agent can operate successfully without long-term memory if the complete task fits inside one bounded run. Memory becomes useful when information must persist across sessions, tasks or long execution horizons.",{},{"id":630,"data":1925,"type":218,"tunes":1927},{"text":1926},"RAG, memory, state and context solve different problems. Treating a vector database as “the agent memory” or conversation history as “the state machine” usually hides important lifecycle and authority boundaries.",{},{"id":635,"data":1929,"type":641,"tunes":1934},{"url":1930,"title":1931,"excerpt":1932,"ctaLabel":1933},"https:\u002F\u002Fstajic.de\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context","AI Agent Memory Is Not RAG: How to Separate Memory, Retrieval, State and Context","A practical architecture separating persistent memory, authoritative application state, retrieval and the context supplied to the model.","Read the memory architecture article",{},{"id":644,"data":1936,"type":42,"tunes":1938},{"text":1937,"level":247},"Context engineering becomes dynamic in agents",{},{"id":649,"data":1940,"type":218,"tunes":1942},{"text":1941},"Every tool call can produce new context. Every step can also make earlier information obsolete. A strong agent runtime therefore rebuilds or curates context as execution progresses rather than replaying everything indefinitely.",{},{"id":654,"data":1944,"type":218,"tunes":1946},{"text":1945},"Tool definitions, task state, retrieved evidence, observations and memory all compete for the model's attention. Long-running agents need trimming, compaction or just-in-time loading so context remains relevant to the current decision.",{},{"id":659,"data":1948,"type":42,"tunes":1950},{"text":1949,"level":247},"Read tools and side-effect tools have different risk",{},{"id":664,"data":1952,"type":369,"tunes":1969},{"rows":1953,"title":1963,"layout":361,"columns":1964},[1954,1957,1960],{"id":668,"label":1955,"values":1956},"Examples",[347,347],{"id":671,"label":1958,"values":1959},"Main risk",[347,347],{"id":675,"label":1961,"values":1962},"Typical control",[347,347],"Information access versus external action",[1965,1967],{"id":681,"label":1966},"Read \u002F observe",{"id":684,"label":1968},"Write \u002F act",{},{"id":688,"data":1971,"type":42,"tunes":1973},{"text":1972,"level":247},"Human-in-the-loop is a control mechanism, not the opposite of agentic AI",{},{"id":693,"data":1975,"type":218,"tunes":1977},{"text":1976},"An agent does not stop being agentic because a human approves consequential steps. The model can still autonomously inspect, reason, search and prepare an action while the runtime requires human confirmation before execution.",{},{"id":698,"data":1979,"type":218,"tunes":1981},{"text":1980},"OpenAI's current agent safety guidance explicitly recommends approvals for tool operations in higher-risk workflows. Anthropic likewise emphasizes checkpoints and human judgment where agents encounter blockers or consequential decisions.",{},{"id":703,"data":1983,"type":218,"tunes":1985},{"text":1984},"The useful architecture question is not “human or autonomous?” but which decisions can be delegated, which require review and which must remain deterministic?",{},{"id":708,"data":1987,"type":42,"tunes":1989},{"text":1988,"level":247},"Agents need explicit stopping conditions",{},{"id":713,"data":1991,"type":361,"tunes":2023},{"content":1992,"stretched":43,"withHeadings":14},[1993,1996,1999,2002,2005,2008,2011,2014,2017,2020],[1994,1995],"Stopping condition","Purpose",[1997,1998],"Successful verified outcome","End when the external target state is confirmed.",[2000,2001],"Maximum steps","Prevent runaway loops.",[2003,2004],"Time budget","Bound wall-clock execution.",[2006,2007],"Cost\u002Ftoken budget","Limit resource consumption.",[2009,2010],"Repeated-action detector","Stop loops that are no longer making progress.",[2012,2013],"Permission boundary","Pause or stop when the next required action is not permitted.",[2015,2016],"Human approval checkpoint","Wait before consequential execution.",[2018,2019],"Unrecoverable tool failure","Escalate instead of retrying indefinitely.",[2021,2022],"Uncertainty threshold","Ask for clarification when the task cannot be safely inferred.",{},{"id":748,"data":2025,"type":42,"tunes":2027},{"text":2026,"level":247},"Recovery is part of agent behavior",{},{"id":753,"data":2029,"type":218,"tunes":2031},{"text":2030},"Agents operate in environments that fail: APIs time out, files change, credentials expire, webpages move and tools return malformed output. A useful agentic system therefore needs recovery behavior, not just a happy-path tool loop.",{},{"id":758,"data":2033,"type":218,"tunes":2035},{"text":2034},"Recovery can include retry with limits, choosing another tool, re-reading current state, asking the user, rolling back a partial action or escalating to a human.",{},{"id":763,"data":2037,"type":218,"tunes":2039},{"text":2038},"Retries also need idempotency awareness. Repeating a read is usually low risk; repeating a payment or message send can create duplicate side effects.",{},{"id":768,"data":2041,"type":42,"tunes":2043},{"text":2042,"level":247},"Agentic AI does not require multiple agents",{},{"id":773,"data":2045,"type":218,"tunes":2047},{"text":2046},"A single agent with a clear tool set is often simpler and easier to evaluate than a multi-agent architecture. Multiple agents are useful when specialization materially improves tool isolation, policy isolation, prompt clarity, ownership or trace legibility.",{},{"id":778,"data":2049,"type":218,"tunes":2051},{"text":2050},"OpenAI's current orchestration guidance explicitly recommends starting with one agent where possible and adding specialists only when the contract or ownership boundary materially changes.",{},{"id":783,"data":2053,"type":218,"tunes":2055},{"text":2054},"Multi-agent systems add new problems: delegation quality, duplicated context, conflicting state, handoff semantics, identity, cost and distributed failure handling.",{},{"id":788,"data":2057,"type":42,"tunes":2059},{"text":2058,"level":247},"Agent protocols are interoperability layers, not the agent itself",{},{"id":793,"data":2061,"type":218,"tunes":2063},{"text":2062},"Protocols such as MCP and A2A can make an agent architecture interoperable, but they do not create the agent loop by themselves. MCP can expose tools and resources. A2A can connect independently implemented agents. The application still needs runtime, authorization, state, evaluation and domain logic.",{},{"id":798,"data":2065,"type":218,"tunes":2067},{"text":2066},"This is why protocol capability must remain separate from business authority. Discovering a tool through MCP does not prove the current principal is allowed to use it. Receiving a task through A2A does not prove the remote agent may perform every requested action.",{},{"id":803,"data":2069,"type":641,"tunes":2073},{"url":805,"title":2070,"excerpt":2071,"ctaLabel":2072},"MCP vs A2A vs UCP vs AP2 vs A2UI: The Agent Protocol Stack Explained","A protocol-responsibility map showing why tool access, agent collaboration, commerce, payment authority and agent-driven UI belong to different interoperability boundaries.","Read the agent protocol stack",{},{"id":811,"data":2075,"type":42,"tunes":2077},{"text":2076,"level":247},"The trajectory is part of agent reliability",{},{"id":816,"data":2079,"type":218,"tunes":2081},{"text":2080},"A final answer is insufficient evidence for an agentic system because an agent can reach the right result through an unsafe or invalid path. It may use an unauthorized tool, skip a required check, retry a side effect, rely on stale state or accidentally succeed.",{},{"id":821,"data":2083,"type":218,"tunes":2085},{"text":2084},"Evaluation therefore needs execution traces: decisions, tool calls, approvals, observations, state changes and final outcome. Current OpenAI safety guidance recommends trace graders and evals; Anthropic's 2026 agent-evaluation guidance similarly treats multi-turn tool trajectories as first-class evaluation objects.",{},{"id":826,"data":2087,"type":218,"tunes":2089},{"text":2088},"The stronger reliability question is: Did the agent reach an acceptable outcome through an acceptable, recoverable and auditable trajectory?",{},{"id":831,"data":2091,"type":641,"tunes":2096},{"url":2092,"title":2093,"excerpt":2094,"ctaLabel":2095},"https:\u002F\u002Fstajic.de\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough","AI Agent Reliability: Why the Final Answer Is Not Enough","Why production evaluation must inspect trajectories, tool use, state transitions and recoverability rather than only final answers.","Read the reliability article",{},{"id":839,"data":2098,"type":42,"tunes":2100},{"text":2099,"level":247},"Agentic systems increase the security surface",{},{"id":844,"data":2102,"type":361,"tunes":2139},{"content":2103,"stretched":43,"withHeadings":14},[2104,2107,2111,2115,2119,2123,2127,2131,2135],[1690,2105,2106],"Why agents amplify it","Architecture response",[2108,2109,2110],"Prompt injection","Untrusted content can influence future tool decisions","Separate instructions from data; constrain tools; sanitize or structure external input where possible",[2112,2113,2114],"Excessive permissions","Reasoning errors can become real side effects","Least privilege, scoped credentials, per-tool policy and approvals",[2116,2117,2118],"Credential exposure","Tools may need powerful secrets","Keep secrets outside model context; broker access through trusted runtime",[2120,2121,2122],"Confused deputy","Agent may act with authority broader than the requesting user","Bind execution to user\u002Fservice identity and re-authorize consequential actions",[2124,2125,2126],"Runaway loops","Model repeatedly calls tools without progress","Step, time and cost budgets plus loop detection",[2128,2129,2130],"State drift","Environment changes after the agent formed a plan","Re-read authoritative state before consequential actions",[2132,2133,2134],"Indirect injection","Tool\u002Fweb\u002Fdocument content contains instructions aimed at the model","Treat external content as untrusted data, not instruction authority",[2136,2137,2138],"Audit gap","Final result cannot show what was executed","Trace tool calls, approvals, identities and state changes",{},{"id":884,"data":2141,"type":42,"tunes":2143},{"text":2142,"level":247},"Agent observability must follow the loop",{},{"id":889,"data":2145,"type":218,"tunes":2147},{"text":2146},"Traditional service observability records requests, latency and errors. Agent observability needs an additional execution model: which agent was active, which model version made the decision, what context was available, which tool was selected, what arguments were sent, what result came back and why execution stopped.",{},{"id":894,"data":2149,"type":218,"tunes":2151},{"text":2150},"For sensitive systems, traces themselves require access control and retention policy because prompts, tool outputs and artifacts can contain confidential data.",{},{"id":899,"data":2153,"type":42,"tunes":2155},{"text":2154,"level":247},"How to evaluate an agentic system",{},{"id":904,"data":2157,"type":361,"tunes":2202},{"content":2158,"stretched":43,"withHeadings":14},[2159,2162,2166,2170,2174,2178,2182,2186,2190,2194,2198],[2160,1832,2161],"Dimension","Example evidence",[2163,2164,2165],"Task success","Did the requested outcome occur?","External state, tests, business outcome",[2167,2168,2169],"Trajectory quality","Were the steps acceptable?","Tool\u002Faction trace",[2171,2172,2173],"Tool selection","Did the agent choose appropriate capabilities?","Expected vs actual tool calls",[2175,2176,2177],"Permission adherence","Did it stay inside allowed authority?","Authorization logs and denied-action tests",[2179,2180,2181],"State handling","Did it use current authoritative state?","Freshness checks and state-change tests",[2183,2184,2185],"Recovery","Did it respond correctly to failures?","Injected timeout\u002Ferror scenarios",[2187,2188,2189],"Stopping behavior","Did it stop at the right point?","Step counts, loop detection, final-state proof",[2191,2192,2193],"Human escalation","Did it ask when review was required?","Approval\u002Fescalation traces",[2195,2196,2197],"Cost\u002Flatency","Was autonomy worth the operational cost?","Tokens, tool calls, duration",[2199,2200,2201],"Robustness","Does it survive realistic environment variation?","Repeated and adversarial trials",{},{"id":952,"data":2204,"type":42,"tunes":2206},{"text":2205,"level":247},"When an agent is appropriate",{},{"id":957,"data":2208,"type":361,"tunes":2234},{"content":2209,"stretched":43,"withHeadings":14},[2210,2213,2216,2219,2222,2225,2228,2231],[2211,2212],"Use an agent when","Prefer a workflow or simple call when",[2214,2215],"The number or order of steps cannot be known reliably in advance","The sequence is stable and deterministic",[2217,2218],"The system must inspect the environment and adapt","A single retrieval + generation step is sufficient",[2220,2221],"Several tools may be useful depending on intermediate results","One known API call solves the task",[2223,2224],"The task benefits from iterative verification or repair","The answer can be produced directly from supplied context",[2226,2227],"Failures require flexible recovery behavior","Failure branches are simple and can be encoded explicitly",[2229,2230],"Human review can be inserted at meaningful checkpoints","Every step is high-risk and must be manually controlled anyway",[2232,2233],"Expected value justifies extra latency, cost and complexity","Predictability and low cost matter more than flexibility",{},{"id":986,"data":2236,"type":218,"tunes":2238},{"text":2237},"A strong default is to start with the simplest solution that works and increase agentic complexity only when flexibility produces measurable value. Agents trade predictability, latency and cost for adaptive execution.",{},{"id":991,"data":2240,"type":42,"tunes":2242},{"text":2241,"level":247},"Original implementation evidence",{},{"id":996,"data":2244,"type":42,"tunes":2246},{"text":2245,"level":246},"Aaasaasa AI Client: model, runtime and permission are separate",{},{"id":1001,"data":2248,"type":218,"tunes":2250},{"text":2249},"Aaasaasa AI Client explicitly separates agent\u002Fclient, provider, model, runtime location and permissions. Its architecture documentation treats permissions as central tool\u002Fworkspace policy rather than a model property.",{},{"id":1006,"data":2252,"type":218,"tunes":2254},{"text":2253},"The same application can expose Direct Chat with no filesystem or shell tools while a Codex runtime operates under a selected workspace and permission profile. This demonstrates a core agentic architecture boundary: changing the runtime\u002Ftool surface changes what the system can do even when model access remains available.",{},{"id":1011,"data":2256,"type":218,"tunes":2258},{"text":2257},"The repository also distinguishes a local Codex runtime from model location: a local runtime can call a cloud model. This prevents the common mistake of equating “agent runs locally” with “inference is local.”",{},{"id":1016,"data":2260,"type":218,"tunes":2262},{"text":2261},"The implementation disables embedded execution paths whose approval semantics do not satisfy the required permission model. This supports the principle that agent capability should not bypass runtime authorization simply because an underlying framework can execute tools.",{},{"id":1021,"data":2264,"type":42,"tunes":2266},{"text":2265,"level":246},"Source of Truth Research Engine: bounded agentic research stages",{},{"id":1026,"data":2268,"type":218,"tunes":2270},{"text":2269},"The Source of Truth Research Engine uses a bounded research pipeline: discover → acquire → extract → verify → contradict → synthesize. Research jobs can execute through an AI runtime while evidence, sources, claims and contradictions remain in an external persistent store.",{},{"id":1031,"data":2272,"type":218,"tunes":2274},{"text":2273},"This is intentionally more controlled than an unconstrained autonomous research agent. The stages provide guardrails around what kind of work should happen next while still allowing model-driven research inside each bounded task.",{},{"id":1036,"data":2276,"type":218,"tunes":2278},{"text":2277},"That distinction is useful evidence for agent design: autonomy can be placed inside a structured delivery envelope rather than applied uniformly to the entire process.",{},{"id":1041,"data":2280,"type":361,"tunes":2303},{"content":2281,"stretched":43,"withHeadings":14},[2282,2285,2288,2291,2294,2297,2300],[2283,2284],"Implemented pattern","Agentic architecture lesson",[2286,2287],"Direct Chat has no OS tools","A model can exist without agentic execution capability.",[2289,2290],"Codex runtime has workspace permission profile","Tool authority belongs to runtime policy, not model capability.",[2292,2293],"Provider\u002Fmodel\u002Fruntime are separate concepts","Agent harness location and inference location are independent decisions.",[2295,2296],"Permission broker for tool-capable runtimes","Capability exposure can be centralized and governed.",[2298,2299],"Bounded research stages","Autonomy can operate inside explicit process boundaries.",[2301,2302],"Persistent claims\u002Fevidence outside model context","Agent state and evidence do not need to live only in conversation history.",{},{"id":1067,"data":2305,"type":226,"tunes":2308},{"body":2306,"title":2307,"variant":240},"These projects demonstrate concrete agent\u002Fruntime, permission and bounded-research patterns. They are not presented as proof of large-scale commercial autonomous-agent deployment.","Evidence boundary",{},{"id":1073,"data":2310,"type":42,"tunes":2312},{"text":2311,"level":247},"Common agentic AI failure modes",{},{"id":1078,"data":2314,"type":361,"tunes":2355},{"content":2315,"stretched":43,"withHeadings":14},[2316,2319,2322,2325,2328,2331,2334,2337,2340,2343,2346,2349,2352],[2317,2318],"Failure mode","What actually failed",[2320,2321],"“Agent” is only a chatbot with tools listed in the prompt","No reliable runtime loop or tool execution architecture exists",[2323,2324],"Tool support is treated as permission","Capability and authorization boundaries are collapsed",[2326,2327],"Agent trusts its own completion statement","Outcome is not verified against external state",[2329,2330],"Every task becomes multi-agent","Complexity increases without a real ownership or specialization boundary",[2332,2333],"Conversation history is used as durable state","Resumability and authoritative state become fragile",[2335,2336],"Agent retries side effects blindly","Duplicate messages, payments or state changes become possible",[2338,2339],"No step\u002Fcost limits","Agent can loop indefinitely or consume uncontrolled resources",[2341,2342],"Tool output is trusted as instruction","Indirect prompt injection can redirect behavior",[2344,2345],"Correct final answer is the only evaluation","Unsafe or invalid trajectories remain invisible",[2347,2348],"Model upgrade is treated as transparent","Tool selection, planning and stopping behavior can change",[2350,2351],"One broad tool exposes many privileged operations","Blast radius increases and intent becomes harder to validate",[2353,2354],"Human approval exists but reviewer lacks context","Approval becomes ceremonial rather than effective",{},{"id":1122,"data":2357,"type":42,"tunes":2359},{"text":2358,"level":247},"Common misconceptions",{},{"id":1127,"data":2361,"type":361,"tunes":2396},{"content":2362,"stretched":43,"withHeadings":14},[2363,2366,2369,2372,2375,2378,2381,2384,2387,2390,2393],[2364,2365],"Misconception","Correction",[2367,2368],"“An LLM is an agent.”","The model is the decision component; the agent is the surrounding system that manages tools, state and iteration.",[2370,2371],"“Tool calling automatically means agentic AI.”","A single bounded tool call may not involve an adaptive multi-step agent loop.",[2373,2374],"“Agents must be fully autonomous.”","Agentic systems can require approvals and operate under narrow permission boundaries.",[2376,2377],"“Agents need long-term memory.”","Memory is optional; many useful agents complete bounded tasks without cross-session memory.",[2379,2380],"“Agents must create a written plan first.”","Planning can be explicit or implicit and can occur one step at a time.",[2382,2383],"“Multi-agent is more advanced than single-agent.”","It is more complex; use it only when specialization or ownership boundaries justify it.",[2385,2386],"“MCP creates an agent.”","MCP exposes tools\u002Fresources; the runtime still needs an agent loop and authorization model.",[2388,2389],"“A local runtime means the model is local.”","Runtime location and inference\u002Fprovider location are separate.",[2391,2392],"“If the final result is correct, the agent worked correctly.”","An unsafe or unauthorized trajectory can still produce a correct result.",[2394,2395],"“Human approval removes autonomy.”","Approval can constrain selected actions while the rest of the process remains model-directed.",{},{"id":1165,"data":2398,"type":42,"tunes":2400},{"text":2399,"level":247},"A practical agent design sequence",{},{"id":1170,"data":2402,"type":317,"tunes":2441},{"steps":2403,"title":2440,"orientation":316},[2404,2407,2410,2413,2416,2419,2422,2425,2428,2431,2434,2437],{"label":2405,"description":2406},"1. Define the outcome","State what external result or artifact proves task success.",{"label":2408,"description":2409},"2. Decide whether an agent is actually needed","Prefer a simple call or deterministic workflow when the path is predictable.",{"label":2411,"description":2412},"3. Identify state and Source of Truth","Define which systems own current facts, task progress and business state.",{"label":2414,"description":2415},"4. Define the tool surface","Expose the smallest set of clear capabilities required for the task.",{"label":2417,"description":2418},"5. Bind identity and permissions","Separate user authority, agent\u002Fruntime permissions and tool capabilities.",{"label":2420,"description":2421},"6. Choose autonomy boundaries","Specify what the model may decide dynamically and what remains deterministic.",{"label":2423,"description":2424},"7. Add approval checkpoints","Require review before consequential or irreversible actions where appropriate.",{"label":2426,"description":2427},"8. Define stopping and recovery","Set success proof, budgets, timeouts, retries, escalation and loop controls.",{"label":2429,"description":2430},"9. Design context\u002Fstate management","Keep current state, memory, tool observations and durable artifacts in the correct layers.",{"label":2432,"description":2433},"10. Trace the trajectory","Record enough execution structure to debug and audit model\u002Ftool decisions.",{"label":2435,"description":2436},"11. Evaluate realistic failures","Test stale state, tool errors, prompt injection, ambiguous requests and changed environments.",{"label":2438,"description":2439},"12. Expand autonomy only from evidence","Increase permissions or execution horizon when evaluation shows the benefit justifies the risk.","Design the agent from authority outward",{},{"id":1212,"data":2443,"type":42,"tunes":2445},{"text":2444,"level":247},"Agentic AI architecture checklist",{},{"id":1217,"data":2447,"type":361,"tunes":2490},{"content":2448,"stretched":43,"withHeadings":14},[2449,2451,2454,2457,2460,2463,2466,2469,2472,2475,2478,2481,2484,2487],[1832,2450],"Expected evidence",[2452,2453],"What proves success?","External outcome, artifact, test or authoritative state.",[2455,2456],"Why is an agent needed?","The path genuinely depends on intermediate observations.",[2458,2459],"Which decisions are model-driven?","Explicit autonomy boundary.",[2461,2462],"Which tools exist?","Small, documented, unambiguous capability set.",[2464,2465],"Who may use each tool?","Identity- and context-aware authorization policy.",[2467,2468],"Which actions need approval?","Consequence-based review rules.",[2470,2471],"Where does task state live?","Application-owned state separate from transient model context.",[2473,2474],"How does the agent recover?","Retry, re-read, rollback, clarification and escalation behavior.",[2476,2477],"How does it stop?","Verified completion plus step\u002Ftime\u002Fcost limits.",[2479,2480],"How are side effects protected?","Validation, idempotency, least privilege and confirmation.",[2482,2483],"Can execution be reconstructed?","Tool, approval and state-transition traces.",[2485,2486],"How is it evaluated?","Outcome + trajectory + robustness tests.",[2488,2489],"What changes after a model\u002Fruntime update?","Regression suite for tool selection, permissions, stopping and recovery.",{},{"id":1263,"data":2492,"type":42,"tunes":2494},{"text":2493,"level":247},"Edge cases and limitations",{},{"id":1268,"data":2496,"type":218,"tunes":2498},{"text":2497},"Some systems are “agentic” only in a narrow routing sense: the model selects one specialist or tool and then the rest of the workflow is deterministic. That can still be useful, but it should not be described as equivalent to a long-running autonomous agent.",{},{"id":1273,"data":2500,"type":218,"tunes":2502},{"text":2501},"Highly consequential domains may intentionally restrict agent autonomy. An AI system can inspect evidence, prepare recommendations and fill structured forms while a human remains the only actor allowed to commit the final transaction.",{},{"id":1278,"data":2504,"type":218,"tunes":2506},{"text":2505},"Some environments are well suited to agents because feedback is objective. Coding agents can run tests; infrastructure agents can inspect metrics; data agents can validate query results. Open-ended domains with weak feedback require more cautious evaluation.",{},{"id":1283,"data":2508,"type":218,"tunes":2510},{"text":2509},"An agent can operate entirely locally, entirely through managed cloud services or in a hybrid architecture. Agentic behavior describes control flow, not hosting location.",{},{"id":1288,"data":2512,"type":218,"tunes":2514},{"text":2513},"The term “reasoning” should not be used as proof that the agent's internal process is correct. Production assurance should rely on observable inputs, actions, outputs, state and evaluation rather than unverifiable claims about hidden reasoning.",{},{"id":1293,"data":2516,"type":42,"tunes":2518},{"text":2517,"level":247},"What would change this answer?",{},{"id":1298,"data":2520,"type":218,"tunes":2522},{"text":2521},"Vendor APIs and agent frameworks will continue to evolve, but the architecture boundary is stable: a model proposes decisions, a runtime manages the loop, tools connect to the environment, permissions constrain actions and external observations determine what actually happened.",{},{"id":1303,"data":2524,"type":218,"tunes":2526},{"text":2525},"As models become more reliable, systems may safely delegate longer horizons or more complex recovery behavior. As runtime verification and authorization improve, some approval steps may become automated. Those are changes in autonomy level, not changes to the fundamental responsibility layers.",{},{"id":1308,"data":2528,"type":218,"tunes":2530},{"text":2529},"The recommended architecture also changes by consequence. A research agent that only reads public sources can tolerate different controls from an agent that writes production configuration or moves money.",{},{"id":1313,"data":2532,"type":42,"tunes":2534},{"text":2533,"level":247},"Related canonical knowledge",{},{"id":1318,"data":2536,"type":218,"tunes":2538},{"text":2537},"Agentic AI sits above several prerequisite layers: context engineering determines what the model sees; Source-of-Truth architecture determines which information is authoritative; retrieval supplies external evidence; runtime architecture determines what can execute.",{},{"id":1323,"data":2540,"type":218,"tunes":2542},{"text":2541},"Downstream nodes include tool calling, MCP, A2A, agent identity, permissions, auditability, human-in-the-loop, orchestration, memory and multi-agent systems.",{},{"id":1328,"data":2544,"type":218,"tunes":2546},{"text":2545},"The protocol stack article should therefore be read after the basic agent concept: protocols standardize boundaries around agents; they do not define agentic behavior itself.",{},{"id":1333,"data":2548,"type":42,"tunes":2550},{"text":2549,"level":247},"Frequently asked questions",{},{"id":1338,"data":2552,"type":1338,"tunes":2582},{"items":2553,"title":2581},[2554,2557,2560,2563,2566,2569,2572,2575,2578],{"id":1342,"answer":2555,"question":2556},"Agentic AI is an AI system in which a model can pursue a goal over multiple steps by choosing actions or tools, observing results, updating its state and continuing until a stopping condition is reached.","What is agentic AI?",{"id":1346,"answer":2558,"question":2559},"An LLM produces outputs from inputs. An agent combines a model with a runtime, tools, state, permissions, context management and an iterative execution loop.","What is the difference between an LLM and an AI agent?",{"id":1350,"answer":2561,"question":2562},"Not necessarily. A single tool-assisted model response can be bounded and non-agentic. Agentic behavior appears when tool observations drive an adaptive multi-step loop.","Does tool calling make a system an agent?",{"id":1354,"answer":2564,"question":2565},"A workflow usually follows a process path defined in application code. An agent has more model-driven control over which steps and tools to use based on intermediate observations.","What is the difference between an agent and an AI workflow?",{"id":1358,"answer":2567,"question":2568},"No. Long-term memory is useful for persistent information across sessions, but many agents complete bounded tasks using only current task state and context.","Do agents need memory?",{"id":1362,"answer":2570,"question":2571},"No. A single agent is often simpler. Multi-agent systems are justified when specialization, tool isolation, policy isolation or ownership boundaries materially improve the system.","Do AI agents need multiple agents?",{"id":1366,"answer":2573,"question":2574},"Yes. The agent can autonomously perform low-risk analysis and preparation while the runtime pauses for human approval before consequential actions.","Can an agent be human-in-the-loop?",{"id":1370,"answer":2576,"question":2577},"No. MCP is an interoperability protocol for exposing tools, resources and prompts. An agent runtime can use MCP, but still needs its own loop, state, authorization and evaluation.","Is MCP an agent framework?",{"id":1374,"answer":2579,"question":2580},"Where possible, verify success through external state, tests, artifacts or authoritative system records rather than trusting the model's own completion statement.","How do you know an agent actually completed a task?","Agentic AI FAQ",{},{"id":1380,"data":2584,"type":42,"tunes":2586},{"text":2585,"level":247},"Glossary",{},{"id":1385,"data":2588,"type":1385,"tunes":2622},{"title":2589,"entries":2590},"Key agentic AI terms",[2591,2594,2597,2599,2601,2604,2607,2610,2613,2616,2619],{"term":2592,"anchor":1391,"definition":2593},"Agentic AI","AI system behavior in which a model dynamically directs multi-step execution using tools, observations and state toward a goal.",{"term":2595,"anchor":1395,"definition":2596},"AI agent","A model-centered system with runtime, tools, state and an execution loop that can pursue a task over multiple steps.",{"term":1731,"anchor":1399,"definition":2598},"Repeated cycle of model decision, tool\u002Faction execution, observation and updated model decision until stopping.",{"term":1769,"anchor":1403,"definition":2600},"The execution layer that manages the model loop, tools, state, approvals, context, errors and stopping conditions.",{"term":2602,"anchor":1407,"definition":2603},"Tool","A capability exposed to the model for reading information, computing, delegating or changing external state.",{"term":2605,"anchor":1411,"definition":2606},"Observation","Information returned from a tool or environment and supplied to a later agent step.",{"term":2608,"anchor":1415,"definition":2609},"Agent state","Persistent task or execution information that exists outside a single model output and may survive across steps or pauses.",{"term":2611,"anchor":1419,"definition":2612},"Autonomy boundary","The explicit limit defining which decisions and actions the model may control dynamically.",{"term":2614,"anchor":1423,"definition":2615},"Human-in-the-loop","A control pattern in which human review, input or approval is required at selected points in an AI-driven process.",{"term":2617,"anchor":1427,"definition":2618},"Trajectory","The sequence of relevant states, decisions, tool calls, actions and observations between task request and final outcome.",{"term":2620,"anchor":1431,"definition":2621},"Idempotency","Property that allows an operation to be repeated without unintentionally applying the same side effect multiple times.",{},{"id":1435,"data":2624,"type":42,"tunes":2626},{"text":2625,"level":247},"Conclusion",{},{"id":1440,"data":2628,"type":218,"tunes":2630},{"text":2629},"Agentic AI is not simply a smarter model or a chatbot with more tools. It is a system architecture in which a model participates in an iterative control loop: decide, act, observe, update and continue.",{},{"id":1445,"data":2632,"type":218,"tunes":2634},{"text":2633},"The model provides flexible decision making, but the surrounding runtime must own execution reality: permissions, tool access, state, approvals, retries, budgets, stopping conditions, tracing and verification.",{},{"id":1450,"data":2636,"type":218,"tunes":2638},{"text":2637},"The most useful design principle is therefore: delegate tactical choice to the model only inside explicit technical and business boundaries. Agentic capability becomes production capability only when autonomy, authority and evidence remain separable.",{},{"id":1455,"data":2640,"type":42,"tunes":2642},{"text":2641,"level":247},"Primary sources and current guidance",{},{"id":1460,"data":2644,"type":218,"tunes":2646},{"text":2645},"The sources below support the current architectural distinctions around agents, workflows, loops, tools, orchestration, safety and evaluation. Project sections are original implementation evidence and are explicitly bounded to what the repositories demonstrate.",{},{"id":1465,"data":2648,"type":1472,"tunes":2653},{"link":1467,"meta":2649},{"image":2650,"title":2651,"description":2652},{"url":347},"OpenAI — Agents","Current developer guidance defining runtime choices for multi-step work, tools, state, orchestration and agent execution.",{},{"id":1475,"data":2655,"type":1472,"tunes":2660},{"link":1477,"meta":2656},{"image":2657,"title":2658,"description":2659},{"url":347},"OpenAI — Agent definitions","Current documentation describing an agent as a model plus instructions and optional runtime behavior including tools, guardrails, MCP servers and handoffs.",{},{"id":1484,"data":2662,"type":1472,"tunes":2667},{"link":1486,"meta":2663},{"image":2664,"title":2665,"description":2666},{"url":347},"OpenAI — Running agents","Current documentation of the agent loop: model call, tool execution or handoff, continuation and final stopping point.",{},{"id":1493,"data":2669,"type":1472,"tunes":2674},{"link":1495,"meta":2670},{"image":2671,"title":2672,"description":2673},{"url":347},"OpenAI — Orchestration and handoffs","Current guidance on handoffs, agents-as-tools and when specialist agents add useful ownership or capability boundaries.",{},{"id":1502,"data":2676,"type":1472,"tunes":2681},{"link":1504,"meta":2677},{"image":2678,"title":2679,"description":2680},{"url":347},"OpenAI — Safety in building agents","Current safety guidance covering tool approvals, prompt injection, guardrails and trace-based evaluation.",{},{"id":1511,"data":2683,"type":1472,"tunes":2688},{"link":1513,"meta":2684},{"image":2685,"title":2686,"description":2687},{"url":347},"Anthropic — Building effective agents","Engineering guidance distinguishing predefined workflows from model-directed agents and describing tool-based environmental feedback loops.",{},{"id":1520,"data":2690,"type":1472,"tunes":2695},{"link":1522,"meta":2691},{"image":2692,"title":2693,"description":2694},{"url":347},"Anthropic — Effective context engineering for AI agents","Practical framing of agents as LLMs autonomously using tools in a loop, with dynamic just-in-time context management.",{},{"id":1529,"data":2697,"type":1472,"tunes":2702},{"link":1531,"meta":2698},{"image":2699,"title":2700,"description":2701},{"url":347},"Anthropic — Demystifying evals for AI agents","2026 guidance on evaluating multi-turn agents that call tools, modify state and adapt to intermediate results.",{},"2.31.6","Agentic AI uses models inside multi-step execution loops where they can choose tools, observe results, update state and adapt their next action within explicit runtime and permission boundaries.",{"lang":7,"title":208,"content":210,"contentJson":2706,"excerpt":1538},{"time":212,"blocks":2707,"version":1537},[2708,2711,2714,2717,2720,2723,2726,2729,2732,2735,2738,2741,2744,2747,2758,2761,2764,2767,2770,2785,2788,2791,2802,2805,2808,2825,2828,2831,2834,2837,2840,2843,2846,2849,2852,2864,2867,2870,2873,2876,2879,2882,2885,2888,2891,2894,2897,2900,2903,2906,2909,2912,2915,2918,2921,2924,2927,2930,2933,2946,2949,2952,2955,2958,2961,2975,2978,2981,2984,2987,2990,2993,2996,2999,3002,3005,3008,3011,3014,3017,3020,3023,3026,3029,3042,3045,3048,3051,3054,3069,3072,3084,3087,3090,3093,3096,3099,3102,3105,3108,3111,3114,3117,3128,3131,3134,3151,3154,3169,3172,3188,3191,3209,3212,3215,3218,3221,3224,3227,3230,3233,3236,3239,3242,3245,3248,3251,3254,3267,3270,3285,3288,3291,3294,3297,3300,3303,3308,3313,3318,3323,3328,3333,3338],{"id":215,"data":2709,"type":218,"tunes":2710},{"text":217},{},{"id":221,"data":2712,"type":226,"tunes":2713},{"body":223,"title":224,"variant":225},{},{"id":229,"data":2715,"type":226,"tunes":2716},{"body":231,"title":232,"variant":233},{},{"id":236,"data":2718,"type":226,"tunes":2719},{"body":238,"title":239,"variant":240},{},{"id":243,"data":2721,"type":248,"tunes":2722},{"title":245,"maxLevel":246,"minLevel":247},{},{"id":251,"data":2724,"type":42,"tunes":2725},{"text":253,"level":247},{},{"id":256,"data":2727,"type":218,"tunes":2728},{"text":258},{},{"id":261,"data":2730,"type":218,"tunes":2731},{"text":263},{},{"id":266,"data":2733,"type":218,"tunes":2734},{"text":268},{},{"id":271,"data":2736,"type":42,"tunes":2737},{"text":273,"level":247},{},{"id":276,"data":2739,"type":218,"tunes":2740},{"text":278},{},{"id":281,"data":2742,"type":218,"tunes":2743},{"text":283},{},{"id":286,"data":2745,"type":218,"tunes":2746},{"text":288},{},{"id":291,"data":2748,"type":317,"tunes":2757},{"steps":2749,"title":315,"orientation":316},[2750,2751,2752,2753,2754,2755,2756],{"label":295,"description":296},{"label":298,"description":299},{"label":301,"description":302},{"label":304,"description":305},{"label":307,"description":308},{"label":310,"description":311},{"label":313,"description":314},{},{"id":320,"data":2759,"type":42,"tunes":2760},{"text":322,"level":247},{},{"id":325,"data":2762,"type":218,"tunes":2763},{"text":327},{},{"id":330,"data":2765,"type":218,"tunes":2766},{"text":332},{},{"id":335,"data":2768,"type":42,"tunes":2769},{"text":337,"level":247},{},{"id":340,"data":2771,"type":369,"tunes":2784},{"rows":2772,"title":360,"layout":361,"columns":2781},[2773,2775,2777,2779],{"id":344,"label":345,"values":2774},[347,347],{"id":349,"label":350,"values":2776},[347,347],{"id":353,"label":354,"values":2778},[347,347],{"id":357,"label":358,"values":2780},[347,347],[2782,2783],{"id":364,"label":365},{"id":367,"label":368},{},{"id":372,"data":2786,"type":218,"tunes":2787},{"text":374},{},{"id":377,"data":2789,"type":42,"tunes":2790},{"text":379,"level":247},{},{"id":382,"data":2792,"type":361,"tunes":2801},{"content":2793,"stretched":43,"withHeadings":14},[2794,2795,2796,2797,2798,2799,2800],[386,387,388],[390,391,392],[394,395,396],[398,399,400],[402,403,404],[406,407,408],[410,411,412],{},{"id":415,"data":2803,"type":218,"tunes":2804},{"text":417},{},{"id":420,"data":2806,"type":42,"tunes":2807},{"text":422,"level":247},{},{"id":425,"data":2809,"type":361,"tunes":2824},{"content":2810,"stretched":43,"withHeadings":14},[2811,2812,2813,2814,2815,2816,2817,2818,2819,2820,2821,2822,2823],[429,430],[432,433],[435,436],[438,439],[441,442],[444,445],[447,448],[450,451],[453,454],[456,457],[459,460],[462,463],[465,466],{},{"id":469,"data":2826,"type":42,"tunes":2827},{"text":471,"level":247},{},{"id":474,"data":2829,"type":218,"tunes":2830},{"text":476},{},{"id":479,"data":2832,"type":218,"tunes":2833},{"text":481},{},{"id":484,"data":2835,"type":226,"tunes":2836},{"body":486,"title":487,"variant":488},{},{"id":491,"data":2838,"type":42,"tunes":2839},{"text":493,"level":247},{},{"id":496,"data":2841,"type":218,"tunes":2842},{"text":498},{},{"id":501,"data":2844,"type":218,"tunes":2845},{"text":503},{},{"id":506,"data":2847,"type":218,"tunes":2848},{"text":508},{},{"id":511,"data":2850,"type":42,"tunes":2851},{"text":513,"level":247},{},{"id":516,"data":2853,"type":361,"tunes":2863},{"content":2854,"stretched":43,"withHeadings":14},[2855,2856,2857,2858,2859,2860,2861,2862],[386,520],[522,523],[525,526],[528,529],[531,532],[534,535],[537,538],[540,541],{},{"id":544,"data":2865,"type":218,"tunes":2866},{"text":546},{},{"id":549,"data":2868,"type":42,"tunes":2869},{"text":551,"level":247},{},{"id":554,"data":2871,"type":218,"tunes":2872},{"text":556},{},{"id":559,"data":2874,"type":218,"tunes":2875},{"text":561},{},{"id":564,"data":2877,"type":218,"tunes":2878},{"text":566},{},{"id":569,"data":2880,"type":42,"tunes":2881},{"text":571,"level":247},{},{"id":574,"data":2883,"type":218,"tunes":2884},{"text":576},{},{"id":579,"data":2886,"type":218,"tunes":2887},{"text":581},{},{"id":584,"data":2889,"type":42,"tunes":2890},{"text":586,"level":247},{},{"id":589,"data":2892,"type":218,"tunes":2893},{"text":591},{},{"id":594,"data":2895,"type":218,"tunes":2896},{"text":596},{},{"id":599,"data":2898,"type":226,"tunes":2899},{"body":601,"title":602,"variant":233},{},{"id":605,"data":2901,"type":42,"tunes":2902},{"text":607,"level":247},{},{"id":610,"data":2904,"type":218,"tunes":2905},{"text":612},{},{"id":615,"data":2907,"type":218,"tunes":2908},{"text":617},{},{"id":620,"data":2910,"type":42,"tunes":2911},{"text":622,"level":247},{},{"id":625,"data":2913,"type":218,"tunes":2914},{"text":627},{},{"id":630,"data":2916,"type":218,"tunes":2917},{"text":632},{},{"id":635,"data":2919,"type":641,"tunes":2920},{"url":637,"title":638,"excerpt":639,"ctaLabel":640},{},{"id":644,"data":2922,"type":42,"tunes":2923},{"text":646,"level":247},{},{"id":649,"data":2925,"type":218,"tunes":2926},{"text":651},{},{"id":654,"data":2928,"type":218,"tunes":2929},{"text":656},{},{"id":659,"data":2931,"type":42,"tunes":2932},{"text":661,"level":247},{},{"id":664,"data":2934,"type":369,"tunes":2945},{"rows":2935,"title":678,"layout":361,"columns":2942},[2936,2938,2940],{"id":668,"label":387,"values":2937},[347,347],{"id":671,"label":672,"values":2939},[347,347],{"id":675,"label":676,"values":2941},[347,347],[2943,2944],{"id":681,"label":682},{"id":684,"label":685},{},{"id":688,"data":2947,"type":42,"tunes":2948},{"text":690,"level":247},{},{"id":693,"data":2950,"type":218,"tunes":2951},{"text":695},{},{"id":698,"data":2953,"type":218,"tunes":2954},{"text":700},{},{"id":703,"data":2956,"type":218,"tunes":2957},{"text":705},{},{"id":708,"data":2959,"type":42,"tunes":2960},{"text":710,"level":247},{},{"id":713,"data":2962,"type":361,"tunes":2974},{"content":2963,"stretched":43,"withHeadings":14},[2964,2965,2966,2967,2968,2969,2970,2971,2972,2973],[717,718],[720,721],[723,724],[726,727],[729,730],[732,733],[735,736],[738,739],[741,742],[744,745],{},{"id":748,"data":2976,"type":42,"tunes":2977},{"text":750,"level":247},{},{"id":753,"data":2979,"type":218,"tunes":2980},{"text":755},{},{"id":758,"data":2982,"type":218,"tunes":2983},{"text":760},{},{"id":763,"data":2985,"type":218,"tunes":2986},{"text":765},{},{"id":768,"data":2988,"type":42,"tunes":2989},{"text":770,"level":247},{},{"id":773,"data":2991,"type":218,"tunes":2992},{"text":775},{},{"id":778,"data":2994,"type":218,"tunes":2995},{"text":780},{},{"id":783,"data":2997,"type":218,"tunes":2998},{"text":785},{},{"id":788,"data":3000,"type":42,"tunes":3001},{"text":790,"level":247},{},{"id":793,"data":3003,"type":218,"tunes":3004},{"text":795},{},{"id":798,"data":3006,"type":218,"tunes":3007},{"text":800},{},{"id":803,"data":3009,"type":641,"tunes":3010},{"url":805,"title":806,"excerpt":807,"ctaLabel":808},{},{"id":811,"data":3012,"type":42,"tunes":3013},{"text":813,"level":247},{},{"id":816,"data":3015,"type":218,"tunes":3016},{"text":818},{},{"id":821,"data":3018,"type":218,"tunes":3019},{"text":823},{},{"id":826,"data":3021,"type":218,"tunes":3022},{"text":828},{},{"id":831,"data":3024,"type":641,"tunes":3025},{"url":833,"title":834,"excerpt":835,"ctaLabel":836},{},{"id":839,"data":3027,"type":42,"tunes":3028},{"text":841,"level":247},{},{"id":844,"data":3030,"type":361,"tunes":3041},{"content":3031,"stretched":43,"withHeadings":14},[3032,3033,3034,3035,3036,3037,3038,3039,3040],[358,848,849],[851,852,853],[855,856,857],[859,860,861],[863,864,865],[867,868,869],[871,872,873],[875,876,877],[879,880,881],{},{"id":884,"data":3043,"type":42,"tunes":3044},{"text":886,"level":247},{},{"id":889,"data":3046,"type":218,"tunes":3047},{"text":891},{},{"id":894,"data":3049,"type":218,"tunes":3050},{"text":896},{},{"id":899,"data":3052,"type":42,"tunes":3053},{"text":901,"level":247},{},{"id":904,"data":3055,"type":361,"tunes":3068},{"content":3056,"stretched":43,"withHeadings":14},[3057,3058,3059,3060,3061,3062,3063,3064,3065,3066,3067],[908,520,909],[911,912,913],[915,916,917],[919,920,921],[923,924,925],[927,928,929],[931,932,933],[935,936,937],[939,940,941],[943,944,945],[947,948,949],{},{"id":952,"data":3070,"type":42,"tunes":3071},{"text":954,"level":247},{},{"id":957,"data":3073,"type":361,"tunes":3083},{"content":3074,"stretched":43,"withHeadings":14},[3075,3076,3077,3078,3079,3080,3081,3082],[961,962],[964,965],[967,968],[970,971],[973,974],[976,977],[979,980],[982,983],{},{"id":986,"data":3085,"type":218,"tunes":3086},{"text":988},{},{"id":991,"data":3088,"type":42,"tunes":3089},{"text":993,"level":247},{},{"id":996,"data":3091,"type":42,"tunes":3092},{"text":998,"level":246},{},{"id":1001,"data":3094,"type":218,"tunes":3095},{"text":1003},{},{"id":1006,"data":3097,"type":218,"tunes":3098},{"text":1008},{},{"id":1011,"data":3100,"type":218,"tunes":3101},{"text":1013},{},{"id":1016,"data":3103,"type":218,"tunes":3104},{"text":1018},{},{"id":1021,"data":3106,"type":42,"tunes":3107},{"text":1023,"level":246},{},{"id":1026,"data":3109,"type":218,"tunes":3110},{"text":1028},{},{"id":1031,"data":3112,"type":218,"tunes":3113},{"text":1033},{},{"id":1036,"data":3115,"type":218,"tunes":3116},{"text":1038},{},{"id":1041,"data":3118,"type":361,"tunes":3127},{"content":3119,"stretched":43,"withHeadings":14},[3120,3121,3122,3123,3124,3125,3126],[1045,1046],[1048,1049],[1051,1052],[1054,1055],[1057,1058],[1060,1061],[1063,1064],{},{"id":1067,"data":3129,"type":226,"tunes":3130},{"body":1069,"title":1070,"variant":240},{},{"id":1073,"data":3132,"type":42,"tunes":3133},{"text":1075,"level":247},{},{"id":1078,"data":3135,"type":361,"tunes":3150},{"content":3136,"stretched":43,"withHeadings":14},[3137,3138,3139,3140,3141,3142,3143,3144,3145,3146,3147,3148,3149],[1082,1083],[1085,1086],[1088,1089],[1091,1092],[1094,1095],[1097,1098],[1100,1101],[1103,1104],[1106,1107],[1109,1110],[1112,1113],[1115,1116],[1118,1119],{},{"id":1122,"data":3152,"type":42,"tunes":3153},{"text":1124,"level":247},{},{"id":1127,"data":3155,"type":361,"tunes":3168},{"content":3156,"stretched":43,"withHeadings":14},[3157,3158,3159,3160,3161,3162,3163,3164,3165,3166,3167],[1131,1132],[1134,1135],[1137,1138],[1140,1141],[1143,1144],[1146,1147],[1149,1150],[1152,1153],[1155,1156],[1158,1159],[1161,1162],{},{"id":1165,"data":3170,"type":42,"tunes":3171},{"text":1167,"level":247},{},{"id":1170,"data":3173,"type":317,"tunes":3187},{"steps":3174,"title":1209,"orientation":316},[3175,3176,3177,3178,3179,3180,3181,3182,3183,3184,3185,3186],{"label":1174,"description":1175},{"label":1177,"description":1178},{"label":1180,"description":1181},{"label":1183,"description":1184},{"label":1186,"description":1187},{"label":1189,"description":1190},{"label":1192,"description":1193},{"label":1195,"description":1196},{"label":1198,"description":1199},{"label":1201,"description":1202},{"label":1204,"description":1205},{"label":1207,"description":1208},{},{"id":1212,"data":3189,"type":42,"tunes":3190},{"text":1214,"level":247},{},{"id":1217,"data":3192,"type":361,"tunes":3208},{"content":3193,"stretched":43,"withHeadings":14},[3194,3195,3196,3197,3198,3199,3200,3201,3202,3203,3204,3205,3206,3207],[520,1221],[1223,1224],[1226,1227],[1229,1230],[1232,1233],[1235,1236],[1238,1239],[1241,1242],[1244,1245],[1247,1248],[1250,1251],[1253,1254],[1256,1257],[1259,1260],{},{"id":1263,"data":3210,"type":42,"tunes":3211},{"text":1265,"level":247},{},{"id":1268,"data":3213,"type":218,"tunes":3214},{"text":1270},{},{"id":1273,"data":3216,"type":218,"tunes":3217},{"text":1275},{},{"id":1278,"data":3219,"type":218,"tunes":3220},{"text":1280},{},{"id":1283,"data":3222,"type":218,"tunes":3223},{"text":1285},{},{"id":1288,"data":3225,"type":218,"tunes":3226},{"text":1290},{},{"id":1293,"data":3228,"type":42,"tunes":3229},{"text":1295,"level":247},{},{"id":1298,"data":3231,"type":218,"tunes":3232},{"text":1300},{},{"id":1303,"data":3234,"type":218,"tunes":3235},{"text":1305},{},{"id":1308,"data":3237,"type":218,"tunes":3238},{"text":1310},{},{"id":1313,"data":3240,"type":42,"tunes":3241},{"text":1315,"level":247},{},{"id":1318,"data":3243,"type":218,"tunes":3244},{"text":1320},{},{"id":1323,"data":3246,"type":218,"tunes":3247},{"text":1325},{},{"id":1328,"data":3249,"type":218,"tunes":3250},{"text":1330},{},{"id":1333,"data":3252,"type":42,"tunes":3253},{"text":1335,"level":247},{},{"id":1338,"data":3255,"type":1338,"tunes":3266},{"items":3256,"title":1377},[3257,3258,3259,3260,3261,3262,3263,3264,3265],{"id":1342,"answer":1343,"question":1344},{"id":1346,"answer":1347,"question":1348},{"id":1350,"answer":1351,"question":1352},{"id":1354,"answer":1355,"question":1356},{"id":1358,"answer":1359,"question":1360},{"id":1362,"answer":1363,"question":1364},{"id":1366,"answer":1367,"question":1368},{"id":1370,"answer":1371,"question":1372},{"id":1374,"answer":1375,"question":1376},{},{"id":1380,"data":3268,"type":42,"tunes":3269},{"text":1382,"level":247},{},{"id":1385,"data":3271,"type":1385,"tunes":3284},{"title":1387,"entries":3272},[3273,3274,3275,3276,3277,3278,3279,3280,3281,3282,3283],{"term":1390,"anchor":1391,"definition":1392},{"term":1394,"anchor":1395,"definition":1396},{"term":1398,"anchor":1399,"definition":1400},{"term":1402,"anchor":1403,"definition":1404},{"term":1406,"anchor":1407,"definition":1408},{"term":1410,"anchor":1411,"definition":1412},{"term":1414,"anchor":1415,"definition":1416},{"term":1418,"anchor":1419,"definition":1420},{"term":1422,"anchor":1423,"definition":1424},{"term":1426,"anchor":1427,"definition":1428},{"term":1430,"anchor":1431,"definition":1432},{},{"id":1435,"data":3286,"type":42,"tunes":3287},{"text":1437,"level":247},{},{"id":1440,"data":3289,"type":218,"tunes":3290},{"text":1442},{},{"id":1445,"data":3292,"type":218,"tunes":3293},{"text":1447},{},{"id":1450,"data":3295,"type":218,"tunes":3296},{"text":1452},{},{"id":1455,"data":3298,"type":42,"tunes":3299},{"text":1457,"level":247},{},{"id":1460,"data":3301,"type":218,"tunes":3302},{"text":1462},{},{"id":1465,"data":3304,"type":1472,"tunes":3307},{"link":1467,"meta":3305},{"image":3306,"title":1470,"description":1471},{"url":347},{},{"id":1475,"data":3309,"type":1472,"tunes":3312},{"link":1477,"meta":3310},{"image":3311,"title":1480,"description":1481},{"url":347},{},{"id":1484,"data":3314,"type":1472,"tunes":3317},{"link":1486,"meta":3315},{"image":3316,"title":1489,"description":1490},{"url":347},{},{"id":1493,"data":3319,"type":1472,"tunes":3322},{"link":1495,"meta":3320},{"image":3321,"title":1498,"description":1499},{"url":347},{},{"id":1502,"data":3324,"type":1472,"tunes":3327},{"link":1504,"meta":3325},{"image":3326,"title":1507,"description":1508},{"url":347},{},{"id":1511,"data":3329,"type":1472,"tunes":3332},{"link":1513,"meta":3330},{"image":3331,"title":1516,"description":1517},{"url":347},{},{"id":1520,"data":3334,"type":1472,"tunes":3337},{"link":1522,"meta":3335},{"image":3336,"title":1525,"description":1526},{"url":347},{},{"id":1529,"data":3339,"type":1472,"tunes":3342},{"link":1531,"meta":3340},{"image":3341,"title":1534,"description":1535},{"url":347},{},"Post erfolgreich abgerufen",{"items":3345,"source":3430,"manualIds":3431,"manualMatchedIds":3432},[3346,3353,3360,3367,3374,3381,3388,3395,3402,3409,3416,3423],{"id":3347,"slug":3348,"title":3349,"excerpt":3350,"featuredImage":3351,"publishedAt":3352},"483","what-is-an-ai-solution-architect-system-boundaries-responsibilities-and-trade-offs","什么是AI解决方案架构师？系统边界、职责与权衡","AI解决方案架构师将业务需求转化为生产就绪的AI系统，涵盖数据、模型、工具、安全、运行时、评估和运维。","\u002Fuploads\u002F2026\u002F10\u002Fwhat-is-an-ai-solution-architect-system-boundaries-responsibilities-and-trade-offs-1791476643267-1st5xz.webp","2026-10-08T12:23:00.000Z",{"id":3354,"slug":3355,"title":3356,"excerpt":3357,"featuredImage":3358,"publishedAt":3359},"486","source-of-truth-in-ai-systems-where-reliable-knowledge-actually-comes-from","AI系统中的真相来源：可靠知识究竟从何而来","事实来源（Source of Truth）定义了对于特定事实或状态，哪个来源具有权威性。了解它与RAG、溯源、记忆、上下文、向量数据库和记录系统有何不同。","\u002Fuploads\u002F2026\u002F10\u002Fsource-of-truth-in-ai-systems-where-reliable-knowledge-actually-comes-from-1791479103235-6bq9em.webp","2026-10-08T13:02:00.000Z",{"id":3361,"slug":3362,"title":3363,"excerpt":3364,"featuredImage":3365,"publishedAt":3366},"466","the-gpu-is-not-the-product-future-proof-private-ai-architecture","GPU 不是产品：面向未来的私有 AI 架构","私有 AI 基础设施不应围绕单一 GPU 或单一模型来设计。更具韧性的做法是将快速推理 GPU、内存充裕的 AI 系统、物理 AI 节点以及可选的前沿云模型，统一置于一个具备能力感知的路由层之后。","\u002Fuploads\u002F2026\u002F09\u002Fthe-gpu-is-not-the-product-future-proof-private-ai-architecture-1790140878812-8hsl39.webp","2026-09-23T01:19:00.000Z",{"id":3368,"slug":3369,"title":3370,"excerpt":3371,"featuredImage":3372,"publishedAt":3373},"481","generative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing","生成式人工智能解析：模型、检索、工具与应用并非同一回事","生成式AI不仅仅是一个模型。了解模型、检索、工具、上下文、运行时和应用程序如何在生产AI系统中协同工作。","\u002Fuploads\u002F2026\u002F10\u002Fgenerative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing-1791475411822-pp0dvz.webp","2026-10-08T12:00:00.000Z",{"id":3375,"slug":3376,"title":3377,"excerpt":3378,"featuredImage":3379,"publishedAt":3380},"494","air-gapped-ai-how-ai-systems-work-without-internet-or-cloud-access","气隙AI：AI系统如何在没有互联网或云访问的情况下工作","气隙AI在隔离的安全域内运行模型、RAG和AI应用，无需互联网或云依赖。了解模型、数据、更新和工具如何离线运行。","\u002Fuploads\u002F2026\u002F10\u002Fair-gapped-ai-how-ai-systems-work-without-internet-or-cloud-access-1791487983978-e6xqf0.webp","2026-10-08T11:32:00.000Z",{"id":3382,"slug":3383,"title":3384,"excerpt":3385,"featuredImage":3386,"publishedAt":3387},"479","where-does-an-llm-get-its-data-rag-data-sources-in-python","LLM从哪里获取数据？Python中的RAG数据源","LLM 并不会神奇地知道你的文件、数据库或 API。这个 RAG 系列的实用续篇用简单的 Python 展示了外部数据如何变成可检索的证据：从文本文件和 SQL 到全文搜索、嵌入、上下文组装以及最终的 LLM 调用。","\u002Fuploads\u002F2026\u002F09\u002Fwhere-does-an-llm-get-its-data-rag-data-sources-in-python-1790517200521-nfsi5i.webp","2026-09-27T05:51:00.000Z",{"id":3389,"slug":3390,"title":3391,"excerpt":3392,"featuredImage":3393,"publishedAt":3394},"492","mcp-explained-what-it-connects-what-it-does-not-do-and-where-it-fits","MCP 解析：它连接什么、不做什么以及它适用于何处","模型上下文协议通过标准的客户端-服务器边界，将AI应用程序连接到外部工具、资源和提示。了解MCP能做什么、不能做什么，以及它在智能体架构中的定位。","\u002Fuploads\u002F2026\u002F10\u002Fmcp-explained-what-it-connects-what-it-does-not-do-and-where-it-fits-1791486640275-7ub1cq.webp","2026-10-08T15:09:00.000Z",{"id":3396,"slug":3397,"title":3398,"excerpt":3399,"featuredImage":3400,"publishedAt":3401},"484","what-is-an-ai-platform-architect-models-data-runtime-security-and-operations","什么是AI平台架构师？模型、数据、运行时、安全与运维","AI平台架构师负责跨模型、提供商、检索、智能体、身份、安全、评估、可观测性和运营设计可复用的AI基础。","\u002Fuploads\u002F2026\u002F10\u002Fwhat-is-an-ai-platform-architect-models-data-runtime-security-and-operations-1791477229171-ou3zcc.webp","2026-10-08T12:32:00.000Z",{"id":3403,"slug":3404,"title":3405,"excerpt":3406,"featuredImage":3407,"publishedAt":3408},"470","what-should-an-ai-agent-remember-forget-recompute-or-retrieve-again","AI代理应该记住、遗忘、重新计算还是再次检索什么？","长时间运行的代理不应记住所有内容。本文提供了一个实用的生命周期模型，用于决定哪些内容应属于持久记忆、哪些内容应重新检索、哪些内容重新计算更安全，以及哪些内容应过期或被取代。","\u002Fuploads\u002F2026\u002F09\u002Fwhat-should-an-ai-agent-remember-forget-recompute-or-retrieve-again-1790351131087-iehz28.webp","2026-09-25T09:43:00.000Z",{"id":3410,"slug":3411,"title":3412,"excerpt":3413,"featuredImage":3414,"publishedAt":3415},"495","sovereign-ai-control-of-models-data-infrastructure-and-dependencies","主权人工智能：模型、数据、基础设施与依赖关系的控制","主权人工智能关乎对模型、数据、基础设施、软件、运营和战略依赖的有效控制——而不仅仅是人工智能模型托管在哪里。","\u002Fuploads\u002F2026\u002F10\u002Fsovereign-ai-control-of-models-data-infrastructure-and-dependencies-1791488833132-niy85x.webp","2026-10-08T15:45:00.000Z",{"id":3417,"slug":3418,"title":3419,"excerpt":3420,"featuredImage":3421,"publishedAt":3422},"485","enterprise-ai-architecture-what-changes-when-ai-enters-a-company","企业AI架构：当AI进入公司时会发生什么变化","企业AI架构阐释了AI如何在数据权限、身份、许可、提供商、风险、治理、评估、合规和运营方面改变公司系统。","\u002Fuploads\u002F2026\u002F10\u002Fenterprise-ai-architecture-what-changes-when-ai-enters-a-company-1791478161363-czrwaq.webp","2026-10-08T10:48:00.000Z",{"id":3424,"slug":3425,"title":3426,"excerpt":3427,"featuredImage":3428,"publishedAt":3429},"467","the-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers","答案有效性边界：相关性到可靠AI答案之间缺失的层级","一个来源可能相关、权威，但对于所提出的问题仍然是错误的。缺失的层次是适用性：答案成立的条件，以及迫使其被重新考虑的变化。本文介绍了“答案有效性边界”这一面向人类、AI搜索和RAG系统的来源设计模式。","\u002Fuploads\u002F2026\u002F09\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers-1790272901306-1g5jly.webp","2026-09-24T11:59:00.000Z","fallback",[],[]]