[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"portal-settings:stajic:zh":3,"public-menus:all":38,"post:generative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing:zh":205,"related:post:generative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing:zh:1":2241},{"statusCode":4,"data":5,"message":37},200,{"tenantId":6,"lang":7,"defaultLang":8,"siteUrl":9,"contactEmail":10,"brandName":11,"logoUrl":12,"siteName":11,"siteDescription":13,"ogImage":10,"robotsIndex":14,"socialLinks":10,"reservedSlugs":10,"seoPolicy":15},"stajic","zh","de","https:\u002F\u002Fstajic.de",null,"Stajic Platform","\u002FLogo_Planet.svg","Stajic Portal",true,{"branding":16,"relatedContent":17,"crossDomainLinks":18},{"logoUrl":12},{"enabled":14},[19,22,25,28,31,34],{"url":20,"label":21,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Ffigure.rocks","figure.rocks",{"url":23,"label":24,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Floving.rocks","loving.rocks",{"url":26,"label":27,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.com","bazify.com",{"url":29,"label":30,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.de","bazify.de",{"url":32,"label":33,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.at","bazify.at",{"url":35,"label":36,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.ba","bazify.ba","Portal settings resolved",[39,45],{"id":40,"name":41,"location":42,"isActive":14,"isDefault":43,"items":44},1,"main-navigation","header",false,[],{"id":46,"name":47,"location":48,"isActive":14,"isDefault":14,"items":49},4,"main-menu","sidebar",[50,66,79,93,103,118,133],{"id":51,"title":52,"url":60,"target":61,"icon":62,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":64,"portfolioId":10,"children":65},"item-18",{"de":53,"en":54,"es":55,"fr":56,"it":54,"ru":57,"sr":58,"zh":59},"Startseite","Home","Inicio","Accueil","Главная","Почетна","首页","\u002Ffull-stack-web-developer-munich-performance-seo-and-maintainable-builds","_self","i-lucide-home","page",111,[],{"id":67,"title":68,"url":75,"target":61,"icon":76,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":77,"portfolioId":10,"children":78},"item-22",{"de":69,"en":69,"es":70,"fr":69,"it":71,"ru":72,"sr":73,"zh":74},"Vision","Visión","Visione","Видение","Визија","想象","\u002Fueber-uns-webdesign-muenchen-webaplikation","i-lucide-eye",113,[],{"id":80,"title":81,"url":89,"target":61,"icon":90,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":91,"portfolioId":10,"children":92},"item-19",{"de":82,"en":83,"es":84,"fr":83,"it":85,"ru":86,"sr":87,"zh":88},"Leistungen","Services","Servicios","Servizi","Услуги","Услуге","服务","\u002Fservices-dienstleistungen-muenchen","i-lucide-wrench",116,[],{"id":94,"title":95,"url":99,"target":61,"icon":100,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":101,"portfolioId":10,"children":102},"item-23",{"de":96,"en":96,"es":96,"fr":96,"it":96,"ru":97,"sr":97,"zh":98},"Blog","Блог","博客","\u002Fblog","i-lucide-book-open",112,[],{"id":104,"title":105,"url":114,"target":61,"icon":115,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":116,"portfolioId":10,"children":117},"item-32",{"de":106,"en":107,"es":108,"fr":109,"it":110,"ru":111,"sr":112,"zh":113},"Neue Technologien","New Technologies","Nuevas tecnologías","Nouvelles technologies","Nuove tecnologie","Новые технологии","Нове технологије","新技术！","\u002Fneue-webtechnologien","i-lucide-sparkles",122,[],{"id":119,"title":120,"url":129,"target":61,"icon":130,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":131,"portfolioId":10,"children":132},"item-20",{"de":121,"en":122,"es":123,"fr":124,"it":125,"ru":126,"sr":127,"zh":128},"Kontakt","Contact us!","Contacto","Contact","Contatto","Контакт","Контактирајте нас","联系我们！","\u002Fcontact","i-lucide-mail",115,[],{"id":134,"title":135,"url":144,"target":61,"icon":145,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":147},"item-21",{"de":136,"en":137,"es":138,"fr":139,"it":140,"ru":141,"sr":142,"zh":143},"Unsere Arbeit","Our Work","Nuestro trabajo","Nos réalisations","I nostri lavori","Наши работы","Наши радови","文件夹","\u002Fportfolio","i-lucide-briefcase",114,[148,161,175,181,193],{"id":149,"title":150,"url":144,"target":61,"icon":159,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":160},"item-24",{"de":151,"en":152,"es":153,"fr":154,"it":155,"ru":156,"sr":157,"zh":158},"Alle Projekte","All Projects","Todos los proyectos","Tous les projets","Tutti i progetti","Все проекты","Сви пројекти","所有项目","i-lucide-grid-3x3",[],{"id":162,"title":163,"url":171,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":174},"item-29",{"de":164,"en":165,"es":166,"fr":167,"it":168,"ru":169,"sr":170,"zh":143},"Local Roots, Global Reach","Local Roots - Global Reach","Empresa local ","Entreprise locale","Azienda locale","Местная компания","Локално предузеће глобално тржиште","\u002Fportfolio\u002Flocal-roots-global-reach-communication-media-systems-for-modern-business","i-lucide-folder","custom",[],{"id":176,"title":177,"url":179,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":180},"item-28",{"de":178,"en":178,"es":178,"fr":178,"it":178,"ru":178,"sr":178,"zh":178},"Solr Suggester","\u002Fportfolio\u002Fsolr-fuzzy-suggester-und-solr-infix-suggester-abfrage-ueber-ajax-und-filterung",[],{"id":182,"title":183,"url":191,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":192},"item-27",{"de":184,"en":185,"es":186,"fr":187,"it":188,"ru":189,"sr":190,"zh":185},"Firmenwebseite SEO","Company Website SEO","Sitio web corporativo SEO","Site web d’entreprise SEO","Sito web aziendale SEO","Корпоративный сайт SEO","Пословна веб-страница SEO","\u002Fportfolio\u002Fseo-sem-branding-mobile-webseite-muenchen",[],{"id":194,"title":195,"url":203,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":204},"item-31",{"de":196,"en":197,"es":198,"fr":199,"it":200,"ru":201,"sr":202,"zh":197},"Digitalisierungsportal","Digitalization Portal","Portal de digitalización","Portail de numérisation","Portale di digitalizzazione","Портал цифровизации","Портал за дигитализацију","\u002Fportfolio\u002Fdigitalisierungsportal-archiv-museum-bibliothek-ead-lido-mets-mods",[],{"statusCode":4,"data":206,"message":2240},{"id":207,"title":208,"slug":209,"content":210,"contentJson":211,"excerpt":1066,"featuredImage":1067,"featuredImageAlt":1068,"featuredImageCaption":10,"featuredImageTitle":10,"featuredImageCopyright":10,"featuredImageAuthor":10,"featuredImageSourceUrl":10,"featuredImageLicense":10,"featuredImageIsAiGenerated":43,"status":1069,"publishedAt":1070,"createdAt":1071,"updatedAt":1072,"seoLocalePaths":1073,"categories":1082,"author":1103,"translations":1108},"481","生成式人工智能解析：模型、检索、工具与应用并非同一回事","generative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing","\u003Cp>生成式人工智能并非单一组件。一个生产级生成式人工智能系统通常将生成模型与应用程序代码相结合，后者提供指令和上下文，在需要时检索外部知识，暴露用于读取或更改外部系统的工具，管理运行时状态和权限，并将结果转化为可用的产品。将模型、检索、工具、上下文、运行时和应用程序视为同一事物，会掩盖决定新鲜度、安全性、可靠性、成本和控制的边界。\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--info my-6 rounded-xl border p-5 border-blue-300 bg-blue-50 dark:border-blue-900 dark:bg-blue-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">直接回答\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">\u003Cstrong>模型负责生成；检索负责查找外部证据；工具负责访问数据或执行操作；上下文是模型在当前推理中能看到的内容；运行时协调执行；应用程序拥有产品规则、状态、权限、持久化和用户体验。\u003C\u002Fstrong>这些层可以由供应商打包在一起，但它们的职责仍然不同。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Caside class=\"editorjs-callout editorjs-callout--note my-6 rounded-xl border p-5 border-gray-300 bg-gray-50 dark:border-gray-700 dark:bg-gray-900\u002F40\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">术语和版本说明\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">本文定义的是持久的架构职责，而非某一供应商的技术栈。当前实现示例已于\u003Cstrong>2026年10月8日\u003C\u002Fstrong>重新核对。供应商API和产品名称可能会变化；职责边界比任何单个SDK或端点都更稳定。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Cnav class=\"editorjs-toc\" data-editorjs-toc=\"true\" aria-label=\"目录\">\u003Cstrong class=\"editorjs-toc__title\">目录\u003C\u002Fstrong>\u003Col class=\"editorjs-toc__list editorjs-toc__list--depth-0\">\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-5\" class=\"editorjs-toc__link\">“生成式人工智能”究竟意味着什么？\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-9\" class=\"editorjs-toc__link\">生成式人工智能系统的最简单有用模型\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-13\" class=\"editorjs-toc__link\">六个重要的边界\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-15\" class=\"editorjs-toc__link\">1. 模型：生成是其核心职责\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-19\" class=\"editorjs-toc__link\">2. 检索：查找外部证据是一项独立操作\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-24\" class=\"editorjs-toc__link\">3. 工具：访问和操作不是模型知识\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-29\" class=\"editorjs-toc__link\">4. 上下文：模型当前能看到什么\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-33\" class=\"editorjs-toc__link\">5. 运行时与编排：协调循环\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-37\" class=\"editorjs-toc__link\">6. 应用：AI 成为产品的地方\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-41\" class=\"editorjs-toc__link\">各部分如何在真实请求中协同工作\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-45\" class=\"editorjs-toc__link\">不同的 AI 产品使用不同的组合\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-48\" class=\"editorjs-toc__link\">实现证据：Aaasaasa AI Client\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-55\" class=\"editorjs-toc__link\">常见的类别错误\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-57\" class=\"editorjs-toc__link\">边界崩溃时的故障模式\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-60\" class=\"editorjs-toc__link\">什么是稳定的，什么对版本敏感？\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-64\" class=\"editorjs-toc__link\">AI 组件边界测试\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-67\" class=\"editorjs-toc__link\">生成式 AI 不是什么\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-71\" class=\"editorjs-toc__link\">知识图谱中接下来去哪里\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-75\" class=\"editorjs-toc__link\">局限性\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-79\" class=\"editorjs-toc__link\">什么会改变这个答案？\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-82\" class=\"editorjs-toc__link\">结论\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-86\" class=\"editorjs-toc__link\">常见问题\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-88\" class=\"editorjs-toc__link\">术语表\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-90\" class=\"editorjs-toc__link\">主要来源和实现证据\u003C\u002Fa>\u003C\u002Fli>\u003C\u002Fol>\u003C\u002Fnav>\n\u003Ch2 id=\"section-5\">“生成式人工智能”究竟意味着什么？\u003C\u002Fh2>\n\u003Cp>在模型层面，生成式人工智能指的是能够生成衍生合成内容（如文本、图像、音频、视频、代码或其他数字输出）的人工智能模型。NIST AI 600-1使用了这种面向模型的含义，并分别在模型、系统、应用程序和用例层面讨论风险。\u003C\u002Fp>\n\u003Cp>这种区分很重要，因为人工智能模型与完整的人工智能系统不是一回事。NIST当前术语表将人工智能模型定义为使用计算、统计或机器学习技术从输入产生输出的组件，而人工智能系统可以包括使用人工智能运行的软件、硬件、应用程序、工具或实用程序。\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--success my-6 rounded-xl border p-5 border-emerald-300 bg-emerald-50 dark:border-emerald-900 dark:bg-emerald-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">一个有用的边界\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">\u003Cstrong>生成模型 ≠ 生成式人工智能应用。\u003C\u002Fstrong>\u003Cbr>模型是一个计算组件。可用的AI产品是围绕该组件构建的系统。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-9\">生成式人工智能系统的最简单有用模型\u003C\u002Fh2>\n\u003Cp>作为第一个心智模型，想象一个公司助手回答：“这位客户今天能获得退款吗？”一个有用的答案可能需要几种不同的职责。语言模型可以解释问题并撰写说明，但当前订单状态可能来自数据库工具，退款政策可能来自文档检索，权限可能由应用程序强制执行，而最终操作可能需要受控的API调用。\u003C\u002Fp>\n\u003Csection class=\"editorjs-process my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">一种常见的执行路径\u003C\u002Fh3>\u003Cdiv class=\"grid grid-cols-1 md:grid-cols-2 xl:grid-cols-3 gap-4\">\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">1\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">1. 用户请求\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">应用程序接收自然语言问题或任务。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">2\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">2. 应用程序策略和状态\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">身份、租户、权限、当前工作流状态和产品规则定义了该请求被允许做什么。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">3\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">3. 检索或直接数据访问\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">当模型知识不足时，系统获取外部证据或当前事实。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">4\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">4. 上下文构建\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">为模型组装指令、用户输入、选定的证据、相关状态和工具定义。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">5\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">5. 模型推理\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">生成模型解释提供的上下文并产生文本、结构化输出或工具请求。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">6\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">6. 需要时执行工具\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">运行时或应用程序在模型外部验证并执行已批准的工具调用。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">7\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">7. 观察与继续\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">工具结果可以作为新上下文返回给模型，用于下一步推理。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">8\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">8. 验证和产品输出\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">应用程序验证结果，记录所需状态或审计数据，并呈现或执行最终结果。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Cp>真实系统并不总是严格遵循这个顺序。检索可以在第一次模型调用之前发生，工具可以在代理循环中选择，确定性的应用程序逻辑可以完全绕过模型，验证可以在多个阶段进行。重点是将职责分开，而不是强加一个通用的工作流。\u003C\u002Fp>\n\u003Ch2 id=\"section-13\">六个重要的边界\u003C\u002Fh2>\n\u003Csection class=\"editorjs-comparison my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">一个AI产品中的六项职责\u003C\u002Fh3>\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left dark:border-gray-700 dark:bg-gray-900\">\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">主要工作\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">典型输入\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">不同于\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">模型\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">检索\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">工具\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">上下文\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">运行时\u002F编排器\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">应用程序\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-15\">1. 模型：生成是其核心职责\u003C\u002Fh2>\n\u003Cp>生成模型将提供的输入映射为生成的输出。对于语言模型，这可以包括自然语言文本、结构化JSON、代码、分类、摘要、计划或工具调用参数。多模态生成模型可以处理额外的输入和输出类型。\u003C\u002Fp>\n\u003Cp>模型可以在其参数中包含大量学习到的知识，但参数化知识不是实时数据库。除非通过当前输入路径提供信息，否则模型不会自动知道五分钟前创建的文档、当前库存水平、私人客户记录或应用程序状态。\u003C\u002Fp>\n\u003Cp>这就是为什么更改模型不能自动解决知识过时、权限缺失、检索损坏、状态所有权不正确或不安全的工具执行问题。这些故障通常属于其他层。\u003C\u002Fp>\n\u003Ch2 id=\"section-19\">2. 检索：查找外部证据是一项独立操作\u003C\u002Fh2>\n\u003Cp>检索在生成之前或生成过程中从外部来源选择信息。Lewis 等人于 2020 年提出的检索增强生成工作通过将参数化生成模型与检索到的非参数化记忆相结合，使这种分离变得明确。现代生产系统使用许多检索变体，但架构理念保持不变：有用的证据可以在推理时获取，而不是仅依赖模型在训练期间学到的内容。\u003C\u002Fp>\n\u003Cp>检索可以使用词法搜索、嵌入、向量搜索、混合搜索、SQL、知识图谱、元数据过滤器、API 或其他选择机制。因此，向量数据库是一种可能的检索组件，而不是 RAG 的定义。\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--warning my-6 rounded-xl border p-5 border-amber-300 bg-amber-50 dark:border-amber-900 dark:bg-amber-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">相关性不等于权威性\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">检索到的段落可能高度相关，但仍然过时、未经授权、来自错误版本，或不足以支持某项主张。检索质量和证据质量必须分开评估。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Caside class=\"editorjs-referral my-6\">\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works\" class=\"flex flex-col sm:flex-row gap-4 rounded-xl border border-gray-200 dark:border-gray-700 p-4 transition hover:border-primary-500\">\u003Cdiv class=\"min-w-0 flex-1\">\u003Cstrong class=\"block text-lg text-gray-900 dark:text-gray-100\">什么是 RAG？对其工作原理的最简单解释\u003C\u002Fstrong>\u003Cp class=\"mt-2 text-sm text-gray-600 dark:text-gray-300\">对检索增强生成的经典通俗解释，包括 LLM、知识、状态、记忆和工具之间的分离。\u003C\u002Fp>\u003Cspan class=\"mt-3 inline-flex text-sm font-medium text-primary-600 dark:text-primary-400\">阅读 RAG 基础 →\u003C\u002Fspan>\u003C\u002Fdiv>\u003C\u002Fa>\u003C\u002Faside>\n\u003Ch2 id=\"section-24\">3. 工具：访问和操作不是模型知识\u003C\u002Fh2>\n\u003Cp>工具是一种接口，AI 运行时可以通过它请求模型之外的功能。工具可以查询数据库、搜索网络、读取文件、计算数值、调用内部服务、创建工单、发送消息、修改记录或触发另一项受控操作。\u003C\u002Fp>\n\u003Cp>OpenAI 当前的函数调用文档明确了这一边界：函数调用让模型能够与外部系统交互，并访问应用程序提供的数据或操作。模型可以提议或选择调用，但真正执行操作的是外部系统。\u003C\u002Fp>\n\u003Cp>因此，工具使用产生了两个独立问题：模型能否请求此能力？以及应用程序是否会授权并执行它？生产系统不应将模型意图与造成副作用的权限混为一谈。\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--success my-6 rounded-xl border p-5 border-emerald-300 bg-emerald-50 dark:border-emerald-900 dark:bg-emerald-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">模型意图不是执行权限\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">模型可以发出有效的工具请求，但仍可能被拒绝。授权、参数验证、速率限制、事务规则、审计要求和回滚都属于模型之外。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-29\">4. 上下文：模型当前能看到什么\u003C\u002Fh2>\n\u003Cp>上下文是模型在特定推理步骤中可用的信息。Anthropic 的上下文工程指南将上下文描述为从 LLM 采样时包含的 token 集合。在实践中，该集合可以包含系统指令、用户消息、对话历史、检索到的证据、工具定义、工具结果、记忆摘要和选定的应用程序状态。\u003C\u002Fp>\n\u003Cp>因此，上下文既不是完整的知识库，也不是长期记忆。一家公司可能存储一千万份文档，但只有少数段落进入一次模型调用。运行时可能保留一年的对话历史，但只暴露当前任务所需的部分。\u003C\u002Fp>\n\u003Cp>上下文窗口还带来工程约束。添加更多文本并不能保证更好的答案；无关、过时、矛盾或低权威性的信息可能会稀释真正重要的证据。\u003C\u002Fp>\n\u003Ch2 id=\"section-33\">5. 运行时与编排：协调循环\u003C\u002Fh2>\n\u003Cp>运行时或编排层协调模型如何参与任务。根据架构的不同，它可以管理会话、模型请求、工具发现、工具调用循环、重试、交接、流式事件、超时、检查点、压缩或执行环境。\u003C\u002Fp>\n\u003Cp>有些运行时只是围绕模型 API 的薄应用代码。另一些则是完整的智能体框架。托管供应商运行时可以拥有循环的一部分，而应用程序仍然拥有领域真相、授权、业务副作用和产品生命周期。\u003C\u002Fp>\n\u003Cp>这一边界很重要，因为运行时在哪里运行和推理在哪里运行是分开的决策。本地运行的客户端或智能体进程仍然可以调用远程模型，而远程应用程序可以调用托管在组织控制的基础设施上的模型。\u003C\u002Fp>\n\u003Ch2 id=\"section-37\">6. 应用：AI 成为产品的地方\u003C\u002Fh2>\n\u003Cp>应用是围绕 AI 组件的产品边界。它拥有用户体验、领域模型、当前状态、身份、租户范围、权限、持久化、服务集成、验证、可观测性、计费或配额逻辑（如相关），以及决定 AI 被允许看到或做什么的规则。\u003C\u002Fp>\n\u003Cp>这一层将“模型可以产生有用的输出”转变为“系统可以提供可靠的能力”。同一个模型可以参与私人研究助手、支持工作流、代码代理或商业应用，因为周围的应用程序改变了数据、工具、策略、状态和执行契约。\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--info my-6 rounded-xl border p-5 border-blue-300 bg-blue-50 dark:border-blue-900 dark:bg-blue-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">模型是可替换的；产品边界不是\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">提供商和模型的替换可以是一个架构目标。应用程序的权威状态、权限、领域规则、审计跟踪和用户契约不能简单地委托给当前选择的任何模型。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-41\">各部分如何在真实请求中协同工作\u003C\u002Fh2>\n\u003Cp>考虑一个支持助手被问到：“如果订单 4711 仍然符合条件，请退款，并解释原因。”该请求结合了知识、当前状态、授权、推理和副作用。\u003C\u002Fp>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">需求\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">正确的层\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">原因\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">退款政策\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">检索\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">系统必须找到当前适用的政策并保留其来源。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">订单 4711 状态\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">直接数据\u002F工具访问\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">当前订单记录是易变的权威状态，不应从模型知识中猜测。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">用户的退款权限\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">应用\u002F授权\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">权限必须独立于模型的要求来执行。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">根据订单事实解释政策\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型 + 上下文\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型可以对提供给它的政策证据和当前订单状态进行推理。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">执行退款\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">工具 + 应用事务规则\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">受控的外部操作改变真实状态。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">解释结果\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型可以根据验证结果生成面向用户的解释。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">审计发生了什么\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">应用\u002F运行时\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">系统按要求记录证据、调用、决策、副作用和错误。\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Cp>如果助手只有语言模型，它可以讨论退款，但无法安全地知道订单 4711 当前是否符合条件或执行交易。如果它只有检索，它可能找到政策，但仍然缺乏实时订单状态。如果它有工具但没有应用授权，它可能变得有能力但不安全。可靠性来自于以明确的所有权组合各层。\u003C\u002Fp>\n\u003Ch2 id=\"section-45\">不同的 AI 产品使用不同的组合\u003C\u002Fh2>\n\u003Csection class=\"editorjs-comparison my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">模型的存在并不定义整个架构\u003C\u002Fh3>\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left dark:border-gray-700 dark:bg-gray-900\">\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">检索\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">工具\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">权威状态\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">典型能力\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">仅模型助手\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">检索增强助手\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">使用工具的助手\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">代理应用\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Cp>这些是架构模式，不是成熟度排名。当任务不需要外部事实或操作时，仅模型功能可以是正确的设计。只有当任务需要这些能力时，添加检索、工具、记忆或代理循环才是合理的。\u003C\u002Fp>\n\u003Ch2 id=\"section-48\">实现证据：Aaasaasa AI Client\u003C\u002Fh2>\n\u003Caside class=\"editorjs-callout editorjs-callout--note my-6 rounded-xl border p-5 border-gray-300 bg-gray-50 dark:border-gray-700 dark:bg-gray-900\u002F40\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">主要实现证据\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">以下部分描述了我根据截至 \u003Cstrong>2026 年 7 月 26 日\u003C\u002Fstrong> 的 Aaasaasa AI Client 代码库和架构文档构建并审查的实现。它是这些边界有用性的证据，而不是声称某个实现是通用标准或商业部署的企业产品。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Cp>Aaasaasa AI Client 是一个使用 Nuxt 4、Electron 和 TypeScript 构建的本地优先桌面 AI 工作区。其 AI Hub 有意将代理\u002F客户端、提供商、模型、运行时位置、权限和 Web 客户端分开，而不是将它们视为一个“AI”设置。\u003C\u002Fp>\n\u003Cp>这种分离创造了具体的行为。Direct Chat 可以与模型对话，而无需文件系统或 shell 工具。Codex 代理可以使用选定的工作区和权限配置文件。Ollama 可以提供直接的本地推理，而 LM Studio 和可配置的 OpenAI 兼容端点代表其他提供商路径。本地运行的 Codex 进程仍然可以使用云模型，因此 UI 和架构并不将本地运行时等同于本地推理。\u003C\u002Fp>\n\u003Cp>该实现还包含 Qdrant\u002F向量支持、文档提取功能和经过身份验证的目录 MCP 代理。这些组件说明了另一个边界：检索基础设施和工具访问可以存在于同一产品中，而不会成为模型本身的属性。\u003C\u002Fp>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">A01 概念\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Aaasaasa AI Client 实现证据\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">特定于提供商的模型标识符与提供商和运行时分开选择。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">提供商\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Ollama、LM Studio、OpenAI 兼容服务和其他提供商路径被分别表示。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">运行时\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">本地或远程代理\u002F运行时位置独立于模型进行跟踪。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">工具\u002F访问\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Direct Chat 没有文件系统或 shell 工具；受控的目录访问被单独代理。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">权限\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">工作区权限配置文件是应用\u002F会话策略，而不是模型能力。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">检索基础设施\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">向量支持和文档提取作为数据\u002F检索能力存在，而不是模型功能。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">应用\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Electron\u002FNuxt 产品协调 UI、凭据、提供商、运行时发现、权限、工具和模型交互。\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Caside class=\"editorjs-callout editorjs-callout--success my-6 rounded-xl border p-5 border-emerald-300 bg-emerald-50 dark:border-emerald-900 dark:bg-emerald-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">实现教训\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">一旦 \u003Cstrong>模型、提供商、运行时、权限、工具、数据和客户端\u003C\u002Fstrong> 不再被表示为一个配置选择，架构就变得更容易理解。这种区分是操作性的：它决定了什么可以在本地运行、什么可以访问文件、什么可以调用付费云推理，以及哪一层拥有授权。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-55\">常见的类别错误\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">类别错误\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">实际发生的情况\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“AI 知道我们的文档。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">应用程序或检索层将选定的文档内容提供给模型。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“RAG 就是我们的向量数据库。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">向量数据库可以是检索管道使用的一个索引或存储；RAG 是检索加生成的模式。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“模型调用了我们的 CRM。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型生成了工具请求；运行时\u002F应用程序授权并执行了外部调用。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“它是本地 AI，因为桌面代理在本地运行。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">运行时位置和推理位置是分开的。本地运行时仍然可以调用远程模型。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“模型有权限编辑文件。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">应用程序\u002F运行时在权限策略下授予工具能力；权限不是模型的内在属性。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“更多上下文意味着更多知识。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">上下文是为一次推理提供的有限输入。更大的上下文可能包含更多噪音、冲突或过时信息。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“聊天机器人就是 AI 架构。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">聊天界面只是一种接口。系统还可以包括身份、状态、检索、工具、运行时、验证、持久化和可观测性。\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-57\">边界崩溃时的故障模式\u003C\u002Fh2>\n\u003Cp>边界错误不仅仅是术语问题。它们会造成不同的生产故障，需要不同的修复方法。\u003C\u002Fp>\n\u003Csection class=\"editorjs-comparison my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">在更换模型之前先诊断故障层\u003C\u002Fh3>\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left dark:border-gray-700 dark:bg-gray-900\">\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">症状\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">可能的边界问题\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">首要架构检查\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">答案过时\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">缺少公司事实\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">不安全的副作用\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">提供了大量文本但答案混乱\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">意外的云使用\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">代理停滞或重复\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-60\">什么是稳定的，什么对版本敏感？\u003C\u002Fh2>\n\u003Cp>本文中的架构区分有意保持供应商中立。下面的当前示例是实施事实，在 API 演变时应重新检查。\u003C\u002Fp>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">领域\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">稳定的架构理念\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">2026 年 10 月 8 日验证的当前示例\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">AI 模型与系统\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型是更广泛系统中的一个组件\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">NIST 当前的术语表分别定义了 AI 模型和 AI 系统。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">RAG\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">生成可以以检索到的外部信息为条件\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Lewis 等人 2020 年的表述仍然是基础参考；生产检索方法现在已远远超出单一密集索引设计。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">托管检索\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">检索可以作为托管工具暴露\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">OpenAI File Search 目前是一个 Responses API 工具，使用语义和关键词检索来搜索上传文件的知识库。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">函数\u002F工具调用\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型可以请求应用程序定义的外部能力\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">OpenAI 目前将函数调用记录为与外部系统、数据和操作的接口。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">上下文工程\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型行为取决于为当前推理提供的有限信息\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Anthropic 当前的工程指南将上下文定义为从 LLM 采样时包含的 token 集合，并专注于策划该集合。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">供应商 API\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">SDK、工具名称、端点形态和支持的功能会变化\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">即使责任边界保持稳定，也要将供应商文档视为对版本敏感的。\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Cp>因此，一篇权威文章应同时保留两个层面：用于架构的稳定概念，以及用于当前实现的带日期证据。将两者混为一谈会使文章不必要地快速过时。\u003C\u002Fp>\n\u003Ch2 id=\"section-64\">AI 组件边界测试\u003C\u002Fh2>\n\u003Cp>在评估 AI 功能时，请按顺序提出以下问题。答案揭示了系统实际拥有哪些组件，以及哪些责任仍然是隐含的。\u003C\u002Fp>\n\u003Csection class=\"editorjs-process my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">生产设计的七个问题\u003C\u002Fh3>\u003Cdiv class=\"grid grid-cols-1 md:grid-cols-2 xl:grid-cols-3 gap-4\">\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">1\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">1. 什么生成输出？\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">确定确切的模型及其提供的模态或结构化输出。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">2\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">2. 模型之外哪些事实是权威的？\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">确定文档、数据库、API、当前状态和其他事实来源。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">3\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">3. 如何选择相关信息？\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">区分直接查找、搜索、检索、排序和上下文构建。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">4\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">4. 什么可能导致真实的副作用？\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">列出工具和外部操作，然后确定谁验证和授权它们。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">5\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">5. 什么作为上下文到达模型？\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">明确指令、证据、状态、历史、记忆和工具定义。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">6\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">6. 谁拥有循环？\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">确定管理调用、事件、重试、工具循环和会话的运行时或框架。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">7\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">7. 什么仍然是应用程序的责任？\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">明确身份、权限、领域状态、验证、持久化、可观测性和用户体验。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-67\">生成式 AI 不是什么\u003C\u002Fh2>\n\u003Cp>生成式 AI 不是 LLM 的同义词，尽管 LLM 是生成式模型的主要类别。它也不是 RAG、向量数据库、代理、工具协议、聊天机器人界面或应用程序的同义词。\u003C\u002Fp>\n\u003Cp>这些概念可以相互关联，但每个概念回答不同的架构问题。LLM 问的是语言输出如何产生。检索问的是外部证据来自哪里。工具问的是外部能力如何暴露。上下文问的是模型能看到什么。运行时问的是执行如何协调。应用程序问的是能力如何成为受控产品。\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--success my-6 rounded-xl border p-5 border-emerald-300 bg-emerald-50 dark:border-emerald-900 dark:bg-emerald-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">如果你只记住一个模型\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">\u003Cstrong>模型 = 生成。\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>检索 = 寻找证据。\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>工具 = 在模型之外读取或行动。\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>上下文 = 模型现在看到的内容。\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>运行时 = 协调执行。\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>应用程序 = 拥有产品、状态、规则和权限。\u003C\u002Fstrong>\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-71\">知识图谱中接下来去哪里\u003C\u002Fh2>\n\u003Cp>一旦这些边界清晰，更深入的主题就更容易定位。RAG 属于检索和上下文构建。检索触发器决定何时需要外部证据。代理记忆关注跨时间持续存在的内容。工具调用和 MCP 属于能力访问。代理框架属于运行时编排。RBAC、租户隔离和领域授权属于应用程序和平台安全边界。\u003C\u002Fp>\n\u003Caside class=\"editorjs-referral my-6\">\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fwhere-does-an-llm-get-its-data-rag-data-sources-in-python\" class=\"flex flex-col sm:flex-row gap-4 rounded-xl border border-gray-200 dark:border-gray-700 p-4 transition hover:border-primary-500\">\u003Cdiv class=\"min-w-0 flex-1\">\u003Cstrong class=\"block text-lg text-gray-900 dark:text-gray-100\">LLM 从哪里获取数据？Python 中的 RAG 数据源\u003C\u002Fstrong>\u003Cp class=\"mt-2 text-sm text-gray-600 dark:text-gray-300\">一个实用的续篇，展示文件、SQL、API、全文搜索、嵌入和上下文组装如何将外部数据连接到 LLM。\u003C\u002Fp>\u003Cspan class=\"mt-3 inline-flex text-sm font-medium text-primary-600 dark:text-primary-400\">在代码中查看数据路径 →\u003C\u002Fspan>\u003C\u002Fdiv>\u003C\u002Fa>\u003C\u002Faside>\n\u003Caside class=\"editorjs-referral my-6\">\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fwhen-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger\" class=\"flex flex-col sm:flex-row gap-4 rounded-xl border border-gray-200 dark:border-gray-700 p-4 transition hover:border-primary-500\">\u003Cdiv class=\"min-w-0 flex-1\">\u003Cstrong class=\"block text-lg text-gray-900 dark:text-gray-100\">AI 何时应停止信任自身知识？——检索触发机制\u003C\u002Fstrong>\u003Cp class=\"mt-2 text-sm text-gray-600 dark:text-gray-300\">一个决策模型，用于判断 AI 系统何时应停止仅依赖模型知识并获取外部证据。\u003C\u002Fp>\u003Cspan class=\"mt-3 inline-flex text-sm font-medium text-primary-600 dark:text-primary-400\">阅读检索决策模型 →\u003C\u002Fspan>\u003C\u002Fdiv>\u003C\u002Fa>\u003C\u002Faside>\n\u003Ch2 id=\"section-75\">局限性\u003C\u002Fh2>\n\u003Cp>六层模型是一张责任地图，而不是要求每个产品都部署六个独立服务。小型应用可以在一个进程内实现上下文构建、检索和编排。托管平台可以将多项职责捆绑在一个 API 之后。物理部署可以合并，而语义归属仍然保持独立。\u003C\u002Fp>\n\u003Cp>不同供应商和研究中的术语也各不相同。“智能体”、“运行时”、“记忆”、“工具”、“连接器”和“上下文”可能有不同的定义。这里的定义旨在明确操作归属和故障诊断，而不是声称每个框架都使用相同的词汇。\u003C\u002Fp>\n\u003Cp>Aaasaasa AI Client 部分记录了一种实现模式。它表明明确的边界是可行的，但并不能证明相同的组件布局对每个 AI 产品都是最优的。\u003C\u002Fp>\n\u003Ch2 id=\"section-79\">什么会改变这个答案？\u003C\u002Fh2>\n\u003Cp>如果模型架构本身开始将权威外部状态、权限、持久事务副作用和可验证的源访问作为内在属性，而不是由周围系统提供的能力，那么责任地图将需要修订。当前的生产架构并未使这成为一个安全的普遍假设。\u003C\u002Fp>\n\u003Cp>个别实现示例会更快地发生变化。托管检索工具、智能体 API、MCP 集成、上下文管理功能和提供商能力都在快速演进。这些细节应更新，而不应破坏生成、证据、能力访问、上下文、执行和应用控制之间的基本区别。\u003C\u002Fp>\n\u003Ch2 id=\"section-82\">结论\u003C\u002Fh2>\n\u003Cp>一旦“AI”不再被视为一个黑箱，生成式 AI 的设计就会变得更容易。模型是生成组件，而不是完整产品。检索提供外部证据。工具暴露能力。上下文将选定的信息带入当前推理。运行时协调执行。应用程序拥有权威的产品边界。\u003C\u002Fp>\n\u003Cp>这种分离不仅仅用于解释。它告诉工程师过时事实的来源、授权应属于哪里、为什么本地运行时仍可使用云推理、为什么 RAG 不等于向量数据库、为什么工具调用需要验证，以及为什么更改模型无法修复所有系统故障。\u003C\u002Fp>\n\u003Cp>因此，持久的架构问题不是“我们使用哪个 AI 模型？”而是：每个组件拥有哪些职责，哪些证据跨越每个边界，以及哪一层被允许改变真实状态？\u003C\u002Fp>\n\u003Ch2 id=\"section-86\">常见问题\u003C\u002Fh2>\n\u003Csection class=\"editorjs-faq my-6 rounded-xl border border-gray-200 p-5 dark:border-gray-700\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">生成式 AI 系统边界\u003C\u002Fh3>\u003Cdiv id=\"faq1\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">生成式 AI 和 LLM 是一回事吗？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">不是。LLM 是一种生成模型。生成式 AI 还包括其他模态，生产级生成式 AI 系统可以在模型周围包含检索、工具、运行时逻辑、应用状态、权限、持久化和用户界面。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq2\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">RAG 是模型的一部分吗？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">通常不是。RAG 是一种应用\u002F系统模式，它检索外部信息并向模型提供选定的证据。一些平台将检索与模型 API 紧密打包，但职责仍然不同。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq3\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">RAG 需要向量数据库吗？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">不需要。RAG 可以使用向量搜索、词法搜索、混合检索、SQL、API、知识图谱或其他方法。其定义属性是为生成而检索外部信息，而不是某一种存储技术。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq4\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">工具和上下文是一回事吗？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">不是。工具是一种外部能力。它的定义可以在上下文中表示，其结果稍后也可能进入上下文，但实际能力在模型外部执行。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq5\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">在本地运行 AI 客户端意味着模型是本地吗？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">不是。运行时位置和推理位置是分开的。本地桌面应用或智能体可以调用远程模型，而远程应用可以调用内部托管的模型。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq6\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">谁应该为 AI 工具强制执行权限？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">应用程序或运行时安全边界应强制执行授权。模型可以请求操作，但模型意图绝不应被视为足够的执行权限。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq7\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">当前应用状态应属于哪里？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">权威的易失状态通常应保留在拥有它的应用程序或领域系统中。AI 可以在需要时通过受控上下文或工具访问接收相关状态。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-88\">术语表\u003C\u002Fh2>\n\u003Csection class=\"editorjs-glossary my-6 rounded-xl border border-gray-200 dark:border-gray-700 p-5\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">核心术语\u003C\u002Fh3>\u003Cdl>\u003Cdiv id=\"generative-model\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">生成模型\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">一种 AI 模型，旨在生成衍生的合成内容，如文本、图像、音频、视频、代码或结构化输出。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"retrieval\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">检索\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">从外部源或存储中为当前任务选择相关信息的过程。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"rag\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">RAG\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">检索增强生成：一种模式，将检索到的外部信息提供给生成模型以改进当前输出。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"tool\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">工具\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">暴露给 AI 运行时用于读取数据、计算、搜索或执行外部操作的能力。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"context\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">上下文\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">模型在特定推理步骤中可用的信息。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"runtime-orchestrator\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">运行时 \u002F 编排器\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">协调模型调用、工具调用、任务循环、会话、重试、事件或执行环境的软件层。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"application\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">应用程序\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">拥有用户交互、权威状态、权限、验证、持久化和业务行为的产品和领域层。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"provider\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">提供商\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">暴露对一个或多个模型访问权限的服务或运行时；提供商身份和模型身份是分开的关注点。\u003C\u002Fdd>\u003C\u002Fdiv>\u003C\u002Fdl>\u003C\u002Fsection>\n\u003Ch2 id=\"section-90\">主要来源和实现证据\u003C\u002Fh2>\n\u003Cp>以下稳定定义以标准\u002F研究为依据；快速变化的实现示例使用当前官方工程文档。Aaasaasa AI Client 是原始实现证据，并已对照其截至 2026 年 7 月 26 日的代码库\u002F文档状态进行核查。\u003C\u002Fp>\n\u003Ca href=\"https:\u002F\u002Fnvlpubs.nist.gov\u002Fnistpubs\u002Fai\u002FNIST.AI.600-1.pdf\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">NIST AI 600-1 — 生成式人工智能概况\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">NIST 的生成式 AI 概况，包括生成式 AI 的定义，并明确区分模型级、系统级、应用级和用例级关注点。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fcsrc.nist.gov\u002Fglossary\u002Fterm\u002Fartificial_intelligence_model\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">NIST — 人工智能模型\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">NIST 术语表中对 AI 模型的当前定义：信息系统的组成部分，使用 AI 技术从输入产生输出。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fcsrc.nist.gov\u002Fglossary\u002Fterm\u002Fartificial_intelligence_system\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">NIST — 人工智能系统\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">NIST 术语表中当前的 AI 系统定义，表明 AI 系统可以包括使用 AI 的数据系统、软件、硬件、应用程序、工具或实用程序。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2005.11401\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">Lewis 等 — 面向知识密集型 NLP 任务的检索增强生成\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">2020 年提出 RAG 公式的论文，该公式将生成模型与检索到的非参数记忆相结合。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Ftools-file-search\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">OpenAI — 文件搜索\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">Responses API 中托管文件检索的当前官方文档，使用上传文件知识库、语义搜索和关键词搜索。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Ffunction-calling\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">OpenAI — 函数调用\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">当前官方文档，将工具\u002F函数调用描述为模型与外部系统、数据和操作之间的接口。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Feffective-context-engineering-for-ai-agents\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">Anthropic — 面向 AI 智能体的有效上下文工程\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">工程指南将上下文定义为 LLM 采样期间可用的 token 集合，并解释为什么上下文选择是一个有限资源问题。\u003C\u002Fp>\u003C\u002Fa>",{"time":212,"blocks":213,"version":1065},1791475747074,[214,220,228,235,243,248,253,258,265,270,275,307,312,317,360,365,370,375,380,385,390,395,402,411,416,421,426,431,437,442,447,452,457,462,467,472,477,482,487,492,498,503,508,543,548,553,584,589,594,600,605,610,615,643,649,654,683,688,693,733,738,743,776,781,786,791,818,823,828,833,839,844,849,857,865,870,875,880,885,890,895,900,905,910,915,920,925,959,964,991,996,1001,1011,1020,1029,1038,1047,1056],{"id":215,"data":216,"type":218,"tunes":219},"intro",{"text":217},"生成式人工智能并非单一组件。一个生产级生成式人工智能系统通常将生成模型与应用程序代码相结合，后者提供指令和上下文，在需要时检索外部知识，暴露用于读取或更改外部系统的工具，管理运行时状态和权限，并将结果转化为可用的产品。将模型、检索、工具、上下文、运行时和应用程序视为同一事物，会掩盖决定新鲜度、安全性、可靠性、成本和控制的边界。","paragraph",{},{"id":221,"data":222,"type":226,"tunes":227},"direct",{"body":223,"title":224,"variant":225},"\u003Cstrong>模型负责生成；检索负责查找外部证据；工具负责访问数据或执行操作；上下文是模型在当前推理中能看到的内容；运行时协调执行；应用程序拥有产品规则、状态、权限、持久化和用户体验。\u003C\u002Fstrong>这些层可以由供应商打包在一起，但它们的职责仍然不同。","直接回答","info","callout",{},{"id":229,"data":230,"type":226,"tunes":234},"scope-note",{"body":231,"title":232,"variant":233},"本文定义的是持久的架构职责，而非某一供应商的技术栈。当前实现示例已于\u003Cstrong>2026年10月8日\u003C\u002Fstrong>重新核对。供应商API和产品名称可能会变化；职责边界比任何单个SDK或端点都更稳定。","术语和版本说明","note",{},{"id":236,"data":237,"type":241,"tunes":242},"toc",{"title":238,"maxLevel":239,"minLevel":240},"目录",3,2,"tableOfContents",{},{"id":244,"data":245,"type":42,"tunes":247},"h-meaning",{"text":246,"level":240},"“生成式人工智能”究竟意味着什么？",{},{"id":249,"data":250,"type":218,"tunes":252},"p-meaning-1",{"text":251},"在模型层面，生成式人工智能指的是能够生成衍生合成内容（如文本、图像、音频、视频、代码或其他数字输出）的人工智能模型。NIST AI 600-1使用了这种面向模型的含义，并分别在模型、系统、应用程序和用例层面讨论风险。",{},{"id":254,"data":255,"type":218,"tunes":257},"p-meaning-2",{"text":256},"这种区分很重要，因为人工智能模型与完整的人工智能系统不是一回事。NIST当前术语表将人工智能模型定义为使用计算、统计或机器学习技术从输入产生输出的组件，而人工智能系统可以包括使用人工智能运行的软件、硬件、应用程序、工具或实用程序。",{},{"id":259,"data":260,"type":226,"tunes":264},"model-system-rule",{"body":261,"title":262,"variant":263},"\u003Cstrong>生成模型 ≠ 生成式人工智能应用。\u003C\u002Fstrong>\u003Cbr>模型是一个计算组件。可用的AI产品是围绕该组件构建的系统。","一个有用的边界","success",{},{"id":266,"data":267,"type":42,"tunes":269},"h-simple",{"text":268,"level":240},"生成式人工智能系统的最简单有用模型",{},{"id":271,"data":272,"type":218,"tunes":274},"p-simple-1",{"text":273},"作为第一个心智模型，想象一个公司助手回答：“这位客户今天能获得退款吗？”一个有用的答案可能需要几种不同的职责。语言模型可以解释问题并撰写说明，但当前订单状态可能来自数据库工具，退款政策可能来自文档检索，权限可能由应用程序强制执行，而最终操作可能需要受控的API调用。",{},{"id":276,"data":277,"type":305,"tunes":306},"simple-flow",{"steps":278,"title":303,"orientation":304},[279,282,285,288,291,294,297,300],{"label":280,"description":281},"1. 用户请求","应用程序接收自然语言问题或任务。",{"label":283,"description":284},"2. 应用程序策略和状态","身份、租户、权限、当前工作流状态和产品规则定义了该请求被允许做什么。",{"label":286,"description":287},"3. 检索或直接数据访问","当模型知识不足时，系统获取外部证据或当前事实。",{"label":289,"description":290},"4. 上下文构建","为模型组装指令、用户输入、选定的证据、相关状态和工具定义。",{"label":292,"description":293},"5. 模型推理","生成模型解释提供的上下文并产生文本、结构化输出或工具请求。",{"label":295,"description":296},"6. 需要时执行工具","运行时或应用程序在模型外部验证并执行已批准的工具调用。",{"label":298,"description":299},"7. 观察与继续","工具结果可以作为新上下文返回给模型，用于下一步推理。",{"label":301,"description":302},"8. 验证和产品输出","应用程序验证结果，记录所需状态或审计数据，并呈现或执行最终结果。","一种常见的执行路径","auto","processFlow",{},{"id":308,"data":309,"type":218,"tunes":311},"p-simple-2",{"text":310},"真实系统并不总是严格遵循这个顺序。检索可以在第一次模型调用之前发生，工具可以在代理循环中选择，确定性的应用程序逻辑可以完全绕过模型，验证可以在多个阶段进行。重点是将职责分开，而不是强加一个通用的工作流。",{},{"id":313,"data":314,"type":42,"tunes":316},"h-boundaries",{"text":315,"level":240},"六个重要的边界",{},{"id":318,"data":319,"type":358,"tunes":359},"boundary-comparison",{"rows":320,"title":346,"layout":347,"columns":348},[321,326,330,334,338,342],{"id":322,"label":323,"values":324},"model","模型",[325,325,325],"",{"id":327,"label":328,"values":329},"retrieval","检索",[325,325,325],{"id":331,"label":332,"values":333},"tools","工具",[325,325,325],{"id":335,"label":336,"values":337},"context","上下文",[325,325,325],{"id":339,"label":340,"values":341},"runtime","运行时\u002F编排器",[325,325,325],{"id":343,"label":344,"values":345},"application","应用程序",[325,325,325],"一个AI产品中的六项职责","table",[349,352,355],{"id":350,"label":351},"job","主要工作",{"id":353,"label":354},"input","典型输入",{"id":356,"label":357},"not","不同于","comparison",{},{"id":361,"data":362,"type":42,"tunes":364},"h-model",{"text":363,"level":240},"1. 模型：生成是其核心职责",{},{"id":366,"data":367,"type":218,"tunes":369},"p-model-1",{"text":368},"生成模型将提供的输入映射为生成的输出。对于语言模型，这可以包括自然语言文本、结构化JSON、代码、分类、摘要、计划或工具调用参数。多模态生成模型可以处理额外的输入和输出类型。",{},{"id":371,"data":372,"type":218,"tunes":374},"p-model-2",{"text":373},"模型可以在其参数中包含大量学习到的知识，但参数化知识不是实时数据库。除非通过当前输入路径提供信息，否则模型不会自动知道五分钟前创建的文档、当前库存水平、私人客户记录或应用程序状态。",{},{"id":376,"data":377,"type":218,"tunes":379},"p-model-3",{"text":378},"这就是为什么更改模型不能自动解决知识过时、权限缺失、检索损坏、状态所有权不正确或不安全的工具执行问题。这些故障通常属于其他层。",{},{"id":381,"data":382,"type":42,"tunes":384},"h-retrieval",{"text":383,"level":240},"2. 检索：查找外部证据是一项独立操作",{},{"id":386,"data":387,"type":218,"tunes":389},"p-retrieval-1",{"text":388},"检索在生成之前或生成过程中从外部来源选择信息。Lewis 等人于 2020 年提出的检索增强生成工作通过将参数化生成模型与检索到的非参数化记忆相结合，使这种分离变得明确。现代生产系统使用许多检索变体，但架构理念保持不变：有用的证据可以在推理时获取，而不是仅依赖模型在训练期间学到的内容。",{},{"id":391,"data":392,"type":218,"tunes":394},"p-retrieval-2",{"text":393},"检索可以使用词法搜索、嵌入、向量搜索、混合搜索、SQL、知识图谱、元数据过滤器、API 或其他选择机制。因此，向量数据库是一种可能的检索组件，而不是 RAG 的定义。",{},{"id":396,"data":397,"type":226,"tunes":401},"retrieval-rule",{"body":398,"title":399,"variant":400},"检索到的段落可能高度相关，但仍然过时、未经授权、来自错误版本，或不足以支持某项主张。检索质量和证据质量必须分开评估。","相关性不等于权威性","warning",{},{"id":403,"data":404,"type":409,"tunes":410},"ref-rag",{"url":405,"title":406,"excerpt":407,"ctaLabel":408},"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works","什么是 RAG？对其工作原理的最简单解释","对检索增强生成的经典通俗解释，包括 LLM、知识、状态、记忆和工具之间的分离。","阅读 RAG 基础","referralArticle",{},{"id":412,"data":413,"type":42,"tunes":415},"h-tools",{"text":414,"level":240},"3. 工具：访问和操作不是模型知识",{},{"id":417,"data":418,"type":218,"tunes":420},"p-tools-1",{"text":419},"工具是一种接口，AI 运行时可以通过它请求模型之外的功能。工具可以查询数据库、搜索网络、读取文件、计算数值、调用内部服务、创建工单、发送消息、修改记录或触发另一项受控操作。",{},{"id":422,"data":423,"type":218,"tunes":425},"p-tools-2",{"text":424},"OpenAI 当前的函数调用文档明确了这一边界：函数调用让模型能够与外部系统交互，并访问应用程序提供的数据或操作。模型可以提议或选择调用，但真正执行操作的是外部系统。",{},{"id":427,"data":428,"type":218,"tunes":430},"p-tools-3",{"text":429},"因此，工具使用产生了两个独立问题：模型能否请求此能力？以及应用程序是否会授权并执行它？生产系统不应将模型意图与造成副作用的权限混为一谈。",{},{"id":432,"data":433,"type":226,"tunes":436},"tool-rule",{"body":434,"title":435,"variant":263},"模型可以发出有效的工具请求，但仍可能被拒绝。授权、参数验证、速率限制、事务规则、审计要求和回滚都属于模型之外。","模型意图不是执行权限",{},{"id":438,"data":439,"type":42,"tunes":441},"h-context",{"text":440,"level":240},"4. 上下文：模型当前能看到什么",{},{"id":443,"data":444,"type":218,"tunes":446},"p-context-1",{"text":445},"上下文是模型在特定推理步骤中可用的信息。Anthropic 的上下文工程指南将上下文描述为从 LLM 采样时包含的 token 集合。在实践中，该集合可以包含系统指令、用户消息、对话历史、检索到的证据、工具定义、工具结果、记忆摘要和选定的应用程序状态。",{},{"id":448,"data":449,"type":218,"tunes":451},"p-context-2",{"text":450},"因此，上下文既不是完整的知识库，也不是长期记忆。一家公司可能存储一千万份文档，但只有少数段落进入一次模型调用。运行时可能保留一年的对话历史，但只暴露当前任务所需的部分。",{},{"id":453,"data":454,"type":218,"tunes":456},"p-context-3",{"text":455},"上下文窗口还带来工程约束。添加更多文本并不能保证更好的答案；无关、过时、矛盾或低权威性的信息可能会稀释真正重要的证据。",{},{"id":458,"data":459,"type":42,"tunes":461},"h-runtime",{"text":460,"level":240},"5. 运行时与编排：协调循环",{},{"id":463,"data":464,"type":218,"tunes":466},"p-runtime-1",{"text":465},"运行时或编排层协调模型如何参与任务。根据架构的不同，它可以管理会话、模型请求、工具发现、工具调用循环、重试、交接、流式事件、超时、检查点、压缩或执行环境。",{},{"id":468,"data":469,"type":218,"tunes":471},"p-runtime-2",{"text":470},"有些运行时只是围绕模型 API 的薄应用代码。另一些则是完整的智能体框架。托管供应商运行时可以拥有循环的一部分，而应用程序仍然拥有领域真相、授权、业务副作用和产品生命周期。",{},{"id":473,"data":474,"type":218,"tunes":476},"p-runtime-3",{"text":475},"这一边界很重要，因为运行时在哪里运行和推理在哪里运行是分开的决策。本地运行的客户端或智能体进程仍然可以调用远程模型，而远程应用程序可以调用托管在组织控制的基础设施上的模型。",{},{"id":478,"data":479,"type":42,"tunes":481},"h-application",{"text":480,"level":240},"6. 应用：AI 成为产品的地方",{},{"id":483,"data":484,"type":218,"tunes":486},"p-app-1",{"text":485},"应用是围绕 AI 组件的产品边界。它拥有用户体验、领域模型、当前状态、身份、租户范围、权限、持久化、服务集成、验证、可观测性、计费或配额逻辑（如相关），以及决定 AI 被允许看到或做什么的规则。",{},{"id":488,"data":489,"type":218,"tunes":491},"p-app-2",{"text":490},"这一层将“模型可以产生有用的输出”转变为“系统可以提供可靠的能力”。同一个模型可以参与私人研究助手、支持工作流、代码代理或商业应用，因为周围的应用程序改变了数据、工具、策略、状态和执行契约。",{},{"id":493,"data":494,"type":226,"tunes":497},"app-rule",{"body":495,"title":496,"variant":225},"提供商和模型的替换可以是一个架构目标。应用程序的权威状态、权限、领域规则、审计跟踪和用户契约不能简单地委托给当前选择的任何模型。","模型是可替换的；产品边界不是",{},{"id":499,"data":500,"type":42,"tunes":502},"h-work-together",{"text":501,"level":240},"各部分如何在真实请求中协同工作",{},{"id":504,"data":505,"type":218,"tunes":507},"p-together-1",{"text":506},"考虑一个支持助手被问到：“如果订单 4711 仍然符合条件，请退款，并解释原因。”该请求结合了知识、当前状态、授权、推理和副作用。",{},{"id":509,"data":510,"type":347,"tunes":542},"support-table",{"content":511,"stretched":43,"withHeadings":14},[512,516,519,523,527,531,535,538],[513,514,515],"需求","正确的层","原因",[517,328,518],"退款政策","系统必须找到当前适用的政策并保留其来源。",[520,521,522],"订单 4711 状态","直接数据\u002F工具访问","当前订单记录是易变的权威状态，不应从模型知识中猜测。",[524,525,526],"用户的退款权限","应用\u002F授权","权限必须独立于模型的要求来执行。",[528,529,530],"根据订单事实解释政策","模型 + 上下文","模型可以对提供给它的政策证据和当前订单状态进行推理。",[532,533,534],"执行退款","工具 + 应用事务规则","受控的外部操作改变真实状态。",[536,323,537],"解释结果","模型可以根据验证结果生成面向用户的解释。",[539,540,541],"审计发生了什么","应用\u002F运行时","系统按要求记录证据、调用、决策、副作用和错误。",{},{"id":544,"data":545,"type":218,"tunes":547},"p-together-2",{"text":546},"如果助手只有语言模型，它可以讨论退款，但无法安全地知道订单 4711 当前是否符合条件或执行交易。如果它只有检索，它可能找到政策，但仍然缺乏实时订单状态。如果它有工具但没有应用授权，它可能变得有能力但不安全。可靠性来自于以明确的所有权组合各层。",{},{"id":549,"data":550,"type":42,"tunes":552},"h-configs",{"text":551,"level":240},"不同的 AI 产品使用不同的组合",{},{"id":554,"data":555,"type":358,"tunes":583},"config-comparison",{"rows":556,"title":573,"layout":347,"columns":574},[557,561,565,569],{"id":558,"label":559,"values":560},"bare","仅模型助手",[325,325,325,325],{"id":562,"label":563,"values":564},"rag","检索增强助手",[325,325,325,325],{"id":566,"label":567,"values":568},"tool","使用工具的助手",[325,325,325,325],{"id":570,"label":571,"values":572},"agent","代理应用",[325,325,325,325],"模型的存在并不定义整个架构",[575,576,577,580],{"id":327,"label":328},{"id":331,"label":332},{"id":578,"label":579},"state","权威状态",{"id":581,"label":582},"result","典型能力",{},{"id":585,"data":586,"type":218,"tunes":588},"p-configs-1",{"text":587},"这些是架构模式，不是成熟度排名。当任务不需要外部事实或操作时，仅模型功能可以是正确的设计。只有当任务需要这些能力时，添加检索、工具、记忆或代理循环才是合理的。",{},{"id":590,"data":591,"type":42,"tunes":593},"h-implementation",{"text":592,"level":240},"实现证据：Aaasaasa AI Client",{},{"id":595,"data":596,"type":226,"tunes":599},"implementation-scope",{"body":597,"title":598,"variant":233},"以下部分描述了我根据截至 \u003Cstrong>2026 年 7 月 26 日\u003C\u002Fstrong> 的 Aaasaasa AI Client 代码库和架构文档构建并审查的实现。它是这些边界有用性的证据，而不是声称某个实现是通用标准或商业部署的企业产品。","主要实现证据",{},{"id":601,"data":602,"type":218,"tunes":604},"p-impl-1",{"text":603},"Aaasaasa AI Client 是一个使用 Nuxt 4、Electron 和 TypeScript 构建的本地优先桌面 AI 工作区。其 AI Hub 有意将代理\u002F客户端、提供商、模型、运行时位置、权限和 Web 客户端分开，而不是将它们视为一个“AI”设置。",{},{"id":606,"data":607,"type":218,"tunes":609},"p-impl-2",{"text":608},"这种分离创造了具体的行为。Direct Chat 可以与模型对话，而无需文件系统或 shell 工具。Codex 代理可以使用选定的工作区和权限配置文件。Ollama 可以提供直接的本地推理，而 LM Studio 和可配置的 OpenAI 兼容端点代表其他提供商路径。本地运行的 Codex 进程仍然可以使用云模型，因此 UI 和架构并不将本地运行时等同于本地推理。",{},{"id":611,"data":612,"type":218,"tunes":614},"p-impl-3",{"text":613},"该实现还包含 Qdrant\u002F向量支持、文档提取功能和经过身份验证的目录 MCP 代理。这些组件说明了另一个边界：检索基础设施和工具访问可以存在于同一产品中，而不会成为模型本身的属性。",{},{"id":616,"data":617,"type":347,"tunes":642},"impl-map",{"content":618,"stretched":43,"withHeadings":14},[619,622,624,627,630,633,636,639],[620,621],"A01 概念","Aaasaasa AI Client 实现证据",[323,623],"特定于提供商的模型标识符与提供商和运行时分开选择。",[625,626],"提供商","Ollama、LM Studio、OpenAI 兼容服务和其他提供商路径被分别表示。",[628,629],"运行时","本地或远程代理\u002F运行时位置独立于模型进行跟踪。",[631,632],"工具\u002F访问","Direct Chat 没有文件系统或 shell 工具；受控的目录访问被单独代理。",[634,635],"权限","工作区权限配置文件是应用\u002F会话策略，而不是模型能力。",[637,638],"检索基础设施","向量支持和文档提取作为数据\u002F检索能力存在，而不是模型功能。",[640,641],"应用","Electron\u002FNuxt 产品协调 UI、凭据、提供商、运行时发现、权限、工具和模型交互。",{},{"id":644,"data":645,"type":226,"tunes":648},"impl-lesson",{"body":646,"title":647,"variant":263},"一旦 \u003Cstrong>模型、提供商、运行时、权限、工具、数据和客户端\u003C\u002Fstrong> 不再被表示为一个配置选择，架构就变得更容易理解。这种区分是操作性的：它决定了什么可以在本地运行、什么可以访问文件、什么可以调用付费云推理，以及哪一层拥有授权。","实现教训",{},{"id":650,"data":651,"type":42,"tunes":653},"h-errors",{"text":652,"level":240},"常见的类别错误",{},{"id":655,"data":656,"type":347,"tunes":682},"errors-table",{"content":657,"stretched":43,"withHeadings":14},[658,661,664,667,670,673,676,679],[659,660],"类别错误","实际发生的情况",[662,663],"“AI 知道我们的文档。”","应用程序或检索层将选定的文档内容提供给模型。",[665,666],"“RAG 就是我们的向量数据库。”","向量数据库可以是检索管道使用的一个索引或存储；RAG 是检索加生成的模式。",[668,669],"“模型调用了我们的 CRM。”","模型生成了工具请求；运行时\u002F应用程序授权并执行了外部调用。",[671,672],"“它是本地 AI，因为桌面代理在本地运行。”","运行时位置和推理位置是分开的。本地运行时仍然可以调用远程模型。",[674,675],"“模型有权限编辑文件。”","应用程序\u002F运行时在权限策略下授予工具能力；权限不是模型的内在属性。",[677,678],"“更多上下文意味着更多知识。”","上下文是为一次推理提供的有限输入。更大的上下文可能包含更多噪音、冲突或过时信息。",[680,681],"“聊天机器人就是 AI 架构。”","聊天界面只是一种接口。系统还可以包括身份、状态、检索、工具、运行时、验证、持久化和可观测性。",{},{"id":684,"data":685,"type":42,"tunes":687},"h-failures",{"text":686,"level":240},"边界崩溃时的故障模式",{},{"id":689,"data":690,"type":218,"tunes":692},"p-failure-intro",{"text":691},"边界错误不仅仅是术语问题。它们会造成不同的生产故障，需要不同的修复方法。",{},{"id":694,"data":695,"type":358,"tunes":732},"failure-comparison",{"rows":696,"title":721,"layout":347,"columns":722},[697,701,705,709,713,717],{"id":698,"label":699,"values":700},"stale","答案过时",[325,325,325],{"id":702,"label":703,"values":704},"missing","缺少公司事实",[325,325,325],{"id":706,"label":707,"values":708},"unsafe","不安全的副作用",[325,325,325],{"id":710,"label":711,"values":712},"noise","提供了大量文本但答案混乱",[325,325,325],{"id":714,"label":715,"values":716},"route","意外的云使用",[325,325,325],{"id":718,"label":719,"values":720},"loop","代理停滞或重复",[325,325,325],"在更换模型之前先诊断故障层",[723,726,729],{"id":724,"label":725},"symptom","症状",{"id":727,"label":728},"likely","可能的边界问题",{"id":730,"label":731},"fix","首要架构检查",{},{"id":734,"data":735,"type":42,"tunes":737},"h-version",{"text":736,"level":240},"什么是稳定的，什么对版本敏感？",{},{"id":739,"data":740,"type":218,"tunes":742},"p-version-1",{"text":741},"本文中的架构区分有意保持供应商中立。下面的当前示例是实施事实，在 API 演变时应重新检查。",{},{"id":744,"data":745,"type":347,"tunes":775},"version-table",{"content":746,"stretched":43,"withHeadings":14},[747,751,755,759,763,767,771],[748,749,750],"领域","稳定的架构理念","2026 年 10 月 8 日验证的当前示例",[752,753,754],"AI 模型与系统","模型是更广泛系统中的一个组件","NIST 当前的术语表分别定义了 AI 模型和 AI 系统。",[756,757,758],"RAG","生成可以以检索到的外部信息为条件","Lewis 等人 2020 年的表述仍然是基础参考；生产检索方法现在已远远超出单一密集索引设计。",[760,761,762],"托管检索","检索可以作为托管工具暴露","OpenAI File Search 目前是一个 Responses API 工具，使用语义和关键词检索来搜索上传文件的知识库。",[764,765,766],"函数\u002F工具调用","模型可以请求应用程序定义的外部能力","OpenAI 目前将函数调用记录为与外部系统、数据和操作的接口。",[768,769,770],"上下文工程","模型行为取决于为当前推理提供的有限信息","Anthropic 当前的工程指南将上下文定义为从 LLM 采样时包含的 token 集合，并专注于策划该集合。",[772,773,774],"供应商 API","SDK、工具名称、端点形态和支持的功能会变化","即使责任边界保持稳定，也要将供应商文档视为对版本敏感的。",{},{"id":777,"data":778,"type":218,"tunes":780},"p-version-2",{"text":779},"因此，一篇权威文章应同时保留两个层面：用于架构的稳定概念，以及用于当前实现的带日期证据。将两者混为一谈会使文章不必要地快速过时。",{},{"id":782,"data":783,"type":42,"tunes":785},"h-test",{"text":784,"level":240},"AI 组件边界测试",{},{"id":787,"data":788,"type":218,"tunes":790},"p-test-1",{"text":789},"在评估 AI 功能时，请按顺序提出以下问题。答案揭示了系统实际拥有哪些组件，以及哪些责任仍然是隐含的。",{},{"id":792,"data":793,"type":305,"tunes":817},"boundary-test",{"steps":794,"title":816,"orientation":304},[795,798,801,804,807,810,813],{"label":796,"description":797},"1. 什么生成输出？","确定确切的模型及其提供的模态或结构化输出。",{"label":799,"description":800},"2. 模型之外哪些事实是权威的？","确定文档、数据库、API、当前状态和其他事实来源。",{"label":802,"description":803},"3. 如何选择相关信息？","区分直接查找、搜索、检索、排序和上下文构建。",{"label":805,"description":806},"4. 什么可能导致真实的副作用？","列出工具和外部操作，然后确定谁验证和授权它们。",{"label":808,"description":809},"5. 什么作为上下文到达模型？","明确指令、证据、状态、历史、记忆和工具定义。",{"label":811,"description":812},"6. 谁拥有循环？","确定管理调用、事件、重试、工具循环和会话的运行时或框架。",{"label":814,"description":815},"7. 什么仍然是应用程序的责任？","明确身份、权限、领域状态、验证、持久化、可观测性和用户体验。","生产设计的七个问题",{},{"id":819,"data":820,"type":42,"tunes":822},"h-not",{"text":821,"level":240},"生成式 AI 不是什么",{},{"id":824,"data":825,"type":218,"tunes":827},"p-not-1",{"text":826},"生成式 AI 不是 LLM 的同义词，尽管 LLM 是生成式模型的主要类别。它也不是 RAG、向量数据库、代理、工具协议、聊天机器人界面或应用程序的同义词。",{},{"id":829,"data":830,"type":218,"tunes":832},"p-not-2",{"text":831},"这些概念可以相互关联，但每个概念回答不同的架构问题。LLM 问的是语言输出如何产生。检索问的是外部证据来自哪里。工具问的是外部能力如何暴露。上下文问的是模型能看到什么。运行时问的是执行如何协调。应用程序问的是能力如何成为受控产品。",{},{"id":834,"data":835,"type":226,"tunes":838},"remember",{"body":836,"title":837,"variant":263},"\u003Cstrong>模型 = 生成。\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>检索 = 寻找证据。\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>工具 = 在模型之外读取或行动。\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>上下文 = 模型现在看到的内容。\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>运行时 = 协调执行。\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>应用程序 = 拥有产品、状态、规则和权限。\u003C\u002Fstrong>","如果你只记住一个模型",{},{"id":840,"data":841,"type":42,"tunes":843},"h-next",{"text":842,"level":240},"知识图谱中接下来去哪里",{},{"id":845,"data":846,"type":218,"tunes":848},"p-next-1",{"text":847},"一旦这些边界清晰，更深入的主题就更容易定位。RAG 属于检索和上下文构建。检索触发器决定何时需要外部证据。代理记忆关注跨时间持续存在的内容。工具调用和 MCP 属于能力访问。代理框架属于运行时编排。RBAC、租户隔离和领域授权属于应用程序和平台安全边界。",{},{"id":850,"data":851,"type":409,"tunes":856},"ref-data",{"url":852,"title":853,"excerpt":854,"ctaLabel":855},"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fwhere-does-an-llm-get-its-data-rag-data-sources-in-python","LLM 从哪里获取数据？Python 中的 RAG 数据源","一个实用的续篇，展示文件、SQL、API、全文搜索、嵌入和上下文组装如何将外部数据连接到 LLM。","在代码中查看数据路径",{},{"id":858,"data":859,"type":409,"tunes":864},"ref-trigger",{"url":860,"title":861,"excerpt":862,"ctaLabel":863},"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fwhen-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger","AI 何时应停止信任自身知识？——检索触发机制","一个决策模型，用于判断 AI 系统何时应停止仅依赖模型知识并获取外部证据。","阅读检索决策模型",{},{"id":866,"data":867,"type":42,"tunes":869},"h-limit",{"text":868,"level":240},"局限性",{},{"id":871,"data":872,"type":218,"tunes":874},"p-limit-1",{"text":873},"六层模型是一张责任地图，而不是要求每个产品都部署六个独立服务。小型应用可以在一个进程内实现上下文构建、检索和编排。托管平台可以将多项职责捆绑在一个 API 之后。物理部署可以合并，而语义归属仍然保持独立。",{},{"id":876,"data":877,"type":218,"tunes":879},"p-limit-2",{"text":878},"不同供应商和研究中的术语也各不相同。“智能体”、“运行时”、“记忆”、“工具”、“连接器”和“上下文”可能有不同的定义。这里的定义旨在明确操作归属和故障诊断，而不是声称每个框架都使用相同的词汇。",{},{"id":881,"data":882,"type":218,"tunes":884},"p-limit-3",{"text":883},"Aaasaasa AI Client 部分记录了一种实现模式。它表明明确的边界是可行的，但并不能证明相同的组件布局对每个 AI 产品都是最优的。",{},{"id":886,"data":887,"type":42,"tunes":889},"h-change",{"text":888,"level":240},"什么会改变这个答案？",{},{"id":891,"data":892,"type":218,"tunes":894},"p-change-1",{"text":893},"如果模型架构本身开始将权威外部状态、权限、持久事务副作用和可验证的源访问作为内在属性，而不是由周围系统提供的能力，那么责任地图将需要修订。当前的生产架构并未使这成为一个安全的普遍假设。",{},{"id":896,"data":897,"type":218,"tunes":899},"p-change-2",{"text":898},"个别实现示例会更快地发生变化。托管检索工具、智能体 API、MCP 集成、上下文管理功能和提供商能力都在快速演进。这些细节应更新，而不应破坏生成、证据、能力访问、上下文、执行和应用控制之间的基本区别。",{},{"id":901,"data":902,"type":42,"tunes":904},"h-conclusion",{"text":903,"level":240},"结论",{},{"id":906,"data":907,"type":218,"tunes":909},"p-conclusion-1",{"text":908},"一旦“AI”不再被视为一个黑箱，生成式 AI 的设计就会变得更容易。模型是生成组件，而不是完整产品。检索提供外部证据。工具暴露能力。上下文将选定的信息带入当前推理。运行时协调执行。应用程序拥有权威的产品边界。",{},{"id":911,"data":912,"type":218,"tunes":914},"p-conclusion-2",{"text":913},"这种分离不仅仅用于解释。它告诉工程师过时事实的来源、授权应属于哪里、为什么本地运行时仍可使用云推理、为什么 RAG 不等于向量数据库、为什么工具调用需要验证，以及为什么更改模型无法修复所有系统故障。",{},{"id":916,"data":917,"type":218,"tunes":919},"p-conclusion-3",{"text":918},"因此，持久的架构问题不是“我们使用哪个 AI 模型？”而是：每个组件拥有哪些职责，哪些证据跨越每个边界，以及哪一层被允许改变真实状态？",{},{"id":921,"data":922,"type":42,"tunes":924},"h-faq",{"text":923,"level":240},"常见问题",{},{"id":926,"data":927,"type":926,"tunes":958},"faq",{"items":928,"title":957},[929,933,937,941,945,949,953],{"id":930,"answer":931,"question":932},"faq1","不是。LLM 是一种生成模型。生成式 AI 还包括其他模态，生产级生成式 AI 系统可以在模型周围包含检索、工具、运行时逻辑、应用状态、权限、持久化和用户界面。","生成式 AI 和 LLM 是一回事吗？",{"id":934,"answer":935,"question":936},"faq2","通常不是。RAG 是一种应用\u002F系统模式，它检索外部信息并向模型提供选定的证据。一些平台将检索与模型 API 紧密打包，但职责仍然不同。","RAG 是模型的一部分吗？",{"id":938,"answer":939,"question":940},"faq3","不需要。RAG 可以使用向量搜索、词法搜索、混合检索、SQL、API、知识图谱或其他方法。其定义属性是为生成而检索外部信息，而不是某一种存储技术。","RAG 需要向量数据库吗？",{"id":942,"answer":943,"question":944},"faq4","不是。工具是一种外部能力。它的定义可以在上下文中表示，其结果稍后也可能进入上下文，但实际能力在模型外部执行。","工具和上下文是一回事吗？",{"id":946,"answer":947,"question":948},"faq5","不是。运行时位置和推理位置是分开的。本地桌面应用或智能体可以调用远程模型，而远程应用可以调用内部托管的模型。","在本地运行 AI 客户端意味着模型是本地吗？",{"id":950,"answer":951,"question":952},"faq6","应用程序或运行时安全边界应强制执行授权。模型可以请求操作，但模型意图绝不应被视为足够的执行权限。","谁应该为 AI 工具强制执行权限？",{"id":954,"answer":955,"question":956},"faq7","权威的易失状态通常应保留在拥有它的应用程序或领域系统中。AI 可以在需要时通过受控上下文或工具访问接收相关状态。","当前应用状态应属于哪里？","生成式 AI 系统边界",{},{"id":960,"data":961,"type":42,"tunes":963},"h-glossary",{"text":962,"level":240},"术语表",{},{"id":965,"data":966,"type":965,"tunes":990},"glossary",{"title":967,"entries":968},"核心术语",[969,973,975,977,979,981,985,987],{"term":970,"anchor":971,"definition":972},"生成模型","generative-model","一种 AI 模型，旨在生成衍生的合成内容，如文本、图像、音频、视频、代码或结构化输出。",{"term":328,"anchor":327,"definition":974},"从外部源或存储中为当前任务选择相关信息的过程。",{"term":756,"anchor":562,"definition":976},"检索增强生成：一种模式，将检索到的外部信息提供给生成模型以改进当前输出。",{"term":332,"anchor":566,"definition":978},"暴露给 AI 运行时用于读取数据、计算、搜索或执行外部操作的能力。",{"term":336,"anchor":335,"definition":980},"模型在特定推理步骤中可用的信息。",{"term":982,"anchor":983,"definition":984},"运行时 \u002F 编排器","runtime-orchestrator","协调模型调用、工具调用、任务循环、会话、重试、事件或执行环境的软件层。",{"term":344,"anchor":343,"definition":986},"拥有用户交互、权威状态、权限、验证、持久化和业务行为的产品和领域层。",{"term":625,"anchor":988,"definition":989},"provider","暴露对一个或多个模型访问权限的服务或运行时；提供商身份和模型身份是分开的关注点。",{},{"id":992,"data":993,"type":42,"tunes":995},"h-sources",{"text":994,"level":240},"主要来源和实现证据",{},{"id":997,"data":998,"type":218,"tunes":1000},"p-sources-note",{"text":999},"以下稳定定义以标准\u002F研究为依据；快速变化的实现示例使用当前官方工程文档。Aaasaasa AI Client 是原始实现证据，并已对照其截至 2026 年 7 月 26 日的代码库\u002F文档状态进行核查。",{},{"id":1002,"data":1003,"type":1009,"tunes":1010},"src-nist-profile",{"link":1004,"meta":1005},"https:\u002F\u002Fnvlpubs.nist.gov\u002Fnistpubs\u002Fai\u002FNIST.AI.600-1.pdf",{"image":1006,"title":1007,"description":1008},{"url":325},"NIST AI 600-1 — 生成式人工智能概况","NIST 的生成式 AI 概况，包括生成式 AI 的定义，并明确区分模型级、系统级、应用级和用例级关注点。","linkTool",{},{"id":1012,"data":1013,"type":1009,"tunes":1019},"src-nist-model",{"link":1014,"meta":1015},"https:\u002F\u002Fcsrc.nist.gov\u002Fglossary\u002Fterm\u002Fartificial_intelligence_model",{"image":1016,"title":1017,"description":1018},{"url":325},"NIST — 人工智能模型","NIST 术语表中对 AI 模型的当前定义：信息系统的组成部分，使用 AI 技术从输入产生输出。",{},{"id":1021,"data":1022,"type":1009,"tunes":1028},"src-nist-system",{"link":1023,"meta":1024},"https:\u002F\u002Fcsrc.nist.gov\u002Fglossary\u002Fterm\u002Fartificial_intelligence_system",{"image":1025,"title":1026,"description":1027},{"url":325},"NIST — 人工智能系统","NIST 术语表中当前的 AI 系统定义，表明 AI 系统可以包括使用 AI 的数据系统、软件、硬件、应用程序、工具或实用程序。",{},{"id":1030,"data":1031,"type":1009,"tunes":1037},"src-rag-paper",{"link":1032,"meta":1033},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2005.11401",{"image":1034,"title":1035,"description":1036},{"url":325},"Lewis 等 — 面向知识密集型 NLP 任务的检索增强生成","2020 年提出 RAG 公式的论文，该公式将生成模型与检索到的非参数记忆相结合。",{},{"id":1039,"data":1040,"type":1009,"tunes":1046},"src-openai-file-search",{"link":1041,"meta":1042},"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Ftools-file-search",{"image":1043,"title":1044,"description":1045},{"url":325},"OpenAI — 文件搜索","Responses API 中托管文件检索的当前官方文档，使用上传文件知识库、语义搜索和关键词搜索。",{},{"id":1048,"data":1049,"type":1009,"tunes":1055},"src-openai-functions",{"link":1050,"meta":1051},"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Ffunction-calling",{"image":1052,"title":1053,"description":1054},{"url":325},"OpenAI — 函数调用","当前官方文档，将工具\u002F函数调用描述为模型与外部系统、数据和操作之间的接口。",{},{"id":1057,"data":1058,"type":1009,"tunes":1064},"src-anthropic-context",{"link":1059,"meta":1060},"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Feffective-context-engineering-for-ai-agents",{"image":1061,"title":1062,"description":1063},{"url":325},"Anthropic — 面向 AI 智能体的有效上下文工程","工程指南将上下文定义为 LLM 采样期间可用的 token 集合，并解释为什么上下文选择是一个有限资源问题。",{},"2.31","生成式AI不仅仅是一个模型。了解模型、检索、工具、上下文、运行时和应用程序如何在生产AI系统中协同工作。","\u002Fuploads\u002F2026\u002F10\u002Fgenerative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing-1791475411822-pp0dvz.webp","generative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing-1791475411822-pp0dvz","PUBLISHED","2026-10-08T12:00:00.000Z","2026-10-08T16:00:39.350Z","2026-10-08T16:10:51.437Z",{"en":1074,"de":1075,"sr":1076,"es":1077,"fr":1078,"it":1079,"ru":1080,"zh":1081},"\u002Fblog\u002Fgenerative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing","\u002Fde\u002Fblog\u002Fgenerative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing","\u002Fsr\u002Fblog\u002Fgenerative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing","\u002Fes\u002Fblog\u002Fgenerative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing","\u002Ffr\u002Fblog\u002Fgenerative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing","\u002Fit\u002Fblog\u002Fgenerative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing","\u002Fru\u002Fblog\u002Fgenerative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing","\u002Fzh\u002Fblog\u002Fgenerative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing",[1083,1087,1091,1095,1099],{"id":1084,"name":1085,"slug":1086},84,"策略与数据边界","policy-and-data",{"id":1088,"name":1089,"slug":1090},57,"数据边界","data-boundaries",{"id":1092,"name":1093,"slug":1094},80,"访问与身份","access-and-identity",{"id":1096,"name":1097,"slug":1098},68,"风险、控制与证据","risks-and-controls",{"id":1100,"name":1101,"slug":1102},54,"威胁模型","threat-model",{"id":1104,"login":1105,"email":1106,"displayName":1107},"20","rooth8233","aleksandar@stajic.de","Aleksandar Stajić",[1109,1812],{"lang":1110,"title":1111,"content":1112,"contentJson":1113,"excerpt":1811},"en","Generative AI Explained: Models, Retrieval, Tools and Applications Are Not the Same Thing","{\"time\":1791475413504,\"blocks\":[{\"id\":\"intro\",\"type\":\"paragraph\",\"data\":{\"text\":\"Generative AI is not one component. A production generative AI system usually combines a generative model with application code that supplies instructions and context, retrieves external knowledge when needed, exposes tools for reading or changing external systems, manages runtime state and permissions, and turns the result into a usable product. Treating the model, retrieval, tools, context, runtime, and application as the same thing hides the boundaries that determine freshness, security, reliability, cost, and control.\"},\"tunes\":{}},{\"id\":\"direct\",\"type\":\"callout\",\"data\":{\"variant\":\"info\",\"title\":\"Direct answer\",\"body\":\"\u003Cstrong>The model generates; retrieval finds external evidence; tools access data or perform actions; context is what the model can see for the current inference; the runtime coordinates execution; the application owns product rules, state, permissions, persistence, and user experience.\u003C\u002Fstrong> These layers can be packaged together by a vendor, but their responsibilities remain different.\"},\"tunes\":{}},{\"id\":\"scope-note\",\"type\":\"callout\",\"data\":{\"variant\":\"note\",\"title\":\"Terminology and version note\",\"body\":\"This article defines durable architectural responsibilities rather than one vendor stack. Current implementation examples were re-checked on \u003Cstrong>8 October 2026\u003C\u002Fstrong>. Vendor APIs and product names can change; the responsibility boundaries are more stable than any individual SDK or endpoint.\"},\"tunes\":{}},{\"id\":\"toc\",\"type\":\"tableOfContents\",\"data\":{\"title\":\"Contents\",\"minLevel\":2,\"maxLevel\":3},\"tunes\":{}},{\"id\":\"h-meaning\",\"type\":\"header\",\"data\":{\"text\":\"What does “generative AI” actually mean?\",\"level\":2},\"tunes\":{}},{\"id\":\"p-meaning-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"At the model level, generative AI refers to AI models that generate derived synthetic content such as text, images, audio, video, code, or other digital output. NIST AI 600-1 uses this model-oriented meaning and separately discusses risks at model, system, application, and use-case levels.\"},\"tunes\":{}},{\"id\":\"p-meaning-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"That distinction matters because an AI model is not the same thing as the complete AI system. NIST's current glossary defines an AI model as a component that produces outputs from inputs using computational, statistical, or machine-learning techniques, while an AI system can include software, hardware, applications, tools, or utilities that operate using AI.\"},\"tunes\":{}},{\"id\":\"model-system-rule\",\"type\":\"callout\",\"data\":{\"variant\":\"success\",\"title\":\"A useful boundary\",\"body\":\"\u003Cstrong>Generative model ≠ generative AI application.\u003C\u002Fstrong>\u003Cbr>A model is one computational component. A usable AI product is a system built around that component.\"},\"tunes\":{}},{\"id\":\"h-simple\",\"type\":\"header\",\"data\":{\"text\":\"The simplest useful model of a generative AI system\",\"level\":2},\"tunes\":{}},{\"id\":\"p-simple-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"For a first mental model, imagine a company assistant answering: “Can this customer receive a refund today?” A useful answer may require several different responsibilities. The language model can interpret the question and write the explanation, but the current order state may come from a database tool, the refund policy may come from document retrieval, permissions may be enforced by the application, and the final action may require a controlled API call.\"},\"tunes\":{}},{\"id\":\"simple-flow\",\"type\":\"processFlow\",\"data\":{\"title\":\"One common execution path\",\"orientation\":\"auto\",\"steps\":[{\"label\":\"1. User request\",\"description\":\"The application receives a natural-language question or task.\"},{\"label\":\"2. Application policy and state\",\"description\":\"Identity, tenant, permissions, current workflow state, and product rules define what the request is allowed to do.\"},{\"label\":\"3. Retrieval or direct data access\",\"description\":\"The system obtains external evidence or current facts when model knowledge is insufficient.\"},{\"label\":\"4. Context construction\",\"description\":\"Instructions, user input, selected evidence, relevant state, and tool definitions are assembled for the model.\"},{\"label\":\"5. Model inference\",\"description\":\"The generative model interprets the supplied context and produces text, structured output, or a tool request.\"},{\"label\":\"6. Tool execution when needed\",\"description\":\"The runtime or application validates and executes approved tool calls outside the model.\"},{\"label\":\"7. Observation and continuation\",\"description\":\"Tool results can return to the model as new context for another inference step.\"},{\"label\":\"8. Validation and product output\",\"description\":\"The application validates the result, records required state or audit data, and presents or executes the final outcome.\"}]},\"tunes\":{}},{\"id\":\"p-simple-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Real systems do not always follow this sequence exactly. Retrieval can happen before the first model call, tools can be selected during an agent loop, deterministic application logic can bypass the model entirely, and validation can occur at several stages. The point is to separate responsibilities, not to impose one universal workflow.\"},\"tunes\":{}},{\"id\":\"h-boundaries\",\"type\":\"header\",\"data\":{\"text\":\"The six boundaries that matter\",\"level\":2},\"tunes\":{}},{\"id\":\"boundary-comparison\",\"type\":\"comparison\",\"data\":{\"title\":\"Six responsibilities inside one AI product\",\"layout\":\"table\",\"columns\":[{\"id\":\"job\",\"label\":\"Primary job\"},{\"id\":\"input\",\"label\":\"Typical inputs\"},{\"id\":\"not\",\"label\":\"Not the same as\"}],\"rows\":[{\"id\":\"model\",\"label\":\"Model\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"retrieval\",\"label\":\"Retrieval\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"tools\",\"label\":\"Tools\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"context\",\"label\":\"Context\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"runtime\",\"label\":\"Runtime \u002F orchestrator\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"application\",\"label\":\"Application\",\"values\":[\"\",\"\",\"\"]}]},\"tunes\":{}},{\"id\":\"h-model\",\"type\":\"header\",\"data\":{\"text\":\"1. The model: generation is its core responsibility\",\"level\":2},\"tunes\":{}},{\"id\":\"p-model-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A generative model maps supplied inputs to generated outputs. For a language model, that can include natural-language text, structured JSON, code, classifications, summaries, plans, or tool-call arguments. Multimodal generative models can work with additional input and output types.\"},\"tunes\":{}},{\"id\":\"p-model-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The model can contain substantial learned knowledge in its parameters, but parameterized knowledge is not a live database. The model does not automatically know a document created five minutes ago, the current stock level, a private customer record, or the state of an application unless that information is supplied through the current input path.\"},\"tunes\":{}},{\"id\":\"p-model-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"This is why changing the model does not automatically solve stale knowledge, missing permissions, broken retrieval, incorrect state ownership, or unsafe tool execution. Those failures often belong to other layers.\"},\"tunes\":{}},{\"id\":\"h-retrieval\",\"type\":\"header\",\"data\":{\"text\":\"2. Retrieval: finding external evidence is a separate operation\",\"level\":2},\"tunes\":{}},{\"id\":\"p-retrieval-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Retrieval selects information from an external source before or during generation. The 2020 Retrieval-Augmented Generation work by Lewis et al. made the separation explicit by combining a parametric generative model with retrieved non-parametric memory. Modern production systems use many retrieval variants, but the architectural idea remains: useful evidence can be fetched at inference time instead of relying only on what the model learned during training.\"},\"tunes\":{}},{\"id\":\"p-retrieval-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Retrieval can use lexical search, embeddings, vector search, hybrid search, SQL, knowledge graphs, metadata filters, APIs, or other selection mechanisms. A vector database is therefore one possible retrieval component, not the definition of RAG.\"},\"tunes\":{}},{\"id\":\"retrieval-rule\",\"type\":\"callout\",\"data\":{\"variant\":\"warning\",\"title\":\"Relevance is not authority\",\"body\":\"A retrieved passage can be highly relevant and still be stale, unauthorized, from the wrong version, or insufficient to support a claim. Retrieval quality and evidence quality must be evaluated separately.\"},\"tunes\":{}},{\"id\":\"ref-rag\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works\",\"title\":\"What Is RAG? The Simplest Explanation of How It Works\",\"excerpt\":\"The canonical plain-English explanation of retrieval-augmented generation, including the separation between LLM, knowledge, state, memory and tools.\",\"ctaLabel\":\"Read the RAG foundation\"},\"tunes\":{}},{\"id\":\"h-tools\",\"type\":\"header\",\"data\":{\"text\":\"3. Tools: access and action are not model knowledge\",\"level\":2},\"tunes\":{}},{\"id\":\"p-tools-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A tool is an interface through which an AI runtime can request functionality outside the model. A tool can query a database, search the web, read a file, calculate a value, call an internal service, create a ticket, send a message, modify a record, or trigger another controlled operation.\"},\"tunes\":{}},{\"id\":\"p-tools-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"OpenAI's current function-calling documentation makes this boundary explicit: function calling lets models interface with external systems and access data or actions provided by the application. The model can propose or select a call, but the external system performs the real operation.\"},\"tunes\":{}},{\"id\":\"p-tools-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"Tool use therefore creates two separate questions: Can the model request this capability? and Will the application authorize and execute it? A production system should not confuse model intent with permission to cause a side effect.\"},\"tunes\":{}},{\"id\":\"tool-rule\",\"type\":\"callout\",\"data\":{\"variant\":\"success\",\"title\":\"Model intent is not execution authority\",\"body\":\"A model can emit a valid tool request and still be denied. Authorization, argument validation, rate limits, transaction rules, audit requirements, and rollback belong outside the model.\"},\"tunes\":{}},{\"id\":\"h-context\",\"type\":\"header\",\"data\":{\"text\":\"4. Context: what the model can see right now\",\"level\":2},\"tunes\":{}},{\"id\":\"p-context-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Context is the information available to the model for a particular inference step. Anthropic's context-engineering guidance describes context as the set of tokens included when sampling from an LLM. In practice, that set can contain system instructions, user messages, conversation history, retrieved evidence, tool definitions, tool results, memory summaries, and selected application state.\"},\"tunes\":{}},{\"id\":\"p-context-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Context is therefore neither the complete knowledge base nor long-term memory. A company may store ten million documents while only a handful of passages enter one model call. A runtime may persist a year of conversation history while exposing only the pieces needed for the current task.\"},\"tunes\":{}},{\"id\":\"p-context-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The context window also creates an engineering constraint. Adding more text does not guarantee a better answer; irrelevant, stale, contradictory, or low-authority information can dilute the evidence that actually matters.\"},\"tunes\":{}},{\"id\":\"h-runtime\",\"type\":\"header\",\"data\":{\"text\":\"5. Runtime and orchestration: coordinating the loop\",\"level\":2},\"tunes\":{}},{\"id\":\"p-runtime-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The runtime or orchestration layer coordinates how the model participates in a task. Depending on the architecture, it can manage sessions, model requests, tool discovery, tool-call loops, retries, handoffs, streaming events, timeouts, checkpoints, compaction, or execution environments.\"},\"tunes\":{}},{\"id\":\"p-runtime-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Some runtimes are thin application code around a model API. Others are full agent harnesses. A managed vendor runtime can own part of the loop while the application still owns domain truth, authorization, business side effects, and product lifecycle.\"},\"tunes\":{}},{\"id\":\"p-runtime-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"This boundary is important because where the runtime runs and where inference runs are separate decisions. A locally running client or agent process can still call a remote model, while a remote application can call a model hosted on infrastructure under the organization's control.\"},\"tunes\":{}},{\"id\":\"h-application\",\"type\":\"header\",\"data\":{\"text\":\"6. The application: where AI becomes a product\",\"level\":2},\"tunes\":{}},{\"id\":\"p-app-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The application is the product boundary around the AI components. It owns the user experience, domain model, current state, identity, tenant scope, permissions, persistence, service integrations, validation, observability, billing or quota logic where relevant, and the rules that determine what the AI is allowed to see or do.\"},\"tunes\":{}},{\"id\":\"p-app-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"This is the layer that turns “a model can produce useful output” into “a system can deliver a reliable capability.” The same model can participate in a private research assistant, a support workflow, a code agent, or a commerce application because the surrounding application changes the data, tools, policies, state, and execution contract.\"},\"tunes\":{}},{\"id\":\"app-rule\",\"type\":\"callout\",\"data\":{\"variant\":\"info\",\"title\":\"The model is replaceable; the product boundary is not\",\"body\":\"Provider and model substitution can be an architectural goal. The application's authoritative state, permissions, domain rules, audit trail, and user contract cannot simply be delegated to whichever model is currently selected.\"},\"tunes\":{}},{\"id\":\"h-work-together\",\"type\":\"header\",\"data\":{\"text\":\"How the parts work together in a real request\",\"level\":2},\"tunes\":{}},{\"id\":\"p-together-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Consider a support assistant asked: “Refund order 4711 if it is still eligible, and explain why.” The request combines knowledge, current state, authorization, reasoning, and a side effect.\"},\"tunes\":{}},{\"id\":\"support-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Need\",\"Correct layer\",\"Why\"],[\"Refund policy\",\"Retrieval\",\"The system must find the current applicable policy and preserve its provenance.\"],[\"Order 4711 status\",\"Direct data\u002Ftool access\",\"The current order record is volatile authoritative state, not something to guess from model knowledge.\"],[\"User's authority to refund\",\"Application \u002F authorization\",\"Permissions must be enforced independently of what the model asks for.\"],[\"Interpret policy against order facts\",\"Model + context\",\"The model can reason over the policy evidence and current order state supplied to it.\"],[\"Execute refund\",\"Tool + application transaction rules\",\"A controlled external operation changes real state.\"],[\"Explain outcome\",\"Model\",\"The model can generate the user-facing explanation from validated results.\"],[\"Audit what happened\",\"Application \u002F runtime\",\"The system records evidence, calls, decisions, side effects, and errors as required.\"]]},\"tunes\":{}},{\"id\":\"p-together-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"If the assistant only has the language model, it can discuss refunds but cannot safely know whether order 4711 is currently eligible or perform the transaction. If it only has retrieval, it may find the policy but still lack live order state. If it has tools without application authorization, it may become capable but unsafe. Reliability comes from composing the layers with explicit ownership.\"},\"tunes\":{}},{\"id\":\"h-configs\",\"type\":\"header\",\"data\":{\"text\":\"Different AI products use different combinations\",\"level\":2},\"tunes\":{}},{\"id\":\"config-comparison\",\"type\":\"comparison\",\"data\":{\"title\":\"The presence of a model does not define the whole architecture\",\"layout\":\"table\",\"columns\":[{\"id\":\"retrieval\",\"label\":\"Retrieval\"},{\"id\":\"tools\",\"label\":\"Tools\"},{\"id\":\"state\",\"label\":\"Authoritative state\"},{\"id\":\"result\",\"label\":\"Typical capability\"}],\"rows\":[{\"id\":\"bare\",\"label\":\"Model-only assistant\",\"values\":[\"\",\"\",\"\",\"\"]},{\"id\":\"rag\",\"label\":\"Retrieval-grounded assistant\",\"values\":[\"\",\"\",\"\",\"\"]},{\"id\":\"tool\",\"label\":\"Tool-using assistant\",\"values\":[\"\",\"\",\"\",\"\"]},{\"id\":\"agent\",\"label\":\"Agentic application\",\"values\":[\"\",\"\",\"\",\"\"]}]},\"tunes\":{}},{\"id\":\"p-configs-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"These are architecture patterns, not maturity rankings. A model-only feature can be the correct design when the task needs no external facts or actions. Adding retrieval, tools, memory, or an agent loop is justified only when the task requires those capabilities.\"},\"tunes\":{}},{\"id\":\"h-implementation\",\"type\":\"header\",\"data\":{\"text\":\"Implementation evidence: Aaasaasa AI Client\",\"level\":2},\"tunes\":{}},{\"id\":\"implementation-scope\",\"type\":\"callout\",\"data\":{\"variant\":\"note\",\"title\":\"Primary implementation evidence\",\"body\":\"The following section describes an implementation I built and reviewed against the Aaasaasa AI Client codebase and architecture documentation as of \u003Cstrong>26 July 2026\u003C\u002Fstrong>. It is evidence for the usefulness of these boundaries, not a claim that one implementation is a universal standard or a commercially deployed enterprise product.\"},\"tunes\":{}},{\"id\":\"p-impl-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Aaasaasa AI Client is a local-first desktop AI workspace built with Nuxt 4, Electron and TypeScript. Its AI Hub deliberately separates agent\u002Fclient, provider, model, runtime location, permissions, and web client instead of treating them as one “AI” setting.\"},\"tunes\":{}},{\"id\":\"p-impl-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"That separation creates concrete behavior. Direct Chat can talk to models without filesystem or shell tools. A Codex agent can use a selected workspace and permission profile. Ollama can provide direct local inference, while LM Studio and configurable OpenAI-compatible endpoints represent other provider paths. A locally running Codex process can still use a cloud model, so the UI and architecture do not equate local runtime with local inference.\"},\"tunes\":{}},{\"id\":\"p-impl-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The implementation also contains Qdrant\u002Fvector support, document-extraction capabilities and an authenticated directory MCP broker. Those components illustrate another boundary: retrieval infrastructure and tool access can live in the same product without becoming properties of the model itself.\"},\"tunes\":{}},{\"id\":\"impl-map\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"A01 concept\",\"Aaasaasa AI Client implementation evidence\"],[\"Model\",\"A provider-specific model identifier is selected separately from provider and runtime.\"],[\"Provider\",\"Ollama, LM Studio, OpenAI-compatible services and other provider paths are represented separately.\"],[\"Runtime\",\"Local or remote agent\u002Fruntime location is tracked independently of the model.\"],[\"Tools \u002F access\",\"Direct Chat has no filesystem or shell tools; controlled directory access is brokered separately.\"],[\"Permissions\",\"Workspace permission profiles are application\u002Fsession policy, not model capability.\"],[\"Retrieval infrastructure\",\"Vector support and document extraction exist as data\u002Fretrieval capabilities rather than model features.\"],[\"Application\",\"The Electron\u002FNuxt product coordinates UI, credentials, providers, runtime discovery, permissions, tools and model interaction.\"]]},\"tunes\":{}},{\"id\":\"impl-lesson\",\"type\":\"callout\",\"data\":{\"variant\":\"success\",\"title\":\"Implementation lesson\",\"body\":\"The architecture became easier to reason about once \u003Cstrong>model, provider, runtime, permissions, tools, data and client\u003C\u002Fstrong> stopped being represented as one configuration choice. The distinction is operational: it determines what can run locally, what can access files, what may call paid cloud inference, and which layer owns authorization.\"},\"tunes\":{}},{\"id\":\"h-errors\",\"type\":\"header\",\"data\":{\"text\":\"Common category errors\",\"level\":2},\"tunes\":{}},{\"id\":\"errors-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Category error\",\"What is actually happening\"],[\"“The AI knows our documents.”\",\"The application or retrieval layer makes selected document content available to the model.\"],[\"“RAG is our vector database.”\",\"The vector database can be one index or store used by a retrieval pipeline; RAG is the retrieval-plus-generation pattern.\"],[\"“The model called our CRM.”\",\"The model produced a tool request; the runtime\u002Fapplication authorized and executed the external call.\"],[\"“It is local AI because the desktop agent runs locally.”\",\"Runtime location and inference location are separate. A local runtime can still invoke a remote model.\"],[\"“The model has permission to edit files.”\",\"The application\u002Fruntime grants a tool capability under a permission policy; permission is not an intrinsic model property.\"],[\"“More context means more knowledge.”\",\"Context is the finite input made available for one inference. Larger context can contain more noise, conflict or stale information.\"],[\"“The chatbot is the AI architecture.”\",\"The chat UI is one interface. The system can also include identity, state, retrieval, tools, runtime, validation, persistence and observability.\"]]},\"tunes\":{}},{\"id\":\"h-failures\",\"type\":\"header\",\"data\":{\"text\":\"Failure modes when the boundaries collapse\",\"level\":2},\"tunes\":{}},{\"id\":\"p-failure-intro\",\"type\":\"paragraph\",\"data\":{\"text\":\"Boundary mistakes are not merely terminology problems. They create distinct production failures that require different fixes.\"},\"tunes\":{}},{\"id\":\"failure-comparison\",\"type\":\"comparison\",\"data\":{\"title\":\"Diagnose the failing layer before replacing the model\",\"layout\":\"table\",\"columns\":[{\"id\":\"symptom\",\"label\":\"Symptom\"},{\"id\":\"likely\",\"label\":\"Likely boundary problem\"},{\"id\":\"fix\",\"label\":\"First architectural check\"}],\"rows\":[{\"id\":\"stale\",\"label\":\"Stale answer\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"missing\",\"label\":\"Missing company fact\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"unsafe\",\"label\":\"Unsafe side effect\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"noise\",\"label\":\"Confused answer with lots of supplied text\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"route\",\"label\":\"Unexpected cloud use\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"loop\",\"label\":\"Agent stalls or repeats\",\"values\":[\"\",\"\",\"\"]}]},\"tunes\":{}},{\"id\":\"h-version\",\"type\":\"header\",\"data\":{\"text\":\"What is stable and what is version-sensitive?\",\"level\":2},\"tunes\":{}},{\"id\":\"p-version-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The architectural distinctions in this article are intentionally vendor-neutral. The current examples below are implementation facts that should be re-checked when APIs evolve.\"},\"tunes\":{}},{\"id\":\"version-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Area\",\"Stable architectural idea\",\"Verified current example on 8 Oct 2026\"],[\"AI model vs system\",\"A model is a component inside a broader system\",\"NIST's current glossary separately defines AI model and AI system.\"],[\"RAG\",\"Generation can be conditioned on retrieved external information\",\"The Lewis et al. 2020 formulation remains the foundational reference; production retrieval methods now extend far beyond one dense index design.\"],[\"Hosted retrieval\",\"Retrieval can be exposed as a managed tool\",\"OpenAI File Search is currently a Responses API tool that searches uploaded-file knowledge bases using semantic and keyword retrieval.\"],[\"Function\u002Ftool calling\",\"A model can request application-defined external capabilities\",\"OpenAI currently documents function calling as an interface to external systems, data and actions.\"],[\"Context engineering\",\"Model behavior depends on the finite information supplied for the current inference\",\"Anthropic's current engineering guidance defines context as the token set included when sampling from the LLM and focuses on curating that set.\"],[\"Vendor APIs\",\"SDKs, tool names, endpoint shapes and supported features change\",\"Treat vendor documentation as version-sensitive even when the responsibility boundary remains stable.\"]]},\"tunes\":{}},{\"id\":\"p-version-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"A source-of-truth article should therefore preserve both levels: stable concepts for architecture, and dated evidence for current implementations. Mixing the two makes an article age unnecessarily fast.\"},\"tunes\":{}},{\"id\":\"h-test\",\"type\":\"header\",\"data\":{\"text\":\"The AI component-boundary test\",\"level\":2},\"tunes\":{}},{\"id\":\"p-test-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"When evaluating an AI feature, ask the following questions in order. The answers reveal which components the system actually has and which responsibilities are still implicit.\"},\"tunes\":{}},{\"id\":\"boundary-test\",\"type\":\"processFlow\",\"data\":{\"title\":\"Seven questions for a production design\",\"orientation\":\"auto\",\"steps\":[{\"label\":\"1. What generates the output?\",\"description\":\"Identify the exact model and the modalities or structured outputs it provides.\"},{\"label\":\"2. What facts are authoritative outside the model?\",\"description\":\"Identify documents, databases, APIs, current state and other sources of truth.\"},{\"label\":\"3. How is relevant information selected?\",\"description\":\"Separate direct lookup, search, retrieval, ranking and context construction.\"},{\"label\":\"4. What can cause real side effects?\",\"description\":\"List tools and external actions, then identify who validates and authorizes them.\"},{\"label\":\"5. What reaches the model as context?\",\"description\":\"Make instructions, evidence, state, history, memory and tool definitions explicit.\"},{\"label\":\"6. Who owns the loop?\",\"description\":\"Identify the runtime or harness that manages calls, events, retries, tool loops and sessions.\"},{\"label\":\"7. What remains the application's responsibility?\",\"description\":\"Make identity, permissions, domain state, validation, persistence, observability and UX explicit.\"}]},\"tunes\":{}},{\"id\":\"h-not\",\"type\":\"header\",\"data\":{\"text\":\"What generative AI is not\",\"level\":2},\"tunes\":{}},{\"id\":\"p-not-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Generative AI is not synonymous with an LLM, even though LLMs are a major class of generative model. It is also not synonymous with RAG, a vector database, an agent, a tool protocol, a chatbot UI, or an application.\"},\"tunes\":{}},{\"id\":\"p-not-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Those concepts can be connected, but each answers a different architectural question. An LLM asks how language output is produced. Retrieval asks where external evidence comes from. Tools ask how external capabilities are exposed. Context asks what the model can see. Runtime asks how execution is coordinated. The application asks how the capability becomes a controlled product.\"},\"tunes\":{}},{\"id\":\"remember\",\"type\":\"callout\",\"data\":{\"variant\":\"success\",\"title\":\"If you remember only one model\",\"body\":\"\u003Cstrong>Model = generate.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Retrieval = find evidence.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Tools = read or act outside the model.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Context = what the model sees now.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Runtime = coordinate execution.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Application = own the product, state, rules and permissions.\u003C\u002Fstrong>\"},\"tunes\":{}},{\"id\":\"h-next\",\"type\":\"header\",\"data\":{\"text\":\"Where to go next in the knowledge graph\",\"level\":2},\"tunes\":{}},{\"id\":\"p-next-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Once these boundaries are clear, deeper topics become easier to place. RAG belongs in retrieval and context construction. Retrieval Trigger decides when external evidence is required. Agent memory concerns what persists across time. Tool calling and MCP belong to capability access. Agent harnesses belong to runtime orchestration. RBAC, tenant isolation and domain authorization belong to the application and platform security boundary.\"},\"tunes\":{}},{\"id\":\"ref-data\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fwhere-does-an-llm-get-its-data-rag-data-sources-in-python\",\"title\":\"Where Does an LLM Get Its Data? RAG Data Sources in Python\",\"excerpt\":\"A practical continuation showing how files, SQL, APIs, full-text search, embeddings and context assembly connect external data to an LLM.\",\"ctaLabel\":\"See the data path in code\"},\"tunes\":{}},{\"id\":\"ref-trigger\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fwhen-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger\",\"title\":\"When Should an AI Stop Trusting Its Own Knowledge? — The Retrieval Trigger\",\"excerpt\":\"A decision model for when an AI system should stop relying only on model knowledge and obtain external evidence.\",\"ctaLabel\":\"Read the retrieval decision model\"},\"tunes\":{}},{\"id\":\"h-limit\",\"type\":\"header\",\"data\":{\"text\":\"Limitations\",\"level\":2},\"tunes\":{}},{\"id\":\"p-limit-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The six-layer model is a responsibility map, not a requirement that every product deploy six separate services. A small application may implement context construction, retrieval and orchestration inside one process. A managed platform may bundle several responsibilities behind one API. Physical deployment can be combined while semantic ownership remains distinct.\"},\"tunes\":{}},{\"id\":\"p-limit-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Terminology also varies across vendors and research. “Agent,” “runtime,” “memory,” “tool,” “connector,” and “context” can be defined differently. The definitions here are chosen to make operational ownership and failure diagnosis explicit rather than to claim that every framework uses identical vocabulary.\"},\"tunes\":{}},{\"id\":\"p-limit-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The Aaasaasa AI Client section documents one implementation pattern. It demonstrates that explicit boundaries are practical, but it does not prove that the same component layout is optimal for every AI product.\"},\"tunes\":{}},{\"id\":\"h-change\",\"type\":\"header\",\"data\":{\"text\":\"What would change this answer?\",\"level\":2},\"tunes\":{}},{\"id\":\"p-change-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The responsibility map would need revision if model architectures themselves began to own authoritative external state, permissions, durable transactional side effects, and verifiable source access as intrinsic properties rather than capabilities supplied by a surrounding system. Current production architectures do not make that a safe general assumption.\"},\"tunes\":{}},{\"id\":\"p-change-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Individual implementation examples will change much sooner. Hosted retrieval tools, agent APIs, MCP integrations, context-management features and provider capabilities evolve quickly. Those details should be updated without collapsing the underlying distinctions between generation, evidence, capability access, context, execution and application control.\"},\"tunes\":{}},{\"id\":\"h-conclusion\",\"type\":\"header\",\"data\":{\"text\":\"Conclusion\",\"level\":2},\"tunes\":{}},{\"id\":\"p-conclusion-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Generative AI becomes easier to design once “the AI” stops being treated as one black box. The model is the generative component, not the complete product. Retrieval provides external evidence. Tools expose capabilities. Context carries selected information into the current inference. The runtime coordinates execution. The application owns the authoritative product boundary.\"},\"tunes\":{}},{\"id\":\"p-conclusion-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"That separation is useful for more than explanation. It tells engineers where stale facts originate, where authorization belongs, why a local runtime can still use cloud inference, why RAG does not equal a vector database, why tool calls require validation, and why changing the model cannot repair every system failure.\"},\"tunes\":{}},{\"id\":\"p-conclusion-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The durable architecture question is therefore not “Which AI model are we using?” It is: Which responsibility does each component own, what evidence crosses each boundary, and which layer is allowed to change real state?\"},\"tunes\":{}},{\"id\":\"h-faq\",\"type\":\"header\",\"data\":{\"text\":\"FAQ\",\"level\":2},\"tunes\":{}},{\"id\":\"faq\",\"type\":\"faq\",\"data\":{\"title\":\"Generative AI system boundaries\",\"items\":[{\"id\":\"faq1\",\"question\":\"Is generative AI the same as an LLM?\",\"answer\":\"No. An LLM is one type of generative model. Generative AI also includes other modalities, and a production generative AI system can include retrieval, tools, runtime logic, application state, permissions, persistence and user interfaces around the model.\"},{\"id\":\"faq2\",\"question\":\"Is RAG part of the model?\",\"answer\":\"Usually no. RAG is an application\u002Fsystem pattern that retrieves external information and supplies selected evidence to the model. Some platforms package retrieval tightly with model APIs, but the responsibility remains distinct.\"},{\"id\":\"faq3\",\"question\":\"Is a vector database required for RAG?\",\"answer\":\"No. RAG can use vector search, lexical search, hybrid retrieval, SQL, APIs, knowledge graphs or other methods. The defining property is retrieval of external information for generation, not one storage technology.\"},{\"id\":\"faq4\",\"question\":\"Are tools the same as context?\",\"answer\":\"No. A tool is an external capability. Its definition may be represented in context, and its result may later enter context, but the actual capability executes outside the model.\"},{\"id\":\"faq5\",\"question\":\"Does running an AI client locally mean the model is local?\",\"answer\":\"No. Runtime location and inference location are separate. A local desktop application or agent can call a remote model, while a remote application can call an internally hosted model.\"},{\"id\":\"faq6\",\"question\":\"Who should enforce permissions for AI tools?\",\"answer\":\"The application or runtime security boundary should enforce authorization. A model can request an operation, but model intent should never be treated as sufficient execution authority.\"},{\"id\":\"faq7\",\"question\":\"Where does current application state belong?\",\"answer\":\"Authoritative volatile state should normally remain in the application or domain system that owns it. The AI can receive the relevant state through controlled context or tool access when needed.\"}]},\"tunes\":{}},{\"id\":\"h-glossary\",\"type\":\"header\",\"data\":{\"text\":\"Glossary\",\"level\":2},\"tunes\":{}},{\"id\":\"glossary\",\"type\":\"glossary\",\"data\":{\"title\":\"Core terms\",\"entries\":[{\"term\":\"Generative model\",\"definition\":\"An AI model designed to generate derived synthetic content such as text, images, audio, video, code or structured output.\",\"anchor\":\"generative-model\"},{\"term\":\"Retrieval\",\"definition\":\"The process of selecting relevant information from an external source or store for the current task.\",\"anchor\":\"retrieval\"},{\"term\":\"RAG\",\"definition\":\"Retrieval-Augmented Generation: a pattern in which retrieved external information is supplied to a generative model to improve the current output.\",\"anchor\":\"rag\"},{\"term\":\"Tool\",\"definition\":\"A capability exposed to an AI runtime for reading data, calculating, searching, or performing an external action.\",\"anchor\":\"tool\"},{\"term\":\"Context\",\"definition\":\"The information available to the model for a particular inference step.\",\"anchor\":\"context\"},{\"term\":\"Runtime \u002F orchestrator\",\"definition\":\"The software layer that coordinates model calls, tool calls, task loops, sessions, retries, events or execution environments.\",\"anchor\":\"runtime-orchestrator\"},{\"term\":\"Application\",\"definition\":\"The product and domain layer that owns user interaction, authoritative state, permissions, validation, persistence and business behavior.\",\"anchor\":\"application\"},{\"term\":\"Provider\",\"definition\":\"The service or runtime that exposes access to one or more models; provider identity and model identity are separate concerns.\",\"anchor\":\"provider\"}]},\"tunes\":{}},{\"id\":\"h-sources\",\"type\":\"header\",\"data\":{\"text\":\"Primary sources and implementation evidence\",\"level\":2},\"tunes\":{}},{\"id\":\"p-sources-note\",\"type\":\"paragraph\",\"data\":{\"text\":\"Stable definitions below are anchored in standards\u002Fresearch; fast-moving implementation examples use current official engineering documentation. Aaasaasa AI Client is original implementation evidence and was checked against its codebase\u002Fdocumentation state dated 26 July 2026.\"},\"tunes\":{}},{\"id\":\"src-nist-profile\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fnvlpubs.nist.gov\u002Fnistpubs\u002Fai\u002FNIST.AI.600-1.pdf\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"NIST AI 600-1 — Generative Artificial Intelligence Profile\",\"description\":\"NIST's Generative AI profile, including the generative-AI definition and explicit distinction between model-, system-, application- and use-case-level concerns.\"}},\"tunes\":{}},{\"id\":\"src-nist-model\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fcsrc.nist.gov\u002Fglossary\u002Fterm\u002Fartificial_intelligence_model\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"NIST — Artificial Intelligence Model\",\"description\":\"Current NIST glossary definition of an AI model as a component of an information system that produces outputs from inputs using AI techniques.\"}},\"tunes\":{}},{\"id\":\"src-nist-system\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fcsrc.nist.gov\u002Fglossary\u002Fterm\u002Fartificial_intelligence_system\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"NIST — Artificial Intelligence System\",\"description\":\"Current NIST glossary definition showing that an AI system can include data systems, software, hardware, applications, tools or utilities using AI.\"}},\"tunes\":{}},{\"id\":\"src-rag-paper\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2005.11401\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Lewis et al. — Retrieval-Augmented Generation for Knowledge-Intensive NLP Tasks\",\"description\":\"The 2020 paper introducing the RAG formulation that combines a generative model with retrieved non-parametric memory.\"}},\"tunes\":{}},{\"id\":\"src-openai-file-search\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Ftools-file-search\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — File Search\",\"description\":\"Current official documentation for hosted file retrieval in the Responses API using uploaded-file knowledge bases, semantic search and keyword search.\"}},\"tunes\":{}},{\"id\":\"src-openai-functions\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Ffunction-calling\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — Function Calling\",\"description\":\"Current official documentation describing tool\u002Ffunction calling as the interface between models and external systems, data and actions.\"}},\"tunes\":{}},{\"id\":\"src-anthropic-context\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Feffective-context-engineering-for-ai-agents\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Anthropic — Effective Context Engineering for AI Agents\",\"description\":\"Engineering guidance defining context as the token set available during LLM sampling and explaining why context selection is a finite-resource problem.\"}},\"tunes\":{}}],\"version\":\"2.31.6\"}",{"time":1114,"blocks":1115,"version":1810},1791475413504,[1116,1120,1125,1130,1134,1138,1142,1146,1151,1155,1159,1188,1192,1196,1226,1230,1234,1238,1242,1246,1250,1254,1259,1266,1270,1274,1278,1282,1287,1291,1295,1299,1303,1307,1311,1315,1319,1323,1327,1331,1336,1340,1344,1378,1382,1386,1410,1414,1418,1423,1427,1431,1435,1461,1466,1470,1498,1502,1506,1536,1540,1544,1575,1579,1583,1587,1613,1617,1621,1625,1630,1634,1638,1645,1652,1656,1660,1664,1668,1672,1676,1680,1684,1688,1692,1696,1700,1726,1730,1753,1757,1761,1768,1775,1782,1789,1796,1803],{"id":215,"data":1117,"type":218,"tunes":1119},{"text":1118},"Generative AI is not one component. A production generative AI system usually combines a generative model with application code that supplies instructions and context, retrieves external knowledge when needed, exposes tools for reading or changing external systems, manages runtime state and permissions, and turns the result into a usable product. Treating the model, retrieval, tools, context, runtime, and application as the same thing hides the boundaries that determine freshness, security, reliability, cost, and control.",{},{"id":221,"data":1121,"type":226,"tunes":1124},{"body":1122,"title":1123,"variant":225},"\u003Cstrong>The model generates; retrieval finds external evidence; tools access data or perform actions; context is what the model can see for the current inference; the runtime coordinates execution; the application owns product rules, state, permissions, persistence, and user experience.\u003C\u002Fstrong> These layers can be packaged together by a vendor, but their responsibilities remain different.","Direct answer",{},{"id":229,"data":1126,"type":226,"tunes":1129},{"body":1127,"title":1128,"variant":233},"This article defines durable architectural responsibilities rather than one vendor stack. Current implementation examples were re-checked on \u003Cstrong>8 October 2026\u003C\u002Fstrong>. Vendor APIs and product names can change; the responsibility boundaries are more stable than any individual SDK or endpoint.","Terminology and version note",{},{"id":236,"data":1131,"type":241,"tunes":1133},{"title":1132,"maxLevel":239,"minLevel":240},"Contents",{},{"id":244,"data":1135,"type":42,"tunes":1137},{"text":1136,"level":240},"What does “generative AI” actually mean?",{},{"id":249,"data":1139,"type":218,"tunes":1141},{"text":1140},"At the model level, generative AI refers to AI models that generate derived synthetic content such as text, images, audio, video, code, or other digital output. NIST AI 600-1 uses this model-oriented meaning and separately discusses risks at model, system, application, and use-case levels.",{},{"id":254,"data":1143,"type":218,"tunes":1145},{"text":1144},"That distinction matters because an AI model is not the same thing as the complete AI system. NIST's current glossary defines an AI model as a component that produces outputs from inputs using computational, statistical, or machine-learning techniques, while an AI system can include software, hardware, applications, tools, or utilities that operate using AI.",{},{"id":259,"data":1147,"type":226,"tunes":1150},{"body":1148,"title":1149,"variant":263},"\u003Cstrong>Generative model ≠ generative AI application.\u003C\u002Fstrong>\u003Cbr>A model is one computational component. A usable AI product is a system built around that component.","A useful boundary",{},{"id":266,"data":1152,"type":42,"tunes":1154},{"text":1153,"level":240},"The simplest useful model of a generative AI system",{},{"id":271,"data":1156,"type":218,"tunes":1158},{"text":1157},"For a first mental model, imagine a company assistant answering: “Can this customer receive a refund today?” A useful answer may require several different responsibilities. The language model can interpret the question and write the explanation, but the current order state may come from a database tool, the refund policy may come from document retrieval, permissions may be enforced by the application, and the final action may require a controlled API call.",{},{"id":276,"data":1160,"type":305,"tunes":1187},{"steps":1161,"title":1186,"orientation":304},[1162,1165,1168,1171,1174,1177,1180,1183],{"label":1163,"description":1164},"1. User request","The application receives a natural-language question or task.",{"label":1166,"description":1167},"2. Application policy and state","Identity, tenant, permissions, current workflow state, and product rules define what the request is allowed to do.",{"label":1169,"description":1170},"3. Retrieval or direct data access","The system obtains external evidence or current facts when model knowledge is insufficient.",{"label":1172,"description":1173},"4. Context construction","Instructions, user input, selected evidence, relevant state, and tool definitions are assembled for the model.",{"label":1175,"description":1176},"5. Model inference","The generative model interprets the supplied context and produces text, structured output, or a tool request.",{"label":1178,"description":1179},"6. Tool execution when needed","The runtime or application validates and executes approved tool calls outside the model.",{"label":1181,"description":1182},"7. Observation and continuation","Tool results can return to the model as new context for another inference step.",{"label":1184,"description":1185},"8. Validation and product output","The application validates the result, records required state or audit data, and presents or executes the final outcome.","One common execution path",{},{"id":308,"data":1189,"type":218,"tunes":1191},{"text":1190},"Real systems do not always follow this sequence exactly. Retrieval can happen before the first model call, tools can be selected during an agent loop, deterministic application logic can bypass the model entirely, and validation can occur at several stages. The point is to separate responsibilities, not to impose one universal workflow.",{},{"id":313,"data":1193,"type":42,"tunes":1195},{"text":1194,"level":240},"The six boundaries that matter",{},{"id":318,"data":1197,"type":358,"tunes":1225},{"rows":1198,"title":1217,"layout":347,"columns":1218},[1199,1202,1205,1208,1211,1214],{"id":322,"label":1200,"values":1201},"Model",[325,325,325],{"id":327,"label":1203,"values":1204},"Retrieval",[325,325,325],{"id":331,"label":1206,"values":1207},"Tools",[325,325,325],{"id":335,"label":1209,"values":1210},"Context",[325,325,325],{"id":339,"label":1212,"values":1213},"Runtime \u002F orchestrator",[325,325,325],{"id":343,"label":1215,"values":1216},"Application",[325,325,325],"Six responsibilities inside one AI product",[1219,1221,1223],{"id":350,"label":1220},"Primary job",{"id":353,"label":1222},"Typical inputs",{"id":356,"label":1224},"Not the same as",{},{"id":361,"data":1227,"type":42,"tunes":1229},{"text":1228,"level":240},"1. The model: generation is its core responsibility",{},{"id":366,"data":1231,"type":218,"tunes":1233},{"text":1232},"A generative model maps supplied inputs to generated outputs. For a language model, that can include natural-language text, structured JSON, code, classifications, summaries, plans, or tool-call arguments. Multimodal generative models can work with additional input and output types.",{},{"id":371,"data":1235,"type":218,"tunes":1237},{"text":1236},"The model can contain substantial learned knowledge in its parameters, but parameterized knowledge is not a live database. The model does not automatically know a document created five minutes ago, the current stock level, a private customer record, or the state of an application unless that information is supplied through the current input path.",{},{"id":376,"data":1239,"type":218,"tunes":1241},{"text":1240},"This is why changing the model does not automatically solve stale knowledge, missing permissions, broken retrieval, incorrect state ownership, or unsafe tool execution. Those failures often belong to other layers.",{},{"id":381,"data":1243,"type":42,"tunes":1245},{"text":1244,"level":240},"2. Retrieval: finding external evidence is a separate operation",{},{"id":386,"data":1247,"type":218,"tunes":1249},{"text":1248},"Retrieval selects information from an external source before or during generation. The 2020 Retrieval-Augmented Generation work by Lewis et al. made the separation explicit by combining a parametric generative model with retrieved non-parametric memory. Modern production systems use many retrieval variants, but the architectural idea remains: useful evidence can be fetched at inference time instead of relying only on what the model learned during training.",{},{"id":391,"data":1251,"type":218,"tunes":1253},{"text":1252},"Retrieval can use lexical search, embeddings, vector search, hybrid search, SQL, knowledge graphs, metadata filters, APIs, or other selection mechanisms. A vector database is therefore one possible retrieval component, not the definition of RAG.",{},{"id":396,"data":1255,"type":226,"tunes":1258},{"body":1256,"title":1257,"variant":400},"A retrieved passage can be highly relevant and still be stale, unauthorized, from the wrong version, or insufficient to support a claim. Retrieval quality and evidence quality must be evaluated separately.","Relevance is not authority",{},{"id":403,"data":1260,"type":409,"tunes":1265},{"url":1261,"title":1262,"excerpt":1263,"ctaLabel":1264},"https:\u002F\u002Fstajic.de\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works","What Is RAG? The Simplest Explanation of How It Works","The canonical plain-English explanation of retrieval-augmented generation, including the separation between LLM, knowledge, state, memory and tools.","Read the RAG foundation",{},{"id":412,"data":1267,"type":42,"tunes":1269},{"text":1268,"level":240},"3. Tools: access and action are not model knowledge",{},{"id":417,"data":1271,"type":218,"tunes":1273},{"text":1272},"A tool is an interface through which an AI runtime can request functionality outside the model. A tool can query a database, search the web, read a file, calculate a value, call an internal service, create a ticket, send a message, modify a record, or trigger another controlled operation.",{},{"id":422,"data":1275,"type":218,"tunes":1277},{"text":1276},"OpenAI's current function-calling documentation makes this boundary explicit: function calling lets models interface with external systems and access data or actions provided by the application. The model can propose or select a call, but the external system performs the real operation.",{},{"id":427,"data":1279,"type":218,"tunes":1281},{"text":1280},"Tool use therefore creates two separate questions: Can the model request this capability? and Will the application authorize and execute it? A production system should not confuse model intent with permission to cause a side effect.",{},{"id":432,"data":1283,"type":226,"tunes":1286},{"body":1284,"title":1285,"variant":263},"A model can emit a valid tool request and still be denied. Authorization, argument validation, rate limits, transaction rules, audit requirements, and rollback belong outside the model.","Model intent is not execution authority",{},{"id":438,"data":1288,"type":42,"tunes":1290},{"text":1289,"level":240},"4. Context: what the model can see right now",{},{"id":443,"data":1292,"type":218,"tunes":1294},{"text":1293},"Context is the information available to the model for a particular inference step. Anthropic's context-engineering guidance describes context as the set of tokens included when sampling from an LLM. In practice, that set can contain system instructions, user messages, conversation history, retrieved evidence, tool definitions, tool results, memory summaries, and selected application state.",{},{"id":448,"data":1296,"type":218,"tunes":1298},{"text":1297},"Context is therefore neither the complete knowledge base nor long-term memory. A company may store ten million documents while only a handful of passages enter one model call. A runtime may persist a year of conversation history while exposing only the pieces needed for the current task.",{},{"id":453,"data":1300,"type":218,"tunes":1302},{"text":1301},"The context window also creates an engineering constraint. Adding more text does not guarantee a better answer; irrelevant, stale, contradictory, or low-authority information can dilute the evidence that actually matters.",{},{"id":458,"data":1304,"type":42,"tunes":1306},{"text":1305,"level":240},"5. Runtime and orchestration: coordinating the loop",{},{"id":463,"data":1308,"type":218,"tunes":1310},{"text":1309},"The runtime or orchestration layer coordinates how the model participates in a task. Depending on the architecture, it can manage sessions, model requests, tool discovery, tool-call loops, retries, handoffs, streaming events, timeouts, checkpoints, compaction, or execution environments.",{},{"id":468,"data":1312,"type":218,"tunes":1314},{"text":1313},"Some runtimes are thin application code around a model API. Others are full agent harnesses. A managed vendor runtime can own part of the loop while the application still owns domain truth, authorization, business side effects, and product lifecycle.",{},{"id":473,"data":1316,"type":218,"tunes":1318},{"text":1317},"This boundary is important because where the runtime runs and where inference runs are separate decisions. A locally running client or agent process can still call a remote model, while a remote application can call a model hosted on infrastructure under the organization's control.",{},{"id":478,"data":1320,"type":42,"tunes":1322},{"text":1321,"level":240},"6. The application: where AI becomes a product",{},{"id":483,"data":1324,"type":218,"tunes":1326},{"text":1325},"The application is the product boundary around the AI components. It owns the user experience, domain model, current state, identity, tenant scope, permissions, persistence, service integrations, validation, observability, billing or quota logic where relevant, and the rules that determine what the AI is allowed to see or do.",{},{"id":488,"data":1328,"type":218,"tunes":1330},{"text":1329},"This is the layer that turns “a model can produce useful output” into “a system can deliver a reliable capability.” The same model can participate in a private research assistant, a support workflow, a code agent, or a commerce application because the surrounding application changes the data, tools, policies, state, and execution contract.",{},{"id":493,"data":1332,"type":226,"tunes":1335},{"body":1333,"title":1334,"variant":225},"Provider and model substitution can be an architectural goal. The application's authoritative state, permissions, domain rules, audit trail, and user contract cannot simply be delegated to whichever model is currently selected.","The model is replaceable; the product boundary is not",{},{"id":499,"data":1337,"type":42,"tunes":1339},{"text":1338,"level":240},"How the parts work together in a real request",{},{"id":504,"data":1341,"type":218,"tunes":1343},{"text":1342},"Consider a support assistant asked: “Refund order 4711 if it is still eligible, and explain why.” The request combines knowledge, current state, authorization, reasoning, and a side effect.",{},{"id":509,"data":1345,"type":347,"tunes":1377},{"content":1346,"stretched":43,"withHeadings":14},[1347,1351,1354,1358,1362,1366,1370,1373],[1348,1349,1350],"Need","Correct layer","Why",[1352,1203,1353],"Refund policy","The system must find the current applicable policy and preserve its provenance.",[1355,1356,1357],"Order 4711 status","Direct data\u002Ftool access","The current order record is volatile authoritative state, not something to guess from model knowledge.",[1359,1360,1361],"User's authority to refund","Application \u002F authorization","Permissions must be enforced independently of what the model asks for.",[1363,1364,1365],"Interpret policy against order facts","Model + context","The model can reason over the policy evidence and current order state supplied to it.",[1367,1368,1369],"Execute refund","Tool + application transaction rules","A controlled external operation changes real state.",[1371,1200,1372],"Explain outcome","The model can generate the user-facing explanation from validated results.",[1374,1375,1376],"Audit what happened","Application \u002F runtime","The system records evidence, calls, decisions, side effects, and errors as required.",{},{"id":544,"data":1379,"type":218,"tunes":1381},{"text":1380},"If the assistant only has the language model, it can discuss refunds but cannot safely know whether order 4711 is currently eligible or perform the transaction. If it only has retrieval, it may find the policy but still lack live order state. If it has tools without application authorization, it may become capable but unsafe. Reliability comes from composing the layers with explicit ownership.",{},{"id":549,"data":1383,"type":42,"tunes":1385},{"text":1384,"level":240},"Different AI products use different combinations",{},{"id":554,"data":1387,"type":358,"tunes":1409},{"rows":1388,"title":1401,"layout":347,"columns":1402},[1389,1392,1395,1398],{"id":558,"label":1390,"values":1391},"Model-only assistant",[325,325,325,325],{"id":562,"label":1393,"values":1394},"Retrieval-grounded assistant",[325,325,325,325],{"id":566,"label":1396,"values":1397},"Tool-using assistant",[325,325,325,325],{"id":570,"label":1399,"values":1400},"Agentic application",[325,325,325,325],"The presence of a model does not define the whole architecture",[1403,1404,1405,1407],{"id":327,"label":1203},{"id":331,"label":1206},{"id":578,"label":1406},"Authoritative state",{"id":581,"label":1408},"Typical capability",{},{"id":585,"data":1411,"type":218,"tunes":1413},{"text":1412},"These are architecture patterns, not maturity rankings. A model-only feature can be the correct design when the task needs no external facts or actions. Adding retrieval, tools, memory, or an agent loop is justified only when the task requires those capabilities.",{},{"id":590,"data":1415,"type":42,"tunes":1417},{"text":1416,"level":240},"Implementation evidence: Aaasaasa AI Client",{},{"id":595,"data":1419,"type":226,"tunes":1422},{"body":1420,"title":1421,"variant":233},"The following section describes an implementation I built and reviewed against the Aaasaasa AI Client codebase and architecture documentation as of \u003Cstrong>26 July 2026\u003C\u002Fstrong>. It is evidence for the usefulness of these boundaries, not a claim that one implementation is a universal standard or a commercially deployed enterprise product.","Primary implementation evidence",{},{"id":601,"data":1424,"type":218,"tunes":1426},{"text":1425},"Aaasaasa AI Client is a local-first desktop AI workspace built with Nuxt 4, Electron and TypeScript. Its AI Hub deliberately separates agent\u002Fclient, provider, model, runtime location, permissions, and web client instead of treating them as one “AI” setting.",{},{"id":606,"data":1428,"type":218,"tunes":1430},{"text":1429},"That separation creates concrete behavior. Direct Chat can talk to models without filesystem or shell tools. A Codex agent can use a selected workspace and permission profile. Ollama can provide direct local inference, while LM Studio and configurable OpenAI-compatible endpoints represent other provider paths. A locally running Codex process can still use a cloud model, so the UI and architecture do not equate local runtime with local inference.",{},{"id":611,"data":1432,"type":218,"tunes":1434},{"text":1433},"The implementation also contains Qdrant\u002Fvector support, document-extraction capabilities and an authenticated directory MCP broker. Those components illustrate another boundary: retrieval infrastructure and tool access can live in the same product without becoming properties of the model itself.",{},{"id":616,"data":1436,"type":347,"tunes":1460},{"content":1437,"stretched":43,"withHeadings":14},[1438,1441,1443,1446,1449,1452,1455,1458],[1439,1440],"A01 concept","Aaasaasa AI Client implementation evidence",[1200,1442],"A provider-specific model identifier is selected separately from provider and runtime.",[1444,1445],"Provider","Ollama, LM Studio, OpenAI-compatible services and other provider paths are represented separately.",[1447,1448],"Runtime","Local or remote agent\u002Fruntime location is tracked independently of the model.",[1450,1451],"Tools \u002F access","Direct Chat has no filesystem or shell tools; controlled directory access is brokered separately.",[1453,1454],"Permissions","Workspace permission profiles are application\u002Fsession policy, not model capability.",[1456,1457],"Retrieval infrastructure","Vector support and document extraction exist as data\u002Fretrieval capabilities rather than model features.",[1215,1459],"The Electron\u002FNuxt product coordinates UI, credentials, providers, runtime discovery, permissions, tools and model interaction.",{},{"id":644,"data":1462,"type":226,"tunes":1465},{"body":1463,"title":1464,"variant":263},"The architecture became easier to reason about once \u003Cstrong>model, provider, runtime, permissions, tools, data and client\u003C\u002Fstrong> stopped being represented as one configuration choice. The distinction is operational: it determines what can run locally, what can access files, what may call paid cloud inference, and which layer owns authorization.","Implementation lesson",{},{"id":650,"data":1467,"type":42,"tunes":1469},{"text":1468,"level":240},"Common category errors",{},{"id":655,"data":1471,"type":347,"tunes":1497},{"content":1472,"stretched":43,"withHeadings":14},[1473,1476,1479,1482,1485,1488,1491,1494],[1474,1475],"Category error","What is actually happening",[1477,1478],"“The AI knows our documents.”","The application or retrieval layer makes selected document content available to the model.",[1480,1481],"“RAG is our vector database.”","The vector database can be one index or store used by a retrieval pipeline; RAG is the retrieval-plus-generation pattern.",[1483,1484],"“The model called our CRM.”","The model produced a tool request; the runtime\u002Fapplication authorized and executed the external call.",[1486,1487],"“It is local AI because the desktop agent runs locally.”","Runtime location and inference location are separate. A local runtime can still invoke a remote model.",[1489,1490],"“The model has permission to edit files.”","The application\u002Fruntime grants a tool capability under a permission policy; permission is not an intrinsic model property.",[1492,1493],"“More context means more knowledge.”","Context is the finite input made available for one inference. Larger context can contain more noise, conflict or stale information.",[1495,1496],"“The chatbot is the AI architecture.”","The chat UI is one interface. The system can also include identity, state, retrieval, tools, runtime, validation, persistence and observability.",{},{"id":684,"data":1499,"type":42,"tunes":1501},{"text":1500,"level":240},"Failure modes when the boundaries collapse",{},{"id":689,"data":1503,"type":218,"tunes":1505},{"text":1504},"Boundary mistakes are not merely terminology problems. They create distinct production failures that require different fixes.",{},{"id":694,"data":1507,"type":358,"tunes":1535},{"rows":1508,"title":1527,"layout":347,"columns":1528},[1509,1512,1515,1518,1521,1524],{"id":698,"label":1510,"values":1511},"Stale answer",[325,325,325],{"id":702,"label":1513,"values":1514},"Missing company fact",[325,325,325],{"id":706,"label":1516,"values":1517},"Unsafe side effect",[325,325,325],{"id":710,"label":1519,"values":1520},"Confused answer with lots of supplied text",[325,325,325],{"id":714,"label":1522,"values":1523},"Unexpected cloud use",[325,325,325],{"id":718,"label":1525,"values":1526},"Agent stalls or repeats",[325,325,325],"Diagnose the failing layer before replacing the model",[1529,1531,1533],{"id":724,"label":1530},"Symptom",{"id":727,"label":1532},"Likely boundary problem",{"id":730,"label":1534},"First architectural check",{},{"id":734,"data":1537,"type":42,"tunes":1539},{"text":1538,"level":240},"What is stable and what is version-sensitive?",{},{"id":739,"data":1541,"type":218,"tunes":1543},{"text":1542},"The architectural distinctions in this article are intentionally vendor-neutral. The current examples below are implementation facts that should be re-checked when APIs evolve.",{},{"id":744,"data":1545,"type":347,"tunes":1574},{"content":1546,"stretched":43,"withHeadings":14},[1547,1551,1555,1558,1562,1566,1570],[1548,1549,1550],"Area","Stable architectural idea","Verified current example on 8 Oct 2026",[1552,1553,1554],"AI model vs system","A model is a component inside a broader system","NIST's current glossary separately defines AI model and AI system.",[756,1556,1557],"Generation can be conditioned on retrieved external information","The Lewis et al. 2020 formulation remains the foundational reference; production retrieval methods now extend far beyond one dense index design.",[1559,1560,1561],"Hosted retrieval","Retrieval can be exposed as a managed tool","OpenAI File Search is currently a Responses API tool that searches uploaded-file knowledge bases using semantic and keyword retrieval.",[1563,1564,1565],"Function\u002Ftool calling","A model can request application-defined external capabilities","OpenAI currently documents function calling as an interface to external systems, data and actions.",[1567,1568,1569],"Context engineering","Model behavior depends on the finite information supplied for the current inference","Anthropic's current engineering guidance defines context as the token set included when sampling from the LLM and focuses on curating that set.",[1571,1572,1573],"Vendor APIs","SDKs, tool names, endpoint shapes and supported features change","Treat vendor documentation as version-sensitive even when the responsibility boundary remains stable.",{},{"id":777,"data":1576,"type":218,"tunes":1578},{"text":1577},"A source-of-truth article should therefore preserve both levels: stable concepts for architecture, and dated evidence for current implementations. Mixing the two makes an article age unnecessarily fast.",{},{"id":782,"data":1580,"type":42,"tunes":1582},{"text":1581,"level":240},"The AI component-boundary test",{},{"id":787,"data":1584,"type":218,"tunes":1586},{"text":1585},"When evaluating an AI feature, ask the following questions in order. The answers reveal which components the system actually has and which responsibilities are still implicit.",{},{"id":792,"data":1588,"type":305,"tunes":1612},{"steps":1589,"title":1611,"orientation":304},[1590,1593,1596,1599,1602,1605,1608],{"label":1591,"description":1592},"1. What generates the output?","Identify the exact model and the modalities or structured outputs it provides.",{"label":1594,"description":1595},"2. What facts are authoritative outside the model?","Identify documents, databases, APIs, current state and other sources of truth.",{"label":1597,"description":1598},"3. How is relevant information selected?","Separate direct lookup, search, retrieval, ranking and context construction.",{"label":1600,"description":1601},"4. What can cause real side effects?","List tools and external actions, then identify who validates and authorizes them.",{"label":1603,"description":1604},"5. What reaches the model as context?","Make instructions, evidence, state, history, memory and tool definitions explicit.",{"label":1606,"description":1607},"6. Who owns the loop?","Identify the runtime or harness that manages calls, events, retries, tool loops and sessions.",{"label":1609,"description":1610},"7. What remains the application's responsibility?","Make identity, permissions, domain state, validation, persistence, observability and UX explicit.","Seven questions for a production design",{},{"id":819,"data":1614,"type":42,"tunes":1616},{"text":1615,"level":240},"What generative AI is not",{},{"id":824,"data":1618,"type":218,"tunes":1620},{"text":1619},"Generative AI is not synonymous with an LLM, even though LLMs are a major class of generative model. It is also not synonymous with RAG, a vector database, an agent, a tool protocol, a chatbot UI, or an application.",{},{"id":829,"data":1622,"type":218,"tunes":1624},{"text":1623},"Those concepts can be connected, but each answers a different architectural question. An LLM asks how language output is produced. Retrieval asks where external evidence comes from. Tools ask how external capabilities are exposed. Context asks what the model can see. Runtime asks how execution is coordinated. The application asks how the capability becomes a controlled product.",{},{"id":834,"data":1626,"type":226,"tunes":1629},{"body":1627,"title":1628,"variant":263},"\u003Cstrong>Model = generate.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Retrieval = find evidence.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Tools = read or act outside the model.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Context = what the model sees now.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Runtime = coordinate execution.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Application = own the product, state, rules and permissions.\u003C\u002Fstrong>","If you remember only one model",{},{"id":840,"data":1631,"type":42,"tunes":1633},{"text":1632,"level":240},"Where to go next in the knowledge graph",{},{"id":845,"data":1635,"type":218,"tunes":1637},{"text":1636},"Once these boundaries are clear, deeper topics become easier to place. RAG belongs in retrieval and context construction. Retrieval Trigger decides when external evidence is required. Agent memory concerns what persists across time. Tool calling and MCP belong to capability access. Agent harnesses belong to runtime orchestration. RBAC, tenant isolation and domain authorization belong to the application and platform security boundary.",{},{"id":850,"data":1639,"type":409,"tunes":1644},{"url":1640,"title":1641,"excerpt":1642,"ctaLabel":1643},"https:\u002F\u002Fstajic.de\u002Fblog\u002Fwhere-does-an-llm-get-its-data-rag-data-sources-in-python","Where Does an LLM Get Its Data? RAG Data Sources in Python","A practical continuation showing how files, SQL, APIs, full-text search, embeddings and context assembly connect external data to an LLM.","See the data path in code",{},{"id":858,"data":1646,"type":409,"tunes":1651},{"url":1647,"title":1648,"excerpt":1649,"ctaLabel":1650},"https:\u002F\u002Fstajic.de\u002Fblog\u002Fwhen-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger","When Should an AI Stop Trusting Its Own Knowledge? — The Retrieval Trigger","A decision model for when an AI system should stop relying only on model knowledge and obtain external evidence.","Read the retrieval decision model",{},{"id":866,"data":1653,"type":42,"tunes":1655},{"text":1654,"level":240},"Limitations",{},{"id":871,"data":1657,"type":218,"tunes":1659},{"text":1658},"The six-layer model is a responsibility map, not a requirement that every product deploy six separate services. A small application may implement context construction, retrieval and orchestration inside one process. A managed platform may bundle several responsibilities behind one API. Physical deployment can be combined while semantic ownership remains distinct.",{},{"id":876,"data":1661,"type":218,"tunes":1663},{"text":1662},"Terminology also varies across vendors and research. “Agent,” “runtime,” “memory,” “tool,” “connector,” and “context” can be defined differently. The definitions here are chosen to make operational ownership and failure diagnosis explicit rather than to claim that every framework uses identical vocabulary.",{},{"id":881,"data":1665,"type":218,"tunes":1667},{"text":1666},"The Aaasaasa AI Client section documents one implementation pattern. It demonstrates that explicit boundaries are practical, but it does not prove that the same component layout is optimal for every AI product.",{},{"id":886,"data":1669,"type":42,"tunes":1671},{"text":1670,"level":240},"What would change this answer?",{},{"id":891,"data":1673,"type":218,"tunes":1675},{"text":1674},"The responsibility map would need revision if model architectures themselves began to own authoritative external state, permissions, durable transactional side effects, and verifiable source access as intrinsic properties rather than capabilities supplied by a surrounding system. Current production architectures do not make that a safe general assumption.",{},{"id":896,"data":1677,"type":218,"tunes":1679},{"text":1678},"Individual implementation examples will change much sooner. Hosted retrieval tools, agent APIs, MCP integrations, context-management features and provider capabilities evolve quickly. Those details should be updated without collapsing the underlying distinctions between generation, evidence, capability access, context, execution and application control.",{},{"id":901,"data":1681,"type":42,"tunes":1683},{"text":1682,"level":240},"Conclusion",{},{"id":906,"data":1685,"type":218,"tunes":1687},{"text":1686},"Generative AI becomes easier to design once “the AI” stops being treated as one black box. The model is the generative component, not the complete product. Retrieval provides external evidence. Tools expose capabilities. Context carries selected information into the current inference. The runtime coordinates execution. The application owns the authoritative product boundary.",{},{"id":911,"data":1689,"type":218,"tunes":1691},{"text":1690},"That separation is useful for more than explanation. It tells engineers where stale facts originate, where authorization belongs, why a local runtime can still use cloud inference, why RAG does not equal a vector database, why tool calls require validation, and why changing the model cannot repair every system failure.",{},{"id":916,"data":1693,"type":218,"tunes":1695},{"text":1694},"The durable architecture question is therefore not “Which AI model are we using?” It is: Which responsibility does each component own, what evidence crosses each boundary, and which layer is allowed to change real state?",{},{"id":921,"data":1697,"type":42,"tunes":1699},{"text":1698,"level":240},"FAQ",{},{"id":926,"data":1701,"type":926,"tunes":1725},{"items":1702,"title":1724},[1703,1706,1709,1712,1715,1718,1721],{"id":930,"answer":1704,"question":1705},"No. An LLM is one type of generative model. Generative AI also includes other modalities, and a production generative AI system can include retrieval, tools, runtime logic, application state, permissions, persistence and user interfaces around the model.","Is generative AI the same as an LLM?",{"id":934,"answer":1707,"question":1708},"Usually no. RAG is an application\u002Fsystem pattern that retrieves external information and supplies selected evidence to the model. Some platforms package retrieval tightly with model APIs, but the responsibility remains distinct.","Is RAG part of the model?",{"id":938,"answer":1710,"question":1711},"No. RAG can use vector search, lexical search, hybrid retrieval, SQL, APIs, knowledge graphs or other methods. The defining property is retrieval of external information for generation, not one storage technology.","Is a vector database required for RAG?",{"id":942,"answer":1713,"question":1714},"No. A tool is an external capability. Its definition may be represented in context, and its result may later enter context, but the actual capability executes outside the model.","Are tools the same as context?",{"id":946,"answer":1716,"question":1717},"No. Runtime location and inference location are separate. A local desktop application or agent can call a remote model, while a remote application can call an internally hosted model.","Does running an AI client locally mean the model is local?",{"id":950,"answer":1719,"question":1720},"The application or runtime security boundary should enforce authorization. A model can request an operation, but model intent should never be treated as sufficient execution authority.","Who should enforce permissions for AI tools?",{"id":954,"answer":1722,"question":1723},"Authoritative volatile state should normally remain in the application or domain system that owns it. The AI can receive the relevant state through controlled context or tool access when needed.","Where does current application state belong?","Generative AI system boundaries",{},{"id":960,"data":1727,"type":42,"tunes":1729},{"text":1728,"level":240},"Glossary",{},{"id":965,"data":1731,"type":965,"tunes":1752},{"title":1732,"entries":1733},"Core terms",[1734,1737,1739,1741,1744,1746,1748,1750],{"term":1735,"anchor":971,"definition":1736},"Generative model","An AI model designed to generate derived synthetic content such as text, images, audio, video, code or structured output.",{"term":1203,"anchor":327,"definition":1738},"The process of selecting relevant information from an external source or store for the current task.",{"term":756,"anchor":562,"definition":1740},"Retrieval-Augmented Generation: a pattern in which retrieved external information is supplied to a generative model to improve the current output.",{"term":1742,"anchor":566,"definition":1743},"Tool","A capability exposed to an AI runtime for reading data, calculating, searching, or performing an external action.",{"term":1209,"anchor":335,"definition":1745},"The information available to the model for a particular inference step.",{"term":1212,"anchor":983,"definition":1747},"The software layer that coordinates model calls, tool calls, task loops, sessions, retries, events or execution environments.",{"term":1215,"anchor":343,"definition":1749},"The product and domain layer that owns user interaction, authoritative state, permissions, validation, persistence and business behavior.",{"term":1444,"anchor":988,"definition":1751},"The service or runtime that exposes access to one or more models; provider identity and model identity are separate concerns.",{},{"id":992,"data":1754,"type":42,"tunes":1756},{"text":1755,"level":240},"Primary sources and implementation evidence",{},{"id":997,"data":1758,"type":218,"tunes":1760},{"text":1759},"Stable definitions below are anchored in standards\u002Fresearch; fast-moving implementation examples use current official engineering documentation. Aaasaasa AI Client is original implementation evidence and was checked against its codebase\u002Fdocumentation state dated 26 July 2026.",{},{"id":1002,"data":1762,"type":1009,"tunes":1767},{"link":1004,"meta":1763},{"image":1764,"title":1765,"description":1766},{"url":325},"NIST AI 600-1 — Generative Artificial Intelligence Profile","NIST's Generative AI profile, including the generative-AI definition and explicit distinction between model-, system-, application- and use-case-level concerns.",{},{"id":1012,"data":1769,"type":1009,"tunes":1774},{"link":1014,"meta":1770},{"image":1771,"title":1772,"description":1773},{"url":325},"NIST — Artificial Intelligence Model","Current NIST glossary definition of an AI model as a component of an information system that produces outputs from inputs using AI techniques.",{},{"id":1021,"data":1776,"type":1009,"tunes":1781},{"link":1023,"meta":1777},{"image":1778,"title":1779,"description":1780},{"url":325},"NIST — Artificial Intelligence System","Current NIST glossary definition showing that an AI system can include data systems, software, hardware, applications, tools or utilities using AI.",{},{"id":1030,"data":1783,"type":1009,"tunes":1788},{"link":1032,"meta":1784},{"image":1785,"title":1786,"description":1787},{"url":325},"Lewis et al. — Retrieval-Augmented Generation for Knowledge-Intensive NLP Tasks","The 2020 paper introducing the RAG formulation that combines a generative model with retrieved non-parametric memory.",{},{"id":1039,"data":1790,"type":1009,"tunes":1795},{"link":1041,"meta":1791},{"image":1792,"title":1793,"description":1794},{"url":325},"OpenAI — File Search","Current official documentation for hosted file retrieval in the Responses API using uploaded-file knowledge bases, semantic search and keyword search.",{},{"id":1048,"data":1797,"type":1009,"tunes":1802},{"link":1050,"meta":1798},{"image":1799,"title":1800,"description":1801},{"url":325},"OpenAI — Function Calling","Current official documentation describing tool\u002Ffunction calling as the interface between models and external systems, data and actions.",{},{"id":1057,"data":1804,"type":1009,"tunes":1809},{"link":1059,"meta":1805},{"image":1806,"title":1807,"description":1808},{"url":325},"Anthropic — Effective Context Engineering for AI Agents","Engineering guidance defining context as the token set available during LLM sampling and explaining why context selection is a finite-resource problem.",{},"2.31.6","Generative AI is more than a model. Learn how models, retrieval, tools, context, runtimes and applications fit together in production AI systems.",{"lang":7,"title":208,"content":210,"contentJson":1813,"excerpt":1066},{"time":212,"blocks":1814,"version":1065},[1815,1818,1821,1824,1827,1830,1833,1836,1839,1842,1845,1857,1860,1863,1883,1886,1889,1892,1895,1898,1901,1904,1907,1910,1913,1916,1919,1922,1925,1928,1931,1934,1937,1940,1943,1946,1949,1952,1955,1958,1961,1964,1967,1979,1982,1985,2002,2005,2008,2011,2014,2017,2020,2032,2035,2038,2050,2053,2056,2076,2079,2082,2093,2096,2099,2102,2113,2116,2119,2122,2125,2128,2131,2134,2137,2140,2143,2146,2149,2152,2155,2158,2161,2164,2167,2170,2173,2184,2187,2199,2202,2205,2210,2215,2220,2225,2230,2235],{"id":215,"data":1816,"type":218,"tunes":1817},{"text":217},{},{"id":221,"data":1819,"type":226,"tunes":1820},{"body":223,"title":224,"variant":225},{},{"id":229,"data":1822,"type":226,"tunes":1823},{"body":231,"title":232,"variant":233},{},{"id":236,"data":1825,"type":241,"tunes":1826},{"title":238,"maxLevel":239,"minLevel":240},{},{"id":244,"data":1828,"type":42,"tunes":1829},{"text":246,"level":240},{},{"id":249,"data":1831,"type":218,"tunes":1832},{"text":251},{},{"id":254,"data":1834,"type":218,"tunes":1835},{"text":256},{},{"id":259,"data":1837,"type":226,"tunes":1838},{"body":261,"title":262,"variant":263},{},{"id":266,"data":1840,"type":42,"tunes":1841},{"text":268,"level":240},{},{"id":271,"data":1843,"type":218,"tunes":1844},{"text":273},{},{"id":276,"data":1846,"type":305,"tunes":1856},{"steps":1847,"title":303,"orientation":304},[1848,1849,1850,1851,1852,1853,1854,1855],{"label":280,"description":281},{"label":283,"description":284},{"label":286,"description":287},{"label":289,"description":290},{"label":292,"description":293},{"label":295,"description":296},{"label":298,"description":299},{"label":301,"description":302},{},{"id":308,"data":1858,"type":218,"tunes":1859},{"text":310},{},{"id":313,"data":1861,"type":42,"tunes":1862},{"text":315,"level":240},{},{"id":318,"data":1864,"type":358,"tunes":1882},{"rows":1865,"title":346,"layout":347,"columns":1878},[1866,1868,1870,1872,1874,1876],{"id":322,"label":323,"values":1867},[325,325,325],{"id":327,"label":328,"values":1869},[325,325,325],{"id":331,"label":332,"values":1871},[325,325,325],{"id":335,"label":336,"values":1873},[325,325,325],{"id":339,"label":340,"values":1875},[325,325,325],{"id":343,"label":344,"values":1877},[325,325,325],[1879,1880,1881],{"id":350,"label":351},{"id":353,"label":354},{"id":356,"label":357},{},{"id":361,"data":1884,"type":42,"tunes":1885},{"text":363,"level":240},{},{"id":366,"data":1887,"type":218,"tunes":1888},{"text":368},{},{"id":371,"data":1890,"type":218,"tunes":1891},{"text":373},{},{"id":376,"data":1893,"type":218,"tunes":1894},{"text":378},{},{"id":381,"data":1896,"type":42,"tunes":1897},{"text":383,"level":240},{},{"id":386,"data":1899,"type":218,"tunes":1900},{"text":388},{},{"id":391,"data":1902,"type":218,"tunes":1903},{"text":393},{},{"id":396,"data":1905,"type":226,"tunes":1906},{"body":398,"title":399,"variant":400},{},{"id":403,"data":1908,"type":409,"tunes":1909},{"url":405,"title":406,"excerpt":407,"ctaLabel":408},{},{"id":412,"data":1911,"type":42,"tunes":1912},{"text":414,"level":240},{},{"id":417,"data":1914,"type":218,"tunes":1915},{"text":419},{},{"id":422,"data":1917,"type":218,"tunes":1918},{"text":424},{},{"id":427,"data":1920,"type":218,"tunes":1921},{"text":429},{},{"id":432,"data":1923,"type":226,"tunes":1924},{"body":434,"title":435,"variant":263},{},{"id":438,"data":1926,"type":42,"tunes":1927},{"text":440,"level":240},{},{"id":443,"data":1929,"type":218,"tunes":1930},{"text":445},{},{"id":448,"data":1932,"type":218,"tunes":1933},{"text":450},{},{"id":453,"data":1935,"type":218,"tunes":1936},{"text":455},{},{"id":458,"data":1938,"type":42,"tunes":1939},{"text":460,"level":240},{},{"id":463,"data":1941,"type":218,"tunes":1942},{"text":465},{},{"id":468,"data":1944,"type":218,"tunes":1945},{"text":470},{},{"id":473,"data":1947,"type":218,"tunes":1948},{"text":475},{},{"id":478,"data":1950,"type":42,"tunes":1951},{"text":480,"level":240},{},{"id":483,"data":1953,"type":218,"tunes":1954},{"text":485},{},{"id":488,"data":1956,"type":218,"tunes":1957},{"text":490},{},{"id":493,"data":1959,"type":226,"tunes":1960},{"body":495,"title":496,"variant":225},{},{"id":499,"data":1962,"type":42,"tunes":1963},{"text":501,"level":240},{},{"id":504,"data":1965,"type":218,"tunes":1966},{"text":506},{},{"id":509,"data":1968,"type":347,"tunes":1978},{"content":1969,"stretched":43,"withHeadings":14},[1970,1971,1972,1973,1974,1975,1976,1977],[513,514,515],[517,328,518],[520,521,522],[524,525,526],[528,529,530],[532,533,534],[536,323,537],[539,540,541],{},{"id":544,"data":1980,"type":218,"tunes":1981},{"text":546},{},{"id":549,"data":1983,"type":42,"tunes":1984},{"text":551,"level":240},{},{"id":554,"data":1986,"type":358,"tunes":2001},{"rows":1987,"title":573,"layout":347,"columns":1996},[1988,1990,1992,1994],{"id":558,"label":559,"values":1989},[325,325,325,325],{"id":562,"label":563,"values":1991},[325,325,325,325],{"id":566,"label":567,"values":1993},[325,325,325,325],{"id":570,"label":571,"values":1995},[325,325,325,325],[1997,1998,1999,2000],{"id":327,"label":328},{"id":331,"label":332},{"id":578,"label":579},{"id":581,"label":582},{},{"id":585,"data":2003,"type":218,"tunes":2004},{"text":587},{},{"id":590,"data":2006,"type":42,"tunes":2007},{"text":592,"level":240},{},{"id":595,"data":2009,"type":226,"tunes":2010},{"body":597,"title":598,"variant":233},{},{"id":601,"data":2012,"type":218,"tunes":2013},{"text":603},{},{"id":606,"data":2015,"type":218,"tunes":2016},{"text":608},{},{"id":611,"data":2018,"type":218,"tunes":2019},{"text":613},{},{"id":616,"data":2021,"type":347,"tunes":2031},{"content":2022,"stretched":43,"withHeadings":14},[2023,2024,2025,2026,2027,2028,2029,2030],[620,621],[323,623],[625,626],[628,629],[631,632],[634,635],[637,638],[640,641],{},{"id":644,"data":2033,"type":226,"tunes":2034},{"body":646,"title":647,"variant":263},{},{"id":650,"data":2036,"type":42,"tunes":2037},{"text":652,"level":240},{},{"id":655,"data":2039,"type":347,"tunes":2049},{"content":2040,"stretched":43,"withHeadings":14},[2041,2042,2043,2044,2045,2046,2047,2048],[659,660],[662,663],[665,666],[668,669],[671,672],[674,675],[677,678],[680,681],{},{"id":684,"data":2051,"type":42,"tunes":2052},{"text":686,"level":240},{},{"id":689,"data":2054,"type":218,"tunes":2055},{"text":691},{},{"id":694,"data":2057,"type":358,"tunes":2075},{"rows":2058,"title":721,"layout":347,"columns":2071},[2059,2061,2063,2065,2067,2069],{"id":698,"label":699,"values":2060},[325,325,325],{"id":702,"label":703,"values":2062},[325,325,325],{"id":706,"label":707,"values":2064},[325,325,325],{"id":710,"label":711,"values":2066},[325,325,325],{"id":714,"label":715,"values":2068},[325,325,325],{"id":718,"label":719,"values":2070},[325,325,325],[2072,2073,2074],{"id":724,"label":725},{"id":727,"label":728},{"id":730,"label":731},{},{"id":734,"data":2077,"type":42,"tunes":2078},{"text":736,"level":240},{},{"id":739,"data":2080,"type":218,"tunes":2081},{"text":741},{},{"id":744,"data":2083,"type":347,"tunes":2092},{"content":2084,"stretched":43,"withHeadings":14},[2085,2086,2087,2088,2089,2090,2091],[748,749,750],[752,753,754],[756,757,758],[760,761,762],[764,765,766],[768,769,770],[772,773,774],{},{"id":777,"data":2094,"type":218,"tunes":2095},{"text":779},{},{"id":782,"data":2097,"type":42,"tunes":2098},{"text":784,"level":240},{},{"id":787,"data":2100,"type":218,"tunes":2101},{"text":789},{},{"id":792,"data":2103,"type":305,"tunes":2112},{"steps":2104,"title":816,"orientation":304},[2105,2106,2107,2108,2109,2110,2111],{"label":796,"description":797},{"label":799,"description":800},{"label":802,"description":803},{"label":805,"description":806},{"label":808,"description":809},{"label":811,"description":812},{"label":814,"description":815},{},{"id":819,"data":2114,"type":42,"tunes":2115},{"text":821,"level":240},{},{"id":824,"data":2117,"type":218,"tunes":2118},{"text":826},{},{"id":829,"data":2120,"type":218,"tunes":2121},{"text":831},{},{"id":834,"data":2123,"type":226,"tunes":2124},{"body":836,"title":837,"variant":263},{},{"id":840,"data":2126,"type":42,"tunes":2127},{"text":842,"level":240},{},{"id":845,"data":2129,"type":218,"tunes":2130},{"text":847},{},{"id":850,"data":2132,"type":409,"tunes":2133},{"url":852,"title":853,"excerpt":854,"ctaLabel":855},{},{"id":858,"data":2135,"type":409,"tunes":2136},{"url":860,"title":861,"excerpt":862,"ctaLabel":863},{},{"id":866,"data":2138,"type":42,"tunes":2139},{"text":868,"level":240},{},{"id":871,"data":2141,"type":218,"tunes":2142},{"text":873},{},{"id":876,"data":2144,"type":218,"tunes":2145},{"text":878},{},{"id":881,"data":2147,"type":218,"tunes":2148},{"text":883},{},{"id":886,"data":2150,"type":42,"tunes":2151},{"text":888,"level":240},{},{"id":891,"data":2153,"type":218,"tunes":2154},{"text":893},{},{"id":896,"data":2156,"type":218,"tunes":2157},{"text":898},{},{"id":901,"data":2159,"type":42,"tunes":2160},{"text":903,"level":240},{},{"id":906,"data":2162,"type":218,"tunes":2163},{"text":908},{},{"id":911,"data":2165,"type":218,"tunes":2166},{"text":913},{},{"id":916,"data":2168,"type":218,"tunes":2169},{"text":918},{},{"id":921,"data":2171,"type":42,"tunes":2172},{"text":923,"level":240},{},{"id":926,"data":2174,"type":926,"tunes":2183},{"items":2175,"title":957},[2176,2177,2178,2179,2180,2181,2182],{"id":930,"answer":931,"question":932},{"id":934,"answer":935,"question":936},{"id":938,"answer":939,"question":940},{"id":942,"answer":943,"question":944},{"id":946,"answer":947,"question":948},{"id":950,"answer":951,"question":952},{"id":954,"answer":955,"question":956},{},{"id":960,"data":2185,"type":42,"tunes":2186},{"text":962,"level":240},{},{"id":965,"data":2188,"type":965,"tunes":2198},{"title":967,"entries":2189},[2190,2191,2192,2193,2194,2195,2196,2197],{"term":970,"anchor":971,"definition":972},{"term":328,"anchor":327,"definition":974},{"term":756,"anchor":562,"definition":976},{"term":332,"anchor":566,"definition":978},{"term":336,"anchor":335,"definition":980},{"term":982,"anchor":983,"definition":984},{"term":344,"anchor":343,"definition":986},{"term":625,"anchor":988,"definition":989},{},{"id":992,"data":2200,"type":42,"tunes":2201},{"text":994,"level":240},{},{"id":997,"data":2203,"type":218,"tunes":2204},{"text":999},{},{"id":1002,"data":2206,"type":1009,"tunes":2209},{"link":1004,"meta":2207},{"image":2208,"title":1007,"description":1008},{"url":325},{},{"id":1012,"data":2211,"type":1009,"tunes":2214},{"link":1014,"meta":2212},{"image":2213,"title":1017,"description":1018},{"url":325},{},{"id":1021,"data":2216,"type":1009,"tunes":2219},{"link":1023,"meta":2217},{"image":2218,"title":1026,"description":1027},{"url":325},{},{"id":1030,"data":2221,"type":1009,"tunes":2224},{"link":1032,"meta":2222},{"image":2223,"title":1035,"description":1036},{"url":325},{},{"id":1039,"data":2226,"type":1009,"tunes":2229},{"link":1041,"meta":2227},{"image":2228,"title":1044,"description":1045},{"url":325},{},{"id":1048,"data":2231,"type":1009,"tunes":2234},{"link":1050,"meta":2232},{"image":2233,"title":1053,"description":1054},{"url":325},{},{"id":1057,"data":2236,"type":1009,"tunes":2239},{"link":1059,"meta":2237},{"image":2238,"title":1062,"description":1063},{"url":325},{},"Post erfolgreich abgerufen",{"items":2242,"source":2327,"manualIds":2328,"manualMatchedIds":2329},[2243,2250,2257,2264,2271,2278,2285,2292,2299,2306,2313,2320],{"id":2244,"slug":2245,"title":2246,"excerpt":2247,"featuredImage":2248,"publishedAt":2249},"487","vector-databases-embeddings-and-reranking-three-different-parts-of-retrieval","向量数据库、嵌入和重排序：检索的三个不同部分","嵌入表示含义，向量数据库检索候选结果，重排序器则精炼结果。了解这三个检索层在RAG中如何不同并协同工作。","\u002Fuploads\u002F2026\u002F10\u002Fvector-databases-embeddings-and-reranking-three-different-parts-of-retrieval-1791480129884-9dtasz.webp","2026-10-08T11:21:00.000Z",{"id":2251,"slug":2252,"title":2253,"excerpt":2254,"featuredImage":2255,"publishedAt":2256},"476","mcp-vs-a2a-vs-ucp-vs-ap2-vs-a2ui-the-agent-protocol-stack-explained","MCP vs A2A vs UCP vs AP2 vs A2UI：智能体协议栈详解","MCP、A2A、UCP、AP2 和 A2UI 常被描述为相互竞争的智能体标准。它们大多解决的是不同的互操作性问题。本指南将每个协议映射到其实际标准化的边界，并展示它们如何在同一个生产系统中协同工作。","\u002Fuploads\u002F2026\u002F09\u002Fmcp-vs-a2a-vs-ucp-vs-ap2-vs-a2ui-the-agent-protocol-stack-explained-1790352625869-2ezle0.webp","2026-09-25T12:09:00.000Z",{"id":2258,"slug":2259,"title":2260,"excerpt":2261,"featuredImage":2262,"publishedAt":2263},"468","ai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context","AI代理记忆不是RAG：如何区分记忆、检索、状态和上下文","代理记忆、RAG、状态和上下文经常被当作可以互换的概念来使用。它们并不是。这个实用的架构模型将这四个层次区分开来，展示了每一层各自应处的位置，并解释了当系统将它们合并为一层时会出现什么问题。","\u002Fuploads\u002F2026\u002F09\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context-1790350560308-np0xy6.webp","2026-09-25T11:34:00.000Z",{"id":2265,"slug":2266,"title":2267,"excerpt":2268,"featuredImage":2269,"publishedAt":2270},"490","rbac-vs-tenant-isolation-two-different-security-boundaries","RBAC与租户隔离：两种不同的安全边界","RBAC 控制用户可以做什么；租户隔离控制该操作可以触及哪个租户的资源。了解为什么多租户 SaaS 安全需要这两道边界。","\u002Fuploads\u002F2026\u002F10\u002Frbac-vs-tenant-isolation-two-different-security-boundaries-1791485111528-qqtzby.webp","2026-10-08T14:43:00.000Z",{"id":2272,"slug":2273,"title":2274,"excerpt":2275,"featuredImage":2276,"publishedAt":2277},"478","what-is-rag-the-simplest-explanation-of-how-it-works","什么是RAG？对其工作原理的最简单解释","RAG听起来很复杂，但想法很简单：在AI回答之前，它先从知识源查找有用的信息，并将该信息提供给语言模型。本指南使用一个简单的思维模型来解释RAG、LLM、状态、记忆和工具。","\u002Fuploads\u002F2026\u002F09\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works-1790377492124-khjagt.webp","2026-09-25T19:03:00.000Z",{"id":2279,"slug":2280,"title":2281,"excerpt":2282,"featuredImage":2283,"publishedAt":2284},"485","enterprise-ai-architecture-what-changes-when-ai-enters-a-company","企业AI架构：当AI进入公司时会发生什么变化","企业AI架构阐释了AI如何在数据权限、身份、许可、提供商、风险、治理、评估、合规和运营方面改变公司系统。","\u002Fuploads\u002F2026\u002F10\u002Fenterprise-ai-architecture-what-changes-when-ai-enters-a-company-1791478161363-czrwaq.webp","2026-10-08T10:48:00.000Z",{"id":2286,"slug":2287,"title":2288,"excerpt":2289,"featuredImage":2290,"publishedAt":2291},"486","source-of-truth-in-ai-systems-where-reliable-knowledge-actually-comes-from","AI系统中的真相来源：可靠知识究竟从何而来","事实来源（Source of Truth）定义了对于特定事实或状态，哪个来源具有权威性。了解它与RAG、溯源、记忆、上下文、向量数据库和记录系统有何不同。","\u002Fuploads\u002F2026\u002F10\u002Fsource-of-truth-in-ai-systems-where-reliable-knowledge-actually-comes-from-1791479103235-6bq9em.webp","2026-10-08T13:02:00.000Z",{"id":2293,"slug":2294,"title":2295,"excerpt":2296,"featuredImage":2297,"publishedAt":2298},"495","sovereign-ai-control-of-models-data-infrastructure-and-dependencies","主权人工智能：模型、数据、基础设施与依赖关系的控制","主权人工智能关乎对模型、数据、基础设施、软件、运营和战略依赖的有效控制——而不仅仅是人工智能模型托管在哪里。","\u002Fuploads\u002F2026\u002F10\u002Fsovereign-ai-control-of-models-data-infrastructure-and-dependencies-1791488833132-niy85x.webp","2026-10-08T15:45:00.000Z",{"id":2300,"slug":2301,"title":2302,"excerpt":2303,"featuredImage":2304,"publishedAt":2305},"484","what-is-an-ai-platform-architect-models-data-runtime-security-and-operations","什么是AI平台架构师？模型、数据、运行时、安全与运维","AI平台架构师负责跨模型、提供商、检索、智能体、身份、安全、评估、可观测性和运营设计可复用的AI基础。","\u002Fuploads\u002F2026\u002F10\u002Fwhat-is-an-ai-platform-architect-models-data-runtime-security-and-operations-1791477229171-ou3zcc.webp","2026-10-08T12:32:00.000Z",{"id":2307,"slug":2308,"title":2309,"excerpt":2310,"featuredImage":2311,"publishedAt":2312},"467","the-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers","答案有效性边界：相关性到可靠AI答案之间缺失的层级","一个来源可能相关、权威，但对于所提出的问题仍然是错误的。缺失的层次是适用性：答案成立的条件，以及迫使其被重新考虑的变化。本文介绍了“答案有效性边界”这一面向人类、AI搜索和RAG系统的来源设计模式。","\u002Fuploads\u002F2026\u002F09\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers-1790272901306-1g5jly.webp","2026-09-24T11:59:00.000Z",{"id":2314,"slug":2315,"title":2316,"excerpt":2317,"featuredImage":2318,"publishedAt":2319},"471","how-to-know-whether-an-ai-agent-actually-used-the-right-evidence","如何判断一个AI智能体是否真正使用了正确的证据","AI代理可以引用来源，却仍然使用错误的证据。本文介绍一种实用方法，用于核查主张支持、来源权威性、适用性、出处，以及证据是否实际影响了答案。","\u002Fuploads\u002F2026\u002F09\u002Fhow-to-know-whether-an-ai-agent-actually-used-the-right-evidence-1790351317188-o5z9ve.webp","2026-09-25T11:47:00.000Z",{"id":2321,"slug":2322,"title":2323,"excerpt":2324,"featuredImage":2325,"publishedAt":2326},"480","when-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger","人工智能何时应停止信任自身知识？——检索触发机制","AI 模型并非每个问题都需要检索。重要的问题在于知道何时其内部知识已不再足够。检索触发器是一个实用的决策边界，它决定 AI 系统何时应停止仅依赖模型知识，并在回答前获取外部证据。","\u002Fuploads\u002F2026\u002F09\u002Fwhen-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger-1790574991244-f4rpyg.webp","2026-09-28T01:49:00.000Z","fallback",[],[]]