[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"portal-settings:stajic:zh":3,"public-menus:all":38,"post:what-is-an-ai-platform-architect-models-data-runtime-security-and-operations:zh":205,"related:post:what-is-an-ai-platform-architect-models-data-runtime-security-and-operations:zh:1":2452},{"statusCode":4,"data":5,"message":37},200,{"tenantId":6,"lang":7,"defaultLang":8,"siteUrl":9,"contactEmail":10,"brandName":11,"logoUrl":12,"siteName":11,"siteDescription":13,"ogImage":10,"robotsIndex":14,"socialLinks":10,"reservedSlugs":10,"seoPolicy":15},"stajic","zh","de","https:\u002F\u002Fstajic.de",null,"Stajic Platform","\u002FLogo_Planet.svg","Stajic Portal",true,{"branding":16,"relatedContent":17,"crossDomainLinks":18},{"logoUrl":12},{"enabled":14},[19,22,25,28,31,34],{"url":20,"label":21,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Ffigure.rocks","figure.rocks",{"url":23,"label":24,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Floving.rocks","loving.rocks",{"url":26,"label":27,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.com","bazify.com",{"url":29,"label":30,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.de","bazify.de",{"url":32,"label":33,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.at","bazify.at",{"url":35,"label":36,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.ba","bazify.ba","Portal settings resolved",[39,45],{"id":40,"name":41,"location":42,"isActive":14,"isDefault":43,"items":44},1,"main-navigation","header",false,[],{"id":46,"name":47,"location":48,"isActive":14,"isDefault":14,"items":49},4,"main-menu","sidebar",[50,66,79,93,103,118,133],{"id":51,"title":52,"url":60,"target":61,"icon":62,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":64,"portfolioId":10,"children":65},"item-18",{"de":53,"en":54,"es":55,"fr":56,"it":54,"ru":57,"sr":58,"zh":59},"Startseite","Home","Inicio","Accueil","Главная","Почетна","首页","\u002Ffull-stack-web-developer-munich-performance-seo-and-maintainable-builds","_self","i-lucide-home","page",111,[],{"id":67,"title":68,"url":75,"target":61,"icon":76,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":77,"portfolioId":10,"children":78},"item-22",{"de":69,"en":69,"es":70,"fr":69,"it":71,"ru":72,"sr":73,"zh":74},"Vision","Visión","Visione","Видение","Визија","想象","\u002Fueber-uns-webdesign-muenchen-webaplikation","i-lucide-eye",113,[],{"id":80,"title":81,"url":89,"target":61,"icon":90,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":91,"portfolioId":10,"children":92},"item-19",{"de":82,"en":83,"es":84,"fr":83,"it":85,"ru":86,"sr":87,"zh":88},"Leistungen","Services","Servicios","Servizi","Услуги","Услуге","服务","\u002Fservices-dienstleistungen-muenchen","i-lucide-wrench",116,[],{"id":94,"title":95,"url":99,"target":61,"icon":100,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":101,"portfolioId":10,"children":102},"item-23",{"de":96,"en":96,"es":96,"fr":96,"it":96,"ru":97,"sr":97,"zh":98},"Blog","Блог","博客","\u002Fblog","i-lucide-book-open",112,[],{"id":104,"title":105,"url":114,"target":61,"icon":115,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":116,"portfolioId":10,"children":117},"item-32",{"de":106,"en":107,"es":108,"fr":109,"it":110,"ru":111,"sr":112,"zh":113},"Neue Technologien","New Technologies","Nuevas tecnologías","Nouvelles technologies","Nuove tecnologie","Новые технологии","Нове технологије","新技术！","\u002Fneue-webtechnologien","i-lucide-sparkles",122,[],{"id":119,"title":120,"url":129,"target":61,"icon":130,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":131,"portfolioId":10,"children":132},"item-20",{"de":121,"en":122,"es":123,"fr":124,"it":125,"ru":126,"sr":127,"zh":128},"Kontakt","Contact us!","Contacto","Contact","Contatto","Контакт","Контактирајте нас","联系我们！","\u002Fcontact","i-lucide-mail",115,[],{"id":134,"title":135,"url":144,"target":61,"icon":145,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":147},"item-21",{"de":136,"en":137,"es":138,"fr":139,"it":140,"ru":141,"sr":142,"zh":143},"Unsere Arbeit","Our Work","Nuestro trabajo","Nos réalisations","I nostri lavori","Наши работы","Наши радови","文件夹","\u002Fportfolio","i-lucide-briefcase",114,[148,161,175,181,193],{"id":149,"title":150,"url":144,"target":61,"icon":159,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":160},"item-24",{"de":151,"en":152,"es":153,"fr":154,"it":155,"ru":156,"sr":157,"zh":158},"Alle Projekte","All Projects","Todos los proyectos","Tous les projets","Tutti i progetti","Все проекты","Сви пројекти","所有项目","i-lucide-grid-3x3",[],{"id":162,"title":163,"url":171,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":174},"item-29",{"de":164,"en":165,"es":166,"fr":167,"it":168,"ru":169,"sr":170,"zh":143},"Local Roots, Global Reach","Local Roots - Global Reach","Empresa local ","Entreprise locale","Azienda locale","Местная компания","Локално предузеће глобално тржиште","\u002Fportfolio\u002Flocal-roots-global-reach-communication-media-systems-for-modern-business","i-lucide-folder","custom",[],{"id":176,"title":177,"url":179,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":180},"item-28",{"de":178,"en":178,"es":178,"fr":178,"it":178,"ru":178,"sr":178,"zh":178},"Solr Suggester","\u002Fportfolio\u002Fsolr-fuzzy-suggester-und-solr-infix-suggester-abfrage-ueber-ajax-und-filterung",[],{"id":182,"title":183,"url":191,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":192},"item-27",{"de":184,"en":185,"es":186,"fr":187,"it":188,"ru":189,"sr":190,"zh":185},"Firmenwebseite SEO","Company Website SEO","Sitio web corporativo SEO","Site web d’entreprise SEO","Sito web aziendale SEO","Корпоративный сайт SEO","Пословна веб-страница SEO","\u002Fportfolio\u002Fseo-sem-branding-mobile-webseite-muenchen",[],{"id":194,"title":195,"url":203,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":204},"item-31",{"de":196,"en":197,"es":198,"fr":199,"it":200,"ru":201,"sr":202,"zh":197},"Digitalisierungsportal","Digitalization Portal","Portal de digitalización","Portail de numérisation","Portale di digitalizzazione","Портал цифровизации","Портал за дигитализацију","\u002Fportfolio\u002Fdigitalisierungsportal-archiv-museum-bibliothek-ead-lido-mets-mods",[],{"statusCode":4,"data":206,"message":2451},{"id":207,"title":208,"slug":209,"content":210,"contentJson":211,"excerpt":1202,"featuredImage":1203,"featuredImageAlt":1204,"featuredImageCaption":10,"featuredImageTitle":10,"featuredImageCopyright":10,"featuredImageAuthor":10,"featuredImageSourceUrl":10,"featuredImageLicense":10,"featuredImageIsAiGenerated":43,"status":1205,"publishedAt":1206,"createdAt":1207,"updatedAt":1208,"seoLocalePaths":1209,"categories":1218,"author":1231,"translations":1236},"484","什么是AI平台架构师？模型、数据、运行时、安全与运维","what-is-an-ai-platform-architect-models-data-runtime-security-and-operations","\u003Cp>\u003Cstrong>AI 平台架构师\u003C\u002Fstrong>设计可复用的 AI 基础，使多个应用、团队或租户上下文能够通过它访问模型、数据与检索、智能体与工具运行时、身份与权限、评估、可观测性、配额、密钥以及部署能力。该角色的范围比基础设施更广，但比拥有每一个 AI 赋能产品更窄：其核心职责是决定\u003Cstrong>哪些应当共享、共享能力如何治理与隔离，以及哪些必须保持为解决方案专属\u003C\u002Fstrong>。\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--info my-6 rounded-xl border p-5 border-blue-300 bg-blue-50 dark:border-blue-900 dark:bg-blue-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">直接回答\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">\u003Cstrong>AI 平台架构师为 AI 系统设计共享的技术与运营基座。\u003C\u002Fstrong>该角色不是为某一个助手或某一个工作流做架构，而是为模型\u002F提供商访问、网关与路由、检索服务、智能体运行时、工具访问、身份与租户隔离、密钥、评估、遥测、部署与生命周期管理定义可复用的契约与边界。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Caside class=\"editorjs-callout editorjs-callout--note my-6 rounded-xl border p-5 border-gray-300 bg-gray-50 dark:border-gray-700 dark:bg-gray-900\u002F40\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">术语说明\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">\u003Cstrong>AI 平台架构师是一个实用的角色标签，并非普遍标准化的职位名称。\u003C\u002Fstrong>ISO\u002FIEC\u002FIEEE 42010:2022 定义的是架构描述的概念，而非这一角色。不同组织可能会将这些职责拆分给平台架构师、解决方案架构师、企业架构师、安全架构师、MLOps\u002FLLMOps 专家以及平台工程团队。本文使用该术语来指代对可复用 AI 平台层的架构职责。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Caside class=\"editorjs-callout editorjs-callout--note my-6 rounded-xl border p-5 border-gray-300 bg-gray-50 dark:border-gray-700 dark:bg-gray-900\u002F40\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">当前来源说明 — 2026 年 10 月 8 日\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">此处稳定的架构原则与供应商无关。当前的 Microsoft、AWS 和 NIST 指南被用作外部实现与治理证据。NIST 表示 AI RMF 1.0 正在修订；供应商平台功能、网关产品、智能体运行时和模型能力的演进速度快于架构原则，因此对版本敏感的实现选择必须在部署前重新核查。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Cnav class=\"editorjs-toc\" data-editorjs-toc=\"true\" aria-label=\"目录\">\u003Cstrong class=\"editorjs-toc__title\">目录\u003C\u002Fstrong>\u003Col class=\"editorjs-toc__list editorjs-toc__list--depth-0\">\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-6\" class=\"editorjs-toc__link\">AI 平台架构师实际上架构什么？\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-10\" class=\"editorjs-toc__link\">最简单的例子\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-15\" class=\"editorjs-toc__link\">简单例子止步之处\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-18\" class=\"editorjs-toc__link\">最重要的平台决策：共享与解决方案专属\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-21\" class=\"editorjs-toc__link\">架构职责映射\u003C\u002Fa>\u003Col class=\"editorjs-toc__list editorjs-toc__list--depth-1\">\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-22\" class=\"editorjs-toc__link\">1. 模型与提供商访问\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-26\" class=\"editorjs-toc__link\">2. 网关、路由、配额与成本控制\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-30\" class=\"editorjs-toc__link\">3. 共享数据、检索与接地服务\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-34\" class=\"editorjs-toc__link\">4. 智能体与工具运行时\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-38\" class=\"editorjs-toc__link\">5. 身份、租户隔离与授权\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-42\" class=\"editorjs-toc__link\">6. 密钥、凭证与信任边界\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-45\" class=\"editorjs-toc__link\">7. 评估、可观测性与可审计性\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-49\" class=\"editorjs-toc__link\">8. 运行时、部署与位置\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-52\" class=\"editorjs-toc__link\">9. 平台生命周期、兼容性与入门\u003C\u002Fa>\u003C\u002Fli>\u003C\u002Fol>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-55\" class=\"editorjs-toc__link\">一个实用的控制平面\u002F执行平面\u002F解决方案平面模型\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-59\" class=\"editorjs-toc__link\">AI 平台架构师应该产出什么？\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-61\" class=\"editorjs-toc__link\">工作主要是权衡，而不是最大程度的集中化\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-63\" class=\"editorjs-toc__link\">这与相邻角色有何不同？\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-66\" class=\"editorjs-toc__link\">实现证据：这些平台边界如何出现在我自己的工作中\u003C\u002Fa>\u003Col class=\"editorjs-toc__list editorjs-toc__list--depth-1\">\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-68\" class=\"editorjs-toc__link\">Aaasaasa AI Client：提供商、运行时和权限分离\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-73\" class=\"editorjs-toc__link\">Aaasaasa AI CMS：作为平台边界的租户范围授权\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-77\" class=\"editorjs-toc__link\">真相源研究引擎：共享检索机制而不共享真相\u003C\u002Fa>\u003C\u002Fli>\u003C\u002Fol>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-82\" class=\"editorjs-toc__link\">当前架构指南如何支持这一平台范围\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-88\" class=\"editorjs-toc__link\">常见误解\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-90\" class=\"editorjs-toc__link\">AI 平台架构师应预防的故障模式\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-92\" class=\"editorjs-toc__link\">实用的平台架构决策序列\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-94\" class=\"editorjs-toc__link\">边缘情况和角色限制\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-100\" class=\"editorjs-toc__link\">什么会改变这个答案？\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-103\" class=\"editorjs-toc__link\">AI平台架构师检查清单\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-105\" class=\"editorjs-toc__link\">结论\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-109\" class=\"editorjs-toc__link\">相关规范知识\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-114\" class=\"editorjs-toc__link\">常见问题\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-116\" class=\"editorjs-toc__link\">术语表\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-118\" class=\"editorjs-toc__link\">主要来源和当前架构指南\u003C\u002Fa>\u003C\u002Fli>\u003C\u002Fol>\u003C\u002Fnav>\n\u003Ch2 id=\"section-6\">AI 平台架构师实际上架构什么？\u003C\u002Fh2>\n\u003Cp>工作的对象是\u003Cstrong>平台\u003C\u002Fstrong>：一组共享能力，在减少重复集成工作的同时，保留明确的安全、数据和运营边界。平台可以向众多消费方解决方案暴露模型访问、提供商适配器、检索原语、智能体执行、工具代理、策略执行、评估、遥测和部署服务。\u003C\u002Fp>\n\u003Cp>平台的价值并不只是因为组件被集中化。当消费方获得具有清晰契约、所有权、隔离、可观测性和生命周期规则的稳定能力时，平台才有价值。因此，关键的架构问题不是“每个人都应该使用哪个模型？”，而是\u003Cstrong>“哪些职责可以被安全地标准化和复用，同时不抹去每个解决方案的需求？”\u003C\u002Fstrong>。\u003C\u002Fp>\n\u003Csection class=\"editorjs-comparison my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">解决方案架构与平台架构解决不同范围的问题\u003C\u002Fh3>\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left dark:border-gray-700 dark:bg-gray-900\">\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">AI 解决方案架构师\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">AI 平台架构师\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">主要范围\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">One concrete AI-enabled product, workflow or application.\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">Reusable AI capabilities consumed by multiple solutions, teams or tenant contexts.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">核心问题\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">How should this solution meet its business, data, security, quality and operational requirements?\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">Which shared capabilities and controls should solutions consume, and where must solution-specific ownership remain?\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">数据权威\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">Defines which domain data is authoritative and how the solution may use it.\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">Provides storage, retrieval, provenance or access primitives without automatically becoming the authority for every domain.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">评估\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">Defines task-specific quality and acceptance criteria.\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">Provides reusable evaluation, telemetry and release mechanisms; it cannot define every domain&#39;s success threshold.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">生命周期\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">Owns the lifecycle of the specific workload.\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">Owns shared capability versions, compatibility, onboarding, quotas, policy and operational contracts.\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-10\">最简单的例子\u003C\u002Fh2>\n\u003Cp>设想一个组织有五个 AI 赋能产品：一个内部文档助手、一个客户支持副驾驶、一个软件工程智能体、一个合同审查工作流和一个产品搜索助手。每个产品都可以独立集成模型 API、保存凭据、实现重试、收集令牌指标、创建检索代码并构建自己的工具权限。\u003C\u002Fp>\n\u003Cp>当每个团队都发明不同的安全和运营模型时，这种重复既昂贵又危险。共享平台则可以提供经批准的提供商连接、模型发现、配额、凭据、租户感知访问、通用遥测、可复用检索服务以及智能体\u002F工具运行时契约。\u003C\u002Fp>\n\u003Cp>但平台必须在正确的边界处停止。合同审查解决方案可能需要法律文档权威和引用规则，而软件智能体不需要。产品搜索助手可能需要特定于商务的新鲜度和授权规则。\u003Cstrong>可复用基础设施并不会让所有领域事实都变得可复用。\u003C\u002Fstrong>\u003C\u002Fp>\n\u003Csection class=\"editorjs-process my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">共享 AI 请求路径\u003C\u002Fh3>\u003Cdiv class=\"grid grid-cols-1 md:grid-cols-2 xl:grid-cols-3 gap-4\">\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">1\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">1. 消费方标识自身\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">调用应用、用户、服务、团队或租户通过经过身份验证的身份和明确范围进入。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">2\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">2. 平台策略生效\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">网关和策略层确定允许的提供商、模型、配额、数据路径、工具和执行模式。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">3\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">3. 共享能力执行\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">请求可能使用推理、检索、智能体运行时、工具访问或其他可复用平台服务。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">4\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">4. 解决方案专属上下文保持权威\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">消费方解决方案提供领域规则、用户意图、数据权威、任务特定约束和验收逻辑。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">5\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">5. 遥测和证据被捕获\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">平台记录身份、路由、模型\u002F提供商、延迟、成本、错误、工具活动以及其他允许的可观测性信号。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">6\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">6. 结果在解决方案契约下返回\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">解决方案仍然负责判断输出对其用户和领域是否可接受。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-15\">简单例子止步之处\u003C\u002Fh2>\n\u003Cp>集中化并不自动等于架构。在多个模型 API 前放置一个端点是有用的，但它本身并不会创建 AI 平台。生产平台还需要身份边界、能力契约、提供商健康与生命周期处理、配额、密钥所有权、可观测性、兼容性规则、安全控制、发布纪律和明确的运营责任。\u003C\u002Fp>\n\u003Cp>相反的失败也很常见：把每个提示词、向量索引、业务规则、智能体和应用工作流都放进一个“AI 后端”。这会创建一个单体，其共享状态是偶然的而非架构性的。\u003Cstrong>平台应当标准化横切能力，而不是仅仅因为涉及 AI 就吸收领域所有权。\u003C\u002Fstrong>\u003C\u002Fp>\n\u003Ch2 id=\"section-18\">最重要的平台决策：共享与解决方案专属\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">能力领域\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">适合由共享平台负责\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">通常仍属于解决方案特定范围\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型访问\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">已批准的提供商连接、适配器、凭据、健康检查、路由原语、配额\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">任务特定的模型验收、提示行为、质量阈值\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">检索\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">摄取原语、提取、索引、搜索 API、来源契约、授权钩子\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">权威语料库、新鲜度规则、领域元数据、证据充分性\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">智能体与工具\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">运行时生命周期、工具注册表\u002F代理、权限执行、追踪、取消\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">业务工作流、允许的操作语义、升级策略、任务成功\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">安全\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">身份集成、密钥存储、策略执行、审计契约、租户隔离机制\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">数据分类、业务授权规则、领域特定风险接受\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">评估\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">测试框架、数据集\u002F版本机制、遥测、实验\u002F发布工作流\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">基准真值、领域测试集、验收阈值、用户结果\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">运维\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">部署模式、健康检查、指标、事件集成、容量控制\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">存在差异时的解决方案 SLO、业务连续性影响、工作负载特定运行手册\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Caside class=\"editorjs-callout editorjs-callout--success my-6 rounded-xl border p-5 border-emerald-300 bg-emerald-50 dark:border-emerald-900 dark:bg-emerald-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">平台原则\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">\u003Cstrong>在复用真实存在的地方共享机制和控制；将权威和验收保留在领域拥有它们的地方。\u003C\u002Fstrong> 这可以防止两种相反的错误：到处重复建设基础设施，以及中央平台错误地成为每个应用的数据、策略和质量的所有者。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-21\">架构职责映射\u003C\u002Fh2>\n\u003Ch3 id=\"section-22\">1. 模型与提供商访问\u003C\u002Fh3>\n\u003Cp>平台架构师定义消费者如何发现和调用模型，而不强迫每个应用硬编码某一个提供商。这包括提供商适配器、模型标识符、能力元数据、身份验证、健康检查、端点配置、请求规范化和兼容行为。\u003C\u002Fp>\n\u003Cp>提供商抽象必须保持诚实。不同提供商暴露不同的上下文限制、工具语义、结构化输出行为、多模态能力、安全控制、缓存、定价和故障模式。好的抽象会创建稳定的平台契约，同时保留对无法被有意义地扁平化的能力的访问。\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--warning my-6 rounded-xl border p-5 border-amber-300 bg-amber-50 dark:border-amber-900 dark:bg-amber-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">不要将抽象与假装提供商完全相同混为一谈\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">最低共同标准的 API 可以让迁移更容易，但也可能抹去重要的能力。架构应定义哪些功能可移植、哪些是提供商特定的，以及消费者如何发现这种差异。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch3 id=\"section-26\">2. 网关、路由、配额与成本控制\u003C\u002Fh3>\n\u003Cp>共享 AI 网关可以集中处理身份验证、路由、限流、重试、令牌限制、使用归因和策略执行。Microsoft 当前的 AI Gateway 指南明确将每分钟令牌限制、配额和多项目隔离视为平台关注点；AWS 同样暴露账户和模型配额以及集中控制。\u003C\u002Fp>\n\u003Cp>因此，当网关承载 AI 特定策略和运维语义时，它就不只是反向代理。但它不应静默地做出业务决策。路由策略可能偏好健康的本地模型、成本更低的提供商或符合区域合规的端点；该路由对特定任务是否可接受，仍然是平台与解决方案之间的契约。\u003C\u002Fp>\n\u003Cp>路由还需要故障语义。如果首选模型不可用，平台必须知道是否允许回退、云路由是否需要明确同意、能力较低的模型是否有效，以及该决策如何暴露给可观测性。\u003C\u002Fp>\n\u003Ch3 id=\"section-30\">3. 共享数据、检索与接地服务\u003C\u002Fh3>\n\u003Cp>检索服务是强有力的平台候选，因为解析、分块、索引、词法搜索、语义搜索、元数据过滤、来源和引用机制都可复用。然而，平台不能将共享检索引擎与共享事实来源混为一谈。\u003C\u002Fp>\n\u003Cp>解决方案仍然拥有以下问题：哪个语料库是权威的？哪个版本有效？此用户能否看到此文档？数据必须有多新鲜？什么算作充分证据？检索失败时能否生成答案？即使平台提供检索机制，这些也是领域和解决方案需求。\u003C\u002Fp>\n\u003Cp>这种边界在多租户系统中尤为重要。技术上共享的索引或向量服务并不证明跨租户可见性合理。授权上下文必须在检索过程中保留，而不是仅在搜索结果已经跨越边界之后才添加。\u003C\u002Fp>\n\u003Ch3 id=\"section-34\">4. 智能体与工具运行时\u003C\u002Fh3>\n\u003Cp>智能体系统增加了可复用的运行时关注点：线程\u002F会话生命周期、规划循环、工具注册、工具调用、取消、超时、人工审批、内存\u002F状态接口、远程智能体协议和追踪关联。平台可以提供这些机制，使每个产品无需重新构建它们。\u003C\u002Fp>\n\u003Cp>平台还必须将工具权限与模型能力分开。模型能够生成 shell 命令并不意味着运行时应允许 shell 执行。权限边界属于应用\u002F运行时架构，并且必须独立于模型可执行。\u003C\u002Fp>\n\u003Cp>当前 AWS Agentic AI 指南强调有界代理、明确授权、端到端追踪、版本化行为工件以及与后果相称的人类监督。这些都是平台赋能关注点，但消费方解决方案仍然定义其领域内哪些操作是合法的。\u003C\u002Fp>\n\u003Ch3 id=\"section-38\">5. 身份、租户隔离与授权\u003C\u002Fh3>\n\u003Cp>AI 平台通常位于高价值模型、专有数据和具备操作能力的工具之前。因此，身份验证只是开始。架构必须在每个需要它的特权操作中携带用户、服务、应用和租户上下文。\u003C\u002Fp>\n\u003Cp>\u003Cstrong>RBAC 和租户隔离解决不同的问题。\u003C\u002Fstrong>RBAC 回答一个身份可以做什么；租户隔离回答该身份可以对哪个租户的资源进行操作。一个检查角色但丢失租户上下文的平台仍然可能暴露错误的数据。\u003C\u002Fp>\n\u003Cp>Microsoft 当前的 AI 工作负载指南明确建议身份分段和授权感知的内容访问。AWS 的多租户生成式 AI 平台指南同样将逻辑隔离、集中控制和可审计性视为平台关注点。\u003C\u002Fp>\n\u003Ch3 id=\"section-42\">6. 密钥、凭证与信任边界\u003C\u002Fh3>\n\u003Cp>平台应定义谁拥有提供商密钥、远程持有者令牌、签名材料和工具凭证，它们存储在哪里，哪个进程可以访问它们，如何轮换它们，以及它们是否可能到达浏览器或不受信任的渲染器。\u003C\u002Fp>\n\u003Cp>这是一个架构边界，而不是实现细节。如果每个消费方应用都将提供商凭证复制到自己的配置中，组织就重复了运营负担和爆炸半径。只有当平台本身具有更窄、可审计的访问路径时，集中化才能降低这种风险。\u003C\u002Fp>\n\u003Ch3 id=\"section-45\">7. 评估、可观测性与可审计性\u003C\u002Fh3>\n\u003Cp>可复用平台可以提供评估工具、追踪 ID、模型\u002F提供商元数据、令牌和成本指标、延迟、错误率、提示\u002F模型版本关联、代理\u002F工具追踪以及受控日志记录。AWS 和 Microsoft 都将可观测性和评估视为 AI 工作负载的核心生产关注点。\u003C\u002Fp>\n\u003Cp>平台评估和解决方案评估必须保持分离。平台可以验证端点是否健康、模型版本是否通过通用回归套件以及追踪是否完整。如果没有领域特定的基准真相和验收标准，它无法判定法律答案、医疗工作流或产品推荐是否可接受。\u003C\u002Fp>\n\u003Cp>日志记录还创建了隐私边界。提示和响应日志可能包含敏感或专有数据。因此，平台架构师必须决定记录什么、脱敏什么、采样什么、保留什么以及可访问什么，而不是假设更多遥测总是更安全。\u003C\u002Fp>\n\u003Ch3 id=\"section-49\">8. 运行时、部署与位置\u003C\u002Fh3>\n\u003Cp>平台架构师决定共享 AI 能力如何部署和访问：托管云服务、自托管端点、本地推理、混合路由、容器化服务、桌面运行时、私有网络或气隙环境。重要的区别在于\u003Cstrong>控制\u002F运行时进程在哪里运行\u003C\u002Fstrong>以及\u003Cstrong>推理和数据处理实际发生在哪里\u003C\u002Fstrong>。\u003C\u002Fp>\n\u003Cp>本地客户端仍然可能调用云模型。云控制平面可能路由到本地模型。远程代理可能在客户网络内执行工具。因此，架构图必须显示信任和数据流边界，而不是将“本地”和“云”用作模糊的标签。\u003C\u002Fp>\n\u003Ch3 id=\"section-52\">9. 平台生命周期、兼容性与入门\u003C\u002Fh3>\n\u003Cp>只有当消费方能够长期依赖可复用能力时，它才成为平台。这需要版本化契约、迁移规则、兼容性策略、弃用、发布测试、回滚、事件所有权、容量规划、文档以及新团队或应用的入门路径。\u003C\u002Fp>\n\u003Cp>快速发展的 AI 生态系统使这一点尤为重要。模型名称、SDK、协议版本、提供商 API 和安全能力各自独立变化。平台必须吸收其中一些波动性，同时不隐藏对解决方案行为有重大影响的变更。\u003C\u002Fp>\n\u003Ch2 id=\"section-55\">一个实用的控制平面\u002F执行平面\u002F解决方案平面模型\u003C\u002Fh2>\n\u003Caside class=\"editorjs-callout editorjs-callout--note my-6 rounded-xl border p-5 border-gray-300 bg-gray-50 dark:border-gray-700 dark:bg-gray-900\u002F40\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">提议的架构模型\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">下面的三平面模型是一种实用的责任推理方式；它不是 ISO、NIST、Microsoft 或 AWS 标准。其目的是明确所有权边界。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">平面\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">典型职责\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">不应静默拥有\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">平台控制平面\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">提供商注册表、模型策略、配额、租户配置、身份、密钥、路由规则、能力版本、部署配置\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">应用业务逻辑或领域真相\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">平台执行\u002F数据平面\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">推理请求、检索操作、代理\u002F工具执行、提取、索引、遥测发射、策略执行\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">仅因基础设施共享而进行跨租户访问\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">解决方案平面\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">用户工作流、提示\u002F指令、权威语料选择、领域授权、业务规则、任务评估与验收\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">平台明确拥有的低级提供商集成\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Cp>这种分离有助于诊断平台漂移。如果应用程序必须知道每个提供商特定的凭据和端点，那么平台契约就太薄弱了。如果平台决定哪个客户记录在法律上是权威的，或者某个领域答案是否可接受，那么平台就已经越界进入了解决方案所有权。\u003C\u002Fp>\n\u003Ch2 id=\"section-59\">AI 平台架构师应该产出什么？\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">架构工件\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">目的\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">平台能力地图\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">定义平台提供什么、谁消费它以及哪些能力仍在范围之外。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">提供商\u002F模型契约\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">定义提供商、模型、能力、抽象边界、路由元数据和回退语义。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">身份与租户模型\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">定义用户\u002F服务\u002F应用身份、租户上下文、RBAC\u002FABAC 钩子和资源隔离。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">网关与配额策略\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">定义速率限制、令牌\u002F成本预算、路由控制、重试和容量行为。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">检索\u002F数据契约\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">定义摄取、来源、搜索、元数据、授权传播以及领域权威保留在哪里。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">代理\u002F工具契约\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">定义运行时生命周期、工具注册、权限、审批、取消和跟踪行为。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">密钥与信任边界模型\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">定义凭据所有权、存储、进程边界、轮换和敏感数据路径。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">评估与遥测契约\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">定义通用指标、跟踪、数据集\u002F版本链接、日志策略和解决方案扩展点。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">生命周期与兼容性策略\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">定义版本、迁移、弃用、发布、回滚、事件所有权和入门。\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-61\">工作主要是权衡，而不是最大程度的集中化\u003C\u002Fh2>\n\u003Csection class=\"editorjs-comparison my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">常见的平台权衡\u003C\u002Fh3>\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left dark:border-gray-700 dark:bg-gray-900\">\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">压力 A\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">压力 B\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">提供商抽象\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">Stable portable platform API\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">Access to provider-specific capabilities and fast innovation\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">复用\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">Shared services reduce duplication\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">Isolation and domain autonomy prevent unsafe coupling\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">治理\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">Central policy and auditability\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">Team speed and local experimentation\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">可观测性\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">Rich traces for debugging and evaluation\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">Privacy, data minimization and logging cost\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">可用性\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">Fallback and multi-provider resilience\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">Predictable quality, compliance and data-location guarantees\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">平台范围\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">More reusable capabilities\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">Smaller blast radius and less platform lock-in\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-63\">这与相邻角色有何不同？\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">角色\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">主要架构范围\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">AI 解决方案架构师\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">一个具体的 AI 赋能解决方案及其端到端需求、边界、权衡和生产验收。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">AI 平台架构师\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">跨多个解决方案或团队消费的可复用 AI 能力以及运营\u002F安全契约。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">企业架构师\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">在更广泛层面上的组织级业务\u002F技术组合、能力和治理对齐。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">MLOps \u002F LLMOps 架构师或专家\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型和 AI 生命周期、部署、实验、可观测性、发布和运营实践；可能强烈重叠，但不自动拥有整个共享应用平台。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">平台工程师 \u002F SRE\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">实现和运营平台基础设施、可靠性、自动化和开发者体验；架构责任可能与平台架构师共享。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">AI \u002F 软件工程师\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">在商定的架构内实现模型、集成、服务、代理、检索和产品功能。\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Cp>这些边界是组织性的，不是普遍的。在小型团队中，一个人可能承担多项职责。在受监管的企业中，它们可能分散在架构、安全、平台、数据和运营组中。有用的区别是\u003Cstrong>架构责任的范围\u003C\u002Fstrong>，而不是组织图上印的职位名称。\u003C\u002Fp>\n\u003Ch2 id=\"section-66\">实现证据：这些平台边界如何出现在我自己的工作中\u003C\u002Fh2>\n\u003Caside class=\"editorjs-callout editorjs-callout--note my-6 rounded-xl border p-5 border-gray-300 bg-gray-50 dark:border-gray-700 dark:bg-gray-900\u002F40\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">原始实现证据\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">以下部分描述了我自己项目中的具体模式。它们是这些架构边界已在真实代码和项目系统中实现或明确设计的证据。它们\u003Cstrong>不是\u003C\u002Fstrong>声称这些项目一起已经构成商业部署的企业 AI 平台。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch3 id=\"section-68\">Aaasaasa AI Client：提供商、运行时和权限分离\u003C\u002Fh3>\n\u003Cp>Aaasaasa AI Client 是一个使用 Nuxt 4、Electron 和 TypeScript 构建的本地优先桌面 AI 工作区。其 AI Hub 有意分离\u003Cstrong>代理\u002F客户端、提供商、模型、连接\u002F运行时位置、权限和 Web 客户端\u003C\u002Fstrong>，而不是将它们视为一个配置值。\u003C\u002Fp>\n\u003Cp>该实现包括直接提供商适配器、Codex 代理运行时集成、本地 Ollama\u002FLM Studio 路径、OpenAI 兼容服务、集中式工作区权限、主进程凭据存储、DuckDB、Qdrant\u002F向量支持、PDF\u002F可读性提取以及基于 MCP 的认证目录访问。\u003C\u002Fp>\n\u003Cp>两个平台经验尤其相关。首先，本地运行时与本地推理不同：本地 Codex 进程仍然可以使用云模型。其次，自动路由不会静默地从本地回退到付费云推理。这使得路由策略和运行时局部性变得明确，而不是从 UI 标签推断。\u003C\u002Fp>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">已实现的边界\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">平台架构含义\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">代理 vs 提供商 vs 模型\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">不同的职责可以独立演进，而不是隐藏在一个“AI”选择器后面。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">权限与模型分离\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">文件系统\u002F工具权限属于运行时策略，而不是模型能力。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">主进程密钥\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">凭据所有权遵循特权进程边界，而不是渲染器\u002FUI。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">提供商健康与模型发现\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">路由和可用性是运行时\u002F平台关注点。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">无静默云回退\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">成本、局部性和数据传输语义保持为明确的策略决策。\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch3 id=\"section-73\">Aaasaasa AI CMS：作为平台边界的租户范围授权\u003C\u002Fh3>\n\u003Cp>Aaasaasa AI CMS 代码库提供了一个独立的实现示例：租户范围的 RBAC 通过绑定到租户标识符的角色、权限和用户角色分配来表示。系统权限按能力分组，角色查找和更新保持租户范围。\u003C\u002Fp>\n\u003Cp>这本身并不能证明一个完整的 AI 平台，但它与最困难的共享平台边界之一直接相关：可复用服务必须保留\u003Cstrong>谁可以做什么\u003C\u002Fstrong>以及\u003Cstrong>针对哪个租户\u003C\u002Fstrong>。在应用平台之上添加 AI 推理或检索并不会消除这一要求。\u003C\u002Fp>\n\u003Cp>架构上的含义是，模型网关、检索服务和代理应使用已建立的身份\u002F租户上下文，而不是发明一个并行的、仅限 AI 的授权体系。\u003C\u002Fp>\n\u003Ch3 id=\"section-77\">真相源研究引擎：共享检索机制而不共享真相\u003C\u002Fh3>\n\u003Cp>真相源研究引擎提供了第三个实现示例。不同的研究模式共享一个共同的证据核心：来源、工件、溯源、主张、关系、矛盾、参考模型和审计追踪。该系统还提供本地词汇检索、可选的语义检索、提取、快照和基于 SHA-256 的溯源。\u003C\u002Fp>\n\u003Cp>该项目明确将搜索和语义相似性视为发现信号而非证据。结果必须追溯到具体的来源和定位符，然后才能支持一项主张。这正是 AI 平台所需的区分：\u003Cstrong>可复用的检索机制可以共享，而证据权威仍由消费方法和领域来治理。\u003C\u002Fstrong>\u003C\u002Fp>\n\u003Cp>该引擎还说明了为什么一个共享平台不需要一种共享解释。历史、科学\u002F技术、市场情报和监控模式可以复用核心证据基础设施，同时保留特定模式的方法论。\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--success my-6 rounded-xl border p-5 border-emerald-300 bg-emerald-50 dark:border-emerald-900 dark:bg-emerald-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">这些实现共同展示了什么\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">在这些项目中，可复用的模式不是“一个后端包办一切”。而是\u003Cstrong>关注点分离加上显式契约\u003C\u002Fstrong>：提供者\u002F模型\u002F运行时分离、租户感知授权、凭证边界、可复用数据\u002F检索原语、溯源以及领域特定权威。未来的集成平台将需要在这些能力之间建立稳定的契约，而不是代码库之间的直接耦合。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-82\">当前架构指南如何支持这一平台范围\u003C\u002Fh2>\n\u003Cp>ISO\u002FIEC\u002FIEEE 42010:2022 为软件、系统、企业及相关实体的架构描述提供了一般性规范。它并未定义 AI 平台架构师，但它强化了表达架构关注点、关系和视角的必要性，而不是将架构简化为技术清单。\u003C\u002Fp>\n\u003Cp>NIST AI RMF 1.0 和生成式 AI 配置文件将 AI 风险管理框定为贯穿生命周期，而不仅仅是在模型选择时。因此，治理、映射、测量和管理与一个平台架构兼容，该架构在许多消费工作负载中承载共享控制和证据。\u003C\u002Fp>\n\u003Cp>微软当前的 AI 工作负载指南将应用程序设计、数据、安全、运营、测试\u002F评估和 GenAIOps 视为相互关联的架构领域。其当前的 AI 网关指南还展示了实际的平台关注点，例如集中式模型访问、项目特定的令牌限制、配额和多团队隔离。\u003C\u002Fp>\n\u003Cp>AWS 当前的生成式 AI 透镜和多租户平台场景同样将基础平台控制与消费应用程序所有权分开。AWS 明确指出，中央平台可以强制执行共享护栏和可审计性，而数据质量和工作负载特定的可观测性仍然是消费应用程序或数据生产者的责任。\u003C\u002Fp>\n\u003Cp>供应商产品各不相同，但跨来源的模式是稳定的：生产 AI 平台必须协调身份、数据访问、模型、策略、评估、可观测性、容量、成本和生命周期。GPU 集群或模型端点仅覆盖该责任的一部分。\u003C\u002Fp>\n\u003Ch2 id=\"section-88\">常见误解\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">误解\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">为什么它是错的\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“AI 平台就是 GPU 集群。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">计算是一种基础。平台还需要身份、模型访问、数据、策略、评估、可观测性和生命周期的契约。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“AI 网关只是一个反向代理。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">它还可能承载模型路由、令牌配额、成本归属、策略执行、身份和 AI 特定的遥测。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“共享意味着全局共享。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">服务可以在物理上共享，同时在逻辑上按租户、应用程序、区域、分类或风险级别进行分段。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“一个中央向量数据库成为公司真相。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">向量存储或检索服务是基础设施。领域权威、新鲜度、溯源和访问仍然是独立的关注点。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“平台评估取代解决方案评估。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">一般回归和遥测无法定义特定领域的答案或行动是否可接受。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“提供者抽象应隐藏所有差异。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">某些差异是实质性能力、安全语义或故障模式，必须保持可见。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“RBAC 解决了多租户。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">RBAC 控制操作；租户隔离控制资源边界。两者都可能需要。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“AI 平台架构师只是 MLOps 的另一个名称。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">MLOps\u002FLLMOps 是一个主要的重叠学科，但共享应用程序\u002F运行时、身份、网关、检索和工具边界可以超越模型生命周期操作。\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-90\">AI 平台架构师应预防的故障模式\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">故障模式\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">架构后果\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">每个团队存储自己的提供商密钥\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">重复的密钥处理、不一致的轮换以及更大的影响范围。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">提供商抽象隐藏了所需能力\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">消费者无法使用他们需要的功能，或在不知情的情况下收到与假设不同的行为。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">共享检索忽略租户\u002F用户上下文\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">在应用程序有机会过滤结果之前，就可能发生跨边界数据泄露。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">回退静默更改提供商或位置\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">成本、合规性、数据位置和输出质量可能在调用方不知情的情况下发生变化。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">代理工具通过模型选择授予\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">一个能力强的模型变得权限过高，因为运行时权限没有被独立强制执行。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">所有提示\u002F响应默认记录日志\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">可观测性可能创建新的敏感数据存储库和合规问题。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">平台拥有一个通用质量分数\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">领域故障仍然隐藏在平台健康指标背后。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">平台能力没有版本契约\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型\u002F提供商\u002F运行时变更会不可预测地破坏消费者。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">所有与AI相关的内容都集中化\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">平台成为瓶颈和单体，而不是可复用的能力层。\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-92\">实用的平台架构决策序列\u003C\u002Fh2>\n\u003Csection class=\"editorjs-process my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">从平台需求到可操作的共享能力\u003C\u002Fh3>\u003Cdiv class=\"grid grid-cols-1 md:grid-cols-2 xl:grid-cols-3 gap-4\">\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">1\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">1. 识别真实消费者\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">列出将消费该平台的解决方案、团队、租户和工作负载；避免为假设的复用构建平台。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">2\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">2. 定义共享边界\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">将横切机制与解决方案特定的领域权限、工作流和验收分开。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">3\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">3. 首先定义身份和隔离\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">在共享检索或工具能力之前，建立用户、服务、应用程序、租户、区域和数据分类。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">4\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">4. 定义能力契约\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">指定模型\u002F提供商、检索、代理\u002F工具、网关和遥测API，并明确所有权和版本控制。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">5\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">5. 决定提供商和运行时策略\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">选择托管、自托管、本地或混合执行，并记录回退、位置和能力语义。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">6\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">6. 设计数据和检索边界\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">定义来源、授权传播、语料库所有权、索引和证据责任。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">7\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">7. 添加配额、密钥和策略\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">控制成本、容量、凭证、工具权限、安全控制和影响范围。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">8\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">8. 构建评估和可观测性契约\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">提供平台指标和追踪，同时将领域真实基准和验收留给解决方案。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">9\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">9. 定义生命周期和运维\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">对能力进行版本控制、测试升级、记录弃用、回滚、事件、容量和消费者入门。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">10\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">10. 用多个消费者验证\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">当共享能力实际上服务于不同的工作负载，而不强迫它们进入同一领域模型时，平台声明才变得可信。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-94\">边缘情况和角色限制\u003C\u002Fh2>\n\u003Cp>只有一个AI应用程序的小型组织可能不需要独立的AI平台或平台架构师。过早平台化可能产生比价值更多的抽象。正确的架构可能是一个设计良好的解决方案，带有几个可复用模块。\u003C\u002Fp>\n\u003Cp>气隙或主权部署会显著改变提供商、更新和可观测性模型。模型托管、工件分发、身份集成和遥测导出可能都需要本地等效方案。\u003C\u002Fp>\n\u003Cp>高度监管或高后果的工作负载可能需要更强的物理或组织隔离，而不是逻辑共享平台。复用从来不是削弱所需安全边界的充分理由。\u003C\u002Fp>\n\u003Cp>托管云AI服务可以减轻实现负担，但不会消除架构责任。组织仍然决定身份、数据访问、日志记录、保留、配额、模型资格、回退、评估和解决方案验收。\u003C\u002Fp>\n\u003Cp>平台边界也可能因模态而异。文本推理、多模态生成、语音、计算机使用和自主代理即使共享提供商和身份基础设施，也可能有不同的延迟、数据、权限和可观测性要求。\u003C\u002Fp>\n\u003Ch2 id=\"section-100\">什么会改变这个答案？\u003C\u002Fh2>\n\u003Cp>如果组织范围发生变化，核心定义也会改变。如果架构师负责一个工作负载，角色就更接近AI解决方案架构师。如果职责扩展到组织范围内的能力战略、投资、标准和目标状态组合，则转向企业AI架构。\u003C\u002Fp>\n\u003Cp>每当提供商、网关产品、代理协议、监管义务、模型能力或部署约束发生变化时，实施指南就会改变。这就是为什么平台架构应该将稳定的职责和契约与当前的供应商机制分开表达。\u003C\u002Fp>\n\u003Ch2 id=\"section-103\">AI平台架构师检查清单\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">问题\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">预期答案\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">谁是实际的平台消费者？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">具有不同但重叠需求的命名解决方案、团队或租户上下文。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">什么是真正共享的？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">明确的能力列表，而不是模糊的“AI后端”。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">什么必须保持解决方案特定？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">领域权限、业务工作流、任务验收和其他工作负载拥有的关注点。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型\u002F提供商如何表示？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">带能力版本化的提供商\u002F模型契约和明确的回退语义。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">身份如何传播？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">用户\u002F服务\u002F应用程序\u002F租户上下文在每条特权请求路径中得以保留。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">租户隔离如何强制执行？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">资源作用域与角色权限检查分开。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">密钥如何处理？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">特权存储、轮换、有限暴露和可审计的所有权。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">检索如何保持权限？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">共享机制与授权、来源和领域拥有的证据规则。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">工具和代理如何受到约束？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">运行时权限、有界工具契约、审批、取消和可追溯性。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">成本和容量如何控制？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">配额、令牌\u002F速率控制、使用归因和过载行为。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">质量如何衡量？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">平台回归\u002F评估加上解决方案特定的真实基准和验收。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">变更如何推出？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">版本控制、兼容性、迁移、弃用、回滚和事件所有权。\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-105\">结论\u003C\u002Fh2>\n\u003Cp>AI平台架构师负责\u003Cstrong>AI能力与消费它们的解决方案之间\u003C\u002Fstrong>的可复用架构。该角色定义模型、提供商、检索、代理、工具、身份、租户、密钥、评估、可观测性、配额和运行时操作如何成为可靠的平台服务，而不是重复的一次性集成。\u003C\u002Fp>\n\u003Cp>困难的部分不是最大化复用，而是选择正确的边界。强大的平台在多个消费者真正受益的地方标准化机制、策略和操作，同时保留解决方案特定的数据权限、业务逻辑、安全要求和验收标准。\u003C\u002Fp>\n\u003Cp>这种区别也解释了与AI解决方案架构的关系：\u003Cstrong>解决方案架构师使一个AI赋能的系统适合其目的；平台架构师使共享的AI能力在许多此类系统中安全、可复用、可操作和可演进。\u003C\u002Fstrong>\u003C\u002Fp>\n\u003Ch2 id=\"section-109\">相关规范知识\u003C\u002Fh2>\n\u003Cp>本文位于生成式AI组件、ADR与NFR、以及AI解决方案架构的规范基础之后。这些概念是前提条件，因为平台的存在是为了提供可复用的系统能力，并针对明确的质量和运营需求编码架构决策。\u003C\u002Fp>\n\u003Cp>检索增强生成是可能通过平台提供的能力的一个例子，但平台不应将检索基础设施、领域知识和答案有效性混为一谈。\u003C\u002Fp>\n\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">什么是RAG？其工作原理的最简解释\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">检索增强生成的规范介绍，以及模型生成与外部知识检索之间的边界。\u003C\u002Fp>\u003C\u002Fa>\n\u003Cp>代理协议、租户隔离、AI治理、模型路由、上下文工程和MLOps\u002FLLMOps是下游或相邻的知识节点。一旦平台边界明确，它们就更容易推理。\u003C\u002Fp>\n\u003Ch2 id=\"section-114\">常见问题\u003C\u002Fh2>\n\u003Csection class=\"editorjs-faq my-6 rounded-xl border border-gray-200 p-5 dark:border-gray-700\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">AI平台架构师FAQ\u003C\u002Fh3>\u003Cdiv id=\"faq-1\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">AI平台架构师与AI解决方案架构师相同吗？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">不同。解决方案架构师专注于一个具体的AI赋能解决方案。平台架构师专注于可复用的AI能力、控制和运营契约，以支持多个解决方案。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq-2\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">AI平台需要托管自己的模型吗？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">不需要。平台可以使用托管云模型、自托管模型、本地推理或混合策略。架构必须明确提供商、位置、身份、路由、数据和运营后果。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq-3\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">AI网关足以成为AI平台吗？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">通常不够。网关可以是重要的平台组件，但完整的平台还需要身份、密钥、数据\u002F检索、评估、可观测性、生命周期和运营所有权的契约。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq-4\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">检索应该集中化吗？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">检索机制通常可以共享，但领域权威、授权、新鲜度、证据充分性和语料库所有权应保持明确。共享基础设施并不意味着共享真相。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq-5\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">平台评估能替代应用评估吗？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">不能。平台评估可以测试共享能力和回归。每个解决方案仍然需要特定任务的真实基准、验收标准和领域质量阈值。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq-6\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">多租户只是RBAC吗？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">不是。RBAC决定身份可以做什么。租户隔离决定身份可以对哪个租户的资源进行操作。平台通常需要两者。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-116\">术语表\u003C\u002Fh2>\n\u003Csection class=\"editorjs-glossary my-6 rounded-xl border border-gray-200 dark:border-gray-700 p-5\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">关键AI平台架构术语\u003C\u002Fh3>\u003Cdl>\u003Cdiv id=\"ai-platform\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">AI平台\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">由多个应用、团队或租户上下文消费的一组可复用的AI相关技术和运营能力。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"ai-gateway\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">AI网关\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">AI端点的网关层，可在基本代理之外添加认证、路由、配额、策略、重试、成本归属和AI特定遥测。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"provider-adapter\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">提供商适配器\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">将平台契约映射到模型提供商的API、能力、健康状况和故障语义的组件。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"tenant-isolation\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">租户隔离\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">防止一个租户上下文访问另一个租户资源的边界，独立于角色权限。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"capability-contract\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">能力契约\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">描述共享平台服务提供什么以及消费者必须提供或拥有什么的版本化接口和行为协议。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"grounding-service\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">接地\u002F检索服务\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">为AI工作负载查找和提供外部信息的共享机制；它不会自动定义哪些信息对某个领域是权威的。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"evaluation-harness\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">评估工具\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">用于运行测试、数据集、模型\u002F提示版本和指标的可复用基础设施；领域验收仍然特定于解决方案。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"control-plane\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">控制平面\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">管理平台能力、身份、策略、配额、版本和部署状态的配置和治理层。\u003C\u002Fdd>\u003C\u002Fdiv>\u003C\u002Fdl>\u003C\u002Fsection>\n\u003Ch2 id=\"section-118\">主要来源和当前架构指南\u003C\u002Fh2>\n\u003Cp>以下来源支持一般架构和生产平台声明。Aaasaasa AI客户端、Aaasaasa AI CMS和真相源研究引擎部分明确为原创实现证据。当前状态的外部参考已于2026年10月8日检查。\u003C\u002Fp>\n\u003Ca href=\"https:\u002F\u002Fwww.iso.org\u002Fstandard\u002F74393.html\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">ISO\u002FIEC\u002FIEEE 42010:2022 — 架构描述\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">当前发布的架构描述概念和关系的国际标准。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fwww.nist.gov\u002Fitl\u002Fai-risk-management-framework\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">NIST AI风险管理框架\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">NIST的AI RMF资源和当前状态；截至2026年10月，AI RMF 1.0正在修订中。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fwww.nist.gov\u002Fpublications\u002Fartificial-intelligence-risk-management-framework-generative-artificial-intelligence\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">NIST AI 600-1 — 生成式AI概况\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">在整个AI生命周期中应用AI风险管理考虑的生成式AI概况。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Flearn.microsoft.com\u002Fen-us\u002Fazure\u002Fwell-architected\u002Fai\u002Fget-started\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">Microsoft Azure Well-Architected — AI工作负载\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">涵盖AI应用、数据、运营、评估、负责任AI和生命周期问题的当前架构指南。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Flearn.microsoft.com\u002Fen-us\u002Fazure\u002Fwell-architected\u002Fai\u002Fdesign-principles\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">Microsoft — AI工作负载设计原则\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">关于身份分段、安全边界、遥测、性能、数据和平台权衡的当前指南。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Flearn.microsoft.com\u002Fen-us\u002Fazure\u002Fai-foundry\u002Fconfiguration\u002Fenable-ai-api-management-gateway-portal?view=foundry\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">Microsoft Foundry — AI网关架构\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">关于共享项目访问、令牌遏制、配额和治理的当前AI网关指南。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Flearn.microsoft.com\u002Fen-us\u002Fazure\u002Farchitecture\u002Fai-ml\u002Fguide\u002Fazure-openai-gateway-guide\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">Azure架构中心 — 通过网关访问模型\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">关于集中模型访问、路由、限流、故障转移和客户端\u002F平台责任的架构指南。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fdocs.aws.amazon.com\u002Fwellarchitected\u002Flatest\u002Fgenerative-ai-lens\u002F\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">AWS Well-Architected — 生成式 AI 透镜\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">面向生成式 AI 工作负载的当前生产架构指导，涵盖安全性、可靠性、运维、性能和成本。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fdocs.aws.amazon.com\u002Fwellarchitected\u002Flatest\u002Fgenerative-ai-lens\u002Fmulti-tenant-generative-ai-platform-scenario.html\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">AWS — 多租户生成式 AI 平台场景\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">当前示例，将中央平台控制与可审计性，同消费应用的数据质量及工作负载特定职责区分开来。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fdocs.aws.amazon.com\u002Fwellarchitected\u002Flatest\u002Fagentic-ai-lens\u002Fdesign-principles.html\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">AWS Well-Architected — 代理式 AI 设计原则\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">关于受限代理权限、可追溯性、版本化行为、显式契约和人工监督的当前指导。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fdocs.aws.amazon.com\u002FAmazonCloudWatch\u002Flatest\u002Fmonitoring\u002FGenAI-observability.html\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">AWS CloudWatch — 生成式 AI 可观测性\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">针对模型、代理、知识库、工具以及成本\u002F延迟\u002F错误分析的当前可观测性能力和生产指标。\u003C\u002Fp>\u003C\u002Fa>",{"time":212,"blocks":213,"version":1201},1791477707099,[214,219,226,232,237,244,248,252,256,300,304,308,312,316,341,345,349,353,357,388,394,398,402,406,410,416,420,424,428,432,436,440,444,448,452,456,460,464,468,472,476,480,484,488,492,496,500,504,508,512,516,520,524,528,532,536,541,561,565,569,603,607,655,659,682,686,690,695,699,703,707,711,733,737,741,745,749,753,757,761,765,770,774,778,782,786,790,794,798,829,833,867,871,906,910,914,918,922,926,930,934,938,942,946,989,993,997,1001,1005,1009,1013,1017,1027,1031,1035,1064,1068,1105,1109,1113,1121,1129,1137,1145,1153,1161,1169,1177,1185,1193],{"id":215,"data":216,"type":218},"intro",{"text":217},"\u003Cstrong>AI 平台架构师\u003C\u002Fstrong>设计可复用的 AI 基础，使多个应用、团队或租户上下文能够通过它访问模型、数据与检索、智能体与工具运行时、身份与权限、评估、可观测性、配额、密钥以及部署能力。该角色的范围比基础设施更广，但比拥有每一个 AI 赋能产品更窄：其核心职责是决定\u003Cstrong>哪些应当共享、共享能力如何治理与隔离，以及哪些必须保持为解决方案专属\u003C\u002Fstrong>。","paragraph",{"id":220,"data":221,"type":225},"direct",{"body":222,"title":223,"variant":224},"\u003Cstrong>AI 平台架构师为 AI 系统设计共享的技术与运营基座。\u003C\u002Fstrong>该角色不是为某一个助手或某一个工作流做架构，而是为模型\u002F提供商访问、网关与路由、检索服务、智能体运行时、工具访问、身份与租户隔离、密钥、评估、遥测、部署与生命周期管理定义可复用的契约与边界。","直接回答","info","callout",{"id":227,"data":228,"type":225},"term-note",{"body":229,"title":230,"variant":231},"\u003Cstrong>AI 平台架构师是一个实用的角色标签，并非普遍标准化的职位名称。\u003C\u002Fstrong>ISO\u002FIEC\u002FIEEE 42010:2022 定义的是架构描述的概念，而非这一角色。不同组织可能会将这些职责拆分给平台架构师、解决方案架构师、企业架构师、安全架构师、MLOps\u002FLLMOps 专家以及平台工程团队。本文使用该术语来指代对可复用 AI 平台层的架构职责。","术语说明","note",{"id":233,"data":234,"type":225},"version-note",{"body":235,"title":236,"variant":231},"此处稳定的架构原则与供应商无关。当前的 Microsoft、AWS 和 NIST 指南被用作外部实现与治理证据。NIST 表示 AI RMF 1.0 正在修订；供应商平台功能、网关产品、智能体运行时和模型能力的演进速度快于架构原则，因此对版本敏感的实现选择必须在部署前重新核查。","当前来源说明 — 2026 年 10 月 8 日",{"id":238,"data":239,"type":243},"toc",{"title":240,"maxLevel":241,"minLevel":242},"目录",3,2,"tableOfContents",{"id":245,"data":246,"type":42},"h-meaning",{"text":247,"level":242},"AI 平台架构师实际上架构什么？",{"id":249,"data":250,"type":218},"p-meaning-1",{"text":251},"工作的对象是\u003Cstrong>平台\u003C\u002Fstrong>：一组共享能力，在减少重复集成工作的同时，保留明确的安全、数据和运营边界。平台可以向众多消费方解决方案暴露模型访问、提供商适配器、检索原语、智能体执行、工具代理、策略执行、评估、遥测和部署服务。",{"id":253,"data":254,"type":218},"p-meaning-2",{"text":255},"平台的价值并不只是因为组件被集中化。当消费方获得具有清晰契约、所有权、隔离、可观测性和生命周期规则的稳定能力时，平台才有价值。因此，关键的架构问题不是“每个人都应该使用哪个模型？”，而是\u003Cstrong>“哪些职责可以被安全地标准化和复用，同时不抹去每个解决方案的需求？”\u003C\u002Fstrong>。",{"id":257,"data":258,"type":299},"solution-vs-platform",{"rows":259,"title":290,"layout":291,"columns":292},[260,266,272,278,284],{"id":261,"label":262,"values":263},"c1","主要范围",{"platform":264,"solution":265},"Reusable AI capabilities consumed by multiple solutions, teams or tenant contexts.","One concrete AI-enabled product, workflow or application.",{"id":267,"label":268,"values":269},"c2","核心问题",{"platform":270,"solution":271},"Which shared capabilities and controls should solutions consume, and where must solution-specific ownership remain?","How should this solution meet its business, data, security, quality and operational requirements?",{"id":273,"label":274,"values":275},"c3","数据权威",{"platform":276,"solution":277},"Provides storage, retrieval, provenance or access primitives without automatically becoming the authority for every domain.","Defines which domain data is authoritative and how the solution may use it.",{"id":279,"label":280,"values":281},"c4","评估",{"platform":282,"solution":283},"Provides reusable evaluation, telemetry and release mechanisms; it cannot define every domain's success threshold.","Defines task-specific quality and acceptance criteria.",{"id":285,"label":286,"values":287},"c5","生命周期",{"platform":288,"solution":289},"Owns shared capability versions, compatibility, onboarding, quotas, policy and operational contracts.","Owns the lifecycle of the specific workload.","解决方案架构与平台架构解决不同范围的问题","table",[293,296],{"id":294,"label":295},"solution","AI 解决方案架构师",{"id":297,"label":298},"platform","AI 平台架构师","comparison",{"id":301,"data":302,"type":42},"h-simple",{"text":303,"level":242},"最简单的例子",{"id":305,"data":306,"type":218},"p-simple-1",{"text":307},"设想一个组织有五个 AI 赋能产品：一个内部文档助手、一个客户支持副驾驶、一个软件工程智能体、一个合同审查工作流和一个产品搜索助手。每个产品都可以独立集成模型 API、保存凭据、实现重试、收集令牌指标、创建检索代码并构建自己的工具权限。",{"id":309,"data":310,"type":218},"p-simple-2",{"text":311},"当每个团队都发明不同的安全和运营模型时，这种重复既昂贵又危险。共享平台则可以提供经批准的提供商连接、模型发现、配额、凭据、租户感知访问、通用遥测、可复用检索服务以及智能体\u002F工具运行时契约。",{"id":313,"data":314,"type":218},"p-simple-3",{"text":315},"但平台必须在正确的边界处停止。合同审查解决方案可能需要法律文档权威和引用规则，而软件智能体不需要。产品搜索助手可能需要特定于商务的新鲜度和授权规则。\u003Cstrong>可复用基础设施并不会让所有领域事实都变得可复用。\u003C\u002Fstrong>",{"id":317,"data":318,"type":340},"simple-flow",{"steps":319,"title":338,"orientation":339},[320,323,326,329,332,335],{"label":321,"description":322},"1. 消费方标识自身","调用应用、用户、服务、团队或租户通过经过身份验证的身份和明确范围进入。",{"label":324,"description":325},"2. 平台策略生效","网关和策略层确定允许的提供商、模型、配额、数据路径、工具和执行模式。",{"label":327,"description":328},"3. 共享能力执行","请求可能使用推理、检索、智能体运行时、工具访问或其他可复用平台服务。",{"label":330,"description":331},"4. 解决方案专属上下文保持权威","消费方解决方案提供领域规则、用户意图、数据权威、任务特定约束和验收逻辑。",{"label":333,"description":334},"5. 遥测和证据被捕获","平台记录身份、路由、模型\u002F提供商、延迟、成本、错误、工具活动以及其他允许的可观测性信号。",{"label":336,"description":337},"6. 结果在解决方案契约下返回","解决方案仍然负责判断输出对其用户和领域是否可接受。","共享 AI 请求路径","auto","processFlow",{"id":342,"data":343,"type":42},"h-stops",{"text":344,"level":242},"简单例子止步之处",{"id":346,"data":347,"type":218},"p-stops-1",{"text":348},"集中化并不自动等于架构。在多个模型 API 前放置一个端点是有用的，但它本身并不会创建 AI 平台。生产平台还需要身份边界、能力契约、提供商健康与生命周期处理、配额、密钥所有权、可观测性、兼容性规则、安全控制、发布纪律和明确的运营责任。",{"id":350,"data":351,"type":218},"p-stops-2",{"text":352},"相反的失败也很常见：把每个提示词、向量索引、业务规则、智能体和应用工作流都放进一个“AI 后端”。这会创建一个单体，其共享状态是偶然的而非架构性的。\u003Cstrong>平台应当标准化横切能力，而不是仅仅因为涉及 AI 就吸收领域所有权。\u003C\u002Fstrong>",{"id":354,"data":355,"type":42},"h-boundary",{"text":356,"level":242},"最重要的平台决策：共享与解决方案专属",{"id":358,"data":359,"type":291},"shared-boundary-table",{"content":360,"stretched":43,"withHeadings":14},[361,365,369,373,377,381,384],[362,363,364],"能力领域","适合由共享平台负责","通常仍属于解决方案特定范围",[366,367,368],"模型访问","已批准的提供商连接、适配器、凭据、健康检查、路由原语、配额","任务特定的模型验收、提示行为、质量阈值",[370,371,372],"检索","摄取原语、提取、索引、搜索 API、来源契约、授权钩子","权威语料库、新鲜度规则、领域元数据、证据充分性",[374,375,376],"智能体与工具","运行时生命周期、工具注册表\u002F代理、权限执行、追踪、取消","业务工作流、允许的操作语义、升级策略、任务成功",[378,379,380],"安全","身份集成、密钥存储、策略执行、审计契约、租户隔离机制","数据分类、业务授权规则、领域特定风险接受",[280,382,383],"测试框架、数据集\u002F版本机制、遥测、实验\u002F发布工作流","基准真值、领域测试集、验收阈值、用户结果",[385,386,387],"运维","部署模式、健康检查、指标、事件集成、容量控制","存在差异时的解决方案 SLO、业务连续性影响、工作负载特定运行手册",{"id":389,"data":390,"type":225},"boundary-principle",{"body":391,"title":392,"variant":393},"\u003Cstrong>在复用真实存在的地方共享机制和控制；将权威和验收保留在领域拥有它们的地方。\u003C\u002Fstrong> 这可以防止两种相反的错误：到处重复建设基础设施，以及中央平台错误地成为每个应用的数据、策略和质量的所有者。","平台原则","success",{"id":395,"data":396,"type":42},"h-responsibility-map",{"text":397,"level":242},"架构职责映射",{"id":399,"data":400,"type":42},"h-provider",{"text":401,"level":241},"1. 模型与提供商访问",{"id":403,"data":404,"type":218},"p-provider-1",{"text":405},"平台架构师定义消费者如何发现和调用模型，而不强迫每个应用硬编码某一个提供商。这包括提供商适配器、模型标识符、能力元数据、身份验证、健康检查、端点配置、请求规范化和兼容行为。",{"id":407,"data":408,"type":218},"p-provider-2",{"text":409},"提供商抽象必须保持诚实。不同提供商暴露不同的上下文限制、工具语义、结构化输出行为、多模态能力、安全控制、缓存、定价和故障模式。好的抽象会创建稳定的平台契约，同时保留对无法被有意义地扁平化的能力的访问。",{"id":411,"data":412,"type":225},"provider-warning",{"body":413,"title":414,"variant":415},"最低共同标准的 API 可以让迁移更容易，但也可能抹去重要的能力。架构应定义哪些功能可移植、哪些是提供商特定的，以及消费者如何发现这种差异。","不要将抽象与假装提供商完全相同混为一谈","warning",{"id":417,"data":418,"type":42},"h-gateway",{"text":419,"level":241},"2. 网关、路由、配额与成本控制",{"id":421,"data":422,"type":218},"p-gateway-1",{"text":423},"共享 AI 网关可以集中处理身份验证、路由、限流、重试、令牌限制、使用归因和策略执行。Microsoft 当前的 AI Gateway 指南明确将每分钟令牌限制、配额和多项目隔离视为平台关注点；AWS 同样暴露账户和模型配额以及集中控制。",{"id":425,"data":426,"type":218},"p-gateway-2",{"text":427},"因此，当网关承载 AI 特定策略和运维语义时，它就不只是反向代理。但它不应静默地做出业务决策。路由策略可能偏好健康的本地模型、成本更低的提供商或符合区域合规的端点；该路由对特定任务是否可接受，仍然是平台与解决方案之间的契约。",{"id":429,"data":430,"type":218},"p-gateway-3",{"text":431},"路由还需要故障语义。如果首选模型不可用，平台必须知道是否允许回退、云路由是否需要明确同意、能力较低的模型是否有效，以及该决策如何暴露给可观测性。",{"id":433,"data":434,"type":42},"h-data",{"text":435,"level":241},"3. 共享数据、检索与接地服务",{"id":437,"data":438,"type":218},"p-data-1",{"text":439},"检索服务是强有力的平台候选，因为解析、分块、索引、词法搜索、语义搜索、元数据过滤、来源和引用机制都可复用。然而，平台不能将共享检索引擎与共享事实来源混为一谈。",{"id":441,"data":442,"type":218},"p-data-2",{"text":443},"解决方案仍然拥有以下问题：哪个语料库是权威的？哪个版本有效？此用户能否看到此文档？数据必须有多新鲜？什么算作充分证据？检索失败时能否生成答案？即使平台提供检索机制，这些也是领域和解决方案需求。",{"id":445,"data":446,"type":218},"p-data-3",{"text":447},"这种边界在多租户系统中尤为重要。技术上共享的索引或向量服务并不证明跨租户可见性合理。授权上下文必须在检索过程中保留，而不是仅在搜索结果已经跨越边界之后才添加。",{"id":449,"data":450,"type":42},"h-agent-runtime",{"text":451,"level":241},"4. 智能体与工具运行时",{"id":453,"data":454,"type":218},"p-agent-1",{"text":455},"智能体系统增加了可复用的运行时关注点：线程\u002F会话生命周期、规划循环、工具注册、工具调用、取消、超时、人工审批、内存\u002F状态接口、远程智能体协议和追踪关联。平台可以提供这些机制，使每个产品无需重新构建它们。",{"id":457,"data":458,"type":218},"p-agent-2",{"text":459},"平台还必须将工具权限与模型能力分开。模型能够生成 shell 命令并不意味着运行时应允许 shell 执行。权限边界属于应用\u002F运行时架构，并且必须独立于模型可执行。",{"id":461,"data":462,"type":218},"p-agent-3",{"text":463},"当前 AWS Agentic AI 指南强调有界代理、明确授权、端到端追踪、版本化行为工件以及与后果相称的人类监督。这些都是平台赋能关注点，但消费方解决方案仍然定义其领域内哪些操作是合法的。",{"id":465,"data":466,"type":42},"h-identity",{"text":467,"level":241},"5. 身份、租户隔离与授权",{"id":469,"data":470,"type":218},"p-identity-1",{"text":471},"AI 平台通常位于高价值模型、专有数据和具备操作能力的工具之前。因此，身份验证只是开始。架构必须在每个需要它的特权操作中携带用户、服务、应用和租户上下文。",{"id":473,"data":474,"type":218},"p-identity-2",{"text":475},"\u003Cstrong>RBAC 和租户隔离解决不同的问题。\u003C\u002Fstrong>RBAC 回答一个身份可以做什么；租户隔离回答该身份可以对哪个租户的资源进行操作。一个检查角色但丢失租户上下文的平台仍然可能暴露错误的数据。",{"id":477,"data":478,"type":218},"p-identity-3",{"text":479},"Microsoft 当前的 AI 工作负载指南明确建议身份分段和授权感知的内容访问。AWS 的多租户生成式 AI 平台指南同样将逻辑隔离、集中控制和可审计性视为平台关注点。",{"id":481,"data":482,"type":42},"h-secrets",{"text":483,"level":241},"6. 密钥、凭证与信任边界",{"id":485,"data":486,"type":218},"p-secrets-1",{"text":487},"平台应定义谁拥有提供商密钥、远程持有者令牌、签名材料和工具凭证，它们存储在哪里，哪个进程可以访问它们，如何轮换它们，以及它们是否可能到达浏览器或不受信任的渲染器。",{"id":489,"data":490,"type":218},"p-secrets-2",{"text":491},"这是一个架构边界，而不是实现细节。如果每个消费方应用都将提供商凭证复制到自己的配置中，组织就重复了运营负担和爆炸半径。只有当平台本身具有更窄、可审计的访问路径时，集中化才能降低这种风险。",{"id":493,"data":494,"type":42},"h-eval",{"text":495,"level":241},"7. 评估、可观测性与可审计性",{"id":497,"data":498,"type":218},"p-eval-1",{"text":499},"可复用平台可以提供评估工具、追踪 ID、模型\u002F提供商元数据、令牌和成本指标、延迟、错误率、提示\u002F模型版本关联、代理\u002F工具追踪以及受控日志记录。AWS 和 Microsoft 都将可观测性和评估视为 AI 工作负载的核心生产关注点。",{"id":501,"data":502,"type":218},"p-eval-2",{"text":503},"平台评估和解决方案评估必须保持分离。平台可以验证端点是否健康、模型版本是否通过通用回归套件以及追踪是否完整。如果没有领域特定的基准真相和验收标准，它无法判定法律答案、医疗工作流或产品推荐是否可接受。",{"id":505,"data":506,"type":218},"p-eval-3",{"text":507},"日志记录还创建了隐私边界。提示和响应日志可能包含敏感或专有数据。因此，平台架构师必须决定记录什么、脱敏什么、采样什么、保留什么以及可访问什么，而不是假设更多遥测总是更安全。",{"id":509,"data":510,"type":42},"h-runtime",{"text":511,"level":241},"8. 运行时、部署与位置",{"id":513,"data":514,"type":218},"p-runtime-1",{"text":515},"平台架构师决定共享 AI 能力如何部署和访问：托管云服务、自托管端点、本地推理、混合路由、容器化服务、桌面运行时、私有网络或气隙环境。重要的区别在于\u003Cstrong>控制\u002F运行时进程在哪里运行\u003C\u002Fstrong>以及\u003Cstrong>推理和数据处理实际发生在哪里\u003C\u002Fstrong>。",{"id":517,"data":518,"type":218},"p-runtime-2",{"text":519},"本地客户端仍然可能调用云模型。云控制平面可能路由到本地模型。远程代理可能在客户网络内执行工具。因此，架构图必须显示信任和数据流边界，而不是将“本地”和“云”用作模糊的标签。",{"id":521,"data":522,"type":42},"h-lifecycle",{"text":523,"level":241},"9. 平台生命周期、兼容性与入门",{"id":525,"data":526,"type":218},"p-lifecycle-1",{"text":527},"只有当消费方能够长期依赖可复用能力时，它才成为平台。这需要版本化契约、迁移规则、兼容性策略、弃用、发布测试、回滚、事件所有权、容量规划、文档以及新团队或应用的入门路径。",{"id":529,"data":530,"type":218},"p-lifecycle-2",{"text":531},"快速发展的 AI 生态系统使这一点尤为重要。模型名称、SDK、协议版本、提供商 API 和安全能力各自独立变化。平台必须吸收其中一些波动性，同时不隐藏对解决方案行为有重大影响的变更。",{"id":533,"data":534,"type":42},"h-control-plane",{"text":535,"level":242},"一个实用的控制平面\u002F执行平面\u002F解决方案平面模型",{"id":537,"data":538,"type":225},"model-note",{"body":539,"title":540,"variant":231},"下面的三平面模型是一种实用的责任推理方式；它不是 ISO、NIST、Microsoft 或 AWS 标准。其目的是明确所有权边界。","提议的架构模型",{"id":542,"data":543,"type":291},"planes-table",{"content":544,"stretched":43,"withHeadings":14},[545,549,553,557],[546,547,548],"平面","典型职责","不应静默拥有",[550,551,552],"平台控制平面","提供商注册表、模型策略、配额、租户配置、身份、密钥、路由规则、能力版本、部署配置","应用业务逻辑或领域真相",[554,555,556],"平台执行\u002F数据平面","推理请求、检索操作、代理\u002F工具执行、提取、索引、遥测发射、策略执行","仅因基础设施共享而进行跨租户访问",[558,559,560],"解决方案平面","用户工作流、提示\u002F指令、权威语料选择、领域授权、业务规则、任务评估与验收","平台明确拥有的低级提供商集成",{"id":562,"data":563,"type":218},"p-control-plane-1",{"text":564},"这种分离有助于诊断平台漂移。如果应用程序必须知道每个提供商特定的凭据和端点，那么平台契约就太薄弱了。如果平台决定哪个客户记录在法律上是权威的，或者某个领域答案是否可接受，那么平台就已经越界进入了解决方案所有权。",{"id":566,"data":567,"type":42},"h-artifacts",{"text":568,"level":242},"AI 平台架构师应该产出什么？",{"id":570,"data":571,"type":291},"artifacts-table",{"content":572,"stretched":43,"withHeadings":14},[573,576,579,582,585,588,591,594,597,600],[574,575],"架构工件","目的",[577,578],"平台能力地图","定义平台提供什么、谁消费它以及哪些能力仍在范围之外。",[580,581],"提供商\u002F模型契约","定义提供商、模型、能力、抽象边界、路由元数据和回退语义。",[583,584],"身份与租户模型","定义用户\u002F服务\u002F应用身份、租户上下文、RBAC\u002FABAC 钩子和资源隔离。",[586,587],"网关与配额策略","定义速率限制、令牌\u002F成本预算、路由控制、重试和容量行为。",[589,590],"检索\u002F数据契约","定义摄取、来源、搜索、元数据、授权传播以及领域权威保留在哪里。",[592,593],"代理\u002F工具契约","定义运行时生命周期、工具注册、权限、审批、取消和跟踪行为。",[595,596],"密钥与信任边界模型","定义凭据所有权、存储、进程边界、轮换和敏感数据路径。",[598,599],"评估与遥测契约","定义通用指标、跟踪、数据集\u002F版本链接、日志策略和解决方案扩展点。",[601,602],"生命周期与兼容性策略","定义版本、迁移、弃用、发布、回滚、事件所有权和入门。",{"id":604,"data":605,"type":42},"h-tradeoffs",{"text":606,"level":242},"工作主要是权衡，而不是最大程度的集中化",{"id":608,"data":609,"type":299},"tradeoff-comparison",{"rows":610,"title":647,"layout":291,"columns":648},[611,617,623,629,635,641],{"id":612,"label":613,"values":614},"t1","提供商抽象",{"pressureA":615,"pressureB":616},"Stable portable platform API","Access to provider-specific capabilities and fast innovation",{"id":618,"label":619,"values":620},"t2","复用",{"pressureA":621,"pressureB":622},"Shared services reduce duplication","Isolation and domain autonomy prevent unsafe coupling",{"id":624,"label":625,"values":626},"t3","治理",{"pressureA":627,"pressureB":628},"Central policy and auditability","Team speed and local experimentation",{"id":630,"label":631,"values":632},"t4","可观测性",{"pressureA":633,"pressureB":634},"Rich traces for debugging and evaluation","Privacy, data minimization and logging cost",{"id":636,"label":637,"values":638},"t5","可用性",{"pressureA":639,"pressureB":640},"Fallback and multi-provider resilience","Predictable quality, compliance and data-location guarantees",{"id":642,"label":643,"values":644},"t6","平台范围",{"pressureA":645,"pressureB":646},"More reusable capabilities","Smaller blast radius and less platform lock-in","常见的平台权衡",[649,652],{"id":650,"label":651},"pressureA","压力 A",{"id":653,"label":654},"pressureB","压力 B",{"id":656,"data":657,"type":42},"h-adjacent",{"text":658,"level":242},"这与相邻角色有何不同？",{"id":660,"data":661,"type":291},"roles-table",{"content":662,"stretched":43,"withHeadings":14},[663,666,668,670,673,676,679],[664,665],"角色","主要架构范围",[295,667],"一个具体的 AI 赋能解决方案及其端到端需求、边界、权衡和生产验收。",[298,669],"跨多个解决方案或团队消费的可复用 AI 能力以及运营\u002F安全契约。",[671,672],"企业架构师","在更广泛层面上的组织级业务\u002F技术组合、能力和治理对齐。",[674,675],"MLOps \u002F LLMOps 架构师或专家","模型和 AI 生命周期、部署、实验、可观测性、发布和运营实践；可能强烈重叠，但不自动拥有整个共享应用平台。",[677,678],"平台工程师 \u002F SRE","实现和运营平台基础设施、可靠性、自动化和开发者体验；架构责任可能与平台架构师共享。",[680,681],"AI \u002F 软件工程师","在商定的架构内实现模型、集成、服务、代理、检索和产品功能。",{"id":683,"data":684,"type":218},"p-adjacent-1",{"text":685},"这些边界是组织性的，不是普遍的。在小型团队中，一个人可能承担多项职责。在受监管的企业中，它们可能分散在架构、安全、平台、数据和运营组中。有用的区别是\u003Cstrong>架构责任的范围\u003C\u002Fstrong>，而不是组织图上印的职位名称。",{"id":687,"data":688,"type":42},"h-evidence",{"text":689,"level":242},"实现证据：这些平台边界如何出现在我自己的工作中",{"id":691,"data":692,"type":225},"evidence-note",{"body":693,"title":694,"variant":231},"以下部分描述了我自己项目中的具体模式。它们是这些架构边界已在真实代码和项目系统中实现或明确设计的证据。它们\u003Cstrong>不是\u003C\u002Fstrong>声称这些项目一起已经构成商业部署的企业 AI 平台。","原始实现证据",{"id":696,"data":697,"type":42},"h-ai-client",{"text":698,"level":241},"Aaasaasa AI Client：提供商、运行时和权限分离",{"id":700,"data":701,"type":218},"p-ai-client-1",{"text":702},"Aaasaasa AI Client 是一个使用 Nuxt 4、Electron 和 TypeScript 构建的本地优先桌面 AI 工作区。其 AI Hub 有意分离\u003Cstrong>代理\u002F客户端、提供商、模型、连接\u002F运行时位置、权限和 Web 客户端\u003C\u002Fstrong>，而不是将它们视为一个配置值。",{"id":704,"data":705,"type":218},"p-ai-client-2",{"text":706},"该实现包括直接提供商适配器、Codex 代理运行时集成、本地 Ollama\u002FLM Studio 路径、OpenAI 兼容服务、集中式工作区权限、主进程凭据存储、DuckDB、Qdrant\u002F向量支持、PDF\u002F可读性提取以及基于 MCP 的认证目录访问。",{"id":708,"data":709,"type":218},"p-ai-client-3",{"text":710},"两个平台经验尤其相关。首先，本地运行时与本地推理不同：本地 Codex 进程仍然可以使用云模型。其次，自动路由不会静默地从本地回退到付费云推理。这使得路由策略和运行时局部性变得明确，而不是从 UI 标签推断。",{"id":712,"data":713,"type":291},"ai-client-evidence-table",{"content":714,"stretched":43,"withHeadings":14},[715,718,721,724,727,730],[716,717],"已实现的边界","平台架构含义",[719,720],"代理 vs 提供商 vs 模型","不同的职责可以独立演进，而不是隐藏在一个“AI”选择器后面。",[722,723],"权限与模型分离","文件系统\u002F工具权限属于运行时策略，而不是模型能力。",[725,726],"主进程密钥","凭据所有权遵循特权进程边界，而不是渲染器\u002FUI。",[728,729],"提供商健康与模型发现","路由和可用性是运行时\u002F平台关注点。",[731,732],"无静默云回退","成本、局部性和数据传输语义保持为明确的策略决策。",{"id":734,"data":735,"type":42},"h-cms",{"text":736,"level":241},"Aaasaasa AI CMS：作为平台边界的租户范围授权",{"id":738,"data":739,"type":218},"p-cms-1",{"text":740},"Aaasaasa AI CMS 代码库提供了一个独立的实现示例：租户范围的 RBAC 通过绑定到租户标识符的角色、权限和用户角色分配来表示。系统权限按能力分组，角色查找和更新保持租户范围。",{"id":742,"data":743,"type":218},"p-cms-2",{"text":744},"这本身并不能证明一个完整的 AI 平台，但它与最困难的共享平台边界之一直接相关：可复用服务必须保留\u003Cstrong>谁可以做什么\u003C\u002Fstrong>以及\u003Cstrong>针对哪个租户\u003C\u002Fstrong>。在应用平台之上添加 AI 推理或检索并不会消除这一要求。",{"id":746,"data":747,"type":218},"p-cms-3",{"text":748},"架构上的含义是，模型网关、检索服务和代理应使用已建立的身份\u002F租户上下文，而不是发明一个并行的、仅限 AI 的授权体系。",{"id":750,"data":751,"type":42},"h-sot",{"text":752,"level":241},"真相源研究引擎：共享检索机制而不共享真相",{"id":754,"data":755,"type":218},"p-sot-1",{"text":756},"真相源研究引擎提供了第三个实现示例。不同的研究模式共享一个共同的证据核心：来源、工件、溯源、主张、关系、矛盾、参考模型和审计追踪。该系统还提供本地词汇检索、可选的语义检索、提取、快照和基于 SHA-256 的溯源。",{"id":758,"data":759,"type":218},"p-sot-2",{"text":760},"该项目明确将搜索和语义相似性视为发现信号而非证据。结果必须追溯到具体的来源和定位符，然后才能支持一项主张。这正是 AI 平台所需的区分：\u003Cstrong>可复用的检索机制可以共享，而证据权威仍由消费方法和领域来治理。\u003C\u002Fstrong>",{"id":762,"data":763,"type":218},"p-sot-3",{"text":764},"该引擎还说明了为什么一个共享平台不需要一种共享解释。历史、科学\u002F技术、市场情报和监控模式可以复用核心证据基础设施，同时保留特定模式的方法论。",{"id":766,"data":767,"type":225},"evidence-synthesis",{"body":768,"title":769,"variant":393},"在这些项目中，可复用的模式不是“一个后端包办一切”。而是\u003Cstrong>关注点分离加上显式契约\u003C\u002Fstrong>：提供者\u002F模型\u002F运行时分离、租户感知授权、凭证边界、可复用数据\u002F检索原语、溯源以及领域特定权威。未来的集成平台将需要在这些能力之间建立稳定的契约，而不是代码库之间的直接耦合。","这些实现共同展示了什么",{"id":771,"data":772,"type":42},"h-frameworks",{"text":773,"level":242},"当前架构指南如何支持这一平台范围",{"id":775,"data":776,"type":218},"p-frameworks-1",{"text":777},"ISO\u002FIEC\u002FIEEE 42010:2022 为软件、系统、企业及相关实体的架构描述提供了一般性规范。它并未定义 AI 平台架构师，但它强化了表达架构关注点、关系和视角的必要性，而不是将架构简化为技术清单。",{"id":779,"data":780,"type":218},"p-frameworks-2",{"text":781},"NIST AI RMF 1.0 和生成式 AI 配置文件将 AI 风险管理框定为贯穿生命周期，而不仅仅是在模型选择时。因此，治理、映射、测量和管理与一个平台架构兼容，该架构在许多消费工作负载中承载共享控制和证据。",{"id":783,"data":784,"type":218},"p-frameworks-3",{"text":785},"微软当前的 AI 工作负载指南将应用程序设计、数据、安全、运营、测试\u002F评估和 GenAIOps 视为相互关联的架构领域。其当前的 AI 网关指南还展示了实际的平台关注点，例如集中式模型访问、项目特定的令牌限制、配额和多团队隔离。",{"id":787,"data":788,"type":218},"p-frameworks-4",{"text":789},"AWS 当前的生成式 AI 透镜和多租户平台场景同样将基础平台控制与消费应用程序所有权分开。AWS 明确指出，中央平台可以强制执行共享护栏和可审计性，而数据质量和工作负载特定的可观测性仍然是消费应用程序或数据生产者的责任。",{"id":791,"data":792,"type":218},"p-frameworks-5",{"text":793},"供应商产品各不相同，但跨来源的模式是稳定的：生产 AI 平台必须协调身份、数据访问、模型、策略、评估、可观测性、容量、成本和生命周期。GPU 集群或模型端点仅覆盖该责任的一部分。",{"id":795,"data":796,"type":42},"h-misconceptions",{"text":797,"level":242},"常见误解",{"id":799,"data":800,"type":291},"misconceptions-table",{"content":801,"stretched":43,"withHeadings":14},[802,805,808,811,814,817,820,823,826],[803,804],"误解","为什么它是错的",[806,807],"“AI 平台就是 GPU 集群。”","计算是一种基础。平台还需要身份、模型访问、数据、策略、评估、可观测性和生命周期的契约。",[809,810],"“AI 网关只是一个反向代理。”","它还可能承载模型路由、令牌配额、成本归属、策略执行、身份和 AI 特定的遥测。",[812,813],"“共享意味着全局共享。”","服务可以在物理上共享，同时在逻辑上按租户、应用程序、区域、分类或风险级别进行分段。",[815,816],"“一个中央向量数据库成为公司真相。”","向量存储或检索服务是基础设施。领域权威、新鲜度、溯源和访问仍然是独立的关注点。",[818,819],"“平台评估取代解决方案评估。”","一般回归和遥测无法定义特定领域的答案或行动是否可接受。",[821,822],"“提供者抽象应隐藏所有差异。”","某些差异是实质性能力、安全语义或故障模式，必须保持可见。",[824,825],"“RBAC 解决了多租户。”","RBAC 控制操作；租户隔离控制资源边界。两者都可能需要。",[827,828],"“AI 平台架构师只是 MLOps 的另一个名称。”","MLOps\u002FLLMOps 是一个主要的重叠学科，但共享应用程序\u002F运行时、身份、网关、检索和工具边界可以超越模型生命周期操作。",{"id":830,"data":831,"type":42},"h-failure",{"text":832,"level":242},"AI 平台架构师应预防的故障模式",{"id":834,"data":835,"type":291},"failures-table",{"content":836,"stretched":43,"withHeadings":14},[837,840,843,846,849,852,855,858,861,864],[838,839],"故障模式","架构后果",[841,842],"每个团队存储自己的提供商密钥","重复的密钥处理、不一致的轮换以及更大的影响范围。",[844,845],"提供商抽象隐藏了所需能力","消费者无法使用他们需要的功能，或在不知情的情况下收到与假设不同的行为。",[847,848],"共享检索忽略租户\u002F用户上下文","在应用程序有机会过滤结果之前，就可能发生跨边界数据泄露。",[850,851],"回退静默更改提供商或位置","成本、合规性、数据位置和输出质量可能在调用方不知情的情况下发生变化。",[853,854],"代理工具通过模型选择授予","一个能力强的模型变得权限过高，因为运行时权限没有被独立强制执行。",[856,857],"所有提示\u002F响应默认记录日志","可观测性可能创建新的敏感数据存储库和合规问题。",[859,860],"平台拥有一个通用质量分数","领域故障仍然隐藏在平台健康指标背后。",[862,863],"平台能力没有版本契约","模型\u002F提供商\u002F运行时变更会不可预测地破坏消费者。",[865,866],"所有与AI相关的内容都集中化","平台成为瓶颈和单体，而不是可复用的能力层。",{"id":868,"data":869,"type":42},"h-decision",{"text":870,"level":242},"实用的平台架构决策序列",{"id":872,"data":873,"type":340},"decision-flow",{"steps":874,"title":905,"orientation":339},[875,878,881,884,887,890,893,896,899,902],{"label":876,"description":877},"1. 识别真实消费者","列出将消费该平台的解决方案、团队、租户和工作负载；避免为假设的复用构建平台。",{"label":879,"description":880},"2. 定义共享边界","将横切机制与解决方案特定的领域权限、工作流和验收分开。",{"label":882,"description":883},"3. 首先定义身份和隔离","在共享检索或工具能力之前，建立用户、服务、应用程序、租户、区域和数据分类。",{"label":885,"description":886},"4. 定义能力契约","指定模型\u002F提供商、检索、代理\u002F工具、网关和遥测API，并明确所有权和版本控制。",{"label":888,"description":889},"5. 决定提供商和运行时策略","选择托管、自托管、本地或混合执行，并记录回退、位置和能力语义。",{"label":891,"description":892},"6. 设计数据和检索边界","定义来源、授权传播、语料库所有权、索引和证据责任。",{"label":894,"description":895},"7. 添加配额、密钥和策略","控制成本、容量、凭证、工具权限、安全控制和影响范围。",{"label":897,"description":898},"8. 构建评估和可观测性契约","提供平台指标和追踪，同时将领域真实基准和验收留给解决方案。",{"label":900,"description":901},"9. 定义生命周期和运维","对能力进行版本控制、测试升级、记录弃用、回滚、事件、容量和消费者入门。",{"label":903,"description":904},"10. 用多个消费者验证","当共享能力实际上服务于不同的工作负载，而不强迫它们进入同一领域模型时，平台声明才变得可信。","从平台需求到可操作的共享能力",{"id":907,"data":908,"type":42},"h-edge",{"text":909,"level":242},"边缘情况和角色限制",{"id":911,"data":912,"type":218},"p-edge-1",{"text":913},"只有一个AI应用程序的小型组织可能不需要独立的AI平台或平台架构师。过早平台化可能产生比价值更多的抽象。正确的架构可能是一个设计良好的解决方案，带有几个可复用模块。",{"id":915,"data":916,"type":218},"p-edge-2",{"text":917},"气隙或主权部署会显著改变提供商、更新和可观测性模型。模型托管、工件分发、身份集成和遥测导出可能都需要本地等效方案。",{"id":919,"data":920,"type":218},"p-edge-3",{"text":921},"高度监管或高后果的工作负载可能需要更强的物理或组织隔离，而不是逻辑共享平台。复用从来不是削弱所需安全边界的充分理由。",{"id":923,"data":924,"type":218},"p-edge-4",{"text":925},"托管云AI服务可以减轻实现负担，但不会消除架构责任。组织仍然决定身份、数据访问、日志记录、保留、配额、模型资格、回退、评估和解决方案验收。",{"id":927,"data":928,"type":218},"p-edge-5",{"text":929},"平台边界也可能因模态而异。文本推理、多模态生成、语音、计算机使用和自主代理即使共享提供商和身份基础设施，也可能有不同的延迟、数据、权限和可观测性要求。",{"id":931,"data":932,"type":42},"h-change",{"text":933,"level":242},"什么会改变这个答案？",{"id":935,"data":936,"type":218},"p-change-1",{"text":937},"如果组织范围发生变化，核心定义也会改变。如果架构师负责一个工作负载，角色就更接近AI解决方案架构师。如果职责扩展到组织范围内的能力战略、投资、标准和目标状态组合，则转向企业AI架构。",{"id":939,"data":940,"type":218},"p-change-2",{"text":941},"每当提供商、网关产品、代理协议、监管义务、模型能力或部署约束发生变化时，实施指南就会改变。这就是为什么平台架构应该将稳定的职责和契约与当前的供应商机制分开表达。",{"id":943,"data":944,"type":42},"h-checklist",{"text":945,"level":242},"AI平台架构师检查清单",{"id":947,"data":948,"type":291},"checklist-table",{"content":949,"stretched":43,"withHeadings":14},[950,953,956,959,962,965,968,971,974,977,980,983,986],[951,952],"问题","预期答案",[954,955],"谁是实际的平台消费者？","具有不同但重叠需求的命名解决方案、团队或租户上下文。",[957,958],"什么是真正共享的？","明确的能力列表，而不是模糊的“AI后端”。",[960,961],"什么必须保持解决方案特定？","领域权限、业务工作流、任务验收和其他工作负载拥有的关注点。",[963,964],"模型\u002F提供商如何表示？","带能力版本化的提供商\u002F模型契约和明确的回退语义。",[966,967],"身份如何传播？","用户\u002F服务\u002F应用程序\u002F租户上下文在每条特权请求路径中得以保留。",[969,970],"租户隔离如何强制执行？","资源作用域与角色权限检查分开。",[972,973],"密钥如何处理？","特权存储、轮换、有限暴露和可审计的所有权。",[975,976],"检索如何保持权限？","共享机制与授权、来源和领域拥有的证据规则。",[978,979],"工具和代理如何受到约束？","运行时权限、有界工具契约、审批、取消和可追溯性。",[981,982],"成本和容量如何控制？","配额、令牌\u002F速率控制、使用归因和过载行为。",[984,985],"质量如何衡量？","平台回归\u002F评估加上解决方案特定的真实基准和验收。",[987,988],"变更如何推出？","版本控制、兼容性、迁移、弃用、回滚和事件所有权。",{"id":990,"data":991,"type":42},"h-conclusion",{"text":992,"level":242},"结论",{"id":994,"data":995,"type":218},"p-conclusion-1",{"text":996},"AI平台架构师负责\u003Cstrong>AI能力与消费它们的解决方案之间\u003C\u002Fstrong>的可复用架构。该角色定义模型、提供商、检索、代理、工具、身份、租户、密钥、评估、可观测性、配额和运行时操作如何成为可靠的平台服务，而不是重复的一次性集成。",{"id":998,"data":999,"type":218},"p-conclusion-2",{"text":1000},"困难的部分不是最大化复用，而是选择正确的边界。强大的平台在多个消费者真正受益的地方标准化机制、策略和操作，同时保留解决方案特定的数据权限、业务逻辑、安全要求和验收标准。",{"id":1002,"data":1003,"type":218},"p-conclusion-3",{"text":1004},"这种区别也解释了与AI解决方案架构的关系：\u003Cstrong>解决方案架构师使一个AI赋能的系统适合其目的；平台架构师使共享的AI能力在许多此类系统中安全、可复用、可操作和可演进。\u003C\u002Fstrong>",{"id":1006,"data":1007,"type":42},"h-related",{"text":1008,"level":242},"相关规范知识",{"id":1010,"data":1011,"type":218},"p-related-1",{"text":1012},"本文位于生成式AI组件、ADR与NFR、以及AI解决方案架构的规范基础之后。这些概念是前提条件，因为平台的存在是为了提供可复用的系统能力，并针对明确的质量和运营需求编码架构决策。",{"id":1014,"data":1015,"type":218},"p-related-2",{"text":1016},"检索增强生成是可能通过平台提供的能力的一个例子，但平台不应将检索基础设施、领域知识和答案有效性混为一谈。",{"id":1018,"data":1019,"type":1026},"related-rag",{"link":1020,"meta":1021},"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works",{"image":1022,"title":1024,"description":1025},{"url":1023},"","什么是RAG？其工作原理的最简解释","检索增强生成的规范介绍，以及模型生成与外部知识检索之间的边界。","linkTool",{"id":1028,"data":1029,"type":218},"p-related-3",{"text":1030},"代理协议、租户隔离、AI治理、模型路由、上下文工程和MLOps\u002FLLMOps是下游或相邻的知识节点。一旦平台边界明确，它们就更容易推理。",{"id":1032,"data":1033,"type":42},"h-faq",{"text":1034,"level":242},"常见问题",{"id":1036,"data":1037,"type":1036},"faq",{"items":1038,"title":1063},[1039,1043,1047,1051,1055,1059],{"id":1040,"answer":1041,"question":1042},"faq-1","不同。解决方案架构师专注于一个具体的AI赋能解决方案。平台架构师专注于可复用的AI能力、控制和运营契约，以支持多个解决方案。","AI平台架构师与AI解决方案架构师相同吗？",{"id":1044,"answer":1045,"question":1046},"faq-2","不需要。平台可以使用托管云模型、自托管模型、本地推理或混合策略。架构必须明确提供商、位置、身份、路由、数据和运营后果。","AI平台需要托管自己的模型吗？",{"id":1048,"answer":1049,"question":1050},"faq-3","通常不够。网关可以是重要的平台组件，但完整的平台还需要身份、密钥、数据\u002F检索、评估、可观测性、生命周期和运营所有权的契约。","AI网关足以成为AI平台吗？",{"id":1052,"answer":1053,"question":1054},"faq-4","检索机制通常可以共享，但领域权威、授权、新鲜度、证据充分性和语料库所有权应保持明确。共享基础设施并不意味着共享真相。","检索应该集中化吗？",{"id":1056,"answer":1057,"question":1058},"faq-5","不能。平台评估可以测试共享能力和回归。每个解决方案仍然需要特定任务的真实基准、验收标准和领域质量阈值。","平台评估能替代应用评估吗？",{"id":1060,"answer":1061,"question":1062},"faq-6","不是。RBAC决定身份可以做什么。租户隔离决定身份可以对哪个租户的资源进行操作。平台通常需要两者。","多租户只是RBAC吗？","AI平台架构师FAQ",{"id":1065,"data":1066,"type":42},"h-glossary",{"text":1067,"level":242},"术语表",{"id":1069,"data":1070,"type":1069},"glossary",{"title":1071,"entries":1072},"关键AI平台架构术语",[1073,1077,1081,1085,1089,1093,1097,1101],{"term":1074,"anchor":1075,"definition":1076},"AI平台","ai-platform","由多个应用、团队或租户上下文消费的一组可复用的AI相关技术和运营能力。",{"term":1078,"anchor":1079,"definition":1080},"AI网关","ai-gateway","AI端点的网关层，可在基本代理之外添加认证、路由、配额、策略、重试、成本归属和AI特定遥测。",{"term":1082,"anchor":1083,"definition":1084},"提供商适配器","provider-adapter","将平台契约映射到模型提供商的API、能力、健康状况和故障语义的组件。",{"term":1086,"anchor":1087,"definition":1088},"租户隔离","tenant-isolation","防止一个租户上下文访问另一个租户资源的边界，独立于角色权限。",{"term":1090,"anchor":1091,"definition":1092},"能力契约","capability-contract","描述共享平台服务提供什么以及消费者必须提供或拥有什么的版本化接口和行为协议。",{"term":1094,"anchor":1095,"definition":1096},"接地\u002F检索服务","grounding-service","为AI工作负载查找和提供外部信息的共享机制；它不会自动定义哪些信息对某个领域是权威的。",{"term":1098,"anchor":1099,"definition":1100},"评估工具","evaluation-harness","用于运行测试、数据集、模型\u002F提示版本和指标的可复用基础设施；领域验收仍然特定于解决方案。",{"term":1102,"anchor":1103,"definition":1104},"控制平面","control-plane","管理平台能力、身份、策略、配额、版本和部署状态的配置和治理层。",{"id":1106,"data":1107,"type":42},"h-sources",{"text":1108,"level":242},"主要来源和当前架构指南",{"id":1110,"data":1111,"type":218},"p-sources-note",{"text":1112},"以下来源支持一般架构和生产平台声明。Aaasaasa AI客户端、Aaasaasa AI CMS和真相源研究引擎部分明确为原创实现证据。当前状态的外部参考已于2026年10月8日检查。",{"id":1114,"data":1115,"type":1026},"src-iso-42010",{"link":1116,"meta":1117},"https:\u002F\u002Fwww.iso.org\u002Fstandard\u002F74393.html",{"image":1118,"title":1119,"description":1120},{"url":1023},"ISO\u002FIEC\u002FIEEE 42010:2022 — 架构描述","当前发布的架构描述概念和关系的国际标准。",{"id":1122,"data":1123,"type":1026},"src-nist-rmf",{"link":1124,"meta":1125},"https:\u002F\u002Fwww.nist.gov\u002Fitl\u002Fai-risk-management-framework",{"image":1126,"title":1127,"description":1128},{"url":1023},"NIST AI风险管理框架","NIST的AI RMF资源和当前状态；截至2026年10月，AI RMF 1.0正在修订中。",{"id":1130,"data":1131,"type":1026},"src-nist-gai",{"link":1132,"meta":1133},"https:\u002F\u002Fwww.nist.gov\u002Fpublications\u002Fartificial-intelligence-risk-management-framework-generative-artificial-intelligence",{"image":1134,"title":1135,"description":1136},{"url":1023},"NIST AI 600-1 — 生成式AI概况","在整个AI生命周期中应用AI风险管理考虑的生成式AI概况。",{"id":1138,"data":1139,"type":1026},"src-ms-ai",{"link":1140,"meta":1141},"https:\u002F\u002Flearn.microsoft.com\u002Fen-us\u002Fazure\u002Fwell-architected\u002Fai\u002Fget-started",{"image":1142,"title":1143,"description":1144},{"url":1023},"Microsoft Azure Well-Architected — AI工作负载","涵盖AI应用、数据、运营、评估、负责任AI和生命周期问题的当前架构指南。",{"id":1146,"data":1147,"type":1026},"src-ms-principles",{"link":1148,"meta":1149},"https:\u002F\u002Flearn.microsoft.com\u002Fen-us\u002Fazure\u002Fwell-architected\u002Fai\u002Fdesign-principles",{"image":1150,"title":1151,"description":1152},{"url":1023},"Microsoft — AI工作负载设计原则","关于身份分段、安全边界、遥测、性能、数据和平台权衡的当前指南。",{"id":1154,"data":1155,"type":1026},"src-ms-gateway",{"link":1156,"meta":1157},"https:\u002F\u002Flearn.microsoft.com\u002Fen-us\u002Fazure\u002Fai-foundry\u002Fconfiguration\u002Fenable-ai-api-management-gateway-portal?view=foundry",{"image":1158,"title":1159,"description":1160},{"url":1023},"Microsoft Foundry — AI网关架构","关于共享项目访问、令牌遏制、配额和治理的当前AI网关指南。",{"id":1162,"data":1163,"type":1026},"src-ms-gateway-guide",{"link":1164,"meta":1165},"https:\u002F\u002Flearn.microsoft.com\u002Fen-us\u002Fazure\u002Farchitecture\u002Fai-ml\u002Fguide\u002Fazure-openai-gateway-guide",{"image":1166,"title":1167,"description":1168},{"url":1023},"Azure架构中心 — 通过网关访问模型","关于集中模型访问、路由、限流、故障转移和客户端\u002F平台责任的架构指南。",{"id":1170,"data":1171,"type":1026},"src-aws-genai",{"link":1172,"meta":1173},"https:\u002F\u002Fdocs.aws.amazon.com\u002Fwellarchitected\u002Flatest\u002Fgenerative-ai-lens\u002F",{"image":1174,"title":1175,"description":1176},{"url":1023},"AWS Well-Architected — 生成式 AI 透镜","面向生成式 AI 工作负载的当前生产架构指导，涵盖安全性、可靠性、运维、性能和成本。",{"id":1178,"data":1179,"type":1026},"src-aws-multitenant",{"link":1180,"meta":1181},"https:\u002F\u002Fdocs.aws.amazon.com\u002Fwellarchitected\u002Flatest\u002Fgenerative-ai-lens\u002Fmulti-tenant-generative-ai-platform-scenario.html",{"image":1182,"title":1183,"description":1184},{"url":1023},"AWS — 多租户生成式 AI 平台场景","当前示例，将中央平台控制与可审计性，同消费应用的数据质量及工作负载特定职责区分开来。",{"id":1186,"data":1187,"type":1026},"src-aws-agentic",{"link":1188,"meta":1189},"https:\u002F\u002Fdocs.aws.amazon.com\u002Fwellarchitected\u002Flatest\u002Fagentic-ai-lens\u002Fdesign-principles.html",{"image":1190,"title":1191,"description":1192},{"url":1023},"AWS Well-Architected — 代理式 AI 设计原则","关于受限代理权限、可追溯性、版本化行为、显式契约和人工监督的当前指导。",{"id":1194,"data":1195,"type":1026},"src-aws-observability",{"link":1196,"meta":1197},"https:\u002F\u002Fdocs.aws.amazon.com\u002FAmazonCloudWatch\u002Flatest\u002Fmonitoring\u002FGenAI-observability.html",{"image":1198,"title":1199,"description":1200},{"url":1023},"AWS CloudWatch — 生成式 AI 可观测性","针对模型、代理、知识库、工具以及成本\u002F延迟\u002F错误分析的当前可观测性能力和生产指标。","2.31","AI平台架构师负责跨模型、提供商、检索、智能体、身份、安全、评估、可观测性和运营设计可复用的AI基础。","\u002Fuploads\u002F2026\u002F10\u002Fwhat-is-an-ai-platform-architect-models-data-runtime-security-and-operations-1791477229171-ou3zcc.webp","what-is-an-ai-platform-architect-models-data-runtime-security-and-operations-1791477229171-ou3zcc","PUBLISHED","2026-10-08T12:32:00.000Z","2026-10-08T16:32:14.856Z","2026-10-08T16:47:57.364Z",{"en":1210,"de":1211,"sr":1212,"es":1213,"fr":1214,"it":1215,"ru":1216,"zh":1217},"\u002Fblog\u002Fwhat-is-an-ai-platform-architect-models-data-runtime-security-and-operations","\u002Fde\u002Fblog\u002Fwhat-is-an-ai-platform-architect-models-data-runtime-security-and-operations","\u002Fsr\u002Fblog\u002Fwhat-is-an-ai-platform-architect-models-data-runtime-security-and-operations","\u002Fes\u002Fblog\u002Fwhat-is-an-ai-platform-architect-models-data-runtime-security-and-operations","\u002Ffr\u002Fblog\u002Fwhat-is-an-ai-platform-architect-models-data-runtime-security-and-operations","\u002Fit\u002Fblog\u002Fwhat-is-an-ai-platform-architect-models-data-runtime-security-and-operations","\u002Fru\u002Fblog\u002Fwhat-is-an-ai-platform-architect-models-data-runtime-security-and-operations","\u002Fzh\u002Fblog\u002Fwhat-is-an-ai-platform-architect-models-data-runtime-security-and-operations",[1219,1223,1227],{"id":1220,"name":1221,"slug":1222},84,"策略与数据边界","policy-and-data",{"id":1224,"name":1225,"slug":1226},57,"数据边界","data-boundaries",{"id":1228,"name":1229,"slug":1230},80,"访问与身份","access-and-identity",{"id":1232,"login":1233,"email":1234,"displayName":1235},"20","rooth8233","aleksandar@stajic.de","Aleksandar Stajić",[1237,2026],{"lang":1238,"title":1239,"content":1240,"contentJson":1241,"excerpt":2025},"en","What Is an AI Platform Architect? Models, Data, Runtime, Security and Operations","{\"time\":1791476955677,\"blocks\":[{\"id\":\"intro\",\"data\":{\"text\":\"An \u003Cstrong>AI Platform Architect\u003C\u002Fstrong> designs the reusable AI foundation through which multiple applications, teams, or tenant contexts access models, data and retrieval, agent and tool runtimes, identity and permissions, evaluation, observability, quotas, secrets, and deployment capabilities. The role is broader than infrastructure but narrower than owning every AI-enabled product: its central responsibility is deciding \u003Cstrong>what should be shared, how shared capabilities are governed and isolated, and what must remain solution-specific\u003C\u002Fstrong>.\"},\"type\":\"paragraph\"},{\"id\":\"direct\",\"data\":{\"body\":\"\u003Cstrong>An AI Platform Architect designs the shared technical and operational substrate for AI systems.\u003C\u002Fstrong> Instead of architecting one assistant or one workflow, the role defines reusable contracts and boundaries for model\u002Fprovider access, gateways and routing, retrieval services, agent runtimes, tool access, identity and tenant isolation, secrets, evaluation, telemetry, deployment and lifecycle management.\",\"title\":\"Direct answer\",\"variant\":\"info\"},\"type\":\"callout\"},{\"id\":\"term-note\",\"data\":{\"body\":\"\u003Cstrong>AI Platform Architect is a practical role label, not a universally standardized job title.\u003C\u002Fstrong> ISO\u002FIEC\u002FIEEE 42010:2022 defines concepts for architecture descriptions, not this role. Different organizations may split these responsibilities among platform architects, solution architects, enterprise architects, security architects, MLOps\u002FLLMOps specialists and platform engineering teams. This article uses the term for the architecture responsibility over a reusable AI platform layer.\",\"title\":\"Terminology note\",\"variant\":\"note\"},\"type\":\"callout\"},{\"id\":\"version-note\",\"data\":{\"body\":\"The stable architectural principles here are vendor-neutral. Current Microsoft, AWS and NIST guidance is used as external implementation and governance evidence. NIST states that AI RMF 1.0 is being revised; vendor platform features, gateway products, agent runtimes and model capabilities evolve faster than the architectural principles, so version-sensitive implementation choices must be rechecked before deployment.\",\"title\":\"Current-source note — 8 October 2026\",\"variant\":\"note\"},\"type\":\"callout\"},{\"id\":\"toc\",\"data\":{\"title\":\"Contents\",\"maxLevel\":3,\"minLevel\":2},\"type\":\"tableOfContents\"},{\"id\":\"h-meaning\",\"data\":{\"text\":\"What does an AI Platform Architect actually architect?\",\"level\":2},\"type\":\"header\"},{\"id\":\"p-meaning-1\",\"data\":{\"text\":\"The object of the work is the \u003Cstrong>platform\u003C\u002Fstrong>: a set of shared capabilities that reduces repeated integration work while preserving explicit security, data and operational boundaries. A platform can expose model access, provider adapters, retrieval primitives, agent execution, tool brokers, policy enforcement, evaluation, telemetry and deployment services to many consuming solutions.\"},\"type\":\"paragraph\"},{\"id\":\"p-meaning-2\",\"data\":{\"text\":\"The platform is not valuable merely because components are centralized. It is valuable when consumers receive stable capabilities with clear contracts, ownership, isolation, observability and lifecycle rules. The key architectural question is therefore not “Which model should everyone use?” but \u003Cstrong>“Which responsibilities can be safely standardized and reused without erasing the requirements of each solution?”\u003C\u002Fstrong>.\"},\"type\":\"paragraph\"},{\"id\":\"solution-vs-platform\",\"data\":{\"rows\":[{\"id\":\"c1\",\"label\":\"Primary scope\",\"values\":{\"platform\":\"Reusable AI capabilities consumed by multiple solutions, teams or tenant contexts.\",\"solution\":\"One concrete AI-enabled product, workflow or application.\"}},{\"id\":\"c2\",\"label\":\"Main question\",\"values\":{\"platform\":\"Which shared capabilities and controls should solutions consume, and where must solution-specific ownership remain?\",\"solution\":\"How should this solution meet its business, data, security, quality and operational requirements?\"}},{\"id\":\"c3\",\"label\":\"Data authority\",\"values\":{\"platform\":\"Provides storage, retrieval, provenance or access primitives without automatically becoming the authority for every domain.\",\"solution\":\"Defines which domain data is authoritative and how the solution may use it.\"}},{\"id\":\"c4\",\"label\":\"Evaluation\",\"values\":{\"platform\":\"Provides reusable evaluation, telemetry and release mechanisms; it cannot define every domain's success threshold.\",\"solution\":\"Defines task-specific quality and acceptance criteria.\"}},{\"id\":\"c5\",\"label\":\"Lifecycle\",\"values\":{\"platform\":\"Owns shared capability versions, compatibility, onboarding, quotas, policy and operational contracts.\",\"solution\":\"Owns the lifecycle of the specific workload.\"}}],\"title\":\"Solution architecture and platform architecture solve different scope problems\",\"layout\":\"table\",\"columns\":[{\"id\":\"solution\",\"label\":\"AI Solution Architect\"},{\"id\":\"platform\",\"label\":\"AI Platform Architect\"}]},\"type\":\"comparison\"},{\"id\":\"h-simple\",\"data\":{\"text\":\"The simplest example\",\"level\":2},\"type\":\"header\"},{\"id\":\"p-simple-1\",\"data\":{\"text\":\"Imagine an organization has five AI-enabled products: an internal document assistant, a customer-support copilot, a software-engineering agent, a contract review workflow and a product-search assistant. Each product could independently integrate model APIs, keep credentials, implement retries, collect token metrics, create retrieval code and build its own tool permissions.\"},\"type\":\"paragraph\"},{\"id\":\"p-simple-2\",\"data\":{\"text\":\"That duplication is expensive and dangerous when every team invents a different security and operational model. A shared platform can instead offer approved provider connections, model discovery, quotas, credentials, tenant-aware access, common telemetry, reusable retrieval services and an agent\u002Ftool runtime contract.\"},\"type\":\"paragraph\"},{\"id\":\"p-simple-3\",\"data\":{\"text\":\"But the platform must stop at the correct boundary. The contract-review solution may require legal-document authority and citation rules that the software agent does not. The product-search assistant may need commerce-specific freshness and authorization rules. \u003Cstrong>Reusable infrastructure does not make all domain truth reusable.\u003C\u002Fstrong>\"},\"type\":\"paragraph\"},{\"id\":\"simple-flow\",\"data\":{\"steps\":[{\"label\":\"1. Consumer identifies itself\",\"description\":\"The calling application, user, service, team or tenant enters through an authenticated identity and explicit scope.\"},{\"label\":\"2. Platform policy applies\",\"description\":\"Gateway and policy layers determine allowed providers, models, quotas, data paths, tools and execution modes.\"},{\"label\":\"3. Shared capability executes\",\"description\":\"The request may use inference, retrieval, agent runtime, tool access or another reusable platform service.\"},{\"label\":\"4. Solution-specific context remains authoritative\",\"description\":\"The consuming solution supplies domain rules, user intent, data authority, task-specific constraints and acceptance logic.\"},{\"label\":\"5. Telemetry and evidence are captured\",\"description\":\"The platform records identity, route, model\u002Fprovider, latency, cost, errors, tool activity and other permitted observability signals.\"},{\"label\":\"6. Result returns under the solution contract\",\"description\":\"The solution remains responsible for whether the output is acceptable for its user and domain.\"}],\"title\":\"A shared AI request path\",\"orientation\":\"auto\"},\"type\":\"processFlow\"},{\"id\":\"h-stops\",\"data\":{\"text\":\"Where the simple example stops\",\"level\":2},\"type\":\"header\"},{\"id\":\"p-stops-1\",\"data\":{\"text\":\"Centralization is not automatically architecture. A single endpoint in front of several model APIs is useful, but it does not by itself create an AI platform. A production platform also needs identity boundaries, capability contracts, provider health and lifecycle handling, quotas, secret ownership, observability, compatibility rules, security controls, release discipline and clear operational responsibility.\"},\"type\":\"paragraph\"},{\"id\":\"p-stops-2\",\"data\":{\"text\":\"The opposite failure is also common: putting every prompt, vector index, business rule, agent and application workflow into one “AI backend.” That creates a monolith whose shared status is accidental rather than architectural. \u003Cstrong>A platform should standardize cross-cutting capabilities, not absorb domain ownership merely because AI is involved.\u003C\u002Fstrong>\"},\"type\":\"paragraph\"},{\"id\":\"h-boundary\",\"data\":{\"text\":\"The most important platform decision: shared versus solution-specific\",\"level\":2},\"type\":\"header\"},{\"id\":\"shared-boundary-table\",\"data\":{\"content\":[[\"Capability area\",\"Good candidate for shared platform ownership\",\"Usually remains solution-specific\"],[\"Model access\",\"Approved provider connections, adapters, credentials, health, routing primitives, quotas\",\"Task-specific model acceptance, prompt behavior, quality threshold\"],[\"Retrieval\",\"Ingestion primitives, extraction, indexing, search APIs, provenance contracts, authorization hooks\",\"Authoritative corpus, freshness rules, domain metadata, evidence sufficiency\"],[\"Agents and tools\",\"Runtime lifecycle, tool registry\u002Fbroker, permission enforcement, tracing, cancellation\",\"Business workflow, allowed action semantics, escalation policy, task success\"],[\"Security\",\"Identity integration, secret storage, policy enforcement, audit contracts, tenant isolation mechanisms\",\"Data classification, business authorization rules, domain-specific risk acceptance\"],[\"Evaluation\",\"Harness, dataset\u002Fversion mechanics, telemetry, experiment\u002Frelease workflow\",\"Ground truth, domain test set, acceptance threshold, user outcome\"],[\"Operations\",\"Deployment pattern, health, metrics, incident integration, capacity controls\",\"Solution SLOs where they differ, business continuity impact, workload-specific runbooks\"]],\"stretched\":false,\"withHeadings\":true},\"type\":\"table\"},{\"id\":\"boundary-principle\",\"data\":{\"body\":\"\u003Cstrong>Share mechanics and controls where reuse is real; keep authority and acceptance where the domain owns them.\u003C\u002Fstrong> This prevents two opposite errors: duplicated infrastructure everywhere, and a central platform that falsely becomes the owner of every application's data, policy and quality.\",\"title\":\"Platform principle\",\"variant\":\"success\"},\"type\":\"callout\"},{\"id\":\"h-responsibility-map\",\"data\":{\"text\":\"Architecture responsibility map\",\"level\":2},\"type\":\"header\"},{\"id\":\"h-provider\",\"data\":{\"text\":\"1. Model and provider access\",\"level\":3},\"type\":\"header\"},{\"id\":\"p-provider-1\",\"data\":{\"text\":\"A platform architect defines how consumers discover and invoke models without forcing every application to hard-code one provider. This includes provider adapters, model identifiers, capability metadata, authentication, health checks, endpoint configuration, request normalization and compatibility behavior.\"},\"type\":\"paragraph\"},{\"id\":\"p-provider-2\",\"data\":{\"text\":\"Provider abstraction must remain honest. Different providers expose different context limits, tool semantics, structured-output behavior, multimodal capabilities, safety controls, caching, pricing and failure modes. A good abstraction creates a stable platform contract while preserving access to capabilities that cannot be meaningfully flattened.\"},\"type\":\"paragraph\"},{\"id\":\"provider-warning\",\"data\":{\"body\":\"A lowest-common-denominator API can make migration easier but can also erase capabilities that matter. The architecture should define which features are portable, which are provider-specific and how consumers discover that difference.\",\"title\":\"Do not confuse abstraction with pretending providers are identical\",\"variant\":\"warning\"},\"type\":\"callout\"},{\"id\":\"h-gateway\",\"data\":{\"text\":\"2. Gateway, routing, quotas and cost controls\",\"level\":3},\"type\":\"header\"},{\"id\":\"p-gateway-1\",\"data\":{\"text\":\"A shared AI gateway can centralize authentication, routing, throttling, retries, token limits, usage attribution and policy enforcement. Microsoft’s current AI Gateway guidance explicitly treats token-per-minute limits, quotas and multi-project containment as platform concerns; AWS likewise exposes account and model quotas and centralized controls.\"},\"type\":\"paragraph\"},{\"id\":\"p-gateway-2\",\"data\":{\"text\":\"The gateway is therefore more than a reverse proxy when it carries AI-specific policy and operational semantics. But it should not silently make business decisions. A routing policy may prefer a healthy local model, a lower-cost provider or a regionally compliant endpoint; whether that route is acceptable for a particular task is still a contract between platform and solution.\"},\"type\":\"paragraph\"},{\"id\":\"p-gateway-3\",\"data\":{\"text\":\"Routing also needs failure semantics. If the preferred model is unavailable, the platform must know whether fallback is permitted, whether a cloud route requires explicit consent, whether a lower-capability model is valid and how the decision is surfaced to observability.\"},\"type\":\"paragraph\"},{\"id\":\"h-data\",\"data\":{\"text\":\"3. Shared data, retrieval and grounding services\",\"level\":3},\"type\":\"header\"},{\"id\":\"p-data-1\",\"data\":{\"text\":\"Retrieval services are strong platform candidates because parsing, chunking, indexing, lexical search, semantic search, metadata filtering, provenance and citation mechanics are reusable. However, the platform must not confuse a shared retrieval engine with a shared source of truth.\"},\"type\":\"paragraph\"},{\"id\":\"p-data-2\",\"data\":{\"text\":\"A solution still owns questions such as: Which corpus is authoritative? Which version is valid? Can this user see this document? How fresh must the data be? What counts as sufficient evidence? Can an answer be generated when retrieval fails? Those are domain and solution requirements even when the platform supplies the retrieval machinery.\"},\"type\":\"paragraph\"},{\"id\":\"p-data-3\",\"data\":{\"text\":\"This boundary is especially important in multi-tenant systems. A technically shared index or vector service does not justify cross-tenant visibility. Authorization context must be preserved through retrieval, not added only after search results have already crossed the boundary.\"},\"type\":\"paragraph\"},{\"id\":\"h-agent-runtime\",\"data\":{\"text\":\"4. Agent and tool runtime\",\"level\":3},\"type\":\"header\"},{\"id\":\"p-agent-1\",\"data\":{\"text\":\"Agentic systems add reusable runtime concerns: thread\u002Fsession lifecycle, planning loops, tool registration, tool invocation, cancellation, timeouts, human approvals, memory\u002Fstate interfaces, remote-agent protocols and trace correlation. A platform can provide these mechanics so each product does not rebuild them.\"},\"type\":\"paragraph\"},{\"id\":\"p-agent-2\",\"data\":{\"text\":\"The platform must also keep tool permission separate from model capability. A model being capable of generating a shell command does not mean the runtime should allow shell execution. The permission boundary belongs to the application\u002Fruntime architecture and must be enforceable independently of the model.\"},\"type\":\"paragraph\"},{\"id\":\"p-agent-3\",\"data\":{\"text\":\"Current AWS Agentic AI guidance emphasizes bounded agents, explicit authority, end-to-end tracing, versioned behavioral artifacts and human oversight proportionate to consequence. Those are platform-enabling concerns, but the consuming solution still defines what actions are legitimate for its domain.\"},\"type\":\"paragraph\"},{\"id\":\"h-identity\",\"data\":{\"text\":\"5. Identity, tenant isolation and authorization\",\"level\":3},\"type\":\"header\"},{\"id\":\"p-identity-1\",\"data\":{\"text\":\"AI platforms often sit in front of high-value models, proprietary data and action-capable tools. Authentication is therefore only the beginning. The architecture must carry user, service, application and tenant context through every privileged operation that needs it.\"},\"type\":\"paragraph\"},{\"id\":\"p-identity-2\",\"data\":{\"text\":\"\u003Cstrong>RBAC and tenant isolation solve different problems.\u003C\u002Fstrong> RBAC answers what an identity may do; tenant isolation answers which tenant’s resources that identity may act on. A platform that checks roles but loses tenant context can still expose the wrong data.\"},\"type\":\"paragraph\"},{\"id\":\"p-identity-3\",\"data\":{\"text\":\"Microsoft’s current AI workload guidance explicitly recommends identity segmentation and authorization-aware access to content. AWS’s multi-tenant generative AI platform guidance similarly treats logical isolation, centralized controls and auditability as platform concerns.\"},\"type\":\"paragraph\"},{\"id\":\"h-secrets\",\"data\":{\"text\":\"6. Secrets, credentials and trust boundaries\",\"level\":3},\"type\":\"header\"},{\"id\":\"p-secrets-1\",\"data\":{\"text\":\"A platform should define who owns provider keys, remote bearer tokens, signing material and tool credentials, where they are stored, which process can access them, how they are rotated and whether they can ever reach a browser or untrusted renderer.\"},\"type\":\"paragraph\"},{\"id\":\"p-secrets-2\",\"data\":{\"text\":\"This is an architectural boundary, not an implementation detail. If every consuming application copies provider credentials into its own configuration, the organization has duplicated both operational burden and blast radius. Centralization can reduce that risk only if the platform itself has narrower, auditable access paths.\"},\"type\":\"paragraph\"},{\"id\":\"h-eval\",\"data\":{\"text\":\"7. Evaluation, observability and auditability\",\"level\":3},\"type\":\"header\"},{\"id\":\"p-eval-1\",\"data\":{\"text\":\"A reusable platform can provide evaluation harnesses, trace IDs, model\u002Fprovider metadata, token and cost metrics, latency, error rates, prompt\u002Fmodel version linkage, agent\u002Ftool traces and controlled logging. AWS and Microsoft both treat observability and evaluation as core production concerns for AI workloads.\"},\"type\":\"paragraph\"},{\"id\":\"p-eval-2\",\"data\":{\"text\":\"Platform evaluation and solution evaluation must remain separate. A platform can verify that an endpoint is healthy, a model version passes a general regression suite and traces are complete. It cannot decide that a legal answer, medical workflow or product recommendation is acceptable without domain-specific ground truth and acceptance criteria.\"},\"type\":\"paragraph\"},{\"id\":\"p-eval-3\",\"data\":{\"text\":\"Logging also creates a privacy boundary. Prompt and response logs may contain sensitive or proprietary data. The platform architect must therefore decide what is logged, redacted, sampled, retained and accessible rather than assuming that more telemetry is always safer.\"},\"type\":\"paragraph\"},{\"id\":\"h-runtime\",\"data\":{\"text\":\"8. Runtime, deployment and locality\",\"level\":3},\"type\":\"header\"},{\"id\":\"p-runtime-1\",\"data\":{\"text\":\"A platform architect decides how shared AI capabilities are deployed and reached: managed cloud services, self-hosted endpoints, local inference, hybrid routing, containerized services, desktop runtimes, private networking or air-gapped environments. The important distinction is between \u003Cstrong>where the control\u002Fruntime process runs\u003C\u002Fstrong> and \u003Cstrong>where inference and data processing actually occur\u003C\u002Fstrong>.\"},\"type\":\"paragraph\"},{\"id\":\"p-runtime-2\",\"data\":{\"text\":\"A local client may still call a cloud model. A cloud control plane may route to an on-premises model. A remote agent may execute tools inside a customer network. Architectural diagrams must therefore show trust and data-flow boundaries rather than using “local” and “cloud” as vague labels.\"},\"type\":\"paragraph\"},{\"id\":\"h-lifecycle\",\"data\":{\"text\":\"9. Platform lifecycle, compatibility and onboarding\",\"level\":3},\"type\":\"header\"},{\"id\":\"p-lifecycle-1\",\"data\":{\"text\":\"Reusable capability becomes a platform only when consumers can depend on it over time. That requires versioned contracts, migration rules, compatibility policy, deprecation, release testing, rollback, incident ownership, capacity planning, documentation and a path for onboarding new teams or applications.\"},\"type\":\"paragraph\"},{\"id\":\"p-lifecycle-2\",\"data\":{\"text\":\"Fast-moving AI ecosystems make this particularly important. Model names, SDKs, protocol versions, provider APIs and safety capabilities change independently. A platform must absorb some of that volatility without hiding changes that materially affect a solution’s behavior.\"},\"type\":\"paragraph\"},{\"id\":\"h-control-plane\",\"data\":{\"text\":\"A practical control-plane \u002F execution-plane \u002F solution-plane model\",\"level\":2},\"type\":\"header\"},{\"id\":\"model-note\",\"data\":{\"body\":\"The three-plane model below is a practical way to reason about responsibilities; it is not an ISO, NIST, Microsoft or AWS standard. Its purpose is to make ownership boundaries explicit.\",\"title\":\"Proposed architecture model\",\"variant\":\"note\"},\"type\":\"callout\"},{\"id\":\"planes-table\",\"data\":{\"content\":[[\"Plane\",\"Typical responsibilities\",\"Should not silently own\"],[\"Platform control plane\",\"Provider registry, model policy, quotas, tenant configuration, identities, secrets, routing rules, capability versions, deployment configuration\",\"Application business logic or domain truth\"],[\"Platform execution\u002Fdata plane\",\"Inference requests, retrieval operations, agent\u002Ftool execution, extraction, indexing, telemetry emission, policy enforcement\",\"Cross-tenant access merely because infrastructure is shared\"],[\"Solution plane\",\"User workflow, prompts\u002Finstructions, authoritative corpus selection, domain authorization, business rules, task evaluation and acceptance\",\"Low-level provider integration that the platform explicitly owns\"]],\"stretched\":false,\"withHeadings\":true},\"type\":\"table\"},{\"id\":\"p-control-plane-1\",\"data\":{\"text\":\"This separation helps diagnose platform drift. If an application must know every provider-specific credential and endpoint, the platform contract is too thin. If the platform decides which customer record is legally authoritative or whether a domain answer is acceptable, the platform has crossed into solution ownership.\"},\"type\":\"paragraph\"},{\"id\":\"h-artifacts\",\"data\":{\"text\":\"What should an AI Platform Architect produce?\",\"level\":2},\"type\":\"header\"},{\"id\":\"artifacts-table\",\"data\":{\"content\":[[\"Architecture artifact\",\"Purpose\"],[\"Platform capability map\",\"Defines what the platform provides, who consumes it and which capabilities remain outside scope.\"],[\"Provider\u002Fmodel contract\",\"Defines providers, models, capabilities, abstraction boundaries, route metadata and fallback semantics.\"],[\"Identity and tenancy model\",\"Defines user\u002Fservice\u002Fapplication identity, tenant context, RBAC\u002FABAC hooks and resource isolation.\"],[\"Gateway and quota policy\",\"Defines rate limits, token\u002Fcost budgets, routing controls, retries and capacity behavior.\"],[\"Retrieval\u002Fdata contract\",\"Defines ingestion, provenance, search, metadata, authorization propagation and where domain authority remains.\"],[\"Agent\u002Ftool contract\",\"Defines runtime lifecycle, tool registration, permissions, approvals, cancellation and trace behavior.\"],[\"Secret and trust-boundary model\",\"Defines credential ownership, storage, process boundaries, rotation and sensitive data paths.\"],[\"Evaluation and telemetry contract\",\"Defines common metrics, traces, datasets\u002Fversion links, logging policy and solution extension points.\"],[\"Lifecycle and compatibility policy\",\"Defines versions, migrations, deprecation, releases, rollback, incident ownership and onboarding.\"]],\"stretched\":false,\"withHeadings\":true},\"type\":\"table\"},{\"id\":\"h-tradeoffs\",\"data\":{\"text\":\"The work is mostly trade-offs, not maximum centralization\",\"level\":2},\"type\":\"header\"},{\"id\":\"tradeoff-comparison\",\"data\":{\"rows\":[{\"id\":\"t1\",\"label\":\"Provider abstraction\",\"values\":{\"pressureA\":\"Stable portable platform API\",\"pressureB\":\"Access to provider-specific capabilities and fast innovation\"}},{\"id\":\"t2\",\"label\":\"Reuse\",\"values\":{\"pressureA\":\"Shared services reduce duplication\",\"pressureB\":\"Isolation and domain autonomy prevent unsafe coupling\"}},{\"id\":\"t3\",\"label\":\"Governance\",\"values\":{\"pressureA\":\"Central policy and auditability\",\"pressureB\":\"Team speed and local experimentation\"}},{\"id\":\"t4\",\"label\":\"Observability\",\"values\":{\"pressureA\":\"Rich traces for debugging and evaluation\",\"pressureB\":\"Privacy, data minimization and logging cost\"}},{\"id\":\"t5\",\"label\":\"Availability\",\"values\":{\"pressureA\":\"Fallback and multi-provider resilience\",\"pressureB\":\"Predictable quality, compliance and data-location guarantees\"}},{\"id\":\"t6\",\"label\":\"Platform scope\",\"values\":{\"pressureA\":\"More reusable capabilities\",\"pressureB\":\"Smaller blast radius and less platform lock-in\"}}],\"title\":\"Common platform trade-offs\",\"layout\":\"table\",\"columns\":[{\"id\":\"pressureA\",\"label\":\"Pressure A\"},{\"id\":\"pressureB\",\"label\":\"Pressure B\"}]},\"type\":\"comparison\"},{\"id\":\"h-adjacent\",\"data\":{\"text\":\"How is this different from adjacent roles?\",\"level\":2},\"type\":\"header\"},{\"id\":\"roles-table\",\"data\":{\"content\":[[\"Role\",\"Primary architectural scope\"],[\"AI Solution Architect\",\"A concrete AI-enabled solution and its end-to-end requirements, boundaries, trade-offs and production acceptance.\"],[\"AI Platform Architect\",\"Reusable AI capabilities and operational\u002Fsecurity contracts consumed across multiple solutions or teams.\"],[\"Enterprise Architect\",\"Organization-wide business\u002Ftechnology portfolio, capability and governance alignment at a broader level.\"],[\"MLOps \u002F LLMOps Architect or specialist\",\"Model and AI lifecycle, deployment, experiments, observability, release and operational practices; may overlap strongly but does not automatically own the whole shared application platform.\"],[\"Platform Engineer \u002F SRE\",\"Implements and operates platform infrastructure, reliability, automation and developer experience; architecture responsibility may be shared with the platform architect.\"],[\"AI \u002F Software Engineer\",\"Implements models, integrations, services, agents, retrieval and product functionality inside the agreed architecture.\"]],\"stretched\":false,\"withHeadings\":true},\"type\":\"table\"},{\"id\":\"p-adjacent-1\",\"data\":{\"text\":\"These boundaries are organizational, not universal. In a small team one person may hold several responsibilities. In a regulated enterprise they may be split across architecture, security, platform, data and operations groups. The useful distinction is the \u003Cstrong>scope of architectural responsibility\u003C\u002Fstrong>, not the job title printed on an org chart.\"},\"type\":\"paragraph\"},{\"id\":\"h-evidence\",\"data\":{\"text\":\"Implementation evidence: how these platform boundaries appear in my own work\",\"level\":2},\"type\":\"header\"},{\"id\":\"evidence-note\",\"data\":{\"body\":\"The following sections describe concrete patterns from my own projects. They are evidence that these architectural boundaries have been implemented or explicitly designed in real code and project systems. They are \u003Cstrong>not\u003C\u002Fstrong> claims that the projects together already constitute a commercially deployed enterprise AI platform.\",\"title\":\"Original implementation evidence\",\"variant\":\"note\"},\"type\":\"callout\"},{\"id\":\"h-ai-client\",\"data\":{\"text\":\"Aaasaasa AI Client: provider, runtime and permission separation\",\"level\":3},\"type\":\"header\"},{\"id\":\"p-ai-client-1\",\"data\":{\"text\":\"Aaasaasa AI Client is a local-first desktop AI workspace built with Nuxt 4, Electron and TypeScript. Its AI Hub deliberately separates \u003Cstrong>agent\u002Fclient, provider, model, connection\u002Fruntime location, permissions and web client\u003C\u002Fstrong> instead of treating them as one configuration value.\"},\"type\":\"paragraph\"},{\"id\":\"p-ai-client-2\",\"data\":{\"text\":\"The implementation includes direct provider adapters, Codex agent runtime integration, local Ollama\u002FLM Studio paths, OpenAI-compatible services, centralized workspace permissions, main-process credential storage, DuckDB, Qdrant\u002Fvector support, PDF\u002Freadability extraction and authenticated MCP-based directory access.\"},\"type\":\"paragraph\"},{\"id\":\"p-ai-client-3\",\"data\":{\"text\":\"Two platform lessons are especially relevant. First, a local runtime is not the same as local inference: a local Codex process can still use a cloud model. Second, automatic routing does not silently fall back from local to paid cloud inference. That makes routing policy and runtime locality explicit rather than inferred from UI labels.\"},\"type\":\"paragraph\"},{\"id\":\"ai-client-evidence-table\",\"data\":{\"content\":[[\"Implemented boundary\",\"Platform-architecture meaning\"],[\"Agent vs provider vs model\",\"Different responsibilities can evolve independently instead of being hidden behind one “AI” selector.\"],[\"Permissions separate from model\",\"Filesystem\u002Ftool authority belongs to the runtime policy, not model capability.\"],[\"Main-process secrets\",\"Credential ownership follows the privileged process boundary rather than the renderer\u002FUI.\"],[\"Provider health and model discovery\",\"Routing and availability are runtime\u002Fplatform concerns.\"],[\"No silent cloud fallback\",\"Cost, locality and data-transfer semantics remain explicit policy decisions.\"]],\"stretched\":false,\"withHeadings\":true},\"type\":\"table\"},{\"id\":\"h-cms\",\"data\":{\"text\":\"Aaasaasa AI CMS: tenant-scoped authorization as a platform boundary\",\"level\":3},\"type\":\"header\"},{\"id\":\"p-cms-1\",\"data\":{\"text\":\"The Aaasaasa AI CMS codebase provides a separate implementation example: tenant-scoped RBAC is represented through roles, permissions and user-role assignments bound to a tenant identifier. System permissions are grouped by capability, and role lookup and updates remain tenant-scoped.\"},\"type\":\"paragraph\"},{\"id\":\"p-cms-2\",\"data\":{\"text\":\"This is not itself proof of a complete AI platform, but it is directly relevant to one of the hardest shared-platform boundaries: a reusable service must preserve \u003Cstrong>who may do what\u003C\u002Fstrong> and \u003Cstrong>for which tenant\u003C\u002Fstrong>. Adding AI inference or retrieval on top of an application platform does not remove that requirement.\"},\"type\":\"paragraph\"},{\"id\":\"p-cms-3\",\"data\":{\"text\":\"The architectural implication is that model gateways, retrieval services and agents should consume established identity\u002Ftenant context rather than inventing a parallel AI-only authorization universe.\"},\"type\":\"paragraph\"},{\"id\":\"h-sot\",\"data\":{\"text\":\"Source of Truth Research Engine: shared retrieval mechanics without shared truth\",\"level\":3},\"type\":\"header\"},{\"id\":\"p-sot-1\",\"data\":{\"text\":\"The Source of Truth Research Engine provides a third implementation example. Different research modes share a common evidence core: Sources, Artifacts, provenance, Claims, Relations, Contradictions, a Reference Model and audit trail. The system also provides local lexical retrieval, optional semantic retrieval, extraction, snapshots and SHA-256-based provenance.\"},\"type\":\"paragraph\"},{\"id\":\"p-sot-2\",\"data\":{\"text\":\"The project explicitly treats search and semantic similarity as discovery signals rather than evidence. A result must be traced back to a concrete source and locator before it can support a claim. This is precisely the distinction an AI platform needs: \u003Cstrong>reusable retrieval machinery can be shared while evidence authority remains governed by the consuming methodology and domain.\u003C\u002Fstrong>\"},\"type\":\"paragraph\"},{\"id\":\"p-sot-3\",\"data\":{\"text\":\"The engine also demonstrates why one shared platform does not require one shared interpretation. Historical, scientific\u002Ftechnical, market-intelligence and monitoring modes can reuse core evidence infrastructure while retaining mode-specific methodology.\"},\"type\":\"paragraph\"},{\"id\":\"evidence-synthesis\",\"data\":{\"body\":\"Across these projects, the reusable pattern is not “one backend for everything.” It is \u003Cstrong>separation of concerns plus explicit contracts\u003C\u002Fstrong>: provider\u002Fmodel\u002Fruntime separation, tenant-aware authorization, credential boundaries, reusable data\u002Fretrieval primitives, provenance, and domain-specific authority. A future integrated platform would need stable contracts between those capabilities rather than direct coupling between codebases.\",\"title\":\"What these implementations demonstrate together\",\"variant\":\"success\"},\"type\":\"callout\"},{\"id\":\"h-frameworks\",\"data\":{\"text\":\"How current architecture guidance supports this platform scope\",\"level\":2},\"type\":\"header\"},{\"id\":\"p-frameworks-1\",\"data\":{\"text\":\"ISO\u002FIEC\u002FIEEE 42010:2022 provides a general discipline for architecture descriptions across software, systems, enterprises and related entities. It does not define an AI Platform Architect, but it reinforces the need to express architectural concerns, relationships and viewpoints rather than reducing architecture to a technology list.\"},\"type\":\"paragraph\"},{\"id\":\"p-frameworks-2\",\"data\":{\"text\":\"NIST AI RMF 1.0 and the Generative AI Profile frame AI risk management across the lifecycle rather than only at model selection time. Governance, mapping, measurement and management are therefore compatible with a platform architecture that carries shared controls and evidence across many consuming workloads.\"},\"type\":\"paragraph\"},{\"id\":\"p-frameworks-3\",\"data\":{\"text\":\"Microsoft’s current AI workload guidance treats application design, data, security, operations, testing\u002Fevaluation and GenAIOps as connected architectural areas. Its current AI Gateway guidance also shows practical platform concerns such as centralized model access, project-specific token limits, quotas and multi-team containment.\"},\"type\":\"paragraph\"},{\"id\":\"p-frameworks-4\",\"data\":{\"text\":\"AWS’s current Generative AI Lens and multi-tenant platform scenario similarly separate foundational platform controls from consuming-application ownership. AWS explicitly notes that a central platform can enforce shared guardrails and auditability while data quality and workload-specific observability still remain responsibilities of consuming applications or data producers.\"},\"type\":\"paragraph\"},{\"id\":\"p-frameworks-5\",\"data\":{\"text\":\"The vendor products differ, but the cross-source pattern is stable: production AI platforms must coordinate identity, data access, models, policy, evaluation, observability, capacity, cost and lifecycle. A GPU cluster or model endpoint covers only part of that responsibility.\"},\"type\":\"paragraph\"},{\"id\":\"h-misconceptions\",\"data\":{\"text\":\"Common misconceptions\",\"level\":2},\"type\":\"header\"},{\"id\":\"misconceptions-table\",\"data\":{\"content\":[[\"Misconception\",\"Why it is wrong\"],[\"“An AI platform is the GPU cluster.”\",\"Compute is one substrate. A platform also needs contracts for identity, model access, data, policy, evaluation, observability and lifecycle.\"],[\"“An AI gateway is just a reverse proxy.”\",\"It may also carry model routing, token quotas, cost attribution, policy enforcement, identity and AI-specific telemetry.\"],[\"“Shared means globally shared.”\",\"A service may be physically shared while logically segmented by tenant, application, region, classification or risk level.\"],[\"“One central vector database becomes the company truth.”\",\"A vector store or retrieval service is infrastructure. Domain authority, freshness, provenance and access remain separate concerns.\"],[\"“Platform evaluation replaces solution evaluation.”\",\"General regression and telemetry cannot define whether a domain-specific answer or action is acceptable.\"],[\"“Provider abstraction should hide every difference.”\",\"Some differences are material capabilities, security semantics or failure modes and must remain visible.\"],[\"“RBAC solves multi-tenancy.”\",\"RBAC controls actions; tenant isolation controls resource boundaries. Both can be required.\"],[\"“AI Platform Architect is just another name for MLOps.”\",\"MLOps\u002FLLMOps is a major overlapping discipline, but shared application\u002Fruntime, identity, gateway, retrieval and tool boundaries can extend beyond model lifecycle operations.\"]],\"stretched\":false,\"withHeadings\":true},\"type\":\"table\"},{\"id\":\"h-failure\",\"data\":{\"text\":\"Failure modes an AI Platform Architect should prevent\",\"level\":2},\"type\":\"header\"},{\"id\":\"failures-table\",\"data\":{\"content\":[[\"Failure mode\",\"Architectural consequence\"],[\"Every team stores its own provider keys\",\"Duplicated secret handling, inconsistent rotation and larger blast radius.\"],[\"Provider abstraction hides required capabilities\",\"Consumers cannot use features they need or silently receive behavior different from assumptions.\"],[\"Shared retrieval ignores tenant\u002Fuser context\",\"Cross-boundary data leakage can occur before the application gets a chance to filter results.\"],[\"Fallback silently changes provider or locality\",\"Cost, compliance, data location and output quality can change without the caller knowing.\"],[\"Agent tools are granted by model choice\",\"A capable model becomes over-privileged because runtime authority is not independently enforced.\"],[\"All prompts\u002Fresponses are logged by default\",\"Observability can create a new sensitive-data repository and compliance problem.\"],[\"Platform owns one generic quality score\",\"Domain failures remain hidden behind platform health metrics.\"],[\"No version contract for platform capabilities\",\"Model\u002Fprovider\u002Fruntime changes break consumers unpredictably.\"],[\"Everything AI-related is centralized\",\"The platform becomes a bottleneck and monolith instead of a reusable capability layer.\"]],\"stretched\":false,\"withHeadings\":true},\"type\":\"table\"},{\"id\":\"h-decision\",\"data\":{\"text\":\"A practical platform-architecture decision sequence\",\"level\":2},\"type\":\"header\"},{\"id\":\"decision-flow\",\"data\":{\"steps\":[{\"label\":\"1. Identify real consumers\",\"description\":\"List solutions, teams, tenants and workloads that would consume the platform; avoid building a platform for hypothetical reuse.\"},{\"label\":\"2. Define the shared boundary\",\"description\":\"Separate cross-cutting mechanics from solution-specific domain authority, workflow and acceptance.\"},{\"label\":\"3. Define identity and isolation first\",\"description\":\"Establish users, services, applications, tenants, regions and data classifications before sharing retrieval or tool capabilities.\"},{\"label\":\"4. Define capability contracts\",\"description\":\"Specify model\u002Fprovider, retrieval, agent\u002Ftool, gateway and telemetry APIs with explicit ownership and versioning.\"},{\"label\":\"5. Decide provider and runtime strategy\",\"description\":\"Choose managed, self-hosted, local or hybrid execution and document fallback, locality and capability semantics.\"},{\"label\":\"6. Design data and retrieval boundaries\",\"description\":\"Define provenance, authorization propagation, corpus ownership, indexing and evidence responsibilities.\"},{\"label\":\"7. Add quotas, secrets and policy\",\"description\":\"Control cost, capacity, credentials, tool permissions, safety controls and blast radius.\"},{\"label\":\"8. Build evaluation and observability contracts\",\"description\":\"Provide platform metrics and tracing while leaving domain ground truth and acceptance to the solution.\"},{\"label\":\"9. Define lifecycle and operations\",\"description\":\"Version capabilities, test upgrades, document deprecation, rollback, incidents, capacity and consumer onboarding.\"},{\"label\":\"10. Validate with more than one consumer\",\"description\":\"A platform claim becomes credible when the shared capability actually serves distinct workloads without forcing them into the same domain model.\"}],\"title\":\"From platform need to operable shared capability\",\"orientation\":\"auto\"},\"type\":\"processFlow\"},{\"id\":\"h-edge\",\"data\":{\"text\":\"Edge cases and limits of the role\",\"level\":2},\"type\":\"header\"},{\"id\":\"p-edge-1\",\"data\":{\"text\":\"A small organization with one AI application may not need a distinct AI platform or platform architect. Premature platforming can create more abstraction than value. The correct architecture may be one well-designed solution with a few reusable modules.\"},\"type\":\"paragraph\"},{\"id\":\"p-edge-2\",\"data\":{\"text\":\"An air-gapped or sovereign deployment changes the provider, update and observability model substantially. Model hosting, artifact distribution, identity integration and telemetry export may all need local equivalents.\"},\"type\":\"paragraph\"},{\"id\":\"p-edge-3\",\"data\":{\"text\":\"Highly regulated or high-consequence workloads may require stronger physical or organizational isolation instead of a logically shared platform. Reuse is never a sufficient reason to weaken a required security boundary.\"},\"type\":\"paragraph\"},{\"id\":\"p-edge-4\",\"data\":{\"text\":\"Managed cloud AI services can remove implementation burden but do not remove architectural accountability. The organization still decides identity, data access, logging, retention, quotas, model eligibility, fallback, evaluation and solution acceptance.\"},\"type\":\"paragraph\"},{\"id\":\"p-edge-5\",\"data\":{\"text\":\"The platform boundary may also differ by modality. Text inference, multimodal generation, speech, computer use and autonomous agents can have different latency, data, permission and observability requirements even when they share provider and identity infrastructure.\"},\"type\":\"paragraph\"},{\"id\":\"h-change\",\"data\":{\"text\":\"What would change this answer?\",\"level\":2},\"type\":\"header\"},{\"id\":\"p-change-1\",\"data\":{\"text\":\"The core definition would change if the organizational scope changes. If the architect owns one workload, the role becomes closer to AI Solution Architect. If the responsibility expands to organization-wide capability strategy, investment, standards and target-state portfolios, it moves toward Enterprise AI Architecture.\"},\"type\":\"paragraph\"},{\"id\":\"p-change-2\",\"data\":{\"text\":\"Implementation guidance changes whenever providers, gateway products, agent protocols, regulatory obligations, model capabilities or deployment constraints change. That is why platform architecture should express stable responsibilities and contracts separately from current vendor mechanisms.\"},\"type\":\"paragraph\"},{\"id\":\"h-checklist\",\"data\":{\"text\":\"AI Platform Architect checklist\",\"level\":2},\"type\":\"header\"},{\"id\":\"checklist-table\",\"data\":{\"content\":[[\"Question\",\"Expected answer\"],[\"Who are the actual platform consumers?\",\"Named solutions, teams or tenant contexts with distinct but overlapping needs.\"],[\"What is genuinely shared?\",\"Explicit capability list, not a vague “AI backend.”\"],[\"What must remain solution-specific?\",\"Domain authority, business workflow, task acceptance and other workload-owned concerns.\"],[\"How are models\u002Fproviders represented?\",\"Versioned provider\u002Fmodel contracts with capabilities and explicit fallback semantics.\"],[\"How is identity propagated?\",\"User\u002Fservice\u002Fapplication\u002Ftenant context survives every privileged request path.\"],[\"How is tenant isolation enforced?\",\"Resource scoping is separate from role permission checks.\"],[\"How are secrets handled?\",\"Privileged storage, rotation, limited exposure and auditable ownership.\"],[\"How does retrieval preserve authority?\",\"Shared mechanics with authorization, provenance and domain-owned evidence rules.\"],[\"How are tools and agents constrained?\",\"Runtime permissions, bounded tool contracts, approvals, cancellation and traceability.\"],[\"How are cost and capacity controlled?\",\"Quotas, token\u002Frate controls, usage attribution and overload behavior.\"],[\"How is quality measured?\",\"Platform regression\u002Fevaluation plus solution-specific ground truth and acceptance.\"],[\"How are changes rolled out?\",\"Versioning, compatibility, migration, deprecation, rollback and incident ownership.\"]],\"stretched\":false,\"withHeadings\":true},\"type\":\"table\"},{\"id\":\"h-conclusion\",\"data\":{\"text\":\"Conclusion\",\"level\":2},\"type\":\"header\"},{\"id\":\"p-conclusion-1\",\"data\":{\"text\":\"An AI Platform Architect is responsible for the reusable architecture \u003Cstrong>between AI capabilities and the solutions that consume them\u003C\u002Fstrong>. The role defines how models, providers, retrieval, agents, tools, identity, tenants, secrets, evaluation, observability, quotas and runtime operations become dependable platform services rather than repeated one-off integrations.\"},\"type\":\"paragraph\"},{\"id\":\"p-conclusion-2\",\"data\":{\"text\":\"The difficult part is not maximizing reuse. It is choosing the correct boundary. A strong platform standardizes mechanics, policy and operations where multiple consumers genuinely benefit, while preserving solution-specific data authority, business logic, security requirements and acceptance criteria.\"},\"type\":\"paragraph\"},{\"id\":\"p-conclusion-3\",\"data\":{\"text\":\"That distinction also explains the relationship with AI Solution Architecture: \u003Cstrong>the solution architect makes one AI-enabled system fit its purpose; the platform architect makes shared AI capabilities safe, reusable, operable and evolvable across many such systems.\u003C\u002Fstrong>\"},\"type\":\"paragraph\"},{\"id\":\"h-related\",\"data\":{\"text\":\"Related canonical knowledge\",\"level\":2},\"type\":\"header\"},{\"id\":\"p-related-1\",\"data\":{\"text\":\"This article sits after the canonical foundations on generative AI components, ADR versus NFR, and AI Solution Architecture. Those concepts are prerequisites because a platform exists to provide reusable system capabilities and to encode architectural decisions against explicit quality and operational requirements.\"},\"type\":\"paragraph\"},{\"id\":\"p-related-2\",\"data\":{\"text\":\"Retrieval-Augmented Generation is one example of a capability that may be offered through a platform, but the platform should not collapse retrieval infrastructure, domain knowledge and answer validity into one concept.\"},\"type\":\"paragraph\"},{\"id\":\"related-rag\",\"data\":{\"link\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"What Is RAG? The Simplest Explanation of How It Works\",\"description\":\"Canonical introduction to retrieval-augmented generation and the boundary between model generation and external knowledge retrieval.\"}},\"type\":\"linkTool\"},{\"id\":\"p-related-3\",\"data\":{\"text\":\"Agent protocols, tenant isolation, AI governance, model routing, Context Engineering and MLOps\u002FLLMOps are downstream or adjacent knowledge nodes. They become easier to reason about once the platform boundary is explicit.\"},\"type\":\"paragraph\"},{\"id\":\"h-faq\",\"data\":{\"text\":\"Frequently asked questions\",\"level\":2},\"type\":\"header\"},{\"id\":\"faq\",\"data\":{\"items\":[{\"id\":\"faq-1\",\"answer\":\"No. The solution architect focuses on one concrete AI-enabled solution. The platform architect focuses on reusable AI capabilities, controls and operational contracts that can support multiple solutions.\",\"question\":\"Is an AI Platform Architect the same as an AI Solution Architect?\"},{\"id\":\"faq-2\",\"answer\":\"No. A platform can use managed cloud models, self-hosted models, local inference or a hybrid strategy. The architecture must make provider, locality, identity, routing, data and operational consequences explicit.\",\"question\":\"Does an AI platform need to host its own models?\"},{\"id\":\"faq-3\",\"answer\":\"Usually not. A gateway can be an important platform component, but a complete platform also needs contracts for identity, secrets, data\u002Fretrieval, evaluation, observability, lifecycle and operational ownership.\",\"question\":\"Is an AI gateway enough to be an AI platform?\"},{\"id\":\"faq-4\",\"answer\":\"Retrieval mechanics can often be shared, but domain authority, authorization, freshness, evidence sufficiency and corpus ownership should remain explicit. Shared infrastructure does not imply shared truth.\",\"question\":\"Should retrieval be centralized?\"},{\"id\":\"faq-5\",\"answer\":\"No. Platform evaluation can test shared capabilities and regressions. Each solution still needs task-specific ground truth, acceptance criteria and domain quality thresholds.\",\"question\":\"Does platform evaluation replace application evaluation?\"},{\"id\":\"faq-6\",\"answer\":\"No. RBAC determines what an identity may do. Tenant isolation determines which tenant's resources the identity may act on. A platform often needs both.\",\"question\":\"Is multi-tenancy just RBAC?\"}],\"title\":\"AI Platform Architect FAQ\"},\"type\":\"faq\"},{\"id\":\"h-glossary\",\"data\":{\"text\":\"Glossary\",\"level\":2},\"type\":\"header\"},{\"id\":\"glossary\",\"data\":{\"title\":\"Key AI platform architecture terms\",\"entries\":[{\"term\":\"AI platform\",\"anchor\":\"ai-platform\",\"definition\":\"A reusable set of AI-related technical and operational capabilities consumed by multiple applications, teams or tenant contexts.\"},{\"term\":\"AI gateway\",\"anchor\":\"ai-gateway\",\"definition\":\"A gateway layer for AI endpoints that may add authentication, routing, quotas, policy, retries, cost attribution and AI-specific telemetry beyond basic proxying.\"},{\"term\":\"Provider adapter\",\"anchor\":\"provider-adapter\",\"definition\":\"A component that maps a platform contract to a model provider's API, capabilities, health and failure semantics.\"},{\"term\":\"Tenant isolation\",\"anchor\":\"tenant-isolation\",\"definition\":\"The boundary that prevents one tenant context from accessing another tenant's resources, independent of role permissions.\"},{\"term\":\"Capability contract\",\"anchor\":\"capability-contract\",\"definition\":\"A versioned interface and behavioral agreement describing what a shared platform service provides and what the consumer must supply or own.\"},{\"term\":\"Grounding \u002F retrieval service\",\"anchor\":\"grounding-service\",\"definition\":\"Shared mechanics for finding and supplying external information to an AI workload; it does not automatically define which information is authoritative for a domain.\"},{\"term\":\"Evaluation harness\",\"anchor\":\"evaluation-harness\",\"definition\":\"Reusable infrastructure for running tests, datasets, model\u002Fprompt versions and metrics; domain acceptance remains solution-specific.\"},{\"term\":\"Control plane\",\"anchor\":\"control-plane\",\"definition\":\"The configuration and governance layer that manages platform capabilities, identities, policies, quotas, versions and deployment state.\"}]},\"type\":\"glossary\"},{\"id\":\"h-sources\",\"data\":{\"text\":\"Primary sources and current architecture guidance\",\"level\":2},\"type\":\"header\"},{\"id\":\"p-sources-note\",\"data\":{\"text\":\"The sources below support the general architecture and production-platform claims. The Aaasaasa AI Client, Aaasaasa AI CMS and Source of Truth Research Engine sections are explicitly original implementation evidence. Current-state external references were checked on 8 October 2026.\"},\"type\":\"paragraph\"},{\"id\":\"src-iso-42010\",\"data\":{\"link\":\"https:\u002F\u002Fwww.iso.org\u002Fstandard\u002F74393.html\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"ISO\u002FIEC\u002FIEEE 42010:2022 — Architecture Description\",\"description\":\"Current published international standard for architecture-description concepts and relationships.\"}},\"type\":\"linkTool\"},{\"id\":\"src-nist-rmf\",\"data\":{\"link\":\"https:\u002F\u002Fwww.nist.gov\u002Fitl\u002Fai-risk-management-framework\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"NIST AI Risk Management Framework\",\"description\":\"NIST's AI RMF resources and current status; AI RMF 1.0 is under revision as of October 2026.\"}},\"type\":\"linkTool\"},{\"id\":\"src-nist-gai\",\"data\":{\"link\":\"https:\u002F\u002Fwww.nist.gov\u002Fpublications\u002Fartificial-intelligence-risk-management-framework-generative-artificial-intelligence\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"NIST AI 600-1 — Generative AI Profile\",\"description\":\"Generative AI profile for applying AI risk-management considerations across the AI lifecycle.\"}},\"type\":\"linkTool\"},{\"id\":\"src-ms-ai\",\"data\":{\"link\":\"https:\u002F\u002Flearn.microsoft.com\u002Fen-us\u002Fazure\u002Fwell-architected\u002Fai\u002Fget-started\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Microsoft Azure Well-Architected — AI Workloads\",\"description\":\"Current architectural guidance covering AI application, data, operations, evaluation, responsible AI and lifecycle concerns.\"}},\"type\":\"linkTool\"},{\"id\":\"src-ms-principles\",\"data\":{\"link\":\"https:\u002F\u002Flearn.microsoft.com\u002Fen-us\u002Fazure\u002Fwell-architected\u002Fai\u002Fdesign-principles\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Microsoft — Design Principles for AI Workloads\",\"description\":\"Current guidance on identity segmentation, security boundaries, telemetry, performance, data and platform trade-offs.\"}},\"type\":\"linkTool\"},{\"id\":\"src-ms-gateway\",\"data\":{\"link\":\"https:\u002F\u002Flearn.microsoft.com\u002Fen-us\u002Fazure\u002Fai-foundry\u002Fconfiguration\u002Fenable-ai-api-management-gateway-portal?view=foundry\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Microsoft Foundry — AI Gateway Architecture\",\"description\":\"Current AI Gateway guidance for shared project access, token containment, quotas and governance.\"}},\"type\":\"linkTool\"},{\"id\":\"src-ms-gateway-guide\",\"data\":{\"link\":\"https:\u002F\u002Flearn.microsoft.com\u002Fen-us\u002Fazure\u002Farchitecture\u002Fai-ml\u002Fguide\u002Fazure-openai-gateway-guide\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Azure Architecture Center — Access Models Through a Gateway\",\"description\":\"Architecture guidance for centralized model access, routing, throttling, failover and client\u002Fplatform responsibilities.\"}},\"type\":\"linkTool\"},{\"id\":\"src-aws-genai\",\"data\":{\"link\":\"https:\u002F\u002Fdocs.aws.amazon.com\u002Fwellarchitected\u002Flatest\u002Fgenerative-ai-lens\u002F\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"AWS Well-Architected — Generative AI Lens\",\"description\":\"Current production architecture guidance for generative AI workloads across security, reliability, operations, performance and cost.\"}},\"type\":\"linkTool\"},{\"id\":\"src-aws-multitenant\",\"data\":{\"link\":\"https:\u002F\u002Fdocs.aws.amazon.com\u002Fwellarchitected\u002Flatest\u002Fgenerative-ai-lens\u002Fmulti-tenant-generative-ai-platform-scenario.html\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"AWS — Multi-tenant Generative AI Platform Scenario\",\"description\":\"Current example separating central platform controls and auditability from consuming-application data quality and workload-specific responsibilities.\"}},\"type\":\"linkTool\"},{\"id\":\"src-aws-agentic\",\"data\":{\"link\":\"https:\u002F\u002Fdocs.aws.amazon.com\u002Fwellarchitected\u002Flatest\u002Fagentic-ai-lens\u002Fdesign-principles.html\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"AWS Well-Architected — Agentic AI Design Principles\",\"description\":\"Current guidance on bounded agent authority, traceability, versioned behavior, explicit contracts and human oversight.\"}},\"type\":\"linkTool\"},{\"id\":\"src-aws-observability\",\"data\":{\"link\":\"https:\u002F\u002Fdocs.aws.amazon.com\u002FAmazonCloudWatch\u002Flatest\u002Fmonitoring\u002FGenAI-observability.html\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"AWS CloudWatch — Generative AI Observability\",\"description\":\"Current observability capabilities and production metrics for models, agents, knowledge bases, tools and cost\u002Flatency\u002Ferror analysis.\"}},\"type\":\"linkTool\"}],\"version\":\"2.31.0\"}",{"time":1242,"blocks":1243,"version":2024},1791476955677,[1244,1247,1251,1255,1259,1262,1265,1268,1271,1295,1298,1301,1304,1307,1329,1332,1335,1338,1341,1371,1375,1378,1381,1384,1387,1391,1394,1397,1400,1403,1406,1409,1412,1415,1418,1421,1424,1427,1430,1433,1436,1439,1442,1445,1448,1451,1454,1457,1460,1463,1466,1469,1472,1475,1478,1481,1485,1504,1507,1510,1543,1546,1573,1576,1598,1601,1604,1608,1611,1614,1617,1620,1641,1644,1647,1650,1653,1656,1659,1662,1665,1669,1672,1675,1678,1681,1684,1687,1690,1720,1723,1756,1759,1793,1796,1799,1802,1805,1808,1811,1814,1817,1820,1823,1865,1868,1871,1874,1877,1880,1883,1886,1893,1896,1899,1921,1924,1952,1955,1958,1964,1970,1976,1982,1988,1994,2000,2006,2012,2018],{"id":215,"data":1245,"type":218},{"text":1246},"An \u003Cstrong>AI Platform Architect\u003C\u002Fstrong> designs the reusable AI foundation through which multiple applications, teams, or tenant contexts access models, data and retrieval, agent and tool runtimes, identity and permissions, evaluation, observability, quotas, secrets, and deployment capabilities. The role is broader than infrastructure but narrower than owning every AI-enabled product: its central responsibility is deciding \u003Cstrong>what should be shared, how shared capabilities are governed and isolated, and what must remain solution-specific\u003C\u002Fstrong>.",{"id":220,"data":1248,"type":225},{"body":1249,"title":1250,"variant":224},"\u003Cstrong>An AI Platform Architect designs the shared technical and operational substrate for AI systems.\u003C\u002Fstrong> Instead of architecting one assistant or one workflow, the role defines reusable contracts and boundaries for model\u002Fprovider access, gateways and routing, retrieval services, agent runtimes, tool access, identity and tenant isolation, secrets, evaluation, telemetry, deployment and lifecycle management.","Direct answer",{"id":227,"data":1252,"type":225},{"body":1253,"title":1254,"variant":231},"\u003Cstrong>AI Platform Architect is a practical role label, not a universally standardized job title.\u003C\u002Fstrong> ISO\u002FIEC\u002FIEEE 42010:2022 defines concepts for architecture descriptions, not this role. Different organizations may split these responsibilities among platform architects, solution architects, enterprise architects, security architects, MLOps\u002FLLMOps specialists and platform engineering teams. This article uses the term for the architecture responsibility over a reusable AI platform layer.","Terminology note",{"id":233,"data":1256,"type":225},{"body":1257,"title":1258,"variant":231},"The stable architectural principles here are vendor-neutral. Current Microsoft, AWS and NIST guidance is used as external implementation and governance evidence. NIST states that AI RMF 1.0 is being revised; vendor platform features, gateway products, agent runtimes and model capabilities evolve faster than the architectural principles, so version-sensitive implementation choices must be rechecked before deployment.","Current-source note — 8 October 2026",{"id":238,"data":1260,"type":243},{"title":1261,"maxLevel":241,"minLevel":242},"Contents",{"id":245,"data":1263,"type":42},{"text":1264,"level":242},"What does an AI Platform Architect actually architect?",{"id":249,"data":1266,"type":218},{"text":1267},"The object of the work is the \u003Cstrong>platform\u003C\u002Fstrong>: a set of shared capabilities that reduces repeated integration work while preserving explicit security, data and operational boundaries. A platform can expose model access, provider adapters, retrieval primitives, agent execution, tool brokers, policy enforcement, evaluation, telemetry and deployment services to many consuming solutions.",{"id":253,"data":1269,"type":218},{"text":1270},"The platform is not valuable merely because components are centralized. It is valuable when consumers receive stable capabilities with clear contracts, ownership, isolation, observability and lifecycle rules. The key architectural question is therefore not “Which model should everyone use?” but \u003Cstrong>“Which responsibilities can be safely standardized and reused without erasing the requirements of each solution?”\u003C\u002Fstrong>.",{"id":257,"data":1272,"type":299},{"rows":1273,"title":1289,"layout":291,"columns":1290},[1274,1277,1280,1283,1286],{"id":261,"label":1275,"values":1276},"Primary scope",{"platform":264,"solution":265},{"id":267,"label":1278,"values":1279},"Main question",{"platform":270,"solution":271},{"id":273,"label":1281,"values":1282},"Data authority",{"platform":276,"solution":277},{"id":279,"label":1284,"values":1285},"Evaluation",{"platform":282,"solution":283},{"id":285,"label":1287,"values":1288},"Lifecycle",{"platform":288,"solution":289},"Solution architecture and platform architecture solve different scope problems",[1291,1293],{"id":294,"label":1292},"AI Solution Architect",{"id":297,"label":1294},"AI Platform Architect",{"id":301,"data":1296,"type":42},{"text":1297,"level":242},"The simplest example",{"id":305,"data":1299,"type":218},{"text":1300},"Imagine an organization has five AI-enabled products: an internal document assistant, a customer-support copilot, a software-engineering agent, a contract review workflow and a product-search assistant. Each product could independently integrate model APIs, keep credentials, implement retries, collect token metrics, create retrieval code and build its own tool permissions.",{"id":309,"data":1302,"type":218},{"text":1303},"That duplication is expensive and dangerous when every team invents a different security and operational model. A shared platform can instead offer approved provider connections, model discovery, quotas, credentials, tenant-aware access, common telemetry, reusable retrieval services and an agent\u002Ftool runtime contract.",{"id":313,"data":1305,"type":218},{"text":1306},"But the platform must stop at the correct boundary. The contract-review solution may require legal-document authority and citation rules that the software agent does not. The product-search assistant may need commerce-specific freshness and authorization rules. \u003Cstrong>Reusable infrastructure does not make all domain truth reusable.\u003C\u002Fstrong>",{"id":317,"data":1308,"type":340},{"steps":1309,"title":1328,"orientation":339},[1310,1313,1316,1319,1322,1325],{"label":1311,"description":1312},"1. Consumer identifies itself","The calling application, user, service, team or tenant enters through an authenticated identity and explicit scope.",{"label":1314,"description":1315},"2. Platform policy applies","Gateway and policy layers determine allowed providers, models, quotas, data paths, tools and execution modes.",{"label":1317,"description":1318},"3. Shared capability executes","The request may use inference, retrieval, agent runtime, tool access or another reusable platform service.",{"label":1320,"description":1321},"4. Solution-specific context remains authoritative","The consuming solution supplies domain rules, user intent, data authority, task-specific constraints and acceptance logic.",{"label":1323,"description":1324},"5. Telemetry and evidence are captured","The platform records identity, route, model\u002Fprovider, latency, cost, errors, tool activity and other permitted observability signals.",{"label":1326,"description":1327},"6. Result returns under the solution contract","The solution remains responsible for whether the output is acceptable for its user and domain.","A shared AI request path",{"id":342,"data":1330,"type":42},{"text":1331,"level":242},"Where the simple example stops",{"id":346,"data":1333,"type":218},{"text":1334},"Centralization is not automatically architecture. A single endpoint in front of several model APIs is useful, but it does not by itself create an AI platform. A production platform also needs identity boundaries, capability contracts, provider health and lifecycle handling, quotas, secret ownership, observability, compatibility rules, security controls, release discipline and clear operational responsibility.",{"id":350,"data":1336,"type":218},{"text":1337},"The opposite failure is also common: putting every prompt, vector index, business rule, agent and application workflow into one “AI backend.” That creates a monolith whose shared status is accidental rather than architectural. \u003Cstrong>A platform should standardize cross-cutting capabilities, not absorb domain ownership merely because AI is involved.\u003C\u002Fstrong>",{"id":354,"data":1339,"type":42},{"text":1340,"level":242},"The most important platform decision: shared versus solution-specific",{"id":358,"data":1342,"type":291},{"content":1343,"stretched":43,"withHeadings":14},[1344,1348,1352,1356,1360,1364,1367],[1345,1346,1347],"Capability area","Good candidate for shared platform ownership","Usually remains solution-specific",[1349,1350,1351],"Model access","Approved provider connections, adapters, credentials, health, routing primitives, quotas","Task-specific model acceptance, prompt behavior, quality threshold",[1353,1354,1355],"Retrieval","Ingestion primitives, extraction, indexing, search APIs, provenance contracts, authorization hooks","Authoritative corpus, freshness rules, domain metadata, evidence sufficiency",[1357,1358,1359],"Agents and tools","Runtime lifecycle, tool registry\u002Fbroker, permission enforcement, tracing, cancellation","Business workflow, allowed action semantics, escalation policy, task success",[1361,1362,1363],"Security","Identity integration, secret storage, policy enforcement, audit contracts, tenant isolation mechanisms","Data classification, business authorization rules, domain-specific risk acceptance",[1284,1365,1366],"Harness, dataset\u002Fversion mechanics, telemetry, experiment\u002Frelease workflow","Ground truth, domain test set, acceptance threshold, user outcome",[1368,1369,1370],"Operations","Deployment pattern, health, metrics, incident integration, capacity controls","Solution SLOs where they differ, business continuity impact, workload-specific runbooks",{"id":389,"data":1372,"type":225},{"body":1373,"title":1374,"variant":393},"\u003Cstrong>Share mechanics and controls where reuse is real; keep authority and acceptance where the domain owns them.\u003C\u002Fstrong> This prevents two opposite errors: duplicated infrastructure everywhere, and a central platform that falsely becomes the owner of every application's data, policy and quality.","Platform principle",{"id":395,"data":1376,"type":42},{"text":1377,"level":242},"Architecture responsibility map",{"id":399,"data":1379,"type":42},{"text":1380,"level":241},"1. Model and provider access",{"id":403,"data":1382,"type":218},{"text":1383},"A platform architect defines how consumers discover and invoke models without forcing every application to hard-code one provider. This includes provider adapters, model identifiers, capability metadata, authentication, health checks, endpoint configuration, request normalization and compatibility behavior.",{"id":407,"data":1385,"type":218},{"text":1386},"Provider abstraction must remain honest. Different providers expose different context limits, tool semantics, structured-output behavior, multimodal capabilities, safety controls, caching, pricing and failure modes. A good abstraction creates a stable platform contract while preserving access to capabilities that cannot be meaningfully flattened.",{"id":411,"data":1388,"type":225},{"body":1389,"title":1390,"variant":415},"A lowest-common-denominator API can make migration easier but can also erase capabilities that matter. The architecture should define which features are portable, which are provider-specific and how consumers discover that difference.","Do not confuse abstraction with pretending providers are identical",{"id":417,"data":1392,"type":42},{"text":1393,"level":241},"2. Gateway, routing, quotas and cost controls",{"id":421,"data":1395,"type":218},{"text":1396},"A shared AI gateway can centralize authentication, routing, throttling, retries, token limits, usage attribution and policy enforcement. Microsoft’s current AI Gateway guidance explicitly treats token-per-minute limits, quotas and multi-project containment as platform concerns; AWS likewise exposes account and model quotas and centralized controls.",{"id":425,"data":1398,"type":218},{"text":1399},"The gateway is therefore more than a reverse proxy when it carries AI-specific policy and operational semantics. But it should not silently make business decisions. A routing policy may prefer a healthy local model, a lower-cost provider or a regionally compliant endpoint; whether that route is acceptable for a particular task is still a contract between platform and solution.",{"id":429,"data":1401,"type":218},{"text":1402},"Routing also needs failure semantics. If the preferred model is unavailable, the platform must know whether fallback is permitted, whether a cloud route requires explicit consent, whether a lower-capability model is valid and how the decision is surfaced to observability.",{"id":433,"data":1404,"type":42},{"text":1405,"level":241},"3. Shared data, retrieval and grounding services",{"id":437,"data":1407,"type":218},{"text":1408},"Retrieval services are strong platform candidates because parsing, chunking, indexing, lexical search, semantic search, metadata filtering, provenance and citation mechanics are reusable. However, the platform must not confuse a shared retrieval engine with a shared source of truth.",{"id":441,"data":1410,"type":218},{"text":1411},"A solution still owns questions such as: Which corpus is authoritative? Which version is valid? Can this user see this document? How fresh must the data be? What counts as sufficient evidence? Can an answer be generated when retrieval fails? Those are domain and solution requirements even when the platform supplies the retrieval machinery.",{"id":445,"data":1413,"type":218},{"text":1414},"This boundary is especially important in multi-tenant systems. A technically shared index or vector service does not justify cross-tenant visibility. Authorization context must be preserved through retrieval, not added only after search results have already crossed the boundary.",{"id":449,"data":1416,"type":42},{"text":1417,"level":241},"4. Agent and tool runtime",{"id":453,"data":1419,"type":218},{"text":1420},"Agentic systems add reusable runtime concerns: thread\u002Fsession lifecycle, planning loops, tool registration, tool invocation, cancellation, timeouts, human approvals, memory\u002Fstate interfaces, remote-agent protocols and trace correlation. A platform can provide these mechanics so each product does not rebuild them.",{"id":457,"data":1422,"type":218},{"text":1423},"The platform must also keep tool permission separate from model capability. A model being capable of generating a shell command does not mean the runtime should allow shell execution. The permission boundary belongs to the application\u002Fruntime architecture and must be enforceable independently of the model.",{"id":461,"data":1425,"type":218},{"text":1426},"Current AWS Agentic AI guidance emphasizes bounded agents, explicit authority, end-to-end tracing, versioned behavioral artifacts and human oversight proportionate to consequence. Those are platform-enabling concerns, but the consuming solution still defines what actions are legitimate for its domain.",{"id":465,"data":1428,"type":42},{"text":1429,"level":241},"5. Identity, tenant isolation and authorization",{"id":469,"data":1431,"type":218},{"text":1432},"AI platforms often sit in front of high-value models, proprietary data and action-capable tools. Authentication is therefore only the beginning. The architecture must carry user, service, application and tenant context through every privileged operation that needs it.",{"id":473,"data":1434,"type":218},{"text":1435},"\u003Cstrong>RBAC and tenant isolation solve different problems.\u003C\u002Fstrong> RBAC answers what an identity may do; tenant isolation answers which tenant’s resources that identity may act on. A platform that checks roles but loses tenant context can still expose the wrong data.",{"id":477,"data":1437,"type":218},{"text":1438},"Microsoft’s current AI workload guidance explicitly recommends identity segmentation and authorization-aware access to content. AWS’s multi-tenant generative AI platform guidance similarly treats logical isolation, centralized controls and auditability as platform concerns.",{"id":481,"data":1440,"type":42},{"text":1441,"level":241},"6. Secrets, credentials and trust boundaries",{"id":485,"data":1443,"type":218},{"text":1444},"A platform should define who owns provider keys, remote bearer tokens, signing material and tool credentials, where they are stored, which process can access them, how they are rotated and whether they can ever reach a browser or untrusted renderer.",{"id":489,"data":1446,"type":218},{"text":1447},"This is an architectural boundary, not an implementation detail. If every consuming application copies provider credentials into its own configuration, the organization has duplicated both operational burden and blast radius. Centralization can reduce that risk only if the platform itself has narrower, auditable access paths.",{"id":493,"data":1449,"type":42},{"text":1450,"level":241},"7. Evaluation, observability and auditability",{"id":497,"data":1452,"type":218},{"text":1453},"A reusable platform can provide evaluation harnesses, trace IDs, model\u002Fprovider metadata, token and cost metrics, latency, error rates, prompt\u002Fmodel version linkage, agent\u002Ftool traces and controlled logging. AWS and Microsoft both treat observability and evaluation as core production concerns for AI workloads.",{"id":501,"data":1455,"type":218},{"text":1456},"Platform evaluation and solution evaluation must remain separate. A platform can verify that an endpoint is healthy, a model version passes a general regression suite and traces are complete. It cannot decide that a legal answer, medical workflow or product recommendation is acceptable without domain-specific ground truth and acceptance criteria.",{"id":505,"data":1458,"type":218},{"text":1459},"Logging also creates a privacy boundary. Prompt and response logs may contain sensitive or proprietary data. The platform architect must therefore decide what is logged, redacted, sampled, retained and accessible rather than assuming that more telemetry is always safer.",{"id":509,"data":1461,"type":42},{"text":1462,"level":241},"8. Runtime, deployment and locality",{"id":513,"data":1464,"type":218},{"text":1465},"A platform architect decides how shared AI capabilities are deployed and reached: managed cloud services, self-hosted endpoints, local inference, hybrid routing, containerized services, desktop runtimes, private networking or air-gapped environments. The important distinction is between \u003Cstrong>where the control\u002Fruntime process runs\u003C\u002Fstrong> and \u003Cstrong>where inference and data processing actually occur\u003C\u002Fstrong>.",{"id":517,"data":1467,"type":218},{"text":1468},"A local client may still call a cloud model. A cloud control plane may route to an on-premises model. A remote agent may execute tools inside a customer network. Architectural diagrams must therefore show trust and data-flow boundaries rather than using “local” and “cloud” as vague labels.",{"id":521,"data":1470,"type":42},{"text":1471,"level":241},"9. Platform lifecycle, compatibility and onboarding",{"id":525,"data":1473,"type":218},{"text":1474},"Reusable capability becomes a platform only when consumers can depend on it over time. That requires versioned contracts, migration rules, compatibility policy, deprecation, release testing, rollback, incident ownership, capacity planning, documentation and a path for onboarding new teams or applications.",{"id":529,"data":1476,"type":218},{"text":1477},"Fast-moving AI ecosystems make this particularly important. Model names, SDKs, protocol versions, provider APIs and safety capabilities change independently. A platform must absorb some of that volatility without hiding changes that materially affect a solution’s behavior.",{"id":533,"data":1479,"type":42},{"text":1480,"level":242},"A practical control-plane \u002F execution-plane \u002F solution-plane model",{"id":537,"data":1482,"type":225},{"body":1483,"title":1484,"variant":231},"The three-plane model below is a practical way to reason about responsibilities; it is not an ISO, NIST, Microsoft or AWS standard. Its purpose is to make ownership boundaries explicit.","Proposed architecture model",{"id":542,"data":1486,"type":291},{"content":1487,"stretched":43,"withHeadings":14},[1488,1492,1496,1500],[1489,1490,1491],"Plane","Typical responsibilities","Should not silently own",[1493,1494,1495],"Platform control plane","Provider registry, model policy, quotas, tenant configuration, identities, secrets, routing rules, capability versions, deployment configuration","Application business logic or domain truth",[1497,1498,1499],"Platform execution\u002Fdata plane","Inference requests, retrieval operations, agent\u002Ftool execution, extraction, indexing, telemetry emission, policy enforcement","Cross-tenant access merely because infrastructure is shared",[1501,1502,1503],"Solution plane","User workflow, prompts\u002Finstructions, authoritative corpus selection, domain authorization, business rules, task evaluation and acceptance","Low-level provider integration that the platform explicitly owns",{"id":562,"data":1505,"type":218},{"text":1506},"This separation helps diagnose platform drift. If an application must know every provider-specific credential and endpoint, the platform contract is too thin. If the platform decides which customer record is legally authoritative or whether a domain answer is acceptable, the platform has crossed into solution ownership.",{"id":566,"data":1508,"type":42},{"text":1509,"level":242},"What should an AI Platform Architect produce?",{"id":570,"data":1511,"type":291},{"content":1512,"stretched":43,"withHeadings":14},[1513,1516,1519,1522,1525,1528,1531,1534,1537,1540],[1514,1515],"Architecture artifact","Purpose",[1517,1518],"Platform capability map","Defines what the platform provides, who consumes it and which capabilities remain outside scope.",[1520,1521],"Provider\u002Fmodel contract","Defines providers, models, capabilities, abstraction boundaries, route metadata and fallback semantics.",[1523,1524],"Identity and tenancy model","Defines user\u002Fservice\u002Fapplication identity, tenant context, RBAC\u002FABAC hooks and resource isolation.",[1526,1527],"Gateway and quota policy","Defines rate limits, token\u002Fcost budgets, routing controls, retries and capacity behavior.",[1529,1530],"Retrieval\u002Fdata contract","Defines ingestion, provenance, search, metadata, authorization propagation and where domain authority remains.",[1532,1533],"Agent\u002Ftool contract","Defines runtime lifecycle, tool registration, permissions, approvals, cancellation and trace behavior.",[1535,1536],"Secret and trust-boundary model","Defines credential ownership, storage, process boundaries, rotation and sensitive data paths.",[1538,1539],"Evaluation and telemetry contract","Defines common metrics, traces, datasets\u002Fversion links, logging policy and solution extension points.",[1541,1542],"Lifecycle and compatibility policy","Defines versions, migrations, deprecation, releases, rollback, incident ownership and onboarding.",{"id":604,"data":1544,"type":42},{"text":1545,"level":242},"The work is mostly trade-offs, not maximum centralization",{"id":608,"data":1547,"type":299},{"rows":1548,"title":1567,"layout":291,"columns":1568},[1549,1552,1555,1558,1561,1564],{"id":612,"label":1550,"values":1551},"Provider abstraction",{"pressureA":615,"pressureB":616},{"id":618,"label":1553,"values":1554},"Reuse",{"pressureA":621,"pressureB":622},{"id":624,"label":1556,"values":1557},"Governance",{"pressureA":627,"pressureB":628},{"id":630,"label":1559,"values":1560},"Observability",{"pressureA":633,"pressureB":634},{"id":636,"label":1562,"values":1563},"Availability",{"pressureA":639,"pressureB":640},{"id":642,"label":1565,"values":1566},"Platform scope",{"pressureA":645,"pressureB":646},"Common platform trade-offs",[1569,1571],{"id":650,"label":1570},"Pressure A",{"id":653,"label":1572},"Pressure B",{"id":656,"data":1574,"type":42},{"text":1575,"level":242},"How is this different from adjacent roles?",{"id":660,"data":1577,"type":291},{"content":1578,"stretched":43,"withHeadings":14},[1579,1582,1584,1586,1589,1592,1595],[1580,1581],"Role","Primary architectural scope",[1292,1583],"A concrete AI-enabled solution and its end-to-end requirements, boundaries, trade-offs and production acceptance.",[1294,1585],"Reusable AI capabilities and operational\u002Fsecurity contracts consumed across multiple solutions or teams.",[1587,1588],"Enterprise Architect","Organization-wide business\u002Ftechnology portfolio, capability and governance alignment at a broader level.",[1590,1591],"MLOps \u002F LLMOps Architect or specialist","Model and AI lifecycle, deployment, experiments, observability, release and operational practices; may overlap strongly but does not automatically own the whole shared application platform.",[1593,1594],"Platform Engineer \u002F SRE","Implements and operates platform infrastructure, reliability, automation and developer experience; architecture responsibility may be shared with the platform architect.",[1596,1597],"AI \u002F Software Engineer","Implements models, integrations, services, agents, retrieval and product functionality inside the agreed architecture.",{"id":683,"data":1599,"type":218},{"text":1600},"These boundaries are organizational, not universal. In a small team one person may hold several responsibilities. In a regulated enterprise they may be split across architecture, security, platform, data and operations groups. The useful distinction is the \u003Cstrong>scope of architectural responsibility\u003C\u002Fstrong>, not the job title printed on an org chart.",{"id":687,"data":1602,"type":42},{"text":1603,"level":242},"Implementation evidence: how these platform boundaries appear in my own work",{"id":691,"data":1605,"type":225},{"body":1606,"title":1607,"variant":231},"The following sections describe concrete patterns from my own projects. They are evidence that these architectural boundaries have been implemented or explicitly designed in real code and project systems. They are \u003Cstrong>not\u003C\u002Fstrong> claims that the projects together already constitute a commercially deployed enterprise AI platform.","Original implementation evidence",{"id":696,"data":1609,"type":42},{"text":1610,"level":241},"Aaasaasa AI Client: provider, runtime and permission separation",{"id":700,"data":1612,"type":218},{"text":1613},"Aaasaasa AI Client is a local-first desktop AI workspace built with Nuxt 4, Electron and TypeScript. Its AI Hub deliberately separates \u003Cstrong>agent\u002Fclient, provider, model, connection\u002Fruntime location, permissions and web client\u003C\u002Fstrong> instead of treating them as one configuration value.",{"id":704,"data":1615,"type":218},{"text":1616},"The implementation includes direct provider adapters, Codex agent runtime integration, local Ollama\u002FLM Studio paths, OpenAI-compatible services, centralized workspace permissions, main-process credential storage, DuckDB, Qdrant\u002Fvector support, PDF\u002Freadability extraction and authenticated MCP-based directory access.",{"id":708,"data":1618,"type":218},{"text":1619},"Two platform lessons are especially relevant. First, a local runtime is not the same as local inference: a local Codex process can still use a cloud model. Second, automatic routing does not silently fall back from local to paid cloud inference. That makes routing policy and runtime locality explicit rather than inferred from UI labels.",{"id":712,"data":1621,"type":291},{"content":1622,"stretched":43,"withHeadings":14},[1623,1626,1629,1632,1635,1638],[1624,1625],"Implemented boundary","Platform-architecture meaning",[1627,1628],"Agent vs provider vs model","Different responsibilities can evolve independently instead of being hidden behind one “AI” selector.",[1630,1631],"Permissions separate from model","Filesystem\u002Ftool authority belongs to the runtime policy, not model capability.",[1633,1634],"Main-process secrets","Credential ownership follows the privileged process boundary rather than the renderer\u002FUI.",[1636,1637],"Provider health and model discovery","Routing and availability are runtime\u002Fplatform concerns.",[1639,1640],"No silent cloud fallback","Cost, locality and data-transfer semantics remain explicit policy decisions.",{"id":734,"data":1642,"type":42},{"text":1643,"level":241},"Aaasaasa AI CMS: tenant-scoped authorization as a platform boundary",{"id":738,"data":1645,"type":218},{"text":1646},"The Aaasaasa AI CMS codebase provides a separate implementation example: tenant-scoped RBAC is represented through roles, permissions and user-role assignments bound to a tenant identifier. System permissions are grouped by capability, and role lookup and updates remain tenant-scoped.",{"id":742,"data":1648,"type":218},{"text":1649},"This is not itself proof of a complete AI platform, but it is directly relevant to one of the hardest shared-platform boundaries: a reusable service must preserve \u003Cstrong>who may do what\u003C\u002Fstrong> and \u003Cstrong>for which tenant\u003C\u002Fstrong>. Adding AI inference or retrieval on top of an application platform does not remove that requirement.",{"id":746,"data":1651,"type":218},{"text":1652},"The architectural implication is that model gateways, retrieval services and agents should consume established identity\u002Ftenant context rather than inventing a parallel AI-only authorization universe.",{"id":750,"data":1654,"type":42},{"text":1655,"level":241},"Source of Truth Research Engine: shared retrieval mechanics without shared truth",{"id":754,"data":1657,"type":218},{"text":1658},"The Source of Truth Research Engine provides a third implementation example. Different research modes share a common evidence core: Sources, Artifacts, provenance, Claims, Relations, Contradictions, a Reference Model and audit trail. The system also provides local lexical retrieval, optional semantic retrieval, extraction, snapshots and SHA-256-based provenance.",{"id":758,"data":1660,"type":218},{"text":1661},"The project explicitly treats search and semantic similarity as discovery signals rather than evidence. A result must be traced back to a concrete source and locator before it can support a claim. This is precisely the distinction an AI platform needs: \u003Cstrong>reusable retrieval machinery can be shared while evidence authority remains governed by the consuming methodology and domain.\u003C\u002Fstrong>",{"id":762,"data":1663,"type":218},{"text":1664},"The engine also demonstrates why one shared platform does not require one shared interpretation. Historical, scientific\u002Ftechnical, market-intelligence and monitoring modes can reuse core evidence infrastructure while retaining mode-specific methodology.",{"id":766,"data":1666,"type":225},{"body":1667,"title":1668,"variant":393},"Across these projects, the reusable pattern is not “one backend for everything.” It is \u003Cstrong>separation of concerns plus explicit contracts\u003C\u002Fstrong>: provider\u002Fmodel\u002Fruntime separation, tenant-aware authorization, credential boundaries, reusable data\u002Fretrieval primitives, provenance, and domain-specific authority. A future integrated platform would need stable contracts between those capabilities rather than direct coupling between codebases.","What these implementations demonstrate together",{"id":771,"data":1670,"type":42},{"text":1671,"level":242},"How current architecture guidance supports this platform scope",{"id":775,"data":1673,"type":218},{"text":1674},"ISO\u002FIEC\u002FIEEE 42010:2022 provides a general discipline for architecture descriptions across software, systems, enterprises and related entities. It does not define an AI Platform Architect, but it reinforces the need to express architectural concerns, relationships and viewpoints rather than reducing architecture to a technology list.",{"id":779,"data":1676,"type":218},{"text":1677},"NIST AI RMF 1.0 and the Generative AI Profile frame AI risk management across the lifecycle rather than only at model selection time. Governance, mapping, measurement and management are therefore compatible with a platform architecture that carries shared controls and evidence across many consuming workloads.",{"id":783,"data":1679,"type":218},{"text":1680},"Microsoft’s current AI workload guidance treats application design, data, security, operations, testing\u002Fevaluation and GenAIOps as connected architectural areas. Its current AI Gateway guidance also shows practical platform concerns such as centralized model access, project-specific token limits, quotas and multi-team containment.",{"id":787,"data":1682,"type":218},{"text":1683},"AWS’s current Generative AI Lens and multi-tenant platform scenario similarly separate foundational platform controls from consuming-application ownership. AWS explicitly notes that a central platform can enforce shared guardrails and auditability while data quality and workload-specific observability still remain responsibilities of consuming applications or data producers.",{"id":791,"data":1685,"type":218},{"text":1686},"The vendor products differ, but the cross-source pattern is stable: production AI platforms must coordinate identity, data access, models, policy, evaluation, observability, capacity, cost and lifecycle. A GPU cluster or model endpoint covers only part of that responsibility.",{"id":795,"data":1688,"type":42},{"text":1689,"level":242},"Common misconceptions",{"id":799,"data":1691,"type":291},{"content":1692,"stretched":43,"withHeadings":14},[1693,1696,1699,1702,1705,1708,1711,1714,1717],[1694,1695],"Misconception","Why it is wrong",[1697,1698],"“An AI platform is the GPU cluster.”","Compute is one substrate. A platform also needs contracts for identity, model access, data, policy, evaluation, observability and lifecycle.",[1700,1701],"“An AI gateway is just a reverse proxy.”","It may also carry model routing, token quotas, cost attribution, policy enforcement, identity and AI-specific telemetry.",[1703,1704],"“Shared means globally shared.”","A service may be physically shared while logically segmented by tenant, application, region, classification or risk level.",[1706,1707],"“One central vector database becomes the company truth.”","A vector store or retrieval service is infrastructure. Domain authority, freshness, provenance and access remain separate concerns.",[1709,1710],"“Platform evaluation replaces solution evaluation.”","General regression and telemetry cannot define whether a domain-specific answer or action is acceptable.",[1712,1713],"“Provider abstraction should hide every difference.”","Some differences are material capabilities, security semantics or failure modes and must remain visible.",[1715,1716],"“RBAC solves multi-tenancy.”","RBAC controls actions; tenant isolation controls resource boundaries. Both can be required.",[1718,1719],"“AI Platform Architect is just another name for MLOps.”","MLOps\u002FLLMOps is a major overlapping discipline, but shared application\u002Fruntime, identity, gateway, retrieval and tool boundaries can extend beyond model lifecycle operations.",{"id":830,"data":1721,"type":42},{"text":1722,"level":242},"Failure modes an AI Platform Architect should prevent",{"id":834,"data":1724,"type":291},{"content":1725,"stretched":43,"withHeadings":14},[1726,1729,1732,1735,1738,1741,1744,1747,1750,1753],[1727,1728],"Failure mode","Architectural consequence",[1730,1731],"Every team stores its own provider keys","Duplicated secret handling, inconsistent rotation and larger blast radius.",[1733,1734],"Provider abstraction hides required capabilities","Consumers cannot use features they need or silently receive behavior different from assumptions.",[1736,1737],"Shared retrieval ignores tenant\u002Fuser context","Cross-boundary data leakage can occur before the application gets a chance to filter results.",[1739,1740],"Fallback silently changes provider or locality","Cost, compliance, data location and output quality can change without the caller knowing.",[1742,1743],"Agent tools are granted by model choice","A capable model becomes over-privileged because runtime authority is not independently enforced.",[1745,1746],"All prompts\u002Fresponses are logged by default","Observability can create a new sensitive-data repository and compliance problem.",[1748,1749],"Platform owns one generic quality score","Domain failures remain hidden behind platform health metrics.",[1751,1752],"No version contract for platform capabilities","Model\u002Fprovider\u002Fruntime changes break consumers unpredictably.",[1754,1755],"Everything AI-related is centralized","The platform becomes a bottleneck and monolith instead of a reusable capability layer.",{"id":868,"data":1757,"type":42},{"text":1758,"level":242},"A practical platform-architecture decision sequence",{"id":872,"data":1760,"type":340},{"steps":1761,"title":1792,"orientation":339},[1762,1765,1768,1771,1774,1777,1780,1783,1786,1789],{"label":1763,"description":1764},"1. Identify real consumers","List solutions, teams, tenants and workloads that would consume the platform; avoid building a platform for hypothetical reuse.",{"label":1766,"description":1767},"2. Define the shared boundary","Separate cross-cutting mechanics from solution-specific domain authority, workflow and acceptance.",{"label":1769,"description":1770},"3. Define identity and isolation first","Establish users, services, applications, tenants, regions and data classifications before sharing retrieval or tool capabilities.",{"label":1772,"description":1773},"4. Define capability contracts","Specify model\u002Fprovider, retrieval, agent\u002Ftool, gateway and telemetry APIs with explicit ownership and versioning.",{"label":1775,"description":1776},"5. Decide provider and runtime strategy","Choose managed, self-hosted, local or hybrid execution and document fallback, locality and capability semantics.",{"label":1778,"description":1779},"6. Design data and retrieval boundaries","Define provenance, authorization propagation, corpus ownership, indexing and evidence responsibilities.",{"label":1781,"description":1782},"7. Add quotas, secrets and policy","Control cost, capacity, credentials, tool permissions, safety controls and blast radius.",{"label":1784,"description":1785},"8. Build evaluation and observability contracts","Provide platform metrics and tracing while leaving domain ground truth and acceptance to the solution.",{"label":1787,"description":1788},"9. Define lifecycle and operations","Version capabilities, test upgrades, document deprecation, rollback, incidents, capacity and consumer onboarding.",{"label":1790,"description":1791},"10. Validate with more than one consumer","A platform claim becomes credible when the shared capability actually serves distinct workloads without forcing them into the same domain model.","From platform need to operable shared capability",{"id":907,"data":1794,"type":42},{"text":1795,"level":242},"Edge cases and limits of the role",{"id":911,"data":1797,"type":218},{"text":1798},"A small organization with one AI application may not need a distinct AI platform or platform architect. Premature platforming can create more abstraction than value. The correct architecture may be one well-designed solution with a few reusable modules.",{"id":915,"data":1800,"type":218},{"text":1801},"An air-gapped or sovereign deployment changes the provider, update and observability model substantially. Model hosting, artifact distribution, identity integration and telemetry export may all need local equivalents.",{"id":919,"data":1803,"type":218},{"text":1804},"Highly regulated or high-consequence workloads may require stronger physical or organizational isolation instead of a logically shared platform. Reuse is never a sufficient reason to weaken a required security boundary.",{"id":923,"data":1806,"type":218},{"text":1807},"Managed cloud AI services can remove implementation burden but do not remove architectural accountability. The organization still decides identity, data access, logging, retention, quotas, model eligibility, fallback, evaluation and solution acceptance.",{"id":927,"data":1809,"type":218},{"text":1810},"The platform boundary may also differ by modality. Text inference, multimodal generation, speech, computer use and autonomous agents can have different latency, data, permission and observability requirements even when they share provider and identity infrastructure.",{"id":931,"data":1812,"type":42},{"text":1813,"level":242},"What would change this answer?",{"id":935,"data":1815,"type":218},{"text":1816},"The core definition would change if the organizational scope changes. If the architect owns one workload, the role becomes closer to AI Solution Architect. If the responsibility expands to organization-wide capability strategy, investment, standards and target-state portfolios, it moves toward Enterprise AI Architecture.",{"id":939,"data":1818,"type":218},{"text":1819},"Implementation guidance changes whenever providers, gateway products, agent protocols, regulatory obligations, model capabilities or deployment constraints change. That is why platform architecture should express stable responsibilities and contracts separately from current vendor mechanisms.",{"id":943,"data":1821,"type":42},{"text":1822,"level":242},"AI Platform Architect checklist",{"id":947,"data":1824,"type":291},{"content":1825,"stretched":43,"withHeadings":14},[1826,1829,1832,1835,1838,1841,1844,1847,1850,1853,1856,1859,1862],[1827,1828],"Question","Expected answer",[1830,1831],"Who are the actual platform consumers?","Named solutions, teams or tenant contexts with distinct but overlapping needs.",[1833,1834],"What is genuinely shared?","Explicit capability list, not a vague “AI backend.”",[1836,1837],"What must remain solution-specific?","Domain authority, business workflow, task acceptance and other workload-owned concerns.",[1839,1840],"How are models\u002Fproviders represented?","Versioned provider\u002Fmodel contracts with capabilities and explicit fallback semantics.",[1842,1843],"How is identity propagated?","User\u002Fservice\u002Fapplication\u002Ftenant context survives every privileged request path.",[1845,1846],"How is tenant isolation enforced?","Resource scoping is separate from role permission checks.",[1848,1849],"How are secrets handled?","Privileged storage, rotation, limited exposure and auditable ownership.",[1851,1852],"How does retrieval preserve authority?","Shared mechanics with authorization, provenance and domain-owned evidence rules.",[1854,1855],"How are tools and agents constrained?","Runtime permissions, bounded tool contracts, approvals, cancellation and traceability.",[1857,1858],"How are cost and capacity controlled?","Quotas, token\u002Frate controls, usage attribution and overload behavior.",[1860,1861],"How is quality measured?","Platform regression\u002Fevaluation plus solution-specific ground truth and acceptance.",[1863,1864],"How are changes rolled out?","Versioning, compatibility, migration, deprecation, rollback and incident ownership.",{"id":990,"data":1866,"type":42},{"text":1867,"level":242},"Conclusion",{"id":994,"data":1869,"type":218},{"text":1870},"An AI Platform Architect is responsible for the reusable architecture \u003Cstrong>between AI capabilities and the solutions that consume them\u003C\u002Fstrong>. The role defines how models, providers, retrieval, agents, tools, identity, tenants, secrets, evaluation, observability, quotas and runtime operations become dependable platform services rather than repeated one-off integrations.",{"id":998,"data":1872,"type":218},{"text":1873},"The difficult part is not maximizing reuse. It is choosing the correct boundary. A strong platform standardizes mechanics, policy and operations where multiple consumers genuinely benefit, while preserving solution-specific data authority, business logic, security requirements and acceptance criteria.",{"id":1002,"data":1875,"type":218},{"text":1876},"That distinction also explains the relationship with AI Solution Architecture: \u003Cstrong>the solution architect makes one AI-enabled system fit its purpose; the platform architect makes shared AI capabilities safe, reusable, operable and evolvable across many such systems.\u003C\u002Fstrong>",{"id":1006,"data":1878,"type":42},{"text":1879,"level":242},"Related canonical knowledge",{"id":1010,"data":1881,"type":218},{"text":1882},"This article sits after the canonical foundations on generative AI components, ADR versus NFR, and AI Solution Architecture. Those concepts are prerequisites because a platform exists to provide reusable system capabilities and to encode architectural decisions against explicit quality and operational requirements.",{"id":1014,"data":1884,"type":218},{"text":1885},"Retrieval-Augmented Generation is one example of a capability that may be offered through a platform, but the platform should not collapse retrieval infrastructure, domain knowledge and answer validity into one concept.",{"id":1018,"data":1887,"type":1026},{"link":1888,"meta":1889},"https:\u002F\u002Fstajic.de\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works",{"image":1890,"title":1891,"description":1892},{"url":1023},"What Is RAG? The Simplest Explanation of How It Works","Canonical introduction to retrieval-augmented generation and the boundary between model generation and external knowledge retrieval.",{"id":1028,"data":1894,"type":218},{"text":1895},"Agent protocols, tenant isolation, AI governance, model routing, Context Engineering and MLOps\u002FLLMOps are downstream or adjacent knowledge nodes. They become easier to reason about once the platform boundary is explicit.",{"id":1032,"data":1897,"type":42},{"text":1898,"level":242},"Frequently asked questions",{"id":1036,"data":1900,"type":1036},{"items":1901,"title":1920},[1902,1905,1908,1911,1914,1917],{"id":1040,"answer":1903,"question":1904},"No. The solution architect focuses on one concrete AI-enabled solution. The platform architect focuses on reusable AI capabilities, controls and operational contracts that can support multiple solutions.","Is an AI Platform Architect the same as an AI Solution Architect?",{"id":1044,"answer":1906,"question":1907},"No. A platform can use managed cloud models, self-hosted models, local inference or a hybrid strategy. The architecture must make provider, locality, identity, routing, data and operational consequences explicit.","Does an AI platform need to host its own models?",{"id":1048,"answer":1909,"question":1910},"Usually not. A gateway can be an important platform component, but a complete platform also needs contracts for identity, secrets, data\u002Fretrieval, evaluation, observability, lifecycle and operational ownership.","Is an AI gateway enough to be an AI platform?",{"id":1052,"answer":1912,"question":1913},"Retrieval mechanics can often be shared, but domain authority, authorization, freshness, evidence sufficiency and corpus ownership should remain explicit. Shared infrastructure does not imply shared truth.","Should retrieval be centralized?",{"id":1056,"answer":1915,"question":1916},"No. Platform evaluation can test shared capabilities and regressions. Each solution still needs task-specific ground truth, acceptance criteria and domain quality thresholds.","Does platform evaluation replace application evaluation?",{"id":1060,"answer":1918,"question":1919},"No. RBAC determines what an identity may do. Tenant isolation determines which tenant's resources the identity may act on. A platform often needs both.","Is multi-tenancy just RBAC?","AI Platform Architect FAQ",{"id":1065,"data":1922,"type":42},{"text":1923,"level":242},"Glossary",{"id":1069,"data":1925,"type":1069},{"title":1926,"entries":1927},"Key AI platform architecture terms",[1928,1931,1934,1937,1940,1943,1946,1949],{"term":1929,"anchor":1075,"definition":1930},"AI platform","A reusable set of AI-related technical and operational capabilities consumed by multiple applications, teams or tenant contexts.",{"term":1932,"anchor":1079,"definition":1933},"AI gateway","A gateway layer for AI endpoints that may add authentication, routing, quotas, policy, retries, cost attribution and AI-specific telemetry beyond basic proxying.",{"term":1935,"anchor":1083,"definition":1936},"Provider adapter","A component that maps a platform contract to a model provider's API, capabilities, health and failure semantics.",{"term":1938,"anchor":1087,"definition":1939},"Tenant isolation","The boundary that prevents one tenant context from accessing another tenant's resources, independent of role permissions.",{"term":1941,"anchor":1091,"definition":1942},"Capability contract","A versioned interface and behavioral agreement describing what a shared platform service provides and what the consumer must supply or own.",{"term":1944,"anchor":1095,"definition":1945},"Grounding \u002F retrieval service","Shared mechanics for finding and supplying external information to an AI workload; it does not automatically define which information is authoritative for a domain.",{"term":1947,"anchor":1099,"definition":1948},"Evaluation harness","Reusable infrastructure for running tests, datasets, model\u002Fprompt versions and metrics; domain acceptance remains solution-specific.",{"term":1950,"anchor":1103,"definition":1951},"Control plane","The configuration and governance layer that manages platform capabilities, identities, policies, quotas, versions and deployment state.",{"id":1106,"data":1953,"type":42},{"text":1954,"level":242},"Primary sources and current architecture guidance",{"id":1110,"data":1956,"type":218},{"text":1957},"The sources below support the general architecture and production-platform claims. The Aaasaasa AI Client, Aaasaasa AI CMS and Source of Truth Research Engine sections are explicitly original implementation evidence. Current-state external references were checked on 8 October 2026.",{"id":1114,"data":1959,"type":1026},{"link":1116,"meta":1960},{"image":1961,"title":1962,"description":1963},{"url":1023},"ISO\u002FIEC\u002FIEEE 42010:2022 — Architecture Description","Current published international standard for architecture-description concepts and relationships.",{"id":1122,"data":1965,"type":1026},{"link":1124,"meta":1966},{"image":1967,"title":1968,"description":1969},{"url":1023},"NIST AI Risk Management Framework","NIST's AI RMF resources and current status; AI RMF 1.0 is under revision as of October 2026.",{"id":1130,"data":1971,"type":1026},{"link":1132,"meta":1972},{"image":1973,"title":1974,"description":1975},{"url":1023},"NIST AI 600-1 — Generative AI Profile","Generative AI profile for applying AI risk-management considerations across the AI lifecycle.",{"id":1138,"data":1977,"type":1026},{"link":1140,"meta":1978},{"image":1979,"title":1980,"description":1981},{"url":1023},"Microsoft Azure Well-Architected — AI Workloads","Current architectural guidance covering AI application, data, operations, evaluation, responsible AI and lifecycle concerns.",{"id":1146,"data":1983,"type":1026},{"link":1148,"meta":1984},{"image":1985,"title":1986,"description":1987},{"url":1023},"Microsoft — Design Principles for AI Workloads","Current guidance on identity segmentation, security boundaries, telemetry, performance, data and platform trade-offs.",{"id":1154,"data":1989,"type":1026},{"link":1156,"meta":1990},{"image":1991,"title":1992,"description":1993},{"url":1023},"Microsoft Foundry — AI Gateway Architecture","Current AI Gateway guidance for shared project access, token containment, quotas and governance.",{"id":1162,"data":1995,"type":1026},{"link":1164,"meta":1996},{"image":1997,"title":1998,"description":1999},{"url":1023},"Azure Architecture Center — Access Models Through a Gateway","Architecture guidance for centralized model access, routing, throttling, failover and client\u002Fplatform responsibilities.",{"id":1170,"data":2001,"type":1026},{"link":1172,"meta":2002},{"image":2003,"title":2004,"description":2005},{"url":1023},"AWS Well-Architected — Generative AI Lens","Current production architecture guidance for generative AI workloads across security, reliability, operations, performance and cost.",{"id":1178,"data":2007,"type":1026},{"link":1180,"meta":2008},{"image":2009,"title":2010,"description":2011},{"url":1023},"AWS — Multi-tenant Generative AI Platform Scenario","Current example separating central platform controls and auditability from consuming-application data quality and workload-specific responsibilities.",{"id":1186,"data":2013,"type":1026},{"link":1188,"meta":2014},{"image":2015,"title":2016,"description":2017},{"url":1023},"AWS Well-Architected — Agentic AI Design Principles","Current guidance on bounded agent authority, traceability, versioned behavior, explicit contracts and human oversight.",{"id":1194,"data":2019,"type":1026},{"link":1196,"meta":2020},{"image":2021,"title":2022,"description":2023},{"url":1023},"AWS CloudWatch — Generative AI Observability","Current observability capabilities and production metrics for models, agents, knowledge bases, tools and cost\u002Flatency\u002Ferror analysis.","2.31.0","An AI Platform Architect designs reusable AI foundations across models, providers, retrieval, agents, identity, security, evaluation, observability and operations.",{"lang":7,"title":208,"content":210,"contentJson":2027,"excerpt":1202},{"time":212,"blocks":2028,"version":1201},[2029,2031,2033,2035,2037,2039,2041,2043,2045,2061,2063,2065,2067,2069,2078,2080,2082,2084,2086,2096,2098,2100,2102,2104,2106,2108,2110,2112,2114,2116,2118,2120,2122,2124,2126,2128,2130,2132,2134,2136,2138,2140,2142,2144,2146,2148,2150,2152,2154,2156,2158,2160,2162,2164,2166,2168,2170,2177,2179,2181,2194,2196,2214,2216,2226,2228,2230,2232,2234,2236,2238,2240,2249,2251,2253,2255,2257,2259,2261,2263,2265,2267,2269,2271,2273,2275,2277,2279,2281,2293,2295,2308,2310,2323,2325,2327,2329,2331,2333,2335,2337,2339,2341,2343,2359,2361,2363,2365,2367,2369,2371,2373,2377,2379,2381,2390,2392,2403,2405,2407,2411,2415,2419,2423,2427,2431,2435,2439,2443,2447],{"id":215,"data":2030,"type":218},{"text":217},{"id":220,"data":2032,"type":225},{"body":222,"title":223,"variant":224},{"id":227,"data":2034,"type":225},{"body":229,"title":230,"variant":231},{"id":233,"data":2036,"type":225},{"body":235,"title":236,"variant":231},{"id":238,"data":2038,"type":243},{"title":240,"maxLevel":241,"minLevel":242},{"id":245,"data":2040,"type":42},{"text":247,"level":242},{"id":249,"data":2042,"type":218},{"text":251},{"id":253,"data":2044,"type":218},{"text":255},{"id":257,"data":2046,"type":299},{"rows":2047,"title":290,"layout":291,"columns":2058},[2048,2050,2052,2054,2056],{"id":261,"label":262,"values":2049},{"platform":264,"solution":265},{"id":267,"label":268,"values":2051},{"platform":270,"solution":271},{"id":273,"label":274,"values":2053},{"platform":276,"solution":277},{"id":279,"label":280,"values":2055},{"platform":282,"solution":283},{"id":285,"label":286,"values":2057},{"platform":288,"solution":289},[2059,2060],{"id":294,"label":295},{"id":297,"label":298},{"id":301,"data":2062,"type":42},{"text":303,"level":242},{"id":305,"data":2064,"type":218},{"text":307},{"id":309,"data":2066,"type":218},{"text":311},{"id":313,"data":2068,"type":218},{"text":315},{"id":317,"data":2070,"type":340},{"steps":2071,"title":338,"orientation":339},[2072,2073,2074,2075,2076,2077],{"label":321,"description":322},{"label":324,"description":325},{"label":327,"description":328},{"label":330,"description":331},{"label":333,"description":334},{"label":336,"description":337},{"id":342,"data":2079,"type":42},{"text":344,"level":242},{"id":346,"data":2081,"type":218},{"text":348},{"id":350,"data":2083,"type":218},{"text":352},{"id":354,"data":2085,"type":42},{"text":356,"level":242},{"id":358,"data":2087,"type":291},{"content":2088,"stretched":43,"withHeadings":14},[2089,2090,2091,2092,2093,2094,2095],[362,363,364],[366,367,368],[370,371,372],[374,375,376],[378,379,380],[280,382,383],[385,386,387],{"id":389,"data":2097,"type":225},{"body":391,"title":392,"variant":393},{"id":395,"data":2099,"type":42},{"text":397,"level":242},{"id":399,"data":2101,"type":42},{"text":401,"level":241},{"id":403,"data":2103,"type":218},{"text":405},{"id":407,"data":2105,"type":218},{"text":409},{"id":411,"data":2107,"type":225},{"body":413,"title":414,"variant":415},{"id":417,"data":2109,"type":42},{"text":419,"level":241},{"id":421,"data":2111,"type":218},{"text":423},{"id":425,"data":2113,"type":218},{"text":427},{"id":429,"data":2115,"type":218},{"text":431},{"id":433,"data":2117,"type":42},{"text":435,"level":241},{"id":437,"data":2119,"type":218},{"text":439},{"id":441,"data":2121,"type":218},{"text":443},{"id":445,"data":2123,"type":218},{"text":447},{"id":449,"data":2125,"type":42},{"text":451,"level":241},{"id":453,"data":2127,"type":218},{"text":455},{"id":457,"data":2129,"type":218},{"text":459},{"id":461,"data":2131,"type":218},{"text":463},{"id":465,"data":2133,"type":42},{"text":467,"level":241},{"id":469,"data":2135,"type":218},{"text":471},{"id":473,"data":2137,"type":218},{"text":475},{"id":477,"data":2139,"type":218},{"text":479},{"id":481,"data":2141,"type":42},{"text":483,"level":241},{"id":485,"data":2143,"type":218},{"text":487},{"id":489,"data":2145,"type":218},{"text":491},{"id":493,"data":2147,"type":42},{"text":495,"level":241},{"id":497,"data":2149,"type":218},{"text":499},{"id":501,"data":2151,"type":218},{"text":503},{"id":505,"data":2153,"type":218},{"text":507},{"id":509,"data":2155,"type":42},{"text":511,"level":241},{"id":513,"data":2157,"type":218},{"text":515},{"id":517,"data":2159,"type":218},{"text":519},{"id":521,"data":2161,"type":42},{"text":523,"level":241},{"id":525,"data":2163,"type":218},{"text":527},{"id":529,"data":2165,"type":218},{"text":531},{"id":533,"data":2167,"type":42},{"text":535,"level":242},{"id":537,"data":2169,"type":225},{"body":539,"title":540,"variant":231},{"id":542,"data":2171,"type":291},{"content":2172,"stretched":43,"withHeadings":14},[2173,2174,2175,2176],[546,547,548],[550,551,552],[554,555,556],[558,559,560],{"id":562,"data":2178,"type":218},{"text":564},{"id":566,"data":2180,"type":42},{"text":568,"level":242},{"id":570,"data":2182,"type":291},{"content":2183,"stretched":43,"withHeadings":14},[2184,2185,2186,2187,2188,2189,2190,2191,2192,2193],[574,575],[577,578],[580,581],[583,584],[586,587],[589,590],[592,593],[595,596],[598,599],[601,602],{"id":604,"data":2195,"type":42},{"text":606,"level":242},{"id":608,"data":2197,"type":299},{"rows":2198,"title":647,"layout":291,"columns":2211},[2199,2201,2203,2205,2207,2209],{"id":612,"label":613,"values":2200},{"pressureA":615,"pressureB":616},{"id":618,"label":619,"values":2202},{"pressureA":621,"pressureB":622},{"id":624,"label":625,"values":2204},{"pressureA":627,"pressureB":628},{"id":630,"label":631,"values":2206},{"pressureA":633,"pressureB":634},{"id":636,"label":637,"values":2208},{"pressureA":639,"pressureB":640},{"id":642,"label":643,"values":2210},{"pressureA":645,"pressureB":646},[2212,2213],{"id":650,"label":651},{"id":653,"label":654},{"id":656,"data":2215,"type":42},{"text":658,"level":242},{"id":660,"data":2217,"type":291},{"content":2218,"stretched":43,"withHeadings":14},[2219,2220,2221,2222,2223,2224,2225],[664,665],[295,667],[298,669],[671,672],[674,675],[677,678],[680,681],{"id":683,"data":2227,"type":218},{"text":685},{"id":687,"data":2229,"type":42},{"text":689,"level":242},{"id":691,"data":2231,"type":225},{"body":693,"title":694,"variant":231},{"id":696,"data":2233,"type":42},{"text":698,"level":241},{"id":700,"data":2235,"type":218},{"text":702},{"id":704,"data":2237,"type":218},{"text":706},{"id":708,"data":2239,"type":218},{"text":710},{"id":712,"data":2241,"type":291},{"content":2242,"stretched":43,"withHeadings":14},[2243,2244,2245,2246,2247,2248],[716,717],[719,720],[722,723],[725,726],[728,729],[731,732],{"id":734,"data":2250,"type":42},{"text":736,"level":241},{"id":738,"data":2252,"type":218},{"text":740},{"id":742,"data":2254,"type":218},{"text":744},{"id":746,"data":2256,"type":218},{"text":748},{"id":750,"data":2258,"type":42},{"text":752,"level":241},{"id":754,"data":2260,"type":218},{"text":756},{"id":758,"data":2262,"type":218},{"text":760},{"id":762,"data":2264,"type":218},{"text":764},{"id":766,"data":2266,"type":225},{"body":768,"title":769,"variant":393},{"id":771,"data":2268,"type":42},{"text":773,"level":242},{"id":775,"data":2270,"type":218},{"text":777},{"id":779,"data":2272,"type":218},{"text":781},{"id":783,"data":2274,"type":218},{"text":785},{"id":787,"data":2276,"type":218},{"text":789},{"id":791,"data":2278,"type":218},{"text":793},{"id":795,"data":2280,"type":42},{"text":797,"level":242},{"id":799,"data":2282,"type":291},{"content":2283,"stretched":43,"withHeadings":14},[2284,2285,2286,2287,2288,2289,2290,2291,2292],[803,804],[806,807],[809,810],[812,813],[815,816],[818,819],[821,822],[824,825],[827,828],{"id":830,"data":2294,"type":42},{"text":832,"level":242},{"id":834,"data":2296,"type":291},{"content":2297,"stretched":43,"withHeadings":14},[2298,2299,2300,2301,2302,2303,2304,2305,2306,2307],[838,839],[841,842],[844,845],[847,848],[850,851],[853,854],[856,857],[859,860],[862,863],[865,866],{"id":868,"data":2309,"type":42},{"text":870,"level":242},{"id":872,"data":2311,"type":340},{"steps":2312,"title":905,"orientation":339},[2313,2314,2315,2316,2317,2318,2319,2320,2321,2322],{"label":876,"description":877},{"label":879,"description":880},{"label":882,"description":883},{"label":885,"description":886},{"label":888,"description":889},{"label":891,"description":892},{"label":894,"description":895},{"label":897,"description":898},{"label":900,"description":901},{"label":903,"description":904},{"id":907,"data":2324,"type":42},{"text":909,"level":242},{"id":911,"data":2326,"type":218},{"text":913},{"id":915,"data":2328,"type":218},{"text":917},{"id":919,"data":2330,"type":218},{"text":921},{"id":923,"data":2332,"type":218},{"text":925},{"id":927,"data":2334,"type":218},{"text":929},{"id":931,"data":2336,"type":42},{"text":933,"level":242},{"id":935,"data":2338,"type":218},{"text":937},{"id":939,"data":2340,"type":218},{"text":941},{"id":943,"data":2342,"type":42},{"text":945,"level":242},{"id":947,"data":2344,"type":291},{"content":2345,"stretched":43,"withHeadings":14},[2346,2347,2348,2349,2350,2351,2352,2353,2354,2355,2356,2357,2358],[951,952],[954,955],[957,958],[960,961],[963,964],[966,967],[969,970],[972,973],[975,976],[978,979],[981,982],[984,985],[987,988],{"id":990,"data":2360,"type":42},{"text":992,"level":242},{"id":994,"data":2362,"type":218},{"text":996},{"id":998,"data":2364,"type":218},{"text":1000},{"id":1002,"data":2366,"type":218},{"text":1004},{"id":1006,"data":2368,"type":42},{"text":1008,"level":242},{"id":1010,"data":2370,"type":218},{"text":1012},{"id":1014,"data":2372,"type":218},{"text":1016},{"id":1018,"data":2374,"type":1026},{"link":1020,"meta":2375},{"image":2376,"title":1024,"description":1025},{"url":1023},{"id":1028,"data":2378,"type":218},{"text":1030},{"id":1032,"data":2380,"type":42},{"text":1034,"level":242},{"id":1036,"data":2382,"type":1036},{"items":2383,"title":1063},[2384,2385,2386,2387,2388,2389],{"id":1040,"answer":1041,"question":1042},{"id":1044,"answer":1045,"question":1046},{"id":1048,"answer":1049,"question":1050},{"id":1052,"answer":1053,"question":1054},{"id":1056,"answer":1057,"question":1058},{"id":1060,"answer":1061,"question":1062},{"id":1065,"data":2391,"type":42},{"text":1067,"level":242},{"id":1069,"data":2393,"type":1069},{"title":1071,"entries":2394},[2395,2396,2397,2398,2399,2400,2401,2402],{"term":1074,"anchor":1075,"definition":1076},{"term":1078,"anchor":1079,"definition":1080},{"term":1082,"anchor":1083,"definition":1084},{"term":1086,"anchor":1087,"definition":1088},{"term":1090,"anchor":1091,"definition":1092},{"term":1094,"anchor":1095,"definition":1096},{"term":1098,"anchor":1099,"definition":1100},{"term":1102,"anchor":1103,"definition":1104},{"id":1106,"data":2404,"type":42},{"text":1108,"level":242},{"id":1110,"data":2406,"type":218},{"text":1112},{"id":1114,"data":2408,"type":1026},{"link":1116,"meta":2409},{"image":2410,"title":1119,"description":1120},{"url":1023},{"id":1122,"data":2412,"type":1026},{"link":1124,"meta":2413},{"image":2414,"title":1127,"description":1128},{"url":1023},{"id":1130,"data":2416,"type":1026},{"link":1132,"meta":2417},{"image":2418,"title":1135,"description":1136},{"url":1023},{"id":1138,"data":2420,"type":1026},{"link":1140,"meta":2421},{"image":2422,"title":1143,"description":1144},{"url":1023},{"id":1146,"data":2424,"type":1026},{"link":1148,"meta":2425},{"image":2426,"title":1151,"description":1152},{"url":1023},{"id":1154,"data":2428,"type":1026},{"link":1156,"meta":2429},{"image":2430,"title":1159,"description":1160},{"url":1023},{"id":1162,"data":2432,"type":1026},{"link":1164,"meta":2433},{"image":2434,"title":1167,"description":1168},{"url":1023},{"id":1170,"data":2436,"type":1026},{"link":1172,"meta":2437},{"image":2438,"title":1175,"description":1176},{"url":1023},{"id":1178,"data":2440,"type":1026},{"link":1180,"meta":2441},{"image":2442,"title":1183,"description":1184},{"url":1023},{"id":1186,"data":2444,"type":1026},{"link":1188,"meta":2445},{"image":2446,"title":1191,"description":1192},{"url":1023},{"id":1194,"data":2448,"type":1026},{"link":1196,"meta":2449},{"image":2450,"title":1199,"description":1200},{"url":1023},"Post erfolgreich abgerufen",{"items":2453,"source":2538,"manualIds":2539,"manualMatchedIds":2540},[2454,2461,2468,2475,2482,2489,2496,2503,2510,2517,2524,2531],{"id":2455,"slug":2456,"title":2457,"excerpt":2458,"featuredImage":2459,"publishedAt":2460},"479","where-does-an-llm-get-its-data-rag-data-sources-in-python","LLM从哪里获取数据？Python中的RAG数据源","LLM 并不会神奇地知道你的文件、数据库或 API。这个 RAG 系列的实用续篇用简单的 Python 展示了外部数据如何变成可检索的证据：从文本文件和 SQL 到全文搜索、嵌入、上下文组装以及最终的 LLM 调用。","\u002Fuploads\u002F2026\u002F09\u002Fwhere-does-an-llm-get-its-data-rag-data-sources-in-python-1790517200521-nfsi5i.webp","2026-09-27T05:51:00.000Z",{"id":2462,"slug":2463,"title":2464,"excerpt":2465,"featuredImage":2466,"publishedAt":2467},"494","air-gapped-ai-how-ai-systems-work-without-internet-or-cloud-access","气隙AI：AI系统如何在没有互联网或云访问的情况下工作","气隙AI在隔离的安全域内运行模型、RAG和AI应用，无需互联网或云依赖。了解模型、数据、更新和工具如何离线运行。","\u002Fuploads\u002F2026\u002F10\u002Fair-gapped-ai-how-ai-systems-work-without-internet-or-cloud-access-1791487983978-e6xqf0.webp","2026-10-08T11:32:00.000Z",{"id":2469,"slug":2470,"title":2471,"excerpt":2472,"featuredImage":2473,"publishedAt":2474},"478","what-is-rag-the-simplest-explanation-of-how-it-works","什么是RAG？对其工作原理的最简单解释","RAG听起来很复杂，但想法很简单：在AI回答之前，它先从知识源查找有用的信息，并将该信息提供给语言模型。本指南使用一个简单的思维模型来解释RAG、LLM、状态、记忆和工具。","\u002Fuploads\u002F2026\u002F09\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works-1790377492124-khjagt.webp","2026-09-25T19:03:00.000Z",{"id":2476,"slug":2477,"title":2478,"excerpt":2479,"featuredImage":2480,"publishedAt":2481},"481","generative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing","生成式人工智能解析：模型、检索、工具与应用并非同一回事","生成式AI不仅仅是一个模型。了解模型、检索、工具、上下文、运行时和应用程序如何在生产AI系统中协同工作。","\u002Fuploads\u002F2026\u002F10\u002Fgenerative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing-1791475411822-pp0dvz.webp","2026-10-08T12:00:00.000Z",{"id":2483,"slug":2484,"title":2485,"excerpt":2486,"featuredImage":2487,"publishedAt":2488},"381","enterprise-grade-multi-tenant-architecture-for-an-international-platform","企业级多租户架构，适用于国际平台","Loving Rocks 是一款企业级婚礼平台，采用真正的多租户架构设计，实现租户间数据库隔离，并内置国际化支持，以确保全球可扩展性、安全性及长期运营稳定性。","\u002Fuploads\u002F2026\u002F01\u002Fenterprise-grade-multi-tenant-architecture-for-an-international-platform-1769789121298-b6v7ak.webp","2026-01-30T12:04:00.000Z",{"id":2490,"slug":2491,"title":2492,"excerpt":2493,"featuredImage":2494,"publishedAt":2495},"492","mcp-explained-what-it-connects-what-it-does-not-do-and-where-it-fits","MCP 解析：它连接什么、不做什么以及它适用于何处","模型上下文协议通过标准的客户端-服务器边界，将AI应用程序连接到外部工具、资源和提示。了解MCP能做什么、不能做什么，以及它在智能体架构中的定位。","\u002Fuploads\u002F2026\u002F10\u002Fmcp-explained-what-it-connects-what-it-does-not-do-and-where-it-fits-1791486640275-7ub1cq.webp","2026-10-08T15:09:00.000Z",{"id":2497,"slug":2498,"title":2499,"excerpt":2500,"featuredImage":2501,"publishedAt":2502},"471","how-to-know-whether-an-ai-agent-actually-used-the-right-evidence","如何判断一个AI智能体是否真正使用了正确的证据","AI代理可以引用来源，却仍然使用错误的证据。本文介绍一种实用方法，用于核查主张支持、来源权威性、适用性、出处，以及证据是否实际影响了答案。","\u002Fuploads\u002F2026\u002F09\u002Fhow-to-know-whether-an-ai-agent-actually-used-the-right-evidence-1790351317188-o5z9ve.webp","2026-09-25T11:47:00.000Z",{"id":2504,"slug":2505,"title":2506,"excerpt":2507,"featuredImage":2508,"publishedAt":2509},"480","when-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger","人工智能何时应停止信任自身知识？——检索触发机制","AI 模型并非每个问题都需要检索。重要的问题在于知道何时其内部知识已不再足够。检索触发器是一个实用的决策边界，它决定 AI 系统何时应停止仅依赖模型知识，并在回答前获取外部证据。","\u002Fuploads\u002F2026\u002F09\u002Fwhen-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger-1790574991244-f4rpyg.webp","2026-09-28T01:49:00.000Z",{"id":2511,"slug":2512,"title":2513,"excerpt":2514,"featuredImage":2515,"publishedAt":2516},"466","the-gpu-is-not-the-product-future-proof-private-ai-architecture","GPU 不是产品：面向未来的私有 AI 架构","私有 AI 基础设施不应围绕单一 GPU 或单一模型来设计。更具韧性的做法是将快速推理 GPU、内存充裕的 AI 系统、物理 AI 节点以及可选的前沿云模型，统一置于一个具备能力感知的路由层之后。","\u002Fuploads\u002F2026\u002F09\u002Fthe-gpu-is-not-the-product-future-proof-private-ai-architecture-1790140878812-8hsl39.webp","2026-09-23T01:19:00.000Z",{"id":2518,"slug":2519,"title":2520,"excerpt":2521,"featuredImage":2522,"publishedAt":2523},"483","what-is-an-ai-solution-architect-system-boundaries-responsibilities-and-trade-offs","什么是AI解决方案架构师？系统边界、职责与权衡","AI解决方案架构师将业务需求转化为生产就绪的AI系统，涵盖数据、模型、工具、安全、运行时、评估和运维。","\u002Fuploads\u002F2026\u002F10\u002Fwhat-is-an-ai-solution-architect-system-boundaries-responsibilities-and-trade-offs-1791476643267-1st5xz.webp","2026-10-08T12:23:00.000Z",{"id":2525,"slug":2526,"title":2527,"excerpt":2528,"featuredImage":2529,"publishedAt":2530},"495","sovereign-ai-control-of-models-data-infrastructure-and-dependencies","主权人工智能：模型、数据、基础设施与依赖关系的控制","主权人工智能关乎对模型、数据、基础设施、软件、运营和战略依赖的有效控制——而不仅仅是人工智能模型托管在哪里。","\u002Fuploads\u002F2026\u002F10\u002Fsovereign-ai-control-of-models-data-infrastructure-and-dependencies-1791488833132-niy85x.webp","2026-10-08T15:45:00.000Z",{"id":2532,"slug":2533,"title":2534,"excerpt":2535,"featuredImage":2536,"publishedAt":2537},"485","enterprise-ai-architecture-what-changes-when-ai-enters-a-company","企业AI架构：当AI进入公司时会发生什么变化","企业AI架构阐释了AI如何在数据权限、身份、许可、提供商、风险、治理、评估、合规和运营方面改变公司系统。","\u002Fuploads\u002F2026\u002F10\u002Fenterprise-ai-architecture-what-changes-when-ai-enters-a-company-1791478161363-czrwaq.webp","2026-10-08T10:48:00.000Z","fallback",[],[]]