[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"portal-settings:stajic:de":3,"public-menus:all":37,"post:what-is-an-ai-platform-architect-models-data-runtime-security-and-operations:de":204,"related:post:what-is-an-ai-platform-architect-models-data-runtime-security-and-operations:de:1":2434},{"statusCode":4,"data":5,"message":36},200,{"tenantId":6,"lang":7,"defaultLang":7,"siteUrl":8,"contactEmail":9,"brandName":10,"logoUrl":11,"siteName":10,"siteDescription":12,"ogImage":9,"robotsIndex":13,"socialLinks":9,"reservedSlugs":9,"seoPolicy":14},"stajic","de","https:\u002F\u002Fstajic.de",null,"Stajic Platform","\u002FLogo_Planet.svg","Stajic Portal",true,{"branding":15,"relatedContent":16,"crossDomainLinks":17},{"logoUrl":11},{"enabled":13},[18,21,24,27,30,33],{"url":19,"label":20,"isActive":13,"showInFooter":13,"includeInSameAs":13},"https:\u002F\u002Ffigure.rocks","figure.rocks",{"url":22,"label":23,"isActive":13,"showInFooter":13,"includeInSameAs":13},"https:\u002F\u002Floving.rocks","loving.rocks",{"url":25,"label":26,"isActive":13,"showInFooter":13,"includeInSameAs":13},"https:\u002F\u002Fbazify.com","bazify.com",{"url":28,"label":29,"isActive":13,"showInFooter":13,"includeInSameAs":13},"https:\u002F\u002Fbazify.de","bazify.de",{"url":31,"label":32,"isActive":13,"showInFooter":13,"includeInSameAs":13},"https:\u002F\u002Fbazify.at","bazify.at",{"url":34,"label":35,"isActive":13,"showInFooter":13,"includeInSameAs":13},"https:\u002F\u002Fbazify.ba","bazify.ba","Portal settings resolved",[38,44],{"id":39,"name":40,"location":41,"isActive":13,"isDefault":42,"items":43},1,"main-navigation","header",false,[],{"id":45,"name":46,"location":47,"isActive":13,"isDefault":13,"items":48},4,"main-menu","sidebar",[49,65,78,92,102,117,132],{"id":50,"title":51,"url":59,"target":60,"icon":61,"isActive":13,"type":62,"productId":9,"categoryId":9,"shopCategoryId":9,"articleId":9,"pageId":63,"portfolioId":9,"children":64},"item-18",{"de":52,"en":53,"es":54,"fr":55,"it":53,"ru":56,"sr":57,"zh":58},"Startseite","Home","Inicio","Accueil","Главная","Почетна","首页","\u002Ffull-stack-web-developer-munich-performance-seo-and-maintainable-builds","_self","i-lucide-home","page",111,[],{"id":66,"title":67,"url":74,"target":60,"icon":75,"isActive":13,"type":62,"productId":9,"categoryId":9,"shopCategoryId":9,"articleId":9,"pageId":76,"portfolioId":9,"children":77},"item-22",{"de":68,"en":68,"es":69,"fr":68,"it":70,"ru":71,"sr":72,"zh":73},"Vision","Visión","Visione","Видение","Визија","想象","\u002Fueber-uns-webdesign-muenchen-webaplikation","i-lucide-eye",113,[],{"id":79,"title":80,"url":88,"target":60,"icon":89,"isActive":13,"type":62,"productId":9,"categoryId":9,"shopCategoryId":9,"articleId":9,"pageId":90,"portfolioId":9,"children":91},"item-19",{"de":81,"en":82,"es":83,"fr":82,"it":84,"ru":85,"sr":86,"zh":87},"Leistungen","Services","Servicios","Servizi","Услуги","Услуге","服务","\u002Fservices-dienstleistungen-muenchen","i-lucide-wrench",116,[],{"id":93,"title":94,"url":98,"target":60,"icon":99,"isActive":13,"type":62,"productId":9,"categoryId":9,"shopCategoryId":9,"articleId":9,"pageId":100,"portfolioId":9,"children":101},"item-23",{"de":95,"en":95,"es":95,"fr":95,"it":95,"ru":96,"sr":96,"zh":97},"Blog","Блог","博客","\u002Fblog","i-lucide-book-open",112,[],{"id":103,"title":104,"url":113,"target":60,"icon":114,"isActive":13,"type":62,"productId":9,"categoryId":9,"shopCategoryId":9,"articleId":9,"pageId":115,"portfolioId":9,"children":116},"item-32",{"de":105,"en":106,"es":107,"fr":108,"it":109,"ru":110,"sr":111,"zh":112},"Neue Technologien","New Technologies","Nuevas tecnologías","Nouvelles technologies","Nuove tecnologie","Новые технологии","Нове технологије","新技术！","\u002Fneue-webtechnologien","i-lucide-sparkles",122,[],{"id":118,"title":119,"url":128,"target":60,"icon":129,"isActive":13,"type":62,"productId":9,"categoryId":9,"shopCategoryId":9,"articleId":9,"pageId":130,"portfolioId":9,"children":131},"item-20",{"de":120,"en":121,"es":122,"fr":123,"it":124,"ru":125,"sr":126,"zh":127},"Kontakt","Contact us!","Contacto","Contact","Contatto","Контакт","Контактирајте нас","联系我们！","\u002Fcontact","i-lucide-mail",115,[],{"id":133,"title":134,"url":143,"target":60,"icon":144,"isActive":13,"type":62,"productId":9,"categoryId":9,"shopCategoryId":9,"articleId":9,"pageId":145,"portfolioId":9,"children":146},"item-21",{"de":135,"en":136,"es":137,"fr":138,"it":139,"ru":140,"sr":141,"zh":142},"Unsere Arbeit","Our Work","Nuestro trabajo","Nos réalisations","I nostri lavori","Наши работы","Наши радови","文件夹","\u002Fportfolio","i-lucide-briefcase",114,[147,160,174,180,192],{"id":148,"title":149,"url":143,"target":60,"icon":158,"isActive":13,"type":62,"productId":9,"categoryId":9,"shopCategoryId":9,"articleId":9,"pageId":145,"portfolioId":9,"children":159},"item-24",{"de":150,"en":151,"es":152,"fr":153,"it":154,"ru":155,"sr":156,"zh":157},"Alle Projekte","All Projects","Todos los proyectos","Tous les projets","Tutti i progetti","Все проекты","Сви пројекти","所有项目","i-lucide-grid-3x3",[],{"id":161,"title":162,"url":170,"target":60,"icon":171,"isActive":13,"type":172,"productId":9,"categoryId":9,"shopCategoryId":9,"articleId":9,"pageId":9,"portfolioId":9,"children":173},"item-29",{"de":163,"en":164,"es":165,"fr":166,"it":167,"ru":168,"sr":169,"zh":142},"Local Roots, Global Reach","Local Roots - Global Reach","Empresa local ","Entreprise locale","Azienda locale","Местная компания","Локално предузеће глобално тржиште","\u002Fportfolio\u002Flocal-roots-global-reach-communication-media-systems-for-modern-business","i-lucide-folder","custom",[],{"id":175,"title":176,"url":178,"target":60,"icon":171,"isActive":13,"type":172,"productId":9,"categoryId":9,"shopCategoryId":9,"articleId":9,"pageId":9,"portfolioId":9,"children":179},"item-28",{"de":177,"en":177,"es":177,"fr":177,"it":177,"ru":177,"sr":177,"zh":177},"Solr Suggester","\u002Fportfolio\u002Fsolr-fuzzy-suggester-und-solr-infix-suggester-abfrage-ueber-ajax-und-filterung",[],{"id":181,"title":182,"url":190,"target":60,"icon":171,"isActive":13,"type":172,"productId":9,"categoryId":9,"shopCategoryId":9,"articleId":9,"pageId":9,"portfolioId":9,"children":191},"item-27",{"de":183,"en":184,"es":185,"fr":186,"it":187,"ru":188,"sr":189,"zh":184},"Firmenwebseite SEO","Company Website SEO","Sitio web corporativo SEO","Site web d’entreprise SEO","Sito web aziendale SEO","Корпоративный сайт SEO","Пословна веб-страница SEO","\u002Fportfolio\u002Fseo-sem-branding-mobile-webseite-muenchen",[],{"id":193,"title":194,"url":202,"target":60,"icon":171,"isActive":13,"type":172,"productId":9,"categoryId":9,"shopCategoryId":9,"articleId":9,"pageId":9,"portfolioId":9,"children":203},"item-31",{"de":195,"en":196,"es":197,"fr":198,"it":199,"ru":200,"sr":201,"zh":196},"Digitalisierungsportal","Digitalization Portal","Portal de digitalización","Portail de numérisation","Portale di digitalizzazione","Портал цифровизации","Портал за дигитализацију","\u002Fportfolio\u002Fdigitalisierungsportal-archiv-museum-bibliothek-ead-lido-mets-mods",[],{"statusCode":4,"data":205,"message":2433},{"id":206,"title":207,"slug":208,"content":209,"contentJson":210,"excerpt":1201,"featuredImage":1202,"featuredImageAlt":1203,"featuredImageCaption":9,"featuredImageTitle":9,"featuredImageCopyright":9,"featuredImageAuthor":9,"featuredImageSourceUrl":9,"featuredImageLicense":9,"featuredImageIsAiGenerated":42,"status":1204,"publishedAt":1205,"createdAt":1206,"updatedAt":1207,"seoLocalePaths":1208,"categories":1217,"author":1230,"translations":1235},"484","Was ist ein KI-Plattform-Architekt? Modelle, Daten, Laufzeitumgebung, Sicherheit und Betrieb","what-is-an-ai-platform-architect-models-data-runtime-security-and-operations","\u003Cp>Ein \u003Cstrong>AI Platform Architect\u003C\u002Fstrong> entwirft das wiederverwendbare KI-Fundament, über das mehrere Anwendungen, Teams oder Mandantenkontexte auf Modelle, Daten und Retrieval, Agenten- und Tool-Runtimes, Identität und Berechtigungen, Evaluierung, Observability, Quotas, Secrets und Deployment-Fähigkeiten zugreifen. Die Rolle ist breiter als Infrastruktur, aber enger als das Eigentum an jedem KI-fähigen Produkt: Ihre zentrale Verantwortung besteht darin, zu entscheiden, \u003Cstrong>was geteilt werden sollte, wie geteilte Fähigkeiten gesteuert und isoliert werden und was lösungsspezifisch bleiben muss\u003C\u002Fstrong>.\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--info my-6 rounded-xl border p-5 border-blue-300 bg-blue-50 dark:border-blue-900 dark:bg-blue-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">Direkte Antwort\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">\u003Cstrong>Ein AI Platform Architect entwirft das gemeinsame technische und operative Substrat für KI-Systeme.\u003C\u002Fstrong> Statt einen einzelnen Assistenten oder einen einzelnen Workflow zu architektieren, definiert die Rolle wiederverwendbare Verträge und Grenzen für Modell-\u002FAnbieterzugriff, Gateways und Routing, Retrieval-Dienste, Agenten-Runtimes, Tool-Zugriff, Identität und Mandantenisolierung, Secrets, Evaluierung, Telemetrie, Deployment und Lifecycle-Management.\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Caside class=\"editorjs-callout editorjs-callout--note my-6 rounded-xl border p-5 border-gray-300 bg-gray-50 dark:border-gray-700 dark:bg-gray-900\u002F40\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">Terminologiehinweis\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">\u003Cstrong>AI Platform Architect ist eine praktische Rollenbezeichnung, kein universell standardisierter Jobtitel.\u003C\u002Fstrong> ISO\u002FIEC\u002FIEEE 42010:2022 definiert Konzepte für Architekturbeschreibungen, nicht diese Rolle. Verschiedene Organisationen können diese Verantwortlichkeiten auf Plattformarchitekten, Lösungsarchitekten, Unternehmensarchitekten, Sicherheitsarchitekten, MLOps\u002FLLMOps-Spezialisten und Plattform-Engineering-Teams aufteilen. Dieser Artikel verwendet den Begriff für die Architekturverantwortung über eine wiederverwendbare KI-Plattformschicht.\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Caside class=\"editorjs-callout editorjs-callout--note my-6 rounded-xl border p-5 border-gray-300 bg-gray-50 dark:border-gray-700 dark:bg-gray-900\u002F40\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">Hinweis zu aktuellen Quellen — 8. Oktober 2026\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">Die hier dargestellten stabilen Architekturprinzipien sind anbieterneutral. Aktuelle Microsoft-, AWS- und NIST-Leitlinien werden als externe Implementierungs- und Governance-Belege verwendet. NIST gibt an, dass AI RMF 1.0 überarbeitet wird; Anbieterplattformfunktionen, Gateway-Produkte, Agenten-Runtimes und Modellfähigkeiten entwickeln sich schneller als die Architekturprinzipien, sodass versionsabhängige Implementierungsentscheidungen vor dem Deployment erneut überprüft werden müssen.\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Cnav class=\"editorjs-toc\" data-editorjs-toc=\"true\" aria-label=\"Inhalt\">\u003Cstrong class=\"editorjs-toc__title\">Inhalt\u003C\u002Fstrong>\u003Col class=\"editorjs-toc__list editorjs-toc__list--depth-0\">\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-6\" class=\"editorjs-toc__link\">Was architektiert ein AI Platform Architect tatsächlich?\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-10\" class=\"editorjs-toc__link\">Das einfachste Beispiel\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-15\" class=\"editorjs-toc__link\">Wo das einfache Beispiel endet\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-18\" class=\"editorjs-toc__link\">Die wichtigste Plattformentscheidung: gemeinsam versus lösungsspezifisch\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-21\" class=\"editorjs-toc__link\">Architektur-Verantwortlichkeitskarte\u003C\u002Fa>\u003Col class=\"editorjs-toc__list editorjs-toc__list--depth-1\">\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-22\" class=\"editorjs-toc__link\">1. Modell- und Anbieterzugriff\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-26\" class=\"editorjs-toc__link\">2. Gateway, Routing, Quoten und Kostenkontrollen\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-30\" class=\"editorjs-toc__link\">3. Gemeinsame Daten-, Retrieval- und Grounding-Dienste\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-34\" class=\"editorjs-toc__link\">4. Agenten- und Tool-Runtime\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-38\" class=\"editorjs-toc__link\">5. Identität, Mandantentrennung und Autorisierung\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-42\" class=\"editorjs-toc__link\">6. Geheimnisse, Anmeldeinformationen und Vertrauensgrenzen\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-45\" class=\"editorjs-toc__link\">7. Evaluierung, Beobachtbarkeit und Auditierbarkeit\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-49\" class=\"editorjs-toc__link\">8. Laufzeit, Bereitstellung und Lokalität\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-52\" class=\"editorjs-toc__link\">9. Plattform-Lebenszyklus, Kompatibilität und Onboarding\u003C\u002Fa>\u003C\u002Fli>\u003C\u002Fol>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-55\" class=\"editorjs-toc__link\">Ein praktisches Modell für Control-Plane \u002F Execution-Plane \u002F Solution-Plane\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-59\" class=\"editorjs-toc__link\">Was sollte ein AI Platform Architect produzieren?\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-61\" class=\"editorjs-toc__link\">Die Arbeit besteht hauptsächlich aus Trade-offs, nicht aus maximaler Zentralisierung\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-63\" class=\"editorjs-toc__link\">Wie unterscheidet sich das von angrenzenden Rollen?\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-66\" class=\"editorjs-toc__link\">Implementierungsnachweise: Wie diese Plattformgrenzen in meiner eigenen Arbeit erscheinen\u003C\u002Fa>\u003Col class=\"editorjs-toc__list editorjs-toc__list--depth-1\">\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-68\" class=\"editorjs-toc__link\">Aaasaasa AI Client: Trennung von Provider, Laufzeit und Berechtigungen\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-73\" class=\"editorjs-toc__link\">Aaasaasa AI CMS: mandantenbezogene Autorisierung als Plattformgrenze\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-77\" class=\"editorjs-toc__link\">Source of Truth Research Engine: gemeinsame Retrieval-Mechanik ohne gemeinsame Wahrheit\u003C\u002Fa>\u003C\u002Fli>\u003C\u002Fol>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-82\" class=\"editorjs-toc__link\">Wie aktuelle Architekturleitlinien diesen Plattformumfang stützen\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-88\" class=\"editorjs-toc__link\">Häufige Missverständnisse\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-90\" class=\"editorjs-toc__link\">Fehlermodi, die ein AI Platform Architect verhindern sollte\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-92\" class=\"editorjs-toc__link\">Eine praktische Entscheidungssequenz für die Plattformarchitektur\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-94\" class=\"editorjs-toc__link\">Randfälle und Grenzen der Rolle\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-100\" class=\"editorjs-toc__link\">Was würde diese Antwort ändern?\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-103\" class=\"editorjs-toc__link\">Checkliste für KI-Plattformarchitekten\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-105\" class=\"editorjs-toc__link\">Fazit\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-109\" class=\"editorjs-toc__link\">Verwandtes kanonisches Wissen\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-114\" class=\"editorjs-toc__link\">Häufig gestellte Fragen\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-116\" class=\"editorjs-toc__link\">Glossar\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-118\" class=\"editorjs-toc__link\">Primärquellen und aktuelle Architekturempfehlungen\u003C\u002Fa>\u003C\u002Fli>\u003C\u002Fol>\u003C\u002Fnav>\n\u003Ch2 id=\"section-6\">Was architektiert ein AI Platform Architect tatsächlich?\u003C\u002Fh2>\n\u003Cp>Der Gegenstand der Arbeit ist die \u003Cstrong>Plattform\u003C\u002Fstrong>: eine Reihe gemeinsamer Fähigkeiten, die wiederholte Integrationsarbeit reduziert und dabei explizite Sicherheits-, Daten- und Betriebsgrenzen wahrt. Eine Plattform kann Modellzugriff, Anbieteradapter, Retrieval-Primitive, Agentenausführung, Tool-Broker, Richtliniendurchsetzung, Evaluierung, Telemetrie und Deployment-Dienste für viele konsumierende Lösungen bereitstellen.\u003C\u002Fp>\n\u003Cp>Die Plattform ist nicht allein deshalb wertvoll, weil Komponenten zentralisiert sind. Sie ist wertvoll, wenn Konsumenten stabile Fähigkeiten mit klaren Verträgen, Eigentum, Isolierung, Observability und Lifecycle-Regeln erhalten. Die zentrale Architekturfrage ist daher nicht „Welches Modell sollte jeder verwenden?“, sondern \u003Cstrong>„Welche Verantwortlichkeiten können sicher standardisiert und wiederverwendet werden, ohne die Anforderungen jeder Lösung zu verwischen?“\u003C\u002Fstrong>.\u003C\u002Fp>\n\u003Csection class=\"editorjs-comparison my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">Lösungsarchitektur und Plattformarchitektur lösen unterschiedliche Umfangsprobleme\u003C\u002Fh3>\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left dark:border-gray-700 dark:bg-gray-900\">\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">AI Solution Architect\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">AI Platform Architect\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Primärer Umfang\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">One concrete AI-enabled product, workflow or application.\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">Reusable AI capabilities consumed by multiple solutions, teams or tenant contexts.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Hauptfrage\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">How should this solution meet its business, data, security, quality and operational requirements?\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">Which shared capabilities and controls should solutions consume, and where must solution-specific ownership remain?\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Datenhoheit\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">Defines which domain data is authoritative and how the solution may use it.\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">Provides storage, retrieval, provenance or access primitives without automatically becoming the authority for every domain.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Evaluierung\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">Defines task-specific quality and acceptance criteria.\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">Provides reusable evaluation, telemetry and release mechanisms; it cannot define every domain&#39;s success threshold.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Lifecycle\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">Owns the lifecycle of the specific workload.\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">Owns shared capability versions, compatibility, onboarding, quotas, policy and operational contracts.\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-10\">Das einfachste Beispiel\u003C\u002Fh2>\n\u003Cp>Stellen Sie sich vor, ein Unternehmen hat fünf KI-fähige Produkte: einen internen Dokumentenassistenten, einen Copiloten für den Kundensupport, einen Software-Engineering-Agenten, einen Vertragsprüfungs-Workflow und einen Produktsuchassistenten. Jedes Produkt könnte unabhängig Modell-APIs integrieren, Anmeldedaten speichern, Wiederholungsversuche implementieren, Token-Metriken sammeln, Retrieval-Code erstellen und eigene Tool-Berechtigungen aufbauen.\u003C\u002Fp>\n\u003Cp>Diese Duplizierung ist teuer und gefährlich, wenn jedes Team ein anderes Sicherheits- und Betriebsmodell erfindet. Eine gemeinsame Plattform kann stattdessen genehmigte Anbieterverbindungen, Modellermittlung, Quotas, Anmeldedaten, mandantenfähigen Zugriff, gemeinsame Telemetrie, wiederverwendbare Retrieval-Dienste und einen Agenten-\u002FTool-Runtime-Vertrag anbieten.\u003C\u002Fp>\n\u003Cp>Aber die Plattform muss an der richtigen Grenze haltmachen. Die Vertragsprüfungslösung kann rechtliche Dokumentenhoheit und Zitierregeln erfordern, die der Software-Agent nicht benötigt. Der Produktsuchassistent kann handelsspezifische Aktualitäts- und Autorisierungsregeln benötigen. \u003Cstrong>Wiederverwendbare Infrastruktur macht nicht alle Domänenwahrheit wiederverwendbar.\u003C\u002Fstrong>\u003C\u002Fp>\n\u003Csection class=\"editorjs-process my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">Ein gemeinsamer KI-Anfragepfad\u003C\u002Fh3>\u003Cdiv class=\"grid grid-cols-1 md:grid-cols-2 xl:grid-cols-3 gap-4\">\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">1\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">1. Konsument identifiziert sich\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Die aufrufende Anwendung, der Benutzer, der Dienst, das Team oder der Mandant tritt über eine authentifizierte Identität und einen expliziten Geltungsbereich ein.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">2\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">2. Plattformrichtlinie wird angewendet\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Gateway- und Richtlinienschichten bestimmen erlaubte Anbieter, Modelle, Quotas, Datenpfade, Tools und Ausführungsmodi.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">3\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">3. Gemeinsame Fähigkeit wird ausgeführt\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Die Anfrage kann Inferenz, Retrieval, Agenten-Runtime, Tool-Zugriff oder einen anderen wiederverwendbaren Plattformdienst nutzen.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">4\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">4. Lösungsspezifischer Kontext bleibt maßgeblich\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Die konsumierende Lösung liefert Domänenregeln, Benutzerabsicht, Datenhoheit, aufgabenspezifische Einschränkungen und Akzeptanzlogik.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">5\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">5. Telemetrie und Belege werden erfasst\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Die Plattform zeichnet Identität, Route, Modell\u002FAnbieter, Latenz, Kosten, Fehler, Tool-Aktivität und andere zulässige Observability-Signale auf.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">6\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">6. Ergebnis kehrt unter dem Lösungsvertrag zurück\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Die Lösung bleibt dafür verantwortlich, ob die Ausgabe für ihren Benutzer und ihre Domäne akzeptabel ist.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-15\">Wo das einfache Beispiel endet\u003C\u002Fh2>\n\u003Cp>Zentralisierung ist nicht automatisch Architektur. Ein einzelner Endpunkt vor mehreren Modell-APIs ist nützlich, aber er schafft für sich genommen noch keine KI-Plattform. Eine Produktionsplattform benötigt außerdem Identitätsgrenzen, Fähigkeitsverträge, Anbieterzustand und Lifecycle-Handhabung, Quotas, Secret-Eigentum, Observability, Kompatibilitätsregeln, Sicherheitskontrollen, Release-Disziplin und klare operative Verantwortung.\u003C\u002Fp>\n\u003Cp>Das gegenteilige Versagen ist ebenfalls häufig: jeden Prompt, jeden Vektorindex, jede Geschäftsregel, jeden Agenten und jeden Anwendungs-Workflow in ein einziges „KI-Backend“ zu packen. Das erzeugt einen Monolithen, dessen gemeinsamer Status zufällig statt architektonisch ist. \u003Cstrong>Eine Plattform sollte querschnittliche Fähigkeiten standardisieren, nicht Domäneneigentum absorbieren, nur weil KI beteiligt ist.\u003C\u002Fstrong>\u003C\u002Fp>\n\u003Ch2 id=\"section-18\">Die wichtigste Plattformentscheidung: gemeinsam versus lösungsspezifisch\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Fähigkeitsbereich\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Guter Kandidat für gemeinsame Plattformverantwortung\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Bleibt üblicherweise lösungsspezifisch\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Modellzugriff\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Genehmigte Anbieterverbindungen, Adapter, Anmeldedaten, Health, Routing-Primitive, Quoten\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Aufgabenspezifische Modellakzeptanz, Prompt-Verhalten, Qualitätsschwelle\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Retrieval\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Ingestion-Primitive, Extraktion, Indexierung, Such-APIs, Provenienz-Verträge, Autorisierungs-Hooks\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Autoritativer Korpus, Aktualitätsregeln, Domänen-Metadaten, Evidenz-Suffizienz\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Agenten und Tools\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Runtime-Lebenszyklus, Tool-Registry\u002FBroker, Berechtigungsdurchsetzung, Tracing, Abbruch\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Geschäftsworkflow, erlaubte Aktionssemantik, Eskalationsrichtlinie, Aufgabenerfolg\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Sicherheit\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Identitätsintegration, Secret-Speicherung, Richtliniendurchsetzung, Audit-Verträge, Mandantenisolationsmechanismen\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Datenklassifizierung, geschäftliche Autorisierungsregeln, domänenspezifische Risikoakzeptanz\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Evaluierung\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Harness, Dataset-\u002FVersionsmechanik, Telemetrie, Experiment-\u002FRelease-Workflow\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Ground Truth, Domänen-Testset, Akzeptanzschwelle, Nutzerergebnis\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Betrieb\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Deployment-Muster, Health, Metriken, Incident-Integration, Kapazitätssteuerung\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Lösungs-SLOs, wo sie abweichen, Auswirkungen auf die Geschäftskontinuität, workloadspezifische Runbooks\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Caside class=\"editorjs-callout editorjs-callout--success my-6 rounded-xl border p-5 border-emerald-300 bg-emerald-50 dark:border-emerald-900 dark:bg-emerald-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">Plattformprinzip\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">\u003Cstrong>Teile Mechaniken und Kontrollen, wo Wiederverwendung real ist; behalte Autorität und Akzeptanz dort, wo die Domäne sie besitzt.\u003C\u002Fstrong> Dies verhindert zwei gegensätzliche Fehler: überall duplizierte Infrastruktur und eine zentrale Plattform, die fälschlicherweise zum Eigentümer der Daten, Richtlinien und Qualität jeder Anwendung wird.\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-21\">Architektur-Verantwortlichkeitskarte\u003C\u002Fh2>\n\u003Ch3 id=\"section-22\">1. Modell- und Anbieterzugriff\u003C\u002Fh3>\n\u003Cp>Ein Plattformarchitekt definiert, wie Konsumenten Modelle entdecken und aufrufen, ohne jede Anwendung zu zwingen, einen Anbieter fest zu codieren. Dies umfasst Anbieteradapter, Modellkennungen, Fähigkeitsmetadaten, Authentifizierung, Health-Checks, Endpunktkonfiguration, Request-Normalisierung und Kompatibilitätsverhalten.\u003C\u002Fp>\n\u003Cp>Anbieterabstraktion muss ehrlich bleiben. Verschiedene Anbieter bieten unterschiedliche Kontextgrenzen, Tool-Semantik, strukturiertes Ausgabeverhalten, multimodale Fähigkeiten, Sicherheitskontrollen, Caching, Preisgestaltung und Fehlermodi. Eine gute Abstraktion schafft einen stabilen Plattformvertrag und bewahrt gleichzeitig den Zugriff auf Fähigkeiten, die nicht sinnvoll vereinheitlicht werden können.\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--warning my-6 rounded-xl border p-5 border-amber-300 bg-amber-50 dark:border-amber-900 dark:bg-amber-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">Verwechsle Abstraktion nicht damit, Anbieter als identisch darzustellen\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">Eine API auf dem kleinsten gemeinsamen Nenner kann die Migration erleichtern, aber auch Fähigkeiten auslöschen, die wichtig sind. Die Architektur sollte definieren, welche Funktionen portabel sind, welche anbieterspezifisch sind und wie Konsumenten diesen Unterschied erkennen.\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch3 id=\"section-26\">2. Gateway, Routing, Quoten und Kostenkontrollen\u003C\u002Fh3>\n\u003Cp>Ein gemeinsames KI-Gateway kann Authentifizierung, Routing, Drosselung, Wiederholungen, Token-Limits, Nutzungszuordnung und Richtliniendurchsetzung zentralisieren. Microsofts aktuelle AI-Gateway-Richtlinien behandeln Token-pro-Minute-Limits, Quoten und Multi-Projekt-Eingrenzung ausdrücklich als Plattformbelange; AWS stellt ebenfalls Konto- und Modellquoten sowie zentrale Kontrollen bereit.\u003C\u002Fp>\n\u003Cp>Das Gateway ist daher mehr als ein Reverse Proxy, wenn es KI-spezifische Richtlinien- und Betriebssemantik trägt. Es sollte jedoch nicht stillschweigend Geschäftsentscheidungen treffen. Eine Routing-Richtlinie kann ein gesundes lokales Modell, einen kostengünstigeren Anbieter oder einen regional konformen Endpunkt bevorzugen; ob diese Route für eine bestimmte Aufgabe akzeptabel ist, bleibt dennoch ein Vertrag zwischen Plattform und Lösung.\u003C\u002Fp>\n\u003Cp>Routing benötigt auch Fehlersemantik. Wenn das bevorzugte Modell nicht verfügbar ist, muss die Plattform wissen, ob ein Fallback erlaubt ist, ob eine Cloud-Route eine ausdrückliche Zustimmung erfordert, ob ein Modell mit geringerer Fähigkeit gültig ist und wie die Entscheidung für die Observability sichtbar gemacht wird.\u003C\u002Fp>\n\u003Ch3 id=\"section-30\">3. Gemeinsame Daten-, Retrieval- und Grounding-Dienste\u003C\u002Fh3>\n\u003Cp>Retrieval-Dienste sind starke Plattformkandidaten, weil Parsing, Chunking, Indexierung, lexikalische Suche, semantische Suche, Metadatenfilterung, Provenienz und Zitationsmechanik wiederverwendbar sind. Die Plattform darf jedoch eine gemeinsame Retrieval-Engine nicht mit einer gemeinsamen Quelle der Wahrheit verwechseln.\u003C\u002Fp>\n\u003Cp>Eine Lösung besitzt weiterhin Fragen wie: Welcher Korpus ist autoritativ? Welche Version ist gültig? Darf dieser Nutzer dieses Dokument sehen? Wie aktuell müssen die Daten sein? Was zählt als ausreichende Evidenz? Kann eine Antwort generiert werden, wenn das Retrieval fehlschlägt? Das sind Domänen- und Lösungsanforderungen, selbst wenn die Plattform die Retrieval-Maschinerie bereitstellt.\u003C\u002Fp>\n\u003Cp>Diese Grenze ist besonders wichtig in Multi-Tenant-Systemen. Ein technisch gemeinsamer Index oder Vektordienst rechtfertigt keine mandantenübergreifende Sichtbarkeit. Der Autorisierungskontext muss durch das Retrieval hindurch erhalten bleiben und darf nicht erst hinzugefügt werden, nachdem Suchergebnisse die Grenze bereits überschritten haben.\u003C\u002Fp>\n\u003Ch3 id=\"section-34\">4. Agenten- und Tool-Runtime\u003C\u002Fh3>\n\u003Cp>Agentische Systeme fügen wiederverwendbare Runtime-Belange hinzu: Thread-\u002FSession-Lebenszyklus, Planungsschleifen, Tool-Registrierung, Tool-Aufruf, Abbruch, Timeouts, menschliche Genehmigungen, Memory-\u002FState-Schnittstellen, Remote-Agent-Protokolle und Trace-Korrelation. Eine Plattform kann diese Mechaniken bereitstellen, damit jedes Produkt sie nicht neu aufbauen muss.\u003C\u002Fp>\n\u003Cp>Die Plattform muss auch die Tool-Berechtigung von der Modellfähigkeit getrennt halten. Dass ein Modell einen Shell-Befehl generieren kann, bedeutet nicht, dass die Runtime die Shell-Ausführung erlauben sollte. Die Berechtigungsgrenze gehört zur Anwendungs-\u002FRuntime-Architektur und muss unabhängig vom Modell durchsetzbar sein.\u003C\u002Fp>\n\u003Cp>Die aktuelle AWS-Agentic-AI-Richtlinie betont begrenzte Agenten, explizite Befugnisse, End-to-End-Tracing, versionierte Verhaltensartefakte und menschliche Aufsicht im Verhältnis zu den Konsequenzen. Das sind plattformermöglichende Belange, aber die konsumierende Lösung definiert weiterhin, welche Aktionen für ihre Domäne legitim sind.\u003C\u002Fp>\n\u003Ch3 id=\"section-38\">5. Identität, Mandantentrennung und Autorisierung\u003C\u002Fh3>\n\u003Cp>KI-Plattformen stehen oft vor hochwertigen Modellen, proprietären Daten und aktionsfähigen Tools. Authentifizierung ist daher nur der Anfang. Die Architektur muss Benutzer-, Dienst-, Anwendungs- und Mandantenkontext durch jeden privilegierten Vorgang tragen, der ihn benötigt.\u003C\u002Fp>\n\u003Cp>\u003Cstrong>RBAC und Mandantentrennung lösen unterschiedliche Probleme.\u003C\u002Fstrong> RBAC beantwortet, was eine Identität tun darf; Mandantentrennung beantwortet, auf welche Ressourcen welches Mandanten diese Identität einwirken darf. Eine Plattform, die Rollen prüft, aber den Mandantenkontext verliert, kann dennoch die falschen Daten offenlegen.\u003C\u002Fp>\n\u003Cp>Die aktuelle KI-Workload-Richtlinie von Microsoft empfiehlt ausdrücklich Identitätssegmentierung und autorisierungsbewussten Zugriff auf Inhalte. Die AWS-Richtlinie für mandantenfähige generative KI-Plattformen behandelt logische Isolierung, zentralisierte Kontrollen und Auditierbarkeit ebenfalls als Plattformbelange.\u003C\u002Fp>\n\u003Ch3 id=\"section-42\">6. Geheimnisse, Anmeldeinformationen und Vertrauensgrenzen\u003C\u002Fh3>\n\u003Cp>Eine Plattform sollte definieren, wem Anbieterschlüssel, entfernte Bearer-Tokens, Signiermaterial und Tool-Anmeldeinformationen gehören, wo sie gespeichert sind, welcher Prozess auf sie zugreifen kann, wie sie rotiert werden und ob sie jemals einen Browser oder einen nicht vertrauenswürdigen Renderer erreichen können.\u003C\u002Fp>\n\u003Cp>Dies ist eine architektonische Grenze, kein Implementierungsdetail. Wenn jede konsumierende Anwendung Anbieter-Anmeldeinformationen in ihre eigene Konfiguration kopiert, hat die Organisation sowohl den operativen Aufwand als auch den Wirkungsradius dupliziert. Zentralisierung kann dieses Risiko nur reduzieren, wenn die Plattform selbst engere, auditierbare Zugriffspfade hat.\u003C\u002Fp>\n\u003Ch3 id=\"section-45\">7. Evaluierung, Beobachtbarkeit und Auditierbarkeit\u003C\u002Fh3>\n\u003Cp>Eine wiederverwendbare Plattform kann Evaluierungs-Harnesses, Trace-IDs, Modell-\u002FAnbieter-Metadaten, Token- und Kostenmetriken, Latenz, Fehlerraten, Verknüpfung von Prompt-\u002FModellversionen, Agenten-\u002FTool-Traces und kontrolliertes Logging bereitstellen. AWS und Microsoft behandeln sowohl Beobachtbarkeit als auch Evaluierung als zentrale Produktionsbelange für KI-Workloads.\u003C\u002Fp>\n\u003Cp>Plattform-Evaluierung und Lösungs-Evaluierung müssen getrennt bleiben. Eine Plattform kann verifizieren, dass ein Endpunkt gesund ist, eine Modellversion eine allgemeine Regressionssuite besteht und Traces vollständig sind. Sie kann nicht entscheiden, dass eine rechtliche Antwort, ein medizinischer Workflow oder eine Produktempfehlung akzeptabel ist, ohne domänenspezifische Ground Truth und Akzeptanzkriterien.\u003C\u002Fp>\n\u003Cp>Logging schafft auch eine Datenschutzgrenze. Prompt- und Antwortprotokolle können sensible oder proprietäre Daten enthalten. Der Plattformarchitekt muss daher entscheiden, was protokolliert, redigiert, gesampelt, aufbewahrt und zugänglich gemacht wird, anstatt anzunehmen, dass mehr Telemetrie immer sicherer ist.\u003C\u002Fp>\n\u003Ch3 id=\"section-49\">8. Laufzeit, Bereitstellung und Lokalität\u003C\u002Fh3>\n\u003Cp>Ein Plattformarchitekt entscheidet, wie gemeinsame KI-Fähigkeiten bereitgestellt und erreicht werden: verwaltete Cloud-Dienste, selbst gehostete Endpunkte, lokale Inferenz, hybrides Routing, containerisierte Dienste, Desktop-Laufzeiten, private Netzwerke oder air-gapped Umgebungen. Die wichtige Unterscheidung ist zwischen \u003Cstrong>wo der Steuerungs-\u002FLaufzeitprozess läuft\u003C\u002Fstrong> und \u003Cstrong>wo Inferenz und Datenverarbeitung tatsächlich stattfinden\u003C\u002Fstrong>.\u003C\u002Fp>\n\u003Cp>Ein lokaler Client kann dennoch ein Cloud-Modell aufrufen. Eine Cloud-Steuerungsebene kann zu einem On-Premises-Modell routen. Ein Remote-Agent kann Tools innerhalb eines Kundennetzwerks ausführen. Architekturdiagramme müssen daher Vertrauens- und Datenflussgrenzen zeigen, anstatt „lokal“ und „Cloud“ als vage Bezeichnungen zu verwenden.\u003C\u002Fp>\n\u003Ch3 id=\"section-52\">9. Plattform-Lebenszyklus, Kompatibilität und Onboarding\u003C\u002Fh3>\n\u003Cp>Wiederverwendbare Fähigkeit wird erst dann zu einer Plattform, wenn Konsumenten sich im Laufe der Zeit darauf verlassen können. Das erfordert versionierte Verträge, Migrationsregeln, Kompatibilitätsrichtlinien, Deprecation, Release-Tests, Rollback, Incident-Verantwortung, Kapazitätsplanung, Dokumentation und einen Weg zum Onboarding neuer Teams oder Anwendungen.\u003C\u002Fp>\n\u003Cp>Sich schnell entwickelnde KI-Ökosysteme machen dies besonders wichtig. Modellnamen, SDKs, Protokollversionen, Anbieter-APIs und Sicherheitsfähigkeiten ändern sich unabhängig voneinander. Eine Plattform muss einen Teil dieser Volatilität absorbieren, ohne Änderungen zu verbergen, die das Verhalten einer Lösung wesentlich beeinflussen.\u003C\u002Fp>\n\u003Ch2 id=\"section-55\">Ein praktisches Modell für Control-Plane \u002F Execution-Plane \u002F Solution-Plane\u003C\u002Fh2>\n\u003Caside class=\"editorjs-callout editorjs-callout--note my-6 rounded-xl border p-5 border-gray-300 bg-gray-50 dark:border-gray-700 dark:bg-gray-900\u002F40\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">Vorgeschlagenes Architekturmodell\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">Das untenstehende Drei-Plane-Modell ist eine praktische Methode, um über Verantwortlichkeiten nachzudenken; es ist kein ISO-, NIST-, Microsoft- oder AWS-Standard. Sein Zweck ist es, Eigentumsgrenzen explizit zu machen.\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Plane\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Typische Verantwortlichkeiten\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Sollte nicht stillschweigend besitzen\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Plattform-Control-Plane\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Provider-Registry, Modellrichtlinie, Quoten, Mandantenkonfiguration, Identitäten, Secrets, Routing-Regeln, Fähigkeitsversionen, Bereitstellungskonfiguration\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Anwendungsgeschäftslogik oder Domänenwahrheit\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Plattform-Execution-\u002FData-Plane\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Inferenzanfragen, Retrieval-Operationen, Agent-\u002FTool-Ausführung, Extraktion, Indexierung, Telemetrie-Emission, Richtliniendurchsetzung\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Mandantenübergreifender Zugriff allein aufgrund gemeinsam genutzter Infrastruktur\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Solution-Plane\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Benutzer-Workflow, Prompts\u002FAnweisungen, autoritative Korpusauswahl, Domänenautorisierung, Geschäftsregeln, Aufgabenbewertung und -akzeptanz\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Low-Level-Provider-Integration, die die Plattform explizit besitzt\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Cp>Diese Trennung hilft, Plattform-Drift zu diagnostizieren. Wenn eine Anwendung jeden providerspezifischen Credential und Endpunkt kennen muss, ist der Plattformvertrag zu dünn. Wenn die Plattform entscheidet, welcher Kundendatensatz rechtlich autoritativ ist oder ob eine Domänenantwort akzeptabel ist, hat die Plattform die Grenze zur Solution-Ownership überschritten.\u003C\u002Fp>\n\u003Ch2 id=\"section-59\">Was sollte ein AI Platform Architect produzieren?\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Architekturartefakt\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Zweck\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Plattform-Fähigkeitskarte\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Definiert, was die Plattform bereitstellt, wer sie nutzt und welche Fähigkeiten außerhalb des Geltungsbereichs bleiben.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Provider-\u002FModellvertrag\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Definiert Provider, Modelle, Fähigkeiten, Abstraktionsgrenzen, Routenmetadaten und Fallback-Semantik.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Identitäts- und Mandantenmodell\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Definiert Benutzer-\u002FDienst-\u002FAnwendungsidentität, Mandantenkontext, RBAC\u002FABAC-Hooks und Ressourcenisolation.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Gateway- und Quotenrichtlinie\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Definiert Ratenlimits, Token-\u002FKostenbudgets, Routing-Steuerung, Wiederholungen und Kapazitätsverhalten.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Retrieval-\u002FDatenvertrag\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Definiert Ingestion, Provenienz, Suche, Metadaten, Autorisierungsweitergabe und wo Domänenautorität bleibt.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Agent-\u002FTool-Vertrag\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Definiert Laufzeit-Lebenszyklus, Tool-Registrierung, Berechtigungen, Genehmigungen, Abbruch und Trace-Verhalten.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Secret- und Trust-Boundary-Modell\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Definiert Credential-Eigentum, Speicherung, Prozessgrenzen, Rotation und Pfade sensibler Daten.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Evaluierungs- und Telemetrievertrag\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Definiert gemeinsame Metriken, Traces, Datensatz-\u002FVersionslinks, Logging-Richtlinie und Solution-Erweiterungspunkte.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Lebenszyklus- und Kompatibilitätsrichtlinie\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Definiert Versionen, Migrationen, Deprecation, Releases, Rollback, Incident-Ownership und Onboarding.\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-61\">Die Arbeit besteht hauptsächlich aus Trade-offs, nicht aus maximaler Zentralisierung\u003C\u002Fh2>\n\u003Csection class=\"editorjs-comparison my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">Häufige Plattform-Trade-offs\u003C\u002Fh3>\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left dark:border-gray-700 dark:bg-gray-900\">\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">Druck A\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">Druck B\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Provider-Abstraktion\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">Stable portable platform API\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">Access to provider-specific capabilities and fast innovation\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Wiederverwendung\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">Shared services reduce duplication\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">Isolation and domain autonomy prevent unsafe coupling\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Governance\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">Central policy and auditability\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">Team speed and local experimentation\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Observability\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">Rich traces for debugging and evaluation\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">Privacy, data minimization and logging cost\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Verfügbarkeit\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">Fallback and multi-provider resilience\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">Predictable quality, compliance and data-location guarantees\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Plattformumfang\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">More reusable capabilities\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">Smaller blast radius and less platform lock-in\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-63\">Wie unterscheidet sich das von angrenzenden Rollen?\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Rolle\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Primärer Architekturumfang\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">AI Solution Architect\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Eine konkrete KI-fähige Lösung und ihre End-to-End-Anforderungen, Grenzen, Trade-offs und Produktionsakzeptanz.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">AI Platform Architect\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Wiederverwendbare KI-Fähigkeiten und betriebliche\u002Fsicherheitstechnische Verträge, die von mehreren Lösungen oder Teams genutzt werden.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Enterprise Architect\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Organisationsweites Geschäfts-\u002FTechnologieportfolio, Fähigkeits- und Governance-Ausrichtung auf breiterer Ebene.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">MLOps \u002F LLMOps Architect oder Spezialist\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Modell- und KI-Lebenszyklus, Bereitstellung, Experimente, Observability, Release- und Betriebspraktiken; kann stark überlappen, besitzt aber nicht automatisch die gesamte gemeinsame Anwendungsplattform.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Platform Engineer \u002F SRE\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Implementiert und betreibt Plattforminfrastruktur, Zuverlässigkeit, Automatisierung und Entwicklererfahrung; Architekturverantwortung kann mit dem Plattformarchitekten geteilt werden.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">AI \u002F Software Engineer\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Implementiert Modelle, Integrationen, Dienste, Agenten, Retrieval und Produktfunktionalität innerhalb der vereinbarten Architektur.\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Cp>Diese Grenzen sind organisatorisch, nicht universell. In einem kleinen Team kann eine Person mehrere Verantwortlichkeiten tragen. In einem regulierten Unternehmen können sie auf Architektur-, Sicherheits-, Plattform-, Daten- und Betriebsgruppen aufgeteilt sein. Die nützliche Unterscheidung ist der \u003Cstrong>Umfang der Architekturverantwortung\u003C\u002Fstrong>, nicht der auf einem Organigramm gedruckte Jobtitel.\u003C\u002Fp>\n\u003Ch2 id=\"section-66\">Implementierungsnachweise: Wie diese Plattformgrenzen in meiner eigenen Arbeit erscheinen\u003C\u002Fh2>\n\u003Caside class=\"editorjs-callout editorjs-callout--note my-6 rounded-xl border p-5 border-gray-300 bg-gray-50 dark:border-gray-700 dark:bg-gray-900\u002F40\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">Originale Implementierungsnachweise\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">Die folgenden Abschnitte beschreiben konkrete Muster aus meinen eigenen Projekten. Sie sind Nachweise, dass diese Architekturgrenzen in echtem Code und Projektsystemen implementiert oder explizit entworfen wurden. Sie sind \u003Cstrong>keine\u003C\u002Fstrong> Behauptungen, dass die Projekte zusammen bereits eine kommerziell eingesetzte Unternehmens-KI-Plattform darstellen.\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch3 id=\"section-68\">Aaasaasa AI Client: Trennung von Provider, Laufzeit und Berechtigungen\u003C\u002Fh3>\n\u003Cp>Aaasaasa AI Client ist ein lokal-first Desktop-KI-Arbeitsbereich, der mit Nuxt 4, Electron und TypeScript erstellt wurde. Sein AI Hub trennt bewusst \u003Cstrong>Agent\u002FClient, Provider, Modell, Verbindungs-\u002FLaufzeitort, Berechtigungen und Web-Client\u003C\u002Fstrong>, anstatt sie als einen Konfigurationswert zu behandeln.\u003C\u002Fp>\n\u003Cp>Die Implementierung umfasst direkte Provider-Adapter, Codex-Agent-Laufzeitintegration, lokale Ollama\u002FLM Studio-Pfade, OpenAI-kompatible Dienste, zentralisierte Arbeitsbereichsberechtigungen, Credential-Speicherung im Hauptprozess, DuckDB, Qdrant\u002FVektor-Unterstützung, PDF-\u002FReadability-Extraktion und authentifizierten MCP-basierten Verzeichniszugriff.\u003C\u002Fp>\n\u003Cp>Zwei Plattformlektionen sind besonders relevant. Erstens ist eine lokale Laufzeit nicht dasselbe wie lokale Inferenz: Ein lokaler Codex-Prozess kann immer noch ein Cloud-Modell verwenden. Zweitens fällt automatisches Routing nicht stillschweigend von lokaler auf kostenpflichtige Cloud-Inferenz zurück. Das macht Routing-Richtlinie und Laufzeitlokalität explizit statt aus UI-Labels abgeleitet.\u003C\u002Fp>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Implementierte Grenze\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Plattformarchitektonische Bedeutung\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Agent vs. Provider vs. Modell\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Unterschiedliche Verantwortlichkeiten können sich unabhängig entwickeln, anstatt hinter einem einzigen „KI“-Selektor verborgen zu werden.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Berechtigungen getrennt vom Modell\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Dateisystem-\u002FTool-Autorität gehört zur Laufzeitrichtlinie, nicht zur Modellfähigkeit.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Secrets im Hauptprozess\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Credential-Eigentum folgt der privilegierten Prozessgrenze statt dem Renderer\u002FUI.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Provider-Zustand und Modell-Erkennung\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Routing und Verfügbarkeit sind Laufzeit-\u002FPlattformbelange.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Kein stiller Cloud-Fallback\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Kosten-, Lokalitäts- und Datenübertragungssemantik bleiben explizite Richtlinienentscheidungen.\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch3 id=\"section-73\">Aaasaasa AI CMS: mandantenbezogene Autorisierung als Plattformgrenze\u003C\u002Fh3>\n\u003Cp>Die Codebasis des Aaasaasa AI CMS bietet ein separates Implementierungsbeispiel: mandantenbezogenes RBAC wird durch Rollen, Berechtigungen und Benutzer-Rollen-Zuweisungen dargestellt, die an eine Mandantenkennung gebunden sind. Systemberechtigungen sind nach Fähigkeiten gruppiert, und Rollensuche und -aktualisierungen bleiben mandantenbezogen.\u003C\u002Fp>\n\u003Cp>Dies ist für sich genommen kein Beweis für eine vollständige KI-Plattform, aber es ist direkt relevant für eine der schwierigsten Grenzen gemeinsamer Plattformen: Ein wiederverwendbarer Dienst muss bewahren, \u003Cstrong>wer was tun darf\u003C\u002Fstrong> und \u003Cstrong>für welchen Mandanten\u003C\u002Fstrong>. Das Hinzufügen von KI-Inferenz oder Retrieval auf einer Anwendungsplattform beseitigt diese Anforderung nicht.\u003C\u002Fp>\n\u003Cp>Die architektonische Implikation ist, dass Modell-Gateways, Retrieval-Dienste und Agenten etablierten Identitäts-\u002FMandantenkontext nutzen sollten, anstatt ein paralleles, nur auf KI ausgerichtetes Autorisierungsuniversum zu erfinden.\u003C\u002Fp>\n\u003Ch3 id=\"section-77\">Source of Truth Research Engine: gemeinsame Retrieval-Mechanik ohne gemeinsame Wahrheit\u003C\u002Fh3>\n\u003Cp>Die Source of Truth Research Engine bietet ein drittes Implementierungsbeispiel. Verschiedene Recherchemodelle teilen einen gemeinsamen Evidenzkern: Quellen, Artefakte, Provenienz, Claims, Relationen, Widersprüche, ein Referenzmodell und Audit-Trail. Das System bietet außerdem lokales lexikalisches Retrieval, optionales semantisches Retrieval, Extraktion, Snapshots und SHA-256-basierte Provenienz.\u003C\u002Fp>\n\u003Cp>Das Projekt behandelt Suche und semantische Ähnlichkeit ausdrücklich als Entdeckungssignale und nicht als Evidenz. Ein Ergebnis muss auf eine konkrete Quelle und einen Locator zurückgeführt werden können, bevor es einen Claim stützen kann. Genau das ist die Unterscheidung, die eine KI-Plattform braucht: \u003Cstrong>wiederverwendbare Retrieval-Maschinerie kann geteilt werden, während die Evidenzautorität weiterhin durch die konsumierende Methodik und Domäne geregelt wird.\u003C\u002Fstrong>\u003C\u002Fp>\n\u003Cp>Die Engine zeigt auch, warum eine gemeinsame Plattform keine gemeinsame Interpretation erfordert. Historische, wissenschaftlich-technische, Marktintelligenz- und Monitoring-Modi können die Kern-Evidenzinfrastruktur wiederverwenden und dennoch modusspezifische Methodik beibehalten.\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--success my-6 rounded-xl border p-5 border-emerald-300 bg-emerald-50 dark:border-emerald-900 dark:bg-emerald-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">Was diese Implementierungen zusammen zeigen\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">Über diese Projekte hinweg ist das wiederverwendbare Muster nicht „ein Backend für alles“. Es ist \u003Cstrong>Trennung von Belangen plus explizite Verträge\u003C\u002Fstrong>: Trennung von Provider\u002FModell\u002FLaufzeit, mandantenbewusste Autorisierung, Credential-Grenzen, wiederverwendbare Daten-\u002FRetrieval-Primitive, Provenienz und domänenspezifische Autorität. Eine zukünftige integrierte Plattform bräuchte stabile Verträge zwischen diesen Fähigkeiten statt direkter Kopplung zwischen Codebasen.\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-82\">Wie aktuelle Architekturleitlinien diesen Plattformumfang stützen\u003C\u002Fh2>\n\u003Cp>ISO\u002FIEC\u002FIEEE 42010:2022 bietet eine allgemeine Disziplin für Architekturbeschreibungen über Software, Systeme, Unternehmen und verwandte Entitäten hinweg. Es definiert keinen AI Platform Architect, aber es stärkt die Notwendigkeit, architektonische Belange, Beziehungen und Sichtweisen auszudrücken, anstatt Architektur auf eine Technologieliste zu reduzieren.\u003C\u002Fp>\n\u003Cp>NIST AI RMF 1.0 und das Generative AI Profile rahmen KI-Risikomanagement über den Lebenszyklus ein und nicht nur zum Zeitpunkt der Modellauswahl. Governance, Mapping, Messung und Management sind daher mit einer Plattformarchitektur vereinbar, die gemeinsame Kontrollen und Evidenz über viele konsumierende Workloads hinweg trägt.\u003C\u002Fp>\n\u003Cp>Die aktuelle AI-Workload-Leitlinie von Microsoft behandelt Anwendungsdesign, Daten, Sicherheit, Betrieb, Test\u002FEvaluierung und GenAIOps als verbundene Architekturbereiche. Die aktuelle AI-Gateway-Leitlinie zeigt zudem praktische Plattformbelange wie zentralisierten Modellzugriff, projektspezifische Token-Limits, Quoten und Multi-Team-Eingrenzung.\u003C\u002Fp>\n\u003Cp>Der aktuelle Generative AI Lens und das Multi-Tenant-Plattformszenario von AWS trennen ebenfalls grundlegende Plattformkontrollen von der Verantwortung konsumierender Anwendungen. AWS weist ausdrücklich darauf hin, dass eine zentrale Plattform gemeinsame Guardrails und Auditierbarkeit durchsetzen kann, während Datenqualität und workloadspezifische Observability weiterhin Verantwortlichkeiten konsumierender Anwendungen oder Datenproduzenten bleiben.\u003C\u002Fp>\n\u003Cp>Die Anbieterprodukte unterscheiden sich, aber das quellenübergreifende Muster ist stabil: Produktions-KI-Plattformen müssen Identität, Datenzugriff, Modelle, Richtlinien, Evaluierung, Observability, Kapazität, Kosten und Lebenszyklus koordinieren. Ein GPU-Cluster oder Modell-Endpunkt deckt nur einen Teil dieser Verantwortung ab.\u003C\u002Fp>\n\u003Ch2 id=\"section-88\">Häufige Missverständnisse\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Missverständnis\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Warum es falsch ist\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">„Eine KI-Plattform ist der GPU-Cluster.“\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Compute ist ein Substrat. Eine Plattform braucht außerdem Verträge für Identität, Modellzugriff, Daten, Richtlinien, Evaluierung, Observability und Lebenszyklus.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">„Ein KI-Gateway ist nur ein Reverse Proxy.“\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Es kann auch Modell-Routing, Token-Quoten, Kostenattribution, Richtliniendurchsetzung, Identität und KI-spezifische Telemetrie tragen.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">„Geteilt bedeutet global geteilt.“\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Ein Dienst kann physisch geteilt sein und logisch nach Mandant, Anwendung, Region, Klassifizierung oder Risikostufe segmentiert werden.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">„Eine zentrale Vektordatenbank wird die Unternehmenswahrheit.“\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Ein Vektorspeicher oder Retrieval-Dienst ist Infrastruktur. Domänenautorität, Aktualität, Provenienz und Zugriff bleiben separate Belange.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">„Plattform-Evaluierung ersetzt Lösungsevaluierung.“\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Allgemeine Regression und Telemetrie können nicht definieren, ob eine domänenspezifische Antwort oder Aktion akzeptabel ist.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">„Provider-Abstraktion sollte jeden Unterschied verbergen.“\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Einige Unterschiede sind wesentliche Fähigkeiten, Sicherheitssemantiken oder Fehlermodi und müssen sichtbar bleiben.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">„RBAC löst Multi-Tenancy.“\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">RBAC steuert Aktionen; Mandantenisolierung steuert Ressourcengrenzen. Beides kann erforderlich sein.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">„AI Platform Architect ist nur ein anderer Name für MLOps.“\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">MLOps\u002FLLMOps ist eine wichtige überlappende Disziplin, aber gemeinsame Anwendungs-\u002FLaufzeit-, Identitäts-, Gateway-, Retrieval- und Tool-Grenzen können über Modelllebenszyklus-Operationen hinausgehen.\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-90\">Fehlermodi, die ein AI Platform Architect verhindern sollte\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Fehlermodus\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Architektonische Konsequenz\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Jedes Team speichert seine eigenen Provider-Schlüssel\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Doppelte Handhabung von Geheimnissen, inkonsistente Rotation und größerer Blast-Radius.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Provider-Abstraktion verbirgt erforderliche Fähigkeiten\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Konsumenten können benötigte Funktionen nicht nutzen oder erhalten stillschweigend ein Verhalten, das von den Annahmen abweicht.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Gemeinsame Retrieval ignoriert Mandanten-\u002FBenutzerkontext\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Grenzüberschreitende Datenlecks können auftreten, bevor die Anwendung die Möglichkeit hat, Ergebnisse zu filtern.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Fallback ändert stillschweigend Provider oder Lokalität\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Kosten, Compliance, Datenstandort und Ausgabequalität können sich ändern, ohne dass der Aufrufer davon weiß.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Agent-Tools werden durch Modellwahl gewährt\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Ein leistungsfähiges Modell wird überprivilegiert, weil die Laufzeitberechtigung nicht unabhängig durchgesetzt wird.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Alle Prompts\u002FAntworten werden standardmäßig protokolliert\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Observability kann ein neues sensibles Datenrepository und Compliance-Problem schaffen.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Plattform besitzt einen generischen Qualitätswert\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Domänenfehler bleiben hinter Plattform-Gesundheitsmetriken verborgen.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Kein Versionsvertrag für Plattformfähigkeiten\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Modell-\u002FProvider-\u002FLaufzeitänderungen brechen Konsumenten unvorhersehbar.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Alles KI-bezogene ist zentralisiert\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Die Plattform wird zum Engpass und Monolithen statt zu einer wiederverwendbaren Fähigkeitsschicht.\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-92\">Eine praktische Entscheidungssequenz für die Plattformarchitektur\u003C\u002Fh2>\n\u003Csection class=\"editorjs-process my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">Vom Plattformbedarf zur betreibbaren gemeinsamen Fähigkeit\u003C\u002Fh3>\u003Cdiv class=\"grid grid-cols-1 md:grid-cols-2 xl:grid-cols-3 gap-4\">\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">1\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">1. Echte Konsumenten identifizieren\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Listen Sie Lösungen, Teams, Mandanten und Workloads auf, die die Plattform nutzen würden; vermeiden Sie den Aufbau einer Plattform für hypothetische Wiederverwendung.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">2\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">2. Die gemeinsame Grenze definieren\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Trennen Sie übergreifende Mechanismen von lösungsspezifischer Domänenautorität, Workflow und Abnahme.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">3\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">3. Zuerst Identität und Isolation definieren\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Etablieren Sie Benutzer, Dienste, Anwendungen, Mandanten, Regionen und Datenklassifizierungen, bevor Sie Retrieval- oder Tool-Fähigkeiten teilen.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">4\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">4. Fähigkeitsverträge definieren\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Spezifizieren Sie Modell-\u002FProvider-, Retrieval-, Agent-\u002FTool-, Gateway- und Telemetrie-APIs mit expliziter Eigentümerschaft und Versionierung.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">5\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">5. Provider- und Laufzeitstrategie entscheiden\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Wählen Sie verwaltete, selbst gehostete, lokale oder hybride Ausführung und dokumentieren Sie Fallback-, Lokalitäts- und Fähigkeitssemantik.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">6\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">6. Daten- und Retrieval-Grenzen entwerfen\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Definieren Sie Herkunft, Autorisierungsweitergabe, Korpus-Eigentümerschaft, Indexierung und Nachweispflichten.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">7\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">7. Quoten, Geheimnisse und Richtlinien hinzufügen\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Steuern Sie Kosten, Kapazität, Anmeldeinformationen, Tool-Berechtigungen, Sicherheitskontrollen und Blast-Radius.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">8\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">8. Evaluierungs- und Observability-Verträge aufbauen\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Bieten Sie Plattformmetriken und Tracing, während Sie Domänen-Ground-Truth und Abnahme der Lösung überlassen.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">9\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">9. Lebenszyklus und Betrieb definieren\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Versionieren Sie Fähigkeiten, testen Sie Upgrades, dokumentieren Sie Deprecation, Rollback, Vorfälle, Kapazität und Konsumenten-Onboarding.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">10\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">10. Mit mehr als einem Konsumenten validieren\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Ein Plattformanspruch wird glaubwürdig, wenn die gemeinsame Fähigkeit tatsächlich unterschiedliche Workloads bedient, ohne sie in dasselbe Domänenmodell zu zwingen.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-94\">Randfälle und Grenzen der Rolle\u003C\u002Fh2>\n\u003Cp>Eine kleine Organisation mit einer einzigen KI-Anwendung benötigt möglicherweise keine eigenständige KI-Plattform oder keinen Plattformarchitekten. Verfrühte Plattformbildung kann mehr Abstraktion als Wert schaffen. Die richtige Architektur kann eine gut entworfene Lösung mit einigen wiederverwendbaren Modulen sein.\u003C\u002Fp>\n\u003Cp>Eine luftgespaltene oder souveräne Bereitstellung ändert das Provider-, Update- und Observability-Modell erheblich. Modell-Hosting, Artefaktverteilung, Identitätsintegration und Telemetrie-Export benötigen möglicherweise alle lokale Äquivalente.\u003C\u002Fp>\n\u003Cp>Hochregulierte oder folgenschwere Workloads können eine stärkere physische oder organisatorische Isolation erfordern, anstatt einer logisch gemeinsamen Plattform. Wiederverwendung ist niemals ein ausreichender Grund, eine erforderliche Sicherheitsgrenze zu schwächen.\u003C\u002Fp>\n\u003Cp>Verwaltete Cloud-KI-Dienste können die Implementierungslast entfernen, aber nicht die architektonische Verantwortung. Die Organisation entscheidet weiterhin über Identität, Datenzugriff, Protokollierung, Aufbewahrung, Quoten, Modellberechtigung, Fallback, Evaluierung und Lösungsabnahme.\u003C\u002Fp>\n\u003Cp>Die Plattformgrenze kann sich auch je nach Modalität unterscheiden. Textinferenz, multimodale Generierung, Sprache, Computernutzung und autonome Agenten können unterschiedliche Latenz-, Daten-, Berechtigungs- und Observability-Anforderungen haben, selbst wenn sie Provider- und Identitätsinfrastruktur teilen.\u003C\u002Fp>\n\u003Ch2 id=\"section-100\">Was würde diese Antwort ändern?\u003C\u002Fh2>\n\u003Cp>Die Kerndefinition würde sich ändern, wenn sich der organisatorische Umfang ändert. Wenn der Architekt einen Workload besitzt, nähert sich die Rolle einem KI-Lösungsarchitekten. Wenn sich die Verantwortung auf organisationsweite Fähigkeitsstrategie, Investitionen, Standards und Zielzustandsportfolios ausdehnt, bewegt sie sich in Richtung Enterprise-KI-Architektur.\u003C\u002Fp>\n\u003Cp>Die Implementierungsanleitung ändert sich, wann immer sich Provider, Gateway-Produkte, Agent-Protokolle, regulatorische Verpflichtungen, Modellfähigkeiten oder Bereitstellungsbeschränkungen ändern. Deshalb sollte die Plattformarchitektur stabile Verantwortlichkeiten und Verträge getrennt von aktuellen Anbietermechanismen ausdrücken.\u003C\u002Fp>\n\u003Ch2 id=\"section-103\">Checkliste für KI-Plattformarchitekten\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Frage\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Erwartete Antwort\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Wer sind die tatsächlichen Plattformkonsumenten?\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Benannte Lösungen, Teams oder Mandantenkontexte mit unterschiedlichen, aber überlappenden Bedürfnissen.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Was ist wirklich gemeinsam?\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Explizite Fähigkeitsliste, kein vages „KI-Backend“.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Was muss lösungsspezifisch bleiben?\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Domänenautorität, Geschäftsworkflow, Aufgabenabnahme und andere vom Workload bestimmte Belange.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Wie werden Modelle\u002FProvider dargestellt?\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Versionierte Provider-\u002FModellverträge mit Fähigkeiten und expliziter Fallback-Semantik.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Wie wird Identität weitergegeben?\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Benutzer-\u002FDienst-\u002FAnwendungs-\u002FMandantenkontext überlebt jeden privilegierten Anforderungspfad.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Wie wird Mandantenisolation durchgesetzt?\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Ressourcen-Scoping ist getrennt von Rollenberechtigungsprüfungen.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Wie werden Geheimnisse behandelt?\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Privilegierter Speicher, Rotation, begrenzte Exposition und auditierbare Eigentümerschaft.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Wie bewahrt Retrieval die Autorität?\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Gemeinsame Mechanismen mit Autorisierung, Herkunft und domäneneigenen Nachweisregeln.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Wie werden Tools und Agenten eingeschränkt?\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Laufzeitberechtigungen, begrenzte Tool-Verträge, Genehmigungen, Abbruch und Rückverfolgbarkeit.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Wie werden Kosten und Kapazität kontrolliert?\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Quoten, Token-\u002FRatenkontrollen, Nutzungszuordnung und Überlastverhalten.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Wie wird Qualität gemessen?\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Plattform-Regression\u002FEvaluierung plus lösungsspezifische Ground-Truth und Abnahme.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Wie werden Änderungen ausgerollt?\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Versionierung, Kompatibilität, Migration, Deprecation, Rollback und Vorfallverantwortung.\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-105\">Fazit\u003C\u002Fh2>\n\u003Cp>Ein KI-Plattformarchitekt ist verantwortlich für die wiederverwendbare Architektur \u003Cstrong>zwischen KI-Fähigkeiten und den Lösungen, die sie nutzen\u003C\u002Fstrong>. Die Rolle definiert, wie Modelle, Provider, Retrieval, Agenten, Tools, Identität, Mandanten, Geheimnisse, Evaluierung, Observability, Quoten und Laufzeitoperationen zu verlässlichen Plattformdiensten werden, anstatt wiederholte Einmalintegrationen zu sein.\u003C\u002Fp>\n\u003Cp>Der schwierige Teil ist nicht die Maximierung der Wiederverwendung. Es ist die Wahl der richtigen Grenze. Eine starke Plattform standardisiert Mechanismen, Richtlinien und Betrieb dort, wo mehrere Konsumenten wirklich profitieren, während sie lösungsspezifische Datenautorität, Geschäftslogik, Sicherheitsanforderungen und Abnahmekriterien bewahrt.\u003C\u002Fp>\n\u003Cp>Diese Unterscheidung erklärt auch die Beziehung zur KI-Lösungsarchitektur: \u003Cstrong>Der Lösungsarchitekt macht ein KI-fähiges System zweckmäßig; der Plattformarchitekt macht gemeinsame KI-Fähigkeiten sicher, wiederverwendbar, betreibbar und über viele solcher Systeme hinweg entwickelbar.\u003C\u002Fstrong>\u003C\u002Fp>\n\u003Ch2 id=\"section-109\">Verwandtes kanonisches Wissen\u003C\u002Fh2>\n\u003Cp>Dieser Artikel baut auf den kanonischen Grundlagen zu generativen KI-Komponenten, ADR versus NFR und KI-Lösungsarchitektur auf. Diese Konzepte sind Voraussetzungen, weil eine Plattform existiert, um wiederverwendbare Systemfähigkeiten bereitzustellen und architektonische Entscheidungen gegen explizite Qualitäts- und Betriebsanforderungen zu kodifizieren.\u003C\u002Fp>\n\u003Cp>Retrieval-Augmented Generation ist ein Beispiel für eine Fähigkeit, die über eine Plattform angeboten werden kann, aber die Plattform sollte Retrieval-Infrastruktur, Domänenwissen und Antwortvalidität nicht zu einem einzigen Konzept verschmelzen.\u003C\u002Fp>\n\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fde\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">Was ist RAG? Die einfachste Erklärung, wie es funktioniert\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">Kanonische Einführung in Retrieval-Augmented Generation und die Grenze zwischen Modellgenerierung und externem Wissensabruf.\u003C\u002Fp>\u003C\u002Fa>\n\u003Cp>Agentenprotokolle, Mandantentrennung, KI-Governance, Modell-Routing, Context Engineering und MLOps\u002FLLMOps sind nachgelagerte oder angrenzende Wissensknoten. Sie lassen sich leichter durchdenken, sobald die Plattformgrenze explizit ist.\u003C\u002Fp>\n\u003Ch2 id=\"section-114\">Häufig gestellte Fragen\u003C\u002Fh2>\n\u003Csection class=\"editorjs-faq my-6 rounded-xl border border-gray-200 p-5 dark:border-gray-700\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">FAQ zum KI-Plattform-Architekten\u003C\u002Fh3>\u003Cdiv id=\"faq-1\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">Ist ein KI-Plattform-Architekt dasselbe wie ein KI-Lösungsarchitekt?\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">Nein. Der Lösungsarchitekt konzentriert sich auf eine konkrete KI-fähige Lösung. Der Plattformarchitekt konzentriert sich auf wiederverwendbare KI-Fähigkeiten, Kontrollen und Betriebsverträge, die mehrere Lösungen unterstützen können.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq-2\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">Muss eine KI-Plattform eigene Modelle hosten?\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">Nein. Eine Plattform kann verwaltete Cloud-Modelle, selbst gehostete Modelle, lokale Inferenz oder eine Hybridstrategie nutzen. Die Architektur muss Anbieter-, Lokalitäts-, Identitäts-, Routing-, Daten- und Betriebskonsequenzen explizit machen.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq-3\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">Reicht ein KI-Gateway aus, um eine KI-Plattform zu sein?\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">Normalerweise nicht. Ein Gateway kann eine wichtige Plattformkomponente sein, aber eine vollständige Plattform benötigt auch Verträge für Identität, Secrets, Daten\u002FRetrieval, Evaluierung, Observability, Lebenszyklus und Betriebsverantwortung.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq-4\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">Sollte Retrieval zentralisiert werden?\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">Retrieval-Mechaniken können oft geteilt werden, aber Domänenautorität, Autorisierung, Aktualität, Evidenzausreichendheit und Korpus-Eigentümerschaft sollten explizit bleiben. Geteilte Infrastruktur impliziert keine geteilte Wahrheit.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq-5\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">Ersetzt die Plattform-Evaluierung die Anwendungs-Evaluierung?\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">Nein. Die Plattform-Evaluierung kann geteilte Fähigkeiten und Regressionen testen. Jede Lösung benötigt weiterhin aufgabenspezifische Ground Truth, Akzeptanzkriterien und Domänen-Qualitätsschwellen.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq-6\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">Ist Multi-Tenancy nur RBAC?\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">Nein. RBAC bestimmt, was eine Identität tun darf. Mandantentrennung bestimmt, auf welche Ressourcen eines Mandanten die Identität einwirken darf. Eine Plattform benötigt oft beides.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-116\">Glossar\u003C\u002Fh2>\n\u003Csection class=\"editorjs-glossary my-6 rounded-xl border border-gray-200 dark:border-gray-700 p-5\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">Wichtige Begriffe der KI-Plattformarchitektur\u003C\u002Fh3>\u003Cdl>\u003Cdiv id=\"ai-platform\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">KI-Plattform\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Eine wiederverwendbare Menge KI-bezogener technischer und betrieblicher Fähigkeiten, die von mehreren Anwendungen, Teams oder Mandantenkontexten genutzt werden.\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"ai-gateway\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">KI-Gateway\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Eine Gateway-Schicht für KI-Endpunkte, die über einfaches Proxying hinaus Authentifizierung, Routing, Quoten, Richtlinien, Wiederholungen, Kostenzuordnung und KI-spezifische Telemetrie hinzufügen kann.\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"provider-adapter\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">Anbieter-Adapter\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Eine Komponente, die einen Plattformvertrag auf die API, Fähigkeiten, Gesundheit und Fehlersemantik eines Modellanbieters abbildet.\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"tenant-isolation\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">Mandantentrennung\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Die Grenze, die verhindert, dass ein Mandantenkontext auf die Ressourcen eines anderen Mandanten zugreift, unabhängig von Rollenberechtigungen.\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"capability-contract\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">Fähigkeitsvertrag\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Eine versionierte Schnittstelle und Verhaltensvereinbarung, die beschreibt, was ein geteilter Plattformdienst bereitstellt und was der Konsument liefern oder besitzen muss.\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"grounding-service\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">Grounding-\u002FRetrieval-Dienst\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Geteilte Mechanismen zum Finden und Bereitstellen externer Informationen für eine KI-Workload; er definiert nicht automatisch, welche Informationen für eine Domäne maßgeblich sind.\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"evaluation-harness\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">Evaluierungs-Harness\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Wiederverwendbare Infrastruktur zum Ausführen von Tests, Datensätzen, Modell-\u002FPrompt-Versionen und Metriken; die Domänenakzeptanz bleibt lösungsspezifisch.\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"control-plane\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">Control Plane\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Die Konfigurations- und Governance-Schicht, die Plattformfähigkeiten, Identitäten, Richtlinien, Quoten, Versionen und Bereitstellungszustand verwaltet.\u003C\u002Fdd>\u003C\u002Fdiv>\u003C\u002Fdl>\u003C\u002Fsection>\n\u003Ch2 id=\"section-118\">Primärquellen und aktuelle Architekturempfehlungen\u003C\u002Fh2>\n\u003Cp>Die folgenden Quellen stützen die allgemeinen Architektur- und Produktionsplattform-Aussagen. Die Abschnitte Aaasaasa AI Client, Aaasaasa AI CMS und Source of Truth Research Engine sind ausdrücklich originäre Implementierungsnachweise. Aktuelle externe Referenzen wurden am 8. Oktober 2026 geprüft.\u003C\u002Fp>\n\u003Ca href=\"https:\u002F\u002Fwww.iso.org\u002Fstandard\u002F74393.html\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">ISO\u002FIEC\u002FIEEE 42010:2022 — Architekturbeschreibung\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">Aktuell veröffentlichter internationaler Standard für Konzepte und Beziehungen der Architekturbeschreibung.\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fwww.nist.gov\u002Fitl\u002Fai-risk-management-framework\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">NIST AI Risk Management Framework\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">NISTs AI RMF-Ressourcen und aktueller Status; AI RMF 1.0 befindet sich Stand Oktober 2026 in Überarbeitung.\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fwww.nist.gov\u002Fpublications\u002Fartificial-intelligence-risk-management-framework-generative-artificial-intelligence\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">NIST AI 600-1 — Generative AI Profile\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">Generative-AI-Profil zur Anwendung von KI-Risikomanagement-Überlegungen über den KI-Lebenszyklus.\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Flearn.microsoft.com\u002Fen-us\u002Fazure\u002Fwell-architected\u002Fai\u002Fget-started\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">Microsoft Azure Well-Architected — AI Workloads\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">Aktuelle Architekturempfehlungen zu KI-Anwendung, Daten, Betrieb, Evaluierung, verantwortungsvoller KI und Lebenszyklusaspekten.\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Flearn.microsoft.com\u002Fen-us\u002Fazure\u002Fwell-architected\u002Fai\u002Fdesign-principles\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">Microsoft — Design Principles for AI Workloads\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">Aktuelle Empfehlungen zu Identitätssegmentierung, Sicherheitsgrenzen, Telemetrie, Leistung, Daten und Plattform-Abwägungen.\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Flearn.microsoft.com\u002Fen-us\u002Fazure\u002Fai-foundry\u002Fconfiguration\u002Fenable-ai-api-management-gateway-portal?view=foundry\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">Microsoft Foundry — AI Gateway Architecture\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">Aktuelle AI-Gateway-Empfehlungen für gemeinsamen Projektzugriff, Token-Begrenzung, Quoten und Governance.\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Flearn.microsoft.com\u002Fen-us\u002Fazure\u002Farchitecture\u002Fai-ml\u002Fguide\u002Fazure-openai-gateway-guide\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">Azure Architecture Center — Access Models Through a Gateway\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">Architekturempfehlungen für zentralisierten Modellzugriff, Routing, Drosselung, Failover und Verantwortlichkeiten von Client und Plattform.\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fdocs.aws.amazon.com\u002Fwellarchitected\u002Flatest\u002Fgenerative-ai-lens\u002F\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">AWS Well-Architected — Generative AI Lens\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">Aktuelle Architekturrichtlinien für generative KI-Workloads in den Bereichen Sicherheit, Zuverlässigkeit, Betrieb, Leistung und Kosten.\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fdocs.aws.amazon.com\u002Fwellarchitected\u002Flatest\u002Fgenerative-ai-lens\u002Fmulti-tenant-generative-ai-platform-scenario.html\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">AWS — Multi-Tenant-Szenario für generative KI-Plattformen\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">Aktuelles Beispiel, das zentrale Plattformkontrollen und Auditierbarkeit von der Datenqualität der nutzenden Anwendungen und workloadspezifischen Verantwortlichkeiten trennt.\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fdocs.aws.amazon.com\u002Fwellarchitected\u002Flatest\u002Fagentic-ai-lens\u002Fdesign-principles.html\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">AWS Well-Architected — Designprinzipien für agentische KI\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">Aktuelle Richtlinien zu begrenzter Agentenautorität, Nachverfolgbarkeit, versioniertem Verhalten, expliziten Verträgen und menschlicher Aufsicht.\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fdocs.aws.amazon.com\u002FAmazonCloudWatch\u002Flatest\u002Fmonitoring\u002FGenAI-observability.html\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">AWS CloudWatch — Observability für generative KI\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">Aktuelle Observability-Funktionen und Produktionsmetriken für Modelle, Agenten, Wissensdatenbanken, Tools sowie Kosten-, Latenz- und Fehleranalysen.\u003C\u002Fp>\u003C\u002Fa>",{"time":211,"blocks":212,"version":1200},1791477334574,[213,218,225,231,236,243,247,251,255,299,303,307,311,315,340,344,348,352,356,387,393,397,401,405,409,415,419,423,427,431,435,439,443,447,451,455,459,463,467,471,475,479,483,487,491,495,499,503,507,511,515,519,523,527,531,535,540,560,564,568,602,606,654,658,681,685,689,694,698,702,706,710,732,736,740,744,748,752,756,760,764,769,773,777,781,785,789,793,797,828,832,866,870,905,909,913,917,921,925,929,933,937,941,945,988,992,996,1000,1004,1008,1012,1016,1026,1030,1034,1063,1067,1104,1108,1112,1120,1128,1136,1144,1152,1160,1168,1176,1184,1192],{"id":214,"data":215,"type":217},"intro",{"text":216},"Ein \u003Cstrong>AI Platform Architect\u003C\u002Fstrong> entwirft das wiederverwendbare KI-Fundament, über das mehrere Anwendungen, Teams oder Mandantenkontexte auf Modelle, Daten und Retrieval, Agenten- und Tool-Runtimes, Identität und Berechtigungen, Evaluierung, Observability, Quotas, Secrets und Deployment-Fähigkeiten zugreifen. Die Rolle ist breiter als Infrastruktur, aber enger als das Eigentum an jedem KI-fähigen Produkt: Ihre zentrale Verantwortung besteht darin, zu entscheiden, \u003Cstrong>was geteilt werden sollte, wie geteilte Fähigkeiten gesteuert und isoliert werden und was lösungsspezifisch bleiben muss\u003C\u002Fstrong>.","paragraph",{"id":219,"data":220,"type":224},"direct",{"body":221,"title":222,"variant":223},"\u003Cstrong>Ein AI Platform Architect entwirft das gemeinsame technische und operative Substrat für KI-Systeme.\u003C\u002Fstrong> Statt einen einzelnen Assistenten oder einen einzelnen Workflow zu architektieren, definiert die Rolle wiederverwendbare Verträge und Grenzen für Modell-\u002FAnbieterzugriff, Gateways und Routing, Retrieval-Dienste, Agenten-Runtimes, Tool-Zugriff, Identität und Mandantenisolierung, Secrets, Evaluierung, Telemetrie, Deployment und Lifecycle-Management.","Direkte Antwort","info","callout",{"id":226,"data":227,"type":224},"term-note",{"body":228,"title":229,"variant":230},"\u003Cstrong>AI Platform Architect ist eine praktische Rollenbezeichnung, kein universell standardisierter Jobtitel.\u003C\u002Fstrong> ISO\u002FIEC\u002FIEEE 42010:2022 definiert Konzepte für Architekturbeschreibungen, nicht diese Rolle. Verschiedene Organisationen können diese Verantwortlichkeiten auf Plattformarchitekten, Lösungsarchitekten, Unternehmensarchitekten, Sicherheitsarchitekten, MLOps\u002FLLMOps-Spezialisten und Plattform-Engineering-Teams aufteilen. Dieser Artikel verwendet den Begriff für die Architekturverantwortung über eine wiederverwendbare KI-Plattformschicht.","Terminologiehinweis","note",{"id":232,"data":233,"type":224},"version-note",{"body":234,"title":235,"variant":230},"Die hier dargestellten stabilen Architekturprinzipien sind anbieterneutral. Aktuelle Microsoft-, AWS- und NIST-Leitlinien werden als externe Implementierungs- und Governance-Belege verwendet. NIST gibt an, dass AI RMF 1.0 überarbeitet wird; Anbieterplattformfunktionen, Gateway-Produkte, Agenten-Runtimes und Modellfähigkeiten entwickeln sich schneller als die Architekturprinzipien, sodass versionsabhängige Implementierungsentscheidungen vor dem Deployment erneut überprüft werden müssen.","Hinweis zu aktuellen Quellen — 8. Oktober 2026",{"id":237,"data":238,"type":242},"toc",{"title":239,"maxLevel":240,"minLevel":241},"Inhalt",3,2,"tableOfContents",{"id":244,"data":245,"type":41},"h-meaning",{"text":246,"level":241},"Was architektiert ein AI Platform Architect tatsächlich?",{"id":248,"data":249,"type":217},"p-meaning-1",{"text":250},"Der Gegenstand der Arbeit ist die \u003Cstrong>Plattform\u003C\u002Fstrong>: eine Reihe gemeinsamer Fähigkeiten, die wiederholte Integrationsarbeit reduziert und dabei explizite Sicherheits-, Daten- und Betriebsgrenzen wahrt. Eine Plattform kann Modellzugriff, Anbieteradapter, Retrieval-Primitive, Agentenausführung, Tool-Broker, Richtliniendurchsetzung, Evaluierung, Telemetrie und Deployment-Dienste für viele konsumierende Lösungen bereitstellen.",{"id":252,"data":253,"type":217},"p-meaning-2",{"text":254},"Die Plattform ist nicht allein deshalb wertvoll, weil Komponenten zentralisiert sind. Sie ist wertvoll, wenn Konsumenten stabile Fähigkeiten mit klaren Verträgen, Eigentum, Isolierung, Observability und Lifecycle-Regeln erhalten. Die zentrale Architekturfrage ist daher nicht „Welches Modell sollte jeder verwenden?“, sondern \u003Cstrong>„Welche Verantwortlichkeiten können sicher standardisiert und wiederverwendet werden, ohne die Anforderungen jeder Lösung zu verwischen?“\u003C\u002Fstrong>.",{"id":256,"data":257,"type":298},"solution-vs-platform",{"rows":258,"title":289,"layout":290,"columns":291},[259,265,271,277,283],{"id":260,"label":261,"values":262},"c1","Primärer Umfang",{"platform":263,"solution":264},"Reusable AI capabilities consumed by multiple solutions, teams or tenant contexts.","One concrete AI-enabled product, workflow or application.",{"id":266,"label":267,"values":268},"c2","Hauptfrage",{"platform":269,"solution":270},"Which shared capabilities and controls should solutions consume, and where must solution-specific ownership remain?","How should this solution meet its business, data, security, quality and operational requirements?",{"id":272,"label":273,"values":274},"c3","Datenhoheit",{"platform":275,"solution":276},"Provides storage, retrieval, provenance or access primitives without automatically becoming the authority for every domain.","Defines which domain data is authoritative and how the solution may use it.",{"id":278,"label":279,"values":280},"c4","Evaluierung",{"platform":281,"solution":282},"Provides reusable evaluation, telemetry and release mechanisms; it cannot define every domain's success threshold.","Defines task-specific quality and acceptance criteria.",{"id":284,"label":285,"values":286},"c5","Lifecycle",{"platform":287,"solution":288},"Owns shared capability versions, compatibility, onboarding, quotas, policy and operational contracts.","Owns the lifecycle of the specific workload.","Lösungsarchitektur und Plattformarchitektur lösen unterschiedliche Umfangsprobleme","table",[292,295],{"id":293,"label":294},"solution","AI Solution Architect",{"id":296,"label":297},"platform","AI Platform Architect","comparison",{"id":300,"data":301,"type":41},"h-simple",{"text":302,"level":241},"Das einfachste Beispiel",{"id":304,"data":305,"type":217},"p-simple-1",{"text":306},"Stellen Sie sich vor, ein Unternehmen hat fünf KI-fähige Produkte: einen internen Dokumentenassistenten, einen Copiloten für den Kundensupport, einen Software-Engineering-Agenten, einen Vertragsprüfungs-Workflow und einen Produktsuchassistenten. Jedes Produkt könnte unabhängig Modell-APIs integrieren, Anmeldedaten speichern, Wiederholungsversuche implementieren, Token-Metriken sammeln, Retrieval-Code erstellen und eigene Tool-Berechtigungen aufbauen.",{"id":308,"data":309,"type":217},"p-simple-2",{"text":310},"Diese Duplizierung ist teuer und gefährlich, wenn jedes Team ein anderes Sicherheits- und Betriebsmodell erfindet. Eine gemeinsame Plattform kann stattdessen genehmigte Anbieterverbindungen, Modellermittlung, Quotas, Anmeldedaten, mandantenfähigen Zugriff, gemeinsame Telemetrie, wiederverwendbare Retrieval-Dienste und einen Agenten-\u002FTool-Runtime-Vertrag anbieten.",{"id":312,"data":313,"type":217},"p-simple-3",{"text":314},"Aber die Plattform muss an der richtigen Grenze haltmachen. Die Vertragsprüfungslösung kann rechtliche Dokumentenhoheit und Zitierregeln erfordern, die der Software-Agent nicht benötigt. Der Produktsuchassistent kann handelsspezifische Aktualitäts- und Autorisierungsregeln benötigen. \u003Cstrong>Wiederverwendbare Infrastruktur macht nicht alle Domänenwahrheit wiederverwendbar.\u003C\u002Fstrong>",{"id":316,"data":317,"type":339},"simple-flow",{"steps":318,"title":337,"orientation":338},[319,322,325,328,331,334],{"label":320,"description":321},"1. Konsument identifiziert sich","Die aufrufende Anwendung, der Benutzer, der Dienst, das Team oder der Mandant tritt über eine authentifizierte Identität und einen expliziten Geltungsbereich ein.",{"label":323,"description":324},"2. Plattformrichtlinie wird angewendet","Gateway- und Richtlinienschichten bestimmen erlaubte Anbieter, Modelle, Quotas, Datenpfade, Tools und Ausführungsmodi.",{"label":326,"description":327},"3. Gemeinsame Fähigkeit wird ausgeführt","Die Anfrage kann Inferenz, Retrieval, Agenten-Runtime, Tool-Zugriff oder einen anderen wiederverwendbaren Plattformdienst nutzen.",{"label":329,"description":330},"4. Lösungsspezifischer Kontext bleibt maßgeblich","Die konsumierende Lösung liefert Domänenregeln, Benutzerabsicht, Datenhoheit, aufgabenspezifische Einschränkungen und Akzeptanzlogik.",{"label":332,"description":333},"5. Telemetrie und Belege werden erfasst","Die Plattform zeichnet Identität, Route, Modell\u002FAnbieter, Latenz, Kosten, Fehler, Tool-Aktivität und andere zulässige Observability-Signale auf.",{"label":335,"description":336},"6. Ergebnis kehrt unter dem Lösungsvertrag zurück","Die Lösung bleibt dafür verantwortlich, ob die Ausgabe für ihren Benutzer und ihre Domäne akzeptabel ist.","Ein gemeinsamer KI-Anfragepfad","auto","processFlow",{"id":341,"data":342,"type":41},"h-stops",{"text":343,"level":241},"Wo das einfache Beispiel endet",{"id":345,"data":346,"type":217},"p-stops-1",{"text":347},"Zentralisierung ist nicht automatisch Architektur. Ein einzelner Endpunkt vor mehreren Modell-APIs ist nützlich, aber er schafft für sich genommen noch keine KI-Plattform. Eine Produktionsplattform benötigt außerdem Identitätsgrenzen, Fähigkeitsverträge, Anbieterzustand und Lifecycle-Handhabung, Quotas, Secret-Eigentum, Observability, Kompatibilitätsregeln, Sicherheitskontrollen, Release-Disziplin und klare operative Verantwortung.",{"id":349,"data":350,"type":217},"p-stops-2",{"text":351},"Das gegenteilige Versagen ist ebenfalls häufig: jeden Prompt, jeden Vektorindex, jede Geschäftsregel, jeden Agenten und jeden Anwendungs-Workflow in ein einziges „KI-Backend“ zu packen. Das erzeugt einen Monolithen, dessen gemeinsamer Status zufällig statt architektonisch ist. \u003Cstrong>Eine Plattform sollte querschnittliche Fähigkeiten standardisieren, nicht Domäneneigentum absorbieren, nur weil KI beteiligt ist.\u003C\u002Fstrong>",{"id":353,"data":354,"type":41},"h-boundary",{"text":355,"level":241},"Die wichtigste Plattformentscheidung: gemeinsam versus lösungsspezifisch",{"id":357,"data":358,"type":290},"shared-boundary-table",{"content":359,"stretched":42,"withHeadings":13},[360,364,368,372,376,380,383],[361,362,363],"Fähigkeitsbereich","Guter Kandidat für gemeinsame Plattformverantwortung","Bleibt üblicherweise lösungsspezifisch",[365,366,367],"Modellzugriff","Genehmigte Anbieterverbindungen, Adapter, Anmeldedaten, Health, Routing-Primitive, Quoten","Aufgabenspezifische Modellakzeptanz, Prompt-Verhalten, Qualitätsschwelle",[369,370,371],"Retrieval","Ingestion-Primitive, Extraktion, Indexierung, Such-APIs, Provenienz-Verträge, Autorisierungs-Hooks","Autoritativer Korpus, Aktualitätsregeln, Domänen-Metadaten, Evidenz-Suffizienz",[373,374,375],"Agenten und Tools","Runtime-Lebenszyklus, Tool-Registry\u002FBroker, Berechtigungsdurchsetzung, Tracing, Abbruch","Geschäftsworkflow, erlaubte Aktionssemantik, Eskalationsrichtlinie, Aufgabenerfolg",[377,378,379],"Sicherheit","Identitätsintegration, Secret-Speicherung, Richtliniendurchsetzung, Audit-Verträge, Mandantenisolationsmechanismen","Datenklassifizierung, geschäftliche Autorisierungsregeln, domänenspezifische Risikoakzeptanz",[279,381,382],"Harness, Dataset-\u002FVersionsmechanik, Telemetrie, Experiment-\u002FRelease-Workflow","Ground Truth, Domänen-Testset, Akzeptanzschwelle, Nutzerergebnis",[384,385,386],"Betrieb","Deployment-Muster, Health, Metriken, Incident-Integration, Kapazitätssteuerung","Lösungs-SLOs, wo sie abweichen, Auswirkungen auf die Geschäftskontinuität, workloadspezifische Runbooks",{"id":388,"data":389,"type":224},"boundary-principle",{"body":390,"title":391,"variant":392},"\u003Cstrong>Teile Mechaniken und Kontrollen, wo Wiederverwendung real ist; behalte Autorität und Akzeptanz dort, wo die Domäne sie besitzt.\u003C\u002Fstrong> Dies verhindert zwei gegensätzliche Fehler: überall duplizierte Infrastruktur und eine zentrale Plattform, die fälschlicherweise zum Eigentümer der Daten, Richtlinien und Qualität jeder Anwendung wird.","Plattformprinzip","success",{"id":394,"data":395,"type":41},"h-responsibility-map",{"text":396,"level":241},"Architektur-Verantwortlichkeitskarte",{"id":398,"data":399,"type":41},"h-provider",{"text":400,"level":240},"1. Modell- und Anbieterzugriff",{"id":402,"data":403,"type":217},"p-provider-1",{"text":404},"Ein Plattformarchitekt definiert, wie Konsumenten Modelle entdecken und aufrufen, ohne jede Anwendung zu zwingen, einen Anbieter fest zu codieren. Dies umfasst Anbieteradapter, Modellkennungen, Fähigkeitsmetadaten, Authentifizierung, Health-Checks, Endpunktkonfiguration, Request-Normalisierung und Kompatibilitätsverhalten.",{"id":406,"data":407,"type":217},"p-provider-2",{"text":408},"Anbieterabstraktion muss ehrlich bleiben. Verschiedene Anbieter bieten unterschiedliche Kontextgrenzen, Tool-Semantik, strukturiertes Ausgabeverhalten, multimodale Fähigkeiten, Sicherheitskontrollen, Caching, Preisgestaltung und Fehlermodi. Eine gute Abstraktion schafft einen stabilen Plattformvertrag und bewahrt gleichzeitig den Zugriff auf Fähigkeiten, die nicht sinnvoll vereinheitlicht werden können.",{"id":410,"data":411,"type":224},"provider-warning",{"body":412,"title":413,"variant":414},"Eine API auf dem kleinsten gemeinsamen Nenner kann die Migration erleichtern, aber auch Fähigkeiten auslöschen, die wichtig sind. Die Architektur sollte definieren, welche Funktionen portabel sind, welche anbieterspezifisch sind und wie Konsumenten diesen Unterschied erkennen.","Verwechsle Abstraktion nicht damit, Anbieter als identisch darzustellen","warning",{"id":416,"data":417,"type":41},"h-gateway",{"text":418,"level":240},"2. Gateway, Routing, Quoten und Kostenkontrollen",{"id":420,"data":421,"type":217},"p-gateway-1",{"text":422},"Ein gemeinsames KI-Gateway kann Authentifizierung, Routing, Drosselung, Wiederholungen, Token-Limits, Nutzungszuordnung und Richtliniendurchsetzung zentralisieren. Microsofts aktuelle AI-Gateway-Richtlinien behandeln Token-pro-Minute-Limits, Quoten und Multi-Projekt-Eingrenzung ausdrücklich als Plattformbelange; AWS stellt ebenfalls Konto- und Modellquoten sowie zentrale Kontrollen bereit.",{"id":424,"data":425,"type":217},"p-gateway-2",{"text":426},"Das Gateway ist daher mehr als ein Reverse Proxy, wenn es KI-spezifische Richtlinien- und Betriebssemantik trägt. Es sollte jedoch nicht stillschweigend Geschäftsentscheidungen treffen. Eine Routing-Richtlinie kann ein gesundes lokales Modell, einen kostengünstigeren Anbieter oder einen regional konformen Endpunkt bevorzugen; ob diese Route für eine bestimmte Aufgabe akzeptabel ist, bleibt dennoch ein Vertrag zwischen Plattform und Lösung.",{"id":428,"data":429,"type":217},"p-gateway-3",{"text":430},"Routing benötigt auch Fehlersemantik. Wenn das bevorzugte Modell nicht verfügbar ist, muss die Plattform wissen, ob ein Fallback erlaubt ist, ob eine Cloud-Route eine ausdrückliche Zustimmung erfordert, ob ein Modell mit geringerer Fähigkeit gültig ist und wie die Entscheidung für die Observability sichtbar gemacht wird.",{"id":432,"data":433,"type":41},"h-data",{"text":434,"level":240},"3. Gemeinsame Daten-, Retrieval- und Grounding-Dienste",{"id":436,"data":437,"type":217},"p-data-1",{"text":438},"Retrieval-Dienste sind starke Plattformkandidaten, weil Parsing, Chunking, Indexierung, lexikalische Suche, semantische Suche, Metadatenfilterung, Provenienz und Zitationsmechanik wiederverwendbar sind. Die Plattform darf jedoch eine gemeinsame Retrieval-Engine nicht mit einer gemeinsamen Quelle der Wahrheit verwechseln.",{"id":440,"data":441,"type":217},"p-data-2",{"text":442},"Eine Lösung besitzt weiterhin Fragen wie: Welcher Korpus ist autoritativ? Welche Version ist gültig? Darf dieser Nutzer dieses Dokument sehen? Wie aktuell müssen die Daten sein? Was zählt als ausreichende Evidenz? Kann eine Antwort generiert werden, wenn das Retrieval fehlschlägt? Das sind Domänen- und Lösungsanforderungen, selbst wenn die Plattform die Retrieval-Maschinerie bereitstellt.",{"id":444,"data":445,"type":217},"p-data-3",{"text":446},"Diese Grenze ist besonders wichtig in Multi-Tenant-Systemen. Ein technisch gemeinsamer Index oder Vektordienst rechtfertigt keine mandantenübergreifende Sichtbarkeit. Der Autorisierungskontext muss durch das Retrieval hindurch erhalten bleiben und darf nicht erst hinzugefügt werden, nachdem Suchergebnisse die Grenze bereits überschritten haben.",{"id":448,"data":449,"type":41},"h-agent-runtime",{"text":450,"level":240},"4. Agenten- und Tool-Runtime",{"id":452,"data":453,"type":217},"p-agent-1",{"text":454},"Agentische Systeme fügen wiederverwendbare Runtime-Belange hinzu: Thread-\u002FSession-Lebenszyklus, Planungsschleifen, Tool-Registrierung, Tool-Aufruf, Abbruch, Timeouts, menschliche Genehmigungen, Memory-\u002FState-Schnittstellen, Remote-Agent-Protokolle und Trace-Korrelation. Eine Plattform kann diese Mechaniken bereitstellen, damit jedes Produkt sie nicht neu aufbauen muss.",{"id":456,"data":457,"type":217},"p-agent-2",{"text":458},"Die Plattform muss auch die Tool-Berechtigung von der Modellfähigkeit getrennt halten. Dass ein Modell einen Shell-Befehl generieren kann, bedeutet nicht, dass die Runtime die Shell-Ausführung erlauben sollte. Die Berechtigungsgrenze gehört zur Anwendungs-\u002FRuntime-Architektur und muss unabhängig vom Modell durchsetzbar sein.",{"id":460,"data":461,"type":217},"p-agent-3",{"text":462},"Die aktuelle AWS-Agentic-AI-Richtlinie betont begrenzte Agenten, explizite Befugnisse, End-to-End-Tracing, versionierte Verhaltensartefakte und menschliche Aufsicht im Verhältnis zu den Konsequenzen. Das sind plattformermöglichende Belange, aber die konsumierende Lösung definiert weiterhin, welche Aktionen für ihre Domäne legitim sind.",{"id":464,"data":465,"type":41},"h-identity",{"text":466,"level":240},"5. Identität, Mandantentrennung und Autorisierung",{"id":468,"data":469,"type":217},"p-identity-1",{"text":470},"KI-Plattformen stehen oft vor hochwertigen Modellen, proprietären Daten und aktionsfähigen Tools. Authentifizierung ist daher nur der Anfang. Die Architektur muss Benutzer-, Dienst-, Anwendungs- und Mandantenkontext durch jeden privilegierten Vorgang tragen, der ihn benötigt.",{"id":472,"data":473,"type":217},"p-identity-2",{"text":474},"\u003Cstrong>RBAC und Mandantentrennung lösen unterschiedliche Probleme.\u003C\u002Fstrong> RBAC beantwortet, was eine Identität tun darf; Mandantentrennung beantwortet, auf welche Ressourcen welches Mandanten diese Identität einwirken darf. Eine Plattform, die Rollen prüft, aber den Mandantenkontext verliert, kann dennoch die falschen Daten offenlegen.",{"id":476,"data":477,"type":217},"p-identity-3",{"text":478},"Die aktuelle KI-Workload-Richtlinie von Microsoft empfiehlt ausdrücklich Identitätssegmentierung und autorisierungsbewussten Zugriff auf Inhalte. Die AWS-Richtlinie für mandantenfähige generative KI-Plattformen behandelt logische Isolierung, zentralisierte Kontrollen und Auditierbarkeit ebenfalls als Plattformbelange.",{"id":480,"data":481,"type":41},"h-secrets",{"text":482,"level":240},"6. Geheimnisse, Anmeldeinformationen und Vertrauensgrenzen",{"id":484,"data":485,"type":217},"p-secrets-1",{"text":486},"Eine Plattform sollte definieren, wem Anbieterschlüssel, entfernte Bearer-Tokens, Signiermaterial und Tool-Anmeldeinformationen gehören, wo sie gespeichert sind, welcher Prozess auf sie zugreifen kann, wie sie rotiert werden und ob sie jemals einen Browser oder einen nicht vertrauenswürdigen Renderer erreichen können.",{"id":488,"data":489,"type":217},"p-secrets-2",{"text":490},"Dies ist eine architektonische Grenze, kein Implementierungsdetail. Wenn jede konsumierende Anwendung Anbieter-Anmeldeinformationen in ihre eigene Konfiguration kopiert, hat die Organisation sowohl den operativen Aufwand als auch den Wirkungsradius dupliziert. Zentralisierung kann dieses Risiko nur reduzieren, wenn die Plattform selbst engere, auditierbare Zugriffspfade hat.",{"id":492,"data":493,"type":41},"h-eval",{"text":494,"level":240},"7. Evaluierung, Beobachtbarkeit und Auditierbarkeit",{"id":496,"data":497,"type":217},"p-eval-1",{"text":498},"Eine wiederverwendbare Plattform kann Evaluierungs-Harnesses, Trace-IDs, Modell-\u002FAnbieter-Metadaten, Token- und Kostenmetriken, Latenz, Fehlerraten, Verknüpfung von Prompt-\u002FModellversionen, Agenten-\u002FTool-Traces und kontrolliertes Logging bereitstellen. AWS und Microsoft behandeln sowohl Beobachtbarkeit als auch Evaluierung als zentrale Produktionsbelange für KI-Workloads.",{"id":500,"data":501,"type":217},"p-eval-2",{"text":502},"Plattform-Evaluierung und Lösungs-Evaluierung müssen getrennt bleiben. Eine Plattform kann verifizieren, dass ein Endpunkt gesund ist, eine Modellversion eine allgemeine Regressionssuite besteht und Traces vollständig sind. Sie kann nicht entscheiden, dass eine rechtliche Antwort, ein medizinischer Workflow oder eine Produktempfehlung akzeptabel ist, ohne domänenspezifische Ground Truth und Akzeptanzkriterien.",{"id":504,"data":505,"type":217},"p-eval-3",{"text":506},"Logging schafft auch eine Datenschutzgrenze. Prompt- und Antwortprotokolle können sensible oder proprietäre Daten enthalten. Der Plattformarchitekt muss daher entscheiden, was protokolliert, redigiert, gesampelt, aufbewahrt und zugänglich gemacht wird, anstatt anzunehmen, dass mehr Telemetrie immer sicherer ist.",{"id":508,"data":509,"type":41},"h-runtime",{"text":510,"level":240},"8. Laufzeit, Bereitstellung und Lokalität",{"id":512,"data":513,"type":217},"p-runtime-1",{"text":514},"Ein Plattformarchitekt entscheidet, wie gemeinsame KI-Fähigkeiten bereitgestellt und erreicht werden: verwaltete Cloud-Dienste, selbst gehostete Endpunkte, lokale Inferenz, hybrides Routing, containerisierte Dienste, Desktop-Laufzeiten, private Netzwerke oder air-gapped Umgebungen. Die wichtige Unterscheidung ist zwischen \u003Cstrong>wo der Steuerungs-\u002FLaufzeitprozess läuft\u003C\u002Fstrong> und \u003Cstrong>wo Inferenz und Datenverarbeitung tatsächlich stattfinden\u003C\u002Fstrong>.",{"id":516,"data":517,"type":217},"p-runtime-2",{"text":518},"Ein lokaler Client kann dennoch ein Cloud-Modell aufrufen. Eine Cloud-Steuerungsebene kann zu einem On-Premises-Modell routen. Ein Remote-Agent kann Tools innerhalb eines Kundennetzwerks ausführen. Architekturdiagramme müssen daher Vertrauens- und Datenflussgrenzen zeigen, anstatt „lokal“ und „Cloud“ als vage Bezeichnungen zu verwenden.",{"id":520,"data":521,"type":41},"h-lifecycle",{"text":522,"level":240},"9. Plattform-Lebenszyklus, Kompatibilität und Onboarding",{"id":524,"data":525,"type":217},"p-lifecycle-1",{"text":526},"Wiederverwendbare Fähigkeit wird erst dann zu einer Plattform, wenn Konsumenten sich im Laufe der Zeit darauf verlassen können. Das erfordert versionierte Verträge, Migrationsregeln, Kompatibilitätsrichtlinien, Deprecation, Release-Tests, Rollback, Incident-Verantwortung, Kapazitätsplanung, Dokumentation und einen Weg zum Onboarding neuer Teams oder Anwendungen.",{"id":528,"data":529,"type":217},"p-lifecycle-2",{"text":530},"Sich schnell entwickelnde KI-Ökosysteme machen dies besonders wichtig. Modellnamen, SDKs, Protokollversionen, Anbieter-APIs und Sicherheitsfähigkeiten ändern sich unabhängig voneinander. Eine Plattform muss einen Teil dieser Volatilität absorbieren, ohne Änderungen zu verbergen, die das Verhalten einer Lösung wesentlich beeinflussen.",{"id":532,"data":533,"type":41},"h-control-plane",{"text":534,"level":241},"Ein praktisches Modell für Control-Plane \u002F Execution-Plane \u002F Solution-Plane",{"id":536,"data":537,"type":224},"model-note",{"body":538,"title":539,"variant":230},"Das untenstehende Drei-Plane-Modell ist eine praktische Methode, um über Verantwortlichkeiten nachzudenken; es ist kein ISO-, NIST-, Microsoft- oder AWS-Standard. Sein Zweck ist es, Eigentumsgrenzen explizit zu machen.","Vorgeschlagenes Architekturmodell",{"id":541,"data":542,"type":290},"planes-table",{"content":543,"stretched":42,"withHeadings":13},[544,548,552,556],[545,546,547],"Plane","Typische Verantwortlichkeiten","Sollte nicht stillschweigend besitzen",[549,550,551],"Plattform-Control-Plane","Provider-Registry, Modellrichtlinie, Quoten, Mandantenkonfiguration, Identitäten, Secrets, Routing-Regeln, Fähigkeitsversionen, Bereitstellungskonfiguration","Anwendungsgeschäftslogik oder Domänenwahrheit",[553,554,555],"Plattform-Execution-\u002FData-Plane","Inferenzanfragen, Retrieval-Operationen, Agent-\u002FTool-Ausführung, Extraktion, Indexierung, Telemetrie-Emission, Richtliniendurchsetzung","Mandantenübergreifender Zugriff allein aufgrund gemeinsam genutzter Infrastruktur",[557,558,559],"Solution-Plane","Benutzer-Workflow, Prompts\u002FAnweisungen, autoritative Korpusauswahl, Domänenautorisierung, Geschäftsregeln, Aufgabenbewertung und -akzeptanz","Low-Level-Provider-Integration, die die Plattform explizit besitzt",{"id":561,"data":562,"type":217},"p-control-plane-1",{"text":563},"Diese Trennung hilft, Plattform-Drift zu diagnostizieren. Wenn eine Anwendung jeden providerspezifischen Credential und Endpunkt kennen muss, ist der Plattformvertrag zu dünn. Wenn die Plattform entscheidet, welcher Kundendatensatz rechtlich autoritativ ist oder ob eine Domänenantwort akzeptabel ist, hat die Plattform die Grenze zur Solution-Ownership überschritten.",{"id":565,"data":566,"type":41},"h-artifacts",{"text":567,"level":241},"Was sollte ein AI Platform Architect produzieren?",{"id":569,"data":570,"type":290},"artifacts-table",{"content":571,"stretched":42,"withHeadings":13},[572,575,578,581,584,587,590,593,596,599],[573,574],"Architekturartefakt","Zweck",[576,577],"Plattform-Fähigkeitskarte","Definiert, was die Plattform bereitstellt, wer sie nutzt und welche Fähigkeiten außerhalb des Geltungsbereichs bleiben.",[579,580],"Provider-\u002FModellvertrag","Definiert Provider, Modelle, Fähigkeiten, Abstraktionsgrenzen, Routenmetadaten und Fallback-Semantik.",[582,583],"Identitäts- und Mandantenmodell","Definiert Benutzer-\u002FDienst-\u002FAnwendungsidentität, Mandantenkontext, RBAC\u002FABAC-Hooks und Ressourcenisolation.",[585,586],"Gateway- und Quotenrichtlinie","Definiert Ratenlimits, Token-\u002FKostenbudgets, Routing-Steuerung, Wiederholungen und Kapazitätsverhalten.",[588,589],"Retrieval-\u002FDatenvertrag","Definiert Ingestion, Provenienz, Suche, Metadaten, Autorisierungsweitergabe und wo Domänenautorität bleibt.",[591,592],"Agent-\u002FTool-Vertrag","Definiert Laufzeit-Lebenszyklus, Tool-Registrierung, Berechtigungen, Genehmigungen, Abbruch und Trace-Verhalten.",[594,595],"Secret- und Trust-Boundary-Modell","Definiert Credential-Eigentum, Speicherung, Prozessgrenzen, Rotation und Pfade sensibler Daten.",[597,598],"Evaluierungs- und Telemetrievertrag","Definiert gemeinsame Metriken, Traces, Datensatz-\u002FVersionslinks, Logging-Richtlinie und Solution-Erweiterungspunkte.",[600,601],"Lebenszyklus- und Kompatibilitätsrichtlinie","Definiert Versionen, Migrationen, Deprecation, Releases, Rollback, Incident-Ownership und Onboarding.",{"id":603,"data":604,"type":41},"h-tradeoffs",{"text":605,"level":241},"Die Arbeit besteht hauptsächlich aus Trade-offs, nicht aus maximaler Zentralisierung",{"id":607,"data":608,"type":298},"tradeoff-comparison",{"rows":609,"title":646,"layout":290,"columns":647},[610,616,622,628,634,640],{"id":611,"label":612,"values":613},"t1","Provider-Abstraktion",{"pressureA":614,"pressureB":615},"Stable portable platform API","Access to provider-specific capabilities and fast innovation",{"id":617,"label":618,"values":619},"t2","Wiederverwendung",{"pressureA":620,"pressureB":621},"Shared services reduce duplication","Isolation and domain autonomy prevent unsafe coupling",{"id":623,"label":624,"values":625},"t3","Governance",{"pressureA":626,"pressureB":627},"Central policy and auditability","Team speed and local experimentation",{"id":629,"label":630,"values":631},"t4","Observability",{"pressureA":632,"pressureB":633},"Rich traces for debugging and evaluation","Privacy, data minimization and logging cost",{"id":635,"label":636,"values":637},"t5","Verfügbarkeit",{"pressureA":638,"pressureB":639},"Fallback and multi-provider resilience","Predictable quality, compliance and data-location guarantees",{"id":641,"label":642,"values":643},"t6","Plattformumfang",{"pressureA":644,"pressureB":645},"More reusable capabilities","Smaller blast radius and less platform lock-in","Häufige Plattform-Trade-offs",[648,651],{"id":649,"label":650},"pressureA","Druck A",{"id":652,"label":653},"pressureB","Druck B",{"id":655,"data":656,"type":41},"h-adjacent",{"text":657,"level":241},"Wie unterscheidet sich das von angrenzenden Rollen?",{"id":659,"data":660,"type":290},"roles-table",{"content":661,"stretched":42,"withHeadings":13},[662,665,667,669,672,675,678],[663,664],"Rolle","Primärer Architekturumfang",[294,666],"Eine konkrete KI-fähige Lösung und ihre End-to-End-Anforderungen, Grenzen, Trade-offs und Produktionsakzeptanz.",[297,668],"Wiederverwendbare KI-Fähigkeiten und betriebliche\u002Fsicherheitstechnische Verträge, die von mehreren Lösungen oder Teams genutzt werden.",[670,671],"Enterprise Architect","Organisationsweites Geschäfts-\u002FTechnologieportfolio, Fähigkeits- und Governance-Ausrichtung auf breiterer Ebene.",[673,674],"MLOps \u002F LLMOps Architect oder Spezialist","Modell- und KI-Lebenszyklus, Bereitstellung, Experimente, Observability, Release- und Betriebspraktiken; kann stark überlappen, besitzt aber nicht automatisch die gesamte gemeinsame Anwendungsplattform.",[676,677],"Platform Engineer \u002F SRE","Implementiert und betreibt Plattforminfrastruktur, Zuverlässigkeit, Automatisierung und Entwicklererfahrung; Architekturverantwortung kann mit dem Plattformarchitekten geteilt werden.",[679,680],"AI \u002F Software Engineer","Implementiert Modelle, Integrationen, Dienste, Agenten, Retrieval und Produktfunktionalität innerhalb der vereinbarten Architektur.",{"id":682,"data":683,"type":217},"p-adjacent-1",{"text":684},"Diese Grenzen sind organisatorisch, nicht universell. In einem kleinen Team kann eine Person mehrere Verantwortlichkeiten tragen. In einem regulierten Unternehmen können sie auf Architektur-, Sicherheits-, Plattform-, Daten- und Betriebsgruppen aufgeteilt sein. Die nützliche Unterscheidung ist der \u003Cstrong>Umfang der Architekturverantwortung\u003C\u002Fstrong>, nicht der auf einem Organigramm gedruckte Jobtitel.",{"id":686,"data":687,"type":41},"h-evidence",{"text":688,"level":241},"Implementierungsnachweise: Wie diese Plattformgrenzen in meiner eigenen Arbeit erscheinen",{"id":690,"data":691,"type":224},"evidence-note",{"body":692,"title":693,"variant":230},"Die folgenden Abschnitte beschreiben konkrete Muster aus meinen eigenen Projekten. Sie sind Nachweise, dass diese Architekturgrenzen in echtem Code und Projektsystemen implementiert oder explizit entworfen wurden. Sie sind \u003Cstrong>keine\u003C\u002Fstrong> Behauptungen, dass die Projekte zusammen bereits eine kommerziell eingesetzte Unternehmens-KI-Plattform darstellen.","Originale Implementierungsnachweise",{"id":695,"data":696,"type":41},"h-ai-client",{"text":697,"level":240},"Aaasaasa AI Client: Trennung von Provider, Laufzeit und Berechtigungen",{"id":699,"data":700,"type":217},"p-ai-client-1",{"text":701},"Aaasaasa AI Client ist ein lokal-first Desktop-KI-Arbeitsbereich, der mit Nuxt 4, Electron und TypeScript erstellt wurde. Sein AI Hub trennt bewusst \u003Cstrong>Agent\u002FClient, Provider, Modell, Verbindungs-\u002FLaufzeitort, Berechtigungen und Web-Client\u003C\u002Fstrong>, anstatt sie als einen Konfigurationswert zu behandeln.",{"id":703,"data":704,"type":217},"p-ai-client-2",{"text":705},"Die Implementierung umfasst direkte Provider-Adapter, Codex-Agent-Laufzeitintegration, lokale Ollama\u002FLM Studio-Pfade, OpenAI-kompatible Dienste, zentralisierte Arbeitsbereichsberechtigungen, Credential-Speicherung im Hauptprozess, DuckDB, Qdrant\u002FVektor-Unterstützung, PDF-\u002FReadability-Extraktion und authentifizierten MCP-basierten Verzeichniszugriff.",{"id":707,"data":708,"type":217},"p-ai-client-3",{"text":709},"Zwei Plattformlektionen sind besonders relevant. Erstens ist eine lokale Laufzeit nicht dasselbe wie lokale Inferenz: Ein lokaler Codex-Prozess kann immer noch ein Cloud-Modell verwenden. Zweitens fällt automatisches Routing nicht stillschweigend von lokaler auf kostenpflichtige Cloud-Inferenz zurück. Das macht Routing-Richtlinie und Laufzeitlokalität explizit statt aus UI-Labels abgeleitet.",{"id":711,"data":712,"type":290},"ai-client-evidence-table",{"content":713,"stretched":42,"withHeadings":13},[714,717,720,723,726,729],[715,716],"Implementierte Grenze","Plattformarchitektonische Bedeutung",[718,719],"Agent vs. Provider vs. Modell","Unterschiedliche Verantwortlichkeiten können sich unabhängig entwickeln, anstatt hinter einem einzigen „KI“-Selektor verborgen zu werden.",[721,722],"Berechtigungen getrennt vom Modell","Dateisystem-\u002FTool-Autorität gehört zur Laufzeitrichtlinie, nicht zur Modellfähigkeit.",[724,725],"Secrets im Hauptprozess","Credential-Eigentum folgt der privilegierten Prozessgrenze statt dem Renderer\u002FUI.",[727,728],"Provider-Zustand und Modell-Erkennung","Routing und Verfügbarkeit sind Laufzeit-\u002FPlattformbelange.",[730,731],"Kein stiller Cloud-Fallback","Kosten-, Lokalitäts- und Datenübertragungssemantik bleiben explizite Richtlinienentscheidungen.",{"id":733,"data":734,"type":41},"h-cms",{"text":735,"level":240},"Aaasaasa AI CMS: mandantenbezogene Autorisierung als Plattformgrenze",{"id":737,"data":738,"type":217},"p-cms-1",{"text":739},"Die Codebasis des Aaasaasa AI CMS bietet ein separates Implementierungsbeispiel: mandantenbezogenes RBAC wird durch Rollen, Berechtigungen und Benutzer-Rollen-Zuweisungen dargestellt, die an eine Mandantenkennung gebunden sind. Systemberechtigungen sind nach Fähigkeiten gruppiert, und Rollensuche und -aktualisierungen bleiben mandantenbezogen.",{"id":741,"data":742,"type":217},"p-cms-2",{"text":743},"Dies ist für sich genommen kein Beweis für eine vollständige KI-Plattform, aber es ist direkt relevant für eine der schwierigsten Grenzen gemeinsamer Plattformen: Ein wiederverwendbarer Dienst muss bewahren, \u003Cstrong>wer was tun darf\u003C\u002Fstrong> und \u003Cstrong>für welchen Mandanten\u003C\u002Fstrong>. Das Hinzufügen von KI-Inferenz oder Retrieval auf einer Anwendungsplattform beseitigt diese Anforderung nicht.",{"id":745,"data":746,"type":217},"p-cms-3",{"text":747},"Die architektonische Implikation ist, dass Modell-Gateways, Retrieval-Dienste und Agenten etablierten Identitäts-\u002FMandantenkontext nutzen sollten, anstatt ein paralleles, nur auf KI ausgerichtetes Autorisierungsuniversum zu erfinden.",{"id":749,"data":750,"type":41},"h-sot",{"text":751,"level":240},"Source of Truth Research Engine: gemeinsame Retrieval-Mechanik ohne gemeinsame Wahrheit",{"id":753,"data":754,"type":217},"p-sot-1",{"text":755},"Die Source of Truth Research Engine bietet ein drittes Implementierungsbeispiel. Verschiedene Recherchemodelle teilen einen gemeinsamen Evidenzkern: Quellen, Artefakte, Provenienz, Claims, Relationen, Widersprüche, ein Referenzmodell und Audit-Trail. Das System bietet außerdem lokales lexikalisches Retrieval, optionales semantisches Retrieval, Extraktion, Snapshots und SHA-256-basierte Provenienz.",{"id":757,"data":758,"type":217},"p-sot-2",{"text":759},"Das Projekt behandelt Suche und semantische Ähnlichkeit ausdrücklich als Entdeckungssignale und nicht als Evidenz. Ein Ergebnis muss auf eine konkrete Quelle und einen Locator zurückgeführt werden können, bevor es einen Claim stützen kann. Genau das ist die Unterscheidung, die eine KI-Plattform braucht: \u003Cstrong>wiederverwendbare Retrieval-Maschinerie kann geteilt werden, während die Evidenzautorität weiterhin durch die konsumierende Methodik und Domäne geregelt wird.\u003C\u002Fstrong>",{"id":761,"data":762,"type":217},"p-sot-3",{"text":763},"Die Engine zeigt auch, warum eine gemeinsame Plattform keine gemeinsame Interpretation erfordert. Historische, wissenschaftlich-technische, Marktintelligenz- und Monitoring-Modi können die Kern-Evidenzinfrastruktur wiederverwenden und dennoch modusspezifische Methodik beibehalten.",{"id":765,"data":766,"type":224},"evidence-synthesis",{"body":767,"title":768,"variant":392},"Über diese Projekte hinweg ist das wiederverwendbare Muster nicht „ein Backend für alles“. Es ist \u003Cstrong>Trennung von Belangen plus explizite Verträge\u003C\u002Fstrong>: Trennung von Provider\u002FModell\u002FLaufzeit, mandantenbewusste Autorisierung, Credential-Grenzen, wiederverwendbare Daten-\u002FRetrieval-Primitive, Provenienz und domänenspezifische Autorität. Eine zukünftige integrierte Plattform bräuchte stabile Verträge zwischen diesen Fähigkeiten statt direkter Kopplung zwischen Codebasen.","Was diese Implementierungen zusammen zeigen",{"id":770,"data":771,"type":41},"h-frameworks",{"text":772,"level":241},"Wie aktuelle Architekturleitlinien diesen Plattformumfang stützen",{"id":774,"data":775,"type":217},"p-frameworks-1",{"text":776},"ISO\u002FIEC\u002FIEEE 42010:2022 bietet eine allgemeine Disziplin für Architekturbeschreibungen über Software, Systeme, Unternehmen und verwandte Entitäten hinweg. Es definiert keinen AI Platform Architect, aber es stärkt die Notwendigkeit, architektonische Belange, Beziehungen und Sichtweisen auszudrücken, anstatt Architektur auf eine Technologieliste zu reduzieren.",{"id":778,"data":779,"type":217},"p-frameworks-2",{"text":780},"NIST AI RMF 1.0 und das Generative AI Profile rahmen KI-Risikomanagement über den Lebenszyklus ein und nicht nur zum Zeitpunkt der Modellauswahl. Governance, Mapping, Messung und Management sind daher mit einer Plattformarchitektur vereinbar, die gemeinsame Kontrollen und Evidenz über viele konsumierende Workloads hinweg trägt.",{"id":782,"data":783,"type":217},"p-frameworks-3",{"text":784},"Die aktuelle AI-Workload-Leitlinie von Microsoft behandelt Anwendungsdesign, Daten, Sicherheit, Betrieb, Test\u002FEvaluierung und GenAIOps als verbundene Architekturbereiche. Die aktuelle AI-Gateway-Leitlinie zeigt zudem praktische Plattformbelange wie zentralisierten Modellzugriff, projektspezifische Token-Limits, Quoten und Multi-Team-Eingrenzung.",{"id":786,"data":787,"type":217},"p-frameworks-4",{"text":788},"Der aktuelle Generative AI Lens und das Multi-Tenant-Plattformszenario von AWS trennen ebenfalls grundlegende Plattformkontrollen von der Verantwortung konsumierender Anwendungen. AWS weist ausdrücklich darauf hin, dass eine zentrale Plattform gemeinsame Guardrails und Auditierbarkeit durchsetzen kann, während Datenqualität und workloadspezifische Observability weiterhin Verantwortlichkeiten konsumierender Anwendungen oder Datenproduzenten bleiben.",{"id":790,"data":791,"type":217},"p-frameworks-5",{"text":792},"Die Anbieterprodukte unterscheiden sich, aber das quellenübergreifende Muster ist stabil: Produktions-KI-Plattformen müssen Identität, Datenzugriff, Modelle, Richtlinien, Evaluierung, Observability, Kapazität, Kosten und Lebenszyklus koordinieren. Ein GPU-Cluster oder Modell-Endpunkt deckt nur einen Teil dieser Verantwortung ab.",{"id":794,"data":795,"type":41},"h-misconceptions",{"text":796,"level":241},"Häufige Missverständnisse",{"id":798,"data":799,"type":290},"misconceptions-table",{"content":800,"stretched":42,"withHeadings":13},[801,804,807,810,813,816,819,822,825],[802,803],"Missverständnis","Warum es falsch ist",[805,806],"„Eine KI-Plattform ist der GPU-Cluster.“","Compute ist ein Substrat. Eine Plattform braucht außerdem Verträge für Identität, Modellzugriff, Daten, Richtlinien, Evaluierung, Observability und Lebenszyklus.",[808,809],"„Ein KI-Gateway ist nur ein Reverse Proxy.“","Es kann auch Modell-Routing, Token-Quoten, Kostenattribution, Richtliniendurchsetzung, Identität und KI-spezifische Telemetrie tragen.",[811,812],"„Geteilt bedeutet global geteilt.“","Ein Dienst kann physisch geteilt sein und logisch nach Mandant, Anwendung, Region, Klassifizierung oder Risikostufe segmentiert werden.",[814,815],"„Eine zentrale Vektordatenbank wird die Unternehmenswahrheit.“","Ein Vektorspeicher oder Retrieval-Dienst ist Infrastruktur. Domänenautorität, Aktualität, Provenienz und Zugriff bleiben separate Belange.",[817,818],"„Plattform-Evaluierung ersetzt Lösungsevaluierung.“","Allgemeine Regression und Telemetrie können nicht definieren, ob eine domänenspezifische Antwort oder Aktion akzeptabel ist.",[820,821],"„Provider-Abstraktion sollte jeden Unterschied verbergen.“","Einige Unterschiede sind wesentliche Fähigkeiten, Sicherheitssemantiken oder Fehlermodi und müssen sichtbar bleiben.",[823,824],"„RBAC löst Multi-Tenancy.“","RBAC steuert Aktionen; Mandantenisolierung steuert Ressourcengrenzen. Beides kann erforderlich sein.",[826,827],"„AI Platform Architect ist nur ein anderer Name für MLOps.“","MLOps\u002FLLMOps ist eine wichtige überlappende Disziplin, aber gemeinsame Anwendungs-\u002FLaufzeit-, Identitäts-, Gateway-, Retrieval- und Tool-Grenzen können über Modelllebenszyklus-Operationen hinausgehen.",{"id":829,"data":830,"type":41},"h-failure",{"text":831,"level":241},"Fehlermodi, die ein AI Platform Architect verhindern sollte",{"id":833,"data":834,"type":290},"failures-table",{"content":835,"stretched":42,"withHeadings":13},[836,839,842,845,848,851,854,857,860,863],[837,838],"Fehlermodus","Architektonische Konsequenz",[840,841],"Jedes Team speichert seine eigenen Provider-Schlüssel","Doppelte Handhabung von Geheimnissen, inkonsistente Rotation und größerer Blast-Radius.",[843,844],"Provider-Abstraktion verbirgt erforderliche Fähigkeiten","Konsumenten können benötigte Funktionen nicht nutzen oder erhalten stillschweigend ein Verhalten, das von den Annahmen abweicht.",[846,847],"Gemeinsame Retrieval ignoriert Mandanten-\u002FBenutzerkontext","Grenzüberschreitende Datenlecks können auftreten, bevor die Anwendung die Möglichkeit hat, Ergebnisse zu filtern.",[849,850],"Fallback ändert stillschweigend Provider oder Lokalität","Kosten, Compliance, Datenstandort und Ausgabequalität können sich ändern, ohne dass der Aufrufer davon weiß.",[852,853],"Agent-Tools werden durch Modellwahl gewährt","Ein leistungsfähiges Modell wird überprivilegiert, weil die Laufzeitberechtigung nicht unabhängig durchgesetzt wird.",[855,856],"Alle Prompts\u002FAntworten werden standardmäßig protokolliert","Observability kann ein neues sensibles Datenrepository und Compliance-Problem schaffen.",[858,859],"Plattform besitzt einen generischen Qualitätswert","Domänenfehler bleiben hinter Plattform-Gesundheitsmetriken verborgen.",[861,862],"Kein Versionsvertrag für Plattformfähigkeiten","Modell-\u002FProvider-\u002FLaufzeitänderungen brechen Konsumenten unvorhersehbar.",[864,865],"Alles KI-bezogene ist zentralisiert","Die Plattform wird zum Engpass und Monolithen statt zu einer wiederverwendbaren Fähigkeitsschicht.",{"id":867,"data":868,"type":41},"h-decision",{"text":869,"level":241},"Eine praktische Entscheidungssequenz für die Plattformarchitektur",{"id":871,"data":872,"type":339},"decision-flow",{"steps":873,"title":904,"orientation":338},[874,877,880,883,886,889,892,895,898,901],{"label":875,"description":876},"1. Echte Konsumenten identifizieren","Listen Sie Lösungen, Teams, Mandanten und Workloads auf, die die Plattform nutzen würden; vermeiden Sie den Aufbau einer Plattform für hypothetische Wiederverwendung.",{"label":878,"description":879},"2. Die gemeinsame Grenze definieren","Trennen Sie übergreifende Mechanismen von lösungsspezifischer Domänenautorität, Workflow und Abnahme.",{"label":881,"description":882},"3. Zuerst Identität und Isolation definieren","Etablieren Sie Benutzer, Dienste, Anwendungen, Mandanten, Regionen und Datenklassifizierungen, bevor Sie Retrieval- oder Tool-Fähigkeiten teilen.",{"label":884,"description":885},"4. Fähigkeitsverträge definieren","Spezifizieren Sie Modell-\u002FProvider-, Retrieval-, Agent-\u002FTool-, Gateway- und Telemetrie-APIs mit expliziter Eigentümerschaft und Versionierung.",{"label":887,"description":888},"5. Provider- und Laufzeitstrategie entscheiden","Wählen Sie verwaltete, selbst gehostete, lokale oder hybride Ausführung und dokumentieren Sie Fallback-, Lokalitäts- und Fähigkeitssemantik.",{"label":890,"description":891},"6. Daten- und Retrieval-Grenzen entwerfen","Definieren Sie Herkunft, Autorisierungsweitergabe, Korpus-Eigentümerschaft, Indexierung und Nachweispflichten.",{"label":893,"description":894},"7. Quoten, Geheimnisse und Richtlinien hinzufügen","Steuern Sie Kosten, Kapazität, Anmeldeinformationen, Tool-Berechtigungen, Sicherheitskontrollen und Blast-Radius.",{"label":896,"description":897},"8. Evaluierungs- und Observability-Verträge aufbauen","Bieten Sie Plattformmetriken und Tracing, während Sie Domänen-Ground-Truth und Abnahme der Lösung überlassen.",{"label":899,"description":900},"9. Lebenszyklus und Betrieb definieren","Versionieren Sie Fähigkeiten, testen Sie Upgrades, dokumentieren Sie Deprecation, Rollback, Vorfälle, Kapazität und Konsumenten-Onboarding.",{"label":902,"description":903},"10. Mit mehr als einem Konsumenten validieren","Ein Plattformanspruch wird glaubwürdig, wenn die gemeinsame Fähigkeit tatsächlich unterschiedliche Workloads bedient, ohne sie in dasselbe Domänenmodell zu zwingen.","Vom Plattformbedarf zur betreibbaren gemeinsamen Fähigkeit",{"id":906,"data":907,"type":41},"h-edge",{"text":908,"level":241},"Randfälle und Grenzen der Rolle",{"id":910,"data":911,"type":217},"p-edge-1",{"text":912},"Eine kleine Organisation mit einer einzigen KI-Anwendung benötigt möglicherweise keine eigenständige KI-Plattform oder keinen Plattformarchitekten. Verfrühte Plattformbildung kann mehr Abstraktion als Wert schaffen. Die richtige Architektur kann eine gut entworfene Lösung mit einigen wiederverwendbaren Modulen sein.",{"id":914,"data":915,"type":217},"p-edge-2",{"text":916},"Eine luftgespaltene oder souveräne Bereitstellung ändert das Provider-, Update- und Observability-Modell erheblich. Modell-Hosting, Artefaktverteilung, Identitätsintegration und Telemetrie-Export benötigen möglicherweise alle lokale Äquivalente.",{"id":918,"data":919,"type":217},"p-edge-3",{"text":920},"Hochregulierte oder folgenschwere Workloads können eine stärkere physische oder organisatorische Isolation erfordern, anstatt einer logisch gemeinsamen Plattform. Wiederverwendung ist niemals ein ausreichender Grund, eine erforderliche Sicherheitsgrenze zu schwächen.",{"id":922,"data":923,"type":217},"p-edge-4",{"text":924},"Verwaltete Cloud-KI-Dienste können die Implementierungslast entfernen, aber nicht die architektonische Verantwortung. Die Organisation entscheidet weiterhin über Identität, Datenzugriff, Protokollierung, Aufbewahrung, Quoten, Modellberechtigung, Fallback, Evaluierung und Lösungsabnahme.",{"id":926,"data":927,"type":217},"p-edge-5",{"text":928},"Die Plattformgrenze kann sich auch je nach Modalität unterscheiden. Textinferenz, multimodale Generierung, Sprache, Computernutzung und autonome Agenten können unterschiedliche Latenz-, Daten-, Berechtigungs- und Observability-Anforderungen haben, selbst wenn sie Provider- und Identitätsinfrastruktur teilen.",{"id":930,"data":931,"type":41},"h-change",{"text":932,"level":241},"Was würde diese Antwort ändern?",{"id":934,"data":935,"type":217},"p-change-1",{"text":936},"Die Kerndefinition würde sich ändern, wenn sich der organisatorische Umfang ändert. Wenn der Architekt einen Workload besitzt, nähert sich die Rolle einem KI-Lösungsarchitekten. Wenn sich die Verantwortung auf organisationsweite Fähigkeitsstrategie, Investitionen, Standards und Zielzustandsportfolios ausdehnt, bewegt sie sich in Richtung Enterprise-KI-Architektur.",{"id":938,"data":939,"type":217},"p-change-2",{"text":940},"Die Implementierungsanleitung ändert sich, wann immer sich Provider, Gateway-Produkte, Agent-Protokolle, regulatorische Verpflichtungen, Modellfähigkeiten oder Bereitstellungsbeschränkungen ändern. Deshalb sollte die Plattformarchitektur stabile Verantwortlichkeiten und Verträge getrennt von aktuellen Anbietermechanismen ausdrücken.",{"id":942,"data":943,"type":41},"h-checklist",{"text":944,"level":241},"Checkliste für KI-Plattformarchitekten",{"id":946,"data":947,"type":290},"checklist-table",{"content":948,"stretched":42,"withHeadings":13},[949,952,955,958,961,964,967,970,973,976,979,982,985],[950,951],"Frage","Erwartete Antwort",[953,954],"Wer sind die tatsächlichen Plattformkonsumenten?","Benannte Lösungen, Teams oder Mandantenkontexte mit unterschiedlichen, aber überlappenden Bedürfnissen.",[956,957],"Was ist wirklich gemeinsam?","Explizite Fähigkeitsliste, kein vages „KI-Backend“.",[959,960],"Was muss lösungsspezifisch bleiben?","Domänenautorität, Geschäftsworkflow, Aufgabenabnahme und andere vom Workload bestimmte Belange.",[962,963],"Wie werden Modelle\u002FProvider dargestellt?","Versionierte Provider-\u002FModellverträge mit Fähigkeiten und expliziter Fallback-Semantik.",[965,966],"Wie wird Identität weitergegeben?","Benutzer-\u002FDienst-\u002FAnwendungs-\u002FMandantenkontext überlebt jeden privilegierten Anforderungspfad.",[968,969],"Wie wird Mandantenisolation durchgesetzt?","Ressourcen-Scoping ist getrennt von Rollenberechtigungsprüfungen.",[971,972],"Wie werden Geheimnisse behandelt?","Privilegierter Speicher, Rotation, begrenzte Exposition und auditierbare Eigentümerschaft.",[974,975],"Wie bewahrt Retrieval die Autorität?","Gemeinsame Mechanismen mit Autorisierung, Herkunft und domäneneigenen Nachweisregeln.",[977,978],"Wie werden Tools und Agenten eingeschränkt?","Laufzeitberechtigungen, begrenzte Tool-Verträge, Genehmigungen, Abbruch und Rückverfolgbarkeit.",[980,981],"Wie werden Kosten und Kapazität kontrolliert?","Quoten, Token-\u002FRatenkontrollen, Nutzungszuordnung und Überlastverhalten.",[983,984],"Wie wird Qualität gemessen?","Plattform-Regression\u002FEvaluierung plus lösungsspezifische Ground-Truth und Abnahme.",[986,987],"Wie werden Änderungen ausgerollt?","Versionierung, Kompatibilität, Migration, Deprecation, Rollback und Vorfallverantwortung.",{"id":989,"data":990,"type":41},"h-conclusion",{"text":991,"level":241},"Fazit",{"id":993,"data":994,"type":217},"p-conclusion-1",{"text":995},"Ein KI-Plattformarchitekt ist verantwortlich für die wiederverwendbare Architektur \u003Cstrong>zwischen KI-Fähigkeiten und den Lösungen, die sie nutzen\u003C\u002Fstrong>. Die Rolle definiert, wie Modelle, Provider, Retrieval, Agenten, Tools, Identität, Mandanten, Geheimnisse, Evaluierung, Observability, Quoten und Laufzeitoperationen zu verlässlichen Plattformdiensten werden, anstatt wiederholte Einmalintegrationen zu sein.",{"id":997,"data":998,"type":217},"p-conclusion-2",{"text":999},"Der schwierige Teil ist nicht die Maximierung der Wiederverwendung. Es ist die Wahl der richtigen Grenze. Eine starke Plattform standardisiert Mechanismen, Richtlinien und Betrieb dort, wo mehrere Konsumenten wirklich profitieren, während sie lösungsspezifische Datenautorität, Geschäftslogik, Sicherheitsanforderungen und Abnahmekriterien bewahrt.",{"id":1001,"data":1002,"type":217},"p-conclusion-3",{"text":1003},"Diese Unterscheidung erklärt auch die Beziehung zur KI-Lösungsarchitektur: \u003Cstrong>Der Lösungsarchitekt macht ein KI-fähiges System zweckmäßig; der Plattformarchitekt macht gemeinsame KI-Fähigkeiten sicher, wiederverwendbar, betreibbar und über viele solcher Systeme hinweg entwickelbar.\u003C\u002Fstrong>",{"id":1005,"data":1006,"type":41},"h-related",{"text":1007,"level":241},"Verwandtes kanonisches Wissen",{"id":1009,"data":1010,"type":217},"p-related-1",{"text":1011},"Dieser Artikel baut auf den kanonischen Grundlagen zu generativen KI-Komponenten, ADR versus NFR und KI-Lösungsarchitektur auf. Diese Konzepte sind Voraussetzungen, weil eine Plattform existiert, um wiederverwendbare Systemfähigkeiten bereitzustellen und architektonische Entscheidungen gegen explizite Qualitäts- und Betriebsanforderungen zu kodifizieren.",{"id":1013,"data":1014,"type":217},"p-related-2",{"text":1015},"Retrieval-Augmented Generation ist ein Beispiel für eine Fähigkeit, die über eine Plattform angeboten werden kann, aber die Plattform sollte Retrieval-Infrastruktur, Domänenwissen und Antwortvalidität nicht zu einem einzigen Konzept verschmelzen.",{"id":1017,"data":1018,"type":1025},"related-rag",{"link":1019,"meta":1020},"https:\u002F\u002Fstajic.de\u002Fde\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works",{"image":1021,"title":1023,"description":1024},{"url":1022},"","Was ist RAG? Die einfachste Erklärung, wie es funktioniert","Kanonische Einführung in Retrieval-Augmented Generation und die Grenze zwischen Modellgenerierung und externem Wissensabruf.","linkTool",{"id":1027,"data":1028,"type":217},"p-related-3",{"text":1029},"Agentenprotokolle, Mandantentrennung, KI-Governance, Modell-Routing, Context Engineering und MLOps\u002FLLMOps sind nachgelagerte oder angrenzende Wissensknoten. Sie lassen sich leichter durchdenken, sobald die Plattformgrenze explizit ist.",{"id":1031,"data":1032,"type":41},"h-faq",{"text":1033,"level":241},"Häufig gestellte Fragen",{"id":1035,"data":1036,"type":1035},"faq",{"items":1037,"title":1062},[1038,1042,1046,1050,1054,1058],{"id":1039,"answer":1040,"question":1041},"faq-1","Nein. Der Lösungsarchitekt konzentriert sich auf eine konkrete KI-fähige Lösung. Der Plattformarchitekt konzentriert sich auf wiederverwendbare KI-Fähigkeiten, Kontrollen und Betriebsverträge, die mehrere Lösungen unterstützen können.","Ist ein KI-Plattform-Architekt dasselbe wie ein KI-Lösungsarchitekt?",{"id":1043,"answer":1044,"question":1045},"faq-2","Nein. Eine Plattform kann verwaltete Cloud-Modelle, selbst gehostete Modelle, lokale Inferenz oder eine Hybridstrategie nutzen. Die Architektur muss Anbieter-, Lokalitäts-, Identitäts-, Routing-, Daten- und Betriebskonsequenzen explizit machen.","Muss eine KI-Plattform eigene Modelle hosten?",{"id":1047,"answer":1048,"question":1049},"faq-3","Normalerweise nicht. Ein Gateway kann eine wichtige Plattformkomponente sein, aber eine vollständige Plattform benötigt auch Verträge für Identität, Secrets, Daten\u002FRetrieval, Evaluierung, Observability, Lebenszyklus und Betriebsverantwortung.","Reicht ein KI-Gateway aus, um eine KI-Plattform zu sein?",{"id":1051,"answer":1052,"question":1053},"faq-4","Retrieval-Mechaniken können oft geteilt werden, aber Domänenautorität, Autorisierung, Aktualität, Evidenzausreichendheit und Korpus-Eigentümerschaft sollten explizit bleiben. Geteilte Infrastruktur impliziert keine geteilte Wahrheit.","Sollte Retrieval zentralisiert werden?",{"id":1055,"answer":1056,"question":1057},"faq-5","Nein. Die Plattform-Evaluierung kann geteilte Fähigkeiten und Regressionen testen. Jede Lösung benötigt weiterhin aufgabenspezifische Ground Truth, Akzeptanzkriterien und Domänen-Qualitätsschwellen.","Ersetzt die Plattform-Evaluierung die Anwendungs-Evaluierung?",{"id":1059,"answer":1060,"question":1061},"faq-6","Nein. RBAC bestimmt, was eine Identität tun darf. Mandantentrennung bestimmt, auf welche Ressourcen eines Mandanten die Identität einwirken darf. Eine Plattform benötigt oft beides.","Ist Multi-Tenancy nur RBAC?","FAQ zum KI-Plattform-Architekten",{"id":1064,"data":1065,"type":41},"h-glossary",{"text":1066,"level":241},"Glossar",{"id":1068,"data":1069,"type":1068},"glossary",{"title":1070,"entries":1071},"Wichtige Begriffe der KI-Plattformarchitektur",[1072,1076,1080,1084,1088,1092,1096,1100],{"term":1073,"anchor":1074,"definition":1075},"KI-Plattform","ai-platform","Eine wiederverwendbare Menge KI-bezogener technischer und betrieblicher Fähigkeiten, die von mehreren Anwendungen, Teams oder Mandantenkontexten genutzt werden.",{"term":1077,"anchor":1078,"definition":1079},"KI-Gateway","ai-gateway","Eine Gateway-Schicht für KI-Endpunkte, die über einfaches Proxying hinaus Authentifizierung, Routing, Quoten, Richtlinien, Wiederholungen, Kostenzuordnung und KI-spezifische Telemetrie hinzufügen kann.",{"term":1081,"anchor":1082,"definition":1083},"Anbieter-Adapter","provider-adapter","Eine Komponente, die einen Plattformvertrag auf die API, Fähigkeiten, Gesundheit und Fehlersemantik eines Modellanbieters abbildet.",{"term":1085,"anchor":1086,"definition":1087},"Mandantentrennung","tenant-isolation","Die Grenze, die verhindert, dass ein Mandantenkontext auf die Ressourcen eines anderen Mandanten zugreift, unabhängig von Rollenberechtigungen.",{"term":1089,"anchor":1090,"definition":1091},"Fähigkeitsvertrag","capability-contract","Eine versionierte Schnittstelle und Verhaltensvereinbarung, die beschreibt, was ein geteilter Plattformdienst bereitstellt und was der Konsument liefern oder besitzen muss.",{"term":1093,"anchor":1094,"definition":1095},"Grounding-\u002FRetrieval-Dienst","grounding-service","Geteilte Mechanismen zum Finden und Bereitstellen externer Informationen für eine KI-Workload; er definiert nicht automatisch, welche Informationen für eine Domäne maßgeblich sind.",{"term":1097,"anchor":1098,"definition":1099},"Evaluierungs-Harness","evaluation-harness","Wiederverwendbare Infrastruktur zum Ausführen von Tests, Datensätzen, Modell-\u002FPrompt-Versionen und Metriken; die Domänenakzeptanz bleibt lösungsspezifisch.",{"term":1101,"anchor":1102,"definition":1103},"Control Plane","control-plane","Die Konfigurations- und Governance-Schicht, die Plattformfähigkeiten, Identitäten, Richtlinien, Quoten, Versionen und Bereitstellungszustand verwaltet.",{"id":1105,"data":1106,"type":41},"h-sources",{"text":1107,"level":241},"Primärquellen und aktuelle Architekturempfehlungen",{"id":1109,"data":1110,"type":217},"p-sources-note",{"text":1111},"Die folgenden Quellen stützen die allgemeinen Architektur- und Produktionsplattform-Aussagen. Die Abschnitte Aaasaasa AI Client, Aaasaasa AI CMS und Source of Truth Research Engine sind ausdrücklich originäre Implementierungsnachweise. Aktuelle externe Referenzen wurden am 8. Oktober 2026 geprüft.",{"id":1113,"data":1114,"type":1025},"src-iso-42010",{"link":1115,"meta":1116},"https:\u002F\u002Fwww.iso.org\u002Fstandard\u002F74393.html",{"image":1117,"title":1118,"description":1119},{"url":1022},"ISO\u002FIEC\u002FIEEE 42010:2022 — Architekturbeschreibung","Aktuell veröffentlichter internationaler Standard für Konzepte und Beziehungen der Architekturbeschreibung.",{"id":1121,"data":1122,"type":1025},"src-nist-rmf",{"link":1123,"meta":1124},"https:\u002F\u002Fwww.nist.gov\u002Fitl\u002Fai-risk-management-framework",{"image":1125,"title":1126,"description":1127},{"url":1022},"NIST AI Risk Management Framework","NISTs AI RMF-Ressourcen und aktueller Status; AI RMF 1.0 befindet sich Stand Oktober 2026 in Überarbeitung.",{"id":1129,"data":1130,"type":1025},"src-nist-gai",{"link":1131,"meta":1132},"https:\u002F\u002Fwww.nist.gov\u002Fpublications\u002Fartificial-intelligence-risk-management-framework-generative-artificial-intelligence",{"image":1133,"title":1134,"description":1135},{"url":1022},"NIST AI 600-1 — Generative AI Profile","Generative-AI-Profil zur Anwendung von KI-Risikomanagement-Überlegungen über den KI-Lebenszyklus.",{"id":1137,"data":1138,"type":1025},"src-ms-ai",{"link":1139,"meta":1140},"https:\u002F\u002Flearn.microsoft.com\u002Fen-us\u002Fazure\u002Fwell-architected\u002Fai\u002Fget-started",{"image":1141,"title":1142,"description":1143},{"url":1022},"Microsoft Azure Well-Architected — AI Workloads","Aktuelle Architekturempfehlungen zu KI-Anwendung, Daten, Betrieb, Evaluierung, verantwortungsvoller KI und Lebenszyklusaspekten.",{"id":1145,"data":1146,"type":1025},"src-ms-principles",{"link":1147,"meta":1148},"https:\u002F\u002Flearn.microsoft.com\u002Fen-us\u002Fazure\u002Fwell-architected\u002Fai\u002Fdesign-principles",{"image":1149,"title":1150,"description":1151},{"url":1022},"Microsoft — Design Principles for AI Workloads","Aktuelle Empfehlungen zu Identitätssegmentierung, Sicherheitsgrenzen, Telemetrie, Leistung, Daten und Plattform-Abwägungen.",{"id":1153,"data":1154,"type":1025},"src-ms-gateway",{"link":1155,"meta":1156},"https:\u002F\u002Flearn.microsoft.com\u002Fen-us\u002Fazure\u002Fai-foundry\u002Fconfiguration\u002Fenable-ai-api-management-gateway-portal?view=foundry",{"image":1157,"title":1158,"description":1159},{"url":1022},"Microsoft Foundry — AI Gateway Architecture","Aktuelle AI-Gateway-Empfehlungen für gemeinsamen Projektzugriff, Token-Begrenzung, Quoten und Governance.",{"id":1161,"data":1162,"type":1025},"src-ms-gateway-guide",{"link":1163,"meta":1164},"https:\u002F\u002Flearn.microsoft.com\u002Fen-us\u002Fazure\u002Farchitecture\u002Fai-ml\u002Fguide\u002Fazure-openai-gateway-guide",{"image":1165,"title":1166,"description":1167},{"url":1022},"Azure Architecture Center — Access Models Through a Gateway","Architekturempfehlungen für zentralisierten Modellzugriff, Routing, Drosselung, Failover und Verantwortlichkeiten von Client und Plattform.",{"id":1169,"data":1170,"type":1025},"src-aws-genai",{"link":1171,"meta":1172},"https:\u002F\u002Fdocs.aws.amazon.com\u002Fwellarchitected\u002Flatest\u002Fgenerative-ai-lens\u002F",{"image":1173,"title":1174,"description":1175},{"url":1022},"AWS Well-Architected — Generative AI Lens","Aktuelle Architekturrichtlinien für generative KI-Workloads in den Bereichen Sicherheit, Zuverlässigkeit, Betrieb, Leistung und Kosten.",{"id":1177,"data":1178,"type":1025},"src-aws-multitenant",{"link":1179,"meta":1180},"https:\u002F\u002Fdocs.aws.amazon.com\u002Fwellarchitected\u002Flatest\u002Fgenerative-ai-lens\u002Fmulti-tenant-generative-ai-platform-scenario.html",{"image":1181,"title":1182,"description":1183},{"url":1022},"AWS — Multi-Tenant-Szenario für generative KI-Plattformen","Aktuelles Beispiel, das zentrale Plattformkontrollen und Auditierbarkeit von der Datenqualität der nutzenden Anwendungen und workloadspezifischen Verantwortlichkeiten trennt.",{"id":1185,"data":1186,"type":1025},"src-aws-agentic",{"link":1187,"meta":1188},"https:\u002F\u002Fdocs.aws.amazon.com\u002Fwellarchitected\u002Flatest\u002Fagentic-ai-lens\u002Fdesign-principles.html",{"image":1189,"title":1190,"description":1191},{"url":1022},"AWS Well-Architected — Designprinzipien für agentische KI","Aktuelle Richtlinien zu begrenzter Agentenautorität, Nachverfolgbarkeit, versioniertem Verhalten, expliziten Verträgen und menschlicher Aufsicht.",{"id":1193,"data":1194,"type":1025},"src-aws-observability",{"link":1195,"meta":1196},"https:\u002F\u002Fdocs.aws.amazon.com\u002FAmazonCloudWatch\u002Flatest\u002Fmonitoring\u002FGenAI-observability.html",{"image":1197,"title":1198,"description":1199},{"url":1022},"AWS CloudWatch — Observability für generative KI","Aktuelle Observability-Funktionen und Produktionsmetriken für Modelle, Agenten, Wissensdatenbanken, Tools sowie Kosten-, Latenz- und Fehleranalysen.","2.31","Ein KI-Plattform-Architekt entwirft wiederverwendbare KI-Grundlagen über Modelle, Anbieter, Retrieval, Agenten, Identität, Sicherheit, Evaluierung, Observability und Betrieb hinweg.","\u002Fuploads\u002F2026\u002F10\u002Fwhat-is-an-ai-platform-architect-models-data-runtime-security-and-operations-1791477229171-ou3zcc.webp","what-is-an-ai-platform-architect-models-data-runtime-security-and-operations-1791477229171-ou3zcc","PUBLISHED","2026-10-08T12:32:00.000Z","2026-10-08T16:32:14.856Z","2026-10-08T16:47:57.364Z",{"en":1209,"de":1210,"sr":1211,"es":1212,"fr":1213,"it":1214,"ru":1215,"zh":1216},"\u002Fblog\u002Fwhat-is-an-ai-platform-architect-models-data-runtime-security-and-operations","\u002Fde\u002Fblog\u002Fwhat-is-an-ai-platform-architect-models-data-runtime-security-and-operations","\u002Fsr\u002Fblog\u002Fwhat-is-an-ai-platform-architect-models-data-runtime-security-and-operations","\u002Fes\u002Fblog\u002Fwhat-is-an-ai-platform-architect-models-data-runtime-security-and-operations","\u002Ffr\u002Fblog\u002Fwhat-is-an-ai-platform-architect-models-data-runtime-security-and-operations","\u002Fit\u002Fblog\u002Fwhat-is-an-ai-platform-architect-models-data-runtime-security-and-operations","\u002Fru\u002Fblog\u002Fwhat-is-an-ai-platform-architect-models-data-runtime-security-and-operations","\u002Fzh\u002Fblog\u002Fwhat-is-an-ai-platform-architect-models-data-runtime-security-and-operations",[1218,1222,1226],{"id":1219,"name":1220,"slug":1221},84,"Policy & Datengrenzen","policy-and-data",{"id":1223,"name":1224,"slug":1225},57,"Daten-Grenzen","data-boundaries",{"id":1227,"name":1228,"slug":1229},80,"Zugriff & Identität","access-and-identity",{"id":1231,"login":1232,"email":1233,"displayName":1234},"20","rooth8233","aleksandar@stajic.de","Aleksandar Stajić",[1236,1661],{"lang":7,"title":207,"content":209,"contentJson":1237,"excerpt":1201},{"time":211,"blocks":1238,"version":1200},[1239,1241,1243,1245,1247,1249,1251,1253,1255,1271,1273,1275,1277,1279,1288,1290,1292,1294,1296,1306,1308,1310,1312,1314,1316,1318,1320,1322,1324,1326,1328,1330,1332,1334,1336,1338,1340,1342,1344,1346,1348,1350,1352,1354,1356,1358,1360,1362,1364,1366,1368,1370,1372,1374,1376,1378,1380,1387,1389,1391,1404,1406,1424,1426,1436,1438,1440,1442,1444,1446,1448,1450,1459,1461,1463,1465,1467,1469,1471,1473,1475,1477,1479,1481,1483,1485,1487,1489,1491,1503,1505,1518,1520,1533,1535,1537,1539,1541,1543,1545,1547,1549,1551,1553,1569,1571,1573,1575,1577,1579,1581,1583,1587,1589,1591,1600,1602,1613,1615,1617,1621,1625,1629,1633,1637,1641,1645,1649,1653,1657],{"id":214,"data":1240,"type":217},{"text":216},{"id":219,"data":1242,"type":224},{"body":221,"title":222,"variant":223},{"id":226,"data":1244,"type":224},{"body":228,"title":229,"variant":230},{"id":232,"data":1246,"type":224},{"body":234,"title":235,"variant":230},{"id":237,"data":1248,"type":242},{"title":239,"maxLevel":240,"minLevel":241},{"id":244,"data":1250,"type":41},{"text":246,"level":241},{"id":248,"data":1252,"type":217},{"text":250},{"id":252,"data":1254,"type":217},{"text":254},{"id":256,"data":1256,"type":298},{"rows":1257,"title":289,"layout":290,"columns":1268},[1258,1260,1262,1264,1266],{"id":260,"label":261,"values":1259},{"platform":263,"solution":264},{"id":266,"label":267,"values":1261},{"platform":269,"solution":270},{"id":272,"label":273,"values":1263},{"platform":275,"solution":276},{"id":278,"label":279,"values":1265},{"platform":281,"solution":282},{"id":284,"label":285,"values":1267},{"platform":287,"solution":288},[1269,1270],{"id":293,"label":294},{"id":296,"label":297},{"id":300,"data":1272,"type":41},{"text":302,"level":241},{"id":304,"data":1274,"type":217},{"text":306},{"id":308,"data":1276,"type":217},{"text":310},{"id":312,"data":1278,"type":217},{"text":314},{"id":316,"data":1280,"type":339},{"steps":1281,"title":337,"orientation":338},[1282,1283,1284,1285,1286,1287],{"label":320,"description":321},{"label":323,"description":324},{"label":326,"description":327},{"label":329,"description":330},{"label":332,"description":333},{"label":335,"description":336},{"id":341,"data":1289,"type":41},{"text":343,"level":241},{"id":345,"data":1291,"type":217},{"text":347},{"id":349,"data":1293,"type":217},{"text":351},{"id":353,"data":1295,"type":41},{"text":355,"level":241},{"id":357,"data":1297,"type":290},{"content":1298,"stretched":42,"withHeadings":13},[1299,1300,1301,1302,1303,1304,1305],[361,362,363],[365,366,367],[369,370,371],[373,374,375],[377,378,379],[279,381,382],[384,385,386],{"id":388,"data":1307,"type":224},{"body":390,"title":391,"variant":392},{"id":394,"data":1309,"type":41},{"text":396,"level":241},{"id":398,"data":1311,"type":41},{"text":400,"level":240},{"id":402,"data":1313,"type":217},{"text":404},{"id":406,"data":1315,"type":217},{"text":408},{"id":410,"data":1317,"type":224},{"body":412,"title":413,"variant":414},{"id":416,"data":1319,"type":41},{"text":418,"level":240},{"id":420,"data":1321,"type":217},{"text":422},{"id":424,"data":1323,"type":217},{"text":426},{"id":428,"data":1325,"type":217},{"text":430},{"id":432,"data":1327,"type":41},{"text":434,"level":240},{"id":436,"data":1329,"type":217},{"text":438},{"id":440,"data":1331,"type":217},{"text":442},{"id":444,"data":1333,"type":217},{"text":446},{"id":448,"data":1335,"type":41},{"text":450,"level":240},{"id":452,"data":1337,"type":217},{"text":454},{"id":456,"data":1339,"type":217},{"text":458},{"id":460,"data":1341,"type":217},{"text":462},{"id":464,"data":1343,"type":41},{"text":466,"level":240},{"id":468,"data":1345,"type":217},{"text":470},{"id":472,"data":1347,"type":217},{"text":474},{"id":476,"data":1349,"type":217},{"text":478},{"id":480,"data":1351,"type":41},{"text":482,"level":240},{"id":484,"data":1353,"type":217},{"text":486},{"id":488,"data":1355,"type":217},{"text":490},{"id":492,"data":1357,"type":41},{"text":494,"level":240},{"id":496,"data":1359,"type":217},{"text":498},{"id":500,"data":1361,"type":217},{"text":502},{"id":504,"data":1363,"type":217},{"text":506},{"id":508,"data":1365,"type":41},{"text":510,"level":240},{"id":512,"data":1367,"type":217},{"text":514},{"id":516,"data":1369,"type":217},{"text":518},{"id":520,"data":1371,"type":41},{"text":522,"level":240},{"id":524,"data":1373,"type":217},{"text":526},{"id":528,"data":1375,"type":217},{"text":530},{"id":532,"data":1377,"type":41},{"text":534,"level":241},{"id":536,"data":1379,"type":224},{"body":538,"title":539,"variant":230},{"id":541,"data":1381,"type":290},{"content":1382,"stretched":42,"withHeadings":13},[1383,1384,1385,1386],[545,546,547],[549,550,551],[553,554,555],[557,558,559],{"id":561,"data":1388,"type":217},{"text":563},{"id":565,"data":1390,"type":41},{"text":567,"level":241},{"id":569,"data":1392,"type":290},{"content":1393,"stretched":42,"withHeadings":13},[1394,1395,1396,1397,1398,1399,1400,1401,1402,1403],[573,574],[576,577],[579,580],[582,583],[585,586],[588,589],[591,592],[594,595],[597,598],[600,601],{"id":603,"data":1405,"type":41},{"text":605,"level":241},{"id":607,"data":1407,"type":298},{"rows":1408,"title":646,"layout":290,"columns":1421},[1409,1411,1413,1415,1417,1419],{"id":611,"label":612,"values":1410},{"pressureA":614,"pressureB":615},{"id":617,"label":618,"values":1412},{"pressureA":620,"pressureB":621},{"id":623,"label":624,"values":1414},{"pressureA":626,"pressureB":627},{"id":629,"label":630,"values":1416},{"pressureA":632,"pressureB":633},{"id":635,"label":636,"values":1418},{"pressureA":638,"pressureB":639},{"id":641,"label":642,"values":1420},{"pressureA":644,"pressureB":645},[1422,1423],{"id":649,"label":650},{"id":652,"label":653},{"id":655,"data":1425,"type":41},{"text":657,"level":241},{"id":659,"data":1427,"type":290},{"content":1428,"stretched":42,"withHeadings":13},[1429,1430,1431,1432,1433,1434,1435],[663,664],[294,666],[297,668],[670,671],[673,674],[676,677],[679,680],{"id":682,"data":1437,"type":217},{"text":684},{"id":686,"data":1439,"type":41},{"text":688,"level":241},{"id":690,"data":1441,"type":224},{"body":692,"title":693,"variant":230},{"id":695,"data":1443,"type":41},{"text":697,"level":240},{"id":699,"data":1445,"type":217},{"text":701},{"id":703,"data":1447,"type":217},{"text":705},{"id":707,"data":1449,"type":217},{"text":709},{"id":711,"data":1451,"type":290},{"content":1452,"stretched":42,"withHeadings":13},[1453,1454,1455,1456,1457,1458],[715,716],[718,719],[721,722],[724,725],[727,728],[730,731],{"id":733,"data":1460,"type":41},{"text":735,"level":240},{"id":737,"data":1462,"type":217},{"text":739},{"id":741,"data":1464,"type":217},{"text":743},{"id":745,"data":1466,"type":217},{"text":747},{"id":749,"data":1468,"type":41},{"text":751,"level":240},{"id":753,"data":1470,"type":217},{"text":755},{"id":757,"data":1472,"type":217},{"text":759},{"id":761,"data":1474,"type":217},{"text":763},{"id":765,"data":1476,"type":224},{"body":767,"title":768,"variant":392},{"id":770,"data":1478,"type":41},{"text":772,"level":241},{"id":774,"data":1480,"type":217},{"text":776},{"id":778,"data":1482,"type":217},{"text":780},{"id":782,"data":1484,"type":217},{"text":784},{"id":786,"data":1486,"type":217},{"text":788},{"id":790,"data":1488,"type":217},{"text":792},{"id":794,"data":1490,"type":41},{"text":796,"level":241},{"id":798,"data":1492,"type":290},{"content":1493,"stretched":42,"withHeadings":13},[1494,1495,1496,1497,1498,1499,1500,1501,1502],[802,803],[805,806],[808,809],[811,812],[814,815],[817,818],[820,821],[823,824],[826,827],{"id":829,"data":1504,"type":41},{"text":831,"level":241},{"id":833,"data":1506,"type":290},{"content":1507,"stretched":42,"withHeadings":13},[1508,1509,1510,1511,1512,1513,1514,1515,1516,1517],[837,838],[840,841],[843,844],[846,847],[849,850],[852,853],[855,856],[858,859],[861,862],[864,865],{"id":867,"data":1519,"type":41},{"text":869,"level":241},{"id":871,"data":1521,"type":339},{"steps":1522,"title":904,"orientation":338},[1523,1524,1525,1526,1527,1528,1529,1530,1531,1532],{"label":875,"description":876},{"label":878,"description":879},{"label":881,"description":882},{"label":884,"description":885},{"label":887,"description":888},{"label":890,"description":891},{"label":893,"description":894},{"label":896,"description":897},{"label":899,"description":900},{"label":902,"description":903},{"id":906,"data":1534,"type":41},{"text":908,"level":241},{"id":910,"data":1536,"type":217},{"text":912},{"id":914,"data":1538,"type":217},{"text":916},{"id":918,"data":1540,"type":217},{"text":920},{"id":922,"data":1542,"type":217},{"text":924},{"id":926,"data":1544,"type":217},{"text":928},{"id":930,"data":1546,"type":41},{"text":932,"level":241},{"id":934,"data":1548,"type":217},{"text":936},{"id":938,"data":1550,"type":217},{"text":940},{"id":942,"data":1552,"type":41},{"text":944,"level":241},{"id":946,"data":1554,"type":290},{"content":1555,"stretched":42,"withHeadings":13},[1556,1557,1558,1559,1560,1561,1562,1563,1564,1565,1566,1567,1568],[950,951],[953,954],[956,957],[959,960],[962,963],[965,966],[968,969],[971,972],[974,975],[977,978],[980,981],[983,984],[986,987],{"id":989,"data":1570,"type":41},{"text":991,"level":241},{"id":993,"data":1572,"type":217},{"text":995},{"id":997,"data":1574,"type":217},{"text":999},{"id":1001,"data":1576,"type":217},{"text":1003},{"id":1005,"data":1578,"type":41},{"text":1007,"level":241},{"id":1009,"data":1580,"type":217},{"text":1011},{"id":1013,"data":1582,"type":217},{"text":1015},{"id":1017,"data":1584,"type":1025},{"link":1019,"meta":1585},{"image":1586,"title":1023,"description":1024},{"url":1022},{"id":1027,"data":1588,"type":217},{"text":1029},{"id":1031,"data":1590,"type":41},{"text":1033,"level":241},{"id":1035,"data":1592,"type":1035},{"items":1593,"title":1062},[1594,1595,1596,1597,1598,1599],{"id":1039,"answer":1040,"question":1041},{"id":1043,"answer":1044,"question":1045},{"id":1047,"answer":1048,"question":1049},{"id":1051,"answer":1052,"question":1053},{"id":1055,"answer":1056,"question":1057},{"id":1059,"answer":1060,"question":1061},{"id":1064,"data":1601,"type":41},{"text":1066,"level":241},{"id":1068,"data":1603,"type":1068},{"title":1070,"entries":1604},[1605,1606,1607,1608,1609,1610,1611,1612],{"term":1073,"anchor":1074,"definition":1075},{"term":1077,"anchor":1078,"definition":1079},{"term":1081,"anchor":1082,"definition":1083},{"term":1085,"anchor":1086,"definition":1087},{"term":1089,"anchor":1090,"definition":1091},{"term":1093,"anchor":1094,"definition":1095},{"term":1097,"anchor":1098,"definition":1099},{"term":1101,"anchor":1102,"definition":1103},{"id":1105,"data":1614,"type":41},{"text":1107,"level":241},{"id":1109,"data":1616,"type":217},{"text":1111},{"id":1113,"data":1618,"type":1025},{"link":1115,"meta":1619},{"image":1620,"title":1118,"description":1119},{"url":1022},{"id":1121,"data":1622,"type":1025},{"link":1123,"meta":1623},{"image":1624,"title":1126,"description":1127},{"url":1022},{"id":1129,"data":1626,"type":1025},{"link":1131,"meta":1627},{"image":1628,"title":1134,"description":1135},{"url":1022},{"id":1137,"data":1630,"type":1025},{"link":1139,"meta":1631},{"image":1632,"title":1142,"description":1143},{"url":1022},{"id":1145,"data":1634,"type":1025},{"link":1147,"meta":1635},{"image":1636,"title":1150,"description":1151},{"url":1022},{"id":1153,"data":1638,"type":1025},{"link":1155,"meta":1639},{"image":1640,"title":1158,"description":1159},{"url":1022},{"id":1161,"data":1642,"type":1025},{"link":1163,"meta":1643},{"image":1644,"title":1166,"description":1167},{"url":1022},{"id":1169,"data":1646,"type":1025},{"link":1171,"meta":1647},{"image":1648,"title":1174,"description":1175},{"url":1022},{"id":1177,"data":1650,"type":1025},{"link":1179,"meta":1651},{"image":1652,"title":1182,"description":1183},{"url":1022},{"id":1185,"data":1654,"type":1025},{"link":1187,"meta":1655},{"image":1656,"title":1190,"description":1191},{"url":1022},{"id":1193,"data":1658,"type":1025},{"link":1195,"meta":1659},{"image":1660,"title":1198,"description":1199},{"url":1022},{"lang":1662,"title":1663,"content":1664,"contentJson":1665,"excerpt":2432},"en","What Is an AI Platform Architect? Models, Data, Runtime, Security and Operations","{\"time\":1791476955677,\"blocks\":[{\"id\":\"intro\",\"data\":{\"text\":\"An \u003Cstrong>AI Platform Architect\u003C\u002Fstrong> designs the reusable AI foundation through which multiple applications, teams, or tenant contexts access models, data and retrieval, agent and tool runtimes, identity and permissions, evaluation, observability, quotas, secrets, and deployment capabilities. The role is broader than infrastructure but narrower than owning every AI-enabled product: its central responsibility is deciding \u003Cstrong>what should be shared, how shared capabilities are governed and isolated, and what must remain solution-specific\u003C\u002Fstrong>.\"},\"type\":\"paragraph\"},{\"id\":\"direct\",\"data\":{\"body\":\"\u003Cstrong>An AI Platform Architect designs the shared technical and operational substrate for AI systems.\u003C\u002Fstrong> Instead of architecting one assistant or one workflow, the role defines reusable contracts and boundaries for model\u002Fprovider access, gateways and routing, retrieval services, agent runtimes, tool access, identity and tenant isolation, secrets, evaluation, telemetry, deployment and lifecycle management.\",\"title\":\"Direct answer\",\"variant\":\"info\"},\"type\":\"callout\"},{\"id\":\"term-note\",\"data\":{\"body\":\"\u003Cstrong>AI Platform Architect is a practical role label, not a universally standardized job title.\u003C\u002Fstrong> ISO\u002FIEC\u002FIEEE 42010:2022 defines concepts for architecture descriptions, not this role. Different organizations may split these responsibilities among platform architects, solution architects, enterprise architects, security architects, MLOps\u002FLLMOps specialists and platform engineering teams. This article uses the term for the architecture responsibility over a reusable AI platform layer.\",\"title\":\"Terminology note\",\"variant\":\"note\"},\"type\":\"callout\"},{\"id\":\"version-note\",\"data\":{\"body\":\"The stable architectural principles here are vendor-neutral. Current Microsoft, AWS and NIST guidance is used as external implementation and governance evidence. NIST states that AI RMF 1.0 is being revised; vendor platform features, gateway products, agent runtimes and model capabilities evolve faster than the architectural principles, so version-sensitive implementation choices must be rechecked before deployment.\",\"title\":\"Current-source note — 8 October 2026\",\"variant\":\"note\"},\"type\":\"callout\"},{\"id\":\"toc\",\"data\":{\"title\":\"Contents\",\"maxLevel\":3,\"minLevel\":2},\"type\":\"tableOfContents\"},{\"id\":\"h-meaning\",\"data\":{\"text\":\"What does an AI Platform Architect actually architect?\",\"level\":2},\"type\":\"header\"},{\"id\":\"p-meaning-1\",\"data\":{\"text\":\"The object of the work is the \u003Cstrong>platform\u003C\u002Fstrong>: a set of shared capabilities that reduces repeated integration work while preserving explicit security, data and operational boundaries. A platform can expose model access, provider adapters, retrieval primitives, agent execution, tool brokers, policy enforcement, evaluation, telemetry and deployment services to many consuming solutions.\"},\"type\":\"paragraph\"},{\"id\":\"p-meaning-2\",\"data\":{\"text\":\"The platform is not valuable merely because components are centralized. It is valuable when consumers receive stable capabilities with clear contracts, ownership, isolation, observability and lifecycle rules. The key architectural question is therefore not “Which model should everyone use?” but \u003Cstrong>“Which responsibilities can be safely standardized and reused without erasing the requirements of each solution?”\u003C\u002Fstrong>.\"},\"type\":\"paragraph\"},{\"id\":\"solution-vs-platform\",\"data\":{\"rows\":[{\"id\":\"c1\",\"label\":\"Primary scope\",\"values\":{\"platform\":\"Reusable AI capabilities consumed by multiple solutions, teams or tenant contexts.\",\"solution\":\"One concrete AI-enabled product, workflow or application.\"}},{\"id\":\"c2\",\"label\":\"Main question\",\"values\":{\"platform\":\"Which shared capabilities and controls should solutions consume, and where must solution-specific ownership remain?\",\"solution\":\"How should this solution meet its business, data, security, quality and operational requirements?\"}},{\"id\":\"c3\",\"label\":\"Data authority\",\"values\":{\"platform\":\"Provides storage, retrieval, provenance or access primitives without automatically becoming the authority for every domain.\",\"solution\":\"Defines which domain data is authoritative and how the solution may use it.\"}},{\"id\":\"c4\",\"label\":\"Evaluation\",\"values\":{\"platform\":\"Provides reusable evaluation, telemetry and release mechanisms; it cannot define every domain's success threshold.\",\"solution\":\"Defines task-specific quality and acceptance criteria.\"}},{\"id\":\"c5\",\"label\":\"Lifecycle\",\"values\":{\"platform\":\"Owns shared capability versions, compatibility, onboarding, quotas, policy and operational contracts.\",\"solution\":\"Owns the lifecycle of the specific workload.\"}}],\"title\":\"Solution architecture and platform architecture solve different scope problems\",\"layout\":\"table\",\"columns\":[{\"id\":\"solution\",\"label\":\"AI Solution Architect\"},{\"id\":\"platform\",\"label\":\"AI Platform Architect\"}]},\"type\":\"comparison\"},{\"id\":\"h-simple\",\"data\":{\"text\":\"The simplest example\",\"level\":2},\"type\":\"header\"},{\"id\":\"p-simple-1\",\"data\":{\"text\":\"Imagine an organization has five AI-enabled products: an internal document assistant, a customer-support copilot, a software-engineering agent, a contract review workflow and a product-search assistant. Each product could independently integrate model APIs, keep credentials, implement retries, collect token metrics, create retrieval code and build its own tool permissions.\"},\"type\":\"paragraph\"},{\"id\":\"p-simple-2\",\"data\":{\"text\":\"That duplication is expensive and dangerous when every team invents a different security and operational model. A shared platform can instead offer approved provider connections, model discovery, quotas, credentials, tenant-aware access, common telemetry, reusable retrieval services and an agent\u002Ftool runtime contract.\"},\"type\":\"paragraph\"},{\"id\":\"p-simple-3\",\"data\":{\"text\":\"But the platform must stop at the correct boundary. The contract-review solution may require legal-document authority and citation rules that the software agent does not. The product-search assistant may need commerce-specific freshness and authorization rules. \u003Cstrong>Reusable infrastructure does not make all domain truth reusable.\u003C\u002Fstrong>\"},\"type\":\"paragraph\"},{\"id\":\"simple-flow\",\"data\":{\"steps\":[{\"label\":\"1. Consumer identifies itself\",\"description\":\"The calling application, user, service, team or tenant enters through an authenticated identity and explicit scope.\"},{\"label\":\"2. Platform policy applies\",\"description\":\"Gateway and policy layers determine allowed providers, models, quotas, data paths, tools and execution modes.\"},{\"label\":\"3. Shared capability executes\",\"description\":\"The request may use inference, retrieval, agent runtime, tool access or another reusable platform service.\"},{\"label\":\"4. Solution-specific context remains authoritative\",\"description\":\"The consuming solution supplies domain rules, user intent, data authority, task-specific constraints and acceptance logic.\"},{\"label\":\"5. Telemetry and evidence are captured\",\"description\":\"The platform records identity, route, model\u002Fprovider, latency, cost, errors, tool activity and other permitted observability signals.\"},{\"label\":\"6. Result returns under the solution contract\",\"description\":\"The solution remains responsible for whether the output is acceptable for its user and domain.\"}],\"title\":\"A shared AI request path\",\"orientation\":\"auto\"},\"type\":\"processFlow\"},{\"id\":\"h-stops\",\"data\":{\"text\":\"Where the simple example stops\",\"level\":2},\"type\":\"header\"},{\"id\":\"p-stops-1\",\"data\":{\"text\":\"Centralization is not automatically architecture. A single endpoint in front of several model APIs is useful, but it does not by itself create an AI platform. A production platform also needs identity boundaries, capability contracts, provider health and lifecycle handling, quotas, secret ownership, observability, compatibility rules, security controls, release discipline and clear operational responsibility.\"},\"type\":\"paragraph\"},{\"id\":\"p-stops-2\",\"data\":{\"text\":\"The opposite failure is also common: putting every prompt, vector index, business rule, agent and application workflow into one “AI backend.” That creates a monolith whose shared status is accidental rather than architectural. \u003Cstrong>A platform should standardize cross-cutting capabilities, not absorb domain ownership merely because AI is involved.\u003C\u002Fstrong>\"},\"type\":\"paragraph\"},{\"id\":\"h-boundary\",\"data\":{\"text\":\"The most important platform decision: shared versus solution-specific\",\"level\":2},\"type\":\"header\"},{\"id\":\"shared-boundary-table\",\"data\":{\"content\":[[\"Capability area\",\"Good candidate for shared platform ownership\",\"Usually remains solution-specific\"],[\"Model access\",\"Approved provider connections, adapters, credentials, health, routing primitives, quotas\",\"Task-specific model acceptance, prompt behavior, quality threshold\"],[\"Retrieval\",\"Ingestion primitives, extraction, indexing, search APIs, provenance contracts, authorization hooks\",\"Authoritative corpus, freshness rules, domain metadata, evidence sufficiency\"],[\"Agents and tools\",\"Runtime lifecycle, tool registry\u002Fbroker, permission enforcement, tracing, cancellation\",\"Business workflow, allowed action semantics, escalation policy, task success\"],[\"Security\",\"Identity integration, secret storage, policy enforcement, audit contracts, tenant isolation mechanisms\",\"Data classification, business authorization rules, domain-specific risk acceptance\"],[\"Evaluation\",\"Harness, dataset\u002Fversion mechanics, telemetry, experiment\u002Frelease workflow\",\"Ground truth, domain test set, acceptance threshold, user outcome\"],[\"Operations\",\"Deployment pattern, health, metrics, incident integration, capacity controls\",\"Solution SLOs where they differ, business continuity impact, workload-specific runbooks\"]],\"stretched\":false,\"withHeadings\":true},\"type\":\"table\"},{\"id\":\"boundary-principle\",\"data\":{\"body\":\"\u003Cstrong>Share mechanics and controls where reuse is real; keep authority and acceptance where the domain owns them.\u003C\u002Fstrong> This prevents two opposite errors: duplicated infrastructure everywhere, and a central platform that falsely becomes the owner of every application's data, policy and quality.\",\"title\":\"Platform principle\",\"variant\":\"success\"},\"type\":\"callout\"},{\"id\":\"h-responsibility-map\",\"data\":{\"text\":\"Architecture responsibility map\",\"level\":2},\"type\":\"header\"},{\"id\":\"h-provider\",\"data\":{\"text\":\"1. Model and provider access\",\"level\":3},\"type\":\"header\"},{\"id\":\"p-provider-1\",\"data\":{\"text\":\"A platform architect defines how consumers discover and invoke models without forcing every application to hard-code one provider. This includes provider adapters, model identifiers, capability metadata, authentication, health checks, endpoint configuration, request normalization and compatibility behavior.\"},\"type\":\"paragraph\"},{\"id\":\"p-provider-2\",\"data\":{\"text\":\"Provider abstraction must remain honest. Different providers expose different context limits, tool semantics, structured-output behavior, multimodal capabilities, safety controls, caching, pricing and failure modes. A good abstraction creates a stable platform contract while preserving access to capabilities that cannot be meaningfully flattened.\"},\"type\":\"paragraph\"},{\"id\":\"provider-warning\",\"data\":{\"body\":\"A lowest-common-denominator API can make migration easier but can also erase capabilities that matter. The architecture should define which features are portable, which are provider-specific and how consumers discover that difference.\",\"title\":\"Do not confuse abstraction with pretending providers are identical\",\"variant\":\"warning\"},\"type\":\"callout\"},{\"id\":\"h-gateway\",\"data\":{\"text\":\"2. Gateway, routing, quotas and cost controls\",\"level\":3},\"type\":\"header\"},{\"id\":\"p-gateway-1\",\"data\":{\"text\":\"A shared AI gateway can centralize authentication, routing, throttling, retries, token limits, usage attribution and policy enforcement. Microsoft’s current AI Gateway guidance explicitly treats token-per-minute limits, quotas and multi-project containment as platform concerns; AWS likewise exposes account and model quotas and centralized controls.\"},\"type\":\"paragraph\"},{\"id\":\"p-gateway-2\",\"data\":{\"text\":\"The gateway is therefore more than a reverse proxy when it carries AI-specific policy and operational semantics. But it should not silently make business decisions. A routing policy may prefer a healthy local model, a lower-cost provider or a regionally compliant endpoint; whether that route is acceptable for a particular task is still a contract between platform and solution.\"},\"type\":\"paragraph\"},{\"id\":\"p-gateway-3\",\"data\":{\"text\":\"Routing also needs failure semantics. If the preferred model is unavailable, the platform must know whether fallback is permitted, whether a cloud route requires explicit consent, whether a lower-capability model is valid and how the decision is surfaced to observability.\"},\"type\":\"paragraph\"},{\"id\":\"h-data\",\"data\":{\"text\":\"3. Shared data, retrieval and grounding services\",\"level\":3},\"type\":\"header\"},{\"id\":\"p-data-1\",\"data\":{\"text\":\"Retrieval services are strong platform candidates because parsing, chunking, indexing, lexical search, semantic search, metadata filtering, provenance and citation mechanics are reusable. However, the platform must not confuse a shared retrieval engine with a shared source of truth.\"},\"type\":\"paragraph\"},{\"id\":\"p-data-2\",\"data\":{\"text\":\"A solution still owns questions such as: Which corpus is authoritative? Which version is valid? Can this user see this document? How fresh must the data be? What counts as sufficient evidence? Can an answer be generated when retrieval fails? Those are domain and solution requirements even when the platform supplies the retrieval machinery.\"},\"type\":\"paragraph\"},{\"id\":\"p-data-3\",\"data\":{\"text\":\"This boundary is especially important in multi-tenant systems. A technically shared index or vector service does not justify cross-tenant visibility. Authorization context must be preserved through retrieval, not added only after search results have already crossed the boundary.\"},\"type\":\"paragraph\"},{\"id\":\"h-agent-runtime\",\"data\":{\"text\":\"4. Agent and tool runtime\",\"level\":3},\"type\":\"header\"},{\"id\":\"p-agent-1\",\"data\":{\"text\":\"Agentic systems add reusable runtime concerns: thread\u002Fsession lifecycle, planning loops, tool registration, tool invocation, cancellation, timeouts, human approvals, memory\u002Fstate interfaces, remote-agent protocols and trace correlation. A platform can provide these mechanics so each product does not rebuild them.\"},\"type\":\"paragraph\"},{\"id\":\"p-agent-2\",\"data\":{\"text\":\"The platform must also keep tool permission separate from model capability. A model being capable of generating a shell command does not mean the runtime should allow shell execution. The permission boundary belongs to the application\u002Fruntime architecture and must be enforceable independently of the model.\"},\"type\":\"paragraph\"},{\"id\":\"p-agent-3\",\"data\":{\"text\":\"Current AWS Agentic AI guidance emphasizes bounded agents, explicit authority, end-to-end tracing, versioned behavioral artifacts and human oversight proportionate to consequence. Those are platform-enabling concerns, but the consuming solution still defines what actions are legitimate for its domain.\"},\"type\":\"paragraph\"},{\"id\":\"h-identity\",\"data\":{\"text\":\"5. Identity, tenant isolation and authorization\",\"level\":3},\"type\":\"header\"},{\"id\":\"p-identity-1\",\"data\":{\"text\":\"AI platforms often sit in front of high-value models, proprietary data and action-capable tools. Authentication is therefore only the beginning. The architecture must carry user, service, application and tenant context through every privileged operation that needs it.\"},\"type\":\"paragraph\"},{\"id\":\"p-identity-2\",\"data\":{\"text\":\"\u003Cstrong>RBAC and tenant isolation solve different problems.\u003C\u002Fstrong> RBAC answers what an identity may do; tenant isolation answers which tenant’s resources that identity may act on. A platform that checks roles but loses tenant context can still expose the wrong data.\"},\"type\":\"paragraph\"},{\"id\":\"p-identity-3\",\"data\":{\"text\":\"Microsoft’s current AI workload guidance explicitly recommends identity segmentation and authorization-aware access to content. AWS’s multi-tenant generative AI platform guidance similarly treats logical isolation, centralized controls and auditability as platform concerns.\"},\"type\":\"paragraph\"},{\"id\":\"h-secrets\",\"data\":{\"text\":\"6. Secrets, credentials and trust boundaries\",\"level\":3},\"type\":\"header\"},{\"id\":\"p-secrets-1\",\"data\":{\"text\":\"A platform should define who owns provider keys, remote bearer tokens, signing material and tool credentials, where they are stored, which process can access them, how they are rotated and whether they can ever reach a browser or untrusted renderer.\"},\"type\":\"paragraph\"},{\"id\":\"p-secrets-2\",\"data\":{\"text\":\"This is an architectural boundary, not an implementation detail. If every consuming application copies provider credentials into its own configuration, the organization has duplicated both operational burden and blast radius. Centralization can reduce that risk only if the platform itself has narrower, auditable access paths.\"},\"type\":\"paragraph\"},{\"id\":\"h-eval\",\"data\":{\"text\":\"7. Evaluation, observability and auditability\",\"level\":3},\"type\":\"header\"},{\"id\":\"p-eval-1\",\"data\":{\"text\":\"A reusable platform can provide evaluation harnesses, trace IDs, model\u002Fprovider metadata, token and cost metrics, latency, error rates, prompt\u002Fmodel version linkage, agent\u002Ftool traces and controlled logging. AWS and Microsoft both treat observability and evaluation as core production concerns for AI workloads.\"},\"type\":\"paragraph\"},{\"id\":\"p-eval-2\",\"data\":{\"text\":\"Platform evaluation and solution evaluation must remain separate. A platform can verify that an endpoint is healthy, a model version passes a general regression suite and traces are complete. It cannot decide that a legal answer, medical workflow or product recommendation is acceptable without domain-specific ground truth and acceptance criteria.\"},\"type\":\"paragraph\"},{\"id\":\"p-eval-3\",\"data\":{\"text\":\"Logging also creates a privacy boundary. Prompt and response logs may contain sensitive or proprietary data. The platform architect must therefore decide what is logged, redacted, sampled, retained and accessible rather than assuming that more telemetry is always safer.\"},\"type\":\"paragraph\"},{\"id\":\"h-runtime\",\"data\":{\"text\":\"8. Runtime, deployment and locality\",\"level\":3},\"type\":\"header\"},{\"id\":\"p-runtime-1\",\"data\":{\"text\":\"A platform architect decides how shared AI capabilities are deployed and reached: managed cloud services, self-hosted endpoints, local inference, hybrid routing, containerized services, desktop runtimes, private networking or air-gapped environments. The important distinction is between \u003Cstrong>where the control\u002Fruntime process runs\u003C\u002Fstrong> and \u003Cstrong>where inference and data processing actually occur\u003C\u002Fstrong>.\"},\"type\":\"paragraph\"},{\"id\":\"p-runtime-2\",\"data\":{\"text\":\"A local client may still call a cloud model. A cloud control plane may route to an on-premises model. A remote agent may execute tools inside a customer network. Architectural diagrams must therefore show trust and data-flow boundaries rather than using “local” and “cloud” as vague labels.\"},\"type\":\"paragraph\"},{\"id\":\"h-lifecycle\",\"data\":{\"text\":\"9. Platform lifecycle, compatibility and onboarding\",\"level\":3},\"type\":\"header\"},{\"id\":\"p-lifecycle-1\",\"data\":{\"text\":\"Reusable capability becomes a platform only when consumers can depend on it over time. That requires versioned contracts, migration rules, compatibility policy, deprecation, release testing, rollback, incident ownership, capacity planning, documentation and a path for onboarding new teams or applications.\"},\"type\":\"paragraph\"},{\"id\":\"p-lifecycle-2\",\"data\":{\"text\":\"Fast-moving AI ecosystems make this particularly important. Model names, SDKs, protocol versions, provider APIs and safety capabilities change independently. A platform must absorb some of that volatility without hiding changes that materially affect a solution’s behavior.\"},\"type\":\"paragraph\"},{\"id\":\"h-control-plane\",\"data\":{\"text\":\"A practical control-plane \u002F execution-plane \u002F solution-plane model\",\"level\":2},\"type\":\"header\"},{\"id\":\"model-note\",\"data\":{\"body\":\"The three-plane model below is a practical way to reason about responsibilities; it is not an ISO, NIST, Microsoft or AWS standard. Its purpose is to make ownership boundaries explicit.\",\"title\":\"Proposed architecture model\",\"variant\":\"note\"},\"type\":\"callout\"},{\"id\":\"planes-table\",\"data\":{\"content\":[[\"Plane\",\"Typical responsibilities\",\"Should not silently own\"],[\"Platform control plane\",\"Provider registry, model policy, quotas, tenant configuration, identities, secrets, routing rules, capability versions, deployment configuration\",\"Application business logic or domain truth\"],[\"Platform execution\u002Fdata plane\",\"Inference requests, retrieval operations, agent\u002Ftool execution, extraction, indexing, telemetry emission, policy enforcement\",\"Cross-tenant access merely because infrastructure is shared\"],[\"Solution plane\",\"User workflow, prompts\u002Finstructions, authoritative corpus selection, domain authorization, business rules, task evaluation and acceptance\",\"Low-level provider integration that the platform explicitly owns\"]],\"stretched\":false,\"withHeadings\":true},\"type\":\"table\"},{\"id\":\"p-control-plane-1\",\"data\":{\"text\":\"This separation helps diagnose platform drift. If an application must know every provider-specific credential and endpoint, the platform contract is too thin. If the platform decides which customer record is legally authoritative or whether a domain answer is acceptable, the platform has crossed into solution ownership.\"},\"type\":\"paragraph\"},{\"id\":\"h-artifacts\",\"data\":{\"text\":\"What should an AI Platform Architect produce?\",\"level\":2},\"type\":\"header\"},{\"id\":\"artifacts-table\",\"data\":{\"content\":[[\"Architecture artifact\",\"Purpose\"],[\"Platform capability map\",\"Defines what the platform provides, who consumes it and which capabilities remain outside scope.\"],[\"Provider\u002Fmodel contract\",\"Defines providers, models, capabilities, abstraction boundaries, route metadata and fallback semantics.\"],[\"Identity and tenancy model\",\"Defines user\u002Fservice\u002Fapplication identity, tenant context, RBAC\u002FABAC hooks and resource isolation.\"],[\"Gateway and quota policy\",\"Defines rate limits, token\u002Fcost budgets, routing controls, retries and capacity behavior.\"],[\"Retrieval\u002Fdata contract\",\"Defines ingestion, provenance, search, metadata, authorization propagation and where domain authority remains.\"],[\"Agent\u002Ftool contract\",\"Defines runtime lifecycle, tool registration, permissions, approvals, cancellation and trace behavior.\"],[\"Secret and trust-boundary model\",\"Defines credential ownership, storage, process boundaries, rotation and sensitive data paths.\"],[\"Evaluation and telemetry contract\",\"Defines common metrics, traces, datasets\u002Fversion links, logging policy and solution extension points.\"],[\"Lifecycle and compatibility policy\",\"Defines versions, migrations, deprecation, releases, rollback, incident ownership and onboarding.\"]],\"stretched\":false,\"withHeadings\":true},\"type\":\"table\"},{\"id\":\"h-tradeoffs\",\"data\":{\"text\":\"The work is mostly trade-offs, not maximum centralization\",\"level\":2},\"type\":\"header\"},{\"id\":\"tradeoff-comparison\",\"data\":{\"rows\":[{\"id\":\"t1\",\"label\":\"Provider abstraction\",\"values\":{\"pressureA\":\"Stable portable platform API\",\"pressureB\":\"Access to provider-specific capabilities and fast innovation\"}},{\"id\":\"t2\",\"label\":\"Reuse\",\"values\":{\"pressureA\":\"Shared services reduce duplication\",\"pressureB\":\"Isolation and domain autonomy prevent unsafe coupling\"}},{\"id\":\"t3\",\"label\":\"Governance\",\"values\":{\"pressureA\":\"Central policy and auditability\",\"pressureB\":\"Team speed and local experimentation\"}},{\"id\":\"t4\",\"label\":\"Observability\",\"values\":{\"pressureA\":\"Rich traces for debugging and evaluation\",\"pressureB\":\"Privacy, data minimization and logging cost\"}},{\"id\":\"t5\",\"label\":\"Availability\",\"values\":{\"pressureA\":\"Fallback and multi-provider resilience\",\"pressureB\":\"Predictable quality, compliance and data-location guarantees\"}},{\"id\":\"t6\",\"label\":\"Platform scope\",\"values\":{\"pressureA\":\"More reusable capabilities\",\"pressureB\":\"Smaller blast radius and less platform lock-in\"}}],\"title\":\"Common platform trade-offs\",\"layout\":\"table\",\"columns\":[{\"id\":\"pressureA\",\"label\":\"Pressure A\"},{\"id\":\"pressureB\",\"label\":\"Pressure B\"}]},\"type\":\"comparison\"},{\"id\":\"h-adjacent\",\"data\":{\"text\":\"How is this different from adjacent roles?\",\"level\":2},\"type\":\"header\"},{\"id\":\"roles-table\",\"data\":{\"content\":[[\"Role\",\"Primary architectural scope\"],[\"AI Solution Architect\",\"A concrete AI-enabled solution and its end-to-end requirements, boundaries, trade-offs and production acceptance.\"],[\"AI Platform Architect\",\"Reusable AI capabilities and operational\u002Fsecurity contracts consumed across multiple solutions or teams.\"],[\"Enterprise Architect\",\"Organization-wide business\u002Ftechnology portfolio, capability and governance alignment at a broader level.\"],[\"MLOps \u002F LLMOps Architect or specialist\",\"Model and AI lifecycle, deployment, experiments, observability, release and operational practices; may overlap strongly but does not automatically own the whole shared application platform.\"],[\"Platform Engineer \u002F SRE\",\"Implements and operates platform infrastructure, reliability, automation and developer experience; architecture responsibility may be shared with the platform architect.\"],[\"AI \u002F Software Engineer\",\"Implements models, integrations, services, agents, retrieval and product functionality inside the agreed architecture.\"]],\"stretched\":false,\"withHeadings\":true},\"type\":\"table\"},{\"id\":\"p-adjacent-1\",\"data\":{\"text\":\"These boundaries are organizational, not universal. In a small team one person may hold several responsibilities. In a regulated enterprise they may be split across architecture, security, platform, data and operations groups. The useful distinction is the \u003Cstrong>scope of architectural responsibility\u003C\u002Fstrong>, not the job title printed on an org chart.\"},\"type\":\"paragraph\"},{\"id\":\"h-evidence\",\"data\":{\"text\":\"Implementation evidence: how these platform boundaries appear in my own work\",\"level\":2},\"type\":\"header\"},{\"id\":\"evidence-note\",\"data\":{\"body\":\"The following sections describe concrete patterns from my own projects. They are evidence that these architectural boundaries have been implemented or explicitly designed in real code and project systems. They are \u003Cstrong>not\u003C\u002Fstrong> claims that the projects together already constitute a commercially deployed enterprise AI platform.\",\"title\":\"Original implementation evidence\",\"variant\":\"note\"},\"type\":\"callout\"},{\"id\":\"h-ai-client\",\"data\":{\"text\":\"Aaasaasa AI Client: provider, runtime and permission separation\",\"level\":3},\"type\":\"header\"},{\"id\":\"p-ai-client-1\",\"data\":{\"text\":\"Aaasaasa AI Client is a local-first desktop AI workspace built with Nuxt 4, Electron and TypeScript. Its AI Hub deliberately separates \u003Cstrong>agent\u002Fclient, provider, model, connection\u002Fruntime location, permissions and web client\u003C\u002Fstrong> instead of treating them as one configuration value.\"},\"type\":\"paragraph\"},{\"id\":\"p-ai-client-2\",\"data\":{\"text\":\"The implementation includes direct provider adapters, Codex agent runtime integration, local Ollama\u002FLM Studio paths, OpenAI-compatible services, centralized workspace permissions, main-process credential storage, DuckDB, Qdrant\u002Fvector support, PDF\u002Freadability extraction and authenticated MCP-based directory access.\"},\"type\":\"paragraph\"},{\"id\":\"p-ai-client-3\",\"data\":{\"text\":\"Two platform lessons are especially relevant. First, a local runtime is not the same as local inference: a local Codex process can still use a cloud model. Second, automatic routing does not silently fall back from local to paid cloud inference. That makes routing policy and runtime locality explicit rather than inferred from UI labels.\"},\"type\":\"paragraph\"},{\"id\":\"ai-client-evidence-table\",\"data\":{\"content\":[[\"Implemented boundary\",\"Platform-architecture meaning\"],[\"Agent vs provider vs model\",\"Different responsibilities can evolve independently instead of being hidden behind one “AI” selector.\"],[\"Permissions separate from model\",\"Filesystem\u002Ftool authority belongs to the runtime policy, not model capability.\"],[\"Main-process secrets\",\"Credential ownership follows the privileged process boundary rather than the renderer\u002FUI.\"],[\"Provider health and model discovery\",\"Routing and availability are runtime\u002Fplatform concerns.\"],[\"No silent cloud fallback\",\"Cost, locality and data-transfer semantics remain explicit policy decisions.\"]],\"stretched\":false,\"withHeadings\":true},\"type\":\"table\"},{\"id\":\"h-cms\",\"data\":{\"text\":\"Aaasaasa AI CMS: tenant-scoped authorization as a platform boundary\",\"level\":3},\"type\":\"header\"},{\"id\":\"p-cms-1\",\"data\":{\"text\":\"The Aaasaasa AI CMS codebase provides a separate implementation example: tenant-scoped RBAC is represented through roles, permissions and user-role assignments bound to a tenant identifier. System permissions are grouped by capability, and role lookup and updates remain tenant-scoped.\"},\"type\":\"paragraph\"},{\"id\":\"p-cms-2\",\"data\":{\"text\":\"This is not itself proof of a complete AI platform, but it is directly relevant to one of the hardest shared-platform boundaries: a reusable service must preserve \u003Cstrong>who may do what\u003C\u002Fstrong> and \u003Cstrong>for which tenant\u003C\u002Fstrong>. Adding AI inference or retrieval on top of an application platform does not remove that requirement.\"},\"type\":\"paragraph\"},{\"id\":\"p-cms-3\",\"data\":{\"text\":\"The architectural implication is that model gateways, retrieval services and agents should consume established identity\u002Ftenant context rather than inventing a parallel AI-only authorization universe.\"},\"type\":\"paragraph\"},{\"id\":\"h-sot\",\"data\":{\"text\":\"Source of Truth Research Engine: shared retrieval mechanics without shared truth\",\"level\":3},\"type\":\"header\"},{\"id\":\"p-sot-1\",\"data\":{\"text\":\"The Source of Truth Research Engine provides a third implementation example. Different research modes share a common evidence core: Sources, Artifacts, provenance, Claims, Relations, Contradictions, a Reference Model and audit trail. The system also provides local lexical retrieval, optional semantic retrieval, extraction, snapshots and SHA-256-based provenance.\"},\"type\":\"paragraph\"},{\"id\":\"p-sot-2\",\"data\":{\"text\":\"The project explicitly treats search and semantic similarity as discovery signals rather than evidence. A result must be traced back to a concrete source and locator before it can support a claim. This is precisely the distinction an AI platform needs: \u003Cstrong>reusable retrieval machinery can be shared while evidence authority remains governed by the consuming methodology and domain.\u003C\u002Fstrong>\"},\"type\":\"paragraph\"},{\"id\":\"p-sot-3\",\"data\":{\"text\":\"The engine also demonstrates why one shared platform does not require one shared interpretation. Historical, scientific\u002Ftechnical, market-intelligence and monitoring modes can reuse core evidence infrastructure while retaining mode-specific methodology.\"},\"type\":\"paragraph\"},{\"id\":\"evidence-synthesis\",\"data\":{\"body\":\"Across these projects, the reusable pattern is not “one backend for everything.” It is \u003Cstrong>separation of concerns plus explicit contracts\u003C\u002Fstrong>: provider\u002Fmodel\u002Fruntime separation, tenant-aware authorization, credential boundaries, reusable data\u002Fretrieval primitives, provenance, and domain-specific authority. A future integrated platform would need stable contracts between those capabilities rather than direct coupling between codebases.\",\"title\":\"What these implementations demonstrate together\",\"variant\":\"success\"},\"type\":\"callout\"},{\"id\":\"h-frameworks\",\"data\":{\"text\":\"How current architecture guidance supports this platform scope\",\"level\":2},\"type\":\"header\"},{\"id\":\"p-frameworks-1\",\"data\":{\"text\":\"ISO\u002FIEC\u002FIEEE 42010:2022 provides a general discipline for architecture descriptions across software, systems, enterprises and related entities. It does not define an AI Platform Architect, but it reinforces the need to express architectural concerns, relationships and viewpoints rather than reducing architecture to a technology list.\"},\"type\":\"paragraph\"},{\"id\":\"p-frameworks-2\",\"data\":{\"text\":\"NIST AI RMF 1.0 and the Generative AI Profile frame AI risk management across the lifecycle rather than only at model selection time. Governance, mapping, measurement and management are therefore compatible with a platform architecture that carries shared controls and evidence across many consuming workloads.\"},\"type\":\"paragraph\"},{\"id\":\"p-frameworks-3\",\"data\":{\"text\":\"Microsoft’s current AI workload guidance treats application design, data, security, operations, testing\u002Fevaluation and GenAIOps as connected architectural areas. Its current AI Gateway guidance also shows practical platform concerns such as centralized model access, project-specific token limits, quotas and multi-team containment.\"},\"type\":\"paragraph\"},{\"id\":\"p-frameworks-4\",\"data\":{\"text\":\"AWS’s current Generative AI Lens and multi-tenant platform scenario similarly separate foundational platform controls from consuming-application ownership. AWS explicitly notes that a central platform can enforce shared guardrails and auditability while data quality and workload-specific observability still remain responsibilities of consuming applications or data producers.\"},\"type\":\"paragraph\"},{\"id\":\"p-frameworks-5\",\"data\":{\"text\":\"The vendor products differ, but the cross-source pattern is stable: production AI platforms must coordinate identity, data access, models, policy, evaluation, observability, capacity, cost and lifecycle. A GPU cluster or model endpoint covers only part of that responsibility.\"},\"type\":\"paragraph\"},{\"id\":\"h-misconceptions\",\"data\":{\"text\":\"Common misconceptions\",\"level\":2},\"type\":\"header\"},{\"id\":\"misconceptions-table\",\"data\":{\"content\":[[\"Misconception\",\"Why it is wrong\"],[\"“An AI platform is the GPU cluster.”\",\"Compute is one substrate. A platform also needs contracts for identity, model access, data, policy, evaluation, observability and lifecycle.\"],[\"“An AI gateway is just a reverse proxy.”\",\"It may also carry model routing, token quotas, cost attribution, policy enforcement, identity and AI-specific telemetry.\"],[\"“Shared means globally shared.”\",\"A service may be physically shared while logically segmented by tenant, application, region, classification or risk level.\"],[\"“One central vector database becomes the company truth.”\",\"A vector store or retrieval service is infrastructure. Domain authority, freshness, provenance and access remain separate concerns.\"],[\"“Platform evaluation replaces solution evaluation.”\",\"General regression and telemetry cannot define whether a domain-specific answer or action is acceptable.\"],[\"“Provider abstraction should hide every difference.”\",\"Some differences are material capabilities, security semantics or failure modes and must remain visible.\"],[\"“RBAC solves multi-tenancy.”\",\"RBAC controls actions; tenant isolation controls resource boundaries. Both can be required.\"],[\"“AI Platform Architect is just another name for MLOps.”\",\"MLOps\u002FLLMOps is a major overlapping discipline, but shared application\u002Fruntime, identity, gateway, retrieval and tool boundaries can extend beyond model lifecycle operations.\"]],\"stretched\":false,\"withHeadings\":true},\"type\":\"table\"},{\"id\":\"h-failure\",\"data\":{\"text\":\"Failure modes an AI Platform Architect should prevent\",\"level\":2},\"type\":\"header\"},{\"id\":\"failures-table\",\"data\":{\"content\":[[\"Failure mode\",\"Architectural consequence\"],[\"Every team stores its own provider keys\",\"Duplicated secret handling, inconsistent rotation and larger blast radius.\"],[\"Provider abstraction hides required capabilities\",\"Consumers cannot use features they need or silently receive behavior different from assumptions.\"],[\"Shared retrieval ignores tenant\u002Fuser context\",\"Cross-boundary data leakage can occur before the application gets a chance to filter results.\"],[\"Fallback silently changes provider or locality\",\"Cost, compliance, data location and output quality can change without the caller knowing.\"],[\"Agent tools are granted by model choice\",\"A capable model becomes over-privileged because runtime authority is not independently enforced.\"],[\"All prompts\u002Fresponses are logged by default\",\"Observability can create a new sensitive-data repository and compliance problem.\"],[\"Platform owns one generic quality score\",\"Domain failures remain hidden behind platform health metrics.\"],[\"No version contract for platform capabilities\",\"Model\u002Fprovider\u002Fruntime changes break consumers unpredictably.\"],[\"Everything AI-related is centralized\",\"The platform becomes a bottleneck and monolith instead of a reusable capability layer.\"]],\"stretched\":false,\"withHeadings\":true},\"type\":\"table\"},{\"id\":\"h-decision\",\"data\":{\"text\":\"A practical platform-architecture decision sequence\",\"level\":2},\"type\":\"header\"},{\"id\":\"decision-flow\",\"data\":{\"steps\":[{\"label\":\"1. Identify real consumers\",\"description\":\"List solutions, teams, tenants and workloads that would consume the platform; avoid building a platform for hypothetical reuse.\"},{\"label\":\"2. Define the shared boundary\",\"description\":\"Separate cross-cutting mechanics from solution-specific domain authority, workflow and acceptance.\"},{\"label\":\"3. Define identity and isolation first\",\"description\":\"Establish users, services, applications, tenants, regions and data classifications before sharing retrieval or tool capabilities.\"},{\"label\":\"4. Define capability contracts\",\"description\":\"Specify model\u002Fprovider, retrieval, agent\u002Ftool, gateway and telemetry APIs with explicit ownership and versioning.\"},{\"label\":\"5. Decide provider and runtime strategy\",\"description\":\"Choose managed, self-hosted, local or hybrid execution and document fallback, locality and capability semantics.\"},{\"label\":\"6. Design data and retrieval boundaries\",\"description\":\"Define provenance, authorization propagation, corpus ownership, indexing and evidence responsibilities.\"},{\"label\":\"7. Add quotas, secrets and policy\",\"description\":\"Control cost, capacity, credentials, tool permissions, safety controls and blast radius.\"},{\"label\":\"8. Build evaluation and observability contracts\",\"description\":\"Provide platform metrics and tracing while leaving domain ground truth and acceptance to the solution.\"},{\"label\":\"9. Define lifecycle and operations\",\"description\":\"Version capabilities, test upgrades, document deprecation, rollback, incidents, capacity and consumer onboarding.\"},{\"label\":\"10. Validate with more than one consumer\",\"description\":\"A platform claim becomes credible when the shared capability actually serves distinct workloads without forcing them into the same domain model.\"}],\"title\":\"From platform need to operable shared capability\",\"orientation\":\"auto\"},\"type\":\"processFlow\"},{\"id\":\"h-edge\",\"data\":{\"text\":\"Edge cases and limits of the role\",\"level\":2},\"type\":\"header\"},{\"id\":\"p-edge-1\",\"data\":{\"text\":\"A small organization with one AI application may not need a distinct AI platform or platform architect. Premature platforming can create more abstraction than value. The correct architecture may be one well-designed solution with a few reusable modules.\"},\"type\":\"paragraph\"},{\"id\":\"p-edge-2\",\"data\":{\"text\":\"An air-gapped or sovereign deployment changes the provider, update and observability model substantially. Model hosting, artifact distribution, identity integration and telemetry export may all need local equivalents.\"},\"type\":\"paragraph\"},{\"id\":\"p-edge-3\",\"data\":{\"text\":\"Highly regulated or high-consequence workloads may require stronger physical or organizational isolation instead of a logically shared platform. Reuse is never a sufficient reason to weaken a required security boundary.\"},\"type\":\"paragraph\"},{\"id\":\"p-edge-4\",\"data\":{\"text\":\"Managed cloud AI services can remove implementation burden but do not remove architectural accountability. The organization still decides identity, data access, logging, retention, quotas, model eligibility, fallback, evaluation and solution acceptance.\"},\"type\":\"paragraph\"},{\"id\":\"p-edge-5\",\"data\":{\"text\":\"The platform boundary may also differ by modality. Text inference, multimodal generation, speech, computer use and autonomous agents can have different latency, data, permission and observability requirements even when they share provider and identity infrastructure.\"},\"type\":\"paragraph\"},{\"id\":\"h-change\",\"data\":{\"text\":\"What would change this answer?\",\"level\":2},\"type\":\"header\"},{\"id\":\"p-change-1\",\"data\":{\"text\":\"The core definition would change if the organizational scope changes. If the architect owns one workload, the role becomes closer to AI Solution Architect. If the responsibility expands to organization-wide capability strategy, investment, standards and target-state portfolios, it moves toward Enterprise AI Architecture.\"},\"type\":\"paragraph\"},{\"id\":\"p-change-2\",\"data\":{\"text\":\"Implementation guidance changes whenever providers, gateway products, agent protocols, regulatory obligations, model capabilities or deployment constraints change. That is why platform architecture should express stable responsibilities and contracts separately from current vendor mechanisms.\"},\"type\":\"paragraph\"},{\"id\":\"h-checklist\",\"data\":{\"text\":\"AI Platform Architect checklist\",\"level\":2},\"type\":\"header\"},{\"id\":\"checklist-table\",\"data\":{\"content\":[[\"Question\",\"Expected answer\"],[\"Who are the actual platform consumers?\",\"Named solutions, teams or tenant contexts with distinct but overlapping needs.\"],[\"What is genuinely shared?\",\"Explicit capability list, not a vague “AI backend.”\"],[\"What must remain solution-specific?\",\"Domain authority, business workflow, task acceptance and other workload-owned concerns.\"],[\"How are models\u002Fproviders represented?\",\"Versioned provider\u002Fmodel contracts with capabilities and explicit fallback semantics.\"],[\"How is identity propagated?\",\"User\u002Fservice\u002Fapplication\u002Ftenant context survives every privileged request path.\"],[\"How is tenant isolation enforced?\",\"Resource scoping is separate from role permission checks.\"],[\"How are secrets handled?\",\"Privileged storage, rotation, limited exposure and auditable ownership.\"],[\"How does retrieval preserve authority?\",\"Shared mechanics with authorization, provenance and domain-owned evidence rules.\"],[\"How are tools and agents constrained?\",\"Runtime permissions, bounded tool contracts, approvals, cancellation and traceability.\"],[\"How are cost and capacity controlled?\",\"Quotas, token\u002Frate controls, usage attribution and overload behavior.\"],[\"How is quality measured?\",\"Platform regression\u002Fevaluation plus solution-specific ground truth and acceptance.\"],[\"How are changes rolled out?\",\"Versioning, compatibility, migration, deprecation, rollback and incident ownership.\"]],\"stretched\":false,\"withHeadings\":true},\"type\":\"table\"},{\"id\":\"h-conclusion\",\"data\":{\"text\":\"Conclusion\",\"level\":2},\"type\":\"header\"},{\"id\":\"p-conclusion-1\",\"data\":{\"text\":\"An AI Platform Architect is responsible for the reusable architecture \u003Cstrong>between AI capabilities and the solutions that consume them\u003C\u002Fstrong>. The role defines how models, providers, retrieval, agents, tools, identity, tenants, secrets, evaluation, observability, quotas and runtime operations become dependable platform services rather than repeated one-off integrations.\"},\"type\":\"paragraph\"},{\"id\":\"p-conclusion-2\",\"data\":{\"text\":\"The difficult part is not maximizing reuse. It is choosing the correct boundary. A strong platform standardizes mechanics, policy and operations where multiple consumers genuinely benefit, while preserving solution-specific data authority, business logic, security requirements and acceptance criteria.\"},\"type\":\"paragraph\"},{\"id\":\"p-conclusion-3\",\"data\":{\"text\":\"That distinction also explains the relationship with AI Solution Architecture: \u003Cstrong>the solution architect makes one AI-enabled system fit its purpose; the platform architect makes shared AI capabilities safe, reusable, operable and evolvable across many such systems.\u003C\u002Fstrong>\"},\"type\":\"paragraph\"},{\"id\":\"h-related\",\"data\":{\"text\":\"Related canonical knowledge\",\"level\":2},\"type\":\"header\"},{\"id\":\"p-related-1\",\"data\":{\"text\":\"This article sits after the canonical foundations on generative AI components, ADR versus NFR, and AI Solution Architecture. Those concepts are prerequisites because a platform exists to provide reusable system capabilities and to encode architectural decisions against explicit quality and operational requirements.\"},\"type\":\"paragraph\"},{\"id\":\"p-related-2\",\"data\":{\"text\":\"Retrieval-Augmented Generation is one example of a capability that may be offered through a platform, but the platform should not collapse retrieval infrastructure, domain knowledge and answer validity into one concept.\"},\"type\":\"paragraph\"},{\"id\":\"related-rag\",\"data\":{\"link\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"What Is RAG? The Simplest Explanation of How It Works\",\"description\":\"Canonical introduction to retrieval-augmented generation and the boundary between model generation and external knowledge retrieval.\"}},\"type\":\"linkTool\"},{\"id\":\"p-related-3\",\"data\":{\"text\":\"Agent protocols, tenant isolation, AI governance, model routing, Context Engineering and MLOps\u002FLLMOps are downstream or adjacent knowledge nodes. They become easier to reason about once the platform boundary is explicit.\"},\"type\":\"paragraph\"},{\"id\":\"h-faq\",\"data\":{\"text\":\"Frequently asked questions\",\"level\":2},\"type\":\"header\"},{\"id\":\"faq\",\"data\":{\"items\":[{\"id\":\"faq-1\",\"answer\":\"No. The solution architect focuses on one concrete AI-enabled solution. The platform architect focuses on reusable AI capabilities, controls and operational contracts that can support multiple solutions.\",\"question\":\"Is an AI Platform Architect the same as an AI Solution Architect?\"},{\"id\":\"faq-2\",\"answer\":\"No. A platform can use managed cloud models, self-hosted models, local inference or a hybrid strategy. The architecture must make provider, locality, identity, routing, data and operational consequences explicit.\",\"question\":\"Does an AI platform need to host its own models?\"},{\"id\":\"faq-3\",\"answer\":\"Usually not. A gateway can be an important platform component, but a complete platform also needs contracts for identity, secrets, data\u002Fretrieval, evaluation, observability, lifecycle and operational ownership.\",\"question\":\"Is an AI gateway enough to be an AI platform?\"},{\"id\":\"faq-4\",\"answer\":\"Retrieval mechanics can often be shared, but domain authority, authorization, freshness, evidence sufficiency and corpus ownership should remain explicit. Shared infrastructure does not imply shared truth.\",\"question\":\"Should retrieval be centralized?\"},{\"id\":\"faq-5\",\"answer\":\"No. Platform evaluation can test shared capabilities and regressions. Each solution still needs task-specific ground truth, acceptance criteria and domain quality thresholds.\",\"question\":\"Does platform evaluation replace application evaluation?\"},{\"id\":\"faq-6\",\"answer\":\"No. RBAC determines what an identity may do. Tenant isolation determines which tenant's resources the identity may act on. A platform often needs both.\",\"question\":\"Is multi-tenancy just RBAC?\"}],\"title\":\"AI Platform Architect FAQ\"},\"type\":\"faq\"},{\"id\":\"h-glossary\",\"data\":{\"text\":\"Glossary\",\"level\":2},\"type\":\"header\"},{\"id\":\"glossary\",\"data\":{\"title\":\"Key AI platform architecture terms\",\"entries\":[{\"term\":\"AI platform\",\"anchor\":\"ai-platform\",\"definition\":\"A reusable set of AI-related technical and operational capabilities consumed by multiple applications, teams or tenant contexts.\"},{\"term\":\"AI gateway\",\"anchor\":\"ai-gateway\",\"definition\":\"A gateway layer for AI endpoints that may add authentication, routing, quotas, policy, retries, cost attribution and AI-specific telemetry beyond basic proxying.\"},{\"term\":\"Provider adapter\",\"anchor\":\"provider-adapter\",\"definition\":\"A component that maps a platform contract to a model provider's API, capabilities, health and failure semantics.\"},{\"term\":\"Tenant isolation\",\"anchor\":\"tenant-isolation\",\"definition\":\"The boundary that prevents one tenant context from accessing another tenant's resources, independent of role permissions.\"},{\"term\":\"Capability contract\",\"anchor\":\"capability-contract\",\"definition\":\"A versioned interface and behavioral agreement describing what a shared platform service provides and what the consumer must supply or own.\"},{\"term\":\"Grounding \u002F retrieval service\",\"anchor\":\"grounding-service\",\"definition\":\"Shared mechanics for finding and supplying external information to an AI workload; it does not automatically define which information is authoritative for a domain.\"},{\"term\":\"Evaluation harness\",\"anchor\":\"evaluation-harness\",\"definition\":\"Reusable infrastructure for running tests, datasets, model\u002Fprompt versions and metrics; domain acceptance remains solution-specific.\"},{\"term\":\"Control plane\",\"anchor\":\"control-plane\",\"definition\":\"The configuration and governance layer that manages platform capabilities, identities, policies, quotas, versions and deployment state.\"}]},\"type\":\"glossary\"},{\"id\":\"h-sources\",\"data\":{\"text\":\"Primary sources and current architecture guidance\",\"level\":2},\"type\":\"header\"},{\"id\":\"p-sources-note\",\"data\":{\"text\":\"The sources below support the general architecture and production-platform claims. The Aaasaasa AI Client, Aaasaasa AI CMS and Source of Truth Research Engine sections are explicitly original implementation evidence. Current-state external references were checked on 8 October 2026.\"},\"type\":\"paragraph\"},{\"id\":\"src-iso-42010\",\"data\":{\"link\":\"https:\u002F\u002Fwww.iso.org\u002Fstandard\u002F74393.html\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"ISO\u002FIEC\u002FIEEE 42010:2022 — Architecture Description\",\"description\":\"Current published international standard for architecture-description concepts and relationships.\"}},\"type\":\"linkTool\"},{\"id\":\"src-nist-rmf\",\"data\":{\"link\":\"https:\u002F\u002Fwww.nist.gov\u002Fitl\u002Fai-risk-management-framework\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"NIST AI Risk Management Framework\",\"description\":\"NIST's AI RMF resources and current status; AI RMF 1.0 is under revision as of October 2026.\"}},\"type\":\"linkTool\"},{\"id\":\"src-nist-gai\",\"data\":{\"link\":\"https:\u002F\u002Fwww.nist.gov\u002Fpublications\u002Fartificial-intelligence-risk-management-framework-generative-artificial-intelligence\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"NIST AI 600-1 — Generative AI Profile\",\"description\":\"Generative AI profile for applying AI risk-management considerations across the AI lifecycle.\"}},\"type\":\"linkTool\"},{\"id\":\"src-ms-ai\",\"data\":{\"link\":\"https:\u002F\u002Flearn.microsoft.com\u002Fen-us\u002Fazure\u002Fwell-architected\u002Fai\u002Fget-started\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Microsoft Azure Well-Architected — AI Workloads\",\"description\":\"Current architectural guidance covering AI application, data, operations, evaluation, responsible AI and lifecycle concerns.\"}},\"type\":\"linkTool\"},{\"id\":\"src-ms-principles\",\"data\":{\"link\":\"https:\u002F\u002Flearn.microsoft.com\u002Fen-us\u002Fazure\u002Fwell-architected\u002Fai\u002Fdesign-principles\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Microsoft — Design Principles for AI Workloads\",\"description\":\"Current guidance on identity segmentation, security boundaries, telemetry, performance, data and platform trade-offs.\"}},\"type\":\"linkTool\"},{\"id\":\"src-ms-gateway\",\"data\":{\"link\":\"https:\u002F\u002Flearn.microsoft.com\u002Fen-us\u002Fazure\u002Fai-foundry\u002Fconfiguration\u002Fenable-ai-api-management-gateway-portal?view=foundry\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Microsoft Foundry — AI Gateway Architecture\",\"description\":\"Current AI Gateway guidance for shared project access, token containment, quotas and governance.\"}},\"type\":\"linkTool\"},{\"id\":\"src-ms-gateway-guide\",\"data\":{\"link\":\"https:\u002F\u002Flearn.microsoft.com\u002Fen-us\u002Fazure\u002Farchitecture\u002Fai-ml\u002Fguide\u002Fazure-openai-gateway-guide\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Azure Architecture Center — Access Models Through a Gateway\",\"description\":\"Architecture guidance for centralized model access, routing, throttling, failover and client\u002Fplatform responsibilities.\"}},\"type\":\"linkTool\"},{\"id\":\"src-aws-genai\",\"data\":{\"link\":\"https:\u002F\u002Fdocs.aws.amazon.com\u002Fwellarchitected\u002Flatest\u002Fgenerative-ai-lens\u002F\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"AWS Well-Architected — Generative AI Lens\",\"description\":\"Current production architecture guidance for generative AI workloads across security, reliability, operations, performance and cost.\"}},\"type\":\"linkTool\"},{\"id\":\"src-aws-multitenant\",\"data\":{\"link\":\"https:\u002F\u002Fdocs.aws.amazon.com\u002Fwellarchitected\u002Flatest\u002Fgenerative-ai-lens\u002Fmulti-tenant-generative-ai-platform-scenario.html\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"AWS — Multi-tenant Generative AI Platform Scenario\",\"description\":\"Current example separating central platform controls and auditability from consuming-application data quality and workload-specific responsibilities.\"}},\"type\":\"linkTool\"},{\"id\":\"src-aws-agentic\",\"data\":{\"link\":\"https:\u002F\u002Fdocs.aws.amazon.com\u002Fwellarchitected\u002Flatest\u002Fagentic-ai-lens\u002Fdesign-principles.html\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"AWS Well-Architected — Agentic AI Design Principles\",\"description\":\"Current guidance on bounded agent authority, traceability, versioned behavior, explicit contracts and human oversight.\"}},\"type\":\"linkTool\"},{\"id\":\"src-aws-observability\",\"data\":{\"link\":\"https:\u002F\u002Fdocs.aws.amazon.com\u002FAmazonCloudWatch\u002Flatest\u002Fmonitoring\u002FGenAI-observability.html\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"AWS CloudWatch — Generative AI Observability\",\"description\":\"Current observability capabilities and production metrics for models, agents, knowledge bases, tools and cost\u002Flatency\u002Ferror analysis.\"}},\"type\":\"linkTool\"}],\"version\":\"2.31.0\"}",{"time":1666,"blocks":1667,"version":2431},1791476955677,[1668,1671,1675,1679,1683,1686,1689,1692,1695,1716,1719,1722,1725,1728,1750,1753,1756,1759,1762,1791,1795,1798,1801,1804,1807,1811,1814,1817,1820,1823,1826,1829,1832,1835,1838,1841,1844,1847,1850,1853,1856,1859,1862,1865,1868,1871,1874,1877,1880,1883,1886,1889,1892,1895,1898,1901,1905,1923,1926,1929,1962,1965,1990,1993,2012,2015,2018,2022,2025,2028,2031,2034,2055,2058,2061,2064,2067,2070,2073,2076,2079,2083,2086,2089,2092,2095,2098,2101,2104,2134,2137,2170,2173,2207,2210,2213,2216,2219,2222,2225,2228,2231,2234,2237,2279,2282,2285,2288,2291,2294,2297,2300,2307,2310,2313,2335,2338,2366,2369,2372,2378,2383,2388,2393,2398,2403,2408,2413,2419,2425],{"id":214,"data":1669,"type":217},{"text":1670},"An \u003Cstrong>AI Platform Architect\u003C\u002Fstrong> designs the reusable AI foundation through which multiple applications, teams, or tenant contexts access models, data and retrieval, agent and tool runtimes, identity and permissions, evaluation, observability, quotas, secrets, and deployment capabilities. The role is broader than infrastructure but narrower than owning every AI-enabled product: its central responsibility is deciding \u003Cstrong>what should be shared, how shared capabilities are governed and isolated, and what must remain solution-specific\u003C\u002Fstrong>.",{"id":219,"data":1672,"type":224},{"body":1673,"title":1674,"variant":223},"\u003Cstrong>An AI Platform Architect designs the shared technical and operational substrate for AI systems.\u003C\u002Fstrong> Instead of architecting one assistant or one workflow, the role defines reusable contracts and boundaries for model\u002Fprovider access, gateways and routing, retrieval services, agent runtimes, tool access, identity and tenant isolation, secrets, evaluation, telemetry, deployment and lifecycle management.","Direct answer",{"id":226,"data":1676,"type":224},{"body":1677,"title":1678,"variant":230},"\u003Cstrong>AI Platform Architect is a practical role label, not a universally standardized job title.\u003C\u002Fstrong> ISO\u002FIEC\u002FIEEE 42010:2022 defines concepts for architecture descriptions, not this role. Different organizations may split these responsibilities among platform architects, solution architects, enterprise architects, security architects, MLOps\u002FLLMOps specialists and platform engineering teams. This article uses the term for the architecture responsibility over a reusable AI platform layer.","Terminology note",{"id":232,"data":1680,"type":224},{"body":1681,"title":1682,"variant":230},"The stable architectural principles here are vendor-neutral. Current Microsoft, AWS and NIST guidance is used as external implementation and governance evidence. NIST states that AI RMF 1.0 is being revised; vendor platform features, gateway products, agent runtimes and model capabilities evolve faster than the architectural principles, so version-sensitive implementation choices must be rechecked before deployment.","Current-source note — 8 October 2026",{"id":237,"data":1684,"type":242},{"title":1685,"maxLevel":240,"minLevel":241},"Contents",{"id":244,"data":1687,"type":41},{"text":1688,"level":241},"What does an AI Platform Architect actually architect?",{"id":248,"data":1690,"type":217},{"text":1691},"The object of the work is the \u003Cstrong>platform\u003C\u002Fstrong>: a set of shared capabilities that reduces repeated integration work while preserving explicit security, data and operational boundaries. A platform can expose model access, provider adapters, retrieval primitives, agent execution, tool brokers, policy enforcement, evaluation, telemetry and deployment services to many consuming solutions.",{"id":252,"data":1693,"type":217},{"text":1694},"The platform is not valuable merely because components are centralized. It is valuable when consumers receive stable capabilities with clear contracts, ownership, isolation, observability and lifecycle rules. The key architectural question is therefore not “Which model should everyone use?” but \u003Cstrong>“Which responsibilities can be safely standardized and reused without erasing the requirements of each solution?”\u003C\u002Fstrong>.",{"id":256,"data":1696,"type":298},{"rows":1697,"title":1712,"layout":290,"columns":1713},[1698,1701,1704,1707,1710],{"id":260,"label":1699,"values":1700},"Primary scope",{"platform":263,"solution":264},{"id":266,"label":1702,"values":1703},"Main question",{"platform":269,"solution":270},{"id":272,"label":1705,"values":1706},"Data authority",{"platform":275,"solution":276},{"id":278,"label":1708,"values":1709},"Evaluation",{"platform":281,"solution":282},{"id":284,"label":285,"values":1711},{"platform":287,"solution":288},"Solution architecture and platform architecture solve different scope problems",[1714,1715],{"id":293,"label":294},{"id":296,"label":297},{"id":300,"data":1717,"type":41},{"text":1718,"level":241},"The simplest example",{"id":304,"data":1720,"type":217},{"text":1721},"Imagine an organization has five AI-enabled products: an internal document assistant, a customer-support copilot, a software-engineering agent, a contract review workflow and a product-search assistant. Each product could independently integrate model APIs, keep credentials, implement retries, collect token metrics, create retrieval code and build its own tool permissions.",{"id":308,"data":1723,"type":217},{"text":1724},"That duplication is expensive and dangerous when every team invents a different security and operational model. A shared platform can instead offer approved provider connections, model discovery, quotas, credentials, tenant-aware access, common telemetry, reusable retrieval services and an agent\u002Ftool runtime contract.",{"id":312,"data":1726,"type":217},{"text":1727},"But the platform must stop at the correct boundary. The contract-review solution may require legal-document authority and citation rules that the software agent does not. The product-search assistant may need commerce-specific freshness and authorization rules. \u003Cstrong>Reusable infrastructure does not make all domain truth reusable.\u003C\u002Fstrong>",{"id":316,"data":1729,"type":339},{"steps":1730,"title":1749,"orientation":338},[1731,1734,1737,1740,1743,1746],{"label":1732,"description":1733},"1. Consumer identifies itself","The calling application, user, service, team or tenant enters through an authenticated identity and explicit scope.",{"label":1735,"description":1736},"2. Platform policy applies","Gateway and policy layers determine allowed providers, models, quotas, data paths, tools and execution modes.",{"label":1738,"description":1739},"3. Shared capability executes","The request may use inference, retrieval, agent runtime, tool access or another reusable platform service.",{"label":1741,"description":1742},"4. Solution-specific context remains authoritative","The consuming solution supplies domain rules, user intent, data authority, task-specific constraints and acceptance logic.",{"label":1744,"description":1745},"5. Telemetry and evidence are captured","The platform records identity, route, model\u002Fprovider, latency, cost, errors, tool activity and other permitted observability signals.",{"label":1747,"description":1748},"6. Result returns under the solution contract","The solution remains responsible for whether the output is acceptable for its user and domain.","A shared AI request path",{"id":341,"data":1751,"type":41},{"text":1752,"level":241},"Where the simple example stops",{"id":345,"data":1754,"type":217},{"text":1755},"Centralization is not automatically architecture. A single endpoint in front of several model APIs is useful, but it does not by itself create an AI platform. A production platform also needs identity boundaries, capability contracts, provider health and lifecycle handling, quotas, secret ownership, observability, compatibility rules, security controls, release discipline and clear operational responsibility.",{"id":349,"data":1757,"type":217},{"text":1758},"The opposite failure is also common: putting every prompt, vector index, business rule, agent and application workflow into one “AI backend.” That creates a monolith whose shared status is accidental rather than architectural. \u003Cstrong>A platform should standardize cross-cutting capabilities, not absorb domain ownership merely because AI is involved.\u003C\u002Fstrong>",{"id":353,"data":1760,"type":41},{"text":1761,"level":241},"The most important platform decision: shared versus solution-specific",{"id":357,"data":1763,"type":290},{"content":1764,"stretched":42,"withHeadings":13},[1765,1769,1773,1776,1780,1784,1787],[1766,1767,1768],"Capability area","Good candidate for shared platform ownership","Usually remains solution-specific",[1770,1771,1772],"Model access","Approved provider connections, adapters, credentials, health, routing primitives, quotas","Task-specific model acceptance, prompt behavior, quality threshold",[369,1774,1775],"Ingestion primitives, extraction, indexing, search APIs, provenance contracts, authorization hooks","Authoritative corpus, freshness rules, domain metadata, evidence sufficiency",[1777,1778,1779],"Agents and tools","Runtime lifecycle, tool registry\u002Fbroker, permission enforcement, tracing, cancellation","Business workflow, allowed action semantics, escalation policy, task success",[1781,1782,1783],"Security","Identity integration, secret storage, policy enforcement, audit contracts, tenant isolation mechanisms","Data classification, business authorization rules, domain-specific risk acceptance",[1708,1785,1786],"Harness, dataset\u002Fversion mechanics, telemetry, experiment\u002Frelease workflow","Ground truth, domain test set, acceptance threshold, user outcome",[1788,1789,1790],"Operations","Deployment pattern, health, metrics, incident integration, capacity controls","Solution SLOs where they differ, business continuity impact, workload-specific runbooks",{"id":388,"data":1792,"type":224},{"body":1793,"title":1794,"variant":392},"\u003Cstrong>Share mechanics and controls where reuse is real; keep authority and acceptance where the domain owns them.\u003C\u002Fstrong> This prevents two opposite errors: duplicated infrastructure everywhere, and a central platform that falsely becomes the owner of every application's data, policy and quality.","Platform principle",{"id":394,"data":1796,"type":41},{"text":1797,"level":241},"Architecture responsibility map",{"id":398,"data":1799,"type":41},{"text":1800,"level":240},"1. Model and provider access",{"id":402,"data":1802,"type":217},{"text":1803},"A platform architect defines how consumers discover and invoke models without forcing every application to hard-code one provider. This includes provider adapters, model identifiers, capability metadata, authentication, health checks, endpoint configuration, request normalization and compatibility behavior.",{"id":406,"data":1805,"type":217},{"text":1806},"Provider abstraction must remain honest. Different providers expose different context limits, tool semantics, structured-output behavior, multimodal capabilities, safety controls, caching, pricing and failure modes. A good abstraction creates a stable platform contract while preserving access to capabilities that cannot be meaningfully flattened.",{"id":410,"data":1808,"type":224},{"body":1809,"title":1810,"variant":414},"A lowest-common-denominator API can make migration easier but can also erase capabilities that matter. The architecture should define which features are portable, which are provider-specific and how consumers discover that difference.","Do not confuse abstraction with pretending providers are identical",{"id":416,"data":1812,"type":41},{"text":1813,"level":240},"2. Gateway, routing, quotas and cost controls",{"id":420,"data":1815,"type":217},{"text":1816},"A shared AI gateway can centralize authentication, routing, throttling, retries, token limits, usage attribution and policy enforcement. Microsoft’s current AI Gateway guidance explicitly treats token-per-minute limits, quotas and multi-project containment as platform concerns; AWS likewise exposes account and model quotas and centralized controls.",{"id":424,"data":1818,"type":217},{"text":1819},"The gateway is therefore more than a reverse proxy when it carries AI-specific policy and operational semantics. But it should not silently make business decisions. A routing policy may prefer a healthy local model, a lower-cost provider or a regionally compliant endpoint; whether that route is acceptable for a particular task is still a contract between platform and solution.",{"id":428,"data":1821,"type":217},{"text":1822},"Routing also needs failure semantics. If the preferred model is unavailable, the platform must know whether fallback is permitted, whether a cloud route requires explicit consent, whether a lower-capability model is valid and how the decision is surfaced to observability.",{"id":432,"data":1824,"type":41},{"text":1825,"level":240},"3. Shared data, retrieval and grounding services",{"id":436,"data":1827,"type":217},{"text":1828},"Retrieval services are strong platform candidates because parsing, chunking, indexing, lexical search, semantic search, metadata filtering, provenance and citation mechanics are reusable. However, the platform must not confuse a shared retrieval engine with a shared source of truth.",{"id":440,"data":1830,"type":217},{"text":1831},"A solution still owns questions such as: Which corpus is authoritative? Which version is valid? Can this user see this document? How fresh must the data be? What counts as sufficient evidence? Can an answer be generated when retrieval fails? Those are domain and solution requirements even when the platform supplies the retrieval machinery.",{"id":444,"data":1833,"type":217},{"text":1834},"This boundary is especially important in multi-tenant systems. A technically shared index or vector service does not justify cross-tenant visibility. Authorization context must be preserved through retrieval, not added only after search results have already crossed the boundary.",{"id":448,"data":1836,"type":41},{"text":1837,"level":240},"4. Agent and tool runtime",{"id":452,"data":1839,"type":217},{"text":1840},"Agentic systems add reusable runtime concerns: thread\u002Fsession lifecycle, planning loops, tool registration, tool invocation, cancellation, timeouts, human approvals, memory\u002Fstate interfaces, remote-agent protocols and trace correlation. A platform can provide these mechanics so each product does not rebuild them.",{"id":456,"data":1842,"type":217},{"text":1843},"The platform must also keep tool permission separate from model capability. A model being capable of generating a shell command does not mean the runtime should allow shell execution. The permission boundary belongs to the application\u002Fruntime architecture and must be enforceable independently of the model.",{"id":460,"data":1845,"type":217},{"text":1846},"Current AWS Agentic AI guidance emphasizes bounded agents, explicit authority, end-to-end tracing, versioned behavioral artifacts and human oversight proportionate to consequence. Those are platform-enabling concerns, but the consuming solution still defines what actions are legitimate for its domain.",{"id":464,"data":1848,"type":41},{"text":1849,"level":240},"5. Identity, tenant isolation and authorization",{"id":468,"data":1851,"type":217},{"text":1852},"AI platforms often sit in front of high-value models, proprietary data and action-capable tools. Authentication is therefore only the beginning. The architecture must carry user, service, application and tenant context through every privileged operation that needs it.",{"id":472,"data":1854,"type":217},{"text":1855},"\u003Cstrong>RBAC and tenant isolation solve different problems.\u003C\u002Fstrong> RBAC answers what an identity may do; tenant isolation answers which tenant’s resources that identity may act on. A platform that checks roles but loses tenant context can still expose the wrong data.",{"id":476,"data":1857,"type":217},{"text":1858},"Microsoft’s current AI workload guidance explicitly recommends identity segmentation and authorization-aware access to content. AWS’s multi-tenant generative AI platform guidance similarly treats logical isolation, centralized controls and auditability as platform concerns.",{"id":480,"data":1860,"type":41},{"text":1861,"level":240},"6. Secrets, credentials and trust boundaries",{"id":484,"data":1863,"type":217},{"text":1864},"A platform should define who owns provider keys, remote bearer tokens, signing material and tool credentials, where they are stored, which process can access them, how they are rotated and whether they can ever reach a browser or untrusted renderer.",{"id":488,"data":1866,"type":217},{"text":1867},"This is an architectural boundary, not an implementation detail. If every consuming application copies provider credentials into its own configuration, the organization has duplicated both operational burden and blast radius. Centralization can reduce that risk only if the platform itself has narrower, auditable access paths.",{"id":492,"data":1869,"type":41},{"text":1870,"level":240},"7. Evaluation, observability and auditability",{"id":496,"data":1872,"type":217},{"text":1873},"A reusable platform can provide evaluation harnesses, trace IDs, model\u002Fprovider metadata, token and cost metrics, latency, error rates, prompt\u002Fmodel version linkage, agent\u002Ftool traces and controlled logging. AWS and Microsoft both treat observability and evaluation as core production concerns for AI workloads.",{"id":500,"data":1875,"type":217},{"text":1876},"Platform evaluation and solution evaluation must remain separate. A platform can verify that an endpoint is healthy, a model version passes a general regression suite and traces are complete. It cannot decide that a legal answer, medical workflow or product recommendation is acceptable without domain-specific ground truth and acceptance criteria.",{"id":504,"data":1878,"type":217},{"text":1879},"Logging also creates a privacy boundary. Prompt and response logs may contain sensitive or proprietary data. The platform architect must therefore decide what is logged, redacted, sampled, retained and accessible rather than assuming that more telemetry is always safer.",{"id":508,"data":1881,"type":41},{"text":1882,"level":240},"8. Runtime, deployment and locality",{"id":512,"data":1884,"type":217},{"text":1885},"A platform architect decides how shared AI capabilities are deployed and reached: managed cloud services, self-hosted endpoints, local inference, hybrid routing, containerized services, desktop runtimes, private networking or air-gapped environments. The important distinction is between \u003Cstrong>where the control\u002Fruntime process runs\u003C\u002Fstrong> and \u003Cstrong>where inference and data processing actually occur\u003C\u002Fstrong>.",{"id":516,"data":1887,"type":217},{"text":1888},"A local client may still call a cloud model. A cloud control plane may route to an on-premises model. A remote agent may execute tools inside a customer network. Architectural diagrams must therefore show trust and data-flow boundaries rather than using “local” and “cloud” as vague labels.",{"id":520,"data":1890,"type":41},{"text":1891,"level":240},"9. Platform lifecycle, compatibility and onboarding",{"id":524,"data":1893,"type":217},{"text":1894},"Reusable capability becomes a platform only when consumers can depend on it over time. That requires versioned contracts, migration rules, compatibility policy, deprecation, release testing, rollback, incident ownership, capacity planning, documentation and a path for onboarding new teams or applications.",{"id":528,"data":1896,"type":217},{"text":1897},"Fast-moving AI ecosystems make this particularly important. Model names, SDKs, protocol versions, provider APIs and safety capabilities change independently. A platform must absorb some of that volatility without hiding changes that materially affect a solution’s behavior.",{"id":532,"data":1899,"type":41},{"text":1900,"level":241},"A practical control-plane \u002F execution-plane \u002F solution-plane model",{"id":536,"data":1902,"type":224},{"body":1903,"title":1904,"variant":230},"The three-plane model below is a practical way to reason about responsibilities; it is not an ISO, NIST, Microsoft or AWS standard. Its purpose is to make ownership boundaries explicit.","Proposed architecture model",{"id":541,"data":1906,"type":290},{"content":1907,"stretched":42,"withHeadings":13},[1908,1911,1915,1919],[545,1909,1910],"Typical responsibilities","Should not silently own",[1912,1913,1914],"Platform control plane","Provider registry, model policy, quotas, tenant configuration, identities, secrets, routing rules, capability versions, deployment configuration","Application business logic or domain truth",[1916,1917,1918],"Platform execution\u002Fdata plane","Inference requests, retrieval operations, agent\u002Ftool execution, extraction, indexing, telemetry emission, policy enforcement","Cross-tenant access merely because infrastructure is shared",[1920,1921,1922],"Solution plane","User workflow, prompts\u002Finstructions, authoritative corpus selection, domain authorization, business rules, task evaluation and acceptance","Low-level provider integration that the platform explicitly owns",{"id":561,"data":1924,"type":217},{"text":1925},"This separation helps diagnose platform drift. If an application must know every provider-specific credential and endpoint, the platform contract is too thin. If the platform decides which customer record is legally authoritative or whether a domain answer is acceptable, the platform has crossed into solution ownership.",{"id":565,"data":1927,"type":41},{"text":1928,"level":241},"What should an AI Platform Architect produce?",{"id":569,"data":1930,"type":290},{"content":1931,"stretched":42,"withHeadings":13},[1932,1935,1938,1941,1944,1947,1950,1953,1956,1959],[1933,1934],"Architecture artifact","Purpose",[1936,1937],"Platform capability map","Defines what the platform provides, who consumes it and which capabilities remain outside scope.",[1939,1940],"Provider\u002Fmodel contract","Defines providers, models, capabilities, abstraction boundaries, route metadata and fallback semantics.",[1942,1943],"Identity and tenancy model","Defines user\u002Fservice\u002Fapplication identity, tenant context, RBAC\u002FABAC hooks and resource isolation.",[1945,1946],"Gateway and quota policy","Defines rate limits, token\u002Fcost budgets, routing controls, retries and capacity behavior.",[1948,1949],"Retrieval\u002Fdata contract","Defines ingestion, provenance, search, metadata, authorization propagation and where domain authority remains.",[1951,1952],"Agent\u002Ftool contract","Defines runtime lifecycle, tool registration, permissions, approvals, cancellation and trace behavior.",[1954,1955],"Secret and trust-boundary model","Defines credential ownership, storage, process boundaries, rotation and sensitive data paths.",[1957,1958],"Evaluation and telemetry contract","Defines common metrics, traces, datasets\u002Fversion links, logging policy and solution extension points.",[1960,1961],"Lifecycle and compatibility policy","Defines versions, migrations, deprecation, releases, rollback, incident ownership and onboarding.",{"id":603,"data":1963,"type":41},{"text":1964,"level":241},"The work is mostly trade-offs, not maximum centralization",{"id":607,"data":1966,"type":298},{"rows":1967,"title":1984,"layout":290,"columns":1985},[1968,1971,1974,1976,1978,1981],{"id":611,"label":1969,"values":1970},"Provider abstraction",{"pressureA":614,"pressureB":615},{"id":617,"label":1972,"values":1973},"Reuse",{"pressureA":620,"pressureB":621},{"id":623,"label":624,"values":1975},{"pressureA":626,"pressureB":627},{"id":629,"label":630,"values":1977},{"pressureA":632,"pressureB":633},{"id":635,"label":1979,"values":1980},"Availability",{"pressureA":638,"pressureB":639},{"id":641,"label":1982,"values":1983},"Platform scope",{"pressureA":644,"pressureB":645},"Common platform trade-offs",[1986,1988],{"id":649,"label":1987},"Pressure A",{"id":652,"label":1989},"Pressure B",{"id":655,"data":1991,"type":41},{"text":1992,"level":241},"How is this different from adjacent roles?",{"id":659,"data":1994,"type":290},{"content":1995,"stretched":42,"withHeadings":13},[1996,1999,2001,2003,2005,2008,2010],[1997,1998],"Role","Primary architectural scope",[294,2000],"A concrete AI-enabled solution and its end-to-end requirements, boundaries, trade-offs and production acceptance.",[297,2002],"Reusable AI capabilities and operational\u002Fsecurity contracts consumed across multiple solutions or teams.",[670,2004],"Organization-wide business\u002Ftechnology portfolio, capability and governance alignment at a broader level.",[2006,2007],"MLOps \u002F LLMOps Architect or specialist","Model and AI lifecycle, deployment, experiments, observability, release and operational practices; may overlap strongly but does not automatically own the whole shared application platform.",[676,2009],"Implements and operates platform infrastructure, reliability, automation and developer experience; architecture responsibility may be shared with the platform architect.",[679,2011],"Implements models, integrations, services, agents, retrieval and product functionality inside the agreed architecture.",{"id":682,"data":2013,"type":217},{"text":2014},"These boundaries are organizational, not universal. In a small team one person may hold several responsibilities. In a regulated enterprise they may be split across architecture, security, platform, data and operations groups. The useful distinction is the \u003Cstrong>scope of architectural responsibility\u003C\u002Fstrong>, not the job title printed on an org chart.",{"id":686,"data":2016,"type":41},{"text":2017,"level":241},"Implementation evidence: how these platform boundaries appear in my own work",{"id":690,"data":2019,"type":224},{"body":2020,"title":2021,"variant":230},"The following sections describe concrete patterns from my own projects. They are evidence that these architectural boundaries have been implemented or explicitly designed in real code and project systems. They are \u003Cstrong>not\u003C\u002Fstrong> claims that the projects together already constitute a commercially deployed enterprise AI platform.","Original implementation evidence",{"id":695,"data":2023,"type":41},{"text":2024,"level":240},"Aaasaasa AI Client: provider, runtime and permission separation",{"id":699,"data":2026,"type":217},{"text":2027},"Aaasaasa AI Client is a local-first desktop AI workspace built with Nuxt 4, Electron and TypeScript. Its AI Hub deliberately separates \u003Cstrong>agent\u002Fclient, provider, model, connection\u002Fruntime location, permissions and web client\u003C\u002Fstrong> instead of treating them as one configuration value.",{"id":703,"data":2029,"type":217},{"text":2030},"The implementation includes direct provider adapters, Codex agent runtime integration, local Ollama\u002FLM Studio paths, OpenAI-compatible services, centralized workspace permissions, main-process credential storage, DuckDB, Qdrant\u002Fvector support, PDF\u002Freadability extraction and authenticated MCP-based directory access.",{"id":707,"data":2032,"type":217},{"text":2033},"Two platform lessons are especially relevant. First, a local runtime is not the same as local inference: a local Codex process can still use a cloud model. Second, automatic routing does not silently fall back from local to paid cloud inference. That makes routing policy and runtime locality explicit rather than inferred from UI labels.",{"id":711,"data":2035,"type":290},{"content":2036,"stretched":42,"withHeadings":13},[2037,2040,2043,2046,2049,2052],[2038,2039],"Implemented boundary","Platform-architecture meaning",[2041,2042],"Agent vs provider vs model","Different responsibilities can evolve independently instead of being hidden behind one “AI” selector.",[2044,2045],"Permissions separate from model","Filesystem\u002Ftool authority belongs to the runtime policy, not model capability.",[2047,2048],"Main-process secrets","Credential ownership follows the privileged process boundary rather than the renderer\u002FUI.",[2050,2051],"Provider health and model discovery","Routing and availability are runtime\u002Fplatform concerns.",[2053,2054],"No silent cloud fallback","Cost, locality and data-transfer semantics remain explicit policy decisions.",{"id":733,"data":2056,"type":41},{"text":2057,"level":240},"Aaasaasa AI CMS: tenant-scoped authorization as a platform boundary",{"id":737,"data":2059,"type":217},{"text":2060},"The Aaasaasa AI CMS codebase provides a separate implementation example: tenant-scoped RBAC is represented through roles, permissions and user-role assignments bound to a tenant identifier. System permissions are grouped by capability, and role lookup and updates remain tenant-scoped.",{"id":741,"data":2062,"type":217},{"text":2063},"This is not itself proof of a complete AI platform, but it is directly relevant to one of the hardest shared-platform boundaries: a reusable service must preserve \u003Cstrong>who may do what\u003C\u002Fstrong> and \u003Cstrong>for which tenant\u003C\u002Fstrong>. Adding AI inference or retrieval on top of an application platform does not remove that requirement.",{"id":745,"data":2065,"type":217},{"text":2066},"The architectural implication is that model gateways, retrieval services and agents should consume established identity\u002Ftenant context rather than inventing a parallel AI-only authorization universe.",{"id":749,"data":2068,"type":41},{"text":2069,"level":240},"Source of Truth Research Engine: shared retrieval mechanics without shared truth",{"id":753,"data":2071,"type":217},{"text":2072},"The Source of Truth Research Engine provides a third implementation example. Different research modes share a common evidence core: Sources, Artifacts, provenance, Claims, Relations, Contradictions, a Reference Model and audit trail. The system also provides local lexical retrieval, optional semantic retrieval, extraction, snapshots and SHA-256-based provenance.",{"id":757,"data":2074,"type":217},{"text":2075},"The project explicitly treats search and semantic similarity as discovery signals rather than evidence. A result must be traced back to a concrete source and locator before it can support a claim. This is precisely the distinction an AI platform needs: \u003Cstrong>reusable retrieval machinery can be shared while evidence authority remains governed by the consuming methodology and domain.\u003C\u002Fstrong>",{"id":761,"data":2077,"type":217},{"text":2078},"The engine also demonstrates why one shared platform does not require one shared interpretation. Historical, scientific\u002Ftechnical, market-intelligence and monitoring modes can reuse core evidence infrastructure while retaining mode-specific methodology.",{"id":765,"data":2080,"type":224},{"body":2081,"title":2082,"variant":392},"Across these projects, the reusable pattern is not “one backend for everything.” It is \u003Cstrong>separation of concerns plus explicit contracts\u003C\u002Fstrong>: provider\u002Fmodel\u002Fruntime separation, tenant-aware authorization, credential boundaries, reusable data\u002Fretrieval primitives, provenance, and domain-specific authority. A future integrated platform would need stable contracts between those capabilities rather than direct coupling between codebases.","What these implementations demonstrate together",{"id":770,"data":2084,"type":41},{"text":2085,"level":241},"How current architecture guidance supports this platform scope",{"id":774,"data":2087,"type":217},{"text":2088},"ISO\u002FIEC\u002FIEEE 42010:2022 provides a general discipline for architecture descriptions across software, systems, enterprises and related entities. It does not define an AI Platform Architect, but it reinforces the need to express architectural concerns, relationships and viewpoints rather than reducing architecture to a technology list.",{"id":778,"data":2090,"type":217},{"text":2091},"NIST AI RMF 1.0 and the Generative AI Profile frame AI risk management across the lifecycle rather than only at model selection time. Governance, mapping, measurement and management are therefore compatible with a platform architecture that carries shared controls and evidence across many consuming workloads.",{"id":782,"data":2093,"type":217},{"text":2094},"Microsoft’s current AI workload guidance treats application design, data, security, operations, testing\u002Fevaluation and GenAIOps as connected architectural areas. Its current AI Gateway guidance also shows practical platform concerns such as centralized model access, project-specific token limits, quotas and multi-team containment.",{"id":786,"data":2096,"type":217},{"text":2097},"AWS’s current Generative AI Lens and multi-tenant platform scenario similarly separate foundational platform controls from consuming-application ownership. AWS explicitly notes that a central platform can enforce shared guardrails and auditability while data quality and workload-specific observability still remain responsibilities of consuming applications or data producers.",{"id":790,"data":2099,"type":217},{"text":2100},"The vendor products differ, but the cross-source pattern is stable: production AI platforms must coordinate identity, data access, models, policy, evaluation, observability, capacity, cost and lifecycle. A GPU cluster or model endpoint covers only part of that responsibility.",{"id":794,"data":2102,"type":41},{"text":2103,"level":241},"Common misconceptions",{"id":798,"data":2105,"type":290},{"content":2106,"stretched":42,"withHeadings":13},[2107,2110,2113,2116,2119,2122,2125,2128,2131],[2108,2109],"Misconception","Why it is wrong",[2111,2112],"“An AI platform is the GPU cluster.”","Compute is one substrate. A platform also needs contracts for identity, model access, data, policy, evaluation, observability and lifecycle.",[2114,2115],"“An AI gateway is just a reverse proxy.”","It may also carry model routing, token quotas, cost attribution, policy enforcement, identity and AI-specific telemetry.",[2117,2118],"“Shared means globally shared.”","A service may be physically shared while logically segmented by tenant, application, region, classification or risk level.",[2120,2121],"“One central vector database becomes the company truth.”","A vector store or retrieval service is infrastructure. Domain authority, freshness, provenance and access remain separate concerns.",[2123,2124],"“Platform evaluation replaces solution evaluation.”","General regression and telemetry cannot define whether a domain-specific answer or action is acceptable.",[2126,2127],"“Provider abstraction should hide every difference.”","Some differences are material capabilities, security semantics or failure modes and must remain visible.",[2129,2130],"“RBAC solves multi-tenancy.”","RBAC controls actions; tenant isolation controls resource boundaries. Both can be required.",[2132,2133],"“AI Platform Architect is just another name for MLOps.”","MLOps\u002FLLMOps is a major overlapping discipline, but shared application\u002Fruntime, identity, gateway, retrieval and tool boundaries can extend beyond model lifecycle operations.",{"id":829,"data":2135,"type":41},{"text":2136,"level":241},"Failure modes an AI Platform Architect should prevent",{"id":833,"data":2138,"type":290},{"content":2139,"stretched":42,"withHeadings":13},[2140,2143,2146,2149,2152,2155,2158,2161,2164,2167],[2141,2142],"Failure mode","Architectural consequence",[2144,2145],"Every team stores its own provider keys","Duplicated secret handling, inconsistent rotation and larger blast radius.",[2147,2148],"Provider abstraction hides required capabilities","Consumers cannot use features they need or silently receive behavior different from assumptions.",[2150,2151],"Shared retrieval ignores tenant\u002Fuser context","Cross-boundary data leakage can occur before the application gets a chance to filter results.",[2153,2154],"Fallback silently changes provider or locality","Cost, compliance, data location and output quality can change without the caller knowing.",[2156,2157],"Agent tools are granted by model choice","A capable model becomes over-privileged because runtime authority is not independently enforced.",[2159,2160],"All prompts\u002Fresponses are logged by default","Observability can create a new sensitive-data repository and compliance problem.",[2162,2163],"Platform owns one generic quality score","Domain failures remain hidden behind platform health metrics.",[2165,2166],"No version contract for platform capabilities","Model\u002Fprovider\u002Fruntime changes break consumers unpredictably.",[2168,2169],"Everything AI-related is centralized","The platform becomes a bottleneck and monolith instead of a reusable capability layer.",{"id":867,"data":2171,"type":41},{"text":2172,"level":241},"A practical platform-architecture decision sequence",{"id":871,"data":2174,"type":339},{"steps":2175,"title":2206,"orientation":338},[2176,2179,2182,2185,2188,2191,2194,2197,2200,2203],{"label":2177,"description":2178},"1. Identify real consumers","List solutions, teams, tenants and workloads that would consume the platform; avoid building a platform for hypothetical reuse.",{"label":2180,"description":2181},"2. Define the shared boundary","Separate cross-cutting mechanics from solution-specific domain authority, workflow and acceptance.",{"label":2183,"description":2184},"3. Define identity and isolation first","Establish users, services, applications, tenants, regions and data classifications before sharing retrieval or tool capabilities.",{"label":2186,"description":2187},"4. Define capability contracts","Specify model\u002Fprovider, retrieval, agent\u002Ftool, gateway and telemetry APIs with explicit ownership and versioning.",{"label":2189,"description":2190},"5. Decide provider and runtime strategy","Choose managed, self-hosted, local or hybrid execution and document fallback, locality and capability semantics.",{"label":2192,"description":2193},"6. Design data and retrieval boundaries","Define provenance, authorization propagation, corpus ownership, indexing and evidence responsibilities.",{"label":2195,"description":2196},"7. Add quotas, secrets and policy","Control cost, capacity, credentials, tool permissions, safety controls and blast radius.",{"label":2198,"description":2199},"8. Build evaluation and observability contracts","Provide platform metrics and tracing while leaving domain ground truth and acceptance to the solution.",{"label":2201,"description":2202},"9. Define lifecycle and operations","Version capabilities, test upgrades, document deprecation, rollback, incidents, capacity and consumer onboarding.",{"label":2204,"description":2205},"10. Validate with more than one consumer","A platform claim becomes credible when the shared capability actually serves distinct workloads without forcing them into the same domain model.","From platform need to operable shared capability",{"id":906,"data":2208,"type":41},{"text":2209,"level":241},"Edge cases and limits of the role",{"id":910,"data":2211,"type":217},{"text":2212},"A small organization with one AI application may not need a distinct AI platform or platform architect. Premature platforming can create more abstraction than value. The correct architecture may be one well-designed solution with a few reusable modules.",{"id":914,"data":2214,"type":217},{"text":2215},"An air-gapped or sovereign deployment changes the provider, update and observability model substantially. Model hosting, artifact distribution, identity integration and telemetry export may all need local equivalents.",{"id":918,"data":2217,"type":217},{"text":2218},"Highly regulated or high-consequence workloads may require stronger physical or organizational isolation instead of a logically shared platform. Reuse is never a sufficient reason to weaken a required security boundary.",{"id":922,"data":2220,"type":217},{"text":2221},"Managed cloud AI services can remove implementation burden but do not remove architectural accountability. The organization still decides identity, data access, logging, retention, quotas, model eligibility, fallback, evaluation and solution acceptance.",{"id":926,"data":2223,"type":217},{"text":2224},"The platform boundary may also differ by modality. Text inference, multimodal generation, speech, computer use and autonomous agents can have different latency, data, permission and observability requirements even when they share provider and identity infrastructure.",{"id":930,"data":2226,"type":41},{"text":2227,"level":241},"What would change this answer?",{"id":934,"data":2229,"type":217},{"text":2230},"The core definition would change if the organizational scope changes. If the architect owns one workload, the role becomes closer to AI Solution Architect. If the responsibility expands to organization-wide capability strategy, investment, standards and target-state portfolios, it moves toward Enterprise AI Architecture.",{"id":938,"data":2232,"type":217},{"text":2233},"Implementation guidance changes whenever providers, gateway products, agent protocols, regulatory obligations, model capabilities or deployment constraints change. That is why platform architecture should express stable responsibilities and contracts separately from current vendor mechanisms.",{"id":942,"data":2235,"type":41},{"text":2236,"level":241},"AI Platform Architect checklist",{"id":946,"data":2238,"type":290},{"content":2239,"stretched":42,"withHeadings":13},[2240,2243,2246,2249,2252,2255,2258,2261,2264,2267,2270,2273,2276],[2241,2242],"Question","Expected answer",[2244,2245],"Who are the actual platform consumers?","Named solutions, teams or tenant contexts with distinct but overlapping needs.",[2247,2248],"What is genuinely shared?","Explicit capability list, not a vague “AI backend.”",[2250,2251],"What must remain solution-specific?","Domain authority, business workflow, task acceptance and other workload-owned concerns.",[2253,2254],"How are models\u002Fproviders represented?","Versioned provider\u002Fmodel contracts with capabilities and explicit fallback semantics.",[2256,2257],"How is identity propagated?","User\u002Fservice\u002Fapplication\u002Ftenant context survives every privileged request path.",[2259,2260],"How is tenant isolation enforced?","Resource scoping is separate from role permission checks.",[2262,2263],"How are secrets handled?","Privileged storage, rotation, limited exposure and auditable ownership.",[2265,2266],"How does retrieval preserve authority?","Shared mechanics with authorization, provenance and domain-owned evidence rules.",[2268,2269],"How are tools and agents constrained?","Runtime permissions, bounded tool contracts, approvals, cancellation and traceability.",[2271,2272],"How are cost and capacity controlled?","Quotas, token\u002Frate controls, usage attribution and overload behavior.",[2274,2275],"How is quality measured?","Platform regression\u002Fevaluation plus solution-specific ground truth and acceptance.",[2277,2278],"How are changes rolled out?","Versioning, compatibility, migration, deprecation, rollback and incident ownership.",{"id":989,"data":2280,"type":41},{"text":2281,"level":241},"Conclusion",{"id":993,"data":2283,"type":217},{"text":2284},"An AI Platform Architect is responsible for the reusable architecture \u003Cstrong>between AI capabilities and the solutions that consume them\u003C\u002Fstrong>. The role defines how models, providers, retrieval, agents, tools, identity, tenants, secrets, evaluation, observability, quotas and runtime operations become dependable platform services rather than repeated one-off integrations.",{"id":997,"data":2286,"type":217},{"text":2287},"The difficult part is not maximizing reuse. It is choosing the correct boundary. A strong platform standardizes mechanics, policy and operations where multiple consumers genuinely benefit, while preserving solution-specific data authority, business logic, security requirements and acceptance criteria.",{"id":1001,"data":2289,"type":217},{"text":2290},"That distinction also explains the relationship with AI Solution Architecture: \u003Cstrong>the solution architect makes one AI-enabled system fit its purpose; the platform architect makes shared AI capabilities safe, reusable, operable and evolvable across many such systems.\u003C\u002Fstrong>",{"id":1005,"data":2292,"type":41},{"text":2293,"level":241},"Related canonical knowledge",{"id":1009,"data":2295,"type":217},{"text":2296},"This article sits after the canonical foundations on generative AI components, ADR versus NFR, and AI Solution Architecture. Those concepts are prerequisites because a platform exists to provide reusable system capabilities and to encode architectural decisions against explicit quality and operational requirements.",{"id":1013,"data":2298,"type":217},{"text":2299},"Retrieval-Augmented Generation is one example of a capability that may be offered through a platform, but the platform should not collapse retrieval infrastructure, domain knowledge and answer validity into one concept.",{"id":1017,"data":2301,"type":1025},{"link":2302,"meta":2303},"https:\u002F\u002Fstajic.de\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works",{"image":2304,"title":2305,"description":2306},{"url":1022},"What Is RAG? The Simplest Explanation of How It Works","Canonical introduction to retrieval-augmented generation and the boundary between model generation and external knowledge retrieval.",{"id":1027,"data":2308,"type":217},{"text":2309},"Agent protocols, tenant isolation, AI governance, model routing, Context Engineering and MLOps\u002FLLMOps are downstream or adjacent knowledge nodes. They become easier to reason about once the platform boundary is explicit.",{"id":1031,"data":2311,"type":41},{"text":2312,"level":241},"Frequently asked questions",{"id":1035,"data":2314,"type":1035},{"items":2315,"title":2334},[2316,2319,2322,2325,2328,2331],{"id":1039,"answer":2317,"question":2318},"No. The solution architect focuses on one concrete AI-enabled solution. The platform architect focuses on reusable AI capabilities, controls and operational contracts that can support multiple solutions.","Is an AI Platform Architect the same as an AI Solution Architect?",{"id":1043,"answer":2320,"question":2321},"No. A platform can use managed cloud models, self-hosted models, local inference or a hybrid strategy. The architecture must make provider, locality, identity, routing, data and operational consequences explicit.","Does an AI platform need to host its own models?",{"id":1047,"answer":2323,"question":2324},"Usually not. A gateway can be an important platform component, but a complete platform also needs contracts for identity, secrets, data\u002Fretrieval, evaluation, observability, lifecycle and operational ownership.","Is an AI gateway enough to be an AI platform?",{"id":1051,"answer":2326,"question":2327},"Retrieval mechanics can often be shared, but domain authority, authorization, freshness, evidence sufficiency and corpus ownership should remain explicit. Shared infrastructure does not imply shared truth.","Should retrieval be centralized?",{"id":1055,"answer":2329,"question":2330},"No. Platform evaluation can test shared capabilities and regressions. Each solution still needs task-specific ground truth, acceptance criteria and domain quality thresholds.","Does platform evaluation replace application evaluation?",{"id":1059,"answer":2332,"question":2333},"No. RBAC determines what an identity may do. Tenant isolation determines which tenant's resources the identity may act on. A platform often needs both.","Is multi-tenancy just RBAC?","AI Platform Architect FAQ",{"id":1064,"data":2336,"type":41},{"text":2337,"level":241},"Glossary",{"id":1068,"data":2339,"type":1068},{"title":2340,"entries":2341},"Key AI platform architecture terms",[2342,2345,2348,2351,2354,2357,2360,2363],{"term":2343,"anchor":1074,"definition":2344},"AI platform","A reusable set of AI-related technical and operational capabilities consumed by multiple applications, teams or tenant contexts.",{"term":2346,"anchor":1078,"definition":2347},"AI gateway","A gateway layer for AI endpoints that may add authentication, routing, quotas, policy, retries, cost attribution and AI-specific telemetry beyond basic proxying.",{"term":2349,"anchor":1082,"definition":2350},"Provider adapter","A component that maps a platform contract to a model provider's API, capabilities, health and failure semantics.",{"term":2352,"anchor":1086,"definition":2353},"Tenant isolation","The boundary that prevents one tenant context from accessing another tenant's resources, independent of role permissions.",{"term":2355,"anchor":1090,"definition":2356},"Capability contract","A versioned interface and behavioral agreement describing what a shared platform service provides and what the consumer must supply or own.",{"term":2358,"anchor":1094,"definition":2359},"Grounding \u002F retrieval service","Shared mechanics for finding and supplying external information to an AI workload; it does not automatically define which information is authoritative for a domain.",{"term":2361,"anchor":1098,"definition":2362},"Evaluation harness","Reusable infrastructure for running tests, datasets, model\u002Fprompt versions and metrics; domain acceptance remains solution-specific.",{"term":2364,"anchor":1102,"definition":2365},"Control plane","The configuration and governance layer that manages platform capabilities, identities, policies, quotas, versions and deployment state.",{"id":1105,"data":2367,"type":41},{"text":2368,"level":241},"Primary sources and current architecture guidance",{"id":1109,"data":2370,"type":217},{"text":2371},"The sources below support the general architecture and production-platform claims. The Aaasaasa AI Client, Aaasaasa AI CMS and Source of Truth Research Engine sections are explicitly original implementation evidence. Current-state external references were checked on 8 October 2026.",{"id":1113,"data":2373,"type":1025},{"link":1115,"meta":2374},{"image":2375,"title":2376,"description":2377},{"url":1022},"ISO\u002FIEC\u002FIEEE 42010:2022 — Architecture Description","Current published international standard for architecture-description concepts and relationships.",{"id":1121,"data":2379,"type":1025},{"link":1123,"meta":2380},{"image":2381,"title":1126,"description":2382},{"url":1022},"NIST's AI RMF resources and current status; AI RMF 1.0 is under revision as of October 2026.",{"id":1129,"data":2384,"type":1025},{"link":1131,"meta":2385},{"image":2386,"title":1134,"description":2387},{"url":1022},"Generative AI profile for applying AI risk-management considerations across the AI lifecycle.",{"id":1137,"data":2389,"type":1025},{"link":1139,"meta":2390},{"image":2391,"title":1142,"description":2392},{"url":1022},"Current architectural guidance covering AI application, data, operations, evaluation, responsible AI and lifecycle concerns.",{"id":1145,"data":2394,"type":1025},{"link":1147,"meta":2395},{"image":2396,"title":1150,"description":2397},{"url":1022},"Current guidance on identity segmentation, security boundaries, telemetry, performance, data and platform trade-offs.",{"id":1153,"data":2399,"type":1025},{"link":1155,"meta":2400},{"image":2401,"title":1158,"description":2402},{"url":1022},"Current AI Gateway guidance for shared project access, token containment, quotas and governance.",{"id":1161,"data":2404,"type":1025},{"link":1163,"meta":2405},{"image":2406,"title":1166,"description":2407},{"url":1022},"Architecture guidance for centralized model access, routing, throttling, failover and client\u002Fplatform responsibilities.",{"id":1169,"data":2409,"type":1025},{"link":1171,"meta":2410},{"image":2411,"title":1174,"description":2412},{"url":1022},"Current production architecture guidance for generative AI workloads across security, reliability, operations, performance and cost.",{"id":1177,"data":2414,"type":1025},{"link":1179,"meta":2415},{"image":2416,"title":2417,"description":2418},{"url":1022},"AWS — Multi-tenant Generative AI Platform Scenario","Current example separating central platform controls and auditability from consuming-application data quality and workload-specific responsibilities.",{"id":1185,"data":2420,"type":1025},{"link":1187,"meta":2421},{"image":2422,"title":2423,"description":2424},{"url":1022},"AWS Well-Architected — Agentic AI Design Principles","Current guidance on bounded agent authority, traceability, versioned behavior, explicit contracts and human oversight.",{"id":1193,"data":2426,"type":1025},{"link":1195,"meta":2427},{"image":2428,"title":2429,"description":2430},{"url":1022},"AWS CloudWatch — Generative AI Observability","Current observability capabilities and production metrics for models, agents, knowledge bases, tools and cost\u002Flatency\u002Ferror analysis.","2.31.0","An AI Platform Architect designs reusable AI foundations across models, providers, retrieval, agents, identity, security, evaluation, observability and operations.","Post erfolgreich abgerufen",{"items":2435,"source":2519,"manualIds":2520,"manualMatchedIds":2521},[2436,2443,2450,2457,2463,2470,2477,2484,2491,2498,2505,2512],{"id":2437,"slug":2438,"title":2439,"excerpt":2440,"featuredImage":2441,"publishedAt":2442},"476","mcp-vs-a2a-vs-ucp-vs-ap2-vs-a2ui-the-agent-protocol-stack-explained","MCP vs A2A vs UCP vs AP2 vs A2UI: Der Agenten-Protokoll-Stack erklärt","MCP, A2A, UCP, AP2 und A2UI werden oft als konkurrierende Agentenstandards dargestellt. Sie lösen größtenteils unterschiedliche Interoperabilitätsprobleme. Dieser Leitfaden ordnet jedes Protokoll der Grenze zu, die es tatsächlich standardisiert—und zeigt, wie sie in einem Produktionssystem zusammenarbeiten können.","\u002Fuploads\u002F2026\u002F09\u002Fmcp-vs-a2a-vs-ucp-vs-ap2-vs-a2ui-the-agent-protocol-stack-explained-1790352625869-2ezle0.webp","2026-09-25T12:09:00.000Z",{"id":2444,"slug":2445,"title":2446,"excerpt":2447,"featuredImage":2448,"publishedAt":2449},"490","rbac-vs-tenant-isolation-two-different-security-boundaries","RBAC vs. Mandantenisolierung: Zwei unterschiedliche Sicherheitsgrenzen","RBAC steuert, was ein Benutzer tun darf; Mandantenisolierung steuert, auf welche Ressourcen eines Mandanten diese Aktion zugreifen darf. Erfahren Sie, warum die Sicherheit von Multi-Tenant-SaaS beide Grenzen erfordert.","\u002Fuploads\u002F2026\u002F10\u002Frbac-vs-tenant-isolation-two-different-security-boundaries-1791485111528-qqtzby.webp","2026-10-08T14:43:00.000Z",{"id":2451,"slug":2452,"title":2453,"excerpt":2454,"featuredImage":2455,"publishedAt":2456},"486","source-of-truth-in-ai-systems-where-reliable-knowledge-actually-comes-from","Wahrheitsquelle in KI-Systemen: Woher verlässliches Wissen tatsächlich stammt","Eine Quelle der Wahrheit definiert, welche Quelle für einen bestimmten Fakt oder Zustand maßgeblich ist. Erfahren Sie, wie sie sich von RAG, Provenienz, Gedächtnis, Kontext, Vektordatenbanken und Systemen of Record unterscheidet.","\u002Fuploads\u002F2026\u002F10\u002Fsource-of-truth-in-ai-systems-where-reliable-knowledge-actually-comes-from-1791479103235-6bq9em.webp","2026-10-08T13:02:00.000Z",{"id":2458,"slug":2459,"title":1023,"excerpt":2460,"featuredImage":2461,"publishedAt":2462},"478","what-is-rag-the-simplest-explanation-of-how-it-works","RAG klingt kompliziert, aber die Idee ist einfach: Bevor eine KI antwortet, sucht sie zunächst nützliche Informationen aus einer Wissensquelle und gibt diese Informationen an das Sprachmodell weiter. Dieser Leitfaden erklärt RAG, LLMs, Zustand, Gedächtnis und Werkzeuge anhand eines einfachen mentalen Modells.","\u002Fuploads\u002F2026\u002F09\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works-1790377492124-khjagt.webp","2026-09-25T19:03:00.000Z",{"id":2464,"slug":2465,"title":2466,"excerpt":2467,"featuredImage":2468,"publishedAt":2469},"470","what-should-an-ai-agent-remember-forget-recompute-or-retrieve-again","Was sollte ein KI-Agent behalten, vergessen, neu berechnen oder erneut abrufen?","Langlaufende Agenten sollten sich nicht alles merken. Dieser Artikel bietet ein praktisches Lebenszyklusmodell für die Entscheidung, was in den dauerhaften Speicher gehört, was erneut abgerufen werden sollte, was sicherer neu zu berechnen ist und was ablaufen oder ersetzt werden sollte.","\u002Fuploads\u002F2026\u002F09\u002Fwhat-should-an-ai-agent-remember-forget-recompute-or-retrieve-again-1790351131087-iehz28.webp","2026-09-25T09:43:00.000Z",{"id":2471,"slug":2472,"title":2473,"excerpt":2474,"featuredImage":2475,"publishedAt":2476},"494","air-gapped-ai-how-ai-systems-work-without-internet-or-cloud-access","Luftgetrennte KI: Wie KI-Systeme ohne Internet- oder Cloud-Zugriff funktionieren","Air-gapped AI führt Modelle, RAG und KI-Anwendungen innerhalb einer isolierten Sicherheitsdomäne ohne Internet- oder Cloud-Abhängigkeiten aus. Erfahren Sie, wie Modelle, Daten, Updates und Tools offline funktionieren.","\u002Fuploads\u002F2026\u002F10\u002Fair-gapped-ai-how-ai-systems-work-without-internet-or-cloud-access-1791487983978-e6xqf0.webp","2026-10-08T11:32:00.000Z",{"id":2478,"slug":2479,"title":2480,"excerpt":2481,"featuredImage":2482,"publishedAt":2483},"381","enterprise-grade-multi-tenant-architecture-for-an-international-platform","Unternehmensfähige mandantenfähige Architektur für eine internationale Plattform","Loving Rocks ist eine Hochzeitsplattform auf Unternehmensniveau, konzipiert mit einer echten Mehrmandantenarchitektur, isolierten Datenbanken pro Mandant und integrierter Internationalisierung für globale Skalierbarkeit, Sicherheit und langfristige Betriebsstabilität.","\u002Fuploads\u002F2026\u002F01\u002Fenterprise-grade-multi-tenant-architecture-for-an-international-platform-1769789121298-b6v7ak.webp","2026-01-30T12:04:00.000Z",{"id":2485,"slug":2486,"title":2487,"excerpt":2488,"featuredImage":2489,"publishedAt":2490},"466","the-gpu-is-not-the-product-future-proof-private-ai-architecture","Die GPU ist nicht das Produkt: Zukunftssichere private KI-Architektur","Private KI-Infrastruktur sollte nicht um eine einzige GPU oder ein einziges Modell herum konzipiert werden. Ein resilienterer Ansatz kombiniert schnelle Inferenz-GPUs, speicherstarke KI-Systeme, physische KI-Knoten und optionale Frontier-Cloud-Modelle hinter einer fähigkeitsbewussten Routing-Schicht.","\u002Fuploads\u002F2026\u002F09\u002Fthe-gpu-is-not-the-product-future-proof-private-ai-architecture-1790140878812-8hsl39.webp","2026-09-23T01:19:00.000Z",{"id":2492,"slug":2493,"title":2494,"excerpt":2495,"featuredImage":2496,"publishedAt":2497},"487","vector-databases-embeddings-and-reranking-three-different-parts-of-retrieval","Vektordatenbanken, Embeddings und Reranking: Drei verschiedene Teile des Retrievals","Embeddings repräsentieren Bedeutung, Vektordatenbanken rufen Kandidaten ab und Reranker verfeinern Ergebnisse. Erfahren Sie, wie sich diese drei Retrieval-Ebenen unterscheiden und in RAG zusammenwirken.","\u002Fuploads\u002F2026\u002F10\u002Fvector-databases-embeddings-and-reranking-three-different-parts-of-retrieval-1791480129884-9dtasz.webp","2026-10-08T11:21:00.000Z",{"id":2499,"slug":2500,"title":2501,"excerpt":2502,"featuredImage":2503,"publishedAt":2504},"471","how-to-know-whether-an-ai-agent-actually-used-the-right-evidence","Wie man erkennt, ob ein KI-Agent tatsächlich die richtigen Belege verwendet hat","Ein KI-Agent kann Quellen zitieren und trotzdem die falschen Belege verwenden. Dieser Artikel stellt eine praktische Methode zur Überprüfung der Belegung von Behauptungen, der Quellenautorität, der Anwendbarkeit, der Herkunft sowie der Frage vor, ob die Belege die Antwort tatsächlich beeinflusst haben.","\u002Fuploads\u002F2026\u002F09\u002Fhow-to-know-whether-an-ai-agent-actually-used-the-right-evidence-1790351317188-o5z9ve.webp","2026-09-25T11:47:00.000Z",{"id":2506,"slug":2507,"title":2508,"excerpt":2509,"featuredImage":2510,"publishedAt":2511},"467","the-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers","Die Antwortgültigkeitsgrenze: Die fehlende Schicht zwischen Relevanz und zuverlässigen KI-Antworten","Eine Quelle kann relevant und maßgeblich sein und dennoch falsch für die gestellte Frage. Die fehlende Ebene ist die Anwendbarkeit: die Bedingungen, unter denen eine Antwort gilt, und die Veränderungen, die erzwingen, dass sie überdacht werden muss. Dieser Artikel führt die Answer Validity Boundary als ein Quellendesign-Muster für Menschen, KI-Suche und RAG-Systeme ein.","\u002Fuploads\u002F2026\u002F09\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers-1790272901306-1g5jly.webp","2026-09-24T11:59:00.000Z",{"id":2513,"slug":2514,"title":2515,"excerpt":2516,"featuredImage":2517,"publishedAt":2518},"489","agentic-ai-explained-when-an-ai-system-can-plan-use-tools-and-act","Agentische KI erklärt: Wenn ein KI-System planen, Werkzeuge nutzen und handeln kann","Agentische KI verwendet Modelle innerhalb mehrstufiger Ausführungsschleifen, in denen sie Werkzeuge auswählen, Ergebnisse beobachten, den Zustand aktualisieren und ihre nächste Aktion innerhalb expliziter Laufzeit- und Berechtigungsgrenzen anpassen können.","\u002Fuploads\u002F2026\u002F10\u002Fagentic-ai-explained-when-an-ai-system-can-plan-use-tools-and-act-1791481499084-wnji2a.webp","2026-10-08T11:43:00.000Z","fallback",[],[]]