[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"portal-settings:stajic:en":3,"public-menus:all":38,"post:generative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing:en":205,"related:post:generative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing:en:1":1537},{"statusCode":4,"data":5,"message":37},200,{"tenantId":6,"lang":7,"defaultLang":8,"siteUrl":9,"contactEmail":10,"brandName":11,"logoUrl":12,"siteName":11,"siteDescription":13,"ogImage":10,"robotsIndex":14,"socialLinks":10,"reservedSlugs":10,"seoPolicy":15},"stajic","en","de","https:\u002F\u002Fstajic.de",null,"Stajic Platform","\u002FLogo_Planet.svg","Stajic Portal",true,{"branding":16,"relatedContent":17,"crossDomainLinks":18},{"logoUrl":12},{"enabled":14},[19,22,25,28,31,34],{"url":20,"label":21,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Ffigure.rocks","figure.rocks",{"url":23,"label":24,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Floving.rocks","loving.rocks",{"url":26,"label":27,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.com","bazify.com",{"url":29,"label":30,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.de","bazify.de",{"url":32,"label":33,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.at","bazify.at",{"url":35,"label":36,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.ba","bazify.ba","Portal settings resolved",[39,45],{"id":40,"name":41,"location":42,"isActive":14,"isDefault":43,"items":44},1,"main-navigation","header",false,[],{"id":46,"name":47,"location":48,"isActive":14,"isDefault":14,"items":49},4,"main-menu","sidebar",[50,66,79,93,103,118,133],{"id":51,"title":52,"url":60,"target":61,"icon":62,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":64,"portfolioId":10,"children":65},"item-18",{"de":53,"en":54,"es":55,"fr":56,"it":54,"ru":57,"sr":58,"zh":59},"Startseite","Home","Inicio","Accueil","Главная","Почетна","首页","\u002Ffull-stack-web-developer-munich-performance-seo-and-maintainable-builds","_self","i-lucide-home","page",111,[],{"id":67,"title":68,"url":75,"target":61,"icon":76,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":77,"portfolioId":10,"children":78},"item-22",{"de":69,"en":69,"es":70,"fr":69,"it":71,"ru":72,"sr":73,"zh":74},"Vision","Visión","Visione","Видение","Визија","想象","\u002Fueber-uns-webdesign-muenchen-webaplikation","i-lucide-eye",113,[],{"id":80,"title":81,"url":89,"target":61,"icon":90,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":91,"portfolioId":10,"children":92},"item-19",{"de":82,"en":83,"es":84,"fr":83,"it":85,"ru":86,"sr":87,"zh":88},"Leistungen","Services","Servicios","Servizi","Услуги","Услуге","服务","\u002Fservices-dienstleistungen-muenchen","i-lucide-wrench",116,[],{"id":94,"title":95,"url":99,"target":61,"icon":100,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":101,"portfolioId":10,"children":102},"item-23",{"de":96,"en":96,"es":96,"fr":96,"it":96,"ru":97,"sr":97,"zh":98},"Blog","Блог","博客","\u002Fblog","i-lucide-book-open",112,[],{"id":104,"title":105,"url":114,"target":61,"icon":115,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":116,"portfolioId":10,"children":117},"item-32",{"de":106,"en":107,"es":108,"fr":109,"it":110,"ru":111,"sr":112,"zh":113},"Neue Technologien","New Technologies","Nuevas tecnologías","Nouvelles technologies","Nuove tecnologie","Новые технологии","Нове технологије","新技术！","\u002Fneue-webtechnologien","i-lucide-sparkles",122,[],{"id":119,"title":120,"url":129,"target":61,"icon":130,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":131,"portfolioId":10,"children":132},"item-20",{"de":121,"en":122,"es":123,"fr":124,"it":125,"ru":126,"sr":127,"zh":128},"Kontakt","Contact us!","Contacto","Contact","Contatto","Контакт","Контактирајте нас","联系我们！","\u002Fcontact","i-lucide-mail",115,[],{"id":134,"title":135,"url":144,"target":61,"icon":145,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":147},"item-21",{"de":136,"en":137,"es":138,"fr":139,"it":140,"ru":141,"sr":142,"zh":143},"Unsere Arbeit","Our Work","Nuestro trabajo","Nos réalisations","I nostri lavori","Наши работы","Наши радови","文件夹","\u002Fportfolio","i-lucide-briefcase",114,[148,161,175,181,193],{"id":149,"title":150,"url":144,"target":61,"icon":159,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":160},"item-24",{"de":151,"en":152,"es":153,"fr":154,"it":155,"ru":156,"sr":157,"zh":158},"Alle Projekte","All Projects","Todos los proyectos","Tous les projets","Tutti i progetti","Все проекты","Сви пројекти","所有项目","i-lucide-grid-3x3",[],{"id":162,"title":163,"url":171,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":174},"item-29",{"de":164,"en":165,"es":166,"fr":167,"it":168,"ru":169,"sr":170,"zh":143},"Local Roots, Global Reach","Local Roots - Global Reach","Empresa local ","Entreprise locale","Azienda locale","Местная компания","Локално предузеће глобално тржиште","\u002Fportfolio\u002Flocal-roots-global-reach-communication-media-systems-for-modern-business","i-lucide-folder","custom",[],{"id":176,"title":177,"url":179,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":180},"item-28",{"de":178,"en":178,"es":178,"fr":178,"it":178,"ru":178,"sr":178,"zh":178},"Solr Suggester","\u002Fportfolio\u002Fsolr-fuzzy-suggester-und-solr-infix-suggester-abfrage-ueber-ajax-und-filterung",[],{"id":182,"title":183,"url":191,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":192},"item-27",{"de":184,"en":185,"es":186,"fr":187,"it":188,"ru":189,"sr":190,"zh":185},"Firmenwebseite SEO","Company Website SEO","Sitio web corporativo SEO","Site web d’entreprise SEO","Sito web aziendale SEO","Корпоративный сайт SEO","Пословна веб-страница SEO","\u002Fportfolio\u002Fseo-sem-branding-mobile-webseite-muenchen",[],{"id":194,"title":195,"url":203,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":204},"item-31",{"de":196,"en":197,"es":198,"fr":199,"it":200,"ru":201,"sr":202,"zh":197},"Digitalisierungsportal","Digitalization Portal","Portal de digitalización","Portail de numérisation","Portale di digitalizzazione","Портал цифровизации","Портал за дигитализацију","\u002Fportfolio\u002Fdigitalisierungsportal-archiv-museum-bibliothek-ead-lido-mets-mods",[],{"statusCode":4,"data":206,"message":1536},{"id":207,"title":208,"slug":209,"content":210,"contentJson":211,"excerpt":1065,"featuredImage":1066,"featuredImageAlt":1067,"featuredImageCaption":10,"featuredImageTitle":10,"featuredImageCopyright":10,"featuredImageAuthor":10,"featuredImageSourceUrl":10,"featuredImageLicense":10,"featuredImageIsAiGenerated":43,"status":1068,"publishedAt":1069,"createdAt":1070,"updatedAt":1071,"seoLocalePaths":1072,"categories":1081,"author":1102,"translations":1107},"481","Generative AI Explained: Models, Retrieval, Tools and Applications Are Not the Same Thing","generative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing","{\"time\":1791475413504,\"blocks\":[{\"id\":\"intro\",\"type\":\"paragraph\",\"data\":{\"text\":\"Generative AI is not one component. A production generative AI system usually combines a generative model with application code that supplies instructions and context, retrieves external knowledge when needed, exposes tools for reading or changing external systems, manages runtime state and permissions, and turns the result into a usable product. Treating the model, retrieval, tools, context, runtime, and application as the same thing hides the boundaries that determine freshness, security, reliability, cost, and control.\"},\"tunes\":{}},{\"id\":\"direct\",\"type\":\"callout\",\"data\":{\"variant\":\"info\",\"title\":\"Direct answer\",\"body\":\"\u003Cstrong>The model generates; retrieval finds external evidence; tools access data or perform actions; context is what the model can see for the current inference; the runtime coordinates execution; the application owns product rules, state, permissions, persistence, and user experience.\u003C\u002Fstrong> These layers can be packaged together by a vendor, but their responsibilities remain different.\"},\"tunes\":{}},{\"id\":\"scope-note\",\"type\":\"callout\",\"data\":{\"variant\":\"note\",\"title\":\"Terminology and version note\",\"body\":\"This article defines durable architectural responsibilities rather than one vendor stack. Current implementation examples were re-checked on \u003Cstrong>8 October 2026\u003C\u002Fstrong>. Vendor APIs and product names can change; the responsibility boundaries are more stable than any individual SDK or endpoint.\"},\"tunes\":{}},{\"id\":\"toc\",\"type\":\"tableOfContents\",\"data\":{\"title\":\"Contents\",\"minLevel\":2,\"maxLevel\":3},\"tunes\":{}},{\"id\":\"h-meaning\",\"type\":\"header\",\"data\":{\"text\":\"What does “generative AI” actually mean?\",\"level\":2},\"tunes\":{}},{\"id\":\"p-meaning-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"At the model level, generative AI refers to AI models that generate derived synthetic content such as text, images, audio, video, code, or other digital output. NIST AI 600-1 uses this model-oriented meaning and separately discusses risks at model, system, application, and use-case levels.\"},\"tunes\":{}},{\"id\":\"p-meaning-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"That distinction matters because an AI model is not the same thing as the complete AI system. NIST's current glossary defines an AI model as a component that produces outputs from inputs using computational, statistical, or machine-learning techniques, while an AI system can include software, hardware, applications, tools, or utilities that operate using AI.\"},\"tunes\":{}},{\"id\":\"model-system-rule\",\"type\":\"callout\",\"data\":{\"variant\":\"success\",\"title\":\"A useful boundary\",\"body\":\"\u003Cstrong>Generative model ≠ generative AI application.\u003C\u002Fstrong>\u003Cbr>A model is one computational component. A usable AI product is a system built around that component.\"},\"tunes\":{}},{\"id\":\"h-simple\",\"type\":\"header\",\"data\":{\"text\":\"The simplest useful model of a generative AI system\",\"level\":2},\"tunes\":{}},{\"id\":\"p-simple-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"For a first mental model, imagine a company assistant answering: “Can this customer receive a refund today?” A useful answer may require several different responsibilities. The language model can interpret the question and write the explanation, but the current order state may come from a database tool, the refund policy may come from document retrieval, permissions may be enforced by the application, and the final action may require a controlled API call.\"},\"tunes\":{}},{\"id\":\"simple-flow\",\"type\":\"processFlow\",\"data\":{\"title\":\"One common execution path\",\"orientation\":\"auto\",\"steps\":[{\"label\":\"1. User request\",\"description\":\"The application receives a natural-language question or task.\"},{\"label\":\"2. Application policy and state\",\"description\":\"Identity, tenant, permissions, current workflow state, and product rules define what the request is allowed to do.\"},{\"label\":\"3. Retrieval or direct data access\",\"description\":\"The system obtains external evidence or current facts when model knowledge is insufficient.\"},{\"label\":\"4. Context construction\",\"description\":\"Instructions, user input, selected evidence, relevant state, and tool definitions are assembled for the model.\"},{\"label\":\"5. Model inference\",\"description\":\"The generative model interprets the supplied context and produces text, structured output, or a tool request.\"},{\"label\":\"6. Tool execution when needed\",\"description\":\"The runtime or application validates and executes approved tool calls outside the model.\"},{\"label\":\"7. Observation and continuation\",\"description\":\"Tool results can return to the model as new context for another inference step.\"},{\"label\":\"8. Validation and product output\",\"description\":\"The application validates the result, records required state or audit data, and presents or executes the final outcome.\"}]},\"tunes\":{}},{\"id\":\"p-simple-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Real systems do not always follow this sequence exactly. Retrieval can happen before the first model call, tools can be selected during an agent loop, deterministic application logic can bypass the model entirely, and validation can occur at several stages. The point is to separate responsibilities, not to impose one universal workflow.\"},\"tunes\":{}},{\"id\":\"h-boundaries\",\"type\":\"header\",\"data\":{\"text\":\"The six boundaries that matter\",\"level\":2},\"tunes\":{}},{\"id\":\"boundary-comparison\",\"type\":\"comparison\",\"data\":{\"title\":\"Six responsibilities inside one AI product\",\"layout\":\"table\",\"columns\":[{\"id\":\"job\",\"label\":\"Primary job\"},{\"id\":\"input\",\"label\":\"Typical inputs\"},{\"id\":\"not\",\"label\":\"Not the same as\"}],\"rows\":[{\"id\":\"model\",\"label\":\"Model\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"retrieval\",\"label\":\"Retrieval\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"tools\",\"label\":\"Tools\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"context\",\"label\":\"Context\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"runtime\",\"label\":\"Runtime \u002F orchestrator\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"application\",\"label\":\"Application\",\"values\":[\"\",\"\",\"\"]}]},\"tunes\":{}},{\"id\":\"h-model\",\"type\":\"header\",\"data\":{\"text\":\"1. The model: generation is its core responsibility\",\"level\":2},\"tunes\":{}},{\"id\":\"p-model-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A generative model maps supplied inputs to generated outputs. For a language model, that can include natural-language text, structured JSON, code, classifications, summaries, plans, or tool-call arguments. Multimodal generative models can work with additional input and output types.\"},\"tunes\":{}},{\"id\":\"p-model-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The model can contain substantial learned knowledge in its parameters, but parameterized knowledge is not a live database. The model does not automatically know a document created five minutes ago, the current stock level, a private customer record, or the state of an application unless that information is supplied through the current input path.\"},\"tunes\":{}},{\"id\":\"p-model-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"This is why changing the model does not automatically solve stale knowledge, missing permissions, broken retrieval, incorrect state ownership, or unsafe tool execution. Those failures often belong to other layers.\"},\"tunes\":{}},{\"id\":\"h-retrieval\",\"type\":\"header\",\"data\":{\"text\":\"2. Retrieval: finding external evidence is a separate operation\",\"level\":2},\"tunes\":{}},{\"id\":\"p-retrieval-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Retrieval selects information from an external source before or during generation. The 2020 Retrieval-Augmented Generation work by Lewis et al. made the separation explicit by combining a parametric generative model with retrieved non-parametric memory. Modern production systems use many retrieval variants, but the architectural idea remains: useful evidence can be fetched at inference time instead of relying only on what the model learned during training.\"},\"tunes\":{}},{\"id\":\"p-retrieval-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Retrieval can use lexical search, embeddings, vector search, hybrid search, SQL, knowledge graphs, metadata filters, APIs, or other selection mechanisms. A vector database is therefore one possible retrieval component, not the definition of RAG.\"},\"tunes\":{}},{\"id\":\"retrieval-rule\",\"type\":\"callout\",\"data\":{\"variant\":\"warning\",\"title\":\"Relevance is not authority\",\"body\":\"A retrieved passage can be highly relevant and still be stale, unauthorized, from the wrong version, or insufficient to support a claim. Retrieval quality and evidence quality must be evaluated separately.\"},\"tunes\":{}},{\"id\":\"ref-rag\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works\",\"title\":\"What Is RAG? The Simplest Explanation of How It Works\",\"excerpt\":\"The canonical plain-English explanation of retrieval-augmented generation, including the separation between LLM, knowledge, state, memory and tools.\",\"ctaLabel\":\"Read the RAG foundation\"},\"tunes\":{}},{\"id\":\"h-tools\",\"type\":\"header\",\"data\":{\"text\":\"3. Tools: access and action are not model knowledge\",\"level\":2},\"tunes\":{}},{\"id\":\"p-tools-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A tool is an interface through which an AI runtime can request functionality outside the model. A tool can query a database, search the web, read a file, calculate a value, call an internal service, create a ticket, send a message, modify a record, or trigger another controlled operation.\"},\"tunes\":{}},{\"id\":\"p-tools-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"OpenAI's current function-calling documentation makes this boundary explicit: function calling lets models interface with external systems and access data or actions provided by the application. The model can propose or select a call, but the external system performs the real operation.\"},\"tunes\":{}},{\"id\":\"p-tools-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"Tool use therefore creates two separate questions: Can the model request this capability? and Will the application authorize and execute it? A production system should not confuse model intent with permission to cause a side effect.\"},\"tunes\":{}},{\"id\":\"tool-rule\",\"type\":\"callout\",\"data\":{\"variant\":\"success\",\"title\":\"Model intent is not execution authority\",\"body\":\"A model can emit a valid tool request and still be denied. Authorization, argument validation, rate limits, transaction rules, audit requirements, and rollback belong outside the model.\"},\"tunes\":{}},{\"id\":\"h-context\",\"type\":\"header\",\"data\":{\"text\":\"4. Context: what the model can see right now\",\"level\":2},\"tunes\":{}},{\"id\":\"p-context-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Context is the information available to the model for a particular inference step. Anthropic's context-engineering guidance describes context as the set of tokens included when sampling from an LLM. In practice, that set can contain system instructions, user messages, conversation history, retrieved evidence, tool definitions, tool results, memory summaries, and selected application state.\"},\"tunes\":{}},{\"id\":\"p-context-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Context is therefore neither the complete knowledge base nor long-term memory. A company may store ten million documents while only a handful of passages enter one model call. A runtime may persist a year of conversation history while exposing only the pieces needed for the current task.\"},\"tunes\":{}},{\"id\":\"p-context-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The context window also creates an engineering constraint. Adding more text does not guarantee a better answer; irrelevant, stale, contradictory, or low-authority information can dilute the evidence that actually matters.\"},\"tunes\":{}},{\"id\":\"h-runtime\",\"type\":\"header\",\"data\":{\"text\":\"5. Runtime and orchestration: coordinating the loop\",\"level\":2},\"tunes\":{}},{\"id\":\"p-runtime-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The runtime or orchestration layer coordinates how the model participates in a task. Depending on the architecture, it can manage sessions, model requests, tool discovery, tool-call loops, retries, handoffs, streaming events, timeouts, checkpoints, compaction, or execution environments.\"},\"tunes\":{}},{\"id\":\"p-runtime-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Some runtimes are thin application code around a model API. Others are full agent harnesses. A managed vendor runtime can own part of the loop while the application still owns domain truth, authorization, business side effects, and product lifecycle.\"},\"tunes\":{}},{\"id\":\"p-runtime-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"This boundary is important because where the runtime runs and where inference runs are separate decisions. A locally running client or agent process can still call a remote model, while a remote application can call a model hosted on infrastructure under the organization's control.\"},\"tunes\":{}},{\"id\":\"h-application\",\"type\":\"header\",\"data\":{\"text\":\"6. The application: where AI becomes a product\",\"level\":2},\"tunes\":{}},{\"id\":\"p-app-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The application is the product boundary around the AI components. It owns the user experience, domain model, current state, identity, tenant scope, permissions, persistence, service integrations, validation, observability, billing or quota logic where relevant, and the rules that determine what the AI is allowed to see or do.\"},\"tunes\":{}},{\"id\":\"p-app-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"This is the layer that turns “a model can produce useful output” into “a system can deliver a reliable capability.” The same model can participate in a private research assistant, a support workflow, a code agent, or a commerce application because the surrounding application changes the data, tools, policies, state, and execution contract.\"},\"tunes\":{}},{\"id\":\"app-rule\",\"type\":\"callout\",\"data\":{\"variant\":\"info\",\"title\":\"The model is replaceable; the product boundary is not\",\"body\":\"Provider and model substitution can be an architectural goal. The application's authoritative state, permissions, domain rules, audit trail, and user contract cannot simply be delegated to whichever model is currently selected.\"},\"tunes\":{}},{\"id\":\"h-work-together\",\"type\":\"header\",\"data\":{\"text\":\"How the parts work together in a real request\",\"level\":2},\"tunes\":{}},{\"id\":\"p-together-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Consider a support assistant asked: “Refund order 4711 if it is still eligible, and explain why.” The request combines knowledge, current state, authorization, reasoning, and a side effect.\"},\"tunes\":{}},{\"id\":\"support-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Need\",\"Correct layer\",\"Why\"],[\"Refund policy\",\"Retrieval\",\"The system must find the current applicable policy and preserve its provenance.\"],[\"Order 4711 status\",\"Direct data\u002Ftool access\",\"The current order record is volatile authoritative state, not something to guess from model knowledge.\"],[\"User's authority to refund\",\"Application \u002F authorization\",\"Permissions must be enforced independently of what the model asks for.\"],[\"Interpret policy against order facts\",\"Model + context\",\"The model can reason over the policy evidence and current order state supplied to it.\"],[\"Execute refund\",\"Tool + application transaction rules\",\"A controlled external operation changes real state.\"],[\"Explain outcome\",\"Model\",\"The model can generate the user-facing explanation from validated results.\"],[\"Audit what happened\",\"Application \u002F runtime\",\"The system records evidence, calls, decisions, side effects, and errors as required.\"]]},\"tunes\":{}},{\"id\":\"p-together-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"If the assistant only has the language model, it can discuss refunds but cannot safely know whether order 4711 is currently eligible or perform the transaction. If it only has retrieval, it may find the policy but still lack live order state. If it has tools without application authorization, it may become capable but unsafe. Reliability comes from composing the layers with explicit ownership.\"},\"tunes\":{}},{\"id\":\"h-configs\",\"type\":\"header\",\"data\":{\"text\":\"Different AI products use different combinations\",\"level\":2},\"tunes\":{}},{\"id\":\"config-comparison\",\"type\":\"comparison\",\"data\":{\"title\":\"The presence of a model does not define the whole architecture\",\"layout\":\"table\",\"columns\":[{\"id\":\"retrieval\",\"label\":\"Retrieval\"},{\"id\":\"tools\",\"label\":\"Tools\"},{\"id\":\"state\",\"label\":\"Authoritative state\"},{\"id\":\"result\",\"label\":\"Typical capability\"}],\"rows\":[{\"id\":\"bare\",\"label\":\"Model-only assistant\",\"values\":[\"\",\"\",\"\",\"\"]},{\"id\":\"rag\",\"label\":\"Retrieval-grounded assistant\",\"values\":[\"\",\"\",\"\",\"\"]},{\"id\":\"tool\",\"label\":\"Tool-using assistant\",\"values\":[\"\",\"\",\"\",\"\"]},{\"id\":\"agent\",\"label\":\"Agentic application\",\"values\":[\"\",\"\",\"\",\"\"]}]},\"tunes\":{}},{\"id\":\"p-configs-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"These are architecture patterns, not maturity rankings. A model-only feature can be the correct design when the task needs no external facts or actions. Adding retrieval, tools, memory, or an agent loop is justified only when the task requires those capabilities.\"},\"tunes\":{}},{\"id\":\"h-implementation\",\"type\":\"header\",\"data\":{\"text\":\"Implementation evidence: Aaasaasa AI Client\",\"level\":2},\"tunes\":{}},{\"id\":\"implementation-scope\",\"type\":\"callout\",\"data\":{\"variant\":\"note\",\"title\":\"Primary implementation evidence\",\"body\":\"The following section describes an implementation I built and reviewed against the Aaasaasa AI Client codebase and architecture documentation as of \u003Cstrong>26 July 2026\u003C\u002Fstrong>. It is evidence for the usefulness of these boundaries, not a claim that one implementation is a universal standard or a commercially deployed enterprise product.\"},\"tunes\":{}},{\"id\":\"p-impl-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Aaasaasa AI Client is a local-first desktop AI workspace built with Nuxt 4, Electron and TypeScript. Its AI Hub deliberately separates agent\u002Fclient, provider, model, runtime location, permissions, and web client instead of treating them as one “AI” setting.\"},\"tunes\":{}},{\"id\":\"p-impl-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"That separation creates concrete behavior. Direct Chat can talk to models without filesystem or shell tools. A Codex agent can use a selected workspace and permission profile. Ollama can provide direct local inference, while LM Studio and configurable OpenAI-compatible endpoints represent other provider paths. A locally running Codex process can still use a cloud model, so the UI and architecture do not equate local runtime with local inference.\"},\"tunes\":{}},{\"id\":\"p-impl-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The implementation also contains Qdrant\u002Fvector support, document-extraction capabilities and an authenticated directory MCP broker. Those components illustrate another boundary: retrieval infrastructure and tool access can live in the same product without becoming properties of the model itself.\"},\"tunes\":{}},{\"id\":\"impl-map\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"A01 concept\",\"Aaasaasa AI Client implementation evidence\"],[\"Model\",\"A provider-specific model identifier is selected separately from provider and runtime.\"],[\"Provider\",\"Ollama, LM Studio, OpenAI-compatible services and other provider paths are represented separately.\"],[\"Runtime\",\"Local or remote agent\u002Fruntime location is tracked independently of the model.\"],[\"Tools \u002F access\",\"Direct Chat has no filesystem or shell tools; controlled directory access is brokered separately.\"],[\"Permissions\",\"Workspace permission profiles are application\u002Fsession policy, not model capability.\"],[\"Retrieval infrastructure\",\"Vector support and document extraction exist as data\u002Fretrieval capabilities rather than model features.\"],[\"Application\",\"The Electron\u002FNuxt product coordinates UI, credentials, providers, runtime discovery, permissions, tools and model interaction.\"]]},\"tunes\":{}},{\"id\":\"impl-lesson\",\"type\":\"callout\",\"data\":{\"variant\":\"success\",\"title\":\"Implementation lesson\",\"body\":\"The architecture became easier to reason about once \u003Cstrong>model, provider, runtime, permissions, tools, data and client\u003C\u002Fstrong> stopped being represented as one configuration choice. The distinction is operational: it determines what can run locally, what can access files, what may call paid cloud inference, and which layer owns authorization.\"},\"tunes\":{}},{\"id\":\"h-errors\",\"type\":\"header\",\"data\":{\"text\":\"Common category errors\",\"level\":2},\"tunes\":{}},{\"id\":\"errors-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Category error\",\"What is actually happening\"],[\"“The AI knows our documents.”\",\"The application or retrieval layer makes selected document content available to the model.\"],[\"“RAG is our vector database.”\",\"The vector database can be one index or store used by a retrieval pipeline; RAG is the retrieval-plus-generation pattern.\"],[\"“The model called our CRM.”\",\"The model produced a tool request; the runtime\u002Fapplication authorized and executed the external call.\"],[\"“It is local AI because the desktop agent runs locally.”\",\"Runtime location and inference location are separate. A local runtime can still invoke a remote model.\"],[\"“The model has permission to edit files.”\",\"The application\u002Fruntime grants a tool capability under a permission policy; permission is not an intrinsic model property.\"],[\"“More context means more knowledge.”\",\"Context is the finite input made available for one inference. Larger context can contain more noise, conflict or stale information.\"],[\"“The chatbot is the AI architecture.”\",\"The chat UI is one interface. The system can also include identity, state, retrieval, tools, runtime, validation, persistence and observability.\"]]},\"tunes\":{}},{\"id\":\"h-failures\",\"type\":\"header\",\"data\":{\"text\":\"Failure modes when the boundaries collapse\",\"level\":2},\"tunes\":{}},{\"id\":\"p-failure-intro\",\"type\":\"paragraph\",\"data\":{\"text\":\"Boundary mistakes are not merely terminology problems. They create distinct production failures that require different fixes.\"},\"tunes\":{}},{\"id\":\"failure-comparison\",\"type\":\"comparison\",\"data\":{\"title\":\"Diagnose the failing layer before replacing the model\",\"layout\":\"table\",\"columns\":[{\"id\":\"symptom\",\"label\":\"Symptom\"},{\"id\":\"likely\",\"label\":\"Likely boundary problem\"},{\"id\":\"fix\",\"label\":\"First architectural check\"}],\"rows\":[{\"id\":\"stale\",\"label\":\"Stale answer\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"missing\",\"label\":\"Missing company fact\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"unsafe\",\"label\":\"Unsafe side effect\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"noise\",\"label\":\"Confused answer with lots of supplied text\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"route\",\"label\":\"Unexpected cloud use\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"loop\",\"label\":\"Agent stalls or repeats\",\"values\":[\"\",\"\",\"\"]}]},\"tunes\":{}},{\"id\":\"h-version\",\"type\":\"header\",\"data\":{\"text\":\"What is stable and what is version-sensitive?\",\"level\":2},\"tunes\":{}},{\"id\":\"p-version-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The architectural distinctions in this article are intentionally vendor-neutral. The current examples below are implementation facts that should be re-checked when APIs evolve.\"},\"tunes\":{}},{\"id\":\"version-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Area\",\"Stable architectural idea\",\"Verified current example on 8 Oct 2026\"],[\"AI model vs system\",\"A model is a component inside a broader system\",\"NIST's current glossary separately defines AI model and AI system.\"],[\"RAG\",\"Generation can be conditioned on retrieved external information\",\"The Lewis et al. 2020 formulation remains the foundational reference; production retrieval methods now extend far beyond one dense index design.\"],[\"Hosted retrieval\",\"Retrieval can be exposed as a managed tool\",\"OpenAI File Search is currently a Responses API tool that searches uploaded-file knowledge bases using semantic and keyword retrieval.\"],[\"Function\u002Ftool calling\",\"A model can request application-defined external capabilities\",\"OpenAI currently documents function calling as an interface to external systems, data and actions.\"],[\"Context engineering\",\"Model behavior depends on the finite information supplied for the current inference\",\"Anthropic's current engineering guidance defines context as the token set included when sampling from the LLM and focuses on curating that set.\"],[\"Vendor APIs\",\"SDKs, tool names, endpoint shapes and supported features change\",\"Treat vendor documentation as version-sensitive even when the responsibility boundary remains stable.\"]]},\"tunes\":{}},{\"id\":\"p-version-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"A source-of-truth article should therefore preserve both levels: stable concepts for architecture, and dated evidence for current implementations. Mixing the two makes an article age unnecessarily fast.\"},\"tunes\":{}},{\"id\":\"h-test\",\"type\":\"header\",\"data\":{\"text\":\"The AI component-boundary test\",\"level\":2},\"tunes\":{}},{\"id\":\"p-test-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"When evaluating an AI feature, ask the following questions in order. The answers reveal which components the system actually has and which responsibilities are still implicit.\"},\"tunes\":{}},{\"id\":\"boundary-test\",\"type\":\"processFlow\",\"data\":{\"title\":\"Seven questions for a production design\",\"orientation\":\"auto\",\"steps\":[{\"label\":\"1. What generates the output?\",\"description\":\"Identify the exact model and the modalities or structured outputs it provides.\"},{\"label\":\"2. What facts are authoritative outside the model?\",\"description\":\"Identify documents, databases, APIs, current state and other sources of truth.\"},{\"label\":\"3. How is relevant information selected?\",\"description\":\"Separate direct lookup, search, retrieval, ranking and context construction.\"},{\"label\":\"4. What can cause real side effects?\",\"description\":\"List tools and external actions, then identify who validates and authorizes them.\"},{\"label\":\"5. What reaches the model as context?\",\"description\":\"Make instructions, evidence, state, history, memory and tool definitions explicit.\"},{\"label\":\"6. Who owns the loop?\",\"description\":\"Identify the runtime or harness that manages calls, events, retries, tool loops and sessions.\"},{\"label\":\"7. What remains the application's responsibility?\",\"description\":\"Make identity, permissions, domain state, validation, persistence, observability and UX explicit.\"}]},\"tunes\":{}},{\"id\":\"h-not\",\"type\":\"header\",\"data\":{\"text\":\"What generative AI is not\",\"level\":2},\"tunes\":{}},{\"id\":\"p-not-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Generative AI is not synonymous with an LLM, even though LLMs are a major class of generative model. It is also not synonymous with RAG, a vector database, an agent, a tool protocol, a chatbot UI, or an application.\"},\"tunes\":{}},{\"id\":\"p-not-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Those concepts can be connected, but each answers a different architectural question. An LLM asks how language output is produced. Retrieval asks where external evidence comes from. Tools ask how external capabilities are exposed. Context asks what the model can see. Runtime asks how execution is coordinated. The application asks how the capability becomes a controlled product.\"},\"tunes\":{}},{\"id\":\"remember\",\"type\":\"callout\",\"data\":{\"variant\":\"success\",\"title\":\"If you remember only one model\",\"body\":\"\u003Cstrong>Model = generate.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Retrieval = find evidence.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Tools = read or act outside the model.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Context = what the model sees now.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Runtime = coordinate execution.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Application = own the product, state, rules and permissions.\u003C\u002Fstrong>\"},\"tunes\":{}},{\"id\":\"h-next\",\"type\":\"header\",\"data\":{\"text\":\"Where to go next in the knowledge graph\",\"level\":2},\"tunes\":{}},{\"id\":\"p-next-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Once these boundaries are clear, deeper topics become easier to place. RAG belongs in retrieval and context construction. Retrieval Trigger decides when external evidence is required. Agent memory concerns what persists across time. Tool calling and MCP belong to capability access. Agent harnesses belong to runtime orchestration. RBAC, tenant isolation and domain authorization belong to the application and platform security boundary.\"},\"tunes\":{}},{\"id\":\"ref-data\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fwhere-does-an-llm-get-its-data-rag-data-sources-in-python\",\"title\":\"Where Does an LLM Get Its Data? RAG Data Sources in Python\",\"excerpt\":\"A practical continuation showing how files, SQL, APIs, full-text search, embeddings and context assembly connect external data to an LLM.\",\"ctaLabel\":\"See the data path in code\"},\"tunes\":{}},{\"id\":\"ref-trigger\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fwhen-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger\",\"title\":\"When Should an AI Stop Trusting Its Own Knowledge? — The Retrieval Trigger\",\"excerpt\":\"A decision model for when an AI system should stop relying only on model knowledge and obtain external evidence.\",\"ctaLabel\":\"Read the retrieval decision model\"},\"tunes\":{}},{\"id\":\"h-limit\",\"type\":\"header\",\"data\":{\"text\":\"Limitations\",\"level\":2},\"tunes\":{}},{\"id\":\"p-limit-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The six-layer model is a responsibility map, not a requirement that every product deploy six separate services. A small application may implement context construction, retrieval and orchestration inside one process. A managed platform may bundle several responsibilities behind one API. Physical deployment can be combined while semantic ownership remains distinct.\"},\"tunes\":{}},{\"id\":\"p-limit-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Terminology also varies across vendors and research. “Agent,” “runtime,” “memory,” “tool,” “connector,” and “context” can be defined differently. The definitions here are chosen to make operational ownership and failure diagnosis explicit rather than to claim that every framework uses identical vocabulary.\"},\"tunes\":{}},{\"id\":\"p-limit-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The Aaasaasa AI Client section documents one implementation pattern. It demonstrates that explicit boundaries are practical, but it does not prove that the same component layout is optimal for every AI product.\"},\"tunes\":{}},{\"id\":\"h-change\",\"type\":\"header\",\"data\":{\"text\":\"What would change this answer?\",\"level\":2},\"tunes\":{}},{\"id\":\"p-change-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The responsibility map would need revision if model architectures themselves began to own authoritative external state, permissions, durable transactional side effects, and verifiable source access as intrinsic properties rather than capabilities supplied by a surrounding system. Current production architectures do not make that a safe general assumption.\"},\"tunes\":{}},{\"id\":\"p-change-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Individual implementation examples will change much sooner. Hosted retrieval tools, agent APIs, MCP integrations, context-management features and provider capabilities evolve quickly. Those details should be updated without collapsing the underlying distinctions between generation, evidence, capability access, context, execution and application control.\"},\"tunes\":{}},{\"id\":\"h-conclusion\",\"type\":\"header\",\"data\":{\"text\":\"Conclusion\",\"level\":2},\"tunes\":{}},{\"id\":\"p-conclusion-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Generative AI becomes easier to design once “the AI” stops being treated as one black box. The model is the generative component, not the complete product. Retrieval provides external evidence. Tools expose capabilities. Context carries selected information into the current inference. The runtime coordinates execution. The application owns the authoritative product boundary.\"},\"tunes\":{}},{\"id\":\"p-conclusion-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"That separation is useful for more than explanation. It tells engineers where stale facts originate, where authorization belongs, why a local runtime can still use cloud inference, why RAG does not equal a vector database, why tool calls require validation, and why changing the model cannot repair every system failure.\"},\"tunes\":{}},{\"id\":\"p-conclusion-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The durable architecture question is therefore not “Which AI model are we using?” It is: Which responsibility does each component own, what evidence crosses each boundary, and which layer is allowed to change real state?\"},\"tunes\":{}},{\"id\":\"h-faq\",\"type\":\"header\",\"data\":{\"text\":\"FAQ\",\"level\":2},\"tunes\":{}},{\"id\":\"faq\",\"type\":\"faq\",\"data\":{\"title\":\"Generative AI system boundaries\",\"items\":[{\"id\":\"faq1\",\"question\":\"Is generative AI the same as an LLM?\",\"answer\":\"No. An LLM is one type of generative model. Generative AI also includes other modalities, and a production generative AI system can include retrieval, tools, runtime logic, application state, permissions, persistence and user interfaces around the model.\"},{\"id\":\"faq2\",\"question\":\"Is RAG part of the model?\",\"answer\":\"Usually no. RAG is an application\u002Fsystem pattern that retrieves external information and supplies selected evidence to the model. Some platforms package retrieval tightly with model APIs, but the responsibility remains distinct.\"},{\"id\":\"faq3\",\"question\":\"Is a vector database required for RAG?\",\"answer\":\"No. RAG can use vector search, lexical search, hybrid retrieval, SQL, APIs, knowledge graphs or other methods. The defining property is retrieval of external information for generation, not one storage technology.\"},{\"id\":\"faq4\",\"question\":\"Are tools the same as context?\",\"answer\":\"No. A tool is an external capability. Its definition may be represented in context, and its result may later enter context, but the actual capability executes outside the model.\"},{\"id\":\"faq5\",\"question\":\"Does running an AI client locally mean the model is local?\",\"answer\":\"No. Runtime location and inference location are separate. A local desktop application or agent can call a remote model, while a remote application can call an internally hosted model.\"},{\"id\":\"faq6\",\"question\":\"Who should enforce permissions for AI tools?\",\"answer\":\"The application or runtime security boundary should enforce authorization. A model can request an operation, but model intent should never be treated as sufficient execution authority.\"},{\"id\":\"faq7\",\"question\":\"Where does current application state belong?\",\"answer\":\"Authoritative volatile state should normally remain in the application or domain system that owns it. The AI can receive the relevant state through controlled context or tool access when needed.\"}]},\"tunes\":{}},{\"id\":\"h-glossary\",\"type\":\"header\",\"data\":{\"text\":\"Glossary\",\"level\":2},\"tunes\":{}},{\"id\":\"glossary\",\"type\":\"glossary\",\"data\":{\"title\":\"Core terms\",\"entries\":[{\"term\":\"Generative model\",\"definition\":\"An AI model designed to generate derived synthetic content such as text, images, audio, video, code or structured output.\",\"anchor\":\"generative-model\"},{\"term\":\"Retrieval\",\"definition\":\"The process of selecting relevant information from an external source or store for the current task.\",\"anchor\":\"retrieval\"},{\"term\":\"RAG\",\"definition\":\"Retrieval-Augmented Generation: a pattern in which retrieved external information is supplied to a generative model to improve the current output.\",\"anchor\":\"rag\"},{\"term\":\"Tool\",\"definition\":\"A capability exposed to an AI runtime for reading data, calculating, searching, or performing an external action.\",\"anchor\":\"tool\"},{\"term\":\"Context\",\"definition\":\"The information available to the model for a particular inference step.\",\"anchor\":\"context\"},{\"term\":\"Runtime \u002F orchestrator\",\"definition\":\"The software layer that coordinates model calls, tool calls, task loops, sessions, retries, events or execution environments.\",\"anchor\":\"runtime-orchestrator\"},{\"term\":\"Application\",\"definition\":\"The product and domain layer that owns user interaction, authoritative state, permissions, validation, persistence and business behavior.\",\"anchor\":\"application\"},{\"term\":\"Provider\",\"definition\":\"The service or runtime that exposes access to one or more models; provider identity and model identity are separate concerns.\",\"anchor\":\"provider\"}]},\"tunes\":{}},{\"id\":\"h-sources\",\"type\":\"header\",\"data\":{\"text\":\"Primary sources and implementation evidence\",\"level\":2},\"tunes\":{}},{\"id\":\"p-sources-note\",\"type\":\"paragraph\",\"data\":{\"text\":\"Stable definitions below are anchored in standards\u002Fresearch; fast-moving implementation examples use current official engineering documentation. Aaasaasa AI Client is original implementation evidence and was checked against its codebase\u002Fdocumentation state dated 26 July 2026.\"},\"tunes\":{}},{\"id\":\"src-nist-profile\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fnvlpubs.nist.gov\u002Fnistpubs\u002Fai\u002FNIST.AI.600-1.pdf\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"NIST AI 600-1 — Generative Artificial Intelligence Profile\",\"description\":\"NIST's Generative AI profile, including the generative-AI definition and explicit distinction between model-, system-, application- and use-case-level concerns.\"}},\"tunes\":{}},{\"id\":\"src-nist-model\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fcsrc.nist.gov\u002Fglossary\u002Fterm\u002Fartificial_intelligence_model\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"NIST — Artificial Intelligence Model\",\"description\":\"Current NIST glossary definition of an AI model as a component of an information system that produces outputs from inputs using AI techniques.\"}},\"tunes\":{}},{\"id\":\"src-nist-system\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fcsrc.nist.gov\u002Fglossary\u002Fterm\u002Fartificial_intelligence_system\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"NIST — Artificial Intelligence System\",\"description\":\"Current NIST glossary definition showing that an AI system can include data systems, software, hardware, applications, tools or utilities using AI.\"}},\"tunes\":{}},{\"id\":\"src-rag-paper\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2005.11401\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Lewis et al. — Retrieval-Augmented Generation for Knowledge-Intensive NLP Tasks\",\"description\":\"The 2020 paper introducing the RAG formulation that combines a generative model with retrieved non-parametric memory.\"}},\"tunes\":{}},{\"id\":\"src-openai-file-search\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Ftools-file-search\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — File Search\",\"description\":\"Current official documentation for hosted file retrieval in the Responses API using uploaded-file knowledge bases, semantic search and keyword search.\"}},\"tunes\":{}},{\"id\":\"src-openai-functions\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Ffunction-calling\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — Function Calling\",\"description\":\"Current official documentation describing tool\u002Ffunction calling as the interface between models and external systems, data and actions.\"}},\"tunes\":{}},{\"id\":\"src-anthropic-context\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Feffective-context-engineering-for-ai-agents\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Anthropic — Effective Context Engineering for AI Agents\",\"description\":\"Engineering guidance defining context as the token set available during LLM sampling and explaining why context selection is a finite-resource problem.\"}},\"tunes\":{}}],\"version\":\"2.31.6\"}",{"time":212,"blocks":213,"version":1064},1791475413504,[214,220,228,235,243,248,253,258,265,270,275,307,312,317,360,365,370,375,380,385,390,395,402,411,416,421,426,431,437,442,447,452,457,462,467,472,477,482,487,492,498,503,508,543,548,553,584,589,594,600,605,610,615,642,648,653,682,687,692,732,737,742,775,780,785,790,817,822,827,832,838,843,848,856,864,869,874,879,884,889,894,899,904,909,914,919,924,958,963,990,995,1000,1010,1019,1028,1037,1046,1055],{"id":215,"data":216,"type":218,"tunes":219},"intro",{"text":217},"Generative AI is not one component. A production generative AI system usually combines a generative model with application code that supplies instructions and context, retrieves external knowledge when needed, exposes tools for reading or changing external systems, manages runtime state and permissions, and turns the result into a usable product. Treating the model, retrieval, tools, context, runtime, and application as the same thing hides the boundaries that determine freshness, security, reliability, cost, and control.","paragraph",{},{"id":221,"data":222,"type":226,"tunes":227},"direct",{"body":223,"title":224,"variant":225},"\u003Cstrong>The model generates; retrieval finds external evidence; tools access data or perform actions; context is what the model can see for the current inference; the runtime coordinates execution; the application owns product rules, state, permissions, persistence, and user experience.\u003C\u002Fstrong> These layers can be packaged together by a vendor, but their responsibilities remain different.","Direct answer","info","callout",{},{"id":229,"data":230,"type":226,"tunes":234},"scope-note",{"body":231,"title":232,"variant":233},"This article defines durable architectural responsibilities rather than one vendor stack. Current implementation examples were re-checked on \u003Cstrong>8 October 2026\u003C\u002Fstrong>. Vendor APIs and product names can change; the responsibility boundaries are more stable than any individual SDK or endpoint.","Terminology and version note","note",{},{"id":236,"data":237,"type":241,"tunes":242},"toc",{"title":238,"maxLevel":239,"minLevel":240},"Contents",3,2,"tableOfContents",{},{"id":244,"data":245,"type":42,"tunes":247},"h-meaning",{"text":246,"level":240},"What does “generative AI” actually mean?",{},{"id":249,"data":250,"type":218,"tunes":252},"p-meaning-1",{"text":251},"At the model level, generative AI refers to AI models that generate derived synthetic content such as text, images, audio, video, code, or other digital output. NIST AI 600-1 uses this model-oriented meaning and separately discusses risks at model, system, application, and use-case levels.",{},{"id":254,"data":255,"type":218,"tunes":257},"p-meaning-2",{"text":256},"That distinction matters because an AI model is not the same thing as the complete AI system. NIST's current glossary defines an AI model as a component that produces outputs from inputs using computational, statistical, or machine-learning techniques, while an AI system can include software, hardware, applications, tools, or utilities that operate using AI.",{},{"id":259,"data":260,"type":226,"tunes":264},"model-system-rule",{"body":261,"title":262,"variant":263},"\u003Cstrong>Generative model ≠ generative AI application.\u003C\u002Fstrong>\u003Cbr>A model is one computational component. A usable AI product is a system built around that component.","A useful boundary","success",{},{"id":266,"data":267,"type":42,"tunes":269},"h-simple",{"text":268,"level":240},"The simplest useful model of a generative AI system",{},{"id":271,"data":272,"type":218,"tunes":274},"p-simple-1",{"text":273},"For a first mental model, imagine a company assistant answering: “Can this customer receive a refund today?” A useful answer may require several different responsibilities. The language model can interpret the question and write the explanation, but the current order state may come from a database tool, the refund policy may come from document retrieval, permissions may be enforced by the application, and the final action may require a controlled API call.",{},{"id":276,"data":277,"type":305,"tunes":306},"simple-flow",{"steps":278,"title":303,"orientation":304},[279,282,285,288,291,294,297,300],{"label":280,"description":281},"1. User request","The application receives a natural-language question or task.",{"label":283,"description":284},"2. Application policy and state","Identity, tenant, permissions, current workflow state, and product rules define what the request is allowed to do.",{"label":286,"description":287},"3. Retrieval or direct data access","The system obtains external evidence or current facts when model knowledge is insufficient.",{"label":289,"description":290},"4. Context construction","Instructions, user input, selected evidence, relevant state, and tool definitions are assembled for the model.",{"label":292,"description":293},"5. Model inference","The generative model interprets the supplied context and produces text, structured output, or a tool request.",{"label":295,"description":296},"6. Tool execution when needed","The runtime or application validates and executes approved tool calls outside the model.",{"label":298,"description":299},"7. Observation and continuation","Tool results can return to the model as new context for another inference step.",{"label":301,"description":302},"8. Validation and product output","The application validates the result, records required state or audit data, and presents or executes the final outcome.","One common execution path","auto","processFlow",{},{"id":308,"data":309,"type":218,"tunes":311},"p-simple-2",{"text":310},"Real systems do not always follow this sequence exactly. Retrieval can happen before the first model call, tools can be selected during an agent loop, deterministic application logic can bypass the model entirely, and validation can occur at several stages. The point is to separate responsibilities, not to impose one universal workflow.",{},{"id":313,"data":314,"type":42,"tunes":316},"h-boundaries",{"text":315,"level":240},"The six boundaries that matter",{},{"id":318,"data":319,"type":358,"tunes":359},"boundary-comparison",{"rows":320,"title":346,"layout":347,"columns":348},[321,326,330,334,338,342],{"id":322,"label":323,"values":324},"model","Model",[325,325,325],"",{"id":327,"label":328,"values":329},"retrieval","Retrieval",[325,325,325],{"id":331,"label":332,"values":333},"tools","Tools",[325,325,325],{"id":335,"label":336,"values":337},"context","Context",[325,325,325],{"id":339,"label":340,"values":341},"runtime","Runtime \u002F orchestrator",[325,325,325],{"id":343,"label":344,"values":345},"application","Application",[325,325,325],"Six responsibilities inside one AI product","table",[349,352,355],{"id":350,"label":351},"job","Primary job",{"id":353,"label":354},"input","Typical inputs",{"id":356,"label":357},"not","Not the same as","comparison",{},{"id":361,"data":362,"type":42,"tunes":364},"h-model",{"text":363,"level":240},"1. The model: generation is its core responsibility",{},{"id":366,"data":367,"type":218,"tunes":369},"p-model-1",{"text":368},"A generative model maps supplied inputs to generated outputs. For a language model, that can include natural-language text, structured JSON, code, classifications, summaries, plans, or tool-call arguments. Multimodal generative models can work with additional input and output types.",{},{"id":371,"data":372,"type":218,"tunes":374},"p-model-2",{"text":373},"The model can contain substantial learned knowledge in its parameters, but parameterized knowledge is not a live database. The model does not automatically know a document created five minutes ago, the current stock level, a private customer record, or the state of an application unless that information is supplied through the current input path.",{},{"id":376,"data":377,"type":218,"tunes":379},"p-model-3",{"text":378},"This is why changing the model does not automatically solve stale knowledge, missing permissions, broken retrieval, incorrect state ownership, or unsafe tool execution. Those failures often belong to other layers.",{},{"id":381,"data":382,"type":42,"tunes":384},"h-retrieval",{"text":383,"level":240},"2. Retrieval: finding external evidence is a separate operation",{},{"id":386,"data":387,"type":218,"tunes":389},"p-retrieval-1",{"text":388},"Retrieval selects information from an external source before or during generation. The 2020 Retrieval-Augmented Generation work by Lewis et al. made the separation explicit by combining a parametric generative model with retrieved non-parametric memory. Modern production systems use many retrieval variants, but the architectural idea remains: useful evidence can be fetched at inference time instead of relying only on what the model learned during training.",{},{"id":391,"data":392,"type":218,"tunes":394},"p-retrieval-2",{"text":393},"Retrieval can use lexical search, embeddings, vector search, hybrid search, SQL, knowledge graphs, metadata filters, APIs, or other selection mechanisms. A vector database is therefore one possible retrieval component, not the definition of RAG.",{},{"id":396,"data":397,"type":226,"tunes":401},"retrieval-rule",{"body":398,"title":399,"variant":400},"A retrieved passage can be highly relevant and still be stale, unauthorized, from the wrong version, or insufficient to support a claim. Retrieval quality and evidence quality must be evaluated separately.","Relevance is not authority","warning",{},{"id":403,"data":404,"type":409,"tunes":410},"ref-rag",{"url":405,"title":406,"excerpt":407,"ctaLabel":408},"https:\u002F\u002Fstajic.de\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works","What Is RAG? The Simplest Explanation of How It Works","The canonical plain-English explanation of retrieval-augmented generation, including the separation between LLM, knowledge, state, memory and tools.","Read the RAG foundation","referralArticle",{},{"id":412,"data":413,"type":42,"tunes":415},"h-tools",{"text":414,"level":240},"3. Tools: access and action are not model knowledge",{},{"id":417,"data":418,"type":218,"tunes":420},"p-tools-1",{"text":419},"A tool is an interface through which an AI runtime can request functionality outside the model. A tool can query a database, search the web, read a file, calculate a value, call an internal service, create a ticket, send a message, modify a record, or trigger another controlled operation.",{},{"id":422,"data":423,"type":218,"tunes":425},"p-tools-2",{"text":424},"OpenAI's current function-calling documentation makes this boundary explicit: function calling lets models interface with external systems and access data or actions provided by the application. The model can propose or select a call, but the external system performs the real operation.",{},{"id":427,"data":428,"type":218,"tunes":430},"p-tools-3",{"text":429},"Tool use therefore creates two separate questions: Can the model request this capability? and Will the application authorize and execute it? A production system should not confuse model intent with permission to cause a side effect.",{},{"id":432,"data":433,"type":226,"tunes":436},"tool-rule",{"body":434,"title":435,"variant":263},"A model can emit a valid tool request and still be denied. Authorization, argument validation, rate limits, transaction rules, audit requirements, and rollback belong outside the model.","Model intent is not execution authority",{},{"id":438,"data":439,"type":42,"tunes":441},"h-context",{"text":440,"level":240},"4. Context: what the model can see right now",{},{"id":443,"data":444,"type":218,"tunes":446},"p-context-1",{"text":445},"Context is the information available to the model for a particular inference step. Anthropic's context-engineering guidance describes context as the set of tokens included when sampling from an LLM. In practice, that set can contain system instructions, user messages, conversation history, retrieved evidence, tool definitions, tool results, memory summaries, and selected application state.",{},{"id":448,"data":449,"type":218,"tunes":451},"p-context-2",{"text":450},"Context is therefore neither the complete knowledge base nor long-term memory. A company may store ten million documents while only a handful of passages enter one model call. A runtime may persist a year of conversation history while exposing only the pieces needed for the current task.",{},{"id":453,"data":454,"type":218,"tunes":456},"p-context-3",{"text":455},"The context window also creates an engineering constraint. Adding more text does not guarantee a better answer; irrelevant, stale, contradictory, or low-authority information can dilute the evidence that actually matters.",{},{"id":458,"data":459,"type":42,"tunes":461},"h-runtime",{"text":460,"level":240},"5. Runtime and orchestration: coordinating the loop",{},{"id":463,"data":464,"type":218,"tunes":466},"p-runtime-1",{"text":465},"The runtime or orchestration layer coordinates how the model participates in a task. Depending on the architecture, it can manage sessions, model requests, tool discovery, tool-call loops, retries, handoffs, streaming events, timeouts, checkpoints, compaction, or execution environments.",{},{"id":468,"data":469,"type":218,"tunes":471},"p-runtime-2",{"text":470},"Some runtimes are thin application code around a model API. Others are full agent harnesses. A managed vendor runtime can own part of the loop while the application still owns domain truth, authorization, business side effects, and product lifecycle.",{},{"id":473,"data":474,"type":218,"tunes":476},"p-runtime-3",{"text":475},"This boundary is important because where the runtime runs and where inference runs are separate decisions. A locally running client or agent process can still call a remote model, while a remote application can call a model hosted on infrastructure under the organization's control.",{},{"id":478,"data":479,"type":42,"tunes":481},"h-application",{"text":480,"level":240},"6. The application: where AI becomes a product",{},{"id":483,"data":484,"type":218,"tunes":486},"p-app-1",{"text":485},"The application is the product boundary around the AI components. It owns the user experience, domain model, current state, identity, tenant scope, permissions, persistence, service integrations, validation, observability, billing or quota logic where relevant, and the rules that determine what the AI is allowed to see or do.",{},{"id":488,"data":489,"type":218,"tunes":491},"p-app-2",{"text":490},"This is the layer that turns “a model can produce useful output” into “a system can deliver a reliable capability.” The same model can participate in a private research assistant, a support workflow, a code agent, or a commerce application because the surrounding application changes the data, tools, policies, state, and execution contract.",{},{"id":493,"data":494,"type":226,"tunes":497},"app-rule",{"body":495,"title":496,"variant":225},"Provider and model substitution can be an architectural goal. The application's authoritative state, permissions, domain rules, audit trail, and user contract cannot simply be delegated to whichever model is currently selected.","The model is replaceable; the product boundary is not",{},{"id":499,"data":500,"type":42,"tunes":502},"h-work-together",{"text":501,"level":240},"How the parts work together in a real request",{},{"id":504,"data":505,"type":218,"tunes":507},"p-together-1",{"text":506},"Consider a support assistant asked: “Refund order 4711 if it is still eligible, and explain why.” The request combines knowledge, current state, authorization, reasoning, and a side effect.",{},{"id":509,"data":510,"type":347,"tunes":542},"support-table",{"content":511,"stretched":43,"withHeadings":14},[512,516,519,523,527,531,535,538],[513,514,515],"Need","Correct layer","Why",[517,328,518],"Refund policy","The system must find the current applicable policy and preserve its provenance.",[520,521,522],"Order 4711 status","Direct data\u002Ftool access","The current order record is volatile authoritative state, not something to guess from model knowledge.",[524,525,526],"User's authority to refund","Application \u002F authorization","Permissions must be enforced independently of what the model asks for.",[528,529,530],"Interpret policy against order facts","Model + context","The model can reason over the policy evidence and current order state supplied to it.",[532,533,534],"Execute refund","Tool + application transaction rules","A controlled external operation changes real state.",[536,323,537],"Explain outcome","The model can generate the user-facing explanation from validated results.",[539,540,541],"Audit what happened","Application \u002F runtime","The system records evidence, calls, decisions, side effects, and errors as required.",{},{"id":544,"data":545,"type":218,"tunes":547},"p-together-2",{"text":546},"If the assistant only has the language model, it can discuss refunds but cannot safely know whether order 4711 is currently eligible or perform the transaction. If it only has retrieval, it may find the policy but still lack live order state. If it has tools without application authorization, it may become capable but unsafe. Reliability comes from composing the layers with explicit ownership.",{},{"id":549,"data":550,"type":42,"tunes":552},"h-configs",{"text":551,"level":240},"Different AI products use different combinations",{},{"id":554,"data":555,"type":358,"tunes":583},"config-comparison",{"rows":556,"title":573,"layout":347,"columns":574},[557,561,565,569],{"id":558,"label":559,"values":560},"bare","Model-only assistant",[325,325,325,325],{"id":562,"label":563,"values":564},"rag","Retrieval-grounded assistant",[325,325,325,325],{"id":566,"label":567,"values":568},"tool","Tool-using assistant",[325,325,325,325],{"id":570,"label":571,"values":572},"agent","Agentic application",[325,325,325,325],"The presence of a model does not define the whole architecture",[575,576,577,580],{"id":327,"label":328},{"id":331,"label":332},{"id":578,"label":579},"state","Authoritative state",{"id":581,"label":582},"result","Typical capability",{},{"id":585,"data":586,"type":218,"tunes":588},"p-configs-1",{"text":587},"These are architecture patterns, not maturity rankings. A model-only feature can be the correct design when the task needs no external facts or actions. Adding retrieval, tools, memory, or an agent loop is justified only when the task requires those capabilities.",{},{"id":590,"data":591,"type":42,"tunes":593},"h-implementation",{"text":592,"level":240},"Implementation evidence: Aaasaasa AI Client",{},{"id":595,"data":596,"type":226,"tunes":599},"implementation-scope",{"body":597,"title":598,"variant":233},"The following section describes an implementation I built and reviewed against the Aaasaasa AI Client codebase and architecture documentation as of \u003Cstrong>26 July 2026\u003C\u002Fstrong>. It is evidence for the usefulness of these boundaries, not a claim that one implementation is a universal standard or a commercially deployed enterprise product.","Primary implementation evidence",{},{"id":601,"data":602,"type":218,"tunes":604},"p-impl-1",{"text":603},"Aaasaasa AI Client is a local-first desktop AI workspace built with Nuxt 4, Electron and TypeScript. Its AI Hub deliberately separates agent\u002Fclient, provider, model, runtime location, permissions, and web client instead of treating them as one “AI” setting.",{},{"id":606,"data":607,"type":218,"tunes":609},"p-impl-2",{"text":608},"That separation creates concrete behavior. Direct Chat can talk to models without filesystem or shell tools. A Codex agent can use a selected workspace and permission profile. Ollama can provide direct local inference, while LM Studio and configurable OpenAI-compatible endpoints represent other provider paths. A locally running Codex process can still use a cloud model, so the UI and architecture do not equate local runtime with local inference.",{},{"id":611,"data":612,"type":218,"tunes":614},"p-impl-3",{"text":613},"The implementation also contains Qdrant\u002Fvector support, document-extraction capabilities and an authenticated directory MCP broker. Those components illustrate another boundary: retrieval infrastructure and tool access can live in the same product without becoming properties of the model itself.",{},{"id":616,"data":617,"type":347,"tunes":641},"impl-map",{"content":618,"stretched":43,"withHeadings":14},[619,622,624,627,630,633,636,639],[620,621],"A01 concept","Aaasaasa AI Client implementation evidence",[323,623],"A provider-specific model identifier is selected separately from provider and runtime.",[625,626],"Provider","Ollama, LM Studio, OpenAI-compatible services and other provider paths are represented separately.",[628,629],"Runtime","Local or remote agent\u002Fruntime location is tracked independently of the model.",[631,632],"Tools \u002F access","Direct Chat has no filesystem or shell tools; controlled directory access is brokered separately.",[634,635],"Permissions","Workspace permission profiles are application\u002Fsession policy, not model capability.",[637,638],"Retrieval infrastructure","Vector support and document extraction exist as data\u002Fretrieval capabilities rather than model features.",[344,640],"The Electron\u002FNuxt product coordinates UI, credentials, providers, runtime discovery, permissions, tools and model interaction.",{},{"id":643,"data":644,"type":226,"tunes":647},"impl-lesson",{"body":645,"title":646,"variant":263},"The architecture became easier to reason about once \u003Cstrong>model, provider, runtime, permissions, tools, data and client\u003C\u002Fstrong> stopped being represented as one configuration choice. The distinction is operational: it determines what can run locally, what can access files, what may call paid cloud inference, and which layer owns authorization.","Implementation lesson",{},{"id":649,"data":650,"type":42,"tunes":652},"h-errors",{"text":651,"level":240},"Common category errors",{},{"id":654,"data":655,"type":347,"tunes":681},"errors-table",{"content":656,"stretched":43,"withHeadings":14},[657,660,663,666,669,672,675,678],[658,659],"Category error","What is actually happening",[661,662],"“The AI knows our documents.”","The application or retrieval layer makes selected document content available to the model.",[664,665],"“RAG is our vector database.”","The vector database can be one index or store used by a retrieval pipeline; RAG is the retrieval-plus-generation pattern.",[667,668],"“The model called our CRM.”","The model produced a tool request; the runtime\u002Fapplication authorized and executed the external call.",[670,671],"“It is local AI because the desktop agent runs locally.”","Runtime location and inference location are separate. A local runtime can still invoke a remote model.",[673,674],"“The model has permission to edit files.”","The application\u002Fruntime grants a tool capability under a permission policy; permission is not an intrinsic model property.",[676,677],"“More context means more knowledge.”","Context is the finite input made available for one inference. Larger context can contain more noise, conflict or stale information.",[679,680],"“The chatbot is the AI architecture.”","The chat UI is one interface. The system can also include identity, state, retrieval, tools, runtime, validation, persistence and observability.",{},{"id":683,"data":684,"type":42,"tunes":686},"h-failures",{"text":685,"level":240},"Failure modes when the boundaries collapse",{},{"id":688,"data":689,"type":218,"tunes":691},"p-failure-intro",{"text":690},"Boundary mistakes are not merely terminology problems. They create distinct production failures that require different fixes.",{},{"id":693,"data":694,"type":358,"tunes":731},"failure-comparison",{"rows":695,"title":720,"layout":347,"columns":721},[696,700,704,708,712,716],{"id":697,"label":698,"values":699},"stale","Stale answer",[325,325,325],{"id":701,"label":702,"values":703},"missing","Missing company fact",[325,325,325],{"id":705,"label":706,"values":707},"unsafe","Unsafe side effect",[325,325,325],{"id":709,"label":710,"values":711},"noise","Confused answer with lots of supplied text",[325,325,325],{"id":713,"label":714,"values":715},"route","Unexpected cloud use",[325,325,325],{"id":717,"label":718,"values":719},"loop","Agent stalls or repeats",[325,325,325],"Diagnose the failing layer before replacing the model",[722,725,728],{"id":723,"label":724},"symptom","Symptom",{"id":726,"label":727},"likely","Likely boundary problem",{"id":729,"label":730},"fix","First architectural check",{},{"id":733,"data":734,"type":42,"tunes":736},"h-version",{"text":735,"level":240},"What is stable and what is version-sensitive?",{},{"id":738,"data":739,"type":218,"tunes":741},"p-version-1",{"text":740},"The architectural distinctions in this article are intentionally vendor-neutral. The current examples below are implementation facts that should be re-checked when APIs evolve.",{},{"id":743,"data":744,"type":347,"tunes":774},"version-table",{"content":745,"stretched":43,"withHeadings":14},[746,750,754,758,762,766,770],[747,748,749],"Area","Stable architectural idea","Verified current example on 8 Oct 2026",[751,752,753],"AI model vs system","A model is a component inside a broader system","NIST's current glossary separately defines AI model and AI system.",[755,756,757],"RAG","Generation can be conditioned on retrieved external information","The Lewis et al. 2020 formulation remains the foundational reference; production retrieval methods now extend far beyond one dense index design.",[759,760,761],"Hosted retrieval","Retrieval can be exposed as a managed tool","OpenAI File Search is currently a Responses API tool that searches uploaded-file knowledge bases using semantic and keyword retrieval.",[763,764,765],"Function\u002Ftool calling","A model can request application-defined external capabilities","OpenAI currently documents function calling as an interface to external systems, data and actions.",[767,768,769],"Context engineering","Model behavior depends on the finite information supplied for the current inference","Anthropic's current engineering guidance defines context as the token set included when sampling from the LLM and focuses on curating that set.",[771,772,773],"Vendor APIs","SDKs, tool names, endpoint shapes and supported features change","Treat vendor documentation as version-sensitive even when the responsibility boundary remains stable.",{},{"id":776,"data":777,"type":218,"tunes":779},"p-version-2",{"text":778},"A source-of-truth article should therefore preserve both levels: stable concepts for architecture, and dated evidence for current implementations. Mixing the two makes an article age unnecessarily fast.",{},{"id":781,"data":782,"type":42,"tunes":784},"h-test",{"text":783,"level":240},"The AI component-boundary test",{},{"id":786,"data":787,"type":218,"tunes":789},"p-test-1",{"text":788},"When evaluating an AI feature, ask the following questions in order. The answers reveal which components the system actually has and which responsibilities are still implicit.",{},{"id":791,"data":792,"type":305,"tunes":816},"boundary-test",{"steps":793,"title":815,"orientation":304},[794,797,800,803,806,809,812],{"label":795,"description":796},"1. What generates the output?","Identify the exact model and the modalities or structured outputs it provides.",{"label":798,"description":799},"2. What facts are authoritative outside the model?","Identify documents, databases, APIs, current state and other sources of truth.",{"label":801,"description":802},"3. How is relevant information selected?","Separate direct lookup, search, retrieval, ranking and context construction.",{"label":804,"description":805},"4. What can cause real side effects?","List tools and external actions, then identify who validates and authorizes them.",{"label":807,"description":808},"5. What reaches the model as context?","Make instructions, evidence, state, history, memory and tool definitions explicit.",{"label":810,"description":811},"6. Who owns the loop?","Identify the runtime or harness that manages calls, events, retries, tool loops and sessions.",{"label":813,"description":814},"7. What remains the application's responsibility?","Make identity, permissions, domain state, validation, persistence, observability and UX explicit.","Seven questions for a production design",{},{"id":818,"data":819,"type":42,"tunes":821},"h-not",{"text":820,"level":240},"What generative AI is not",{},{"id":823,"data":824,"type":218,"tunes":826},"p-not-1",{"text":825},"Generative AI is not synonymous with an LLM, even though LLMs are a major class of generative model. It is also not synonymous with RAG, a vector database, an agent, a tool protocol, a chatbot UI, or an application.",{},{"id":828,"data":829,"type":218,"tunes":831},"p-not-2",{"text":830},"Those concepts can be connected, but each answers a different architectural question. An LLM asks how language output is produced. Retrieval asks where external evidence comes from. Tools ask how external capabilities are exposed. Context asks what the model can see. Runtime asks how execution is coordinated. The application asks how the capability becomes a controlled product.",{},{"id":833,"data":834,"type":226,"tunes":837},"remember",{"body":835,"title":836,"variant":263},"\u003Cstrong>Model = generate.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Retrieval = find evidence.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Tools = read or act outside the model.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Context = what the model sees now.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Runtime = coordinate execution.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Application = own the product, state, rules and permissions.\u003C\u002Fstrong>","If you remember only one model",{},{"id":839,"data":840,"type":42,"tunes":842},"h-next",{"text":841,"level":240},"Where to go next in the knowledge graph",{},{"id":844,"data":845,"type":218,"tunes":847},"p-next-1",{"text":846},"Once these boundaries are clear, deeper topics become easier to place. RAG belongs in retrieval and context construction. Retrieval Trigger decides when external evidence is required. Agent memory concerns what persists across time. Tool calling and MCP belong to capability access. Agent harnesses belong to runtime orchestration. RBAC, tenant isolation and domain authorization belong to the application and platform security boundary.",{},{"id":849,"data":850,"type":409,"tunes":855},"ref-data",{"url":851,"title":852,"excerpt":853,"ctaLabel":854},"https:\u002F\u002Fstajic.de\u002Fblog\u002Fwhere-does-an-llm-get-its-data-rag-data-sources-in-python","Where Does an LLM Get Its Data? RAG Data Sources in Python","A practical continuation showing how files, SQL, APIs, full-text search, embeddings and context assembly connect external data to an LLM.","See the data path in code",{},{"id":857,"data":858,"type":409,"tunes":863},"ref-trigger",{"url":859,"title":860,"excerpt":861,"ctaLabel":862},"https:\u002F\u002Fstajic.de\u002Fblog\u002Fwhen-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger","When Should an AI Stop Trusting Its Own Knowledge? — The Retrieval Trigger","A decision model for when an AI system should stop relying only on model knowledge and obtain external evidence.","Read the retrieval decision model",{},{"id":865,"data":866,"type":42,"tunes":868},"h-limit",{"text":867,"level":240},"Limitations",{},{"id":870,"data":871,"type":218,"tunes":873},"p-limit-1",{"text":872},"The six-layer model is a responsibility map, not a requirement that every product deploy six separate services. A small application may implement context construction, retrieval and orchestration inside one process. A managed platform may bundle several responsibilities behind one API. Physical deployment can be combined while semantic ownership remains distinct.",{},{"id":875,"data":876,"type":218,"tunes":878},"p-limit-2",{"text":877},"Terminology also varies across vendors and research. “Agent,” “runtime,” “memory,” “tool,” “connector,” and “context” can be defined differently. The definitions here are chosen to make operational ownership and failure diagnosis explicit rather than to claim that every framework uses identical vocabulary.",{},{"id":880,"data":881,"type":218,"tunes":883},"p-limit-3",{"text":882},"The Aaasaasa AI Client section documents one implementation pattern. It demonstrates that explicit boundaries are practical, but it does not prove that the same component layout is optimal for every AI product.",{},{"id":885,"data":886,"type":42,"tunes":888},"h-change",{"text":887,"level":240},"What would change this answer?",{},{"id":890,"data":891,"type":218,"tunes":893},"p-change-1",{"text":892},"The responsibility map would need revision if model architectures themselves began to own authoritative external state, permissions, durable transactional side effects, and verifiable source access as intrinsic properties rather than capabilities supplied by a surrounding system. Current production architectures do not make that a safe general assumption.",{},{"id":895,"data":896,"type":218,"tunes":898},"p-change-2",{"text":897},"Individual implementation examples will change much sooner. Hosted retrieval tools, agent APIs, MCP integrations, context-management features and provider capabilities evolve quickly. Those details should be updated without collapsing the underlying distinctions between generation, evidence, capability access, context, execution and application control.",{},{"id":900,"data":901,"type":42,"tunes":903},"h-conclusion",{"text":902,"level":240},"Conclusion",{},{"id":905,"data":906,"type":218,"tunes":908},"p-conclusion-1",{"text":907},"Generative AI becomes easier to design once “the AI” stops being treated as one black box. The model is the generative component, not the complete product. Retrieval provides external evidence. Tools expose capabilities. Context carries selected information into the current inference. The runtime coordinates execution. The application owns the authoritative product boundary.",{},{"id":910,"data":911,"type":218,"tunes":913},"p-conclusion-2",{"text":912},"That separation is useful for more than explanation. It tells engineers where stale facts originate, where authorization belongs, why a local runtime can still use cloud inference, why RAG does not equal a vector database, why tool calls require validation, and why changing the model cannot repair every system failure.",{},{"id":915,"data":916,"type":218,"tunes":918},"p-conclusion-3",{"text":917},"The durable architecture question is therefore not “Which AI model are we using?” It is: Which responsibility does each component own, what evidence crosses each boundary, and which layer is allowed to change real state?",{},{"id":920,"data":921,"type":42,"tunes":923},"h-faq",{"text":922,"level":240},"FAQ",{},{"id":925,"data":926,"type":925,"tunes":957},"faq",{"items":927,"title":956},[928,932,936,940,944,948,952],{"id":929,"answer":930,"question":931},"faq1","No. An LLM is one type of generative model. Generative AI also includes other modalities, and a production generative AI system can include retrieval, tools, runtime logic, application state, permissions, persistence and user interfaces around the model.","Is generative AI the same as an LLM?",{"id":933,"answer":934,"question":935},"faq2","Usually no. RAG is an application\u002Fsystem pattern that retrieves external information and supplies selected evidence to the model. Some platforms package retrieval tightly with model APIs, but the responsibility remains distinct.","Is RAG part of the model?",{"id":937,"answer":938,"question":939},"faq3","No. RAG can use vector search, lexical search, hybrid retrieval, SQL, APIs, knowledge graphs or other methods. The defining property is retrieval of external information for generation, not one storage technology.","Is a vector database required for RAG?",{"id":941,"answer":942,"question":943},"faq4","No. A tool is an external capability. Its definition may be represented in context, and its result may later enter context, but the actual capability executes outside the model.","Are tools the same as context?",{"id":945,"answer":946,"question":947},"faq5","No. Runtime location and inference location are separate. A local desktop application or agent can call a remote model, while a remote application can call an internally hosted model.","Does running an AI client locally mean the model is local?",{"id":949,"answer":950,"question":951},"faq6","The application or runtime security boundary should enforce authorization. A model can request an operation, but model intent should never be treated as sufficient execution authority.","Who should enforce permissions for AI tools?",{"id":953,"answer":954,"question":955},"faq7","Authoritative volatile state should normally remain in the application or domain system that owns it. The AI can receive the relevant state through controlled context or tool access when needed.","Where does current application state belong?","Generative AI system boundaries",{},{"id":959,"data":960,"type":42,"tunes":962},"h-glossary",{"text":961,"level":240},"Glossary",{},{"id":964,"data":965,"type":964,"tunes":989},"glossary",{"title":966,"entries":967},"Core terms",[968,972,974,976,979,981,984,986],{"term":969,"anchor":970,"definition":971},"Generative model","generative-model","An AI model designed to generate derived synthetic content such as text, images, audio, video, code or structured output.",{"term":328,"anchor":327,"definition":973},"The process of selecting relevant information from an external source or store for the current task.",{"term":755,"anchor":562,"definition":975},"Retrieval-Augmented Generation: a pattern in which retrieved external information is supplied to a generative model to improve the current output.",{"term":977,"anchor":566,"definition":978},"Tool","A capability exposed to an AI runtime for reading data, calculating, searching, or performing an external action.",{"term":336,"anchor":335,"definition":980},"The information available to the model for a particular inference step.",{"term":340,"anchor":982,"definition":983},"runtime-orchestrator","The software layer that coordinates model calls, tool calls, task loops, sessions, retries, events or execution environments.",{"term":344,"anchor":343,"definition":985},"The product and domain layer that owns user interaction, authoritative state, permissions, validation, persistence and business behavior.",{"term":625,"anchor":987,"definition":988},"provider","The service or runtime that exposes access to one or more models; provider identity and model identity are separate concerns.",{},{"id":991,"data":992,"type":42,"tunes":994},"h-sources",{"text":993,"level":240},"Primary sources and implementation evidence",{},{"id":996,"data":997,"type":218,"tunes":999},"p-sources-note",{"text":998},"Stable definitions below are anchored in standards\u002Fresearch; fast-moving implementation examples use current official engineering documentation. Aaasaasa AI Client is original implementation evidence and was checked against its codebase\u002Fdocumentation state dated 26 July 2026.",{},{"id":1001,"data":1002,"type":1008,"tunes":1009},"src-nist-profile",{"link":1003,"meta":1004},"https:\u002F\u002Fnvlpubs.nist.gov\u002Fnistpubs\u002Fai\u002FNIST.AI.600-1.pdf",{"image":1005,"title":1006,"description":1007},{"url":325},"NIST AI 600-1 — Generative Artificial Intelligence Profile","NIST's Generative AI profile, including the generative-AI definition and explicit distinction between model-, system-, application- and use-case-level concerns.","linkTool",{},{"id":1011,"data":1012,"type":1008,"tunes":1018},"src-nist-model",{"link":1013,"meta":1014},"https:\u002F\u002Fcsrc.nist.gov\u002Fglossary\u002Fterm\u002Fartificial_intelligence_model",{"image":1015,"title":1016,"description":1017},{"url":325},"NIST — Artificial Intelligence Model","Current NIST glossary definition of an AI model as a component of an information system that produces outputs from inputs using AI techniques.",{},{"id":1020,"data":1021,"type":1008,"tunes":1027},"src-nist-system",{"link":1022,"meta":1023},"https:\u002F\u002Fcsrc.nist.gov\u002Fglossary\u002Fterm\u002Fartificial_intelligence_system",{"image":1024,"title":1025,"description":1026},{"url":325},"NIST — Artificial Intelligence System","Current NIST glossary definition showing that an AI system can include data systems, software, hardware, applications, tools or utilities using AI.",{},{"id":1029,"data":1030,"type":1008,"tunes":1036},"src-rag-paper",{"link":1031,"meta":1032},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2005.11401",{"image":1033,"title":1034,"description":1035},{"url":325},"Lewis et al. — Retrieval-Augmented Generation for Knowledge-Intensive NLP Tasks","The 2020 paper introducing the RAG formulation that combines a generative model with retrieved non-parametric memory.",{},{"id":1038,"data":1039,"type":1008,"tunes":1045},"src-openai-file-search",{"link":1040,"meta":1041},"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Ftools-file-search",{"image":1042,"title":1043,"description":1044},{"url":325},"OpenAI — File Search","Current official documentation for hosted file retrieval in the Responses API using uploaded-file knowledge bases, semantic search and keyword search.",{},{"id":1047,"data":1048,"type":1008,"tunes":1054},"src-openai-functions",{"link":1049,"meta":1050},"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Ffunction-calling",{"image":1051,"title":1052,"description":1053},{"url":325},"OpenAI — Function Calling","Current official documentation describing tool\u002Ffunction calling as the interface between models and external systems, data and actions.",{},{"id":1056,"data":1057,"type":1008,"tunes":1063},"src-anthropic-context",{"link":1058,"meta":1059},"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Feffective-context-engineering-for-ai-agents",{"image":1060,"title":1061,"description":1062},{"url":325},"Anthropic — Effective Context Engineering for AI Agents","Engineering guidance defining context as the token set available during LLM sampling and explaining why context selection is a finite-resource problem.",{},"2.31.6","Generative AI is more than a model. Learn how models, retrieval, tools, context, runtimes and applications fit together in production AI systems.","\u002Fuploads\u002F2026\u002F10\u002Fgenerative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing-1791475411822-pp0dvz.webp","generative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing-1791475411822-pp0dvz","PUBLISHED","2026-10-08T12:00:00.000Z","2026-10-08T16:00:39.350Z","2026-10-08T16:10:51.437Z",{"en":1073,"de":1074,"sr":1075,"es":1076,"fr":1077,"it":1078,"ru":1079,"zh":1080},"\u002Fblog\u002Fgenerative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing","\u002Fde\u002Fblog\u002Fgenerative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing","\u002Fsr\u002Fblog\u002Fgenerative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing","\u002Fes\u002Fblog\u002Fgenerative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing","\u002Ffr\u002Fblog\u002Fgenerative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing","\u002Fit\u002Fblog\u002Fgenerative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing","\u002Fru\u002Fblog\u002Fgenerative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing","\u002Fzh\u002Fblog\u002Fgenerative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing",[1082,1086,1090,1094,1098],{"id":1083,"name":1084,"slug":1085},84,"Policy & Data Boundaries","policy-and-data",{"id":1087,"name":1088,"slug":1089},57,"Data Boundaries","data-boundaries",{"id":1091,"name":1092,"slug":1093},80,"Access & Identity","access-and-identity",{"id":1095,"name":1096,"slug":1097},68,"Risks, Controls & Evidence","risks-and-controls",{"id":1099,"name":1100,"slug":1101},54,"Threat Model","threat-model",{"id":1103,"login":1104,"email":1105,"displayName":1106},"20","rooth8233","aleksandar@stajic.de","Aleksandar Stajić",[1108],{"lang":7,"title":208,"content":210,"contentJson":1109,"excerpt":1065},{"time":212,"blocks":1110,"version":1064},[1111,1114,1117,1120,1123,1126,1129,1132,1135,1138,1141,1153,1156,1159,1179,1182,1185,1188,1191,1194,1197,1200,1203,1206,1209,1212,1215,1218,1221,1224,1227,1230,1233,1236,1239,1242,1245,1248,1251,1254,1257,1260,1263,1275,1278,1281,1298,1301,1304,1307,1310,1313,1316,1328,1331,1334,1346,1349,1352,1372,1375,1378,1389,1392,1395,1398,1409,1412,1415,1418,1421,1424,1427,1430,1433,1436,1439,1442,1445,1448,1451,1454,1457,1460,1463,1466,1469,1480,1483,1495,1498,1501,1506,1511,1516,1521,1526,1531],{"id":215,"data":1112,"type":218,"tunes":1113},{"text":217},{},{"id":221,"data":1115,"type":226,"tunes":1116},{"body":223,"title":224,"variant":225},{},{"id":229,"data":1118,"type":226,"tunes":1119},{"body":231,"title":232,"variant":233},{},{"id":236,"data":1121,"type":241,"tunes":1122},{"title":238,"maxLevel":239,"minLevel":240},{},{"id":244,"data":1124,"type":42,"tunes":1125},{"text":246,"level":240},{},{"id":249,"data":1127,"type":218,"tunes":1128},{"text":251},{},{"id":254,"data":1130,"type":218,"tunes":1131},{"text":256},{},{"id":259,"data":1133,"type":226,"tunes":1134},{"body":261,"title":262,"variant":263},{},{"id":266,"data":1136,"type":42,"tunes":1137},{"text":268,"level":240},{},{"id":271,"data":1139,"type":218,"tunes":1140},{"text":273},{},{"id":276,"data":1142,"type":305,"tunes":1152},{"steps":1143,"title":303,"orientation":304},[1144,1145,1146,1147,1148,1149,1150,1151],{"label":280,"description":281},{"label":283,"description":284},{"label":286,"description":287},{"label":289,"description":290},{"label":292,"description":293},{"label":295,"description":296},{"label":298,"description":299},{"label":301,"description":302},{},{"id":308,"data":1154,"type":218,"tunes":1155},{"text":310},{},{"id":313,"data":1157,"type":42,"tunes":1158},{"text":315,"level":240},{},{"id":318,"data":1160,"type":358,"tunes":1178},{"rows":1161,"title":346,"layout":347,"columns":1174},[1162,1164,1166,1168,1170,1172],{"id":322,"label":323,"values":1163},[325,325,325],{"id":327,"label":328,"values":1165},[325,325,325],{"id":331,"label":332,"values":1167},[325,325,325],{"id":335,"label":336,"values":1169},[325,325,325],{"id":339,"label":340,"values":1171},[325,325,325],{"id":343,"label":344,"values":1173},[325,325,325],[1175,1176,1177],{"id":350,"label":351},{"id":353,"label":354},{"id":356,"label":357},{},{"id":361,"data":1180,"type":42,"tunes":1181},{"text":363,"level":240},{},{"id":366,"data":1183,"type":218,"tunes":1184},{"text":368},{},{"id":371,"data":1186,"type":218,"tunes":1187},{"text":373},{},{"id":376,"data":1189,"type":218,"tunes":1190},{"text":378},{},{"id":381,"data":1192,"type":42,"tunes":1193},{"text":383,"level":240},{},{"id":386,"data":1195,"type":218,"tunes":1196},{"text":388},{},{"id":391,"data":1198,"type":218,"tunes":1199},{"text":393},{},{"id":396,"data":1201,"type":226,"tunes":1202},{"body":398,"title":399,"variant":400},{},{"id":403,"data":1204,"type":409,"tunes":1205},{"url":405,"title":406,"excerpt":407,"ctaLabel":408},{},{"id":412,"data":1207,"type":42,"tunes":1208},{"text":414,"level":240},{},{"id":417,"data":1210,"type":218,"tunes":1211},{"text":419},{},{"id":422,"data":1213,"type":218,"tunes":1214},{"text":424},{},{"id":427,"data":1216,"type":218,"tunes":1217},{"text":429},{},{"id":432,"data":1219,"type":226,"tunes":1220},{"body":434,"title":435,"variant":263},{},{"id":438,"data":1222,"type":42,"tunes":1223},{"text":440,"level":240},{},{"id":443,"data":1225,"type":218,"tunes":1226},{"text":445},{},{"id":448,"data":1228,"type":218,"tunes":1229},{"text":450},{},{"id":453,"data":1231,"type":218,"tunes":1232},{"text":455},{},{"id":458,"data":1234,"type":42,"tunes":1235},{"text":460,"level":240},{},{"id":463,"data":1237,"type":218,"tunes":1238},{"text":465},{},{"id":468,"data":1240,"type":218,"tunes":1241},{"text":470},{},{"id":473,"data":1243,"type":218,"tunes":1244},{"text":475},{},{"id":478,"data":1246,"type":42,"tunes":1247},{"text":480,"level":240},{},{"id":483,"data":1249,"type":218,"tunes":1250},{"text":485},{},{"id":488,"data":1252,"type":218,"tunes":1253},{"text":490},{},{"id":493,"data":1255,"type":226,"tunes":1256},{"body":495,"title":496,"variant":225},{},{"id":499,"data":1258,"type":42,"tunes":1259},{"text":501,"level":240},{},{"id":504,"data":1261,"type":218,"tunes":1262},{"text":506},{},{"id":509,"data":1264,"type":347,"tunes":1274},{"content":1265,"stretched":43,"withHeadings":14},[1266,1267,1268,1269,1270,1271,1272,1273],[513,514,515],[517,328,518],[520,521,522],[524,525,526],[528,529,530],[532,533,534],[536,323,537],[539,540,541],{},{"id":544,"data":1276,"type":218,"tunes":1277},{"text":546},{},{"id":549,"data":1279,"type":42,"tunes":1280},{"text":551,"level":240},{},{"id":554,"data":1282,"type":358,"tunes":1297},{"rows":1283,"title":573,"layout":347,"columns":1292},[1284,1286,1288,1290],{"id":558,"label":559,"values":1285},[325,325,325,325],{"id":562,"label":563,"values":1287},[325,325,325,325],{"id":566,"label":567,"values":1289},[325,325,325,325],{"id":570,"label":571,"values":1291},[325,325,325,325],[1293,1294,1295,1296],{"id":327,"label":328},{"id":331,"label":332},{"id":578,"label":579},{"id":581,"label":582},{},{"id":585,"data":1299,"type":218,"tunes":1300},{"text":587},{},{"id":590,"data":1302,"type":42,"tunes":1303},{"text":592,"level":240},{},{"id":595,"data":1305,"type":226,"tunes":1306},{"body":597,"title":598,"variant":233},{},{"id":601,"data":1308,"type":218,"tunes":1309},{"text":603},{},{"id":606,"data":1311,"type":218,"tunes":1312},{"text":608},{},{"id":611,"data":1314,"type":218,"tunes":1315},{"text":613},{},{"id":616,"data":1317,"type":347,"tunes":1327},{"content":1318,"stretched":43,"withHeadings":14},[1319,1320,1321,1322,1323,1324,1325,1326],[620,621],[323,623],[625,626],[628,629],[631,632],[634,635],[637,638],[344,640],{},{"id":643,"data":1329,"type":226,"tunes":1330},{"body":645,"title":646,"variant":263},{},{"id":649,"data":1332,"type":42,"tunes":1333},{"text":651,"level":240},{},{"id":654,"data":1335,"type":347,"tunes":1345},{"content":1336,"stretched":43,"withHeadings":14},[1337,1338,1339,1340,1341,1342,1343,1344],[658,659],[661,662],[664,665],[667,668],[670,671],[673,674],[676,677],[679,680],{},{"id":683,"data":1347,"type":42,"tunes":1348},{"text":685,"level":240},{},{"id":688,"data":1350,"type":218,"tunes":1351},{"text":690},{},{"id":693,"data":1353,"type":358,"tunes":1371},{"rows":1354,"title":720,"layout":347,"columns":1367},[1355,1357,1359,1361,1363,1365],{"id":697,"label":698,"values":1356},[325,325,325],{"id":701,"label":702,"values":1358},[325,325,325],{"id":705,"label":706,"values":1360},[325,325,325],{"id":709,"label":710,"values":1362},[325,325,325],{"id":713,"label":714,"values":1364},[325,325,325],{"id":717,"label":718,"values":1366},[325,325,325],[1368,1369,1370],{"id":723,"label":724},{"id":726,"label":727},{"id":729,"label":730},{},{"id":733,"data":1373,"type":42,"tunes":1374},{"text":735,"level":240},{},{"id":738,"data":1376,"type":218,"tunes":1377},{"text":740},{},{"id":743,"data":1379,"type":347,"tunes":1388},{"content":1380,"stretched":43,"withHeadings":14},[1381,1382,1383,1384,1385,1386,1387],[747,748,749],[751,752,753],[755,756,757],[759,760,761],[763,764,765],[767,768,769],[771,772,773],{},{"id":776,"data":1390,"type":218,"tunes":1391},{"text":778},{},{"id":781,"data":1393,"type":42,"tunes":1394},{"text":783,"level":240},{},{"id":786,"data":1396,"type":218,"tunes":1397},{"text":788},{},{"id":791,"data":1399,"type":305,"tunes":1408},{"steps":1400,"title":815,"orientation":304},[1401,1402,1403,1404,1405,1406,1407],{"label":795,"description":796},{"label":798,"description":799},{"label":801,"description":802},{"label":804,"description":805},{"label":807,"description":808},{"label":810,"description":811},{"label":813,"description":814},{},{"id":818,"data":1410,"type":42,"tunes":1411},{"text":820,"level":240},{},{"id":823,"data":1413,"type":218,"tunes":1414},{"text":825},{},{"id":828,"data":1416,"type":218,"tunes":1417},{"text":830},{},{"id":833,"data":1419,"type":226,"tunes":1420},{"body":835,"title":836,"variant":263},{},{"id":839,"data":1422,"type":42,"tunes":1423},{"text":841,"level":240},{},{"id":844,"data":1425,"type":218,"tunes":1426},{"text":846},{},{"id":849,"data":1428,"type":409,"tunes":1429},{"url":851,"title":852,"excerpt":853,"ctaLabel":854},{},{"id":857,"data":1431,"type":409,"tunes":1432},{"url":859,"title":860,"excerpt":861,"ctaLabel":862},{},{"id":865,"data":1434,"type":42,"tunes":1435},{"text":867,"level":240},{},{"id":870,"data":1437,"type":218,"tunes":1438},{"text":872},{},{"id":875,"data":1440,"type":218,"tunes":1441},{"text":877},{},{"id":880,"data":1443,"type":218,"tunes":1444},{"text":882},{},{"id":885,"data":1446,"type":42,"tunes":1447},{"text":887,"level":240},{},{"id":890,"data":1449,"type":218,"tunes":1450},{"text":892},{},{"id":895,"data":1452,"type":218,"tunes":1453},{"text":897},{},{"id":900,"data":1455,"type":42,"tunes":1456},{"text":902,"level":240},{},{"id":905,"data":1458,"type":218,"tunes":1459},{"text":907},{},{"id":910,"data":1461,"type":218,"tunes":1462},{"text":912},{},{"id":915,"data":1464,"type":218,"tunes":1465},{"text":917},{},{"id":920,"data":1467,"type":42,"tunes":1468},{"text":922,"level":240},{},{"id":925,"data":1470,"type":925,"tunes":1479},{"items":1471,"title":956},[1472,1473,1474,1475,1476,1477,1478],{"id":929,"answer":930,"question":931},{"id":933,"answer":934,"question":935},{"id":937,"answer":938,"question":939},{"id":941,"answer":942,"question":943},{"id":945,"answer":946,"question":947},{"id":949,"answer":950,"question":951},{"id":953,"answer":954,"question":955},{},{"id":959,"data":1481,"type":42,"tunes":1482},{"text":961,"level":240},{},{"id":964,"data":1484,"type":964,"tunes":1494},{"title":966,"entries":1485},[1486,1487,1488,1489,1490,1491,1492,1493],{"term":969,"anchor":970,"definition":971},{"term":328,"anchor":327,"definition":973},{"term":755,"anchor":562,"definition":975},{"term":977,"anchor":566,"definition":978},{"term":336,"anchor":335,"definition":980},{"term":340,"anchor":982,"definition":983},{"term":344,"anchor":343,"definition":985},{"term":625,"anchor":987,"definition":988},{},{"id":991,"data":1496,"type":42,"tunes":1497},{"text":993,"level":240},{},{"id":996,"data":1499,"type":218,"tunes":1500},{"text":998},{},{"id":1001,"data":1502,"type":1008,"tunes":1505},{"link":1003,"meta":1503},{"image":1504,"title":1006,"description":1007},{"url":325},{},{"id":1011,"data":1507,"type":1008,"tunes":1510},{"link":1013,"meta":1508},{"image":1509,"title":1016,"description":1017},{"url":325},{},{"id":1020,"data":1512,"type":1008,"tunes":1515},{"link":1022,"meta":1513},{"image":1514,"title":1025,"description":1026},{"url":325},{},{"id":1029,"data":1517,"type":1008,"tunes":1520},{"link":1031,"meta":1518},{"image":1519,"title":1034,"description":1035},{"url":325},{},{"id":1038,"data":1522,"type":1008,"tunes":1525},{"link":1040,"meta":1523},{"image":1524,"title":1043,"description":1044},{"url":325},{},{"id":1047,"data":1527,"type":1008,"tunes":1530},{"link":1049,"meta":1528},{"image":1529,"title":1052,"description":1053},{"url":325},{},{"id":1056,"data":1532,"type":1008,"tunes":1535},{"link":1058,"meta":1533},{"image":1534,"title":1061,"description":1062},{"url":325},{},"Post erfolgreich abgerufen",{"items":1538,"source":1621,"manualIds":1622,"manualMatchedIds":1623},[1539,1546,1553,1560,1567,1573,1580,1586,1593,1600,1607,1614],{"id":1540,"slug":1541,"title":1542,"excerpt":1543,"featuredImage":1544,"publishedAt":1545},"470","what-should-an-ai-agent-remember-forget-recompute-or-retrieve-again","What Should an AI Agent Remember, Forget, Recompute or Retrieve Again?","Long-running agents should not remember everything. This article provides a practical lifecycle model for deciding what belongs in durable memory, what should be retrieved again, what is safer to recompute, and what should expire or be superseded.","\u002Fuploads\u002F2026\u002F09\u002Fwhat-should-an-ai-agent-remember-forget-recompute-or-retrieve-again-1790351131087-iehz28.webp","2026-09-25T09:43:00.000Z",{"id":1547,"slug":1548,"title":1549,"excerpt":1550,"featuredImage":1551,"publishedAt":1552},"485","enterprise-ai-architecture-what-changes-when-ai-enters-a-company","Enterprise AI Architecture: What Changes When AI Enters a Company","Enterprise AI architecture explains how AI changes company systems across data authority, identity, permissions, providers, risk, governance, evaluation, compliance and operations.","\u002Fuploads\u002F2026\u002F10\u002Fenterprise-ai-architecture-what-changes-when-ai-enters-a-company-1791478161363-czrwaq.webp","2026-10-08T10:48:00.000Z",{"id":1554,"slug":1555,"title":1556,"excerpt":1557,"featuredImage":1558,"publishedAt":1559},"381","enterprise-grade-multi-tenant-architecture-for-an-international-platform","Enterprise-Grade Multi-Tenant Architecture for an International Platform","Loving Rocks is an enterprise-grade wedding platform designed with a true multi-tenant architecture, isolated databases per tenant, and built-in internationalization for global scalability, security, and long-term operational stability.","\u002Fuploads\u002F2026\u002F01\u002Fenterprise-grade-multi-tenant-architecture-for-an-international-platform-1769789121298-b6v7ak.webp","2026-01-30T12:04:00.000Z",{"id":1561,"slug":1562,"title":1563,"excerpt":1564,"featuredImage":1565,"publishedAt":1566},"476","mcp-vs-a2a-vs-ucp-vs-ap2-vs-a2ui-the-agent-protocol-stack-explained","MCP vs A2A vs UCP vs AP2 vs A2UI: The Agent Protocol Stack Explained","MCP, A2A, UCP, AP2 and A2UI are often presented as competing agent standards. They mostly solve different interoperability problems. This guide maps each protocol to the boundary it actually standardizes—and shows how they can work together in one production system.","\u002Fuploads\u002F2026\u002F09\u002Fmcp-vs-a2a-vs-ucp-vs-ap2-vs-a2ui-the-agent-protocol-stack-explained-1790352625869-2ezle0.webp","2026-09-25T12:09:00.000Z",{"id":1568,"slug":1569,"title":860,"excerpt":1570,"featuredImage":1571,"publishedAt":1572},"480","when-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger","An AI model does not need retrieval for every question. The important problem is knowing when its internal knowledge is no longer enough. The Retrieval Trigger is a practical decision boundary that determines when an AI system should stop relying solely on model knowledge and obtain external evidence before answering.","\u002Fuploads\u002F2026\u002F09\u002Fwhen-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger-1790574991244-f4rpyg.webp","2026-09-28T01:49:00.000Z",{"id":1574,"slug":1575,"title":1576,"excerpt":1577,"featuredImage":1578,"publishedAt":1579},"484","what-is-an-ai-platform-architect-models-data-runtime-security-and-operations","What Is an AI Platform Architect? Models, Data, Runtime, Security and Operations","An AI Platform Architect designs reusable AI foundations across models, providers, retrieval, agents, identity, security, evaluation, observability and operations.","\u002Fuploads\u002F2026\u002F10\u002Fwhat-is-an-ai-platform-architect-models-data-runtime-security-and-operations-1791477229171-ou3zcc.webp","2026-10-08T12:32:00.000Z",{"id":1581,"slug":1582,"title":852,"excerpt":1583,"featuredImage":1584,"publishedAt":1585},"479","where-does-an-llm-get-its-data-rag-data-sources-in-python","An LLM does not magically know your files, databases or APIs. This practical continuation of the RAG series shows, with simple Python, how external data becomes retrievable evidence: from text files and SQL to full-text search, embeddings, context assembly and the final LLM call.","\u002Fuploads\u002F2026\u002F09\u002Fwhere-does-an-llm-get-its-data-rag-data-sources-in-python-1790517200521-nfsi5i.webp","2026-09-27T05:51:00.000Z",{"id":1587,"slug":1588,"title":1589,"excerpt":1590,"featuredImage":1591,"publishedAt":1592},"494","air-gapped-ai-how-ai-systems-work-without-internet-or-cloud-access","Air-Gapped AI: How AI Systems Work Without Internet or Cloud Access","Air-gapped AI runs models, RAG and AI applications inside an isolated security domain without internet or cloud dependencies. Learn how models, data, updates and tools operate offline.","\u002Fuploads\u002F2026\u002F10\u002Fair-gapped-ai-how-ai-systems-work-without-internet-or-cloud-access-1791487983978-e6xqf0.webp","2026-10-08T11:32:00.000Z",{"id":1594,"slug":1595,"title":1596,"excerpt":1597,"featuredImage":1598,"publishedAt":1599},"468","ai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context","AI Agent Memory Is Not RAG: How to Separate Memory, Retrieval, State and Context","Agent memory, RAG, state, and context are often used as if they were interchangeable. They are not. This practical architecture model separates the four layers, shows where each belongs, and explains what breaks when systems collapse them into one.","\u002Fuploads\u002F2026\u002F09\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context-1790350560308-np0xy6.webp","2026-09-25T11:34:00.000Z",{"id":1601,"slug":1602,"title":1603,"excerpt":1604,"featuredImage":1605,"publishedAt":1606},"467","the-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers","The Answer Validity Boundary: The Missing Layer Between Relevance and Reliable AI Answers","A source can be relevant, authoritative and still be wrong for the question being asked. The missing layer is applicability: the conditions under which an answer holds, and the changes that force it to be reconsidered. This article introduces the Answer Validity Boundary as a source-design pattern for humans, AI search and RAG systems.","\u002Fuploads\u002F2026\u002F09\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers-1790272901306-1g5jly.webp","2026-09-24T11:59:00.000Z",{"id":1608,"slug":1609,"title":1610,"excerpt":1611,"featuredImage":1612,"publishedAt":1613},"492","mcp-explained-what-it-connects-what-it-does-not-do-and-where-it-fits","MCP Explained: What It Connects, What It Does Not Do and Where It Fits","Model Context Protocol connects AI applications to external tools, resources and prompts through a standard client-server boundary. Learn what MCP does, what it does not do, and where it fits in agent architecture.","\u002Fuploads\u002F2026\u002F10\u002Fmcp-explained-what-it-connects-what-it-does-not-do-and-where-it-fits-1791486640275-7ub1cq.webp","2026-10-08T15:09:00.000Z",{"id":1615,"slug":1616,"title":1617,"excerpt":1618,"featuredImage":1619,"publishedAt":1620},"489","agentic-ai-explained-when-an-ai-system-can-plan-use-tools-and-act","Agentic AI Explained: When an AI System Can Plan, Use Tools and Act","Agentic AI uses models inside multi-step execution loops where they can choose tools, observe results, update state and adapt their next action within explicit runtime and permission boundaries.","\u002Fuploads\u002F2026\u002F10\u002Fagentic-ai-explained-when-an-ai-system-can-plan-use-tools-and-act-1791481499084-wnji2a.webp","2026-10-08T11:43:00.000Z","fallback",[],[]]