[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"portal-settings:stajic:en":3,"public-menus:all":38,"post:mlops-vs-llmops-what-changes-when-the-model-is-an-llm:en":205,"related:post:mlops-vs-llmops-what-changes-when-the-model-is-an-llm:en:1":2281},{"statusCode":4,"data":5,"message":37},200,{"tenantId":6,"lang":7,"defaultLang":8,"siteUrl":9,"contactEmail":10,"brandName":11,"logoUrl":12,"siteName":11,"siteDescription":13,"ogImage":10,"robotsIndex":14,"socialLinks":10,"reservedSlugs":10,"seoPolicy":15},"stajic","en","de","https:\u002F\u002Fstajic.de",null,"Stajic Platform","\u002FLogo_Planet.svg","Stajic Portal",true,{"branding":16,"relatedContent":17,"crossDomainLinks":18},{"logoUrl":12},{"enabled":14},[19,22,25,28,31,34],{"url":20,"label":21,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Ffigure.rocks","figure.rocks",{"url":23,"label":24,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Floving.rocks","loving.rocks",{"url":26,"label":27,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.com","bazify.com",{"url":29,"label":30,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.de","bazify.de",{"url":32,"label":33,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.at","bazify.at",{"url":35,"label":36,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.ba","bazify.ba","Portal settings resolved",[39,45],{"id":40,"name":41,"location":42,"isActive":14,"isDefault":43,"items":44},1,"main-navigation","header",false,[],{"id":46,"name":47,"location":48,"isActive":14,"isDefault":14,"items":49},4,"main-menu","sidebar",[50,66,79,93,103,118,133],{"id":51,"title":52,"url":60,"target":61,"icon":62,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":64,"portfolioId":10,"children":65},"item-18",{"de":53,"en":54,"es":55,"fr":56,"it":54,"ru":57,"sr":58,"zh":59},"Startseite","Home","Inicio","Accueil","Главная","Почетна","首页","\u002Ffull-stack-web-developer-munich-performance-seo-and-maintainable-builds","_self","i-lucide-home","page",111,[],{"id":67,"title":68,"url":75,"target":61,"icon":76,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":77,"portfolioId":10,"children":78},"item-22",{"de":69,"en":69,"es":70,"fr":69,"it":71,"ru":72,"sr":73,"zh":74},"Vision","Visión","Visione","Видение","Визија","想象","\u002Fueber-uns-webdesign-muenchen-webaplikation","i-lucide-eye",113,[],{"id":80,"title":81,"url":89,"target":61,"icon":90,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":91,"portfolioId":10,"children":92},"item-19",{"de":82,"en":83,"es":84,"fr":83,"it":85,"ru":86,"sr":87,"zh":88},"Leistungen","Services","Servicios","Servizi","Услуги","Услуге","服务","\u002Fservices-dienstleistungen-muenchen","i-lucide-wrench",116,[],{"id":94,"title":95,"url":99,"target":61,"icon":100,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":101,"portfolioId":10,"children":102},"item-23",{"de":96,"en":96,"es":96,"fr":96,"it":96,"ru":97,"sr":97,"zh":98},"Blog","Блог","博客","\u002Fblog","i-lucide-book-open",112,[],{"id":104,"title":105,"url":114,"target":61,"icon":115,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":116,"portfolioId":10,"children":117},"item-32",{"de":106,"en":107,"es":108,"fr":109,"it":110,"ru":111,"sr":112,"zh":113},"Neue Technologien","New Technologies","Nuevas tecnologías","Nouvelles technologies","Nuove tecnologie","Новые технологии","Нове технологије","新技术！","\u002Fneue-webtechnologien","i-lucide-sparkles",122,[],{"id":119,"title":120,"url":129,"target":61,"icon":130,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":131,"portfolioId":10,"children":132},"item-20",{"de":121,"en":122,"es":123,"fr":124,"it":125,"ru":126,"sr":127,"zh":128},"Kontakt","Contact us!","Contacto","Contact","Contatto","Контакт","Контактирајте нас","联系我们！","\u002Fcontact","i-lucide-mail",115,[],{"id":134,"title":135,"url":144,"target":61,"icon":145,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":147},"item-21",{"de":136,"en":137,"es":138,"fr":139,"it":140,"ru":141,"sr":142,"zh":143},"Unsere Arbeit","Our Work","Nuestro trabajo","Nos réalisations","I nostri lavori","Наши работы","Наши радови","文件夹","\u002Fportfolio","i-lucide-briefcase",114,[148,161,175,181,193],{"id":149,"title":150,"url":144,"target":61,"icon":159,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":160},"item-24",{"de":151,"en":152,"es":153,"fr":154,"it":155,"ru":156,"sr":157,"zh":158},"Alle Projekte","All Projects","Todos los proyectos","Tous les projets","Tutti i progetti","Все проекты","Сви пројекти","所有项目","i-lucide-grid-3x3",[],{"id":162,"title":163,"url":171,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":174},"item-29",{"de":164,"en":165,"es":166,"fr":167,"it":168,"ru":169,"sr":170,"zh":143},"Local Roots, Global Reach","Local Roots - Global Reach","Empresa local ","Entreprise locale","Azienda locale","Местная компания","Локално предузеће глобално тржиште","\u002Fportfolio\u002Flocal-roots-global-reach-communication-media-systems-for-modern-business","i-lucide-folder","custom",[],{"id":176,"title":177,"url":179,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":180},"item-28",{"de":178,"en":178,"es":178,"fr":178,"it":178,"ru":178,"sr":178,"zh":178},"Solr Suggester","\u002Fportfolio\u002Fsolr-fuzzy-suggester-und-solr-infix-suggester-abfrage-ueber-ajax-und-filterung",[],{"id":182,"title":183,"url":191,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":192},"item-27",{"de":184,"en":185,"es":186,"fr":187,"it":188,"ru":189,"sr":190,"zh":185},"Firmenwebseite SEO","Company Website SEO","Sitio web corporativo SEO","Site web d’entreprise SEO","Sito web aziendale SEO","Корпоративный сайт SEO","Пословна веб-страница SEO","\u002Fportfolio\u002Fseo-sem-branding-mobile-webseite-muenchen",[],{"id":194,"title":195,"url":203,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":204},"item-31",{"de":196,"en":197,"es":198,"fr":199,"it":200,"ru":201,"sr":202,"zh":197},"Digitalisierungsportal","Digitalization Portal","Portal de digitalización","Portail de numérisation","Portale di digitalizzazione","Портал цифровизации","Портал за дигитализацију","\u002Fportfolio\u002Fdigitalisierungsportal-archiv-museum-bibliothek-ead-lido-mets-mods",[],{"statusCode":4,"data":206,"message":2280},{"id":207,"title":208,"slug":209,"content":210,"contentJson":211,"excerpt":1559,"featuredImage":1560,"featuredImageAlt":1561,"featuredImageCaption":10,"featuredImageTitle":10,"featuredImageCopyright":10,"featuredImageAuthor":10,"featuredImageSourceUrl":10,"featuredImageLicense":10,"featuredImageIsAiGenerated":43,"status":1562,"publishedAt":1563,"createdAt":1564,"updatedAt":1565,"seoLocalePaths":1566,"categories":1575,"author":1592,"translations":1597},"493","MLOps vs LLMOps: What Changes When the Model Is an LLM","mlops-vs-llmops-what-changes-when-the-model-is-an-llm","{\"time\":1791487321430,\"blocks\":[{\"id\":\"intro\",\"type\":\"paragraph\",\"data\":{\"text\":\"MLOps is the engineering discipline for reliably developing, deploying, versioning and operating machine-learning systems; LLMOps extends that discipline to applications built around large language models, where production behavior depends not only on a model artifact but also on prompts, context, retrieval, provider\u002Fmodel versions, tool calls, safety controls and evaluation pipelines. LLMOps does not replace MLOps. It changes the operational unit from “a model plus serving pipeline” toward “an evolving LLM application whose behavior emerges from several independently changing components.”\"},\"tunes\":{}},{\"id\":\"direct\",\"type\":\"callout\",\"data\":{\"variant\":\"info\",\"title\":\"Direct answer\",\"body\":\"\u003Cstrong>MLOps operates ML systems. LLMOps operates LLM applications.\u003C\u002Fstrong>\u003Cbr>\u003Cbr>Classical MLOps commonly centers on data pipelines, training, validation, model registry, deployment, drift and retraining. LLMOps keeps those disciplines where relevant, but often adds prompt\u002Fcontext versioning, model\u002Fprovider abstraction, RAG indexes, agent\u002Ftool traces, semantic evaluations, safety tests, token\u002Fcost monitoring and regression testing across rapidly changing model snapshots.\"},\"tunes\":{}},{\"id\":\"boundary\",\"type\":\"callout\",\"data\":{\"variant\":\"warning\",\"title\":\"LLMOps is not just prompt management\",\"body\":\"A production LLM application can fail even when the prompt is unchanged: the provider can change a model snapshot, a RAG corpus can become stale, a reranker can regress, tool permissions can change, context assembly can drop evidence, or an agent can take a wrong trajectory. LLMOps therefore has to observe and version the system around the model, not only prompt text.\"},\"tunes\":{}},{\"id\":\"term-note\",\"type\":\"callout\",\"data\":{\"variant\":\"note\",\"title\":\"Terminology boundary\",\"body\":\"\u003Cstrong>LLMOps\u003C\u002Fstrong>, \u003Cstrong>GenAIOps\u003C\u002Fstrong> and related terms are widely used engineering labels, but they are not one universal formal standard with a single canonical lifecycle. Microsoft currently describes GenAIOps as “sometimes called LLMOps,” while MLflow groups operational tooling around agents and LLM applications. This article uses LLMOps as a practical architecture term for operating production systems whose behavior materially depends on LLMs.\"},\"tunes\":{}},{\"id\":\"current\",\"type\":\"callout\",\"data\":{\"variant\":\"note\",\"title\":\"Current-source note — 8 October 2026\",\"body\":\"The operational surface is changing quickly. OpenAI currently recommends pinning model snapshots and running evals because prompting behavior can change between snapshots, and several older platform-specific prompt\u002Feval surfaces are being retired in 2026. The stable architectural lesson is to keep prompts, tests and evals portable and versioned with the application rather than depend on one provider's dashboard object model.\"},\"tunes\":{}},{\"id\":\"toc\",\"type\":\"tableOfContents\",\"data\":{\"title\":\"Contents\",\"minLevel\":2,\"maxLevel\":3},\"tunes\":{}},{\"id\":\"h-meaning\",\"type\":\"header\",\"data\":{\"text\":\"What MLOps really means\",\"level\":2},\"tunes\":{}},{\"id\":\"p-mlops-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"MLOps applies software-engineering and operational discipline to machine-learning systems. The production challenge is broader than training a model: data collection, data validation, experimentation, reproducibility, model evaluation, deployment, infrastructure and monitoring all have to work together.\"},\"tunes\":{}},{\"id\":\"p-mlops-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Google's MLOps architecture guidance frames the discipline around continuous integration, continuous delivery and continuous training. CI validates not only code but also data, schemas and models; CD deploys ML pipelines and prediction services; CT can retrain and redeploy models as data or implementations change.\"},\"tunes\":{}},{\"id\":\"p-mlops-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"AWS guidance adds the same operational concerns from another angle: model lineage, model\u002Fversion traceability, drift monitoring and production-quality monitoring are core parts of keeping ML systems reliable after deployment.\"},\"tunes\":{}},{\"id\":\"h-llmops\",\"type\":\"header\",\"data\":{\"text\":\"What changes when the model is an LLM\",\"level\":2},\"tunes\":{}},{\"id\":\"p-llmops-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Large language models change the production problem because the application often does not own the complete model-training lifecycle. A team may call a hosted model API, run an open model locally, switch between providers or use several models for different tasks.\"},\"tunes\":{}},{\"id\":\"p-llmops-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The model is therefore only one versioned dependency inside a larger behavioral system. Prompts, retrieval results, context order, tools, model snapshot, temperature\u002Freasoning settings, safety filters and runtime orchestration can all change the output.\"},\"tunes\":{}},{\"id\":\"p-llmops-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"This creates a broader operational question: which combination of model, context, data, prompt, tools and runtime produced this behavior? LLMOps exists to make that question answerable and the answer reproducible enough for engineering work.\"},\"tunes\":{}},{\"id\":\"h-simple\",\"type\":\"header\",\"data\":{\"text\":\"The simplest example\",\"level\":2},\"tunes\":{}},{\"id\":\"p-simple-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Suppose an application answers internal policy questions.\"},\"tunes\":{}},{\"id\":\"p-simple-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"In a classical ML framing, you might version a trained classifier, deploy it and monitor prediction quality. In an LLM application, the answer might depend on a hosted model snapshot, a system prompt, an embedding model, a vector index, retrieval filters, a reranker and the final selected context.\"},\"tunes\":{}},{\"id\":\"p-simple-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"Changing any one of those components can change the final answer even though the application endpoint and user question stay identical.\"},\"tunes\":{}},{\"id\":\"simple-flow\",\"type\":\"processFlow\",\"data\":{\"title\":\"A typical LLMOps release path\",\"orientation\":\"auto\",\"steps\":[{\"label\":\"1. Change one component\",\"description\":\"Prompt, model, provider, retrieval setting, tool schema or application code changes.\"},{\"label\":\"2. Run deterministic tests\",\"description\":\"Validate schemas, permissions, tool contracts, retrieval filters and application behavior.\"},{\"label\":\"3. Run behavioral evals\",\"description\":\"Compare representative outputs, retrieval quality and agent\u002Ftool trajectories against acceptance criteria.\"},{\"label\":\"4. Compare cost and latency\",\"description\":\"Measure token use, model calls, retrieval\u002Ftool overhead and response latency.\"},{\"label\":\"5. Deploy controlled version\",\"description\":\"Ship the concrete application configuration with model\u002Fprovider versions recorded.\"},{\"label\":\"6. Trace production behavior\",\"description\":\"Capture relevant model, retrieval, tool and runtime spans.\"},{\"label\":\"7. Evaluate production traces\",\"description\":\"Sample real executions for quality, grounding, safety and task success.\"},{\"label\":\"8. Roll back or iterate\",\"description\":\"Use regression evidence and operational signals to decide the next release.\"}]},\"tunes\":{}},{\"id\":\"h-stops\",\"type\":\"header\",\"data\":{\"text\":\"Where the simple example stops\",\"level\":2},\"tunes\":{}},{\"id\":\"p-stops-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Some LLM systems still train or fine-tune their own models, so traditional MLOps practices such as training pipelines, model registry and data lineage remain directly relevant.\"},\"tunes\":{}},{\"id\":\"p-stops-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Other systems use only external foundation-model APIs and never run continuous training. Their main operational workload is application evaluation, model\u002Fprovider change management, prompt\u002Fcontext versioning, retrieval quality and observability.\"},\"tunes\":{}},{\"id\":\"p-stops-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"There is therefore no single universal “LLMOps pipeline.” The exact lifecycle depends on whether you train, fine-tune, self-host, retrieve external knowledge, run agents or depend on managed model APIs.\"},\"tunes\":{}},{\"id\":\"h-compare\",\"type\":\"header\",\"data\":{\"text\":\"MLOps vs LLMOps\",\"level\":2},\"tunes\":{}},{\"id\":\"main-comparison\",\"type\":\"comparison\",\"data\":{\"title\":\"What stays the same and what expands\",\"layout\":\"table\",\"columns\":[{\"id\":\"mlops\",\"label\":\"MLOps\"},{\"id\":\"llmops\",\"label\":\"LLMOps\"}],\"rows\":[{\"id\":\"unit\",\"label\":\"Primary operational unit\",\"values\":[\"\",\"\"]},{\"id\":\"model\",\"label\":\"Model ownership\",\"values\":[\"\",\"\"]},{\"id\":\"change\",\"label\":\"Typical change\",\"values\":[\"\",\"\"]},{\"id\":\"eval\",\"label\":\"Evaluation\",\"values\":[\"\",\"\"]},{\"id\":\"monitor\",\"label\":\"Production monitoring\",\"values\":[\"\",\"\"]},{\"id\":\"training\",\"label\":\"Continuous training\",\"values\":[\"\",\"\"]},{\"id\":\"registry\",\"label\":\"Versioned artifacts\",\"values\":[\"\",\"\"]},{\"id\":\"rollback\",\"label\":\"Rollback target\",\"values\":[\"\",\"\"]}]},\"tunes\":{}},{\"id\":\"h-extension\",\"type\":\"header\",\"data\":{\"text\":\"LLMOps extends MLOps rather than replacing it\",\"level\":2},\"tunes\":{}},{\"id\":\"p-extension-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The core operational principles do not disappear: source control, CI\u002FCD, reproducibility, lineage, deployment controls, monitoring, rollback and measurable acceptance criteria remain essential.\"},\"tunes\":{}},{\"id\":\"p-extension-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The extension is that more behavior-defining artifacts now sit outside the model weights. A managed foundation model can change behavior through snapshot upgrades, while application output can change through prompt or retrieval changes without any model retraining.\"},\"tunes\":{}},{\"id\":\"p-extension-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"This is why the useful hierarchy is usually DevOps → MLOps → LLMOps\u002FGenAIOps as increasingly specialized operational concerns, not three mutually exclusive practices.\"},\"tunes\":{}},{\"id\":\"h-artifacts\",\"type\":\"header\",\"data\":{\"text\":\"What has to be versioned in LLMOps?\",\"level\":2},\"tunes\":{}},{\"id\":\"artifact-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Artifact\",\"Why it matters\"],[\"Application code\",\"Defines orchestration, validation, retries and business behavior\"],[\"Model family + snapshot\u002Fversion\",\"Different snapshots can produce different behavior\"],[\"Provider \u002F endpoint\",\"Changes data flow, latency, limits, pricing and availability\"],[\"Prompt\u002Finstruction code\",\"Changes model behavior even with same model\"],[\"Generation\u002Freasoning parameters\",\"Can alter determinism, latency, depth and cost\"],[\"Eval dataset\",\"Defines what “good enough” is tested against\"],[\"Scorers \u002F graders\",\"Define how quality is measured\"],[\"Embedding model\",\"Changes vector representation and retrieval behavior\"],[\"Chunking\u002Findex configuration\",\"Changes what can be retrieved\"],[\"Reranker \u002F retrieval fusion\",\"Changes result ordering\"],[\"Tool schemas\",\"Change what the model can request and how\"],[\"Permission profile\",\"Changes what tool actions may actually execute\"],[\"Context assembly rules\",\"Change what evidence and state reach the model\"],[\"Safety\u002Fguardrail configuration\",\"Changes allowed or blocked behavior\"]]},\"tunes\":{}},{\"id\":\"h-model-version\",\"type\":\"header\",\"data\":{\"text\":\"Model snapshots become release dependencies\",\"level\":2},\"tunes\":{}},{\"id\":\"p-model-version-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"With hosted LLMs, the team may not control model training, but it still controls which model or snapshot the application calls.\"},\"tunes\":{}},{\"id\":\"p-model-version-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"OpenAI's current API guidance explicitly warns that prompting behavior can change between model snapshots and recommends pinning production applications to specific snapshots where consistency matters, then running evals when upgrading.\"},\"tunes\":{}},{\"id\":\"p-model-version-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The operational consequence is straightforward: model upgrades should be treated as application releases, not invisible infrastructure maintenance.\"},\"tunes\":{}},{\"id\":\"h-provider\",\"type\":\"header\",\"data\":{\"text\":\"Provider lifecycle becomes part of operations\",\"level\":2},\"tunes\":{}},{\"id\":\"p-provider-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"LLM applications often depend on provider rate limits, deprecation schedules, API semantics, context limits, data-handling rules and pricing.\"},\"tunes\":{}},{\"id\":\"p-provider-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"A provider can deprecate a model while your application code remains unchanged. OpenAI's current deprecation schedule, for example, includes 2026 retirement dates for older model snapshots and platform surfaces.\"},\"tunes\":{}},{\"id\":\"p-provider-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"LLMOps therefore needs provider lifecycle tracking, migration testing and fallback decisions in addition to model-quality monitoring.\"},\"tunes\":{}},{\"id\":\"h-prompt\",\"type\":\"header\",\"data\":{\"text\":\"Prompts behave like production code\",\"level\":2},\"tunes\":{}},{\"id\":\"p-prompt-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Prompts are executable behavioral configuration. Small changes can alter output quality, tool selection and policy interpretation.\"},\"tunes\":{}},{\"id\":\"p-prompt-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"OpenAI's current guidance recommends storing production prompts in application code, reviewing prompt changes through pull requests, using typed inputs and covering changes with tests and evaluation checks.\"},\"tunes\":{}},{\"id\":\"p-prompt-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"That makes prompt versioning less like editing marketing copy and more like changing a function whose output is probabilistic and model-dependent.\"},\"tunes\":{}},{\"id\":\"h-context\",\"type\":\"header\",\"data\":{\"text\":\"Context engineering becomes an operational concern\",\"level\":2},\"tunes\":{}},{\"id\":\"p-context-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The production model rarely receives only a static prompt. It may receive conversation history, retrieved documents, tool outputs, memory, current application state and policy instructions.\"},\"tunes\":{}},{\"id\":\"p-context-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"LLMOps must therefore observe context assembly: which evidence was selected, which state version was current, whether truncation occurred and whether important instructions survived compaction.\"},\"tunes\":{}},{\"id\":\"p-context-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"A model regression and a context regression can look identical at the final answer. Tracing the actual context path is what lets the team separate them.\"},\"tunes\":{}},{\"id\":\"h-rag\",\"type\":\"header\",\"data\":{\"text\":\"RAG creates its own operational lifecycle\",\"level\":2},\"tunes\":{}},{\"id\":\"p-rag-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A RAG system introduces a second production pipeline beside model inference: ingestion, extraction, chunking, metadata, embeddings, indexes, retrieval, reranking and context selection.\"},\"tunes\":{}},{\"id\":\"p-rag-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The knowledge corpus can change every day even when the model and prompt do not. A stale index or broken metadata filter can therefore degrade answer quality without any model drift.\"},\"tunes\":{}},{\"id\":\"p-rag-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"LLMOps for RAG should track corpus\u002Findex version, embedding model, chunking policy, retrieval configuration, source freshness and retrieval metrics separately from generation quality.\"},\"tunes\":{}},{\"id\":\"ref-rag-diagnostic\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method\",\"title\":\"RAG Failed — But Which Layer Actually Failed? A Diagnostic Method\",\"excerpt\":\"A production LLM pipeline needs separate observability for source coverage, retrieval, ranking, context assembly and generation.\",\"ctaLabel\":\"Read the RAG diagnostic method\"},\"tunes\":{}},{\"id\":\"h-evals\",\"type\":\"header\",\"data\":{\"text\":\"Evals replace “looks good to me” with release evidence\",\"level\":2},\"tunes\":{}},{\"id\":\"p-eval-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Generative outputs are often open-ended, so exact-match tests are insufficient for many tasks. LLMOps adds evaluation datasets and scorers that can measure task success, correctness, safety, groundedness, style or domain-specific acceptance criteria.\"},\"tunes\":{}},{\"id\":\"p-eval-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"MLflow's current GenAI evaluation stack supports versioned evaluation datasets, prompt\u002Fmodel comparisons, custom scorers and evaluation over complete traces.\"},\"tunes\":{}},{\"id\":\"p-eval-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The strongest practice is evaluation-driven development: define representative cases and acceptance criteria before or alongside changes, then compare releases against the same evidence.\"},\"tunes\":{}},{\"id\":\"eval-rule\",\"type\":\"callout\",\"data\":{\"variant\":\"success\",\"title\":\"Behavioral changes need behavioral tests\",\"body\":\"A deployment should not be considered equivalent merely because the API contract still works. If the prompt, model, retrieval or tools changed, the behavioral regression suite should run again.\"},\"tunes\":{}},{\"id\":\"h-judges\",\"type\":\"header\",\"data\":{\"text\":\"LLM-as-a-judge is useful but not ground truth\",\"level\":2},\"tunes\":{}},{\"id\":\"p-judge-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"LLM judges can scale evaluation for qualities that are expensive to encode as deterministic assertions, such as relevance, tone or groundedness.\"},\"tunes\":{}},{\"id\":\"p-judge-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"However, the judge is another model with its own bias, version and prompt. Judge configuration should therefore be versioned and calibrated against human or deterministic reference cases where consequence matters.\"},\"tunes\":{}},{\"id\":\"p-judge-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"A production eval can mix deterministic checks, reference-based metrics, model judges and human review rather than asking one metric to represent every quality dimension.\"},\"tunes\":{}},{\"id\":\"h-tracing\",\"type\":\"header\",\"data\":{\"text\":\"Tracing becomes more important than endpoint logs\",\"level\":2},\"tunes\":{}},{\"id\":\"p-trace-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Traditional API logs can tell you that a request took two seconds and returned HTTP 200. They cannot tell you which retrieved chunks were selected, which tool the agent called or which model span consumed most tokens.\"},\"tunes\":{}},{\"id\":\"p-trace-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"MLflow's current GenAI tracing captures prompts, retrievals, tool calls and application spans, and its production evaluation flow can score intermediate trajectory information rather than only final text.\"},\"tunes\":{}},{\"id\":\"p-trace-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"This is a major LLMOps shift: observability follows the behavioral graph of the application, not only the serving endpoint.\"},\"tunes\":{}},{\"id\":\"h-agent\",\"type\":\"header\",\"data\":{\"text\":\"Agents expand LLMOps into runtime operations\",\"level\":2},\"tunes\":{}},{\"id\":\"p-agent-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"An agentic application can perform several model calls, tool invocations and state transitions before producing a result.\"},\"tunes\":{}},{\"id\":\"p-agent-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Operating agents therefore requires step counts, tool-call traces, permission denials, retries, loop detection, human approvals and verified final state in addition to ordinary model latency and token metrics.\"},\"tunes\":{}},{\"id\":\"p-agent-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"A correct final answer can hide a bad trajectory, so agent evaluation must inspect the path as well as the result.\"},\"tunes\":{}},{\"id\":\"ref-agent-reliability\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough\",\"title\":\"AI Agent Reliability: Why the Final Answer Is Not Enough\",\"excerpt\":\"Why agent production evaluation must include tool calls, state transitions, approvals and recoverability.\",\"ctaLabel\":\"Read the agent reliability article\"},\"tunes\":{}},{\"id\":\"h-cost\",\"type\":\"header\",\"data\":{\"text\":\"Tokens, model calls and context become cost variables\",\"level\":2},\"tunes\":{}},{\"id\":\"p-cost-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Classical ML inference cost is often dominated by serving infrastructure or per-prediction compute. LLM applications can add provider token pricing, repeated agent calls, embedding calls, reranking and tool\u002Fruntime overhead.\"},\"tunes\":{}},{\"id\":\"p-cost-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Cost therefore has to be attributed to task or trace, not only to one endpoint. A workflow that makes eight hidden model calls can be functionally correct but operationally unacceptable.\"},\"tunes\":{}},{\"id\":\"p-cost-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"Latency behaves the same way: model latency, retrieval, reranking and external tools compose into end-to-end user latency.\"},\"tunes\":{}},{\"id\":\"h-cache\",\"type\":\"header\",\"data\":{\"text\":\"Caching becomes semantic, not only technical\",\"level\":2},\"tunes\":{}},{\"id\":\"p-cache-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"LLM systems can cache prompts, embeddings, retrieval results or full responses, but the cache key must reflect the semantics that can change the result.\"},\"tunes\":{}},{\"id\":\"p-cache-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"A response cache that ignores model version, tenant, permissions or source freshness can return a technically valid but semantically invalid answer.\"},\"tunes\":{}},{\"id\":\"p-cache-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"LLMOps therefore treats cache invalidation as part of model\u002Fcontext\u002Fdata versioning rather than only infrastructure optimization.\"},\"tunes\":{}},{\"id\":\"h-safety\",\"type\":\"header\",\"data\":{\"text\":\"Safety and permissions become release criteria\",\"level\":2},\"tunes\":{}},{\"id\":\"p-safety-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Generative systems can produce unbounded text and agents can trigger external actions. Safety testing therefore sits closer to ordinary CI\u002FCD than in many classical predictive ML systems.\"},\"tunes\":{}},{\"id\":\"p-safety-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Permission checks, prompt-injection tests, tenant-isolation tests and side-effect approvals should be reproducible regression tests where those risks exist.\"},\"tunes\":{}},{\"id\":\"p-safety-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The model may suggest an operation, but the runtime still has to enforce authorization. LLMOps owns the evidence that those controls continue to work after model, prompt or tool changes.\"},\"tunes\":{}},{\"id\":\"h-ci\",\"type\":\"header\",\"data\":{\"text\":\"What CI looks like in LLMOps\",\"level\":2},\"tunes\":{}},{\"id\":\"ci-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"CI layer\",\"Example checks\"],[\"Code\",\"Unit tests, type checks, schema validation\"],[\"Prompts\",\"Template rendering, required variables, policy text, snapshot review\"],[\"Models\u002Fproviders\",\"Compatibility, output schema, capability and regression tests\"],[\"RAG\",\"Chunking fixtures, filter tests, Recall@k, reranker regression\"],[\"Tools\",\"Input\u002Foutput schema tests, permission tests, idempotency tests\"],[\"Agents\",\"Trajectory fixtures, loop limits, handoff\u002Ftool-selection tests\"],[\"Security\",\"Prompt injection, unauthorized tools, cross-tenant negative tests\"],[\"Behavioral evals\",\"Task success, correctness, grounding, safety, domain criteria\"],[\"Operational\",\"Latency, token\u002Fcost budgets, timeout\u002Ffallback behavior\"]]},\"tunes\":{}},{\"id\":\"h-cd\",\"type\":\"header\",\"data\":{\"text\":\"What CD looks like in LLMOps\",\"level\":2},\"tunes\":{}},{\"id\":\"p-cd-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A production release may deploy no new model artifact at all. It may simply ship a new prompt, retrieval configuration, tool set or provider mapping.\"},\"tunes\":{}},{\"id\":\"p-cd-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The release bundle should therefore identify the complete behavior-defining configuration rather than only the application container image.\"},\"tunes\":{}},{\"id\":\"p-cd-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"Feature flags, staged rollout, shadow evaluation, canary traffic and rollback are useful because LLM behavior can regress in ways that static contract tests do not detect.\"},\"tunes\":{}},{\"id\":\"h-ct\",\"type\":\"header\",\"data\":{\"text\":\"Continuous training becomes optional; continuous evaluation becomes central\",\"level\":2},\"tunes\":{}},{\"id\":\"p-ct-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Traditional MLOps often emphasizes continuous training when new data or drift justifies retraining.\"},\"tunes\":{}},{\"id\":\"p-ct-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Many LLM applications never train the foundation model. Their equivalent continuous loop is continuous evaluation: collect failures and representative production cases, add them to evaluation datasets, test candidate prompt\u002Fmodel\u002Fretrieval changes and redeploy only when evidence improves.\"},\"tunes\":{}},{\"id\":\"p-ct-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"Fine-tuning can reintroduce a training lifecycle, but it should sit inside the same broader evaluation and release process.\"},\"tunes\":{}},{\"id\":\"h-monitor\",\"type\":\"header\",\"data\":{\"text\":\"What should be monitored in production?\",\"level\":2},\"tunes\":{}},{\"id\":\"monitor-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Signal class\",\"Examples\"],[\"System health\",\"Errors, timeouts, endpoint availability\"],[\"Model\u002Fprovider\",\"Model ID, snapshot, rate limits, provider errors\"],[\"Latency\",\"End-to-end, model, retrieval, tool and reranker spans\"],[\"Cost\",\"Input\u002Foutput tokens, embeddings, tool\u002FAPI spend\"],[\"Quality\",\"Sampled task success, correctness, relevance, groundedness\"],[\"RAG\",\"Retrieval recall proxies, empty retrieval, stale sources, citation coverage\"],[\"Agents\",\"Tool selection, retries, loops, handoffs, approval frequency\"],[\"Security\",\"Denied actions, prompt-injection indicators, tenant-boundary failures\"],[\"User feedback\",\"Corrections, abandonment, escalation, explicit ratings\"],[\"Change drift\",\"Provider\u002Fmodel\u002Fconfig changes relative to approved release\"]]},\"tunes\":{}},{\"id\":\"h-prod-eval\",\"type\":\"header\",\"data\":{\"text\":\"Production traces can become evaluation data\",\"level\":2},\"tunes\":{}},{\"id\":\"p-prod-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"One of the most useful modern LLMOps patterns is to turn sampled production traces into evaluation records.\"},\"tunes\":{}},{\"id\":\"p-prod-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"MLflow currently supports retrieving production traces and scoring not only outputs but intermediate spans such as retrieval or tool-call trajectories.\"},\"tunes\":{}},{\"id\":\"p-prod-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"This closes the loop between observability and development: real failures can become regression cases in the next release rather than disappear inside logs.\"},\"tunes\":{}},{\"id\":\"h-repro\",\"type\":\"header\",\"data\":{\"text\":\"Reproducibility becomes conditional rather than exact\",\"level\":2},\"tunes\":{}},{\"id\":\"p-repro-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Classical ML reproducibility often aims to recreate a model from versioned code, data, environment and training parameters.\"},\"tunes\":{}},{\"id\":\"p-repro-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Hosted LLM applications cannot always reproduce identical output token-for-token because generation is probabilistic and providers may control infrastructure.\"},\"tunes\":{}},{\"id\":\"p-repro-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"LLMOps therefore aims for behavioral reproducibility: record enough model\u002Fprovider\u002Fversion, prompt, context inputs, retrieval state and runtime configuration to reproduce the conditions and validate behavior within expected tolerances.\"},\"tunes\":{}},{\"id\":\"h-lineage\",\"type\":\"header\",\"data\":{\"text\":\"Lineage expands from model lineage to application lineage\",\"level\":2},\"tunes\":{}},{\"id\":\"p-lineage-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"AWS's MLOps guidance treats model lineage as the history of code, data, model and infrastructure artifacts needed for diagnosis and reproducibility.\"},\"tunes\":{}},{\"id\":\"p-lineage-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"For LLM applications, lineage should additionally connect prompts, eval datasets, retrieval\u002Findex versions, tool schemas, agent\u002Fruntime configuration and provider\u002Fmodel snapshots.\"},\"tunes\":{}},{\"id\":\"p-lineage-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The target question becomes: Which exact application configuration produced this trace?\"},\"tunes\":{}},{\"id\":\"h-routing\",\"type\":\"header\",\"data\":{\"text\":\"Multi-provider and model routing create operational policy\",\"level\":2},\"tunes\":{}},{\"id\":\"p-route-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Once an application can use several providers or local models, routing becomes an operational policy rather than a simple model string.\"},\"tunes\":{}},{\"id\":\"p-route-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Routing may depend on capability, latency, cost, privacy, context length, availability, tool support or locality. A fallback can preserve uptime while changing answer quality or data-processing assumptions.\"},\"tunes\":{}},{\"id\":\"p-route-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"LLMOps should therefore log which route was actually selected and evaluate routes independently rather than treat every compatible endpoint as behaviorally interchangeable.\"},\"tunes\":{}},{\"id\":\"h-implementation\",\"type\":\"header\",\"data\":{\"text\":\"Original implementation evidence\",\"level\":2},\"tunes\":{}},{\"id\":\"h-client\",\"type\":\"header\",\"data\":{\"text\":\"Aaasaasa AI Client: provider, model and runtime are separate operational objects\",\"level\":3},\"tunes\":{}},{\"id\":\"p-client-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Aaasaasa AI Client separates agent\u002Fclient, provider, model, runtime location and permissions. Its AI Hub supports Ollama, LM Studio\u002FOpenAI-compatible endpoints and other provider protocols rather than treating “the model” as one global setting.\"},\"tunes\":{}},{\"id\":\"p-client-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The implementation includes dynamic local model discovery, streaming, thinking output and explicit Ollama warm\u002Fload and unload controls. That is operational evidence that local LLM serving introduces resource lifecycle concerns beyond an API model name.\"},\"tunes\":{}},{\"id\":\"p-client-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"Provider status is queried through provider adapters, and connection types distinguish local, cloud API, account-backed, remote-agent and web-client paths. These are concrete operational dimensions an LLM-aware platform has to surface.\"},\"tunes\":{}},{\"id\":\"p-client-4\",\"type\":\"paragraph\",\"data\":{\"text\":\"The repository also preserves an important boundary: a local runtime is not automatically local inference. Provider\u002Fmodel\u002Fruntime location are versioned or configurable concerns that affect privacy, latency, cost and availability.\"},\"tunes\":{}},{\"id\":\"h-sot\",\"type\":\"header\",\"data\":{\"text\":\"Source of Truth Research Engine: LLM application state extends beyond the model\",\"level\":3},\"tunes\":{}},{\"id\":\"p-sot-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The Source of Truth Research Engine combines lexical search, optional embeddings, source snapshots, SHA-256 identity, claims, provenance and contradiction tracking around local model-assisted research.\"},\"tunes\":{}},{\"id\":\"p-sot-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"This is useful LLMOps evidence because changing the model alone does not define the research system. Retrieval, source acquisition, evidence classification and persistent provenance are independent operational artifacts.\"},\"tunes\":{}},{\"id\":\"p-sot-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The implementation deliberately treats semantic similarity as discovery rather than evidence, showing why LLMOps observability should distinguish retrieval behavior from claim validity.\"},\"tunes\":{}},{\"id\":\"impl-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Observed implementation\",\"LLMOps lesson\"],[\"Multiple provider protocols\",\"Provider identity is an operational dependency\"],[\"Dynamic model discovery\",\"Available models can change independently of application code\"],[\"Ollama load\u002Funload controls\",\"Local models have memory\u002Fresource lifecycle\"],[\"Provider health\u002Fstatus adapters\",\"Model availability needs runtime observability\"],[\"Separate runtime and inference location\",\"Deployment topology is not one boolean “local\u002Fcloud”\"],[\"Central permissions\",\"Model capability and tool authority must remain separate\"],[\"Lexical + semantic retrieval pipeline\",\"Retrieval configuration is part of application behavior\"],[\"Source\u002Fprovenance persistence\",\"Operational state and evidence live outside model weights\"]]},\"tunes\":{}},{\"id\":\"impl-boundary\",\"type\":\"callout\",\"data\":{\"variant\":\"note\",\"title\":\"Evidence boundary\",\"body\":\"These projects demonstrate multi-provider\u002Flocal-model operations, permission separation, retrieval infrastructure and evidence persistence. They are not presented as a complete commercial LLMOps platform or proof of large-scale production traffic.\"},\"tunes\":{}},{\"id\":\"h-failures\",\"type\":\"header\",\"data\":{\"text\":\"Common LLMOps failure modes\",\"level\":2},\"tunes\":{}},{\"id\":\"failure-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Failure mode\",\"What actually went wrong\"],[\"Model alias upgraded silently\",\"Behavior changed without controlled release\"],[\"Prompt changed without evals\",\"Behavioral regression passed normal unit tests\"],[\"RAG index stale\",\"Generation model was blamed for retrieval\u002Fdata failure\"],[\"Only final answer is logged\",\"Root cause in retrieval\u002Ftool\u002Fcontext trajectory is invisible\"],[\"Provider fallback is silent\",\"Different model\u002Fdata path changes behavior without attribution\"],[\"Token cost tracked globally\",\"Expensive workflows cannot be localized\"],[\"Judge model changed\",\"Evaluation scores drift without application change\"],[\"Production traces never become tests\",\"Known failures repeatedly return\"],[\"Local model stays loaded indefinitely\",\"VRAM\u002Fresource pressure becomes operational instability\"],[\"Permissions encoded only in prompt\",\"Model behavior is mistaken for authorization\"],[\"One eval score gates everything\",\"Different quality dimensions are collapsed into a misleading number\"],[\"Model registry exists but prompt\u002Findex versions do not\",\"Application lineage remains incomplete\"]]},\"tunes\":{}},{\"id\":\"h-misconceptions\",\"type\":\"header\",\"data\":{\"text\":\"Common misconceptions\",\"level\":2},\"tunes\":{}},{\"id\":\"misconceptions-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Misconception\",\"Correction\"],[\"“LLMOps replaces MLOps.”\",\"LLMOps extends MLOps principles to LLM-specific application behavior.\"],[\"“LLMOps is prompt engineering.”\",\"Prompts are one artifact among models, providers, context, retrieval, tools, evals and runtime.\"],[\"“Hosted APIs remove operations work.”\",\"They remove some model-serving\u002Ftraining work but add provider lifecycle, version and dependency management.\"],[\"“If the API is stable, the app is stable.”\",\"Model behavior and provider\u002Fmodel snapshots can change independently of API schema.\"],[\"“RAG is just data preprocessing.”\",\"In production it has its own ingestion, index, retrieval and freshness lifecycle.\"],[\"“LLM outputs cannot be tested.”\",\"They can be evaluated with deterministic, reference, judge and human criteria.\"],[\"“LLM judges are objective ground truth.”\",\"They are model-based evaluators that also require calibration and version control.\"],[\"“A local model eliminates LLMOps.”\",\"Local serving adds model files, VRAM, load\u002Funload, runtime health and upgrade concerns.\"],[\"“Observability means token counts.”\",\"Useful observability follows prompts, retrievals, tools, model spans and outcomes.\"],[\"“Continuous training is mandatory.”\",\"Many LLM apps use continuous evaluation without training the foundation model.\"]]},\"tunes\":{}},{\"id\":\"h-design\",\"type\":\"header\",\"data\":{\"text\":\"A practical LLMOps design sequence\",\"level\":2},\"tunes\":{}},{\"id\":\"design-flow\",\"type\":\"processFlow\",\"data\":{\"title\":\"Operate the complete behavior-producing system\",\"orientation\":\"auto\",\"steps\":[{\"label\":\"1. Define the behavior unit\",\"description\":\"List every component that can materially change output: model, prompt, retrieval, tools, context and policy.\"},{\"label\":\"2. Establish application lineage\",\"description\":\"Version code, model\u002Fprovider, prompts, eval datasets, retrieval configuration and tool contracts.\"},{\"label\":\"3. Build representative eval datasets\",\"description\":\"Use expected success\u002Ffailure cases from design and production.\"},{\"label\":\"4. Separate deterministic and behavioral tests\",\"description\":\"Keep schema\u002Fsecurity assertions distinct from semantic output evaluation.\"},{\"label\":\"5. Trace end-to-end execution\",\"description\":\"Instrument model, retrieval, reranking, tools and agent\u002Fruntime spans.\"},{\"label\":\"6. Define release gates\",\"description\":\"Set quality, safety, latency and cost thresholds.\"},{\"label\":\"7. Pin or explicitly record model versions\",\"description\":\"Treat model\u002Fprovider changes as release events.\"},{\"label\":\"8. Deploy progressively\",\"description\":\"Use flags, canaries or staged rollout where consequence warrants it.\"},{\"label\":\"9. Evaluate production traces\",\"description\":\"Measure real task behavior and identify recurrent failures.\"},{\"label\":\"10. Feed failures back into eval datasets\",\"description\":\"Turn incidents and corrections into permanent regression coverage.\"},{\"label\":\"11. Monitor provider and data lifecycles\",\"description\":\"Track deprecations, index freshness, source changes and runtime availability.\"},{\"label\":\"12. Retire obsolete versions cleanly\",\"description\":\"Remove old prompts\u002Fmodels\u002Findexes\u002Fcredentials after migration and evidence retention decisions.\"}]},\"tunes\":{}},{\"id\":\"h-checklist\",\"type\":\"header\",\"data\":{\"text\":\"LLMOps architecture checklist\",\"level\":2},\"tunes\":{}},{\"id\":\"checklist-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Question\",\"Expected evidence\"],[\"Which model\u002Fprovider\u002Fversion served the request?\",\"Traceable model identity\"],[\"Which prompt\u002Finstructions were active?\",\"Versioned application code\u002Fconfig\"],[\"Which context reached the model?\",\"Context\u002Fretrieval trace\"],[\"Which corpus\u002Findex version was used?\",\"Retrieval lineage\"],[\"Which tools were available and called?\",\"Tool schema + trajectory trace\"],[\"Which permissions applied?\",\"Runtime authorization record\"],[\"How is quality measured?\",\"Versioned eval dataset + scorers\"],[\"How are model upgrades tested?\",\"Behavioral regression suite\"],[\"How is production quality sampled?\",\"Trace evaluation\u002Ffeedback process\"],[\"Can one failure be reproduced approximately?\",\"Model\u002Fcontext\u002Fprovider\u002Fapplication lineage\"],[\"Where is cost spent?\",\"Per-trace model\u002Ftool\u002Fretrieval attribution\"],[\"What triggers rollback?\",\"Defined quality\u002Fsafety\u002Fcost\u002Favailability threshold\"],[\"How are provider deprecations handled?\",\"Migration\u002Ffallback process\"],[\"How are local models operated?\",\"Health, resource, load\u002Funload and version controls\"]]},\"tunes\":{}},{\"id\":\"h-edge\",\"type\":\"header\",\"data\":{\"text\":\"Edge cases and limitations\",\"level\":2},\"tunes\":{}},{\"id\":\"p-edge-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A simple application that calls one fixed hosted model with no retrieval or tools may need only lightweight LLMOps: versioned prompt code, evals, model pinning, basic tracing and provider monitoring.\"},\"tunes\":{}},{\"id\":\"p-edge-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"A self-hosted fine-tuned model may require nearly the full classical MLOps stack plus LLM-specific application evaluation, making the boundary between MLOps and LLMOps intentionally blurry.\"},\"tunes\":{}},{\"id\":\"p-edge-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"An agent platform can have minimal model-training operations but substantial runtime operations because failures occur in tool selection, state and orchestration.\"},\"tunes\":{}},{\"id\":\"p-edge-4\",\"type\":\"paragraph\",\"data\":{\"text\":\"A RAG-heavy system can be operationally dominated by document ingestion and retrieval quality rather than model serving.\"},\"tunes\":{}},{\"id\":\"p-edge-5\",\"type\":\"paragraph\",\"data\":{\"text\":\"Terminology will continue to evolve. The durable architecture question is not which “Ops” label wins, but which artifacts produce behavior and therefore must be versioned, evaluated, observed and governed.\"},\"tunes\":{}},{\"id\":\"h-change\",\"type\":\"header\",\"data\":{\"text\":\"What would change this answer?\",\"level\":2},\"tunes\":{}},{\"id\":\"p-change-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"If foundation-model providers standardize perfectly stable model behavior and long-term version support, provider\u002Fsnapshot management could become less operationally significant.\"},\"tunes\":{}},{\"id\":\"p-change-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"If applications increasingly own fine-tuning or training, classical MLOps concerns become more central again.\"},\"tunes\":{}},{\"id\":\"p-change-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The operational principle would remain: every component that can materially change production behavior belongs in lineage, testing, observability and change control.\"},\"tunes\":{}},{\"id\":\"h-related\",\"type\":\"header\",\"data\":{\"text\":\"Related canonical knowledge\",\"level\":2},\"tunes\":{}},{\"id\":\"p-related-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"LLMOps sits below AI Governance and Enterprise AI Architecture: governance defines which changes require evidence and approval, while LLMOps provides the operational machinery to version, evaluate, deploy and observe those changes.\"},\"tunes\":{}},{\"id\":\"p-related-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Context Engineering and RAG are operational subdomains inside many LLM applications because context and retrieval can change behavior independently of the model.\"},\"tunes\":{}},{\"id\":\"p-related-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"Agentic AI extends LLMOps further into trajectory, permissions and tool-runtime operations.\"},\"tunes\":{}},{\"id\":\"ref-memory\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context\",\"title\":\"AI Agent Memory Is Not RAG: How to Separate Memory, Retrieval, State and Context\",\"excerpt\":\"Operational reliability improves when memory, retrieval, application state and model context remain separate lifecycle objects.\",\"ctaLabel\":\"Read the architecture article\"},\"tunes\":{}},{\"id\":\"ref-avb\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers\",\"title\":\"The Answer Validity Boundary: The Missing Layer Between Relevance and Reliable AI Answers\",\"excerpt\":\"LLMOps evaluation should preserve the version, scope and evidence conditions under which an answer remains supported.\",\"ctaLabel\":\"Read the Answer Validity Boundary\"},\"tunes\":{}},{\"id\":\"h-faq\",\"type\":\"header\",\"data\":{\"text\":\"Frequently asked questions\",\"level\":2},\"tunes\":{}},{\"id\":\"faq\",\"type\":\"faq\",\"data\":{\"title\":\"MLOps vs LLMOps FAQ\",\"items\":[{\"id\":\"faq1\",\"question\":\"What is the difference between MLOps and LLMOps?\",\"answer\":\"MLOps operates machine-learning systems across data, training, deployment and monitoring. LLMOps extends those practices to LLM applications where prompts, context, retrieval, providers, tools and evaluations also materially affect behavior.\"},{\"id\":\"faq2\",\"question\":\"Does LLMOps replace MLOps?\",\"answer\":\"No. LLMOps reuses MLOps disciplines such as CI\u002FCD, lineage, evaluation, deployment and monitoring and adds LLM-specific operational concerns.\"},{\"id\":\"faq3\",\"question\":\"Do LLM applications need continuous training?\",\"answer\":\"Not necessarily. Many use external foundation models and instead rely on continuous evaluation of prompts, models, retrieval and application behavior. Fine-tuned or self-trained systems can still require training pipelines.\"},{\"id\":\"faq4\",\"question\":\"Why are evals so important in LLMOps?\",\"answer\":\"Generative outputs are open-ended and model behavior can change across prompts, snapshots and context. Evals provide repeatable evidence that a release still meets defined quality and safety criteria.\"},{\"id\":\"faq5\",\"question\":\"What should be versioned in LLMOps?\",\"answer\":\"At minimum: application code, model\u002Fprovider\u002Fversion, prompts, eval datasets\u002Fscorers, retrieval configuration\u002Findexes, tool schemas, context rules and relevant safety\u002Fpermission configuration.\"},{\"id\":\"faq6\",\"question\":\"Is prompt versioning enough?\",\"answer\":\"No. The same prompt can behave differently with another model, retrieval set, context order, tool surface or provider.\"},{\"id\":\"faq7\",\"question\":\"What is GenAIOps?\",\"answer\":\"GenAIOps is another industry term for operating generative-AI applications. Some vendors use it interchangeably or as a broader label than LLMOps.\"},{\"id\":\"faq8\",\"question\":\"How do you monitor an LLM application?\",\"answer\":\"Monitor end-to-end traces including model calls, prompts\u002Fcontext, retrieval, tools, latency, token\u002Fcost, quality samples, safety and final task outcomes.\"},{\"id\":\"faq9\",\"question\":\"Can local LLMs use LLMOps practices?\",\"answer\":\"Yes. Local models add their own operational concerns such as model files, hardware\u002FVRAM, load\u002Funload, runtime health, quantization and upgrade management.\"}]},\"tunes\":{}},{\"id\":\"h-glossary\",\"type\":\"header\",\"data\":{\"text\":\"Glossary\",\"level\":2},\"tunes\":{}},{\"id\":\"glossary\",\"type\":\"glossary\",\"data\":{\"title\":\"Key MLOps and LLMOps terms\",\"entries\":[{\"term\":\"MLOps\",\"definition\":\"Engineering practices for building, deploying, monitoring and maintaining machine-learning systems and their data\u002Fmodel lifecycle.\",\"anchor\":\"mlops\"},{\"term\":\"LLMOps\",\"definition\":\"Operational practices for production applications whose behavior materially depends on large language models and surrounding prompts, context, retrieval, tools and runtime.\",\"anchor\":\"llmops\"},{\"term\":\"GenAIOps\",\"definition\":\"Operational discipline for generative-AI applications; often used as a broader or alternate label for LLMOps.\",\"anchor\":\"genaiops\"},{\"term\":\"Continuous training\",\"definition\":\"Automated or repeated retraining and serving of ML models as data or implementations change.\",\"anchor\":\"continuous-training\"},{\"term\":\"Continuous evaluation\",\"definition\":\"Repeated evaluation of candidate and production AI behavior against versioned datasets and criteria.\",\"anchor\":\"continuous-evaluation\"},{\"term\":\"Model snapshot\",\"definition\":\"A concrete version of a hosted or packaged model whose behavior can be tested and referenced.\",\"anchor\":\"model-snapshot\"},{\"term\":\"Application lineage\",\"definition\":\"Traceable relationship among code, model\u002Fprovider, prompts, data\u002Fretrieval, tools, runtime and release configuration.\",\"anchor\":\"application-lineage\"},{\"term\":\"Trace\",\"definition\":\"Structured record of one application execution containing spans such as model calls, retrievals and tool operations.\",\"anchor\":\"trace\"},{\"term\":\"Eval dataset\",\"definition\":\"Versioned set of representative inputs, expectations and optionally traces\u002Foutputs used to measure behavior.\",\"anchor\":\"eval-dataset\"},{\"term\":\"LLM judge\",\"definition\":\"A language model used as an evaluator for qualitative or semantic criteria; it is itself a versioned evaluation dependency.\",\"anchor\":\"llm-judge\"},{\"term\":\"Behavioral regression\",\"definition\":\"A degradation in application output or trajectory despite interfaces and code continuing to execute successfully.\",\"anchor\":\"behavioral-regression\"},{\"term\":\"Provider routing\",\"definition\":\"Policy for selecting among available model providers\u002Fendpoints according to capability, cost, latency, privacy or availability.\",\"anchor\":\"provider-routing\"}]},\"tunes\":{}},{\"id\":\"h-conclusion\",\"type\":\"header\",\"data\":{\"text\":\"Conclusion\",\"level\":2},\"tunes\":{}},{\"id\":\"p-conclusion-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"MLOps and LLMOps share the same engineering objective: make AI systems reproducible enough, testable enough and observable enough to operate reliably in production.\"},\"tunes\":{}},{\"id\":\"p-conclusion-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The difference is the shape of the system. Classical MLOps often centers on training and serving model artifacts; LLMOps must operate a behavioral stack in which model snapshots, prompts, context, retrieval, tools, permissions and providers can change independently.\"},\"tunes\":{}},{\"id\":\"p-conclusion-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The shortest useful rule is: version, evaluate and observe everything that can materially change the LLM application's behavior — not only the model.\"},\"tunes\":{}},{\"id\":\"h-sources\",\"type\":\"header\",\"data\":{\"text\":\"Primary sources and current documentation\",\"level\":2},\"tunes\":{}},{\"id\":\"p-sources-note\",\"type\":\"paragraph\",\"data\":{\"text\":\"The sources below ground the MLOps baseline and the current operational patterns for LLM and agent applications. Project sections are original implementation evidence and are intentionally narrower than claims about a complete LLMOps platform.\"},\"tunes\":{}},{\"id\":\"src-google-mlops\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdocs.cloud.google.com\u002Farchitecture\u002Fmlops-continuous-delivery-and-automation-pipelines-in-machine-learning\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Google Cloud — MLOps: Continuous delivery and automation pipelines\",\"description\":\"Reference architecture describing CI, CD, continuous training, model registry, metadata, serving and monitoring for ML systems.\"}},\"tunes\":{}},{\"id\":\"src-aws-lineage\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdocs.aws.amazon.com\u002Fwellarchitected\u002Flatest\u002Fmachine-learning-lens\u002Fmlops02-bp04.html\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"AWS Machine Learning Lens — Model lineage\",\"description\":\"Current guidance for tracking code, data, models, environments and infrastructure across ML releases.\"}},\"tunes\":{}},{\"id\":\"src-aws-monitor\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdocs.aws.amazon.com\u002Fwellarchitected\u002Flatest\u002Fmachine-learning-lens\u002Fmlops06-bp02.html\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"AWS Machine Learning Lens — Model observability and tracking\",\"description\":\"Current guidance for production model monitoring, drift, endpoint health and lineage.\"}},\"tunes\":{}},{\"id\":\"src-azure-llmops\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Flearn.microsoft.com\u002Fen-us\u002Fazure\u002Fmachine-learning\u002Fprompt-flow\u002Fhow-to-end-to-end-llmops-with-prompt-flow\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Microsoft Azure — GenAIOps \u002F LLMOps lifecycle\",\"description\":\"Official guidance describing GenAIOps, sometimes called LLMOps, across initialization, experimentation, evaluation\u002Frefinement and deployment.\"}},\"tunes\":{}},{\"id\":\"src-mlflow-genai\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fmlflow.org\u002Fdocs\u002Flatest\u002Fgenai\u002F\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"MLflow — Agents and LLM applications\",\"description\":\"Current GenAI operations documentation covering tracing, evaluation, prompts and production observability for LLM applications and agents.\"}},\"tunes\":{}},{\"id\":\"src-mlflow-traces\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fwww.mlflow.org\u002Fdocs\u002Flatest\u002Fgenai\u002Feval-monitor\u002Frunning-evaluation\u002Ftraces\u002F\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"MLflow — Evaluating production traces\",\"description\":\"Current guidance for evaluating complete LLM\u002Fagent traces, including retrieval and tool-call trajectories.\"}},\"tunes\":{}},{\"id\":\"src-mlflow-prompt-eval\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fmlflow.org\u002Fdocs\u002Flatest\u002Fgenai\u002Fprompt-registry\u002Fevaluate-prompts\u002F\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"MLflow — Evaluating prompts\",\"description\":\"Current prompt\u002Fmodel evaluation workflow using versioned prompts, datasets, scorers and traces.\"}},\"tunes\":{}},{\"id\":\"src-openai-api\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Freference\u002Foverview\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI API — Versioning and model snapshots\",\"description\":\"Current API guidance recommending pinned model versions and evals because prompting behavior can change between snapshots.\"}},\"tunes\":{}},{\"id\":\"src-openai-prompting\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fprompting\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — Prompting\",\"description\":\"Current guidance to treat production prompts as application code, version them through source control and cover changes with tests and evaluation checks.\"}},\"tunes\":{}},{\"id\":\"src-openai-deprecations\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fdeprecations\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — Deprecations\",\"description\":\"Current provider lifecycle evidence showing model and platform-surface retirement as an operational dependency.\"}},\"tunes\":{}},{\"id\":\"src-openai-promptfoo\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdevelopers.openai.com\u002Fcookbook\u002Fexamples\u002Fevaluation\u002Fmoving-from-openai-evals-to-promptfoo\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — Moving evaluation workflows to Promptfoo\",\"description\":\"Current 2026 migration guidance illustrating why evaluation assets should remain portable as provider tooling changes.\"}},\"tunes\":{}}],\"version\":\"2.31.6\"}",{"time":212,"blocks":213,"version":1558},1791487321430,[214,220,228,235,242,248,256,261,266,271,276,281,286,291,296,301,306,311,316,348,353,358,363,368,373,421,426,431,436,441,446,496,501,506,511,516,521,526,531,536,541,546,551,556,561,566,571,576,581,586,591,596,605,610,615,620,625,632,637,642,647,652,657,662,667,672,677,682,687,692,700,705,710,715,720,725,730,735,740,745,750,755,760,765,800,805,810,815,820,825,830,835,840,845,880,885,890,895,900,905,910,915,920,925,930,935,940,945,950,955,960,965,970,975,980,985,990,995,1000,1005,1010,1042,1048,1053,1097,1102,1140,1145,1187,1192,1242,1247,1252,1257,1262,1267,1272,1277,1282,1287,1292,1297,1302,1307,1312,1320,1328,1333,1375,1380,1428,1433,1438,1443,1448,1453,1458,1468,1477,1486,1495,1504,1513,1522,1531,1540,1549],{"id":215,"data":216,"type":218,"tunes":219},"intro",{"text":217},"MLOps is the engineering discipline for reliably developing, deploying, versioning and operating machine-learning systems; LLMOps extends that discipline to applications built around large language models, where production behavior depends not only on a model artifact but also on prompts, context, retrieval, provider\u002Fmodel versions, tool calls, safety controls and evaluation pipelines. LLMOps does not replace MLOps. It changes the operational unit from “a model plus serving pipeline” toward “an evolving LLM application whose behavior emerges from several independently changing components.”","paragraph",{},{"id":221,"data":222,"type":226,"tunes":227},"direct",{"body":223,"title":224,"variant":225},"\u003Cstrong>MLOps operates ML systems. LLMOps operates LLM applications.\u003C\u002Fstrong>\u003Cbr>\u003Cbr>Classical MLOps commonly centers on data pipelines, training, validation, model registry, deployment, drift and retraining. LLMOps keeps those disciplines where relevant, but often adds prompt\u002Fcontext versioning, model\u002Fprovider abstraction, RAG indexes, agent\u002Ftool traces, semantic evaluations, safety tests, token\u002Fcost monitoring and regression testing across rapidly changing model snapshots.","Direct answer","info","callout",{},{"id":229,"data":230,"type":226,"tunes":234},"boundary",{"body":231,"title":232,"variant":233},"A production LLM application can fail even when the prompt is unchanged: the provider can change a model snapshot, a RAG corpus can become stale, a reranker can regress, tool permissions can change, context assembly can drop evidence, or an agent can take a wrong trajectory. LLMOps therefore has to observe and version the system around the model, not only prompt text.","LLMOps is not just prompt management","warning",{},{"id":236,"data":237,"type":226,"tunes":241},"term-note",{"body":238,"title":239,"variant":240},"\u003Cstrong>LLMOps\u003C\u002Fstrong>, \u003Cstrong>GenAIOps\u003C\u002Fstrong> and related terms are widely used engineering labels, but they are not one universal formal standard with a single canonical lifecycle. Microsoft currently describes GenAIOps as “sometimes called LLMOps,” while MLflow groups operational tooling around agents and LLM applications. This article uses LLMOps as a practical architecture term for operating production systems whose behavior materially depends on LLMs.","Terminology boundary","note",{},{"id":243,"data":244,"type":226,"tunes":247},"current",{"body":245,"title":246,"variant":240},"The operational surface is changing quickly. OpenAI currently recommends pinning model snapshots and running evals because prompting behavior can change between snapshots, and several older platform-specific prompt\u002Feval surfaces are being retired in 2026. The stable architectural lesson is to keep prompts, tests and evals portable and versioned with the application rather than depend on one provider's dashboard object model.","Current-source note — 8 October 2026",{},{"id":249,"data":250,"type":254,"tunes":255},"toc",{"title":251,"maxLevel":252,"minLevel":253},"Contents",3,2,"tableOfContents",{},{"id":257,"data":258,"type":42,"tunes":260},"h-meaning",{"text":259,"level":253},"What MLOps really means",{},{"id":262,"data":263,"type":218,"tunes":265},"p-mlops-1",{"text":264},"MLOps applies software-engineering and operational discipline to machine-learning systems. The production challenge is broader than training a model: data collection, data validation, experimentation, reproducibility, model evaluation, deployment, infrastructure and monitoring all have to work together.",{},{"id":267,"data":268,"type":218,"tunes":270},"p-mlops-2",{"text":269},"Google's MLOps architecture guidance frames the discipline around continuous integration, continuous delivery and continuous training. CI validates not only code but also data, schemas and models; CD deploys ML pipelines and prediction services; CT can retrain and redeploy models as data or implementations change.",{},{"id":272,"data":273,"type":218,"tunes":275},"p-mlops-3",{"text":274},"AWS guidance adds the same operational concerns from another angle: model lineage, model\u002Fversion traceability, drift monitoring and production-quality monitoring are core parts of keeping ML systems reliable after deployment.",{},{"id":277,"data":278,"type":42,"tunes":280},"h-llmops",{"text":279,"level":253},"What changes when the model is an LLM",{},{"id":282,"data":283,"type":218,"tunes":285},"p-llmops-1",{"text":284},"Large language models change the production problem because the application often does not own the complete model-training lifecycle. A team may call a hosted model API, run an open model locally, switch between providers or use several models for different tasks.",{},{"id":287,"data":288,"type":218,"tunes":290},"p-llmops-2",{"text":289},"The model is therefore only one versioned dependency inside a larger behavioral system. Prompts, retrieval results, context order, tools, model snapshot, temperature\u002Freasoning settings, safety filters and runtime orchestration can all change the output.",{},{"id":292,"data":293,"type":218,"tunes":295},"p-llmops-3",{"text":294},"This creates a broader operational question: which combination of model, context, data, prompt, tools and runtime produced this behavior? LLMOps exists to make that question answerable and the answer reproducible enough for engineering work.",{},{"id":297,"data":298,"type":42,"tunes":300},"h-simple",{"text":299,"level":253},"The simplest example",{},{"id":302,"data":303,"type":218,"tunes":305},"p-simple-1",{"text":304},"Suppose an application answers internal policy questions.",{},{"id":307,"data":308,"type":218,"tunes":310},"p-simple-2",{"text":309},"In a classical ML framing, you might version a trained classifier, deploy it and monitor prediction quality. In an LLM application, the answer might depend on a hosted model snapshot, a system prompt, an embedding model, a vector index, retrieval filters, a reranker and the final selected context.",{},{"id":312,"data":313,"type":218,"tunes":315},"p-simple-3",{"text":314},"Changing any one of those components can change the final answer even though the application endpoint and user question stay identical.",{},{"id":317,"data":318,"type":346,"tunes":347},"simple-flow",{"steps":319,"title":344,"orientation":345},[320,323,326,329,332,335,338,341],{"label":321,"description":322},"1. Change one component","Prompt, model, provider, retrieval setting, tool schema or application code changes.",{"label":324,"description":325},"2. Run deterministic tests","Validate schemas, permissions, tool contracts, retrieval filters and application behavior.",{"label":327,"description":328},"3. Run behavioral evals","Compare representative outputs, retrieval quality and agent\u002Ftool trajectories against acceptance criteria.",{"label":330,"description":331},"4. Compare cost and latency","Measure token use, model calls, retrieval\u002Ftool overhead and response latency.",{"label":333,"description":334},"5. Deploy controlled version","Ship the concrete application configuration with model\u002Fprovider versions recorded.",{"label":336,"description":337},"6. Trace production behavior","Capture relevant model, retrieval, tool and runtime spans.",{"label":339,"description":340},"7. Evaluate production traces","Sample real executions for quality, grounding, safety and task success.",{"label":342,"description":343},"8. Roll back or iterate","Use regression evidence and operational signals to decide the next release.","A typical LLMOps release path","auto","processFlow",{},{"id":349,"data":350,"type":42,"tunes":352},"h-stops",{"text":351,"level":253},"Where the simple example stops",{},{"id":354,"data":355,"type":218,"tunes":357},"p-stops-1",{"text":356},"Some LLM systems still train or fine-tune their own models, so traditional MLOps practices such as training pipelines, model registry and data lineage remain directly relevant.",{},{"id":359,"data":360,"type":218,"tunes":362},"p-stops-2",{"text":361},"Other systems use only external foundation-model APIs and never run continuous training. Their main operational workload is application evaluation, model\u002Fprovider change management, prompt\u002Fcontext versioning, retrieval quality and observability.",{},{"id":364,"data":365,"type":218,"tunes":367},"p-stops-3",{"text":366},"There is therefore no single universal “LLMOps pipeline.” The exact lifecycle depends on whether you train, fine-tune, self-host, retrieve external knowledge, run agents or depend on managed model APIs.",{},{"id":369,"data":370,"type":42,"tunes":372},"h-compare",{"text":371,"level":253},"MLOps vs LLMOps",{},{"id":374,"data":375,"type":419,"tunes":420},"main-comparison",{"rows":376,"title":410,"layout":411,"columns":412},[377,382,386,390,394,398,402,406],{"id":378,"label":379,"values":380},"unit","Primary operational unit",[381,381],"",{"id":383,"label":384,"values":385},"model","Model ownership",[381,381],{"id":387,"label":388,"values":389},"change","Typical change",[381,381],{"id":391,"label":392,"values":393},"eval","Evaluation",[381,381],{"id":395,"label":396,"values":397},"monitor","Production monitoring",[381,381],{"id":399,"label":400,"values":401},"training","Continuous training",[381,381],{"id":403,"label":404,"values":405},"registry","Versioned artifacts",[381,381],{"id":407,"label":408,"values":409},"rollback","Rollback target",[381,381],"What stays the same and what expands","table",[413,416],{"id":414,"label":415},"mlops","MLOps",{"id":417,"label":418},"llmops","LLMOps","comparison",{},{"id":422,"data":423,"type":42,"tunes":425},"h-extension",{"text":424,"level":253},"LLMOps extends MLOps rather than replacing it",{},{"id":427,"data":428,"type":218,"tunes":430},"p-extension-1",{"text":429},"The core operational principles do not disappear: source control, CI\u002FCD, reproducibility, lineage, deployment controls, monitoring, rollback and measurable acceptance criteria remain essential.",{},{"id":432,"data":433,"type":218,"tunes":435},"p-extension-2",{"text":434},"The extension is that more behavior-defining artifacts now sit outside the model weights. A managed foundation model can change behavior through snapshot upgrades, while application output can change through prompt or retrieval changes without any model retraining.",{},{"id":437,"data":438,"type":218,"tunes":440},"p-extension-3",{"text":439},"This is why the useful hierarchy is usually DevOps → MLOps → LLMOps\u002FGenAIOps as increasingly specialized operational concerns, not three mutually exclusive practices.",{},{"id":442,"data":443,"type":42,"tunes":445},"h-artifacts",{"text":444,"level":253},"What has to be versioned in LLMOps?",{},{"id":447,"data":448,"type":411,"tunes":495},"artifact-table",{"content":449,"stretched":43,"withHeadings":14},[450,453,456,459,462,465,468,471,474,477,480,483,486,489,492],[451,452],"Artifact","Why it matters",[454,455],"Application code","Defines orchestration, validation, retries and business behavior",[457,458],"Model family + snapshot\u002Fversion","Different snapshots can produce different behavior",[460,461],"Provider \u002F endpoint","Changes data flow, latency, limits, pricing and availability",[463,464],"Prompt\u002Finstruction code","Changes model behavior even with same model",[466,467],"Generation\u002Freasoning parameters","Can alter determinism, latency, depth and cost",[469,470],"Eval dataset","Defines what “good enough” is tested against",[472,473],"Scorers \u002F graders","Define how quality is measured",[475,476],"Embedding model","Changes vector representation and retrieval behavior",[478,479],"Chunking\u002Findex configuration","Changes what can be retrieved",[481,482],"Reranker \u002F retrieval fusion","Changes result ordering",[484,485],"Tool schemas","Change what the model can request and how",[487,488],"Permission profile","Changes what tool actions may actually execute",[490,491],"Context assembly rules","Change what evidence and state reach the model",[493,494],"Safety\u002Fguardrail configuration","Changes allowed or blocked behavior",{},{"id":497,"data":498,"type":42,"tunes":500},"h-model-version",{"text":499,"level":253},"Model snapshots become release dependencies",{},{"id":502,"data":503,"type":218,"tunes":505},"p-model-version-1",{"text":504},"With hosted LLMs, the team may not control model training, but it still controls which model or snapshot the application calls.",{},{"id":507,"data":508,"type":218,"tunes":510},"p-model-version-2",{"text":509},"OpenAI's current API guidance explicitly warns that prompting behavior can change between model snapshots and recommends pinning production applications to specific snapshots where consistency matters, then running evals when upgrading.",{},{"id":512,"data":513,"type":218,"tunes":515},"p-model-version-3",{"text":514},"The operational consequence is straightforward: model upgrades should be treated as application releases, not invisible infrastructure maintenance.",{},{"id":517,"data":518,"type":42,"tunes":520},"h-provider",{"text":519,"level":253},"Provider lifecycle becomes part of operations",{},{"id":522,"data":523,"type":218,"tunes":525},"p-provider-1",{"text":524},"LLM applications often depend on provider rate limits, deprecation schedules, API semantics, context limits, data-handling rules and pricing.",{},{"id":527,"data":528,"type":218,"tunes":530},"p-provider-2",{"text":529},"A provider can deprecate a model while your application code remains unchanged. OpenAI's current deprecation schedule, for example, includes 2026 retirement dates for older model snapshots and platform surfaces.",{},{"id":532,"data":533,"type":218,"tunes":535},"p-provider-3",{"text":534},"LLMOps therefore needs provider lifecycle tracking, migration testing and fallback decisions in addition to model-quality monitoring.",{},{"id":537,"data":538,"type":42,"tunes":540},"h-prompt",{"text":539,"level":253},"Prompts behave like production code",{},{"id":542,"data":543,"type":218,"tunes":545},"p-prompt-1",{"text":544},"Prompts are executable behavioral configuration. Small changes can alter output quality, tool selection and policy interpretation.",{},{"id":547,"data":548,"type":218,"tunes":550},"p-prompt-2",{"text":549},"OpenAI's current guidance recommends storing production prompts in application code, reviewing prompt changes through pull requests, using typed inputs and covering changes with tests and evaluation checks.",{},{"id":552,"data":553,"type":218,"tunes":555},"p-prompt-3",{"text":554},"That makes prompt versioning less like editing marketing copy and more like changing a function whose output is probabilistic and model-dependent.",{},{"id":557,"data":558,"type":42,"tunes":560},"h-context",{"text":559,"level":253},"Context engineering becomes an operational concern",{},{"id":562,"data":563,"type":218,"tunes":565},"p-context-1",{"text":564},"The production model rarely receives only a static prompt. It may receive conversation history, retrieved documents, tool outputs, memory, current application state and policy instructions.",{},{"id":567,"data":568,"type":218,"tunes":570},"p-context-2",{"text":569},"LLMOps must therefore observe context assembly: which evidence was selected, which state version was current, whether truncation occurred and whether important instructions survived compaction.",{},{"id":572,"data":573,"type":218,"tunes":575},"p-context-3",{"text":574},"A model regression and a context regression can look identical at the final answer. Tracing the actual context path is what lets the team separate them.",{},{"id":577,"data":578,"type":42,"tunes":580},"h-rag",{"text":579,"level":253},"RAG creates its own operational lifecycle",{},{"id":582,"data":583,"type":218,"tunes":585},"p-rag-1",{"text":584},"A RAG system introduces a second production pipeline beside model inference: ingestion, extraction, chunking, metadata, embeddings, indexes, retrieval, reranking and context selection.",{},{"id":587,"data":588,"type":218,"tunes":590},"p-rag-2",{"text":589},"The knowledge corpus can change every day even when the model and prompt do not. A stale index or broken metadata filter can therefore degrade answer quality without any model drift.",{},{"id":592,"data":593,"type":218,"tunes":595},"p-rag-3",{"text":594},"LLMOps for RAG should track corpus\u002Findex version, embedding model, chunking policy, retrieval configuration, source freshness and retrieval metrics separately from generation quality.",{},{"id":597,"data":598,"type":603,"tunes":604},"ref-rag-diagnostic",{"url":599,"title":600,"excerpt":601,"ctaLabel":602},"https:\u002F\u002Fstajic.de\u002Fblog\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method","RAG Failed — But Which Layer Actually Failed? A Diagnostic Method","A production LLM pipeline needs separate observability for source coverage, retrieval, ranking, context assembly and generation.","Read the RAG diagnostic method","referralArticle",{},{"id":606,"data":607,"type":42,"tunes":609},"h-evals",{"text":608,"level":253},"Evals replace “looks good to me” with release evidence",{},{"id":611,"data":612,"type":218,"tunes":614},"p-eval-1",{"text":613},"Generative outputs are often open-ended, so exact-match tests are insufficient for many tasks. LLMOps adds evaluation datasets and scorers that can measure task success, correctness, safety, groundedness, style or domain-specific acceptance criteria.",{},{"id":616,"data":617,"type":218,"tunes":619},"p-eval-2",{"text":618},"MLflow's current GenAI evaluation stack supports versioned evaluation datasets, prompt\u002Fmodel comparisons, custom scorers and evaluation over complete traces.",{},{"id":621,"data":622,"type":218,"tunes":624},"p-eval-3",{"text":623},"The strongest practice is evaluation-driven development: define representative cases and acceptance criteria before or alongside changes, then compare releases against the same evidence.",{},{"id":626,"data":627,"type":226,"tunes":631},"eval-rule",{"body":628,"title":629,"variant":630},"A deployment should not be considered equivalent merely because the API contract still works. If the prompt, model, retrieval or tools changed, the behavioral regression suite should run again.","Behavioral changes need behavioral tests","success",{},{"id":633,"data":634,"type":42,"tunes":636},"h-judges",{"text":635,"level":253},"LLM-as-a-judge is useful but not ground truth",{},{"id":638,"data":639,"type":218,"tunes":641},"p-judge-1",{"text":640},"LLM judges can scale evaluation for qualities that are expensive to encode as deterministic assertions, such as relevance, tone or groundedness.",{},{"id":643,"data":644,"type":218,"tunes":646},"p-judge-2",{"text":645},"However, the judge is another model with its own bias, version and prompt. Judge configuration should therefore be versioned and calibrated against human or deterministic reference cases where consequence matters.",{},{"id":648,"data":649,"type":218,"tunes":651},"p-judge-3",{"text":650},"A production eval can mix deterministic checks, reference-based metrics, model judges and human review rather than asking one metric to represent every quality dimension.",{},{"id":653,"data":654,"type":42,"tunes":656},"h-tracing",{"text":655,"level":253},"Tracing becomes more important than endpoint logs",{},{"id":658,"data":659,"type":218,"tunes":661},"p-trace-1",{"text":660},"Traditional API logs can tell you that a request took two seconds and returned HTTP 200. They cannot tell you which retrieved chunks were selected, which tool the agent called or which model span consumed most tokens.",{},{"id":663,"data":664,"type":218,"tunes":666},"p-trace-2",{"text":665},"MLflow's current GenAI tracing captures prompts, retrievals, tool calls and application spans, and its production evaluation flow can score intermediate trajectory information rather than only final text.",{},{"id":668,"data":669,"type":218,"tunes":671},"p-trace-3",{"text":670},"This is a major LLMOps shift: observability follows the behavioral graph of the application, not only the serving endpoint.",{},{"id":673,"data":674,"type":42,"tunes":676},"h-agent",{"text":675,"level":253},"Agents expand LLMOps into runtime operations",{},{"id":678,"data":679,"type":218,"tunes":681},"p-agent-1",{"text":680},"An agentic application can perform several model calls, tool invocations and state transitions before producing a result.",{},{"id":683,"data":684,"type":218,"tunes":686},"p-agent-2",{"text":685},"Operating agents therefore requires step counts, tool-call traces, permission denials, retries, loop detection, human approvals and verified final state in addition to ordinary model latency and token metrics.",{},{"id":688,"data":689,"type":218,"tunes":691},"p-agent-3",{"text":690},"A correct final answer can hide a bad trajectory, so agent evaluation must inspect the path as well as the result.",{},{"id":693,"data":694,"type":603,"tunes":699},"ref-agent-reliability",{"url":695,"title":696,"excerpt":697,"ctaLabel":698},"https:\u002F\u002Fstajic.de\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough","AI Agent Reliability: Why the Final Answer Is Not Enough","Why agent production evaluation must include tool calls, state transitions, approvals and recoverability.","Read the agent reliability article",{},{"id":701,"data":702,"type":42,"tunes":704},"h-cost",{"text":703,"level":253},"Tokens, model calls and context become cost variables",{},{"id":706,"data":707,"type":218,"tunes":709},"p-cost-1",{"text":708},"Classical ML inference cost is often dominated by serving infrastructure or per-prediction compute. LLM applications can add provider token pricing, repeated agent calls, embedding calls, reranking and tool\u002Fruntime overhead.",{},{"id":711,"data":712,"type":218,"tunes":714},"p-cost-2",{"text":713},"Cost therefore has to be attributed to task or trace, not only to one endpoint. A workflow that makes eight hidden model calls can be functionally correct but operationally unacceptable.",{},{"id":716,"data":717,"type":218,"tunes":719},"p-cost-3",{"text":718},"Latency behaves the same way: model latency, retrieval, reranking and external tools compose into end-to-end user latency.",{},{"id":721,"data":722,"type":42,"tunes":724},"h-cache",{"text":723,"level":253},"Caching becomes semantic, not only technical",{},{"id":726,"data":727,"type":218,"tunes":729},"p-cache-1",{"text":728},"LLM systems can cache prompts, embeddings, retrieval results or full responses, but the cache key must reflect the semantics that can change the result.",{},{"id":731,"data":732,"type":218,"tunes":734},"p-cache-2",{"text":733},"A response cache that ignores model version, tenant, permissions or source freshness can return a technically valid but semantically invalid answer.",{},{"id":736,"data":737,"type":218,"tunes":739},"p-cache-3",{"text":738},"LLMOps therefore treats cache invalidation as part of model\u002Fcontext\u002Fdata versioning rather than only infrastructure optimization.",{},{"id":741,"data":742,"type":42,"tunes":744},"h-safety",{"text":743,"level":253},"Safety and permissions become release criteria",{},{"id":746,"data":747,"type":218,"tunes":749},"p-safety-1",{"text":748},"Generative systems can produce unbounded text and agents can trigger external actions. Safety testing therefore sits closer to ordinary CI\u002FCD than in many classical predictive ML systems.",{},{"id":751,"data":752,"type":218,"tunes":754},"p-safety-2",{"text":753},"Permission checks, prompt-injection tests, tenant-isolation tests and side-effect approvals should be reproducible regression tests where those risks exist.",{},{"id":756,"data":757,"type":218,"tunes":759},"p-safety-3",{"text":758},"The model may suggest an operation, but the runtime still has to enforce authorization. LLMOps owns the evidence that those controls continue to work after model, prompt or tool changes.",{},{"id":761,"data":762,"type":42,"tunes":764},"h-ci",{"text":763,"level":253},"What CI looks like in LLMOps",{},{"id":766,"data":767,"type":411,"tunes":799},"ci-table",{"content":768,"stretched":43,"withHeadings":14},[769,772,775,778,781,784,787,790,793,796],[770,771],"CI layer","Example checks",[773,774],"Code","Unit tests, type checks, schema validation",[776,777],"Prompts","Template rendering, required variables, policy text, snapshot review",[779,780],"Models\u002Fproviders","Compatibility, output schema, capability and regression tests",[782,783],"RAG","Chunking fixtures, filter tests, Recall@k, reranker regression",[785,786],"Tools","Input\u002Foutput schema tests, permission tests, idempotency tests",[788,789],"Agents","Trajectory fixtures, loop limits, handoff\u002Ftool-selection tests",[791,792],"Security","Prompt injection, unauthorized tools, cross-tenant negative tests",[794,795],"Behavioral evals","Task success, correctness, grounding, safety, domain criteria",[797,798],"Operational","Latency, token\u002Fcost budgets, timeout\u002Ffallback behavior",{},{"id":801,"data":802,"type":42,"tunes":804},"h-cd",{"text":803,"level":253},"What CD looks like in LLMOps",{},{"id":806,"data":807,"type":218,"tunes":809},"p-cd-1",{"text":808},"A production release may deploy no new model artifact at all. It may simply ship a new prompt, retrieval configuration, tool set or provider mapping.",{},{"id":811,"data":812,"type":218,"tunes":814},"p-cd-2",{"text":813},"The release bundle should therefore identify the complete behavior-defining configuration rather than only the application container image.",{},{"id":816,"data":817,"type":218,"tunes":819},"p-cd-3",{"text":818},"Feature flags, staged rollout, shadow evaluation, canary traffic and rollback are useful because LLM behavior can regress in ways that static contract tests do not detect.",{},{"id":821,"data":822,"type":42,"tunes":824},"h-ct",{"text":823,"level":253},"Continuous training becomes optional; continuous evaluation becomes central",{},{"id":826,"data":827,"type":218,"tunes":829},"p-ct-1",{"text":828},"Traditional MLOps often emphasizes continuous training when new data or drift justifies retraining.",{},{"id":831,"data":832,"type":218,"tunes":834},"p-ct-2",{"text":833},"Many LLM applications never train the foundation model. Their equivalent continuous loop is continuous evaluation: collect failures and representative production cases, add them to evaluation datasets, test candidate prompt\u002Fmodel\u002Fretrieval changes and redeploy only when evidence improves.",{},{"id":836,"data":837,"type":218,"tunes":839},"p-ct-3",{"text":838},"Fine-tuning can reintroduce a training lifecycle, but it should sit inside the same broader evaluation and release process.",{},{"id":841,"data":842,"type":42,"tunes":844},"h-monitor",{"text":843,"level":253},"What should be monitored in production?",{},{"id":846,"data":847,"type":411,"tunes":879},"monitor-table",{"content":848,"stretched":43,"withHeadings":14},[849,852,855,858,861,864,867,869,871,873,876],[850,851],"Signal class","Examples",[853,854],"System health","Errors, timeouts, endpoint availability",[856,857],"Model\u002Fprovider","Model ID, snapshot, rate limits, provider errors",[859,860],"Latency","End-to-end, model, retrieval, tool and reranker spans",[862,863],"Cost","Input\u002Foutput tokens, embeddings, tool\u002FAPI spend",[865,866],"Quality","Sampled task success, correctness, relevance, groundedness",[782,868],"Retrieval recall proxies, empty retrieval, stale sources, citation coverage",[788,870],"Tool selection, retries, loops, handoffs, approval frequency",[791,872],"Denied actions, prompt-injection indicators, tenant-boundary failures",[874,875],"User feedback","Corrections, abandonment, escalation, explicit ratings",[877,878],"Change drift","Provider\u002Fmodel\u002Fconfig changes relative to approved release",{},{"id":881,"data":882,"type":42,"tunes":884},"h-prod-eval",{"text":883,"level":253},"Production traces can become evaluation data",{},{"id":886,"data":887,"type":218,"tunes":889},"p-prod-1",{"text":888},"One of the most useful modern LLMOps patterns is to turn sampled production traces into evaluation records.",{},{"id":891,"data":892,"type":218,"tunes":894},"p-prod-2",{"text":893},"MLflow currently supports retrieving production traces and scoring not only outputs but intermediate spans such as retrieval or tool-call trajectories.",{},{"id":896,"data":897,"type":218,"tunes":899},"p-prod-3",{"text":898},"This closes the loop between observability and development: real failures can become regression cases in the next release rather than disappear inside logs.",{},{"id":901,"data":902,"type":42,"tunes":904},"h-repro",{"text":903,"level":253},"Reproducibility becomes conditional rather than exact",{},{"id":906,"data":907,"type":218,"tunes":909},"p-repro-1",{"text":908},"Classical ML reproducibility often aims to recreate a model from versioned code, data, environment and training parameters.",{},{"id":911,"data":912,"type":218,"tunes":914},"p-repro-2",{"text":913},"Hosted LLM applications cannot always reproduce identical output token-for-token because generation is probabilistic and providers may control infrastructure.",{},{"id":916,"data":917,"type":218,"tunes":919},"p-repro-3",{"text":918},"LLMOps therefore aims for behavioral reproducibility: record enough model\u002Fprovider\u002Fversion, prompt, context inputs, retrieval state and runtime configuration to reproduce the conditions and validate behavior within expected tolerances.",{},{"id":921,"data":922,"type":42,"tunes":924},"h-lineage",{"text":923,"level":253},"Lineage expands from model lineage to application lineage",{},{"id":926,"data":927,"type":218,"tunes":929},"p-lineage-1",{"text":928},"AWS's MLOps guidance treats model lineage as the history of code, data, model and infrastructure artifacts needed for diagnosis and reproducibility.",{},{"id":931,"data":932,"type":218,"tunes":934},"p-lineage-2",{"text":933},"For LLM applications, lineage should additionally connect prompts, eval datasets, retrieval\u002Findex versions, tool schemas, agent\u002Fruntime configuration and provider\u002Fmodel snapshots.",{},{"id":936,"data":937,"type":218,"tunes":939},"p-lineage-3",{"text":938},"The target question becomes: Which exact application configuration produced this trace?",{},{"id":941,"data":942,"type":42,"tunes":944},"h-routing",{"text":943,"level":253},"Multi-provider and model routing create operational policy",{},{"id":946,"data":947,"type":218,"tunes":949},"p-route-1",{"text":948},"Once an application can use several providers or local models, routing becomes an operational policy rather than a simple model string.",{},{"id":951,"data":952,"type":218,"tunes":954},"p-route-2",{"text":953},"Routing may depend on capability, latency, cost, privacy, context length, availability, tool support or locality. A fallback can preserve uptime while changing answer quality or data-processing assumptions.",{},{"id":956,"data":957,"type":218,"tunes":959},"p-route-3",{"text":958},"LLMOps should therefore log which route was actually selected and evaluate routes independently rather than treat every compatible endpoint as behaviorally interchangeable.",{},{"id":961,"data":962,"type":42,"tunes":964},"h-implementation",{"text":963,"level":253},"Original implementation evidence",{},{"id":966,"data":967,"type":42,"tunes":969},"h-client",{"text":968,"level":252},"Aaasaasa AI Client: provider, model and runtime are separate operational objects",{},{"id":971,"data":972,"type":218,"tunes":974},"p-client-1",{"text":973},"Aaasaasa AI Client separates agent\u002Fclient, provider, model, runtime location and permissions. Its AI Hub supports Ollama, LM Studio\u002FOpenAI-compatible endpoints and other provider protocols rather than treating “the model” as one global setting.",{},{"id":976,"data":977,"type":218,"tunes":979},"p-client-2",{"text":978},"The implementation includes dynamic local model discovery, streaming, thinking output and explicit Ollama warm\u002Fload and unload controls. That is operational evidence that local LLM serving introduces resource lifecycle concerns beyond an API model name.",{},{"id":981,"data":982,"type":218,"tunes":984},"p-client-3",{"text":983},"Provider status is queried through provider adapters, and connection types distinguish local, cloud API, account-backed, remote-agent and web-client paths. These are concrete operational dimensions an LLM-aware platform has to surface.",{},{"id":986,"data":987,"type":218,"tunes":989},"p-client-4",{"text":988},"The repository also preserves an important boundary: a local runtime is not automatically local inference. Provider\u002Fmodel\u002Fruntime location are versioned or configurable concerns that affect privacy, latency, cost and availability.",{},{"id":991,"data":992,"type":42,"tunes":994},"h-sot",{"text":993,"level":252},"Source of Truth Research Engine: LLM application state extends beyond the model",{},{"id":996,"data":997,"type":218,"tunes":999},"p-sot-1",{"text":998},"The Source of Truth Research Engine combines lexical search, optional embeddings, source snapshots, SHA-256 identity, claims, provenance and contradiction tracking around local model-assisted research.",{},{"id":1001,"data":1002,"type":218,"tunes":1004},"p-sot-2",{"text":1003},"This is useful LLMOps evidence because changing the model alone does not define the research system. Retrieval, source acquisition, evidence classification and persistent provenance are independent operational artifacts.",{},{"id":1006,"data":1007,"type":218,"tunes":1009},"p-sot-3",{"text":1008},"The implementation deliberately treats semantic similarity as discovery rather than evidence, showing why LLMOps observability should distinguish retrieval behavior from claim validity.",{},{"id":1011,"data":1012,"type":411,"tunes":1041},"impl-table",{"content":1013,"stretched":43,"withHeadings":14},[1014,1017,1020,1023,1026,1029,1032,1035,1038],[1015,1016],"Observed implementation","LLMOps lesson",[1018,1019],"Multiple provider protocols","Provider identity is an operational dependency",[1021,1022],"Dynamic model discovery","Available models can change independently of application code",[1024,1025],"Ollama load\u002Funload controls","Local models have memory\u002Fresource lifecycle",[1027,1028],"Provider health\u002Fstatus adapters","Model availability needs runtime observability",[1030,1031],"Separate runtime and inference location","Deployment topology is not one boolean “local\u002Fcloud”",[1033,1034],"Central permissions","Model capability and tool authority must remain separate",[1036,1037],"Lexical + semantic retrieval pipeline","Retrieval configuration is part of application behavior",[1039,1040],"Source\u002Fprovenance persistence","Operational state and evidence live outside model weights",{},{"id":1043,"data":1044,"type":226,"tunes":1047},"impl-boundary",{"body":1045,"title":1046,"variant":240},"These projects demonstrate multi-provider\u002Flocal-model operations, permission separation, retrieval infrastructure and evidence persistence. They are not presented as a complete commercial LLMOps platform or proof of large-scale production traffic.","Evidence boundary",{},{"id":1049,"data":1050,"type":42,"tunes":1052},"h-failures",{"text":1051,"level":253},"Common LLMOps failure modes",{},{"id":1054,"data":1055,"type":411,"tunes":1096},"failure-table",{"content":1056,"stretched":43,"withHeadings":14},[1057,1060,1063,1066,1069,1072,1075,1078,1081,1084,1087,1090,1093],[1058,1059],"Failure mode","What actually went wrong",[1061,1062],"Model alias upgraded silently","Behavior changed without controlled release",[1064,1065],"Prompt changed without evals","Behavioral regression passed normal unit tests",[1067,1068],"RAG index stale","Generation model was blamed for retrieval\u002Fdata failure",[1070,1071],"Only final answer is logged","Root cause in retrieval\u002Ftool\u002Fcontext trajectory is invisible",[1073,1074],"Provider fallback is silent","Different model\u002Fdata path changes behavior without attribution",[1076,1077],"Token cost tracked globally","Expensive workflows cannot be localized",[1079,1080],"Judge model changed","Evaluation scores drift without application change",[1082,1083],"Production traces never become tests","Known failures repeatedly return",[1085,1086],"Local model stays loaded indefinitely","VRAM\u002Fresource pressure becomes operational instability",[1088,1089],"Permissions encoded only in prompt","Model behavior is mistaken for authorization",[1091,1092],"One eval score gates everything","Different quality dimensions are collapsed into a misleading number",[1094,1095],"Model registry exists but prompt\u002Findex versions do not","Application lineage remains incomplete",{},{"id":1098,"data":1099,"type":42,"tunes":1101},"h-misconceptions",{"text":1100,"level":253},"Common misconceptions",{},{"id":1103,"data":1104,"type":411,"tunes":1139},"misconceptions-table",{"content":1105,"stretched":43,"withHeadings":14},[1106,1109,1112,1115,1118,1121,1124,1127,1130,1133,1136],[1107,1108],"Misconception","Correction",[1110,1111],"“LLMOps replaces MLOps.”","LLMOps extends MLOps principles to LLM-specific application behavior.",[1113,1114],"“LLMOps is prompt engineering.”","Prompts are one artifact among models, providers, context, retrieval, tools, evals and runtime.",[1116,1117],"“Hosted APIs remove operations work.”","They remove some model-serving\u002Ftraining work but add provider lifecycle, version and dependency management.",[1119,1120],"“If the API is stable, the app is stable.”","Model behavior and provider\u002Fmodel snapshots can change independently of API schema.",[1122,1123],"“RAG is just data preprocessing.”","In production it has its own ingestion, index, retrieval and freshness lifecycle.",[1125,1126],"“LLM outputs cannot be tested.”","They can be evaluated with deterministic, reference, judge and human criteria.",[1128,1129],"“LLM judges are objective ground truth.”","They are model-based evaluators that also require calibration and version control.",[1131,1132],"“A local model eliminates LLMOps.”","Local serving adds model files, VRAM, load\u002Funload, runtime health and upgrade concerns.",[1134,1135],"“Observability means token counts.”","Useful observability follows prompts, retrievals, tools, model spans and outcomes.",[1137,1138],"“Continuous training is mandatory.”","Many LLM apps use continuous evaluation without training the foundation model.",{},{"id":1141,"data":1142,"type":42,"tunes":1144},"h-design",{"text":1143,"level":253},"A practical LLMOps design sequence",{},{"id":1146,"data":1147,"type":346,"tunes":1186},"design-flow",{"steps":1148,"title":1185,"orientation":345},[1149,1152,1155,1158,1161,1164,1167,1170,1173,1176,1179,1182],{"label":1150,"description":1151},"1. Define the behavior unit","List every component that can materially change output: model, prompt, retrieval, tools, context and policy.",{"label":1153,"description":1154},"2. Establish application lineage","Version code, model\u002Fprovider, prompts, eval datasets, retrieval configuration and tool contracts.",{"label":1156,"description":1157},"3. Build representative eval datasets","Use expected success\u002Ffailure cases from design and production.",{"label":1159,"description":1160},"4. Separate deterministic and behavioral tests","Keep schema\u002Fsecurity assertions distinct from semantic output evaluation.",{"label":1162,"description":1163},"5. Trace end-to-end execution","Instrument model, retrieval, reranking, tools and agent\u002Fruntime spans.",{"label":1165,"description":1166},"6. Define release gates","Set quality, safety, latency and cost thresholds.",{"label":1168,"description":1169},"7. Pin or explicitly record model versions","Treat model\u002Fprovider changes as release events.",{"label":1171,"description":1172},"8. Deploy progressively","Use flags, canaries or staged rollout where consequence warrants it.",{"label":1174,"description":1175},"9. Evaluate production traces","Measure real task behavior and identify recurrent failures.",{"label":1177,"description":1178},"10. Feed failures back into eval datasets","Turn incidents and corrections into permanent regression coverage.",{"label":1180,"description":1181},"11. Monitor provider and data lifecycles","Track deprecations, index freshness, source changes and runtime availability.",{"label":1183,"description":1184},"12. Retire obsolete versions cleanly","Remove old prompts\u002Fmodels\u002Findexes\u002Fcredentials after migration and evidence retention decisions.","Operate the complete behavior-producing system",{},{"id":1188,"data":1189,"type":42,"tunes":1191},"h-checklist",{"text":1190,"level":253},"LLMOps architecture checklist",{},{"id":1193,"data":1194,"type":411,"tunes":1241},"checklist-table",{"content":1195,"stretched":43,"withHeadings":14},[1196,1199,1202,1205,1208,1211,1214,1217,1220,1223,1226,1229,1232,1235,1238],[1197,1198],"Question","Expected evidence",[1200,1201],"Which model\u002Fprovider\u002Fversion served the request?","Traceable model identity",[1203,1204],"Which prompt\u002Finstructions were active?","Versioned application code\u002Fconfig",[1206,1207],"Which context reached the model?","Context\u002Fretrieval trace",[1209,1210],"Which corpus\u002Findex version was used?","Retrieval lineage",[1212,1213],"Which tools were available and called?","Tool schema + trajectory trace",[1215,1216],"Which permissions applied?","Runtime authorization record",[1218,1219],"How is quality measured?","Versioned eval dataset + scorers",[1221,1222],"How are model upgrades tested?","Behavioral regression suite",[1224,1225],"How is production quality sampled?","Trace evaluation\u002Ffeedback process",[1227,1228],"Can one failure be reproduced approximately?","Model\u002Fcontext\u002Fprovider\u002Fapplication lineage",[1230,1231],"Where is cost spent?","Per-trace model\u002Ftool\u002Fretrieval attribution",[1233,1234],"What triggers rollback?","Defined quality\u002Fsafety\u002Fcost\u002Favailability threshold",[1236,1237],"How are provider deprecations handled?","Migration\u002Ffallback process",[1239,1240],"How are local models operated?","Health, resource, load\u002Funload and version controls",{},{"id":1243,"data":1244,"type":42,"tunes":1246},"h-edge",{"text":1245,"level":253},"Edge cases and limitations",{},{"id":1248,"data":1249,"type":218,"tunes":1251},"p-edge-1",{"text":1250},"A simple application that calls one fixed hosted model with no retrieval or tools may need only lightweight LLMOps: versioned prompt code, evals, model pinning, basic tracing and provider monitoring.",{},{"id":1253,"data":1254,"type":218,"tunes":1256},"p-edge-2",{"text":1255},"A self-hosted fine-tuned model may require nearly the full classical MLOps stack plus LLM-specific application evaluation, making the boundary between MLOps and LLMOps intentionally blurry.",{},{"id":1258,"data":1259,"type":218,"tunes":1261},"p-edge-3",{"text":1260},"An agent platform can have minimal model-training operations but substantial runtime operations because failures occur in tool selection, state and orchestration.",{},{"id":1263,"data":1264,"type":218,"tunes":1266},"p-edge-4",{"text":1265},"A RAG-heavy system can be operationally dominated by document ingestion and retrieval quality rather than model serving.",{},{"id":1268,"data":1269,"type":218,"tunes":1271},"p-edge-5",{"text":1270},"Terminology will continue to evolve. The durable architecture question is not which “Ops” label wins, but which artifacts produce behavior and therefore must be versioned, evaluated, observed and governed.",{},{"id":1273,"data":1274,"type":42,"tunes":1276},"h-change",{"text":1275,"level":253},"What would change this answer?",{},{"id":1278,"data":1279,"type":218,"tunes":1281},"p-change-1",{"text":1280},"If foundation-model providers standardize perfectly stable model behavior and long-term version support, provider\u002Fsnapshot management could become less operationally significant.",{},{"id":1283,"data":1284,"type":218,"tunes":1286},"p-change-2",{"text":1285},"If applications increasingly own fine-tuning or training, classical MLOps concerns become more central again.",{},{"id":1288,"data":1289,"type":218,"tunes":1291},"p-change-3",{"text":1290},"The operational principle would remain: every component that can materially change production behavior belongs in lineage, testing, observability and change control.",{},{"id":1293,"data":1294,"type":42,"tunes":1296},"h-related",{"text":1295,"level":253},"Related canonical knowledge",{},{"id":1298,"data":1299,"type":218,"tunes":1301},"p-related-1",{"text":1300},"LLMOps sits below AI Governance and Enterprise AI Architecture: governance defines which changes require evidence and approval, while LLMOps provides the operational machinery to version, evaluate, deploy and observe those changes.",{},{"id":1303,"data":1304,"type":218,"tunes":1306},"p-related-2",{"text":1305},"Context Engineering and RAG are operational subdomains inside many LLM applications because context and retrieval can change behavior independently of the model.",{},{"id":1308,"data":1309,"type":218,"tunes":1311},"p-related-3",{"text":1310},"Agentic AI extends LLMOps further into trajectory, permissions and tool-runtime operations.",{},{"id":1313,"data":1314,"type":603,"tunes":1319},"ref-memory",{"url":1315,"title":1316,"excerpt":1317,"ctaLabel":1318},"https:\u002F\u002Fstajic.de\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context","AI Agent Memory Is Not RAG: How to Separate Memory, Retrieval, State and Context","Operational reliability improves when memory, retrieval, application state and model context remain separate lifecycle objects.","Read the architecture article",{},{"id":1321,"data":1322,"type":603,"tunes":1327},"ref-avb",{"url":1323,"title":1324,"excerpt":1325,"ctaLabel":1326},"https:\u002F\u002Fstajic.de\u002Fblog\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers","The Answer Validity Boundary: The Missing Layer Between Relevance and Reliable AI Answers","LLMOps evaluation should preserve the version, scope and evidence conditions under which an answer remains supported.","Read the Answer Validity Boundary",{},{"id":1329,"data":1330,"type":42,"tunes":1332},"h-faq",{"text":1331,"level":253},"Frequently asked questions",{},{"id":1334,"data":1335,"type":1334,"tunes":1374},"faq",{"items":1336,"title":1373},[1337,1341,1345,1349,1353,1357,1361,1365,1369],{"id":1338,"answer":1339,"question":1340},"faq1","MLOps operates machine-learning systems across data, training, deployment and monitoring. LLMOps extends those practices to LLM applications where prompts, context, retrieval, providers, tools and evaluations also materially affect behavior.","What is the difference between MLOps and LLMOps?",{"id":1342,"answer":1343,"question":1344},"faq2","No. LLMOps reuses MLOps disciplines such as CI\u002FCD, lineage, evaluation, deployment and monitoring and adds LLM-specific operational concerns.","Does LLMOps replace MLOps?",{"id":1346,"answer":1347,"question":1348},"faq3","Not necessarily. Many use external foundation models and instead rely on continuous evaluation of prompts, models, retrieval and application behavior. Fine-tuned or self-trained systems can still require training pipelines.","Do LLM applications need continuous training?",{"id":1350,"answer":1351,"question":1352},"faq4","Generative outputs are open-ended and model behavior can change across prompts, snapshots and context. Evals provide repeatable evidence that a release still meets defined quality and safety criteria.","Why are evals so important in LLMOps?",{"id":1354,"answer":1355,"question":1356},"faq5","At minimum: application code, model\u002Fprovider\u002Fversion, prompts, eval datasets\u002Fscorers, retrieval configuration\u002Findexes, tool schemas, context rules and relevant safety\u002Fpermission configuration.","What should be versioned in LLMOps?",{"id":1358,"answer":1359,"question":1360},"faq6","No. The same prompt can behave differently with another model, retrieval set, context order, tool surface or provider.","Is prompt versioning enough?",{"id":1362,"answer":1363,"question":1364},"faq7","GenAIOps is another industry term for operating generative-AI applications. Some vendors use it interchangeably or as a broader label than LLMOps.","What is GenAIOps?",{"id":1366,"answer":1367,"question":1368},"faq8","Monitor end-to-end traces including model calls, prompts\u002Fcontext, retrieval, tools, latency, token\u002Fcost, quality samples, safety and final task outcomes.","How do you monitor an LLM application?",{"id":1370,"answer":1371,"question":1372},"faq9","Yes. Local models add their own operational concerns such as model files, hardware\u002FVRAM, load\u002Funload, runtime health, quantization and upgrade management.","Can local LLMs use LLMOps practices?","MLOps vs LLMOps FAQ",{},{"id":1376,"data":1377,"type":42,"tunes":1379},"h-glossary",{"text":1378,"level":253},"Glossary",{},{"id":1381,"data":1382,"type":1381,"tunes":1427},"glossary",{"title":1383,"entries":1384},"Key MLOps and LLMOps terms",[1385,1387,1389,1393,1396,1400,1404,1408,1412,1415,1419,1423],{"term":415,"anchor":414,"definition":1386},"Engineering practices for building, deploying, monitoring and maintaining machine-learning systems and their data\u002Fmodel lifecycle.",{"term":418,"anchor":417,"definition":1388},"Operational practices for production applications whose behavior materially depends on large language models and surrounding prompts, context, retrieval, tools and runtime.",{"term":1390,"anchor":1391,"definition":1392},"GenAIOps","genaiops","Operational discipline for generative-AI applications; often used as a broader or alternate label for LLMOps.",{"term":400,"anchor":1394,"definition":1395},"continuous-training","Automated or repeated retraining and serving of ML models as data or implementations change.",{"term":1397,"anchor":1398,"definition":1399},"Continuous evaluation","continuous-evaluation","Repeated evaluation of candidate and production AI behavior against versioned datasets and criteria.",{"term":1401,"anchor":1402,"definition":1403},"Model snapshot","model-snapshot","A concrete version of a hosted or packaged model whose behavior can be tested and referenced.",{"term":1405,"anchor":1406,"definition":1407},"Application lineage","application-lineage","Traceable relationship among code, model\u002Fprovider, prompts, data\u002Fretrieval, tools, runtime and release configuration.",{"term":1409,"anchor":1410,"definition":1411},"Trace","trace","Structured record of one application execution containing spans such as model calls, retrievals and tool operations.",{"term":469,"anchor":1413,"definition":1414},"eval-dataset","Versioned set of representative inputs, expectations and optionally traces\u002Foutputs used to measure behavior.",{"term":1416,"anchor":1417,"definition":1418},"LLM judge","llm-judge","A language model used as an evaluator for qualitative or semantic criteria; it is itself a versioned evaluation dependency.",{"term":1420,"anchor":1421,"definition":1422},"Behavioral regression","behavioral-regression","A degradation in application output or trajectory despite interfaces and code continuing to execute successfully.",{"term":1424,"anchor":1425,"definition":1426},"Provider routing","provider-routing","Policy for selecting among available model providers\u002Fendpoints according to capability, cost, latency, privacy or availability.",{},{"id":1429,"data":1430,"type":42,"tunes":1432},"h-conclusion",{"text":1431,"level":253},"Conclusion",{},{"id":1434,"data":1435,"type":218,"tunes":1437},"p-conclusion-1",{"text":1436},"MLOps and LLMOps share the same engineering objective: make AI systems reproducible enough, testable enough and observable enough to operate reliably in production.",{},{"id":1439,"data":1440,"type":218,"tunes":1442},"p-conclusion-2",{"text":1441},"The difference is the shape of the system. Classical MLOps often centers on training and serving model artifacts; LLMOps must operate a behavioral stack in which model snapshots, prompts, context, retrieval, tools, permissions and providers can change independently.",{},{"id":1444,"data":1445,"type":218,"tunes":1447},"p-conclusion-3",{"text":1446},"The shortest useful rule is: version, evaluate and observe everything that can materially change the LLM application's behavior — not only the model.",{},{"id":1449,"data":1450,"type":42,"tunes":1452},"h-sources",{"text":1451,"level":253},"Primary sources and current documentation",{},{"id":1454,"data":1455,"type":218,"tunes":1457},"p-sources-note",{"text":1456},"The sources below ground the MLOps baseline and the current operational patterns for LLM and agent applications. Project sections are original implementation evidence and are intentionally narrower than claims about a complete LLMOps platform.",{},{"id":1459,"data":1460,"type":1466,"tunes":1467},"src-google-mlops",{"link":1461,"meta":1462},"https:\u002F\u002Fdocs.cloud.google.com\u002Farchitecture\u002Fmlops-continuous-delivery-and-automation-pipelines-in-machine-learning",{"image":1463,"title":1464,"description":1465},{"url":381},"Google Cloud — MLOps: Continuous delivery and automation pipelines","Reference architecture describing CI, CD, continuous training, model registry, metadata, serving and monitoring for ML systems.","linkTool",{},{"id":1469,"data":1470,"type":1466,"tunes":1476},"src-aws-lineage",{"link":1471,"meta":1472},"https:\u002F\u002Fdocs.aws.amazon.com\u002Fwellarchitected\u002Flatest\u002Fmachine-learning-lens\u002Fmlops02-bp04.html",{"image":1473,"title":1474,"description":1475},{"url":381},"AWS Machine Learning Lens — Model lineage","Current guidance for tracking code, data, models, environments and infrastructure across ML releases.",{},{"id":1478,"data":1479,"type":1466,"tunes":1485},"src-aws-monitor",{"link":1480,"meta":1481},"https:\u002F\u002Fdocs.aws.amazon.com\u002Fwellarchitected\u002Flatest\u002Fmachine-learning-lens\u002Fmlops06-bp02.html",{"image":1482,"title":1483,"description":1484},{"url":381},"AWS Machine Learning Lens — Model observability and tracking","Current guidance for production model monitoring, drift, endpoint health and lineage.",{},{"id":1487,"data":1488,"type":1466,"tunes":1494},"src-azure-llmops",{"link":1489,"meta":1490},"https:\u002F\u002Flearn.microsoft.com\u002Fen-us\u002Fazure\u002Fmachine-learning\u002Fprompt-flow\u002Fhow-to-end-to-end-llmops-with-prompt-flow",{"image":1491,"title":1492,"description":1493},{"url":381},"Microsoft Azure — GenAIOps \u002F LLMOps lifecycle","Official guidance describing GenAIOps, sometimes called LLMOps, across initialization, experimentation, evaluation\u002Frefinement and deployment.",{},{"id":1496,"data":1497,"type":1466,"tunes":1503},"src-mlflow-genai",{"link":1498,"meta":1499},"https:\u002F\u002Fmlflow.org\u002Fdocs\u002Flatest\u002Fgenai\u002F",{"image":1500,"title":1501,"description":1502},{"url":381},"MLflow — Agents and LLM applications","Current GenAI operations documentation covering tracing, evaluation, prompts and production observability for LLM applications and agents.",{},{"id":1505,"data":1506,"type":1466,"tunes":1512},"src-mlflow-traces",{"link":1507,"meta":1508},"https:\u002F\u002Fwww.mlflow.org\u002Fdocs\u002Flatest\u002Fgenai\u002Feval-monitor\u002Frunning-evaluation\u002Ftraces\u002F",{"image":1509,"title":1510,"description":1511},{"url":381},"MLflow — Evaluating production traces","Current guidance for evaluating complete LLM\u002Fagent traces, including retrieval and tool-call trajectories.",{},{"id":1514,"data":1515,"type":1466,"tunes":1521},"src-mlflow-prompt-eval",{"link":1516,"meta":1517},"https:\u002F\u002Fmlflow.org\u002Fdocs\u002Flatest\u002Fgenai\u002Fprompt-registry\u002Fevaluate-prompts\u002F",{"image":1518,"title":1519,"description":1520},{"url":381},"MLflow — Evaluating prompts","Current prompt\u002Fmodel evaluation workflow using versioned prompts, datasets, scorers and traces.",{},{"id":1523,"data":1524,"type":1466,"tunes":1530},"src-openai-api",{"link":1525,"meta":1526},"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Freference\u002Foverview",{"image":1527,"title":1528,"description":1529},{"url":381},"OpenAI API — Versioning and model snapshots","Current API guidance recommending pinned model versions and evals because prompting behavior can change between snapshots.",{},{"id":1532,"data":1533,"type":1466,"tunes":1539},"src-openai-prompting",{"link":1534,"meta":1535},"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fprompting",{"image":1536,"title":1537,"description":1538},{"url":381},"OpenAI — Prompting","Current guidance to treat production prompts as application code, version them through source control and cover changes with tests and evaluation checks.",{},{"id":1541,"data":1542,"type":1466,"tunes":1548},"src-openai-deprecations",{"link":1543,"meta":1544},"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fdeprecations",{"image":1545,"title":1546,"description":1547},{"url":381},"OpenAI — Deprecations","Current provider lifecycle evidence showing model and platform-surface retirement as an operational dependency.",{},{"id":1550,"data":1551,"type":1466,"tunes":1557},"src-openai-promptfoo",{"link":1552,"meta":1553},"https:\u002F\u002Fdevelopers.openai.com\u002Fcookbook\u002Fexamples\u002Fevaluation\u002Fmoving-from-openai-evals-to-promptfoo",{"image":1554,"title":1555,"description":1556},{"url":381},"OpenAI — Moving evaluation workflows to Promptfoo","Current 2026 migration guidance illustrating why evaluation assets should remain portable as provider tooling changes.",{},"2.31.6","MLOps operates machine-learning systems; LLMOps extends those practices to prompts, context, retrieval, providers, tools, evaluations and runtime behavior around large language models.","\u002Fuploads\u002F2026\u002F10\u002Fmlops-vs-llmops-what-changes-when-the-model-is-an-llm-1791487319869-2v7hxo.webp","mlops-vs-llmops-what-changes-when-the-model-is-an-llm-1791487319869-2v7hxo","PUBLISHED","2026-10-08T15:20:00.000Z","2026-10-08T19:20:01.249Z","2026-10-08T19:31:17.955Z",{"en":1567,"de":1568,"sr":1569,"es":1570,"fr":1571,"it":1572,"ru":1573,"zh":1574},"\u002Fblog\u002Fmlops-vs-llmops-what-changes-when-the-model-is-an-llm","\u002Fde\u002Fblog\u002Fmlops-vs-llmops-what-changes-when-the-model-is-an-llm","\u002Fsr\u002Fblog\u002Fmlops-vs-llmops-what-changes-when-the-model-is-an-llm","\u002Fes\u002Fblog\u002Fmlops-vs-llmops-what-changes-when-the-model-is-an-llm","\u002Ffr\u002Fblog\u002Fmlops-vs-llmops-what-changes-when-the-model-is-an-llm","\u002Fit\u002Fblog\u002Fmlops-vs-llmops-what-changes-when-the-model-is-an-llm","\u002Fru\u002Fblog\u002Fmlops-vs-llmops-what-changes-when-the-model-is-an-llm","\u002Fzh\u002Fblog\u002Fmlops-vs-llmops-what-changes-when-the-model-is-an-llm",[1576,1580,1584,1588],{"id":1577,"name":1578,"slug":1579},88,"Versioning (Prompts, Models)","versioning",{"id":1581,"name":1582,"slug":1583},91,"Monitoring (Quality, Drift)","monitoring",{"id":1585,"name":1586,"slug":1587},89,"Evaluation Harness","evaluation-harness",{"id":1589,"name":1590,"slug":1591},58,"Evaluation & Quality Gates","evaluation",{"id":1593,"login":1594,"email":1595,"displayName":1596},"20","rooth8233","aleksandar@stajic.de","Aleksandar Stajić",[1598],{"lang":7,"title":208,"content":210,"contentJson":1599,"excerpt":1559},{"time":212,"blocks":1600,"version":1558},[1601,1604,1607,1610,1613,1616,1619,1622,1625,1628,1631,1634,1637,1640,1643,1646,1649,1652,1655,1667,1670,1673,1676,1679,1682,1705,1708,1711,1714,1717,1720,1739,1742,1745,1748,1751,1754,1757,1760,1763,1766,1769,1772,1775,1778,1781,1784,1787,1790,1793,1796,1799,1802,1805,1808,1811,1814,1817,1820,1823,1826,1829,1832,1835,1838,1841,1844,1847,1850,1853,1856,1859,1862,1865,1868,1871,1874,1877,1880,1883,1886,1889,1892,1895,1909,1912,1915,1918,1921,1924,1927,1930,1933,1936,1951,1954,1957,1960,1963,1966,1969,1972,1975,1978,1981,1984,1987,1990,1993,1996,1999,2002,2005,2008,2011,2014,2017,2020,2023,2026,2029,2042,2045,2048,2065,2068,2083,2086,2102,2105,2124,2127,2130,2133,2136,2139,2142,2145,2148,2151,2154,2157,2160,2163,2166,2169,2172,2175,2188,2191,2207,2210,2213,2216,2219,2222,2225,2230,2235,2240,2245,2250,2255,2260,2265,2270,2275],{"id":215,"data":1602,"type":218,"tunes":1603},{"text":217},{},{"id":221,"data":1605,"type":226,"tunes":1606},{"body":223,"title":224,"variant":225},{},{"id":229,"data":1608,"type":226,"tunes":1609},{"body":231,"title":232,"variant":233},{},{"id":236,"data":1611,"type":226,"tunes":1612},{"body":238,"title":239,"variant":240},{},{"id":243,"data":1614,"type":226,"tunes":1615},{"body":245,"title":246,"variant":240},{},{"id":249,"data":1617,"type":254,"tunes":1618},{"title":251,"maxLevel":252,"minLevel":253},{},{"id":257,"data":1620,"type":42,"tunes":1621},{"text":259,"level":253},{},{"id":262,"data":1623,"type":218,"tunes":1624},{"text":264},{},{"id":267,"data":1626,"type":218,"tunes":1627},{"text":269},{},{"id":272,"data":1629,"type":218,"tunes":1630},{"text":274},{},{"id":277,"data":1632,"type":42,"tunes":1633},{"text":279,"level":253},{},{"id":282,"data":1635,"type":218,"tunes":1636},{"text":284},{},{"id":287,"data":1638,"type":218,"tunes":1639},{"text":289},{},{"id":292,"data":1641,"type":218,"tunes":1642},{"text":294},{},{"id":297,"data":1644,"type":42,"tunes":1645},{"text":299,"level":253},{},{"id":302,"data":1647,"type":218,"tunes":1648},{"text":304},{},{"id":307,"data":1650,"type":218,"tunes":1651},{"text":309},{},{"id":312,"data":1653,"type":218,"tunes":1654},{"text":314},{},{"id":317,"data":1656,"type":346,"tunes":1666},{"steps":1657,"title":344,"orientation":345},[1658,1659,1660,1661,1662,1663,1664,1665],{"label":321,"description":322},{"label":324,"description":325},{"label":327,"description":328},{"label":330,"description":331},{"label":333,"description":334},{"label":336,"description":337},{"label":339,"description":340},{"label":342,"description":343},{},{"id":349,"data":1668,"type":42,"tunes":1669},{"text":351,"level":253},{},{"id":354,"data":1671,"type":218,"tunes":1672},{"text":356},{},{"id":359,"data":1674,"type":218,"tunes":1675},{"text":361},{},{"id":364,"data":1677,"type":218,"tunes":1678},{"text":366},{},{"id":369,"data":1680,"type":42,"tunes":1681},{"text":371,"level":253},{},{"id":374,"data":1683,"type":419,"tunes":1704},{"rows":1684,"title":410,"layout":411,"columns":1701},[1685,1687,1689,1691,1693,1695,1697,1699],{"id":378,"label":379,"values":1686},[381,381],{"id":383,"label":384,"values":1688},[381,381],{"id":387,"label":388,"values":1690},[381,381],{"id":391,"label":392,"values":1692},[381,381],{"id":395,"label":396,"values":1694},[381,381],{"id":399,"label":400,"values":1696},[381,381],{"id":403,"label":404,"values":1698},[381,381],{"id":407,"label":408,"values":1700},[381,381],[1702,1703],{"id":414,"label":415},{"id":417,"label":418},{},{"id":422,"data":1706,"type":42,"tunes":1707},{"text":424,"level":253},{},{"id":427,"data":1709,"type":218,"tunes":1710},{"text":429},{},{"id":432,"data":1712,"type":218,"tunes":1713},{"text":434},{},{"id":437,"data":1715,"type":218,"tunes":1716},{"text":439},{},{"id":442,"data":1718,"type":42,"tunes":1719},{"text":444,"level":253},{},{"id":447,"data":1721,"type":411,"tunes":1738},{"content":1722,"stretched":43,"withHeadings":14},[1723,1724,1725,1726,1727,1728,1729,1730,1731,1732,1733,1734,1735,1736,1737],[451,452],[454,455],[457,458],[460,461],[463,464],[466,467],[469,470],[472,473],[475,476],[478,479],[481,482],[484,485],[487,488],[490,491],[493,494],{},{"id":497,"data":1740,"type":42,"tunes":1741},{"text":499,"level":253},{},{"id":502,"data":1743,"type":218,"tunes":1744},{"text":504},{},{"id":507,"data":1746,"type":218,"tunes":1747},{"text":509},{},{"id":512,"data":1749,"type":218,"tunes":1750},{"text":514},{},{"id":517,"data":1752,"type":42,"tunes":1753},{"text":519,"level":253},{},{"id":522,"data":1755,"type":218,"tunes":1756},{"text":524},{},{"id":527,"data":1758,"type":218,"tunes":1759},{"text":529},{},{"id":532,"data":1761,"type":218,"tunes":1762},{"text":534},{},{"id":537,"data":1764,"type":42,"tunes":1765},{"text":539,"level":253},{},{"id":542,"data":1767,"type":218,"tunes":1768},{"text":544},{},{"id":547,"data":1770,"type":218,"tunes":1771},{"text":549},{},{"id":552,"data":1773,"type":218,"tunes":1774},{"text":554},{},{"id":557,"data":1776,"type":42,"tunes":1777},{"text":559,"level":253},{},{"id":562,"data":1779,"type":218,"tunes":1780},{"text":564},{},{"id":567,"data":1782,"type":218,"tunes":1783},{"text":569},{},{"id":572,"data":1785,"type":218,"tunes":1786},{"text":574},{},{"id":577,"data":1788,"type":42,"tunes":1789},{"text":579,"level":253},{},{"id":582,"data":1791,"type":218,"tunes":1792},{"text":584},{},{"id":587,"data":1794,"type":218,"tunes":1795},{"text":589},{},{"id":592,"data":1797,"type":218,"tunes":1798},{"text":594},{},{"id":597,"data":1800,"type":603,"tunes":1801},{"url":599,"title":600,"excerpt":601,"ctaLabel":602},{},{"id":606,"data":1803,"type":42,"tunes":1804},{"text":608,"level":253},{},{"id":611,"data":1806,"type":218,"tunes":1807},{"text":613},{},{"id":616,"data":1809,"type":218,"tunes":1810},{"text":618},{},{"id":621,"data":1812,"type":218,"tunes":1813},{"text":623},{},{"id":626,"data":1815,"type":226,"tunes":1816},{"body":628,"title":629,"variant":630},{},{"id":633,"data":1818,"type":42,"tunes":1819},{"text":635,"level":253},{},{"id":638,"data":1821,"type":218,"tunes":1822},{"text":640},{},{"id":643,"data":1824,"type":218,"tunes":1825},{"text":645},{},{"id":648,"data":1827,"type":218,"tunes":1828},{"text":650},{},{"id":653,"data":1830,"type":42,"tunes":1831},{"text":655,"level":253},{},{"id":658,"data":1833,"type":218,"tunes":1834},{"text":660},{},{"id":663,"data":1836,"type":218,"tunes":1837},{"text":665},{},{"id":668,"data":1839,"type":218,"tunes":1840},{"text":670},{},{"id":673,"data":1842,"type":42,"tunes":1843},{"text":675,"level":253},{},{"id":678,"data":1845,"type":218,"tunes":1846},{"text":680},{},{"id":683,"data":1848,"type":218,"tunes":1849},{"text":685},{},{"id":688,"data":1851,"type":218,"tunes":1852},{"text":690},{},{"id":693,"data":1854,"type":603,"tunes":1855},{"url":695,"title":696,"excerpt":697,"ctaLabel":698},{},{"id":701,"data":1857,"type":42,"tunes":1858},{"text":703,"level":253},{},{"id":706,"data":1860,"type":218,"tunes":1861},{"text":708},{},{"id":711,"data":1863,"type":218,"tunes":1864},{"text":713},{},{"id":716,"data":1866,"type":218,"tunes":1867},{"text":718},{},{"id":721,"data":1869,"type":42,"tunes":1870},{"text":723,"level":253},{},{"id":726,"data":1872,"type":218,"tunes":1873},{"text":728},{},{"id":731,"data":1875,"type":218,"tunes":1876},{"text":733},{},{"id":736,"data":1878,"type":218,"tunes":1879},{"text":738},{},{"id":741,"data":1881,"type":42,"tunes":1882},{"text":743,"level":253},{},{"id":746,"data":1884,"type":218,"tunes":1885},{"text":748},{},{"id":751,"data":1887,"type":218,"tunes":1888},{"text":753},{},{"id":756,"data":1890,"type":218,"tunes":1891},{"text":758},{},{"id":761,"data":1893,"type":42,"tunes":1894},{"text":763,"level":253},{},{"id":766,"data":1896,"type":411,"tunes":1908},{"content":1897,"stretched":43,"withHeadings":14},[1898,1899,1900,1901,1902,1903,1904,1905,1906,1907],[770,771],[773,774],[776,777],[779,780],[782,783],[785,786],[788,789],[791,792],[794,795],[797,798],{},{"id":801,"data":1910,"type":42,"tunes":1911},{"text":803,"level":253},{},{"id":806,"data":1913,"type":218,"tunes":1914},{"text":808},{},{"id":811,"data":1916,"type":218,"tunes":1917},{"text":813},{},{"id":816,"data":1919,"type":218,"tunes":1920},{"text":818},{},{"id":821,"data":1922,"type":42,"tunes":1923},{"text":823,"level":253},{},{"id":826,"data":1925,"type":218,"tunes":1926},{"text":828},{},{"id":831,"data":1928,"type":218,"tunes":1929},{"text":833},{},{"id":836,"data":1931,"type":218,"tunes":1932},{"text":838},{},{"id":841,"data":1934,"type":42,"tunes":1935},{"text":843,"level":253},{},{"id":846,"data":1937,"type":411,"tunes":1950},{"content":1938,"stretched":43,"withHeadings":14},[1939,1940,1941,1942,1943,1944,1945,1946,1947,1948,1949],[850,851],[853,854],[856,857],[859,860],[862,863],[865,866],[782,868],[788,870],[791,872],[874,875],[877,878],{},{"id":881,"data":1952,"type":42,"tunes":1953},{"text":883,"level":253},{},{"id":886,"data":1955,"type":218,"tunes":1956},{"text":888},{},{"id":891,"data":1958,"type":218,"tunes":1959},{"text":893},{},{"id":896,"data":1961,"type":218,"tunes":1962},{"text":898},{},{"id":901,"data":1964,"type":42,"tunes":1965},{"text":903,"level":253},{},{"id":906,"data":1967,"type":218,"tunes":1968},{"text":908},{},{"id":911,"data":1970,"type":218,"tunes":1971},{"text":913},{},{"id":916,"data":1973,"type":218,"tunes":1974},{"text":918},{},{"id":921,"data":1976,"type":42,"tunes":1977},{"text":923,"level":253},{},{"id":926,"data":1979,"type":218,"tunes":1980},{"text":928},{},{"id":931,"data":1982,"type":218,"tunes":1983},{"text":933},{},{"id":936,"data":1985,"type":218,"tunes":1986},{"text":938},{},{"id":941,"data":1988,"type":42,"tunes":1989},{"text":943,"level":253},{},{"id":946,"data":1991,"type":218,"tunes":1992},{"text":948},{},{"id":951,"data":1994,"type":218,"tunes":1995},{"text":953},{},{"id":956,"data":1997,"type":218,"tunes":1998},{"text":958},{},{"id":961,"data":2000,"type":42,"tunes":2001},{"text":963,"level":253},{},{"id":966,"data":2003,"type":42,"tunes":2004},{"text":968,"level":252},{},{"id":971,"data":2006,"type":218,"tunes":2007},{"text":973},{},{"id":976,"data":2009,"type":218,"tunes":2010},{"text":978},{},{"id":981,"data":2012,"type":218,"tunes":2013},{"text":983},{},{"id":986,"data":2015,"type":218,"tunes":2016},{"text":988},{},{"id":991,"data":2018,"type":42,"tunes":2019},{"text":993,"level":252},{},{"id":996,"data":2021,"type":218,"tunes":2022},{"text":998},{},{"id":1001,"data":2024,"type":218,"tunes":2025},{"text":1003},{},{"id":1006,"data":2027,"type":218,"tunes":2028},{"text":1008},{},{"id":1011,"data":2030,"type":411,"tunes":2041},{"content":2031,"stretched":43,"withHeadings":14},[2032,2033,2034,2035,2036,2037,2038,2039,2040],[1015,1016],[1018,1019],[1021,1022],[1024,1025],[1027,1028],[1030,1031],[1033,1034],[1036,1037],[1039,1040],{},{"id":1043,"data":2043,"type":226,"tunes":2044},{"body":1045,"title":1046,"variant":240},{},{"id":1049,"data":2046,"type":42,"tunes":2047},{"text":1051,"level":253},{},{"id":1054,"data":2049,"type":411,"tunes":2064},{"content":2050,"stretched":43,"withHeadings":14},[2051,2052,2053,2054,2055,2056,2057,2058,2059,2060,2061,2062,2063],[1058,1059],[1061,1062],[1064,1065],[1067,1068],[1070,1071],[1073,1074],[1076,1077],[1079,1080],[1082,1083],[1085,1086],[1088,1089],[1091,1092],[1094,1095],{},{"id":1098,"data":2066,"type":42,"tunes":2067},{"text":1100,"level":253},{},{"id":1103,"data":2069,"type":411,"tunes":2082},{"content":2070,"stretched":43,"withHeadings":14},[2071,2072,2073,2074,2075,2076,2077,2078,2079,2080,2081],[1107,1108],[1110,1111],[1113,1114],[1116,1117],[1119,1120],[1122,1123],[1125,1126],[1128,1129],[1131,1132],[1134,1135],[1137,1138],{},{"id":1141,"data":2084,"type":42,"tunes":2085},{"text":1143,"level":253},{},{"id":1146,"data":2087,"type":346,"tunes":2101},{"steps":2088,"title":1185,"orientation":345},[2089,2090,2091,2092,2093,2094,2095,2096,2097,2098,2099,2100],{"label":1150,"description":1151},{"label":1153,"description":1154},{"label":1156,"description":1157},{"label":1159,"description":1160},{"label":1162,"description":1163},{"label":1165,"description":1166},{"label":1168,"description":1169},{"label":1171,"description":1172},{"label":1174,"description":1175},{"label":1177,"description":1178},{"label":1180,"description":1181},{"label":1183,"description":1184},{},{"id":1188,"data":2103,"type":42,"tunes":2104},{"text":1190,"level":253},{},{"id":1193,"data":2106,"type":411,"tunes":2123},{"content":2107,"stretched":43,"withHeadings":14},[2108,2109,2110,2111,2112,2113,2114,2115,2116,2117,2118,2119,2120,2121,2122],[1197,1198],[1200,1201],[1203,1204],[1206,1207],[1209,1210],[1212,1213],[1215,1216],[1218,1219],[1221,1222],[1224,1225],[1227,1228],[1230,1231],[1233,1234],[1236,1237],[1239,1240],{},{"id":1243,"data":2125,"type":42,"tunes":2126},{"text":1245,"level":253},{},{"id":1248,"data":2128,"type":218,"tunes":2129},{"text":1250},{},{"id":1253,"data":2131,"type":218,"tunes":2132},{"text":1255},{},{"id":1258,"data":2134,"type":218,"tunes":2135},{"text":1260},{},{"id":1263,"data":2137,"type":218,"tunes":2138},{"text":1265},{},{"id":1268,"data":2140,"type":218,"tunes":2141},{"text":1270},{},{"id":1273,"data":2143,"type":42,"tunes":2144},{"text":1275,"level":253},{},{"id":1278,"data":2146,"type":218,"tunes":2147},{"text":1280},{},{"id":1283,"data":2149,"type":218,"tunes":2150},{"text":1285},{},{"id":1288,"data":2152,"type":218,"tunes":2153},{"text":1290},{},{"id":1293,"data":2155,"type":42,"tunes":2156},{"text":1295,"level":253},{},{"id":1298,"data":2158,"type":218,"tunes":2159},{"text":1300},{},{"id":1303,"data":2161,"type":218,"tunes":2162},{"text":1305},{},{"id":1308,"data":2164,"type":218,"tunes":2165},{"text":1310},{},{"id":1313,"data":2167,"type":603,"tunes":2168},{"url":1315,"title":1316,"excerpt":1317,"ctaLabel":1318},{},{"id":1321,"data":2170,"type":603,"tunes":2171},{"url":1323,"title":1324,"excerpt":1325,"ctaLabel":1326},{},{"id":1329,"data":2173,"type":42,"tunes":2174},{"text":1331,"level":253},{},{"id":1334,"data":2176,"type":1334,"tunes":2187},{"items":2177,"title":1373},[2178,2179,2180,2181,2182,2183,2184,2185,2186],{"id":1338,"answer":1339,"question":1340},{"id":1342,"answer":1343,"question":1344},{"id":1346,"answer":1347,"question":1348},{"id":1350,"answer":1351,"question":1352},{"id":1354,"answer":1355,"question":1356},{"id":1358,"answer":1359,"question":1360},{"id":1362,"answer":1363,"question":1364},{"id":1366,"answer":1367,"question":1368},{"id":1370,"answer":1371,"question":1372},{},{"id":1376,"data":2189,"type":42,"tunes":2190},{"text":1378,"level":253},{},{"id":1381,"data":2192,"type":1381,"tunes":2206},{"title":1383,"entries":2193},[2194,2195,2196,2197,2198,2199,2200,2201,2202,2203,2204,2205],{"term":415,"anchor":414,"definition":1386},{"term":418,"anchor":417,"definition":1388},{"term":1390,"anchor":1391,"definition":1392},{"term":400,"anchor":1394,"definition":1395},{"term":1397,"anchor":1398,"definition":1399},{"term":1401,"anchor":1402,"definition":1403},{"term":1405,"anchor":1406,"definition":1407},{"term":1409,"anchor":1410,"definition":1411},{"term":469,"anchor":1413,"definition":1414},{"term":1416,"anchor":1417,"definition":1418},{"term":1420,"anchor":1421,"definition":1422},{"term":1424,"anchor":1425,"definition":1426},{},{"id":1429,"data":2208,"type":42,"tunes":2209},{"text":1431,"level":253},{},{"id":1434,"data":2211,"type":218,"tunes":2212},{"text":1436},{},{"id":1439,"data":2214,"type":218,"tunes":2215},{"text":1441},{},{"id":1444,"data":2217,"type":218,"tunes":2218},{"text":1446},{},{"id":1449,"data":2220,"type":42,"tunes":2221},{"text":1451,"level":253},{},{"id":1454,"data":2223,"type":218,"tunes":2224},{"text":1456},{},{"id":1459,"data":2226,"type":1466,"tunes":2229},{"link":1461,"meta":2227},{"image":2228,"title":1464,"description":1465},{"url":381},{},{"id":1469,"data":2231,"type":1466,"tunes":2234},{"link":1471,"meta":2232},{"image":2233,"title":1474,"description":1475},{"url":381},{},{"id":1478,"data":2236,"type":1466,"tunes":2239},{"link":1480,"meta":2237},{"image":2238,"title":1483,"description":1484},{"url":381},{},{"id":1487,"data":2241,"type":1466,"tunes":2244},{"link":1489,"meta":2242},{"image":2243,"title":1492,"description":1493},{"url":381},{},{"id":1496,"data":2246,"type":1466,"tunes":2249},{"link":1498,"meta":2247},{"image":2248,"title":1501,"description":1502},{"url":381},{},{"id":1505,"data":2251,"type":1466,"tunes":2254},{"link":1507,"meta":2252},{"image":2253,"title":1510,"description":1511},{"url":381},{},{"id":1514,"data":2256,"type":1466,"tunes":2259},{"link":1516,"meta":2257},{"image":2258,"title":1519,"description":1520},{"url":381},{},{"id":1523,"data":2261,"type":1466,"tunes":2264},{"link":1525,"meta":2262},{"image":2263,"title":1528,"description":1529},{"url":381},{},{"id":1532,"data":2266,"type":1466,"tunes":2269},{"link":1534,"meta":2267},{"image":2268,"title":1537,"description":1538},{"url":381},{},{"id":1541,"data":2271,"type":1466,"tunes":2274},{"link":1543,"meta":2272},{"image":2273,"title":1546,"description":1547},{"url":381},{},{"id":1550,"data":2276,"type":1466,"tunes":2279},{"link":1552,"meta":2277},{"image":2278,"title":1555,"description":1556},{"url":381},{},"Post erfolgreich abgerufen",{"items":2282,"source":2364,"manualIds":2365,"manualMatchedIds":2366},[2283,2290,2297,2304,2310,2317,2324,2330,2337,2344,2351,2357],{"id":2284,"slug":2285,"title":2286,"excerpt":2287,"featuredImage":2288,"publishedAt":2289},"384","new-qwen-3-5-plus","New Qwen 3.5-Plus: Open-source AI is getting serious now","Discover the groundbreaking features and benefits of Alibaba's Qwen 3.5-Plus, a revolutionary open-source AI for developers.","\u002Fuploads\u002F2026\u002F02\u002Fnew-qwen-3-5-plus-1771515512741-dcbi9p.webp","2026-02-19T10:23:00.000Z",{"id":2291,"slug":2292,"title":2293,"excerpt":2294,"featuredImage":2295,"publishedAt":2296},"459","ollama-is-not-the-product-building-production-ready-open-llm-applications","Ollama Is Not the Product: Building Production-Ready Open-LLM Applications","Running a local model with Ollama is easy. Building a production-ready Open-LLM application is harder: it requires RAG, access control, provider abstraction, evaluation, logging, deployment discipline and a controlled application layer around the model.\n","\u002Fuploads\u002F2026\u002F06\u002Follama-is-not-the-product-building-production-ready-open-llm-applications-1782679361640-h0usqf.webp","2026-06-28T16:39:00.000Z",{"id":2298,"slug":2299,"title":2300,"excerpt":2301,"featuredImage":2302,"publishedAt":2303},"475","managed-agent-harness-vs-self-hosted-agent-loop-what-you-gain-what-you-lose","Managed Agent Harness vs Self-Hosted Agent Loop: What You Gain, What You Lose","“Self-hosted agent” can mean very different architectures. This guide separates the managed harness, self-hosted execution environment, and fully self-operated agent loop—and shows which control boundary teams actually need.","\u002Fuploads\u002F2026\u002F09\u002Fmanaged-agent-harness-vs-self-hosted-agent-loop-what-you-gain-what-you-lose-1790352403475-kj10jh.webp","2026-09-25T12:05:00.000Z",{"id":2305,"slug":2306,"title":696,"excerpt":2307,"featuredImage":2308,"publishedAt":2309},"460","ai-agent-reliability-why-the-final-answer-is-not-enough","Correct output does not prove correct reasoning, safe execution, or a trustworthy system.","\u002Fuploads\u002F2026\u002F09\u002Fai-agent-reliability-why-the-final-answer-is-not-enough-1788955466306-pl0qhz.webp","2026-09-09T04:01:00.000Z",{"id":2311,"slug":2312,"title":2313,"excerpt":2314,"featuredImage":2315,"publishedAt":2316},"488","what-is-context-engineering-what-the-model-receives-before-it-answers","What Is Context Engineering? What the Model Receives Before It Answers","Context engineering designs what information an AI model receives before inference, including prompts, retrieval, memory, application state, tool results and conversation history.","\u002Fuploads\u002F2026\u002F10\u002Fwhat-is-context-engineering-what-the-model-receives-before-it-answers-1791480653258-018kcv.webp","2026-10-08T13:29:00.000Z",{"id":2318,"slug":2319,"title":2320,"excerpt":2321,"featuredImage":2322,"publishedAt":2323},"480","when-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger","When Should an AI Stop Trusting Its Own Knowledge? — The Retrieval Trigger","An AI model does not need retrieval for every question. The important problem is knowing when its internal knowledge is no longer enough. The Retrieval Trigger is a practical decision boundary that determines when an AI system should stop relying solely on model knowledge and obtain external evidence before answering.","\u002Fuploads\u002F2026\u002F09\u002Fwhen-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger-1790574991244-f4rpyg.webp","2026-09-28T01:49:00.000Z",{"id":2325,"slug":2326,"title":2327,"excerpt":2328,"featuredImage":2288,"publishedAt":2329},"445","qwen-3-6-in-production-release-runbook-ai-rollback-and-llmops-versioning","Qwen 3.6 in Production: Release Runbook, AI Rollback, and LLMOps Versioning","Qwen 3.6 is not just another model upgrade. It is a release event, a rollback scenario, and a versioning problem at the same time. This article explains how Qwen 3.6 should be handled in production through LLMOps discipline, prompt and model traceability, controlled rollout, and evidence-based rollback readiness.","2026-05-04T02:49:00.000Z",{"id":2331,"slug":2332,"title":2333,"excerpt":2334,"featuredImage":2335,"publishedAt":2336},"478","what-is-rag-the-simplest-explanation-of-how-it-works","What Is RAG? The Simplest Explanation of How It Works","RAG sounds complicated, but the idea is simple: before an AI answers, it first looks up useful information from a knowledge source and gives that information to the language model. This guide explains RAG, LLMs, state, memory and tools using one simple mental model.","\u002Fuploads\u002F2026\u002F09\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works-1790377492124-khjagt.webp","2026-09-25T19:03:00.000Z",{"id":2338,"slug":2339,"title":2340,"excerpt":2341,"featuredImage":2342,"publishedAt":2343},"455","zbt-z8102ax-dual-sim-failover-test","ZBT Z8102AX Dual-SIM Failover: What Works, What Is Missing and What Needs Better Firmware","The ZBT Z8102AX is a dual-SIM 5G OpenWrt router, but dual-SIM hardware alone is not the same as intelligent failover. The router recognizes the SIM and connects successfully, but automatic switching, modem recovery, signal-based decisions and clean failover logic still need deeper testing.","\u002Fuploads\u002F2026\u002F06\u002Fopenwrt-router-review-dual-sim-03-1781620592829-7t77j7.webp","2026-06-16T10:40:00.000Z",{"id":2345,"slug":2346,"title":2347,"excerpt":2348,"featuredImage":2349,"publishedAt":2350},"364","tipps-fuer-die-verbesserung-der-seo-suchmaschinenoptimierung","Mastering the SEO Workflow: Essential Optimization Strategies for Organic Growth","A structured SEO workflow is crucial for sustainable organic growth. Learn the ten foundational strategies, from keyword research and technical optimization to content quality and performance analysis.","\u002Fuploads\u002F2026\u002F03\u002Ftipps-fuer-die-verbesserung-der-seo-suchmaschinenoptimierung-1774866098131-hwkzrg.webp","2024-01-26T06:35:00.000Z",{"id":2352,"slug":2353,"title":600,"excerpt":2354,"featuredImage":2355,"publishedAt":2356},"469","rag-failed-but-which-layer-actually-failed-a-diagnostic-method","When a RAG answer is wrong, blaming retrieval or the model is too vague. This diagnostic method isolates source coverage, query construction, retrieval, ranking, context assembly, generation, evidence attribution, and freshness—so the actual failure can be reproduced and fixed.","\u002Fuploads\u002F2026\u002F09\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method-1790350847177-pior4c.webp","2026-09-24T19:39:00.000Z",{"id":2358,"slug":2359,"title":2360,"excerpt":2361,"featuredImage":2362,"publishedAt":2363},"477","computer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system","Computer-Use Agents: Why a Successful Demo Can Still Be an Unreliable System","Computer-use agents can now complete impressive browser and desktop workflows, but one successful run proves capability—not reliability. This article shows how to test repeatability, environmental robustness, long-horizon control, state awareness, outcome verification, and safe goal handling.","\u002Fuploads\u002F2026\u002F09\u002Fcomputer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system-1790352854690-75qnrg.webp","2026-09-25T12:13:00.000Z","fallback",[],[]]