[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"portal-settings:stajic:zh":3,"public-menus:all":38,"post:mlops-vs-llmops-what-changes-when-the-model-is-an-llm:zh":205,"related:post:mlops-vs-llmops-what-changes-when-the-model-is-an-llm:zh:1":3406},{"statusCode":4,"data":5,"message":37},200,{"tenantId":6,"lang":7,"defaultLang":8,"siteUrl":9,"contactEmail":10,"brandName":11,"logoUrl":12,"siteName":11,"siteDescription":13,"ogImage":10,"robotsIndex":14,"socialLinks":10,"reservedSlugs":10,"seoPolicy":15},"stajic","zh","de","https:\u002F\u002Fstajic.de",null,"Stajic Platform","\u002FLogo_Planet.svg","Stajic Portal",true,{"branding":16,"relatedContent":17,"crossDomainLinks":18},{"logoUrl":12},{"enabled":14},[19,22,25,28,31,34],{"url":20,"label":21,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Ffigure.rocks","figure.rocks",{"url":23,"label":24,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Floving.rocks","loving.rocks",{"url":26,"label":27,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.com","bazify.com",{"url":29,"label":30,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.de","bazify.de",{"url":32,"label":33,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.at","bazify.at",{"url":35,"label":36,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.ba","bazify.ba","Portal settings resolved",[39,45],{"id":40,"name":41,"location":42,"isActive":14,"isDefault":43,"items":44},1,"main-navigation","header",false,[],{"id":46,"name":47,"location":48,"isActive":14,"isDefault":14,"items":49},4,"main-menu","sidebar",[50,66,79,93,103,118,133],{"id":51,"title":52,"url":60,"target":61,"icon":62,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":64,"portfolioId":10,"children":65},"item-18",{"de":53,"en":54,"es":55,"fr":56,"it":54,"ru":57,"sr":58,"zh":59},"Startseite","Home","Inicio","Accueil","Главная","Почетна","首页","\u002Ffull-stack-web-developer-munich-performance-seo-and-maintainable-builds","_self","i-lucide-home","page",111,[],{"id":67,"title":68,"url":75,"target":61,"icon":76,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":77,"portfolioId":10,"children":78},"item-22",{"de":69,"en":69,"es":70,"fr":69,"it":71,"ru":72,"sr":73,"zh":74},"Vision","Visión","Visione","Видение","Визија","想象","\u002Fueber-uns-webdesign-muenchen-webaplikation","i-lucide-eye",113,[],{"id":80,"title":81,"url":89,"target":61,"icon":90,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":91,"portfolioId":10,"children":92},"item-19",{"de":82,"en":83,"es":84,"fr":83,"it":85,"ru":86,"sr":87,"zh":88},"Leistungen","Services","Servicios","Servizi","Услуги","Услуге","服务","\u002Fservices-dienstleistungen-muenchen","i-lucide-wrench",116,[],{"id":94,"title":95,"url":99,"target":61,"icon":100,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":101,"portfolioId":10,"children":102},"item-23",{"de":96,"en":96,"es":96,"fr":96,"it":96,"ru":97,"sr":97,"zh":98},"Blog","Блог","博客","\u002Fblog","i-lucide-book-open",112,[],{"id":104,"title":105,"url":114,"target":61,"icon":115,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":116,"portfolioId":10,"children":117},"item-32",{"de":106,"en":107,"es":108,"fr":109,"it":110,"ru":111,"sr":112,"zh":113},"Neue Technologien","New Technologies","Nuevas tecnologías","Nouvelles technologies","Nuove tecnologie","Новые технологии","Нове технологије","新技术！","\u002Fneue-webtechnologien","i-lucide-sparkles",122,[],{"id":119,"title":120,"url":129,"target":61,"icon":130,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":131,"portfolioId":10,"children":132},"item-20",{"de":121,"en":122,"es":123,"fr":124,"it":125,"ru":126,"sr":127,"zh":128},"Kontakt","Contact us!","Contacto","Contact","Contatto","Контакт","Контактирајте нас","联系我们！","\u002Fcontact","i-lucide-mail",115,[],{"id":134,"title":135,"url":144,"target":61,"icon":145,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":147},"item-21",{"de":136,"en":137,"es":138,"fr":139,"it":140,"ru":141,"sr":142,"zh":143},"Unsere Arbeit","Our Work","Nuestro trabajo","Nos réalisations","I nostri lavori","Наши работы","Наши радови","文件夹","\u002Fportfolio","i-lucide-briefcase",114,[148,161,175,181,193],{"id":149,"title":150,"url":144,"target":61,"icon":159,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":160},"item-24",{"de":151,"en":152,"es":153,"fr":154,"it":155,"ru":156,"sr":157,"zh":158},"Alle Projekte","All Projects","Todos los proyectos","Tous les projets","Tutti i progetti","Все проекты","Сви пројекти","所有项目","i-lucide-grid-3x3",[],{"id":162,"title":163,"url":171,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":174},"item-29",{"de":164,"en":165,"es":166,"fr":167,"it":168,"ru":169,"sr":170,"zh":143},"Local Roots, Global Reach","Local Roots - Global Reach","Empresa local ","Entreprise locale","Azienda locale","Местная компания","Локално предузеће глобално тржиште","\u002Fportfolio\u002Flocal-roots-global-reach-communication-media-systems-for-modern-business","i-lucide-folder","custom",[],{"id":176,"title":177,"url":179,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":180},"item-28",{"de":178,"en":178,"es":178,"fr":178,"it":178,"ru":178,"sr":178,"zh":178},"Solr Suggester","\u002Fportfolio\u002Fsolr-fuzzy-suggester-und-solr-infix-suggester-abfrage-ueber-ajax-und-filterung",[],{"id":182,"title":183,"url":191,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":192},"item-27",{"de":184,"en":185,"es":186,"fr":187,"it":188,"ru":189,"sr":190,"zh":185},"Firmenwebseite SEO","Company Website SEO","Sitio web corporativo SEO","Site web d’entreprise SEO","Sito web aziendale SEO","Корпоративный сайт SEO","Пословна веб-страница SEO","\u002Fportfolio\u002Fseo-sem-branding-mobile-webseite-muenchen",[],{"id":194,"title":195,"url":203,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":204},"item-31",{"de":196,"en":197,"es":198,"fr":199,"it":200,"ru":201,"sr":202,"zh":197},"Digitalisierungsportal","Digitalization Portal","Portal de digitalización","Portail de numérisation","Portale di digitalizzazione","Портал цифровизации","Портал за дигитализацију","\u002Fportfolio\u002Fdigitalisierungsportal-archiv-museum-bibliothek-ead-lido-mets-mods",[],{"statusCode":4,"data":206,"message":3405},{"id":207,"title":208,"slug":209,"content":210,"contentJson":211,"excerpt":1558,"featuredImage":1559,"featuredImageAlt":1560,"featuredImageCaption":10,"featuredImageTitle":10,"featuredImageCopyright":10,"featuredImageAuthor":10,"featuredImageSourceUrl":10,"featuredImageLicense":10,"featuredImageIsAiGenerated":43,"status":1561,"publishedAt":1562,"createdAt":1563,"updatedAt":1564,"seoLocalePaths":1565,"categories":1574,"author":1591,"translations":1596},"493","MLOps 与 LLMOps：当模型是 LLM 时，会发生哪些变化","mlops-vs-llmops-what-changes-when-the-model-is-an-llm","\u003Cp>MLOps 是用于可靠地开发、部署、版本管理和运维机器学习系统的工程学科；LLMOps 将这一学科扩展到围绕大语言模型构建的应用，在这类应用中，生产行为不仅取决于模型产物，还取决于提示词、上下文、检索、提供商\u002F模型版本、工具调用、安全控制和评估流水线。LLMOps 并不取代 MLOps。它将运维单元从“一个模型加服务流水线”转变为“一个不断演进的 LLM 应用，其行为由多个独立变化的组件共同涌现”。\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--info my-6 rounded-xl border p-5 border-blue-300 bg-blue-50 dark:border-blue-900 dark:bg-blue-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">直接回答\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">\u003Cstrong>MLOps 运维 ML 系统。LLMOps 运维 LLM 应用。\u003C\u002Fstrong>\u003Cbr>\u003Cbr>经典 MLOps 通常以数据流水线、训练、验证、模型注册表、部署、漂移和再训练为中心。LLMOps 在相关之处保留这些学科，但通常会增加提示词\u002F上下文版本管理、模型\u002F提供商抽象、RAG 索引、智能体\u002F工具追踪、语义评估、安全测试、token\u002F成本监控，以及针对快速变化的模型快照的回归测试。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Caside class=\"editorjs-callout editorjs-callout--warning my-6 rounded-xl border p-5 border-amber-300 bg-amber-50 dark:border-amber-900 dark:bg-amber-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">LLMOps 不只是提示词管理\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">即使提示词没有变化，生产环境中的 LLM 应用也可能失败：提供商可能更改模型快照，RAG 语料库可能过时，重排序器可能退化，工具权限可能变化，上下文组装可能丢失证据，或者智能体可能采取错误轨迹。因此，LLMOps 必须观察并版本化模型周围的系统，而不仅仅是提示词文本。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Caside class=\"editorjs-callout editorjs-callout--note my-6 rounded-xl border p-5 border-gray-300 bg-gray-50 dark:border-gray-700 dark:bg-gray-900\u002F40\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">术语边界\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">\u003Cstrong>LLMOps\u003C\u002Fstrong>、\u003Cstrong>GenAIOps\u003C\u002Fstrong> 及相关术语是被广泛使用的工程标签，但它们并不是一个具有单一规范生命周期的通用正式标准。Microsoft 目前将 GenAIOps 描述为“有时称为 LLMOps”，而 MLflow 则将围绕智能体和 LLM 应用的运维工具归为一类。本文使用 LLMOps 作为一个实用架构术语，指代运维那些行为实质上依赖 LLM 的生产系统。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Caside class=\"editorjs-callout editorjs-callout--note my-6 rounded-xl border p-5 border-gray-300 bg-gray-50 dark:border-gray-700 dark:bg-gray-900\u002F40\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">当前来源说明 — 2026年10月8日\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">运维界面正在快速变化。OpenAI 目前建议固定模型快照并运行评估，因为提示行为可能在不同快照之间发生变化，而且若干较旧的平台特定提示词\u002F评估界面将在 2026 年退役。稳定的架构经验是：让提示词、测试和评估与应用程序一起保持可移植并版本化，而不是依赖某个提供商的仪表盘对象模型。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Cnav class=\"editorjs-toc\" data-editorjs-toc=\"true\" aria-label=\"目录\">\u003Cstrong class=\"editorjs-toc__title\">目录\u003C\u002Fstrong>\u003Col class=\"editorjs-toc__list editorjs-toc__list--depth-0\">\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-7\" class=\"editorjs-toc__link\">MLOps 的真正含义\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-11\" class=\"editorjs-toc__link\">当模型是 LLM 时会发生什么变化\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-15\" class=\"editorjs-toc__link\">最简单的例子\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-20\" class=\"editorjs-toc__link\">简单示例止步之处\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-24\" class=\"editorjs-toc__link\">MLOps 与 LLMOps\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-26\" class=\"editorjs-toc__link\">LLMOps 扩展 MLOps 而非取代它\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-30\" class=\"editorjs-toc__link\">LLMOps 中必须版本化哪些内容？\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-32\" class=\"editorjs-toc__link\">模型快照成为发布依赖项\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-36\" class=\"editorjs-toc__link\">提供商生命周期成为运营的一部分\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-40\" class=\"editorjs-toc__link\">提示词的行为类似于生产代码\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-44\" class=\"editorjs-toc__link\">上下文工程成为一项运营关注点\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-48\" class=\"editorjs-toc__link\">RAG 创建了自己的运营生命周期\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-53\" class=\"editorjs-toc__link\">评估用发布证据取代“在我看来不错”\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-58\" class=\"editorjs-toc__link\">LLM 作为评判者有用但并非绝对真理\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-62\" class=\"editorjs-toc__link\">追踪变得比端点日志更重要\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-66\" class=\"editorjs-toc__link\">代理将 LLMOps 扩展到运行时操作\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-71\" class=\"editorjs-toc__link\">Token、模型调用和上下文成为成本变量\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-75\" class=\"editorjs-toc__link\">缓存变得语义化，而不仅仅是技术性\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-79\" class=\"editorjs-toc__link\">安全与权限成为发布标准\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-83\" class=\"editorjs-toc__link\">LLMOps中的CI是什么样的\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-85\" class=\"editorjs-toc__link\">LLMOps中的CD是什么样的\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-89\" class=\"editorjs-toc__link\">持续训练变为可选；持续评估成为核心\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-93\" class=\"editorjs-toc__link\">生产环境中应监控哪些内容？\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-95\" class=\"editorjs-toc__link\">生产追踪可以成为评估数据\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-99\" class=\"editorjs-toc__link\">可复现性变为有条件的，而非精确的\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-103\" class=\"editorjs-toc__link\">血缘从模型血缘扩展为应用血缘\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-107\" class=\"editorjs-toc__link\">多提供商和模型路由产生运营策略\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-111\" class=\"editorjs-toc__link\">原始实现证据\u003C\u002Fa>\u003Col class=\"editorjs-toc__list editorjs-toc__list--depth-1\">\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-112\" class=\"editorjs-toc__link\">Aaasaasa AI Client：提供商、模型和运行时是独立的运维对象\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-117\" class=\"editorjs-toc__link\">Source of Truth Research Engine：LLM 应用状态超出模型本身\u003C\u002Fa>\u003C\u002Fli>\u003C\u002Fol>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-123\" class=\"editorjs-toc__link\">常见 LLMOps 故障模式\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-125\" class=\"editorjs-toc__link\">常见误解\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-127\" class=\"editorjs-toc__link\">实用的LLMOps设计流程\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-129\" class=\"editorjs-toc__link\">LLMOps架构检查清单\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-131\" class=\"editorjs-toc__link\">边缘情况和限制\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-137\" class=\"editorjs-toc__link\">什么会改变这个答案？\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-141\" class=\"editorjs-toc__link\">相关规范知识\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-147\" class=\"editorjs-toc__link\">常见问题\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-149\" class=\"editorjs-toc__link\">术语表\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-151\" class=\"editorjs-toc__link\">结论\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-155\" class=\"editorjs-toc__link\">主要来源和当前文档\u003C\u002Fa>\u003C\u002Fli>\u003C\u002Fol>\u003C\u002Fnav>\n\u003Ch2 id=\"section-7\">MLOps 的真正含义\u003C\u002Fh2>\n\u003Cp>MLOps 将软件工程和运维纪律应用于机器学习系统。生产挑战不仅仅是训练模型：数据收集、数据验证、实验、可复现性、模型评估、部署、基础设施和监控都必须协同工作。\u003C\u002Fp>\n\u003Cp>Google 的 MLOps 架构指南围绕持续集成、持续交付和持续训练来构建这一学科。CI 不仅验证代码，还验证数据、模式和模型；CD 部署 ML 流水线和预测服务；CT 可以在数据或实现变化时重新训练并重新部署模型。\u003C\u002Fp>\n\u003Cp>AWS 指南从另一个角度补充了相同的运维关注点：模型血缘、模型\u002F版本可追溯性、漂移监控和生产质量监控是部署后保持 ML 系统可靠性的核心部分。\u003C\u002Fp>\n\u003Ch2 id=\"section-11\">当模型是 LLM 时会发生什么变化\u003C\u002Fh2>\n\u003Cp>大语言模型改变了生产问题，因为应用通常并不拥有完整的模型训练生命周期。团队可能调用托管模型 API、在本地运行开源模型、在提供商之间切换，或为不同任务使用多个模型。\u003C\u002Fp>\n\u003Cp>因此，模型只是更大的行为系统中一个带版本的依赖项。提示词、检索结果、上下文顺序、工具、模型快照、温度\u002F推理设置、安全过滤器和运行时编排都可能改变输出。\u003C\u002Fp>\n\u003Cp>这带来了一个更广泛的运维问题：是模型、上下文、数据、提示词、工具和运行时的哪种组合产生了这种行为？LLMOps 的存在就是为了让这个问题可回答，并让答案足够可复现，以支持工程工作。\u003C\u002Fp>\n\u003Ch2 id=\"section-15\">最简单的例子\u003C\u002Fh2>\n\u003Cp>假设一个应用回答内部政策问题。\u003C\u002Fp>\n\u003Cp>在经典 ML 框架中，你可能会对一个训练好的分类器进行版本管理、部署它并监控预测质量。在 LLM 应用中，答案可能取决于托管模型快照、系统提示词、嵌入模型、向量索引、检索过滤器、重排序器以及最终选定的上下文。\u003C\u002Fp>\n\u003Cp>即使应用端点和用户问题保持不变，更改其中任何一个组件都可能改变最终答案。\u003C\u002Fp>\n\u003Csection class=\"editorjs-process my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">典型的 LLMOps 发布路径\u003C\u002Fh3>\u003Cdiv class=\"grid grid-cols-1 md:grid-cols-2 xl:grid-cols-3 gap-4\">\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">1\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">1. 更改一个组件\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">提示词、模型、提供商、检索设置、工具模式或应用代码发生变化。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">2\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">2. 运行确定性测试\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">验证模式、权限、工具契约、检索过滤器和应用行为。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">3\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">3. 运行行为评估\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">根据验收标准比较代表性输出、检索质量以及智能体\u002F工具轨迹。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">4\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">4. 比较成本和延迟\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">测量令牌使用量、模型调用次数、检索\u002F工具开销和响应延迟。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">5\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">5. 部署受控版本\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">发布具体的应用配置，并记录模型\u002F提供商版本。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">6\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">6. 追踪生产行为\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">捕获相关的模型、检索、工具和运行时跨度。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">7\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">7. 评估生产追踪\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">对真实执行进行采样，以评估质量、依据性、安全性和任务成功率。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">8\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">8. 回滚或迭代\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">利用回归证据和运营信号来决定下一次发布。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-20\">简单示例止步之处\u003C\u002Fh2>\n\u003Cp>一些 LLM 系统仍然训练或微调自己的模型，因此训练流水线、模型注册表和数据血缘等传统 MLOps 实践仍然直接相关。\u003C\u002Fp>\n\u003Cp>其他系统仅使用外部基础模型 API，从不运行持续训练。它们的主要运营工作是应用评估、模型\u002F提供商变更管理、提示词\u002F上下文版本控制、检索质量和可观测性。\u003C\u002Fp>\n\u003Cp>因此，不存在单一的通用“LLMOps 流水线”。确切的生命周期取决于你是训练、微调、自托管、检索外部知识、运行智能体，还是依赖托管模型 API。\u003C\u002Fp>\n\u003Ch2 id=\"section-24\">MLOps 与 LLMOps\u003C\u002Fh2>\n\u003Csection class=\"editorjs-comparison my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">哪些保持不变，哪些扩展\u003C\u002Fh3>\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left dark:border-gray-700 dark:bg-gray-900\">\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">MLOps\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">LLMOps\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">主要运营单元\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">模型所有权\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">典型变更\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">评估\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">生产监控\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">持续训练\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">版本化产物\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">回滚目标\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-26\">LLMOps 扩展 MLOps 而非取代它\u003C\u002Fh2>\n\u003Cp>核心运营原则并未消失：源代码控制、CI\u002FCD、可复现性、血缘、部署控制、监控、回滚和可衡量的验收标准仍然至关重要。\u003C\u002Fp>\n\u003Cp>扩展之处在于，更多定义行为的产物现在位于模型权重之外。托管的基础模型可以通过快照升级改变行为，而应用输出可以通过提示词或检索更改而无需任何模型重新训练即可改变。\u003C\u002Fp>\n\u003Cp>这就是为什么有用的层次结构通常是 DevOps → MLOps → LLMOps\u002FGenAIOps，作为日益专业化的运营关注点，而不是三种相互排斥的实践。\u003C\u002Fp>\n\u003Ch2 id=\"section-30\">LLMOps 中必须版本化哪些内容？\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">产物\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">为何重要\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">应用代码\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">定义编排、验证、重试和业务行为\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型家族 + 快照\u002F版本\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">不同的快照可能产生不同的行为\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">提供商 \u002F 端点\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">改变数据流、延迟、限制、定价和可用性\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">提示词\u002F指令代码\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">即使模型相同，也会改变模型行为\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">生成\u002F推理参数\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">可能改变确定性、延迟、深度和成本\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">评估数据集\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">定义“足够好”的测试基准\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">评分器 \u002F 打分器\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">定义如何衡量质量\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">嵌入模型\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">改变向量表示和检索行为\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">分块\u002F索引配置\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">改变可检索的内容\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">重排序器 \u002F 检索融合\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">改变结果排序\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">工具模式\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">改变模型可以请求什么以及如何请求\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">权限配置文件\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">改变哪些工具操作可以实际执行\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">上下文组装规则\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">改变哪些证据和状态到达模型\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">安全\u002F护栏配置\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">改变允许或阻止的行为\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-32\">模型快照成为发布依赖项\u003C\u002Fh2>\n\u003Cp>对于托管的 LLM，团队可能无法控制模型训练，但仍然控制应用程序调用哪个模型或快照。\u003C\u002Fp>\n\u003Cp>OpenAI 当前的 API 指南明确警告，提示行为可能在不同模型快照之间发生变化，并建议在一致性重要的生产应用程序中固定到特定快照，然后在升级时运行评估。\u003C\u002Fp>\n\u003Cp>运营后果很直接：模型升级应被视为应用程序发布，而不是不可见的基础设施维护。\u003C\u002Fp>\n\u003Ch2 id=\"section-36\">提供商生命周期成为运营的一部分\u003C\u002Fh2>\n\u003Cp>LLM 应用通常依赖于提供商的速率限制、弃用时间表、API 语义、上下文限制、数据处理规则和定价。\u003C\u002Fp>\n\u003Cp>即使你的应用程序代码保持不变，提供商也可能弃用某个模型。例如，OpenAI 当前的弃用时间表包括针对旧模型快照和平台界面的 2026 年退役日期。\u003C\u002Fp>\n\u003Cp>因此，LLMOps 除了需要模型质量监控外，还需要提供商生命周期跟踪、迁移测试和回退决策。\u003C\u002Fp>\n\u003Ch2 id=\"section-40\">提示词的行为类似于生产代码\u003C\u002Fh2>\n\u003Cp>提示词是可执行的行为配置。微小的更改可能会改变输出质量、工具选择和策略解释。\u003C\u002Fp>\n\u003Cp>OpenAI 当前的指南建议将生产提示词存储在应用程序代码中，通过拉取请求审查提示词更改，使用类型化输入，并通过测试和评估检查覆盖更改。\u003C\u002Fp>\n\u003Cp>这使得提示词版本管理不再像编辑营销文案，而更像更改一个输出具有概率性且依赖于模型的函数。\u003C\u002Fp>\n\u003Ch2 id=\"section-44\">上下文工程成为一项运营关注点\u003C\u002Fh2>\n\u003Cp>生产模型很少只接收静态提示词。它可能接收对话历史、检索到的文档、工具输出、记忆、当前应用程序状态和策略指令。\u003C\u002Fp>\n\u003Cp>因此，LLMOps 必须观察上下文组装：选择了哪些证据、当前是哪个状态版本、是否发生了截断，以及重要指令是否在压缩后保留下来。\u003C\u002Fp>\n\u003Cp>模型回归和上下文回归在最终答案上可能看起来完全相同。追踪实际的上下文路径才能让团队区分它们。\u003C\u002Fp>\n\u003Ch2 id=\"section-48\">RAG 创建了自己的运营生命周期\u003C\u002Fh2>\n\u003Cp>RAG 系统在模型推理之外引入了第二条生产流水线：摄取、提取、分块、元数据、嵌入、索引、检索、重排序和上下文选择。\u003C\u002Fp>\n\u003Cp>即使模型和提示词没有变化，知识语料库也可能每天发生变化。因此，过时的索引或损坏的元数据过滤器可能会在没有任何模型漂移的情况下降低答案质量。\u003C\u002Fp>\n\u003Cp>针对 RAG 的 LLMOps 应独立于生成质量，分别跟踪语料库\u002F索引版本、嵌入模型、分块策略、检索配置、源新鲜度和检索指标。\u003C\u002Fp>\n\u003Caside class=\"editorjs-referral my-6\">\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method\" class=\"flex flex-col sm:flex-row gap-4 rounded-xl border border-gray-200 dark:border-gray-700 p-4 transition hover:border-primary-500\">\u003Cdiv class=\"min-w-0 flex-1\">\u003Cstrong class=\"block text-lg text-gray-900 dark:text-gray-100\">RAG 失败了——但究竟是哪一层失败了？一种诊断方法\u003C\u002Fstrong>\u003Cp class=\"mt-2 text-sm text-gray-600 dark:text-gray-300\">生产 LLM 流水线需要对源覆盖、检索、排序、上下文组装和生成进行单独的可观测性。\u003C\u002Fp>\u003Cspan class=\"mt-3 inline-flex text-sm font-medium text-primary-600 dark:text-primary-400\">阅读 RAG 诊断方法 →\u003C\u002Fspan>\u003C\u002Fdiv>\u003C\u002Fa>\u003C\u002Faside>\n\u003Ch2 id=\"section-53\">评估用发布证据取代“在我看来不错”\u003C\u002Fh2>\n\u003Cp>生成式输出通常是开放式的，因此精确匹配测试对许多任务来说是不够的。LLMOps 增加了评估数据集和评分器，可以衡量任务成功、正确性、安全性、有据性、风格或特定领域的验收标准。\u003C\u002Fp>\n\u003Cp>MLflow 当前的 GenAI 评估栈支持版本化评估数据集、提示\u002F模型比较、自定义评分器以及对完整追踪的评估。\u003C\u002Fp>\n\u003Cp>最强的实践是评估驱动开发：在变更之前或同时定义代表性案例和验收标准，然后用相同的证据比较各个版本。\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--success my-6 rounded-xl border p-5 border-emerald-300 bg-emerald-50 dark:border-emerald-900 dark:bg-emerald-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">行为变更需要行为测试\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">仅仅因为 API 契约仍然有效，不应认为部署是等价的。如果提示、模型、检索或工具发生了变化，行为回归测试套件应重新运行。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-58\">LLM 作为评判者有用但并非绝对真理\u003C\u002Fh2>\n\u003Cp>LLM 评判者可以扩展对那些难以编码为确定性断言的品质的评估，例如相关性、语气或事实依据。\u003C\u002Fp>\n\u003Cp>然而，评判者是另一个具有自身偏见、版本和提示的模型。因此，在后果重要的地方，评判者配置应进行版本管理，并针对人类或确定性参考案例进行校准。\u003C\u002Fp>\n\u003Cp>生产评估可以混合使用确定性检查、基于参考的指标、模型评判者和人工审查，而不是要求单一指标代表所有质量维度。\u003C\u002Fp>\n\u003Ch2 id=\"section-62\">追踪变得比端点日志更重要\u003C\u002Fh2>\n\u003Cp>传统的 API 日志可以告诉你一个请求耗时两秒并返回了 HTTP 200。它们无法告诉你选择了哪些检索到的块、代理调用了哪个工具，或者哪个模型跨度消耗了最多的 token。\u003C\u002Fp>\n\u003Cp>MLflow 当前的 GenAI 追踪捕获提示、检索、工具调用和应用跨度，其生产评估流程可以对中间轨迹信息进行评分，而不仅仅是最终文本。\u003C\u002Fp>\n\u003Cp>这是一个重大的 LLMOps 转变：可观测性遵循应用的行为图，而不仅仅是服务端点。\u003C\u002Fp>\n\u003Ch2 id=\"section-66\">代理将 LLMOps 扩展到运行时操作\u003C\u002Fh2>\n\u003Cp>代理应用在产生结果之前可以执行多次模型调用、工具调用和状态转换。\u003C\u002Fp>\n\u003Cp>因此，操作代理除了需要常规的模型延迟和 token 指标外，还需要步数、工具调用追踪、权限拒绝、重试、循环检测、人工批准和验证的最终状态。\u003C\u002Fp>\n\u003Cp>正确的最终答案可能掩盖了糟糕的轨迹，因此代理评估必须检查路径以及结果。\u003C\u002Fp>\n\u003Caside class=\"editorjs-referral my-6\">\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough\" class=\"flex flex-col sm:flex-row gap-4 rounded-xl border border-gray-200 dark:border-gray-700 p-4 transition hover:border-primary-500\">\u003Cdiv class=\"min-w-0 flex-1\">\u003Cstrong class=\"block text-lg text-gray-900 dark:text-gray-100\">AI 代理可靠性：为什么最终答案还不够\u003C\u002Fstrong>\u003Cp class=\"mt-2 text-sm text-gray-600 dark:text-gray-300\">为什么代理生产评估必须包括工具调用、状态转换、批准和可恢复性。\u003C\u002Fp>\u003Cspan class=\"mt-3 inline-flex text-sm font-medium text-primary-600 dark:text-primary-400\">阅读代理可靠性文章 →\u003C\u002Fspan>\u003C\u002Fdiv>\u003C\u002Fa>\u003C\u002Faside>\n\u003Ch2 id=\"section-71\">Token、模型调用和上下文成为成本变量\u003C\u002Fh2>\n\u003Cp>经典 ML 推理成本通常由服务基础设施或每次预测的计算主导。LLM 应用可以增加提供商 token 定价、重复的代理调用、嵌入调用、重排序和工具\u002F运行时开销。\u003C\u002Fp>\n\u003Cp>因此，成本必须归因到任务或追踪，而不仅仅是单个端点。一个工作流如果进行了八次隐藏的模型调用，可能在功能上是正确的，但在运营上不可接受。\u003C\u002Fp>\n\u003Cp>延迟的表现也是如此：模型延迟、检索、重排序和外部工具共同构成了端到端的用户延迟。\u003C\u002Fp>\n\u003Ch2 id=\"section-75\">缓存变得语义化，而不仅仅是技术性\u003C\u002Fh2>\n\u003Cp>LLM系统可以缓存提示、嵌入、检索结果或完整响应，但缓存键必须反映可能改变结果的语义。\u003C\u002Fp>\n\u003Cp>一个忽略模型版本、租户、权限或源数据新鲜度的响应缓存，可能返回技术上有效但语义上无效的答案。\u003C\u002Fp>\n\u003Cp>因此，LLMOps将缓存失效视为模型\u002F上下文\u002F数据版本管理的一部分，而不仅仅是基础设施优化。\u003C\u002Fp>\n\u003Ch2 id=\"section-79\">安全与权限成为发布标准\u003C\u002Fh2>\n\u003Cp>生成式系统可以产生无界文本，智能体可以触发外部操作。因此，安全测试比许多经典预测性ML系统更接近常规CI\u002FCD。\u003C\u002Fp>\n\u003Cp>在存在这些风险的地方，权限检查、提示注入测试、租户隔离测试和副作用审批应成为可复现的回归测试。\u003C\u002Fp>\n\u003Cp>模型可以建议操作，但运行时仍然必须强制执行授权。LLMOps负责提供证据，证明这些控制在模型、提示或工具变更后仍然有效。\u003C\u002Fp>\n\u003Ch2 id=\"section-83\">LLMOps中的CI是什么样的\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">CI层\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">示例检查\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">代码\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">单元测试、类型检查、模式验证\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">提示\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模板渲染、必需变量、策略文本、快照审查\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型\u002F提供商\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">兼容性、输出模式、能力和回归测试\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">RAG\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">分块夹具、过滤器测试、Recall@k、重排序器回归\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">工具\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">输入\u002F输出模式测试、权限测试、幂等性测试\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">智能体\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">轨迹夹具、循环限制、交接\u002F工具选择测试\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">安全\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">提示注入、未授权工具、跨租户负面测试\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">行为评估\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">任务成功、正确性、基础性、安全性、领域标准\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">运营\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">延迟、令牌\u002F成本预算、超时\u002F回退行为\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-85\">LLMOps中的CD是什么样的\u003C\u002Fh2>\n\u003Cp>生产发布可能根本不部署新的模型工件。它可能只是发布新的提示、检索配置、工具集或提供商映射。\u003C\u002Fp>\n\u003Cp>因此，发布包应标识完整的定义行为的配置，而不仅仅是应用容器镜像。\u003C\u002Fp>\n\u003Cp>功能标志、分阶段推出、影子评估、金丝雀流量和回滚非常有用，因为LLM行为可能以静态契约测试无法检测的方式发生回归。\u003C\u002Fp>\n\u003Ch2 id=\"section-89\">持续训练变为可选；持续评估成为核心\u003C\u002Fh2>\n\u003Cp>传统MLOps通常强调在新数据或漂移证明需要重新训练时进行持续训练。\u003C\u002Fp>\n\u003Cp>许多LLM应用从不训练基础模型。它们对应的持续循环是持续评估：收集失败案例和具有代表性的生产案例，将其加入评估数据集，测试候选的提示词\u002F模型\u002F检索变更，并且仅在证据表明有改进时才重新部署。\u003C\u002Fp>\n\u003Cp>微调可以重新引入训练生命周期，但它应当置于同一个更广泛的评估与发布流程之中。\u003C\u002Fp>\n\u003Ch2 id=\"section-93\">生产环境中应监控哪些内容？\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">信号类别\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">示例\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">系统健康\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">错误、超时、端点可用性\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型\u002F提供商\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型ID、快照、速率限制、提供商错误\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">延迟\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">端到端、模型、检索、工具和重排序器跨度\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">成本\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">输入\u002F输出令牌、嵌入、工具\u002FAPI支出\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">质量\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">抽样任务成功率、正确性、相关性、有据性\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">RAG\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">检索召回代理指标、空检索、过时来源、引用覆盖率\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">智能体\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">工具选择、重试、循环、交接、审批频率\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">安全\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">被拒绝的操作、提示注入指标、租户边界失效\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">用户反馈\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">纠正、放弃、升级、显式评分\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">变更漂移\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">提供商\u002F模型\u002F配置相对于已批准发布的变更\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-95\">生产追踪可以成为评估数据\u003C\u002Fh2>\n\u003Cp>现代LLMOps中最有用的模式之一，是将抽样的生产追踪转化为评估记录。\u003C\u002Fp>\n\u003Cp>MLflow目前支持检索生产追踪，并且不仅对输出评分，还可以对中间跨度（如检索或工具调用轨迹）进行评分。\u003C\u002Fp>\n\u003Cp>这闭合了可观测性与开发之间的循环：真实失败可以成为下一次发布中的回归案例，而不是消失在日志中。\u003C\u002Fp>\n\u003Ch2 id=\"section-99\">可复现性变为有条件的，而非精确的\u003C\u002Fh2>\n\u003Cp>经典机器学习的可复现性通常旨在从版本化的代码、数据、环境和训练参数中重建模型。\u003C\u002Fp>\n\u003Cp>托管式LLM应用无法始终逐令牌复现完全相同的输出，因为生成是概率性的，并且提供商可能控制基础设施。\u003C\u002Fp>\n\u003Cp>因此，LLMOps追求的是行为可复现性：记录足够的模型\u002F提供商\u002F版本、提示词、上下文输入、检索状态和运行时配置，以便在预期容差范围内复现条件并验证行为。\u003C\u002Fp>\n\u003Ch2 id=\"section-103\">血缘从模型血缘扩展为应用血缘\u003C\u002Fh2>\n\u003Cp>AWS的MLOps指南将模型血缘视为诊断和可复现性所需的代码、数据、模型和基础设施制品的历史记录。\u003C\u002Fp>\n\u003Cp>对于LLM应用，血缘还应连接提示词、评估数据集、检索\u002F索引版本、工具模式、智能体\u002F运行时配置以及提供商\u002F模型快照。\u003C\u002Fp>\n\u003Cp>目标问题变为：究竟是哪个确切的应用配置产生了这条追踪？\u003C\u002Fp>\n\u003Ch2 id=\"section-107\">多提供商和模型路由产生运营策略\u003C\u002Fh2>\n\u003Cp>一旦应用可以使用多个提供商或本地模型，路由就成为一种运营策略，而不再只是一个简单的模型字符串。\u003C\u002Fp>\n\u003Cp>路由可能取决于能力、延迟、成本、隐私、上下文长度、可用性、工具支持或本地性。回退机制可以在保持正常运行时间的同时，改变答案质量或数据处理假设。\u003C\u002Fp>\n\u003Cp>因此，LLMOps 应记录实际选择了哪条路由，并独立评估各条路由，而不是将每个兼容端点视为行为上可互换的。\u003C\u002Fp>\n\u003Ch2 id=\"section-111\">原始实现证据\u003C\u002Fh2>\n\u003Ch3 id=\"section-112\">Aaasaasa AI Client：提供商、模型和运行时是独立的运维对象\u003C\u002Fh3>\n\u003Cp>Aaasaasa AI Client 将代理\u002F客户端、提供商、模型、运行时位置和权限分开。其 AI Hub 支持 Ollama、LM Studio\u002FOpenAI 兼容端点以及其他提供商协议，而不是将“模型”视为一个全局设置。\u003C\u002Fp>\n\u003Cp>该实现包括动态本地模型发现、流式传输、思考输出以及显式的 Ollama 预热\u002F加载和卸载控制。这是运维证据，表明本地 LLM 服务引入了超出 API 模型名称的资源生命周期问题。\u003C\u002Fp>\n\u003Cp>提供商状态通过提供商适配器查询，连接类型区分本地、云 API、账户支持、远程代理和 Web 客户端路径。这些是 LLM 感知平台必须呈现的具体运维维度。\u003C\u002Fp>\n\u003Cp>该仓库还保留了一个重要边界：本地运行时并不自动等于本地推理。提供商\u002F模型\u002F运行时位置是影响隐私、延迟、成本和可用性的版本化或可配置事项。\u003C\u002Fp>\n\u003Ch3 id=\"section-117\">Source of Truth Research Engine：LLM 应用状态超出模型本身\u003C\u002Fh3>\n\u003Cp>Source of Truth Research Engine 围绕本地模型辅助研究，结合了词法搜索、可选嵌入、来源快照、SHA-256 身份、声明、溯源和矛盾跟踪。\u003C\u002Fp>\n\u003Cp>这是有用的 LLMOps 证据，因为仅更改模型并不能定义研究系统。检索、来源获取、证据分类和持久溯源是独立的运维产物。\u003C\u002Fp>\n\u003Cp>该实现有意将语义相似性视为发现而非证据，说明为什么 LLMOps 可观测性应区分检索行为与声明有效性。\u003C\u002Fp>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">观察到的实现\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">LLMOps 经验\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">多种提供商协议\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">提供商身份是一种运维依赖\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">动态模型发现\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">可用模型可以独立于应用代码发生变化\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Ollama 加载\u002F卸载控制\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">本地模型具有内存\u002F资源生命周期\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">提供商健康\u002F状态适配器\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型可用性需要运行时可观测性\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">运行时与推理位置分离\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">部署拓扑不是一个布尔值的“本地\u002F云”\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">集中权限\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型能力与工具权限必须保持分离\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">词法 + 语义检索管道\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">检索配置是应用行为的一部分\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">来源\u002F溯源持久化\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">运维状态和证据存在于模型权重之外\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Caside class=\"editorjs-callout editorjs-callout--note my-6 rounded-xl border p-5 border-gray-300 bg-gray-50 dark:border-gray-700 dark:bg-gray-900\u002F40\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">证据边界\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">这些项目展示了多提供商\u002F本地模型运维、权限分离、检索基础设施和证据持久化。它们并非作为完整的商业 LLMOps 平台或大规模生产流量的证明来呈现。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-123\">常见 LLMOps 故障模式\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">故障模式\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">实际出了什么问题\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型别名被静默升级\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">行为在未受控发布的情况下发生变化\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">提示词更改但未进行评估\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">行为回归通过了常规单元测试\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">RAG 索引过期\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">生成模型被错误归咎于检索\u002F数据故障\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">仅记录最终答案\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">检索\u002F工具\u002F上下文轨迹中的根本原因不可见\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">提供商回退是静默的\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">不同的模型\u002F数据路径在无归因的情况下改变行为\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Token 成本全局跟踪\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">昂贵的工作流无法被定位\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">评判模型发生变化\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">评估分数在应用未更改的情况下漂移\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">生产跟踪从未转化为测试\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">已知故障反复出现\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">本地模型无限期保持加载\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">VRAM\u002F资源压力成为运维不稳定因素\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">权限仅编码在提示词中\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型行为被误认为授权\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">单一评估分数决定一切\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">不同的质量维度被压缩成一个误导性的数字\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">存在模型注册表但提示词\u002F索引版本不存在\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">应用谱系仍然不完整\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-125\">常见误解\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">误解\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">纠正\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“LLMOps 取代 MLOps。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">LLMOps 将 MLOps 原则扩展到 LLM 特定的应用行为。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“LLMOps 就是提示词工程。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">提示词只是模型、提供商、上下文、检索、工具、评估和运行时中的一种产物。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“托管 API 消除了运维工作。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">它们消除了一些模型服务\u002F训练工作，但增加了提供商生命周期、版本和依赖管理。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“如果 API 稳定，应用就稳定。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型行为和提供商\u002F模型快照可以独立于 API 模式发生变化。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“RAG 只是数据预处理。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">在生产环境中，它有自己的摄取、索引、检索和新鲜度生命周期。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“LLM 输出无法测试。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">它们可以通过确定性、参考、评判和人工标准进行评估。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“LLM 评判者是客观的基准真相。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">它们是基于模型的评估器，同样需要校准和版本控制。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“本地模型消除了 LLMOps。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">本地服务增加了模型文件、VRAM、加载\u002F卸载、运行时健康和升级问题。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“可观测性意味着 Token 计数。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">有用的可观测性会跟踪提示词、检索、工具、模型跨度和结果。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“持续训练是强制性的。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">许多 LLM 应用使用持续评估，而不训练基础模型。\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-127\">实用的LLMOps设计流程\u003C\u002Fh2>\n\u003Csection class=\"editorjs-process my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">运营完整的产生行为的系统\u003C\u002Fh3>\u003Cdiv class=\"grid grid-cols-1 md:grid-cols-2 xl:grid-cols-3 gap-4\">\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">1\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">1. 定义行为单元\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">列出所有能实质性改变输出的组件：模型、提示词、检索、工具、上下文和策略。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">2\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">2. 建立应用谱系\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">对代码、模型\u002F提供商、提示词、评估数据集、检索配置和工具契约进行版本控制。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">3\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">3. 构建有代表性的评估数据集\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">使用来自设计和生产的预期成功\u002F失败案例。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">4\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">4. 分离确定性测试和行为测试\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">将模式\u002F安全断言与语义输出评估区分开。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">5\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">5. 追踪端到端执行\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">对模型、检索、重排序、工具和代理\u002F运行时跨度进行埋点。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">6\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">6. 定义发布门禁\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">设置质量、安全、延迟和成本阈值。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">7\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">7. 固定或明确记录模型版本\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">将模型\u002F提供商变更视为发布事件。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">8\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">8. 渐进式部署\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">在后果需要时使用功能标志、金丝雀发布或分阶段推出。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">9\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">9. 评估生产追踪\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">衡量真实任务行为并识别反复出现的故障。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">10\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">10. 将故障反馈到评估数据集\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">将事件和纠正转化为永久的回归覆盖。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">11\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">11. 监控提供商和数据生命周期\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">跟踪弃用、索引新鲜度、源变更和运行时可用性。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">12\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">12. 干净地退役过时版本\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">在迁移和证据保留决策后，移除旧的提示词\u002F模型\u002F索引\u002F凭证。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-129\">LLMOps架构检查清单\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">问题\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">预期证据\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">哪个模型\u002F提供商\u002F版本服务了该请求？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">可追踪的模型身份\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">哪些提示词\u002F指令处于活动状态？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">版本化的应用代码\u002F配置\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">哪些上下文到达了模型？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">上下文\u002F检索追踪\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">使用了哪个语料库\u002F索引版本？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">检索谱系\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">哪些工具可用并被调用？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">工具模式 + 轨迹追踪\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">应用了哪些权限？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">运行时授权记录\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">如何衡量质量？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">版本化的评估数据集 + 评分器\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">如何测试模型升级？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">行为回归测试套件\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">如何抽样生产质量？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">追踪评估\u002F反馈流程\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">能否近似复现一次故障？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型\u002F上下文\u002F提供商\u002F应用谱系\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">成本花在哪里？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">按追踪的模型\u002F工具\u002F检索归因\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">什么触发回滚？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">定义的质量\u002F安全\u002F成本\u002F可用性阈值\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">如何处理提供商弃用？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">迁移\u002F回退流程\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">如何运营本地模型？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">健康、资源、加载\u002F卸载和版本控制\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-131\">边缘情况和限制\u003C\u002Fh2>\n\u003Cp>一个简单的应用，调用一个固定的托管模型，没有检索或工具，可能只需要轻量级的LLMOps：版本化的提示词代码、评估、模型固定、基本追踪和提供商监控。\u003C\u002Fp>\n\u003Cp>一个自托管的微调模型可能需要几乎完整的经典MLOps堆栈，加上LLM特定的应用评估，使得MLOps和LLMOps之间的界限有意模糊。\u003C\u002Fp>\n\u003Cp>一个代理平台可能只有最少的模型训练操作，但需要大量的运行时操作，因为故障发生在工具选择、状态和编排中。\u003C\u002Fp>\n\u003Cp>一个重度依赖RAG的系统，其操作可能主要由文档摄取和检索质量主导，而不是模型服务。\u003C\u002Fp>\n\u003Cp>术语将继续演变。持久的架构问题不是哪个“Ops”标签胜出，而是哪些工件产生行为，因此必须被版本化、评估、观察和治理。\u003C\u002Fp>\n\u003Ch2 id=\"section-137\">什么会改变这个答案？\u003C\u002Fh2>\n\u003Cp>如果基础模型提供商标准化了完全稳定的模型行为和长期版本支持，提供商\u002F快照管理可能在操作上变得不那么重要。\u003C\u002Fp>\n\u003Cp>如果应用越来越多地拥有微调或训练，经典的MLOps关注点将再次变得更加核心。\u003C\u002Fp>\n\u003Cp>操作原则将保持不变：每个能实质性改变生产行为的组件都属于谱系、测试、可观测性和变更控制。\u003C\u002Fp>\n\u003Ch2 id=\"section-141\">相关规范知识\u003C\u002Fh2>\n\u003Cp>LLMOps位于AI治理和企业AI架构之下：治理定义哪些变更需要证据和批准，而LLMOps提供操作机制来版本化、评估、部署和观察这些变更。\u003C\u002Fp>\n\u003Cp>上下文工程和RAG是许多LLM应用内部的操作子领域，因为上下文和检索可以独立于模型改变行为。\u003C\u002Fp>\n\u003Cp>代理式AI将LLMOps进一步扩展到轨迹、权限和工具运行时操作。\u003C\u002Fp>\n\u003Caside class=\"editorjs-referral my-6\">\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context\" class=\"flex flex-col sm:flex-row gap-4 rounded-xl border border-gray-200 dark:border-gray-700 p-4 transition hover:border-primary-500\">\u003Cdiv class=\"min-w-0 flex-1\">\u003Cstrong class=\"block text-lg text-gray-900 dark:text-gray-100\">AI Agent 记忆不是 RAG：如何分离记忆、检索、状态与上下文\u003C\u002Fstrong>\u003Cp class=\"mt-2 text-sm text-gray-600 dark:text-gray-300\">当记忆、检索、应用状态和模型上下文保持为独立生命周期对象时，运营可靠性会提高。\u003C\u002Fp>\u003Cspan class=\"mt-3 inline-flex text-sm font-medium text-primary-600 dark:text-primary-400\">阅读架构文章 →\u003C\u002Fspan>\u003C\u002Fdiv>\u003C\u002Fa>\u003C\u002Faside>\n\u003Caside class=\"editorjs-referral my-6\">\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers\" class=\"flex flex-col sm:flex-row gap-4 rounded-xl border border-gray-200 dark:border-gray-700 p-4 transition hover:border-primary-500\">\u003Cdiv class=\"min-w-0 flex-1\">\u003Cstrong class=\"block text-lg text-gray-900 dark:text-gray-100\">答案有效性边界：相关性与可靠 AI 答案之间缺失的层\u003C\u002Fstrong>\u003Cp class=\"mt-2 text-sm text-gray-600 dark:text-gray-300\">LLMOps 评估应保留答案仍受支持的版本、范围和证据条件。\u003C\u002Fp>\u003Cspan class=\"mt-3 inline-flex text-sm font-medium text-primary-600 dark:text-primary-400\">阅读答案有效性边界 →\u003C\u002Fspan>\u003C\u002Fdiv>\u003C\u002Fa>\u003C\u002Faside>\n\u003Ch2 id=\"section-147\">常见问题\u003C\u002Fh2>\n\u003Csection class=\"editorjs-faq my-6 rounded-xl border border-gray-200 p-5 dark:border-gray-700\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">MLOps 与 LLMOps 常见问题\u003C\u002Fh3>\u003Cdiv id=\"faq1\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">MLOps 和 LLMOps 有什么区别？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">MLOps 在数据、训练、部署和监控方面运营机器学习系统。LLMOps 将这些实践扩展到 LLM 应用，其中提示、上下文、检索、提供商、工具和评估也会实质性地影响行为。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq2\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">LLMOps 会取代 MLOps 吗？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">不会。LLMOps 复用 MLOps 的学科，如 CI\u002FCD、血缘、评估、部署和监控，并增加 LLM 特有的运营关注点。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq3\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">LLM 应用需要持续训练吗？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">不一定。许多应用使用外部基础模型，转而依赖对提示、模型、检索和应用行为的持续评估。微调或自训练系统仍可能需要训练流水线。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq4\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">为什么评估在 LLMOps 中如此重要？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">生成式输出是开放式的，模型行为可能因提示、快照和上下文而变化。评估提供可重复的证据，证明发布仍满足定义的质量和安全标准。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq5\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">在 LLMOps 中应该对什么进行版本控制？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">至少包括：应用代码、模型\u002F提供商\u002F版本、提示、评估数据集\u002F评分器、检索配置\u002F索引、工具模式、上下文规则以及相关的安全\u002F权限配置。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq6\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">提示版本控制就足够了吗？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">不够。同一个提示在另一个模型、检索集、上下文顺序、工具面或提供商下可能表现不同。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq7\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">什么是 GenAIOps？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">GenAIOps 是运营生成式 AI 应用的另一个行业术语。一些供应商将其与 LLMOps 互换使用，或作为比 LLMOps 更广泛的标签。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq8\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">如何监控 LLM 应用？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">监控端到端追踪，包括模型调用、提示\u002F上下文、检索、工具、延迟、令牌\u002F成本、质量样本、安全性以及最终任务结果。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq9\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">本地 LLM 可以使用 LLMOps 实践吗？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">可以。本地模型增加了自身的运营关注点，如模型文件、硬件\u002FVRAM、加载\u002F卸载、运行时健康、量化和升级管理。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-149\">术语表\u003C\u002Fh2>\n\u003Csection class=\"editorjs-glossary my-6 rounded-xl border border-gray-200 dark:border-gray-700 p-5\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">关键 MLOps 和 LLMOps 术语\u003C\u002Fh3>\u003Cdl>\u003Cdiv id=\"mlops\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">MLOps\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">用于构建、部署、监控和维护机器学习系统及其数据\u002F模型生命周期的工程实践。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"llmops\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">LLMOps\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">针对生产应用的运营实践，其行为实质性地依赖于大型语言模型以及周围的提示、上下文、检索、工具和运行时。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"genaiops\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">GenAIOps\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">生成式 AI 应用的运营学科；通常用作 LLMOps 的更广泛或替代标签。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"continuous-training\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">持续训练\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">随着数据或实现变化，对 ML 模型进行自动化或重复的再训练和 serving。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"continuous-evaluation\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">持续评估\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">针对版本化数据集和标准，对候选和生产 AI 行为进行重复评估。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"model-snapshot\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">模型快照\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">托管或打包模型的具体版本，其行为可以被测试和引用。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"application-lineage\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">应用血缘\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">代码、模型\u002F提供商、提示、数据\u002F检索、工具、运行时和发布配置之间的可追溯关系。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"trace\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">追踪\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">一次应用执行的结构化记录，包含模型调用、检索和工具操作等跨度。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"eval-dataset\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">评估数据集\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">版本化的代表性输入、期望值以及可选的追踪\u002F输出集合，用于衡量行为。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"llm-judge\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">LLM 评判器\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">用作定性或语义标准评估器的语言模型；它本身就是一个版本化的评估依赖项。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"behavioral-regression\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">行为回归\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">尽管接口和代码继续成功执行，应用输出或轨迹仍出现退化。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"provider-routing\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">提供商路由\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">根据能力、成本、延迟、隐私或可用性在可用模型提供商\u002F端点之间进行选择的策略。\u003C\u002Fdd>\u003C\u002Fdiv>\u003C\u002Fdl>\u003C\u002Fsection>\n\u003Ch2 id=\"section-151\">结论\u003C\u002Fh2>\n\u003Cp>MLOps 和 LLMOps 共享相同的工程目标：使 AI 系统足够可复现、足够可测试、足够可观测，以便在生产中可靠运行。\u003C\u002Fp>\n\u003Cp>区别在于系统的形态。经典 MLOps 通常以训练和 serving 模型工件为中心；LLMOps 必须运营一个行为栈，其中模型快照、提示、上下文、检索、工具、权限和提供商可以独立变化。\u003C\u002Fp>\n\u003Cp>最简短有用的规则是：对一切可能实质性改变 LLM 应用行为的内容进行版本控制、评估和观测——而不仅仅是模型。\u003C\u002Fp>\n\u003Ch2 id=\"section-155\">主要来源和当前文档\u003C\u002Fh2>\n\u003Cp>以下来源为 MLOps 基线和 LLM 及代理应用的当前运营模式提供依据。项目部分是原始实现证据，并且有意比关于完整 LLMOps 平台的声明更窄。\u003C\u002Fp>\n\u003Ca href=\"https:\u002F\u002Fdocs.cloud.google.com\u002Farchitecture\u002Fmlops-continuous-delivery-and-automation-pipelines-in-machine-learning\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">Google Cloud — MLOps：持续交付和自动化流水线\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">描述 ML 系统的 CI、CD、持续训练、模型注册表、元数据、serving 和监控的参考架构。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fdocs.aws.amazon.com\u002Fwellarchitected\u002Flatest\u002Fmachine-learning-lens\u002Fmlops02-bp04.html\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">AWS Machine Learning Lens — 模型血缘\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">跨 ML 发布跟踪代码、数据、模型、环境和基础设施的当前指南。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fdocs.aws.amazon.com\u002Fwellarchitected\u002Flatest\u002Fmachine-learning-lens\u002Fmlops06-bp02.html\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">AWS Machine Learning Lens — 模型可观测性和跟踪\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">生产模型监控、漂移、端点健康和血缘的当前指南。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Flearn.microsoft.com\u002Fen-us\u002Fazure\u002Fmachine-learning\u002Fprompt-flow\u002Fhow-to-end-to-end-llmops-with-prompt-flow\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">Microsoft Azure — GenAIOps \u002F LLMOps 生命周期\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">官方指南，描述 GenAIOps（有时称为 LLMOps）在初始化、实验、评估\u002F改进和部署方面的内容。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fmlflow.org\u002Fdocs\u002Flatest\u002Fgenai\u002F\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">MLflow — 代理和 LLM 应用\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">当前 GenAI 运营文档，涵盖 LLM 应用和代理的追踪、评估、提示和生产可观测性。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fwww.mlflow.org\u002Fdocs\u002Flatest\u002Fgenai\u002Feval-monitor\u002Frunning-evaluation\u002Ftraces\u002F\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">MLflow — 评估生产追踪\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">评估完整 LLM\u002F代理追踪（包括检索和工具调用轨迹）的当前指南。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fmlflow.org\u002Fdocs\u002Flatest\u002Fgenai\u002Fprompt-registry\u002Fevaluate-prompts\u002F\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">MLflow — 评估提示词\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">当前使用版本化提示词、数据集、评分器和追踪的提示词\u002F模型评估工作流。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Freference\u002Foverview\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">OpenAI API — 版本管理与模型快照\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">当前 API 指南建议固定模型版本并进行评估，因为提示行为可能在不同快照之间发生变化。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fprompting\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">OpenAI — 提示词工程\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">当前指南建议将生产提示词视为应用程序代码，通过源代码控制进行版本管理，并用测试和评估检查覆盖变更。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fdeprecations\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">OpenAI — 弃用\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">当前提供商生命周期证据表明，模型和平台界面的退役是一种运营依赖。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fdevelopers.openai.com\u002Fcookbook\u002Fexamples\u002Fevaluation\u002Fmoving-from-openai-evals-to-promptfoo\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">OpenAI — 将评估工作流迁移到 Promptfoo\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">当前 2026 年迁移指南说明，随着提供商工具的变化，评估资产应保持可移植性。\u003C\u002Fp>\u003C\u002Fa>",{"time":212,"blocks":213,"version":1557},1791487765513,[214,220,228,235,242,248,256,261,266,271,276,281,286,291,296,301,306,311,316,348,353,358,363,368,373,421,426,431,436,441,446,496,501,506,511,516,521,526,531,536,541,546,551,556,561,566,571,576,581,586,591,596,605,610,615,620,625,632,637,642,647,652,657,662,667,672,677,682,687,692,700,705,710,715,720,725,730,735,740,745,750,755,760,765,800,805,810,815,820,825,830,835,840,845,879,884,889,894,899,904,909,914,919,924,929,934,939,944,949,954,959,964,969,974,979,984,989,994,999,1004,1009,1041,1047,1052,1096,1101,1139,1144,1186,1191,1241,1246,1251,1256,1261,1266,1271,1276,1281,1286,1291,1296,1301,1306,1311,1319,1327,1332,1374,1379,1427,1432,1437,1442,1447,1452,1457,1467,1476,1485,1494,1503,1512,1521,1530,1539,1548],{"id":215,"data":216,"type":218,"tunes":219},"intro",{"text":217},"MLOps 是用于可靠地开发、部署、版本管理和运维机器学习系统的工程学科；LLMOps 将这一学科扩展到围绕大语言模型构建的应用，在这类应用中，生产行为不仅取决于模型产物，还取决于提示词、上下文、检索、提供商\u002F模型版本、工具调用、安全控制和评估流水线。LLMOps 并不取代 MLOps。它将运维单元从“一个模型加服务流水线”转变为“一个不断演进的 LLM 应用，其行为由多个独立变化的组件共同涌现”。","paragraph",{},{"id":221,"data":222,"type":226,"tunes":227},"direct",{"body":223,"title":224,"variant":225},"\u003Cstrong>MLOps 运维 ML 系统。LLMOps 运维 LLM 应用。\u003C\u002Fstrong>\u003Cbr>\u003Cbr>经典 MLOps 通常以数据流水线、训练、验证、模型注册表、部署、漂移和再训练为中心。LLMOps 在相关之处保留这些学科，但通常会增加提示词\u002F上下文版本管理、模型\u002F提供商抽象、RAG 索引、智能体\u002F工具追踪、语义评估、安全测试、token\u002F成本监控，以及针对快速变化的模型快照的回归测试。","直接回答","info","callout",{},{"id":229,"data":230,"type":226,"tunes":234},"boundary",{"body":231,"title":232,"variant":233},"即使提示词没有变化，生产环境中的 LLM 应用也可能失败：提供商可能更改模型快照，RAG 语料库可能过时，重排序器可能退化，工具权限可能变化，上下文组装可能丢失证据，或者智能体可能采取错误轨迹。因此，LLMOps 必须观察并版本化模型周围的系统，而不仅仅是提示词文本。","LLMOps 不只是提示词管理","warning",{},{"id":236,"data":237,"type":226,"tunes":241},"term-note",{"body":238,"title":239,"variant":240},"\u003Cstrong>LLMOps\u003C\u002Fstrong>、\u003Cstrong>GenAIOps\u003C\u002Fstrong> 及相关术语是被广泛使用的工程标签，但它们并不是一个具有单一规范生命周期的通用正式标准。Microsoft 目前将 GenAIOps 描述为“有时称为 LLMOps”，而 MLflow 则将围绕智能体和 LLM 应用的运维工具归为一类。本文使用 LLMOps 作为一个实用架构术语，指代运维那些行为实质上依赖 LLM 的生产系统。","术语边界","note",{},{"id":243,"data":244,"type":226,"tunes":247},"current",{"body":245,"title":246,"variant":240},"运维界面正在快速变化。OpenAI 目前建议固定模型快照并运行评估，因为提示行为可能在不同快照之间发生变化，而且若干较旧的平台特定提示词\u002F评估界面将在 2026 年退役。稳定的架构经验是：让提示词、测试和评估与应用程序一起保持可移植并版本化，而不是依赖某个提供商的仪表盘对象模型。","当前来源说明 — 2026年10月8日",{},{"id":249,"data":250,"type":254,"tunes":255},"toc",{"title":251,"maxLevel":252,"minLevel":253},"目录",3,2,"tableOfContents",{},{"id":257,"data":258,"type":42,"tunes":260},"h-meaning",{"text":259,"level":253},"MLOps 的真正含义",{},{"id":262,"data":263,"type":218,"tunes":265},"p-mlops-1",{"text":264},"MLOps 将软件工程和运维纪律应用于机器学习系统。生产挑战不仅仅是训练模型：数据收集、数据验证、实验、可复现性、模型评估、部署、基础设施和监控都必须协同工作。",{},{"id":267,"data":268,"type":218,"tunes":270},"p-mlops-2",{"text":269},"Google 的 MLOps 架构指南围绕持续集成、持续交付和持续训练来构建这一学科。CI 不仅验证代码，还验证数据、模式和模型；CD 部署 ML 流水线和预测服务；CT 可以在数据或实现变化时重新训练并重新部署模型。",{},{"id":272,"data":273,"type":218,"tunes":275},"p-mlops-3",{"text":274},"AWS 指南从另一个角度补充了相同的运维关注点：模型血缘、模型\u002F版本可追溯性、漂移监控和生产质量监控是部署后保持 ML 系统可靠性的核心部分。",{},{"id":277,"data":278,"type":42,"tunes":280},"h-llmops",{"text":279,"level":253},"当模型是 LLM 时会发生什么变化",{},{"id":282,"data":283,"type":218,"tunes":285},"p-llmops-1",{"text":284},"大语言模型改变了生产问题，因为应用通常并不拥有完整的模型训练生命周期。团队可能调用托管模型 API、在本地运行开源模型、在提供商之间切换，或为不同任务使用多个模型。",{},{"id":287,"data":288,"type":218,"tunes":290},"p-llmops-2",{"text":289},"因此，模型只是更大的行为系统中一个带版本的依赖项。提示词、检索结果、上下文顺序、工具、模型快照、温度\u002F推理设置、安全过滤器和运行时编排都可能改变输出。",{},{"id":292,"data":293,"type":218,"tunes":295},"p-llmops-3",{"text":294},"这带来了一个更广泛的运维问题：是模型、上下文、数据、提示词、工具和运行时的哪种组合产生了这种行为？LLMOps 的存在就是为了让这个问题可回答，并让答案足够可复现，以支持工程工作。",{},{"id":297,"data":298,"type":42,"tunes":300},"h-simple",{"text":299,"level":253},"最简单的例子",{},{"id":302,"data":303,"type":218,"tunes":305},"p-simple-1",{"text":304},"假设一个应用回答内部政策问题。",{},{"id":307,"data":308,"type":218,"tunes":310},"p-simple-2",{"text":309},"在经典 ML 框架中，你可能会对一个训练好的分类器进行版本管理、部署它并监控预测质量。在 LLM 应用中，答案可能取决于托管模型快照、系统提示词、嵌入模型、向量索引、检索过滤器、重排序器以及最终选定的上下文。",{},{"id":312,"data":313,"type":218,"tunes":315},"p-simple-3",{"text":314},"即使应用端点和用户问题保持不变，更改其中任何一个组件都可能改变最终答案。",{},{"id":317,"data":318,"type":346,"tunes":347},"simple-flow",{"steps":319,"title":344,"orientation":345},[320,323,326,329,332,335,338,341],{"label":321,"description":322},"1. 更改一个组件","提示词、模型、提供商、检索设置、工具模式或应用代码发生变化。",{"label":324,"description":325},"2. 运行确定性测试","验证模式、权限、工具契约、检索过滤器和应用行为。",{"label":327,"description":328},"3. 运行行为评估","根据验收标准比较代表性输出、检索质量以及智能体\u002F工具轨迹。",{"label":330,"description":331},"4. 比较成本和延迟","测量令牌使用量、模型调用次数、检索\u002F工具开销和响应延迟。",{"label":333,"description":334},"5. 部署受控版本","发布具体的应用配置，并记录模型\u002F提供商版本。",{"label":336,"description":337},"6. 追踪生产行为","捕获相关的模型、检索、工具和运行时跨度。",{"label":339,"description":340},"7. 评估生产追踪","对真实执行进行采样，以评估质量、依据性、安全性和任务成功率。",{"label":342,"description":343},"8. 回滚或迭代","利用回归证据和运营信号来决定下一次发布。","典型的 LLMOps 发布路径","auto","processFlow",{},{"id":349,"data":350,"type":42,"tunes":352},"h-stops",{"text":351,"level":253},"简单示例止步之处",{},{"id":354,"data":355,"type":218,"tunes":357},"p-stops-1",{"text":356},"一些 LLM 系统仍然训练或微调自己的模型，因此训练流水线、模型注册表和数据血缘等传统 MLOps 实践仍然直接相关。",{},{"id":359,"data":360,"type":218,"tunes":362},"p-stops-2",{"text":361},"其他系统仅使用外部基础模型 API，从不运行持续训练。它们的主要运营工作是应用评估、模型\u002F提供商变更管理、提示词\u002F上下文版本控制、检索质量和可观测性。",{},{"id":364,"data":365,"type":218,"tunes":367},"p-stops-3",{"text":366},"因此，不存在单一的通用“LLMOps 流水线”。确切的生命周期取决于你是训练、微调、自托管、检索外部知识、运行智能体，还是依赖托管模型 API。",{},{"id":369,"data":370,"type":42,"tunes":372},"h-compare",{"text":371,"level":253},"MLOps 与 LLMOps",{},{"id":374,"data":375,"type":419,"tunes":420},"main-comparison",{"rows":376,"title":410,"layout":411,"columns":412},[377,382,386,390,394,398,402,406],{"id":378,"label":379,"values":380},"unit","主要运营单元",[381,381],"",{"id":383,"label":384,"values":385},"model","模型所有权",[381,381],{"id":387,"label":388,"values":389},"change","典型变更",[381,381],{"id":391,"label":392,"values":393},"eval","评估",[381,381],{"id":395,"label":396,"values":397},"monitor","生产监控",[381,381],{"id":399,"label":400,"values":401},"training","持续训练",[381,381],{"id":403,"label":404,"values":405},"registry","版本化产物",[381,381],{"id":407,"label":408,"values":409},"rollback","回滚目标",[381,381],"哪些保持不变，哪些扩展","table",[413,416],{"id":414,"label":415},"mlops","MLOps",{"id":417,"label":418},"llmops","LLMOps","comparison",{},{"id":422,"data":423,"type":42,"tunes":425},"h-extension",{"text":424,"level":253},"LLMOps 扩展 MLOps 而非取代它",{},{"id":427,"data":428,"type":218,"tunes":430},"p-extension-1",{"text":429},"核心运营原则并未消失：源代码控制、CI\u002FCD、可复现性、血缘、部署控制、监控、回滚和可衡量的验收标准仍然至关重要。",{},{"id":432,"data":433,"type":218,"tunes":435},"p-extension-2",{"text":434},"扩展之处在于，更多定义行为的产物现在位于模型权重之外。托管的基础模型可以通过快照升级改变行为，而应用输出可以通过提示词或检索更改而无需任何模型重新训练即可改变。",{},{"id":437,"data":438,"type":218,"tunes":440},"p-extension-3",{"text":439},"这就是为什么有用的层次结构通常是 DevOps → MLOps → LLMOps\u002FGenAIOps，作为日益专业化的运营关注点，而不是三种相互排斥的实践。",{},{"id":442,"data":443,"type":42,"tunes":445},"h-artifacts",{"text":444,"level":253},"LLMOps 中必须版本化哪些内容？",{},{"id":447,"data":448,"type":411,"tunes":495},"artifact-table",{"content":449,"stretched":43,"withHeadings":14},[450,453,456,459,462,465,468,471,474,477,480,483,486,489,492],[451,452],"产物","为何重要",[454,455],"应用代码","定义编排、验证、重试和业务行为",[457,458],"模型家族 + 快照\u002F版本","不同的快照可能产生不同的行为",[460,461],"提供商 \u002F 端点","改变数据流、延迟、限制、定价和可用性",[463,464],"提示词\u002F指令代码","即使模型相同，也会改变模型行为",[466,467],"生成\u002F推理参数","可能改变确定性、延迟、深度和成本",[469,470],"评估数据集","定义“足够好”的测试基准",[472,473],"评分器 \u002F 打分器","定义如何衡量质量",[475,476],"嵌入模型","改变向量表示和检索行为",[478,479],"分块\u002F索引配置","改变可检索的内容",[481,482],"重排序器 \u002F 检索融合","改变结果排序",[484,485],"工具模式","改变模型可以请求什么以及如何请求",[487,488],"权限配置文件","改变哪些工具操作可以实际执行",[490,491],"上下文组装规则","改变哪些证据和状态到达模型",[493,494],"安全\u002F护栏配置","改变允许或阻止的行为",{},{"id":497,"data":498,"type":42,"tunes":500},"h-model-version",{"text":499,"level":253},"模型快照成为发布依赖项",{},{"id":502,"data":503,"type":218,"tunes":505},"p-model-version-1",{"text":504},"对于托管的 LLM，团队可能无法控制模型训练，但仍然控制应用程序调用哪个模型或快照。",{},{"id":507,"data":508,"type":218,"tunes":510},"p-model-version-2",{"text":509},"OpenAI 当前的 API 指南明确警告，提示行为可能在不同模型快照之间发生变化，并建议在一致性重要的生产应用程序中固定到特定快照，然后在升级时运行评估。",{},{"id":512,"data":513,"type":218,"tunes":515},"p-model-version-3",{"text":514},"运营后果很直接：模型升级应被视为应用程序发布，而不是不可见的基础设施维护。",{},{"id":517,"data":518,"type":42,"tunes":520},"h-provider",{"text":519,"level":253},"提供商生命周期成为运营的一部分",{},{"id":522,"data":523,"type":218,"tunes":525},"p-provider-1",{"text":524},"LLM 应用通常依赖于提供商的速率限制、弃用时间表、API 语义、上下文限制、数据处理规则和定价。",{},{"id":527,"data":528,"type":218,"tunes":530},"p-provider-2",{"text":529},"即使你的应用程序代码保持不变，提供商也可能弃用某个模型。例如，OpenAI 当前的弃用时间表包括针对旧模型快照和平台界面的 2026 年退役日期。",{},{"id":532,"data":533,"type":218,"tunes":535},"p-provider-3",{"text":534},"因此，LLMOps 除了需要模型质量监控外，还需要提供商生命周期跟踪、迁移测试和回退决策。",{},{"id":537,"data":538,"type":42,"tunes":540},"h-prompt",{"text":539,"level":253},"提示词的行为类似于生产代码",{},{"id":542,"data":543,"type":218,"tunes":545},"p-prompt-1",{"text":544},"提示词是可执行的行为配置。微小的更改可能会改变输出质量、工具选择和策略解释。",{},{"id":547,"data":548,"type":218,"tunes":550},"p-prompt-2",{"text":549},"OpenAI 当前的指南建议将生产提示词存储在应用程序代码中，通过拉取请求审查提示词更改，使用类型化输入，并通过测试和评估检查覆盖更改。",{},{"id":552,"data":553,"type":218,"tunes":555},"p-prompt-3",{"text":554},"这使得提示词版本管理不再像编辑营销文案，而更像更改一个输出具有概率性且依赖于模型的函数。",{},{"id":557,"data":558,"type":42,"tunes":560},"h-context",{"text":559,"level":253},"上下文工程成为一项运营关注点",{},{"id":562,"data":563,"type":218,"tunes":565},"p-context-1",{"text":564},"生产模型很少只接收静态提示词。它可能接收对话历史、检索到的文档、工具输出、记忆、当前应用程序状态和策略指令。",{},{"id":567,"data":568,"type":218,"tunes":570},"p-context-2",{"text":569},"因此，LLMOps 必须观察上下文组装：选择了哪些证据、当前是哪个状态版本、是否发生了截断，以及重要指令是否在压缩后保留下来。",{},{"id":572,"data":573,"type":218,"tunes":575},"p-context-3",{"text":574},"模型回归和上下文回归在最终答案上可能看起来完全相同。追踪实际的上下文路径才能让团队区分它们。",{},{"id":577,"data":578,"type":42,"tunes":580},"h-rag",{"text":579,"level":253},"RAG 创建了自己的运营生命周期",{},{"id":582,"data":583,"type":218,"tunes":585},"p-rag-1",{"text":584},"RAG 系统在模型推理之外引入了第二条生产流水线：摄取、提取、分块、元数据、嵌入、索引、检索、重排序和上下文选择。",{},{"id":587,"data":588,"type":218,"tunes":590},"p-rag-2",{"text":589},"即使模型和提示词没有变化，知识语料库也可能每天发生变化。因此，过时的索引或损坏的元数据过滤器可能会在没有任何模型漂移的情况下降低答案质量。",{},{"id":592,"data":593,"type":218,"tunes":595},"p-rag-3",{"text":594},"针对 RAG 的 LLMOps 应独立于生成质量，分别跟踪语料库\u002F索引版本、嵌入模型、分块策略、检索配置、源新鲜度和检索指标。",{},{"id":597,"data":598,"type":603,"tunes":604},"ref-rag-diagnostic",{"url":599,"title":600,"excerpt":601,"ctaLabel":602},"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method","RAG 失败了——但究竟是哪一层失败了？一种诊断方法","生产 LLM 流水线需要对源覆盖、检索、排序、上下文组装和生成进行单独的可观测性。","阅读 RAG 诊断方法","referralArticle",{},{"id":606,"data":607,"type":42,"tunes":609},"h-evals",{"text":608,"level":253},"评估用发布证据取代“在我看来不错”",{},{"id":611,"data":612,"type":218,"tunes":614},"p-eval-1",{"text":613},"生成式输出通常是开放式的，因此精确匹配测试对许多任务来说是不够的。LLMOps 增加了评估数据集和评分器，可以衡量任务成功、正确性、安全性、有据性、风格或特定领域的验收标准。",{},{"id":616,"data":617,"type":218,"tunes":619},"p-eval-2",{"text":618},"MLflow 当前的 GenAI 评估栈支持版本化评估数据集、提示\u002F模型比较、自定义评分器以及对完整追踪的评估。",{},{"id":621,"data":622,"type":218,"tunes":624},"p-eval-3",{"text":623},"最强的实践是评估驱动开发：在变更之前或同时定义代表性案例和验收标准，然后用相同的证据比较各个版本。",{},{"id":626,"data":627,"type":226,"tunes":631},"eval-rule",{"body":628,"title":629,"variant":630},"仅仅因为 API 契约仍然有效，不应认为部署是等价的。如果提示、模型、检索或工具发生了变化，行为回归测试套件应重新运行。","行为变更需要行为测试","success",{},{"id":633,"data":634,"type":42,"tunes":636},"h-judges",{"text":635,"level":253},"LLM 作为评判者有用但并非绝对真理",{},{"id":638,"data":639,"type":218,"tunes":641},"p-judge-1",{"text":640},"LLM 评判者可以扩展对那些难以编码为确定性断言的品质的评估，例如相关性、语气或事实依据。",{},{"id":643,"data":644,"type":218,"tunes":646},"p-judge-2",{"text":645},"然而，评判者是另一个具有自身偏见、版本和提示的模型。因此，在后果重要的地方，评判者配置应进行版本管理，并针对人类或确定性参考案例进行校准。",{},{"id":648,"data":649,"type":218,"tunes":651},"p-judge-3",{"text":650},"生产评估可以混合使用确定性检查、基于参考的指标、模型评判者和人工审查，而不是要求单一指标代表所有质量维度。",{},{"id":653,"data":654,"type":42,"tunes":656},"h-tracing",{"text":655,"level":253},"追踪变得比端点日志更重要",{},{"id":658,"data":659,"type":218,"tunes":661},"p-trace-1",{"text":660},"传统的 API 日志可以告诉你一个请求耗时两秒并返回了 HTTP 200。它们无法告诉你选择了哪些检索到的块、代理调用了哪个工具，或者哪个模型跨度消耗了最多的 token。",{},{"id":663,"data":664,"type":218,"tunes":666},"p-trace-2",{"text":665},"MLflow 当前的 GenAI 追踪捕获提示、检索、工具调用和应用跨度，其生产评估流程可以对中间轨迹信息进行评分，而不仅仅是最终文本。",{},{"id":668,"data":669,"type":218,"tunes":671},"p-trace-3",{"text":670},"这是一个重大的 LLMOps 转变：可观测性遵循应用的行为图，而不仅仅是服务端点。",{},{"id":673,"data":674,"type":42,"tunes":676},"h-agent",{"text":675,"level":253},"代理将 LLMOps 扩展到运行时操作",{},{"id":678,"data":679,"type":218,"tunes":681},"p-agent-1",{"text":680},"代理应用在产生结果之前可以执行多次模型调用、工具调用和状态转换。",{},{"id":683,"data":684,"type":218,"tunes":686},"p-agent-2",{"text":685},"因此，操作代理除了需要常规的模型延迟和 token 指标外，还需要步数、工具调用追踪、权限拒绝、重试、循环检测、人工批准和验证的最终状态。",{},{"id":688,"data":689,"type":218,"tunes":691},"p-agent-3",{"text":690},"正确的最终答案可能掩盖了糟糕的轨迹，因此代理评估必须检查路径以及结果。",{},{"id":693,"data":694,"type":603,"tunes":699},"ref-agent-reliability",{"url":695,"title":696,"excerpt":697,"ctaLabel":698},"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough","AI 代理可靠性：为什么最终答案还不够","为什么代理生产评估必须包括工具调用、状态转换、批准和可恢复性。","阅读代理可靠性文章",{},{"id":701,"data":702,"type":42,"tunes":704},"h-cost",{"text":703,"level":253},"Token、模型调用和上下文成为成本变量",{},{"id":706,"data":707,"type":218,"tunes":709},"p-cost-1",{"text":708},"经典 ML 推理成本通常由服务基础设施或每次预测的计算主导。LLM 应用可以增加提供商 token 定价、重复的代理调用、嵌入调用、重排序和工具\u002F运行时开销。",{},{"id":711,"data":712,"type":218,"tunes":714},"p-cost-2",{"text":713},"因此，成本必须归因到任务或追踪，而不仅仅是单个端点。一个工作流如果进行了八次隐藏的模型调用，可能在功能上是正确的，但在运营上不可接受。",{},{"id":716,"data":717,"type":218,"tunes":719},"p-cost-3",{"text":718},"延迟的表现也是如此：模型延迟、检索、重排序和外部工具共同构成了端到端的用户延迟。",{},{"id":721,"data":722,"type":42,"tunes":724},"h-cache",{"text":723,"level":253},"缓存变得语义化，而不仅仅是技术性",{},{"id":726,"data":727,"type":218,"tunes":729},"p-cache-1",{"text":728},"LLM系统可以缓存提示、嵌入、检索结果或完整响应，但缓存键必须反映可能改变结果的语义。",{},{"id":731,"data":732,"type":218,"tunes":734},"p-cache-2",{"text":733},"一个忽略模型版本、租户、权限或源数据新鲜度的响应缓存，可能返回技术上有效但语义上无效的答案。",{},{"id":736,"data":737,"type":218,"tunes":739},"p-cache-3",{"text":738},"因此，LLMOps将缓存失效视为模型\u002F上下文\u002F数据版本管理的一部分，而不仅仅是基础设施优化。",{},{"id":741,"data":742,"type":42,"tunes":744},"h-safety",{"text":743,"level":253},"安全与权限成为发布标准",{},{"id":746,"data":747,"type":218,"tunes":749},"p-safety-1",{"text":748},"生成式系统可以产生无界文本，智能体可以触发外部操作。因此，安全测试比许多经典预测性ML系统更接近常规CI\u002FCD。",{},{"id":751,"data":752,"type":218,"tunes":754},"p-safety-2",{"text":753},"在存在这些风险的地方，权限检查、提示注入测试、租户隔离测试和副作用审批应成为可复现的回归测试。",{},{"id":756,"data":757,"type":218,"tunes":759},"p-safety-3",{"text":758},"模型可以建议操作，但运行时仍然必须强制执行授权。LLMOps负责提供证据，证明这些控制在模型、提示或工具变更后仍然有效。",{},{"id":761,"data":762,"type":42,"tunes":764},"h-ci",{"text":763,"level":253},"LLMOps中的CI是什么样的",{},{"id":766,"data":767,"type":411,"tunes":799},"ci-table",{"content":768,"stretched":43,"withHeadings":14},[769,772,775,778,781,784,787,790,793,796],[770,771],"CI层","示例检查",[773,774],"代码","单元测试、类型检查、模式验证",[776,777],"提示","模板渲染、必需变量、策略文本、快照审查",[779,780],"模型\u002F提供商","兼容性、输出模式、能力和回归测试",[782,783],"RAG","分块夹具、过滤器测试、Recall@k、重排序器回归",[785,786],"工具","输入\u002F输出模式测试、权限测试、幂等性测试",[788,789],"智能体","轨迹夹具、循环限制、交接\u002F工具选择测试",[791,792],"安全","提示注入、未授权工具、跨租户负面测试",[794,795],"行为评估","任务成功、正确性、基础性、安全性、领域标准",[797,798],"运营","延迟、令牌\u002F成本预算、超时\u002F回退行为",{},{"id":801,"data":802,"type":42,"tunes":804},"h-cd",{"text":803,"level":253},"LLMOps中的CD是什么样的",{},{"id":806,"data":807,"type":218,"tunes":809},"p-cd-1",{"text":808},"生产发布可能根本不部署新的模型工件。它可能只是发布新的提示、检索配置、工具集或提供商映射。",{},{"id":811,"data":812,"type":218,"tunes":814},"p-cd-2",{"text":813},"因此，发布包应标识完整的定义行为的配置，而不仅仅是应用容器镜像。",{},{"id":816,"data":817,"type":218,"tunes":819},"p-cd-3",{"text":818},"功能标志、分阶段推出、影子评估、金丝雀流量和回滚非常有用，因为LLM行为可能以静态契约测试无法检测的方式发生回归。",{},{"id":821,"data":822,"type":42,"tunes":824},"h-ct",{"text":823,"level":253},"持续训练变为可选；持续评估成为核心",{},{"id":826,"data":827,"type":218,"tunes":829},"p-ct-1",{"text":828},"传统MLOps通常强调在新数据或漂移证明需要重新训练时进行持续训练。",{},{"id":831,"data":832,"type":218,"tunes":834},"p-ct-2",{"text":833},"许多LLM应用从不训练基础模型。它们对应的持续循环是持续评估：收集失败案例和具有代表性的生产案例，将其加入评估数据集，测试候选的提示词\u002F模型\u002F检索变更，并且仅在证据表明有改进时才重新部署。",{},{"id":836,"data":837,"type":218,"tunes":839},"p-ct-3",{"text":838},"微调可以重新引入训练生命周期，但它应当置于同一个更广泛的评估与发布流程之中。",{},{"id":841,"data":842,"type":42,"tunes":844},"h-monitor",{"text":843,"level":253},"生产环境中应监控哪些内容？",{},{"id":846,"data":847,"type":411,"tunes":878},"monitor-table",{"content":848,"stretched":43,"withHeadings":14},[849,852,855,857,860,863,866,868,870,872,875],[850,851],"信号类别","示例",[853,854],"系统健康","错误、超时、端点可用性",[779,856],"模型ID、快照、速率限制、提供商错误",[858,859],"延迟","端到端、模型、检索、工具和重排序器跨度",[861,862],"成本","输入\u002F输出令牌、嵌入、工具\u002FAPI支出",[864,865],"质量","抽样任务成功率、正确性、相关性、有据性",[782,867],"检索召回代理指标、空检索、过时来源、引用覆盖率",[788,869],"工具选择、重试、循环、交接、审批频率",[791,871],"被拒绝的操作、提示注入指标、租户边界失效",[873,874],"用户反馈","纠正、放弃、升级、显式评分",[876,877],"变更漂移","提供商\u002F模型\u002F配置相对于已批准发布的变更",{},{"id":880,"data":881,"type":42,"tunes":883},"h-prod-eval",{"text":882,"level":253},"生产追踪可以成为评估数据",{},{"id":885,"data":886,"type":218,"tunes":888},"p-prod-1",{"text":887},"现代LLMOps中最有用的模式之一，是将抽样的生产追踪转化为评估记录。",{},{"id":890,"data":891,"type":218,"tunes":893},"p-prod-2",{"text":892},"MLflow目前支持检索生产追踪，并且不仅对输出评分，还可以对中间跨度（如检索或工具调用轨迹）进行评分。",{},{"id":895,"data":896,"type":218,"tunes":898},"p-prod-3",{"text":897},"这闭合了可观测性与开发之间的循环：真实失败可以成为下一次发布中的回归案例，而不是消失在日志中。",{},{"id":900,"data":901,"type":42,"tunes":903},"h-repro",{"text":902,"level":253},"可复现性变为有条件的，而非精确的",{},{"id":905,"data":906,"type":218,"tunes":908},"p-repro-1",{"text":907},"经典机器学习的可复现性通常旨在从版本化的代码、数据、环境和训练参数中重建模型。",{},{"id":910,"data":911,"type":218,"tunes":913},"p-repro-2",{"text":912},"托管式LLM应用无法始终逐令牌复现完全相同的输出，因为生成是概率性的，并且提供商可能控制基础设施。",{},{"id":915,"data":916,"type":218,"tunes":918},"p-repro-3",{"text":917},"因此，LLMOps追求的是行为可复现性：记录足够的模型\u002F提供商\u002F版本、提示词、上下文输入、检索状态和运行时配置，以便在预期容差范围内复现条件并验证行为。",{},{"id":920,"data":921,"type":42,"tunes":923},"h-lineage",{"text":922,"level":253},"血缘从模型血缘扩展为应用血缘",{},{"id":925,"data":926,"type":218,"tunes":928},"p-lineage-1",{"text":927},"AWS的MLOps指南将模型血缘视为诊断和可复现性所需的代码、数据、模型和基础设施制品的历史记录。",{},{"id":930,"data":931,"type":218,"tunes":933},"p-lineage-2",{"text":932},"对于LLM应用，血缘还应连接提示词、评估数据集、检索\u002F索引版本、工具模式、智能体\u002F运行时配置以及提供商\u002F模型快照。",{},{"id":935,"data":936,"type":218,"tunes":938},"p-lineage-3",{"text":937},"目标问题变为：究竟是哪个确切的应用配置产生了这条追踪？",{},{"id":940,"data":941,"type":42,"tunes":943},"h-routing",{"text":942,"level":253},"多提供商和模型路由产生运营策略",{},{"id":945,"data":946,"type":218,"tunes":948},"p-route-1",{"text":947},"一旦应用可以使用多个提供商或本地模型，路由就成为一种运营策略，而不再只是一个简单的模型字符串。",{},{"id":950,"data":951,"type":218,"tunes":953},"p-route-2",{"text":952},"路由可能取决于能力、延迟、成本、隐私、上下文长度、可用性、工具支持或本地性。回退机制可以在保持正常运行时间的同时，改变答案质量或数据处理假设。",{},{"id":955,"data":956,"type":218,"tunes":958},"p-route-3",{"text":957},"因此，LLMOps 应记录实际选择了哪条路由，并独立评估各条路由，而不是将每个兼容端点视为行为上可互换的。",{},{"id":960,"data":961,"type":42,"tunes":963},"h-implementation",{"text":962,"level":253},"原始实现证据",{},{"id":965,"data":966,"type":42,"tunes":968},"h-client",{"text":967,"level":252},"Aaasaasa AI Client：提供商、模型和运行时是独立的运维对象",{},{"id":970,"data":971,"type":218,"tunes":973},"p-client-1",{"text":972},"Aaasaasa AI Client 将代理\u002F客户端、提供商、模型、运行时位置和权限分开。其 AI Hub 支持 Ollama、LM Studio\u002FOpenAI 兼容端点以及其他提供商协议，而不是将“模型”视为一个全局设置。",{},{"id":975,"data":976,"type":218,"tunes":978},"p-client-2",{"text":977},"该实现包括动态本地模型发现、流式传输、思考输出以及显式的 Ollama 预热\u002F加载和卸载控制。这是运维证据，表明本地 LLM 服务引入了超出 API 模型名称的资源生命周期问题。",{},{"id":980,"data":981,"type":218,"tunes":983},"p-client-3",{"text":982},"提供商状态通过提供商适配器查询，连接类型区分本地、云 API、账户支持、远程代理和 Web 客户端路径。这些是 LLM 感知平台必须呈现的具体运维维度。",{},{"id":985,"data":986,"type":218,"tunes":988},"p-client-4",{"text":987},"该仓库还保留了一个重要边界：本地运行时并不自动等于本地推理。提供商\u002F模型\u002F运行时位置是影响隐私、延迟、成本和可用性的版本化或可配置事项。",{},{"id":990,"data":991,"type":42,"tunes":993},"h-sot",{"text":992,"level":252},"Source of Truth Research Engine：LLM 应用状态超出模型本身",{},{"id":995,"data":996,"type":218,"tunes":998},"p-sot-1",{"text":997},"Source of Truth Research Engine 围绕本地模型辅助研究，结合了词法搜索、可选嵌入、来源快照、SHA-256 身份、声明、溯源和矛盾跟踪。",{},{"id":1000,"data":1001,"type":218,"tunes":1003},"p-sot-2",{"text":1002},"这是有用的 LLMOps 证据，因为仅更改模型并不能定义研究系统。检索、来源获取、证据分类和持久溯源是独立的运维产物。",{},{"id":1005,"data":1006,"type":218,"tunes":1008},"p-sot-3",{"text":1007},"该实现有意将语义相似性视为发现而非证据，说明为什么 LLMOps 可观测性应区分检索行为与声明有效性。",{},{"id":1010,"data":1011,"type":411,"tunes":1040},"impl-table",{"content":1012,"stretched":43,"withHeadings":14},[1013,1016,1019,1022,1025,1028,1031,1034,1037],[1014,1015],"观察到的实现","LLMOps 经验",[1017,1018],"多种提供商协议","提供商身份是一种运维依赖",[1020,1021],"动态模型发现","可用模型可以独立于应用代码发生变化",[1023,1024],"Ollama 加载\u002F卸载控制","本地模型具有内存\u002F资源生命周期",[1026,1027],"提供商健康\u002F状态适配器","模型可用性需要运行时可观测性",[1029,1030],"运行时与推理位置分离","部署拓扑不是一个布尔值的“本地\u002F云”",[1032,1033],"集中权限","模型能力与工具权限必须保持分离",[1035,1036],"词法 + 语义检索管道","检索配置是应用行为的一部分",[1038,1039],"来源\u002F溯源持久化","运维状态和证据存在于模型权重之外",{},{"id":1042,"data":1043,"type":226,"tunes":1046},"impl-boundary",{"body":1044,"title":1045,"variant":240},"这些项目展示了多提供商\u002F本地模型运维、权限分离、检索基础设施和证据持久化。它们并非作为完整的商业 LLMOps 平台或大规模生产流量的证明来呈现。","证据边界",{},{"id":1048,"data":1049,"type":42,"tunes":1051},"h-failures",{"text":1050,"level":253},"常见 LLMOps 故障模式",{},{"id":1053,"data":1054,"type":411,"tunes":1095},"failure-table",{"content":1055,"stretched":43,"withHeadings":14},[1056,1059,1062,1065,1068,1071,1074,1077,1080,1083,1086,1089,1092],[1057,1058],"故障模式","实际出了什么问题",[1060,1061],"模型别名被静默升级","行为在未受控发布的情况下发生变化",[1063,1064],"提示词更改但未进行评估","行为回归通过了常规单元测试",[1066,1067],"RAG 索引过期","生成模型被错误归咎于检索\u002F数据故障",[1069,1070],"仅记录最终答案","检索\u002F工具\u002F上下文轨迹中的根本原因不可见",[1072,1073],"提供商回退是静默的","不同的模型\u002F数据路径在无归因的情况下改变行为",[1075,1076],"Token 成本全局跟踪","昂贵的工作流无法被定位",[1078,1079],"评判模型发生变化","评估分数在应用未更改的情况下漂移",[1081,1082],"生产跟踪从未转化为测试","已知故障反复出现",[1084,1085],"本地模型无限期保持加载","VRAM\u002F资源压力成为运维不稳定因素",[1087,1088],"权限仅编码在提示词中","模型行为被误认为授权",[1090,1091],"单一评估分数决定一切","不同的质量维度被压缩成一个误导性的数字",[1093,1094],"存在模型注册表但提示词\u002F索引版本不存在","应用谱系仍然不完整",{},{"id":1097,"data":1098,"type":42,"tunes":1100},"h-misconceptions",{"text":1099,"level":253},"常见误解",{},{"id":1102,"data":1103,"type":411,"tunes":1138},"misconceptions-table",{"content":1104,"stretched":43,"withHeadings":14},[1105,1108,1111,1114,1117,1120,1123,1126,1129,1132,1135],[1106,1107],"误解","纠正",[1109,1110],"“LLMOps 取代 MLOps。”","LLMOps 将 MLOps 原则扩展到 LLM 特定的应用行为。",[1112,1113],"“LLMOps 就是提示词工程。”","提示词只是模型、提供商、上下文、检索、工具、评估和运行时中的一种产物。",[1115,1116],"“托管 API 消除了运维工作。”","它们消除了一些模型服务\u002F训练工作，但增加了提供商生命周期、版本和依赖管理。",[1118,1119],"“如果 API 稳定，应用就稳定。”","模型行为和提供商\u002F模型快照可以独立于 API 模式发生变化。",[1121,1122],"“RAG 只是数据预处理。”","在生产环境中，它有自己的摄取、索引、检索和新鲜度生命周期。",[1124,1125],"“LLM 输出无法测试。”","它们可以通过确定性、参考、评判和人工标准进行评估。",[1127,1128],"“LLM 评判者是客观的基准真相。”","它们是基于模型的评估器，同样需要校准和版本控制。",[1130,1131],"“本地模型消除了 LLMOps。”","本地服务增加了模型文件、VRAM、加载\u002F卸载、运行时健康和升级问题。",[1133,1134],"“可观测性意味着 Token 计数。”","有用的可观测性会跟踪提示词、检索、工具、模型跨度和结果。",[1136,1137],"“持续训练是强制性的。”","许多 LLM 应用使用持续评估，而不训练基础模型。",{},{"id":1140,"data":1141,"type":42,"tunes":1143},"h-design",{"text":1142,"level":253},"实用的LLMOps设计流程",{},{"id":1145,"data":1146,"type":346,"tunes":1185},"design-flow",{"steps":1147,"title":1184,"orientation":345},[1148,1151,1154,1157,1160,1163,1166,1169,1172,1175,1178,1181],{"label":1149,"description":1150},"1. 定义行为单元","列出所有能实质性改变输出的组件：模型、提示词、检索、工具、上下文和策略。",{"label":1152,"description":1153},"2. 建立应用谱系","对代码、模型\u002F提供商、提示词、评估数据集、检索配置和工具契约进行版本控制。",{"label":1155,"description":1156},"3. 构建有代表性的评估数据集","使用来自设计和生产的预期成功\u002F失败案例。",{"label":1158,"description":1159},"4. 分离确定性测试和行为测试","将模式\u002F安全断言与语义输出评估区分开。",{"label":1161,"description":1162},"5. 追踪端到端执行","对模型、检索、重排序、工具和代理\u002F运行时跨度进行埋点。",{"label":1164,"description":1165},"6. 定义发布门禁","设置质量、安全、延迟和成本阈值。",{"label":1167,"description":1168},"7. 固定或明确记录模型版本","将模型\u002F提供商变更视为发布事件。",{"label":1170,"description":1171},"8. 渐进式部署","在后果需要时使用功能标志、金丝雀发布或分阶段推出。",{"label":1173,"description":1174},"9. 评估生产追踪","衡量真实任务行为并识别反复出现的故障。",{"label":1176,"description":1177},"10. 将故障反馈到评估数据集","将事件和纠正转化为永久的回归覆盖。",{"label":1179,"description":1180},"11. 监控提供商和数据生命周期","跟踪弃用、索引新鲜度、源变更和运行时可用性。",{"label":1182,"description":1183},"12. 干净地退役过时版本","在迁移和证据保留决策后，移除旧的提示词\u002F模型\u002F索引\u002F凭证。","运营完整的产生行为的系统",{},{"id":1187,"data":1188,"type":42,"tunes":1190},"h-checklist",{"text":1189,"level":253},"LLMOps架构检查清单",{},{"id":1192,"data":1193,"type":411,"tunes":1240},"checklist-table",{"content":1194,"stretched":43,"withHeadings":14},[1195,1198,1201,1204,1207,1210,1213,1216,1219,1222,1225,1228,1231,1234,1237],[1196,1197],"问题","预期证据",[1199,1200],"哪个模型\u002F提供商\u002F版本服务了该请求？","可追踪的模型身份",[1202,1203],"哪些提示词\u002F指令处于活动状态？","版本化的应用代码\u002F配置",[1205,1206],"哪些上下文到达了模型？","上下文\u002F检索追踪",[1208,1209],"使用了哪个语料库\u002F索引版本？","检索谱系",[1211,1212],"哪些工具可用并被调用？","工具模式 + 轨迹追踪",[1214,1215],"应用了哪些权限？","运行时授权记录",[1217,1218],"如何衡量质量？","版本化的评估数据集 + 评分器",[1220,1221],"如何测试模型升级？","行为回归测试套件",[1223,1224],"如何抽样生产质量？","追踪评估\u002F反馈流程",[1226,1227],"能否近似复现一次故障？","模型\u002F上下文\u002F提供商\u002F应用谱系",[1229,1230],"成本花在哪里？","按追踪的模型\u002F工具\u002F检索归因",[1232,1233],"什么触发回滚？","定义的质量\u002F安全\u002F成本\u002F可用性阈值",[1235,1236],"如何处理提供商弃用？","迁移\u002F回退流程",[1238,1239],"如何运营本地模型？","健康、资源、加载\u002F卸载和版本控制",{},{"id":1242,"data":1243,"type":42,"tunes":1245},"h-edge",{"text":1244,"level":253},"边缘情况和限制",{},{"id":1247,"data":1248,"type":218,"tunes":1250},"p-edge-1",{"text":1249},"一个简单的应用，调用一个固定的托管模型，没有检索或工具，可能只需要轻量级的LLMOps：版本化的提示词代码、评估、模型固定、基本追踪和提供商监控。",{},{"id":1252,"data":1253,"type":218,"tunes":1255},"p-edge-2",{"text":1254},"一个自托管的微调模型可能需要几乎完整的经典MLOps堆栈，加上LLM特定的应用评估，使得MLOps和LLMOps之间的界限有意模糊。",{},{"id":1257,"data":1258,"type":218,"tunes":1260},"p-edge-3",{"text":1259},"一个代理平台可能只有最少的模型训练操作，但需要大量的运行时操作，因为故障发生在工具选择、状态和编排中。",{},{"id":1262,"data":1263,"type":218,"tunes":1265},"p-edge-4",{"text":1264},"一个重度依赖RAG的系统，其操作可能主要由文档摄取和检索质量主导，而不是模型服务。",{},{"id":1267,"data":1268,"type":218,"tunes":1270},"p-edge-5",{"text":1269},"术语将继续演变。持久的架构问题不是哪个“Ops”标签胜出，而是哪些工件产生行为，因此必须被版本化、评估、观察和治理。",{},{"id":1272,"data":1273,"type":42,"tunes":1275},"h-change",{"text":1274,"level":253},"什么会改变这个答案？",{},{"id":1277,"data":1278,"type":218,"tunes":1280},"p-change-1",{"text":1279},"如果基础模型提供商标准化了完全稳定的模型行为和长期版本支持，提供商\u002F快照管理可能在操作上变得不那么重要。",{},{"id":1282,"data":1283,"type":218,"tunes":1285},"p-change-2",{"text":1284},"如果应用越来越多地拥有微调或训练，经典的MLOps关注点将再次变得更加核心。",{},{"id":1287,"data":1288,"type":218,"tunes":1290},"p-change-3",{"text":1289},"操作原则将保持不变：每个能实质性改变生产行为的组件都属于谱系、测试、可观测性和变更控制。",{},{"id":1292,"data":1293,"type":42,"tunes":1295},"h-related",{"text":1294,"level":253},"相关规范知识",{},{"id":1297,"data":1298,"type":218,"tunes":1300},"p-related-1",{"text":1299},"LLMOps位于AI治理和企业AI架构之下：治理定义哪些变更需要证据和批准，而LLMOps提供操作机制来版本化、评估、部署和观察这些变更。",{},{"id":1302,"data":1303,"type":218,"tunes":1305},"p-related-2",{"text":1304},"上下文工程和RAG是许多LLM应用内部的操作子领域，因为上下文和检索可以独立于模型改变行为。",{},{"id":1307,"data":1308,"type":218,"tunes":1310},"p-related-3",{"text":1309},"代理式AI将LLMOps进一步扩展到轨迹、权限和工具运行时操作。",{},{"id":1312,"data":1313,"type":603,"tunes":1318},"ref-memory",{"url":1314,"title":1315,"excerpt":1316,"ctaLabel":1317},"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context","AI Agent 记忆不是 RAG：如何分离记忆、检索、状态与上下文","当记忆、检索、应用状态和模型上下文保持为独立生命周期对象时，运营可靠性会提高。","阅读架构文章",{},{"id":1320,"data":1321,"type":603,"tunes":1326},"ref-avb",{"url":1322,"title":1323,"excerpt":1324,"ctaLabel":1325},"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers","答案有效性边界：相关性与可靠 AI 答案之间缺失的层","LLMOps 评估应保留答案仍受支持的版本、范围和证据条件。","阅读答案有效性边界",{},{"id":1328,"data":1329,"type":42,"tunes":1331},"h-faq",{"text":1330,"level":253},"常见问题",{},{"id":1333,"data":1334,"type":1333,"tunes":1373},"faq",{"items":1335,"title":1372},[1336,1340,1344,1348,1352,1356,1360,1364,1368],{"id":1337,"answer":1338,"question":1339},"faq1","MLOps 在数据、训练、部署和监控方面运营机器学习系统。LLMOps 将这些实践扩展到 LLM 应用，其中提示、上下文、检索、提供商、工具和评估也会实质性地影响行为。","MLOps 和 LLMOps 有什么区别？",{"id":1341,"answer":1342,"question":1343},"faq2","不会。LLMOps 复用 MLOps 的学科，如 CI\u002FCD、血缘、评估、部署和监控，并增加 LLM 特有的运营关注点。","LLMOps 会取代 MLOps 吗？",{"id":1345,"answer":1346,"question":1347},"faq3","不一定。许多应用使用外部基础模型，转而依赖对提示、模型、检索和应用行为的持续评估。微调或自训练系统仍可能需要训练流水线。","LLM 应用需要持续训练吗？",{"id":1349,"answer":1350,"question":1351},"faq4","生成式输出是开放式的，模型行为可能因提示、快照和上下文而变化。评估提供可重复的证据，证明发布仍满足定义的质量和安全标准。","为什么评估在 LLMOps 中如此重要？",{"id":1353,"answer":1354,"question":1355},"faq5","至少包括：应用代码、模型\u002F提供商\u002F版本、提示、评估数据集\u002F评分器、检索配置\u002F索引、工具模式、上下文规则以及相关的安全\u002F权限配置。","在 LLMOps 中应该对什么进行版本控制？",{"id":1357,"answer":1358,"question":1359},"faq6","不够。同一个提示在另一个模型、检索集、上下文顺序、工具面或提供商下可能表现不同。","提示版本控制就足够了吗？",{"id":1361,"answer":1362,"question":1363},"faq7","GenAIOps 是运营生成式 AI 应用的另一个行业术语。一些供应商将其与 LLMOps 互换使用，或作为比 LLMOps 更广泛的标签。","什么是 GenAIOps？",{"id":1365,"answer":1366,"question":1367},"faq8","监控端到端追踪，包括模型调用、提示\u002F上下文、检索、工具、延迟、令牌\u002F成本、质量样本、安全性以及最终任务结果。","如何监控 LLM 应用？",{"id":1369,"answer":1370,"question":1371},"faq9","可以。本地模型增加了自身的运营关注点，如模型文件、硬件\u002FVRAM、加载\u002F卸载、运行时健康、量化和升级管理。","本地 LLM 可以使用 LLMOps 实践吗？","MLOps 与 LLMOps 常见问题",{},{"id":1375,"data":1376,"type":42,"tunes":1378},"h-glossary",{"text":1377,"level":253},"术语表",{},{"id":1380,"data":1381,"type":1380,"tunes":1426},"glossary",{"title":1382,"entries":1383},"关键 MLOps 和 LLMOps 术语",[1384,1386,1388,1392,1395,1399,1403,1407,1411,1414,1418,1422],{"term":415,"anchor":414,"definition":1385},"用于构建、部署、监控和维护机器学习系统及其数据\u002F模型生命周期的工程实践。",{"term":418,"anchor":417,"definition":1387},"针对生产应用的运营实践，其行为实质性地依赖于大型语言模型以及周围的提示、上下文、检索、工具和运行时。",{"term":1389,"anchor":1390,"definition":1391},"GenAIOps","genaiops","生成式 AI 应用的运营学科；通常用作 LLMOps 的更广泛或替代标签。",{"term":400,"anchor":1393,"definition":1394},"continuous-training","随着数据或实现变化，对 ML 模型进行自动化或重复的再训练和 serving。",{"term":1396,"anchor":1397,"definition":1398},"持续评估","continuous-evaluation","针对版本化数据集和标准，对候选和生产 AI 行为进行重复评估。",{"term":1400,"anchor":1401,"definition":1402},"模型快照","model-snapshot","托管或打包模型的具体版本，其行为可以被测试和引用。",{"term":1404,"anchor":1405,"definition":1406},"应用血缘","application-lineage","代码、模型\u002F提供商、提示、数据\u002F检索、工具、运行时和发布配置之间的可追溯关系。",{"term":1408,"anchor":1409,"definition":1410},"追踪","trace","一次应用执行的结构化记录，包含模型调用、检索和工具操作等跨度。",{"term":469,"anchor":1412,"definition":1413},"eval-dataset","版本化的代表性输入、期望值以及可选的追踪\u002F输出集合，用于衡量行为。",{"term":1415,"anchor":1416,"definition":1417},"LLM 评判器","llm-judge","用作定性或语义标准评估器的语言模型；它本身就是一个版本化的评估依赖项。",{"term":1419,"anchor":1420,"definition":1421},"行为回归","behavioral-regression","尽管接口和代码继续成功执行，应用输出或轨迹仍出现退化。",{"term":1423,"anchor":1424,"definition":1425},"提供商路由","provider-routing","根据能力、成本、延迟、隐私或可用性在可用模型提供商\u002F端点之间进行选择的策略。",{},{"id":1428,"data":1429,"type":42,"tunes":1431},"h-conclusion",{"text":1430,"level":253},"结论",{},{"id":1433,"data":1434,"type":218,"tunes":1436},"p-conclusion-1",{"text":1435},"MLOps 和 LLMOps 共享相同的工程目标：使 AI 系统足够可复现、足够可测试、足够可观测，以便在生产中可靠运行。",{},{"id":1438,"data":1439,"type":218,"tunes":1441},"p-conclusion-2",{"text":1440},"区别在于系统的形态。经典 MLOps 通常以训练和 serving 模型工件为中心；LLMOps 必须运营一个行为栈，其中模型快照、提示、上下文、检索、工具、权限和提供商可以独立变化。",{},{"id":1443,"data":1444,"type":218,"tunes":1446},"p-conclusion-3",{"text":1445},"最简短有用的规则是：对一切可能实质性改变 LLM 应用行为的内容进行版本控制、评估和观测——而不仅仅是模型。",{},{"id":1448,"data":1449,"type":42,"tunes":1451},"h-sources",{"text":1450,"level":253},"主要来源和当前文档",{},{"id":1453,"data":1454,"type":218,"tunes":1456},"p-sources-note",{"text":1455},"以下来源为 MLOps 基线和 LLM 及代理应用的当前运营模式提供依据。项目部分是原始实现证据，并且有意比关于完整 LLMOps 平台的声明更窄。",{},{"id":1458,"data":1459,"type":1465,"tunes":1466},"src-google-mlops",{"link":1460,"meta":1461},"https:\u002F\u002Fdocs.cloud.google.com\u002Farchitecture\u002Fmlops-continuous-delivery-and-automation-pipelines-in-machine-learning",{"image":1462,"title":1463,"description":1464},{"url":381},"Google Cloud — MLOps：持续交付和自动化流水线","描述 ML 系统的 CI、CD、持续训练、模型注册表、元数据、serving 和监控的参考架构。","linkTool",{},{"id":1468,"data":1469,"type":1465,"tunes":1475},"src-aws-lineage",{"link":1470,"meta":1471},"https:\u002F\u002Fdocs.aws.amazon.com\u002Fwellarchitected\u002Flatest\u002Fmachine-learning-lens\u002Fmlops02-bp04.html",{"image":1472,"title":1473,"description":1474},{"url":381},"AWS Machine Learning Lens — 模型血缘","跨 ML 发布跟踪代码、数据、模型、环境和基础设施的当前指南。",{},{"id":1477,"data":1478,"type":1465,"tunes":1484},"src-aws-monitor",{"link":1479,"meta":1480},"https:\u002F\u002Fdocs.aws.amazon.com\u002Fwellarchitected\u002Flatest\u002Fmachine-learning-lens\u002Fmlops06-bp02.html",{"image":1481,"title":1482,"description":1483},{"url":381},"AWS Machine Learning Lens — 模型可观测性和跟踪","生产模型监控、漂移、端点健康和血缘的当前指南。",{},{"id":1486,"data":1487,"type":1465,"tunes":1493},"src-azure-llmops",{"link":1488,"meta":1489},"https:\u002F\u002Flearn.microsoft.com\u002Fen-us\u002Fazure\u002Fmachine-learning\u002Fprompt-flow\u002Fhow-to-end-to-end-llmops-with-prompt-flow",{"image":1490,"title":1491,"description":1492},{"url":381},"Microsoft Azure — GenAIOps \u002F LLMOps 生命周期","官方指南，描述 GenAIOps（有时称为 LLMOps）在初始化、实验、评估\u002F改进和部署方面的内容。",{},{"id":1495,"data":1496,"type":1465,"tunes":1502},"src-mlflow-genai",{"link":1497,"meta":1498},"https:\u002F\u002Fmlflow.org\u002Fdocs\u002Flatest\u002Fgenai\u002F",{"image":1499,"title":1500,"description":1501},{"url":381},"MLflow — 代理和 LLM 应用","当前 GenAI 运营文档，涵盖 LLM 应用和代理的追踪、评估、提示和生产可观测性。",{},{"id":1504,"data":1505,"type":1465,"tunes":1511},"src-mlflow-traces",{"link":1506,"meta":1507},"https:\u002F\u002Fwww.mlflow.org\u002Fdocs\u002Flatest\u002Fgenai\u002Feval-monitor\u002Frunning-evaluation\u002Ftraces\u002F",{"image":1508,"title":1509,"description":1510},{"url":381},"MLflow — 评估生产追踪","评估完整 LLM\u002F代理追踪（包括检索和工具调用轨迹）的当前指南。",{},{"id":1513,"data":1514,"type":1465,"tunes":1520},"src-mlflow-prompt-eval",{"link":1515,"meta":1516},"https:\u002F\u002Fmlflow.org\u002Fdocs\u002Flatest\u002Fgenai\u002Fprompt-registry\u002Fevaluate-prompts\u002F",{"image":1517,"title":1518,"description":1519},{"url":381},"MLflow — 评估提示词","当前使用版本化提示词、数据集、评分器和追踪的提示词\u002F模型评估工作流。",{},{"id":1522,"data":1523,"type":1465,"tunes":1529},"src-openai-api",{"link":1524,"meta":1525},"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Freference\u002Foverview",{"image":1526,"title":1527,"description":1528},{"url":381},"OpenAI API — 版本管理与模型快照","当前 API 指南建议固定模型版本并进行评估，因为提示行为可能在不同快照之间发生变化。",{},{"id":1531,"data":1532,"type":1465,"tunes":1538},"src-openai-prompting",{"link":1533,"meta":1534},"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fprompting",{"image":1535,"title":1536,"description":1537},{"url":381},"OpenAI — 提示词工程","当前指南建议将生产提示词视为应用程序代码，通过源代码控制进行版本管理，并用测试和评估检查覆盖变更。",{},{"id":1540,"data":1541,"type":1465,"tunes":1547},"src-openai-deprecations",{"link":1542,"meta":1543},"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fdeprecations",{"image":1544,"title":1545,"description":1546},{"url":381},"OpenAI — 弃用","当前提供商生命周期证据表明，模型和平台界面的退役是一种运营依赖。",{},{"id":1549,"data":1550,"type":1465,"tunes":1556},"src-openai-promptfoo",{"link":1551,"meta":1552},"https:\u002F\u002Fdevelopers.openai.com\u002Fcookbook\u002Fexamples\u002Fevaluation\u002Fmoving-from-openai-evals-to-promptfoo",{"image":1553,"title":1554,"description":1555},{"url":381},"OpenAI — 将评估工作流迁移到 Promptfoo","当前 2026 年迁移指南说明，随着提供商工具的变化，评估资产应保持可移植性。",{},"2.31","MLOps 运维机器学习系统；LLMOps 将这些实践扩展到围绕大型语言模型的提示、上下文、检索、提供商、工具、评估和运行时行为。","\u002Fuploads\u002F2026\u002F10\u002Fmlops-vs-llmops-what-changes-when-the-model-is-an-llm-1791487319869-2v7hxo.webp","mlops-vs-llmops-what-changes-when-the-model-is-an-llm-1791487319869-2v7hxo","PUBLISHED","2026-10-08T15:20:00.000Z","2026-10-08T19:20:01.249Z","2026-10-08T19:31:17.955Z",{"en":1566,"de":1567,"sr":1568,"es":1569,"fr":1570,"it":1571,"ru":1572,"zh":1573},"\u002Fblog\u002Fmlops-vs-llmops-what-changes-when-the-model-is-an-llm","\u002Fde\u002Fblog\u002Fmlops-vs-llmops-what-changes-when-the-model-is-an-llm","\u002Fsr\u002Fblog\u002Fmlops-vs-llmops-what-changes-when-the-model-is-an-llm","\u002Fes\u002Fblog\u002Fmlops-vs-llmops-what-changes-when-the-model-is-an-llm","\u002Ffr\u002Fblog\u002Fmlops-vs-llmops-what-changes-when-the-model-is-an-llm","\u002Fit\u002Fblog\u002Fmlops-vs-llmops-what-changes-when-the-model-is-an-llm","\u002Fru\u002Fblog\u002Fmlops-vs-llmops-what-changes-when-the-model-is-an-llm","\u002Fzh\u002Fblog\u002Fmlops-vs-llmops-what-changes-when-the-model-is-an-llm",[1575,1579,1583,1587],{"id":1576,"name":1577,"slug":1578},88,"版本管理（提示\u002F模型）","versioning",{"id":1580,"name":1581,"slug":1582},91,"监控（质量\u002F漂移）","monitoring",{"id":1584,"name":1585,"slug":1586},89,"评估框架","evaluation-harness",{"id":1588,"name":1589,"slug":1590},58,"评估与质量门槛","evaluation",{"id":1592,"login":1593,"email":1594,"displayName":1595},"20","rooth8233","aleksandar@stajic.de","Aleksandar Stajić",[1597,2723],{"lang":1598,"title":1599,"content":1600,"contentJson":1601,"excerpt":2722},"en","MLOps vs LLMOps: What Changes When the Model Is an LLM","{\"time\":1791487321430,\"blocks\":[{\"id\":\"intro\",\"type\":\"paragraph\",\"data\":{\"text\":\"MLOps is the engineering discipline for reliably developing, deploying, versioning and operating machine-learning systems; LLMOps extends that discipline to applications built around large language models, where production behavior depends not only on a model artifact but also on prompts, context, retrieval, provider\u002Fmodel versions, tool calls, safety controls and evaluation pipelines. LLMOps does not replace MLOps. It changes the operational unit from “a model plus serving pipeline” toward “an evolving LLM application whose behavior emerges from several independently changing components.”\"},\"tunes\":{}},{\"id\":\"direct\",\"type\":\"callout\",\"data\":{\"variant\":\"info\",\"title\":\"Direct answer\",\"body\":\"\u003Cstrong>MLOps operates ML systems. LLMOps operates LLM applications.\u003C\u002Fstrong>\u003Cbr>\u003Cbr>Classical MLOps commonly centers on data pipelines, training, validation, model registry, deployment, drift and retraining. LLMOps keeps those disciplines where relevant, but often adds prompt\u002Fcontext versioning, model\u002Fprovider abstraction, RAG indexes, agent\u002Ftool traces, semantic evaluations, safety tests, token\u002Fcost monitoring and regression testing across rapidly changing model snapshots.\"},\"tunes\":{}},{\"id\":\"boundary\",\"type\":\"callout\",\"data\":{\"variant\":\"warning\",\"title\":\"LLMOps is not just prompt management\",\"body\":\"A production LLM application can fail even when the prompt is unchanged: the provider can change a model snapshot, a RAG corpus can become stale, a reranker can regress, tool permissions can change, context assembly can drop evidence, or an agent can take a wrong trajectory. LLMOps therefore has to observe and version the system around the model, not only prompt text.\"},\"tunes\":{}},{\"id\":\"term-note\",\"type\":\"callout\",\"data\":{\"variant\":\"note\",\"title\":\"Terminology boundary\",\"body\":\"\u003Cstrong>LLMOps\u003C\u002Fstrong>, \u003Cstrong>GenAIOps\u003C\u002Fstrong> and related terms are widely used engineering labels, but they are not one universal formal standard with a single canonical lifecycle. Microsoft currently describes GenAIOps as “sometimes called LLMOps,” while MLflow groups operational tooling around agents and LLM applications. This article uses LLMOps as a practical architecture term for operating production systems whose behavior materially depends on LLMs.\"},\"tunes\":{}},{\"id\":\"current\",\"type\":\"callout\",\"data\":{\"variant\":\"note\",\"title\":\"Current-source note — 8 October 2026\",\"body\":\"The operational surface is changing quickly. OpenAI currently recommends pinning model snapshots and running evals because prompting behavior can change between snapshots, and several older platform-specific prompt\u002Feval surfaces are being retired in 2026. The stable architectural lesson is to keep prompts, tests and evals portable and versioned with the application rather than depend on one provider's dashboard object model.\"},\"tunes\":{}},{\"id\":\"toc\",\"type\":\"tableOfContents\",\"data\":{\"title\":\"Contents\",\"minLevel\":2,\"maxLevel\":3},\"tunes\":{}},{\"id\":\"h-meaning\",\"type\":\"header\",\"data\":{\"text\":\"What MLOps really means\",\"level\":2},\"tunes\":{}},{\"id\":\"p-mlops-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"MLOps applies software-engineering and operational discipline to machine-learning systems. The production challenge is broader than training a model: data collection, data validation, experimentation, reproducibility, model evaluation, deployment, infrastructure and monitoring all have to work together.\"},\"tunes\":{}},{\"id\":\"p-mlops-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Google's MLOps architecture guidance frames the discipline around continuous integration, continuous delivery and continuous training. CI validates not only code but also data, schemas and models; CD deploys ML pipelines and prediction services; CT can retrain and redeploy models as data or implementations change.\"},\"tunes\":{}},{\"id\":\"p-mlops-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"AWS guidance adds the same operational concerns from another angle: model lineage, model\u002Fversion traceability, drift monitoring and production-quality monitoring are core parts of keeping ML systems reliable after deployment.\"},\"tunes\":{}},{\"id\":\"h-llmops\",\"type\":\"header\",\"data\":{\"text\":\"What changes when the model is an LLM\",\"level\":2},\"tunes\":{}},{\"id\":\"p-llmops-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Large language models change the production problem because the application often does not own the complete model-training lifecycle. A team may call a hosted model API, run an open model locally, switch between providers or use several models for different tasks.\"},\"tunes\":{}},{\"id\":\"p-llmops-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The model is therefore only one versioned dependency inside a larger behavioral system. Prompts, retrieval results, context order, tools, model snapshot, temperature\u002Freasoning settings, safety filters and runtime orchestration can all change the output.\"},\"tunes\":{}},{\"id\":\"p-llmops-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"This creates a broader operational question: which combination of model, context, data, prompt, tools and runtime produced this behavior? LLMOps exists to make that question answerable and the answer reproducible enough for engineering work.\"},\"tunes\":{}},{\"id\":\"h-simple\",\"type\":\"header\",\"data\":{\"text\":\"The simplest example\",\"level\":2},\"tunes\":{}},{\"id\":\"p-simple-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Suppose an application answers internal policy questions.\"},\"tunes\":{}},{\"id\":\"p-simple-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"In a classical ML framing, you might version a trained classifier, deploy it and monitor prediction quality. In an LLM application, the answer might depend on a hosted model snapshot, a system prompt, an embedding model, a vector index, retrieval filters, a reranker and the final selected context.\"},\"tunes\":{}},{\"id\":\"p-simple-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"Changing any one of those components can change the final answer even though the application endpoint and user question stay identical.\"},\"tunes\":{}},{\"id\":\"simple-flow\",\"type\":\"processFlow\",\"data\":{\"title\":\"A typical LLMOps release path\",\"orientation\":\"auto\",\"steps\":[{\"label\":\"1. Change one component\",\"description\":\"Prompt, model, provider, retrieval setting, tool schema or application code changes.\"},{\"label\":\"2. Run deterministic tests\",\"description\":\"Validate schemas, permissions, tool contracts, retrieval filters and application behavior.\"},{\"label\":\"3. Run behavioral evals\",\"description\":\"Compare representative outputs, retrieval quality and agent\u002Ftool trajectories against acceptance criteria.\"},{\"label\":\"4. Compare cost and latency\",\"description\":\"Measure token use, model calls, retrieval\u002Ftool overhead and response latency.\"},{\"label\":\"5. Deploy controlled version\",\"description\":\"Ship the concrete application configuration with model\u002Fprovider versions recorded.\"},{\"label\":\"6. Trace production behavior\",\"description\":\"Capture relevant model, retrieval, tool and runtime spans.\"},{\"label\":\"7. Evaluate production traces\",\"description\":\"Sample real executions for quality, grounding, safety and task success.\"},{\"label\":\"8. Roll back or iterate\",\"description\":\"Use regression evidence and operational signals to decide the next release.\"}]},\"tunes\":{}},{\"id\":\"h-stops\",\"type\":\"header\",\"data\":{\"text\":\"Where the simple example stops\",\"level\":2},\"tunes\":{}},{\"id\":\"p-stops-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Some LLM systems still train or fine-tune their own models, so traditional MLOps practices such as training pipelines, model registry and data lineage remain directly relevant.\"},\"tunes\":{}},{\"id\":\"p-stops-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Other systems use only external foundation-model APIs and never run continuous training. Their main operational workload is application evaluation, model\u002Fprovider change management, prompt\u002Fcontext versioning, retrieval quality and observability.\"},\"tunes\":{}},{\"id\":\"p-stops-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"There is therefore no single universal “LLMOps pipeline.” The exact lifecycle depends on whether you train, fine-tune, self-host, retrieve external knowledge, run agents or depend on managed model APIs.\"},\"tunes\":{}},{\"id\":\"h-compare\",\"type\":\"header\",\"data\":{\"text\":\"MLOps vs LLMOps\",\"level\":2},\"tunes\":{}},{\"id\":\"main-comparison\",\"type\":\"comparison\",\"data\":{\"title\":\"What stays the same and what expands\",\"layout\":\"table\",\"columns\":[{\"id\":\"mlops\",\"label\":\"MLOps\"},{\"id\":\"llmops\",\"label\":\"LLMOps\"}],\"rows\":[{\"id\":\"unit\",\"label\":\"Primary operational unit\",\"values\":[\"\",\"\"]},{\"id\":\"model\",\"label\":\"Model ownership\",\"values\":[\"\",\"\"]},{\"id\":\"change\",\"label\":\"Typical change\",\"values\":[\"\",\"\"]},{\"id\":\"eval\",\"label\":\"Evaluation\",\"values\":[\"\",\"\"]},{\"id\":\"monitor\",\"label\":\"Production monitoring\",\"values\":[\"\",\"\"]},{\"id\":\"training\",\"label\":\"Continuous training\",\"values\":[\"\",\"\"]},{\"id\":\"registry\",\"label\":\"Versioned artifacts\",\"values\":[\"\",\"\"]},{\"id\":\"rollback\",\"label\":\"Rollback target\",\"values\":[\"\",\"\"]}]},\"tunes\":{}},{\"id\":\"h-extension\",\"type\":\"header\",\"data\":{\"text\":\"LLMOps extends MLOps rather than replacing it\",\"level\":2},\"tunes\":{}},{\"id\":\"p-extension-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The core operational principles do not disappear: source control, CI\u002FCD, reproducibility, lineage, deployment controls, monitoring, rollback and measurable acceptance criteria remain essential.\"},\"tunes\":{}},{\"id\":\"p-extension-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The extension is that more behavior-defining artifacts now sit outside the model weights. A managed foundation model can change behavior through snapshot upgrades, while application output can change through prompt or retrieval changes without any model retraining.\"},\"tunes\":{}},{\"id\":\"p-extension-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"This is why the useful hierarchy is usually DevOps → MLOps → LLMOps\u002FGenAIOps as increasingly specialized operational concerns, not three mutually exclusive practices.\"},\"tunes\":{}},{\"id\":\"h-artifacts\",\"type\":\"header\",\"data\":{\"text\":\"What has to be versioned in LLMOps?\",\"level\":2},\"tunes\":{}},{\"id\":\"artifact-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Artifact\",\"Why it matters\"],[\"Application code\",\"Defines orchestration, validation, retries and business behavior\"],[\"Model family + snapshot\u002Fversion\",\"Different snapshots can produce different behavior\"],[\"Provider \u002F endpoint\",\"Changes data flow, latency, limits, pricing and availability\"],[\"Prompt\u002Finstruction code\",\"Changes model behavior even with same model\"],[\"Generation\u002Freasoning parameters\",\"Can alter determinism, latency, depth and cost\"],[\"Eval dataset\",\"Defines what “good enough” is tested against\"],[\"Scorers \u002F graders\",\"Define how quality is measured\"],[\"Embedding model\",\"Changes vector representation and retrieval behavior\"],[\"Chunking\u002Findex configuration\",\"Changes what can be retrieved\"],[\"Reranker \u002F retrieval fusion\",\"Changes result ordering\"],[\"Tool schemas\",\"Change what the model can request and how\"],[\"Permission profile\",\"Changes what tool actions may actually execute\"],[\"Context assembly rules\",\"Change what evidence and state reach the model\"],[\"Safety\u002Fguardrail configuration\",\"Changes allowed or blocked behavior\"]]},\"tunes\":{}},{\"id\":\"h-model-version\",\"type\":\"header\",\"data\":{\"text\":\"Model snapshots become release dependencies\",\"level\":2},\"tunes\":{}},{\"id\":\"p-model-version-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"With hosted LLMs, the team may not control model training, but it still controls which model or snapshot the application calls.\"},\"tunes\":{}},{\"id\":\"p-model-version-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"OpenAI's current API guidance explicitly warns that prompting behavior can change between model snapshots and recommends pinning production applications to specific snapshots where consistency matters, then running evals when upgrading.\"},\"tunes\":{}},{\"id\":\"p-model-version-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The operational consequence is straightforward: model upgrades should be treated as application releases, not invisible infrastructure maintenance.\"},\"tunes\":{}},{\"id\":\"h-provider\",\"type\":\"header\",\"data\":{\"text\":\"Provider lifecycle becomes part of operations\",\"level\":2},\"tunes\":{}},{\"id\":\"p-provider-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"LLM applications often depend on provider rate limits, deprecation schedules, API semantics, context limits, data-handling rules and pricing.\"},\"tunes\":{}},{\"id\":\"p-provider-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"A provider can deprecate a model while your application code remains unchanged. OpenAI's current deprecation schedule, for example, includes 2026 retirement dates for older model snapshots and platform surfaces.\"},\"tunes\":{}},{\"id\":\"p-provider-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"LLMOps therefore needs provider lifecycle tracking, migration testing and fallback decisions in addition to model-quality monitoring.\"},\"tunes\":{}},{\"id\":\"h-prompt\",\"type\":\"header\",\"data\":{\"text\":\"Prompts behave like production code\",\"level\":2},\"tunes\":{}},{\"id\":\"p-prompt-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Prompts are executable behavioral configuration. Small changes can alter output quality, tool selection and policy interpretation.\"},\"tunes\":{}},{\"id\":\"p-prompt-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"OpenAI's current guidance recommends storing production prompts in application code, reviewing prompt changes through pull requests, using typed inputs and covering changes with tests and evaluation checks.\"},\"tunes\":{}},{\"id\":\"p-prompt-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"That makes prompt versioning less like editing marketing copy and more like changing a function whose output is probabilistic and model-dependent.\"},\"tunes\":{}},{\"id\":\"h-context\",\"type\":\"header\",\"data\":{\"text\":\"Context engineering becomes an operational concern\",\"level\":2},\"tunes\":{}},{\"id\":\"p-context-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The production model rarely receives only a static prompt. It may receive conversation history, retrieved documents, tool outputs, memory, current application state and policy instructions.\"},\"tunes\":{}},{\"id\":\"p-context-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"LLMOps must therefore observe context assembly: which evidence was selected, which state version was current, whether truncation occurred and whether important instructions survived compaction.\"},\"tunes\":{}},{\"id\":\"p-context-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"A model regression and a context regression can look identical at the final answer. Tracing the actual context path is what lets the team separate them.\"},\"tunes\":{}},{\"id\":\"h-rag\",\"type\":\"header\",\"data\":{\"text\":\"RAG creates its own operational lifecycle\",\"level\":2},\"tunes\":{}},{\"id\":\"p-rag-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A RAG system introduces a second production pipeline beside model inference: ingestion, extraction, chunking, metadata, embeddings, indexes, retrieval, reranking and context selection.\"},\"tunes\":{}},{\"id\":\"p-rag-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The knowledge corpus can change every day even when the model and prompt do not. A stale index or broken metadata filter can therefore degrade answer quality without any model drift.\"},\"tunes\":{}},{\"id\":\"p-rag-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"LLMOps for RAG should track corpus\u002Findex version, embedding model, chunking policy, retrieval configuration, source freshness and retrieval metrics separately from generation quality.\"},\"tunes\":{}},{\"id\":\"ref-rag-diagnostic\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method\",\"title\":\"RAG Failed — But Which Layer Actually Failed? A Diagnostic Method\",\"excerpt\":\"A production LLM pipeline needs separate observability for source coverage, retrieval, ranking, context assembly and generation.\",\"ctaLabel\":\"Read the RAG diagnostic method\"},\"tunes\":{}},{\"id\":\"h-evals\",\"type\":\"header\",\"data\":{\"text\":\"Evals replace “looks good to me” with release evidence\",\"level\":2},\"tunes\":{}},{\"id\":\"p-eval-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Generative outputs are often open-ended, so exact-match tests are insufficient for many tasks. LLMOps adds evaluation datasets and scorers that can measure task success, correctness, safety, groundedness, style or domain-specific acceptance criteria.\"},\"tunes\":{}},{\"id\":\"p-eval-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"MLflow's current GenAI evaluation stack supports versioned evaluation datasets, prompt\u002Fmodel comparisons, custom scorers and evaluation over complete traces.\"},\"tunes\":{}},{\"id\":\"p-eval-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The strongest practice is evaluation-driven development: define representative cases and acceptance criteria before or alongside changes, then compare releases against the same evidence.\"},\"tunes\":{}},{\"id\":\"eval-rule\",\"type\":\"callout\",\"data\":{\"variant\":\"success\",\"title\":\"Behavioral changes need behavioral tests\",\"body\":\"A deployment should not be considered equivalent merely because the API contract still works. If the prompt, model, retrieval or tools changed, the behavioral regression suite should run again.\"},\"tunes\":{}},{\"id\":\"h-judges\",\"type\":\"header\",\"data\":{\"text\":\"LLM-as-a-judge is useful but not ground truth\",\"level\":2},\"tunes\":{}},{\"id\":\"p-judge-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"LLM judges can scale evaluation for qualities that are expensive to encode as deterministic assertions, such as relevance, tone or groundedness.\"},\"tunes\":{}},{\"id\":\"p-judge-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"However, the judge is another model with its own bias, version and prompt. Judge configuration should therefore be versioned and calibrated against human or deterministic reference cases where consequence matters.\"},\"tunes\":{}},{\"id\":\"p-judge-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"A production eval can mix deterministic checks, reference-based metrics, model judges and human review rather than asking one metric to represent every quality dimension.\"},\"tunes\":{}},{\"id\":\"h-tracing\",\"type\":\"header\",\"data\":{\"text\":\"Tracing becomes more important than endpoint logs\",\"level\":2},\"tunes\":{}},{\"id\":\"p-trace-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Traditional API logs can tell you that a request took two seconds and returned HTTP 200. They cannot tell you which retrieved chunks were selected, which tool the agent called or which model span consumed most tokens.\"},\"tunes\":{}},{\"id\":\"p-trace-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"MLflow's current GenAI tracing captures prompts, retrievals, tool calls and application spans, and its production evaluation flow can score intermediate trajectory information rather than only final text.\"},\"tunes\":{}},{\"id\":\"p-trace-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"This is a major LLMOps shift: observability follows the behavioral graph of the application, not only the serving endpoint.\"},\"tunes\":{}},{\"id\":\"h-agent\",\"type\":\"header\",\"data\":{\"text\":\"Agents expand LLMOps into runtime operations\",\"level\":2},\"tunes\":{}},{\"id\":\"p-agent-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"An agentic application can perform several model calls, tool invocations and state transitions before producing a result.\"},\"tunes\":{}},{\"id\":\"p-agent-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Operating agents therefore requires step counts, tool-call traces, permission denials, retries, loop detection, human approvals and verified final state in addition to ordinary model latency and token metrics.\"},\"tunes\":{}},{\"id\":\"p-agent-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"A correct final answer can hide a bad trajectory, so agent evaluation must inspect the path as well as the result.\"},\"tunes\":{}},{\"id\":\"ref-agent-reliability\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough\",\"title\":\"AI Agent Reliability: Why the Final Answer Is Not Enough\",\"excerpt\":\"Why agent production evaluation must include tool calls, state transitions, approvals and recoverability.\",\"ctaLabel\":\"Read the agent reliability article\"},\"tunes\":{}},{\"id\":\"h-cost\",\"type\":\"header\",\"data\":{\"text\":\"Tokens, model calls and context become cost variables\",\"level\":2},\"tunes\":{}},{\"id\":\"p-cost-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Classical ML inference cost is often dominated by serving infrastructure or per-prediction compute. LLM applications can add provider token pricing, repeated agent calls, embedding calls, reranking and tool\u002Fruntime overhead.\"},\"tunes\":{}},{\"id\":\"p-cost-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Cost therefore has to be attributed to task or trace, not only to one endpoint. A workflow that makes eight hidden model calls can be functionally correct but operationally unacceptable.\"},\"tunes\":{}},{\"id\":\"p-cost-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"Latency behaves the same way: model latency, retrieval, reranking and external tools compose into end-to-end user latency.\"},\"tunes\":{}},{\"id\":\"h-cache\",\"type\":\"header\",\"data\":{\"text\":\"Caching becomes semantic, not only technical\",\"level\":2},\"tunes\":{}},{\"id\":\"p-cache-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"LLM systems can cache prompts, embeddings, retrieval results or full responses, but the cache key must reflect the semantics that can change the result.\"},\"tunes\":{}},{\"id\":\"p-cache-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"A response cache that ignores model version, tenant, permissions or source freshness can return a technically valid but semantically invalid answer.\"},\"tunes\":{}},{\"id\":\"p-cache-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"LLMOps therefore treats cache invalidation as part of model\u002Fcontext\u002Fdata versioning rather than only infrastructure optimization.\"},\"tunes\":{}},{\"id\":\"h-safety\",\"type\":\"header\",\"data\":{\"text\":\"Safety and permissions become release criteria\",\"level\":2},\"tunes\":{}},{\"id\":\"p-safety-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Generative systems can produce unbounded text and agents can trigger external actions. Safety testing therefore sits closer to ordinary CI\u002FCD than in many classical predictive ML systems.\"},\"tunes\":{}},{\"id\":\"p-safety-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Permission checks, prompt-injection tests, tenant-isolation tests and side-effect approvals should be reproducible regression tests where those risks exist.\"},\"tunes\":{}},{\"id\":\"p-safety-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The model may suggest an operation, but the runtime still has to enforce authorization. LLMOps owns the evidence that those controls continue to work after model, prompt or tool changes.\"},\"tunes\":{}},{\"id\":\"h-ci\",\"type\":\"header\",\"data\":{\"text\":\"What CI looks like in LLMOps\",\"level\":2},\"tunes\":{}},{\"id\":\"ci-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"CI layer\",\"Example checks\"],[\"Code\",\"Unit tests, type checks, schema validation\"],[\"Prompts\",\"Template rendering, required variables, policy text, snapshot review\"],[\"Models\u002Fproviders\",\"Compatibility, output schema, capability and regression tests\"],[\"RAG\",\"Chunking fixtures, filter tests, Recall@k, reranker regression\"],[\"Tools\",\"Input\u002Foutput schema tests, permission tests, idempotency tests\"],[\"Agents\",\"Trajectory fixtures, loop limits, handoff\u002Ftool-selection tests\"],[\"Security\",\"Prompt injection, unauthorized tools, cross-tenant negative tests\"],[\"Behavioral evals\",\"Task success, correctness, grounding, safety, domain criteria\"],[\"Operational\",\"Latency, token\u002Fcost budgets, timeout\u002Ffallback behavior\"]]},\"tunes\":{}},{\"id\":\"h-cd\",\"type\":\"header\",\"data\":{\"text\":\"What CD looks like in LLMOps\",\"level\":2},\"tunes\":{}},{\"id\":\"p-cd-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A production release may deploy no new model artifact at all. It may simply ship a new prompt, retrieval configuration, tool set or provider mapping.\"},\"tunes\":{}},{\"id\":\"p-cd-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The release bundle should therefore identify the complete behavior-defining configuration rather than only the application container image.\"},\"tunes\":{}},{\"id\":\"p-cd-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"Feature flags, staged rollout, shadow evaluation, canary traffic and rollback are useful because LLM behavior can regress in ways that static contract tests do not detect.\"},\"tunes\":{}},{\"id\":\"h-ct\",\"type\":\"header\",\"data\":{\"text\":\"Continuous training becomes optional; continuous evaluation becomes central\",\"level\":2},\"tunes\":{}},{\"id\":\"p-ct-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Traditional MLOps often emphasizes continuous training when new data or drift justifies retraining.\"},\"tunes\":{}},{\"id\":\"p-ct-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Many LLM applications never train the foundation model. Their equivalent continuous loop is continuous evaluation: collect failures and representative production cases, add them to evaluation datasets, test candidate prompt\u002Fmodel\u002Fretrieval changes and redeploy only when evidence improves.\"},\"tunes\":{}},{\"id\":\"p-ct-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"Fine-tuning can reintroduce a training lifecycle, but it should sit inside the same broader evaluation and release process.\"},\"tunes\":{}},{\"id\":\"h-monitor\",\"type\":\"header\",\"data\":{\"text\":\"What should be monitored in production?\",\"level\":2},\"tunes\":{}},{\"id\":\"monitor-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Signal class\",\"Examples\"],[\"System health\",\"Errors, timeouts, endpoint availability\"],[\"Model\u002Fprovider\",\"Model ID, snapshot, rate limits, provider errors\"],[\"Latency\",\"End-to-end, model, retrieval, tool and reranker spans\"],[\"Cost\",\"Input\u002Foutput tokens, embeddings, tool\u002FAPI spend\"],[\"Quality\",\"Sampled task success, correctness, relevance, groundedness\"],[\"RAG\",\"Retrieval recall proxies, empty retrieval, stale sources, citation coverage\"],[\"Agents\",\"Tool selection, retries, loops, handoffs, approval frequency\"],[\"Security\",\"Denied actions, prompt-injection indicators, tenant-boundary failures\"],[\"User feedback\",\"Corrections, abandonment, escalation, explicit ratings\"],[\"Change drift\",\"Provider\u002Fmodel\u002Fconfig changes relative to approved release\"]]},\"tunes\":{}},{\"id\":\"h-prod-eval\",\"type\":\"header\",\"data\":{\"text\":\"Production traces can become evaluation data\",\"level\":2},\"tunes\":{}},{\"id\":\"p-prod-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"One of the most useful modern LLMOps patterns is to turn sampled production traces into evaluation records.\"},\"tunes\":{}},{\"id\":\"p-prod-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"MLflow currently supports retrieving production traces and scoring not only outputs but intermediate spans such as retrieval or tool-call trajectories.\"},\"tunes\":{}},{\"id\":\"p-prod-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"This closes the loop between observability and development: real failures can become regression cases in the next release rather than disappear inside logs.\"},\"tunes\":{}},{\"id\":\"h-repro\",\"type\":\"header\",\"data\":{\"text\":\"Reproducibility becomes conditional rather than exact\",\"level\":2},\"tunes\":{}},{\"id\":\"p-repro-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Classical ML reproducibility often aims to recreate a model from versioned code, data, environment and training parameters.\"},\"tunes\":{}},{\"id\":\"p-repro-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Hosted LLM applications cannot always reproduce identical output token-for-token because generation is probabilistic and providers may control infrastructure.\"},\"tunes\":{}},{\"id\":\"p-repro-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"LLMOps therefore aims for behavioral reproducibility: record enough model\u002Fprovider\u002Fversion, prompt, context inputs, retrieval state and runtime configuration to reproduce the conditions and validate behavior within expected tolerances.\"},\"tunes\":{}},{\"id\":\"h-lineage\",\"type\":\"header\",\"data\":{\"text\":\"Lineage expands from model lineage to application lineage\",\"level\":2},\"tunes\":{}},{\"id\":\"p-lineage-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"AWS's MLOps guidance treats model lineage as the history of code, data, model and infrastructure artifacts needed for diagnosis and reproducibility.\"},\"tunes\":{}},{\"id\":\"p-lineage-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"For LLM applications, lineage should additionally connect prompts, eval datasets, retrieval\u002Findex versions, tool schemas, agent\u002Fruntime configuration and provider\u002Fmodel snapshots.\"},\"tunes\":{}},{\"id\":\"p-lineage-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The target question becomes: Which exact application configuration produced this trace?\"},\"tunes\":{}},{\"id\":\"h-routing\",\"type\":\"header\",\"data\":{\"text\":\"Multi-provider and model routing create operational policy\",\"level\":2},\"tunes\":{}},{\"id\":\"p-route-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Once an application can use several providers or local models, routing becomes an operational policy rather than a simple model string.\"},\"tunes\":{}},{\"id\":\"p-route-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Routing may depend on capability, latency, cost, privacy, context length, availability, tool support or locality. A fallback can preserve uptime while changing answer quality or data-processing assumptions.\"},\"tunes\":{}},{\"id\":\"p-route-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"LLMOps should therefore log which route was actually selected and evaluate routes independently rather than treat every compatible endpoint as behaviorally interchangeable.\"},\"tunes\":{}},{\"id\":\"h-implementation\",\"type\":\"header\",\"data\":{\"text\":\"Original implementation evidence\",\"level\":2},\"tunes\":{}},{\"id\":\"h-client\",\"type\":\"header\",\"data\":{\"text\":\"Aaasaasa AI Client: provider, model and runtime are separate operational objects\",\"level\":3},\"tunes\":{}},{\"id\":\"p-client-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Aaasaasa AI Client separates agent\u002Fclient, provider, model, runtime location and permissions. Its AI Hub supports Ollama, LM Studio\u002FOpenAI-compatible endpoints and other provider protocols rather than treating “the model” as one global setting.\"},\"tunes\":{}},{\"id\":\"p-client-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The implementation includes dynamic local model discovery, streaming, thinking output and explicit Ollama warm\u002Fload and unload controls. That is operational evidence that local LLM serving introduces resource lifecycle concerns beyond an API model name.\"},\"tunes\":{}},{\"id\":\"p-client-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"Provider status is queried through provider adapters, and connection types distinguish local, cloud API, account-backed, remote-agent and web-client paths. These are concrete operational dimensions an LLM-aware platform has to surface.\"},\"tunes\":{}},{\"id\":\"p-client-4\",\"type\":\"paragraph\",\"data\":{\"text\":\"The repository also preserves an important boundary: a local runtime is not automatically local inference. Provider\u002Fmodel\u002Fruntime location are versioned or configurable concerns that affect privacy, latency, cost and availability.\"},\"tunes\":{}},{\"id\":\"h-sot\",\"type\":\"header\",\"data\":{\"text\":\"Source of Truth Research Engine: LLM application state extends beyond the model\",\"level\":3},\"tunes\":{}},{\"id\":\"p-sot-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The Source of Truth Research Engine combines lexical search, optional embeddings, source snapshots, SHA-256 identity, claims, provenance and contradiction tracking around local model-assisted research.\"},\"tunes\":{}},{\"id\":\"p-sot-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"This is useful LLMOps evidence because changing the model alone does not define the research system. Retrieval, source acquisition, evidence classification and persistent provenance are independent operational artifacts.\"},\"tunes\":{}},{\"id\":\"p-sot-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The implementation deliberately treats semantic similarity as discovery rather than evidence, showing why LLMOps observability should distinguish retrieval behavior from claim validity.\"},\"tunes\":{}},{\"id\":\"impl-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Observed implementation\",\"LLMOps lesson\"],[\"Multiple provider protocols\",\"Provider identity is an operational dependency\"],[\"Dynamic model discovery\",\"Available models can change independently of application code\"],[\"Ollama load\u002Funload controls\",\"Local models have memory\u002Fresource lifecycle\"],[\"Provider health\u002Fstatus adapters\",\"Model availability needs runtime observability\"],[\"Separate runtime and inference location\",\"Deployment topology is not one boolean “local\u002Fcloud”\"],[\"Central permissions\",\"Model capability and tool authority must remain separate\"],[\"Lexical + semantic retrieval pipeline\",\"Retrieval configuration is part of application behavior\"],[\"Source\u002Fprovenance persistence\",\"Operational state and evidence live outside model weights\"]]},\"tunes\":{}},{\"id\":\"impl-boundary\",\"type\":\"callout\",\"data\":{\"variant\":\"note\",\"title\":\"Evidence boundary\",\"body\":\"These projects demonstrate multi-provider\u002Flocal-model operations, permission separation, retrieval infrastructure and evidence persistence. They are not presented as a complete commercial LLMOps platform or proof of large-scale production traffic.\"},\"tunes\":{}},{\"id\":\"h-failures\",\"type\":\"header\",\"data\":{\"text\":\"Common LLMOps failure modes\",\"level\":2},\"tunes\":{}},{\"id\":\"failure-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Failure mode\",\"What actually went wrong\"],[\"Model alias upgraded silently\",\"Behavior changed without controlled release\"],[\"Prompt changed without evals\",\"Behavioral regression passed normal unit tests\"],[\"RAG index stale\",\"Generation model was blamed for retrieval\u002Fdata failure\"],[\"Only final answer is logged\",\"Root cause in retrieval\u002Ftool\u002Fcontext trajectory is invisible\"],[\"Provider fallback is silent\",\"Different model\u002Fdata path changes behavior without attribution\"],[\"Token cost tracked globally\",\"Expensive workflows cannot be localized\"],[\"Judge model changed\",\"Evaluation scores drift without application change\"],[\"Production traces never become tests\",\"Known failures repeatedly return\"],[\"Local model stays loaded indefinitely\",\"VRAM\u002Fresource pressure becomes operational instability\"],[\"Permissions encoded only in prompt\",\"Model behavior is mistaken for authorization\"],[\"One eval score gates everything\",\"Different quality dimensions are collapsed into a misleading number\"],[\"Model registry exists but prompt\u002Findex versions do not\",\"Application lineage remains incomplete\"]]},\"tunes\":{}},{\"id\":\"h-misconceptions\",\"type\":\"header\",\"data\":{\"text\":\"Common misconceptions\",\"level\":2},\"tunes\":{}},{\"id\":\"misconceptions-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Misconception\",\"Correction\"],[\"“LLMOps replaces MLOps.”\",\"LLMOps extends MLOps principles to LLM-specific application behavior.\"],[\"“LLMOps is prompt engineering.”\",\"Prompts are one artifact among models, providers, context, retrieval, tools, evals and runtime.\"],[\"“Hosted APIs remove operations work.”\",\"They remove some model-serving\u002Ftraining work but add provider lifecycle, version and dependency management.\"],[\"“If the API is stable, the app is stable.”\",\"Model behavior and provider\u002Fmodel snapshots can change independently of API schema.\"],[\"“RAG is just data preprocessing.”\",\"In production it has its own ingestion, index, retrieval and freshness lifecycle.\"],[\"“LLM outputs cannot be tested.”\",\"They can be evaluated with deterministic, reference, judge and human criteria.\"],[\"“LLM judges are objective ground truth.”\",\"They are model-based evaluators that also require calibration and version control.\"],[\"“A local model eliminates LLMOps.”\",\"Local serving adds model files, VRAM, load\u002Funload, runtime health and upgrade concerns.\"],[\"“Observability means token counts.”\",\"Useful observability follows prompts, retrievals, tools, model spans and outcomes.\"],[\"“Continuous training is mandatory.”\",\"Many LLM apps use continuous evaluation without training the foundation model.\"]]},\"tunes\":{}},{\"id\":\"h-design\",\"type\":\"header\",\"data\":{\"text\":\"A practical LLMOps design sequence\",\"level\":2},\"tunes\":{}},{\"id\":\"design-flow\",\"type\":\"processFlow\",\"data\":{\"title\":\"Operate the complete behavior-producing system\",\"orientation\":\"auto\",\"steps\":[{\"label\":\"1. Define the behavior unit\",\"description\":\"List every component that can materially change output: model, prompt, retrieval, tools, context and policy.\"},{\"label\":\"2. Establish application lineage\",\"description\":\"Version code, model\u002Fprovider, prompts, eval datasets, retrieval configuration and tool contracts.\"},{\"label\":\"3. Build representative eval datasets\",\"description\":\"Use expected success\u002Ffailure cases from design and production.\"},{\"label\":\"4. Separate deterministic and behavioral tests\",\"description\":\"Keep schema\u002Fsecurity assertions distinct from semantic output evaluation.\"},{\"label\":\"5. Trace end-to-end execution\",\"description\":\"Instrument model, retrieval, reranking, tools and agent\u002Fruntime spans.\"},{\"label\":\"6. Define release gates\",\"description\":\"Set quality, safety, latency and cost thresholds.\"},{\"label\":\"7. Pin or explicitly record model versions\",\"description\":\"Treat model\u002Fprovider changes as release events.\"},{\"label\":\"8. Deploy progressively\",\"description\":\"Use flags, canaries or staged rollout where consequence warrants it.\"},{\"label\":\"9. Evaluate production traces\",\"description\":\"Measure real task behavior and identify recurrent failures.\"},{\"label\":\"10. Feed failures back into eval datasets\",\"description\":\"Turn incidents and corrections into permanent regression coverage.\"},{\"label\":\"11. Monitor provider and data lifecycles\",\"description\":\"Track deprecations, index freshness, source changes and runtime availability.\"},{\"label\":\"12. Retire obsolete versions cleanly\",\"description\":\"Remove old prompts\u002Fmodels\u002Findexes\u002Fcredentials after migration and evidence retention decisions.\"}]},\"tunes\":{}},{\"id\":\"h-checklist\",\"type\":\"header\",\"data\":{\"text\":\"LLMOps architecture checklist\",\"level\":2},\"tunes\":{}},{\"id\":\"checklist-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Question\",\"Expected evidence\"],[\"Which model\u002Fprovider\u002Fversion served the request?\",\"Traceable model identity\"],[\"Which prompt\u002Finstructions were active?\",\"Versioned application code\u002Fconfig\"],[\"Which context reached the model?\",\"Context\u002Fretrieval trace\"],[\"Which corpus\u002Findex version was used?\",\"Retrieval lineage\"],[\"Which tools were available and called?\",\"Tool schema + trajectory trace\"],[\"Which permissions applied?\",\"Runtime authorization record\"],[\"How is quality measured?\",\"Versioned eval dataset + scorers\"],[\"How are model upgrades tested?\",\"Behavioral regression suite\"],[\"How is production quality sampled?\",\"Trace evaluation\u002Ffeedback process\"],[\"Can one failure be reproduced approximately?\",\"Model\u002Fcontext\u002Fprovider\u002Fapplication lineage\"],[\"Where is cost spent?\",\"Per-trace model\u002Ftool\u002Fretrieval attribution\"],[\"What triggers rollback?\",\"Defined quality\u002Fsafety\u002Fcost\u002Favailability threshold\"],[\"How are provider deprecations handled?\",\"Migration\u002Ffallback process\"],[\"How are local models operated?\",\"Health, resource, load\u002Funload and version controls\"]]},\"tunes\":{}},{\"id\":\"h-edge\",\"type\":\"header\",\"data\":{\"text\":\"Edge cases and limitations\",\"level\":2},\"tunes\":{}},{\"id\":\"p-edge-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A simple application that calls one fixed hosted model with no retrieval or tools may need only lightweight LLMOps: versioned prompt code, evals, model pinning, basic tracing and provider monitoring.\"},\"tunes\":{}},{\"id\":\"p-edge-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"A self-hosted fine-tuned model may require nearly the full classical MLOps stack plus LLM-specific application evaluation, making the boundary between MLOps and LLMOps intentionally blurry.\"},\"tunes\":{}},{\"id\":\"p-edge-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"An agent platform can have minimal model-training operations but substantial runtime operations because failures occur in tool selection, state and orchestration.\"},\"tunes\":{}},{\"id\":\"p-edge-4\",\"type\":\"paragraph\",\"data\":{\"text\":\"A RAG-heavy system can be operationally dominated by document ingestion and retrieval quality rather than model serving.\"},\"tunes\":{}},{\"id\":\"p-edge-5\",\"type\":\"paragraph\",\"data\":{\"text\":\"Terminology will continue to evolve. The durable architecture question is not which “Ops” label wins, but which artifacts produce behavior and therefore must be versioned, evaluated, observed and governed.\"},\"tunes\":{}},{\"id\":\"h-change\",\"type\":\"header\",\"data\":{\"text\":\"What would change this answer?\",\"level\":2},\"tunes\":{}},{\"id\":\"p-change-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"If foundation-model providers standardize perfectly stable model behavior and long-term version support, provider\u002Fsnapshot management could become less operationally significant.\"},\"tunes\":{}},{\"id\":\"p-change-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"If applications increasingly own fine-tuning or training, classical MLOps concerns become more central again.\"},\"tunes\":{}},{\"id\":\"p-change-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The operational principle would remain: every component that can materially change production behavior belongs in lineage, testing, observability and change control.\"},\"tunes\":{}},{\"id\":\"h-related\",\"type\":\"header\",\"data\":{\"text\":\"Related canonical knowledge\",\"level\":2},\"tunes\":{}},{\"id\":\"p-related-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"LLMOps sits below AI Governance and Enterprise AI Architecture: governance defines which changes require evidence and approval, while LLMOps provides the operational machinery to version, evaluate, deploy and observe those changes.\"},\"tunes\":{}},{\"id\":\"p-related-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Context Engineering and RAG are operational subdomains inside many LLM applications because context and retrieval can change behavior independently of the model.\"},\"tunes\":{}},{\"id\":\"p-related-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"Agentic AI extends LLMOps further into trajectory, permissions and tool-runtime operations.\"},\"tunes\":{}},{\"id\":\"ref-memory\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context\",\"title\":\"AI Agent Memory Is Not RAG: How to Separate Memory, Retrieval, State and Context\",\"excerpt\":\"Operational reliability improves when memory, retrieval, application state and model context remain separate lifecycle objects.\",\"ctaLabel\":\"Read the architecture article\"},\"tunes\":{}},{\"id\":\"ref-avb\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers\",\"title\":\"The Answer Validity Boundary: The Missing Layer Between Relevance and Reliable AI Answers\",\"excerpt\":\"LLMOps evaluation should preserve the version, scope and evidence conditions under which an answer remains supported.\",\"ctaLabel\":\"Read the Answer Validity Boundary\"},\"tunes\":{}},{\"id\":\"h-faq\",\"type\":\"header\",\"data\":{\"text\":\"Frequently asked questions\",\"level\":2},\"tunes\":{}},{\"id\":\"faq\",\"type\":\"faq\",\"data\":{\"title\":\"MLOps vs LLMOps FAQ\",\"items\":[{\"id\":\"faq1\",\"question\":\"What is the difference between MLOps and LLMOps?\",\"answer\":\"MLOps operates machine-learning systems across data, training, deployment and monitoring. LLMOps extends those practices to LLM applications where prompts, context, retrieval, providers, tools and evaluations also materially affect behavior.\"},{\"id\":\"faq2\",\"question\":\"Does LLMOps replace MLOps?\",\"answer\":\"No. LLMOps reuses MLOps disciplines such as CI\u002FCD, lineage, evaluation, deployment and monitoring and adds LLM-specific operational concerns.\"},{\"id\":\"faq3\",\"question\":\"Do LLM applications need continuous training?\",\"answer\":\"Not necessarily. Many use external foundation models and instead rely on continuous evaluation of prompts, models, retrieval and application behavior. Fine-tuned or self-trained systems can still require training pipelines.\"},{\"id\":\"faq4\",\"question\":\"Why are evals so important in LLMOps?\",\"answer\":\"Generative outputs are open-ended and model behavior can change across prompts, snapshots and context. Evals provide repeatable evidence that a release still meets defined quality and safety criteria.\"},{\"id\":\"faq5\",\"question\":\"What should be versioned in LLMOps?\",\"answer\":\"At minimum: application code, model\u002Fprovider\u002Fversion, prompts, eval datasets\u002Fscorers, retrieval configuration\u002Findexes, tool schemas, context rules and relevant safety\u002Fpermission configuration.\"},{\"id\":\"faq6\",\"question\":\"Is prompt versioning enough?\",\"answer\":\"No. The same prompt can behave differently with another model, retrieval set, context order, tool surface or provider.\"},{\"id\":\"faq7\",\"question\":\"What is GenAIOps?\",\"answer\":\"GenAIOps is another industry term for operating generative-AI applications. Some vendors use it interchangeably or as a broader label than LLMOps.\"},{\"id\":\"faq8\",\"question\":\"How do you monitor an LLM application?\",\"answer\":\"Monitor end-to-end traces including model calls, prompts\u002Fcontext, retrieval, tools, latency, token\u002Fcost, quality samples, safety and final task outcomes.\"},{\"id\":\"faq9\",\"question\":\"Can local LLMs use LLMOps practices?\",\"answer\":\"Yes. Local models add their own operational concerns such as model files, hardware\u002FVRAM, load\u002Funload, runtime health, quantization and upgrade management.\"}]},\"tunes\":{}},{\"id\":\"h-glossary\",\"type\":\"header\",\"data\":{\"text\":\"Glossary\",\"level\":2},\"tunes\":{}},{\"id\":\"glossary\",\"type\":\"glossary\",\"data\":{\"title\":\"Key MLOps and LLMOps terms\",\"entries\":[{\"term\":\"MLOps\",\"definition\":\"Engineering practices for building, deploying, monitoring and maintaining machine-learning systems and their data\u002Fmodel lifecycle.\",\"anchor\":\"mlops\"},{\"term\":\"LLMOps\",\"definition\":\"Operational practices for production applications whose behavior materially depends on large language models and surrounding prompts, context, retrieval, tools and runtime.\",\"anchor\":\"llmops\"},{\"term\":\"GenAIOps\",\"definition\":\"Operational discipline for generative-AI applications; often used as a broader or alternate label for LLMOps.\",\"anchor\":\"genaiops\"},{\"term\":\"Continuous training\",\"definition\":\"Automated or repeated retraining and serving of ML models as data or implementations change.\",\"anchor\":\"continuous-training\"},{\"term\":\"Continuous evaluation\",\"definition\":\"Repeated evaluation of candidate and production AI behavior against versioned datasets and criteria.\",\"anchor\":\"continuous-evaluation\"},{\"term\":\"Model snapshot\",\"definition\":\"A concrete version of a hosted or packaged model whose behavior can be tested and referenced.\",\"anchor\":\"model-snapshot\"},{\"term\":\"Application lineage\",\"definition\":\"Traceable relationship among code, model\u002Fprovider, prompts, data\u002Fretrieval, tools, runtime and release configuration.\",\"anchor\":\"application-lineage\"},{\"term\":\"Trace\",\"definition\":\"Structured record of one application execution containing spans such as model calls, retrievals and tool operations.\",\"anchor\":\"trace\"},{\"term\":\"Eval dataset\",\"definition\":\"Versioned set of representative inputs, expectations and optionally traces\u002Foutputs used to measure behavior.\",\"anchor\":\"eval-dataset\"},{\"term\":\"LLM judge\",\"definition\":\"A language model used as an evaluator for qualitative or semantic criteria; it is itself a versioned evaluation dependency.\",\"anchor\":\"llm-judge\"},{\"term\":\"Behavioral regression\",\"definition\":\"A degradation in application output or trajectory despite interfaces and code continuing to execute successfully.\",\"anchor\":\"behavioral-regression\"},{\"term\":\"Provider routing\",\"definition\":\"Policy for selecting among available model providers\u002Fendpoints according to capability, cost, latency, privacy or availability.\",\"anchor\":\"provider-routing\"}]},\"tunes\":{}},{\"id\":\"h-conclusion\",\"type\":\"header\",\"data\":{\"text\":\"Conclusion\",\"level\":2},\"tunes\":{}},{\"id\":\"p-conclusion-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"MLOps and LLMOps share the same engineering objective: make AI systems reproducible enough, testable enough and observable enough to operate reliably in production.\"},\"tunes\":{}},{\"id\":\"p-conclusion-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The difference is the shape of the system. Classical MLOps often centers on training and serving model artifacts; LLMOps must operate a behavioral stack in which model snapshots, prompts, context, retrieval, tools, permissions and providers can change independently.\"},\"tunes\":{}},{\"id\":\"p-conclusion-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The shortest useful rule is: version, evaluate and observe everything that can materially change the LLM application's behavior — not only the model.\"},\"tunes\":{}},{\"id\":\"h-sources\",\"type\":\"header\",\"data\":{\"text\":\"Primary sources and current documentation\",\"level\":2},\"tunes\":{}},{\"id\":\"p-sources-note\",\"type\":\"paragraph\",\"data\":{\"text\":\"The sources below ground the MLOps baseline and the current operational patterns for LLM and agent applications. Project sections are original implementation evidence and are intentionally narrower than claims about a complete LLMOps platform.\"},\"tunes\":{}},{\"id\":\"src-google-mlops\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdocs.cloud.google.com\u002Farchitecture\u002Fmlops-continuous-delivery-and-automation-pipelines-in-machine-learning\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Google Cloud — MLOps: Continuous delivery and automation pipelines\",\"description\":\"Reference architecture describing CI, CD, continuous training, model registry, metadata, serving and monitoring for ML systems.\"}},\"tunes\":{}},{\"id\":\"src-aws-lineage\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdocs.aws.amazon.com\u002Fwellarchitected\u002Flatest\u002Fmachine-learning-lens\u002Fmlops02-bp04.html\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"AWS Machine Learning Lens — Model lineage\",\"description\":\"Current guidance for tracking code, data, models, environments and infrastructure across ML releases.\"}},\"tunes\":{}},{\"id\":\"src-aws-monitor\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdocs.aws.amazon.com\u002Fwellarchitected\u002Flatest\u002Fmachine-learning-lens\u002Fmlops06-bp02.html\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"AWS Machine Learning Lens — Model observability and tracking\",\"description\":\"Current guidance for production model monitoring, drift, endpoint health and lineage.\"}},\"tunes\":{}},{\"id\":\"src-azure-llmops\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Flearn.microsoft.com\u002Fen-us\u002Fazure\u002Fmachine-learning\u002Fprompt-flow\u002Fhow-to-end-to-end-llmops-with-prompt-flow\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Microsoft Azure — GenAIOps \u002F LLMOps lifecycle\",\"description\":\"Official guidance describing GenAIOps, sometimes called LLMOps, across initialization, experimentation, evaluation\u002Frefinement and deployment.\"}},\"tunes\":{}},{\"id\":\"src-mlflow-genai\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fmlflow.org\u002Fdocs\u002Flatest\u002Fgenai\u002F\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"MLflow — Agents and LLM applications\",\"description\":\"Current GenAI operations documentation covering tracing, evaluation, prompts and production observability for LLM applications and agents.\"}},\"tunes\":{}},{\"id\":\"src-mlflow-traces\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fwww.mlflow.org\u002Fdocs\u002Flatest\u002Fgenai\u002Feval-monitor\u002Frunning-evaluation\u002Ftraces\u002F\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"MLflow — Evaluating production traces\",\"description\":\"Current guidance for evaluating complete LLM\u002Fagent traces, including retrieval and tool-call trajectories.\"}},\"tunes\":{}},{\"id\":\"src-mlflow-prompt-eval\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fmlflow.org\u002Fdocs\u002Flatest\u002Fgenai\u002Fprompt-registry\u002Fevaluate-prompts\u002F\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"MLflow — Evaluating prompts\",\"description\":\"Current prompt\u002Fmodel evaluation workflow using versioned prompts, datasets, scorers and traces.\"}},\"tunes\":{}},{\"id\":\"src-openai-api\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Freference\u002Foverview\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI API — Versioning and model snapshots\",\"description\":\"Current API guidance recommending pinned model versions and evals because prompting behavior can change between snapshots.\"}},\"tunes\":{}},{\"id\":\"src-openai-prompting\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fprompting\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — Prompting\",\"description\":\"Current guidance to treat production prompts as application code, version them through source control and cover changes with tests and evaluation checks.\"}},\"tunes\":{}},{\"id\":\"src-openai-deprecations\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fdeprecations\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — Deprecations\",\"description\":\"Current provider lifecycle evidence showing model and platform-surface retirement as an operational dependency.\"}},\"tunes\":{}},{\"id\":\"src-openai-promptfoo\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdevelopers.openai.com\u002Fcookbook\u002Fexamples\u002Fevaluation\u002Fmoving-from-openai-evals-to-promptfoo\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — Moving evaluation workflows to Promptfoo\",\"description\":\"Current 2026 migration guidance illustrating why evaluation assets should remain portable as provider tooling changes.\"}},\"tunes\":{}}],\"version\":\"2.31.6\"}",{"time":1602,"blocks":1603,"version":2721},1791487321430,[1604,1608,1613,1618,1623,1628,1632,1636,1640,1644,1648,1652,1656,1660,1664,1668,1672,1676,1680,1709,1713,1717,1721,1725,1729,1761,1765,1769,1773,1777,1781,1830,1834,1838,1842,1846,1850,1854,1858,1862,1866,1870,1874,1878,1882,1886,1890,1894,1898,1902,1906,1910,1917,1921,1925,1929,1933,1938,1942,1946,1950,1954,1958,1962,1966,1970,1974,1978,1982,1986,1993,1997,2001,2005,2009,2013,2017,2021,2025,2029,2033,2037,2041,2045,2078,2082,2086,2090,2094,2098,2102,2106,2110,2114,2148,2152,2156,2160,2164,2168,2172,2176,2180,2184,2188,2192,2196,2200,2204,2208,2212,2216,2220,2224,2228,2232,2236,2240,2244,2248,2252,2283,2288,2292,2335,2339,2376,2380,2421,2425,2474,2478,2482,2486,2490,2494,2498,2502,2506,2510,2514,2518,2522,2526,2530,2537,2544,2548,2580,2584,2620,2624,2628,2632,2636,2640,2644,2651,2658,2665,2672,2679,2686,2693,2700,2707,2714],{"id":215,"data":1605,"type":218,"tunes":1607},{"text":1606},"MLOps is the engineering discipline for reliably developing, deploying, versioning and operating machine-learning systems; LLMOps extends that discipline to applications built around large language models, where production behavior depends not only on a model artifact but also on prompts, context, retrieval, provider\u002Fmodel versions, tool calls, safety controls and evaluation pipelines. LLMOps does not replace MLOps. It changes the operational unit from “a model plus serving pipeline” toward “an evolving LLM application whose behavior emerges from several independently changing components.”",{},{"id":221,"data":1609,"type":226,"tunes":1612},{"body":1610,"title":1611,"variant":225},"\u003Cstrong>MLOps operates ML systems. LLMOps operates LLM applications.\u003C\u002Fstrong>\u003Cbr>\u003Cbr>Classical MLOps commonly centers on data pipelines, training, validation, model registry, deployment, drift and retraining. LLMOps keeps those disciplines where relevant, but often adds prompt\u002Fcontext versioning, model\u002Fprovider abstraction, RAG indexes, agent\u002Ftool traces, semantic evaluations, safety tests, token\u002Fcost monitoring and regression testing across rapidly changing model snapshots.","Direct answer",{},{"id":229,"data":1614,"type":226,"tunes":1617},{"body":1615,"title":1616,"variant":233},"A production LLM application can fail even when the prompt is unchanged: the provider can change a model snapshot, a RAG corpus can become stale, a reranker can regress, tool permissions can change, context assembly can drop evidence, or an agent can take a wrong trajectory. LLMOps therefore has to observe and version the system around the model, not only prompt text.","LLMOps is not just prompt management",{},{"id":236,"data":1619,"type":226,"tunes":1622},{"body":1620,"title":1621,"variant":240},"\u003Cstrong>LLMOps\u003C\u002Fstrong>, \u003Cstrong>GenAIOps\u003C\u002Fstrong> and related terms are widely used engineering labels, but they are not one universal formal standard with a single canonical lifecycle. Microsoft currently describes GenAIOps as “sometimes called LLMOps,” while MLflow groups operational tooling around agents and LLM applications. This article uses LLMOps as a practical architecture term for operating production systems whose behavior materially depends on LLMs.","Terminology boundary",{},{"id":243,"data":1624,"type":226,"tunes":1627},{"body":1625,"title":1626,"variant":240},"The operational surface is changing quickly. OpenAI currently recommends pinning model snapshots and running evals because prompting behavior can change between snapshots, and several older platform-specific prompt\u002Feval surfaces are being retired in 2026. The stable architectural lesson is to keep prompts, tests and evals portable and versioned with the application rather than depend on one provider's dashboard object model.","Current-source note — 8 October 2026",{},{"id":249,"data":1629,"type":254,"tunes":1631},{"title":1630,"maxLevel":252,"minLevel":253},"Contents",{},{"id":257,"data":1633,"type":42,"tunes":1635},{"text":1634,"level":253},"What MLOps really means",{},{"id":262,"data":1637,"type":218,"tunes":1639},{"text":1638},"MLOps applies software-engineering and operational discipline to machine-learning systems. The production challenge is broader than training a model: data collection, data validation, experimentation, reproducibility, model evaluation, deployment, infrastructure and monitoring all have to work together.",{},{"id":267,"data":1641,"type":218,"tunes":1643},{"text":1642},"Google's MLOps architecture guidance frames the discipline around continuous integration, continuous delivery and continuous training. CI validates not only code but also data, schemas and models; CD deploys ML pipelines and prediction services; CT can retrain and redeploy models as data or implementations change.",{},{"id":272,"data":1645,"type":218,"tunes":1647},{"text":1646},"AWS guidance adds the same operational concerns from another angle: model lineage, model\u002Fversion traceability, drift monitoring and production-quality monitoring are core parts of keeping ML systems reliable after deployment.",{},{"id":277,"data":1649,"type":42,"tunes":1651},{"text":1650,"level":253},"What changes when the model is an LLM",{},{"id":282,"data":1653,"type":218,"tunes":1655},{"text":1654},"Large language models change the production problem because the application often does not own the complete model-training lifecycle. A team may call a hosted model API, run an open model locally, switch between providers or use several models for different tasks.",{},{"id":287,"data":1657,"type":218,"tunes":1659},{"text":1658},"The model is therefore only one versioned dependency inside a larger behavioral system. Prompts, retrieval results, context order, tools, model snapshot, temperature\u002Freasoning settings, safety filters and runtime orchestration can all change the output.",{},{"id":292,"data":1661,"type":218,"tunes":1663},{"text":1662},"This creates a broader operational question: which combination of model, context, data, prompt, tools and runtime produced this behavior? LLMOps exists to make that question answerable and the answer reproducible enough for engineering work.",{},{"id":297,"data":1665,"type":42,"tunes":1667},{"text":1666,"level":253},"The simplest example",{},{"id":302,"data":1669,"type":218,"tunes":1671},{"text":1670},"Suppose an application answers internal policy questions.",{},{"id":307,"data":1673,"type":218,"tunes":1675},{"text":1674},"In a classical ML framing, you might version a trained classifier, deploy it and monitor prediction quality. In an LLM application, the answer might depend on a hosted model snapshot, a system prompt, an embedding model, a vector index, retrieval filters, a reranker and the final selected context.",{},{"id":312,"data":1677,"type":218,"tunes":1679},{"text":1678},"Changing any one of those components can change the final answer even though the application endpoint and user question stay identical.",{},{"id":317,"data":1681,"type":346,"tunes":1708},{"steps":1682,"title":1707,"orientation":345},[1683,1686,1689,1692,1695,1698,1701,1704],{"label":1684,"description":1685},"1. Change one component","Prompt, model, provider, retrieval setting, tool schema or application code changes.",{"label":1687,"description":1688},"2. Run deterministic tests","Validate schemas, permissions, tool contracts, retrieval filters and application behavior.",{"label":1690,"description":1691},"3. Run behavioral evals","Compare representative outputs, retrieval quality and agent\u002Ftool trajectories against acceptance criteria.",{"label":1693,"description":1694},"4. Compare cost and latency","Measure token use, model calls, retrieval\u002Ftool overhead and response latency.",{"label":1696,"description":1697},"5. Deploy controlled version","Ship the concrete application configuration with model\u002Fprovider versions recorded.",{"label":1699,"description":1700},"6. Trace production behavior","Capture relevant model, retrieval, tool and runtime spans.",{"label":1702,"description":1703},"7. Evaluate production traces","Sample real executions for quality, grounding, safety and task success.",{"label":1705,"description":1706},"8. Roll back or iterate","Use regression evidence and operational signals to decide the next release.","A typical LLMOps release path",{},{"id":349,"data":1710,"type":42,"tunes":1712},{"text":1711,"level":253},"Where the simple example stops",{},{"id":354,"data":1714,"type":218,"tunes":1716},{"text":1715},"Some LLM systems still train or fine-tune their own models, so traditional MLOps practices such as training pipelines, model registry and data lineage remain directly relevant.",{},{"id":359,"data":1718,"type":218,"tunes":1720},{"text":1719},"Other systems use only external foundation-model APIs and never run continuous training. Their main operational workload is application evaluation, model\u002Fprovider change management, prompt\u002Fcontext versioning, retrieval quality and observability.",{},{"id":364,"data":1722,"type":218,"tunes":1724},{"text":1723},"There is therefore no single universal “LLMOps pipeline.” The exact lifecycle depends on whether you train, fine-tune, self-host, retrieve external knowledge, run agents or depend on managed model APIs.",{},{"id":369,"data":1726,"type":42,"tunes":1728},{"text":1727,"level":253},"MLOps vs LLMOps",{},{"id":374,"data":1730,"type":419,"tunes":1760},{"rows":1731,"title":1756,"layout":411,"columns":1757},[1732,1735,1738,1741,1744,1747,1750,1753],{"id":378,"label":1733,"values":1734},"Primary operational unit",[381,381],{"id":383,"label":1736,"values":1737},"Model ownership",[381,381],{"id":387,"label":1739,"values":1740},"Typical change",[381,381],{"id":391,"label":1742,"values":1743},"Evaluation",[381,381],{"id":395,"label":1745,"values":1746},"Production monitoring",[381,381],{"id":399,"label":1748,"values":1749},"Continuous training",[381,381],{"id":403,"label":1751,"values":1752},"Versioned artifacts",[381,381],{"id":407,"label":1754,"values":1755},"Rollback target",[381,381],"What stays the same and what expands",[1758,1759],{"id":414,"label":415},{"id":417,"label":418},{},{"id":422,"data":1762,"type":42,"tunes":1764},{"text":1763,"level":253},"LLMOps extends MLOps rather than replacing it",{},{"id":427,"data":1766,"type":218,"tunes":1768},{"text":1767},"The core operational principles do not disappear: source control, CI\u002FCD, reproducibility, lineage, deployment controls, monitoring, rollback and measurable acceptance criteria remain essential.",{},{"id":432,"data":1770,"type":218,"tunes":1772},{"text":1771},"The extension is that more behavior-defining artifacts now sit outside the model weights. A managed foundation model can change behavior through snapshot upgrades, while application output can change through prompt or retrieval changes without any model retraining.",{},{"id":437,"data":1774,"type":218,"tunes":1776},{"text":1775},"This is why the useful hierarchy is usually DevOps → MLOps → LLMOps\u002FGenAIOps as increasingly specialized operational concerns, not three mutually exclusive practices.",{},{"id":442,"data":1778,"type":42,"tunes":1780},{"text":1779,"level":253},"What has to be versioned in LLMOps?",{},{"id":447,"data":1782,"type":411,"tunes":1829},{"content":1783,"stretched":43,"withHeadings":14},[1784,1787,1790,1793,1796,1799,1802,1805,1808,1811,1814,1817,1820,1823,1826],[1785,1786],"Artifact","Why it matters",[1788,1789],"Application code","Defines orchestration, validation, retries and business behavior",[1791,1792],"Model family + snapshot\u002Fversion","Different snapshots can produce different behavior",[1794,1795],"Provider \u002F endpoint","Changes data flow, latency, limits, pricing and availability",[1797,1798],"Prompt\u002Finstruction code","Changes model behavior even with same model",[1800,1801],"Generation\u002Freasoning parameters","Can alter determinism, latency, depth and cost",[1803,1804],"Eval dataset","Defines what “good enough” is tested against",[1806,1807],"Scorers \u002F graders","Define how quality is measured",[1809,1810],"Embedding model","Changes vector representation and retrieval behavior",[1812,1813],"Chunking\u002Findex configuration","Changes what can be retrieved",[1815,1816],"Reranker \u002F retrieval fusion","Changes result ordering",[1818,1819],"Tool schemas","Change what the model can request and how",[1821,1822],"Permission profile","Changes what tool actions may actually execute",[1824,1825],"Context assembly rules","Change what evidence and state reach the model",[1827,1828],"Safety\u002Fguardrail configuration","Changes allowed or blocked behavior",{},{"id":497,"data":1831,"type":42,"tunes":1833},{"text":1832,"level":253},"Model snapshots become release dependencies",{},{"id":502,"data":1835,"type":218,"tunes":1837},{"text":1836},"With hosted LLMs, the team may not control model training, but it still controls which model or snapshot the application calls.",{},{"id":507,"data":1839,"type":218,"tunes":1841},{"text":1840},"OpenAI's current API guidance explicitly warns that prompting behavior can change between model snapshots and recommends pinning production applications to specific snapshots where consistency matters, then running evals when upgrading.",{},{"id":512,"data":1843,"type":218,"tunes":1845},{"text":1844},"The operational consequence is straightforward: model upgrades should be treated as application releases, not invisible infrastructure maintenance.",{},{"id":517,"data":1847,"type":42,"tunes":1849},{"text":1848,"level":253},"Provider lifecycle becomes part of operations",{},{"id":522,"data":1851,"type":218,"tunes":1853},{"text":1852},"LLM applications often depend on provider rate limits, deprecation schedules, API semantics, context limits, data-handling rules and pricing.",{},{"id":527,"data":1855,"type":218,"tunes":1857},{"text":1856},"A provider can deprecate a model while your application code remains unchanged. OpenAI's current deprecation schedule, for example, includes 2026 retirement dates for older model snapshots and platform surfaces.",{},{"id":532,"data":1859,"type":218,"tunes":1861},{"text":1860},"LLMOps therefore needs provider lifecycle tracking, migration testing and fallback decisions in addition to model-quality monitoring.",{},{"id":537,"data":1863,"type":42,"tunes":1865},{"text":1864,"level":253},"Prompts behave like production code",{},{"id":542,"data":1867,"type":218,"tunes":1869},{"text":1868},"Prompts are executable behavioral configuration. Small changes can alter output quality, tool selection and policy interpretation.",{},{"id":547,"data":1871,"type":218,"tunes":1873},{"text":1872},"OpenAI's current guidance recommends storing production prompts in application code, reviewing prompt changes through pull requests, using typed inputs and covering changes with tests and evaluation checks.",{},{"id":552,"data":1875,"type":218,"tunes":1877},{"text":1876},"That makes prompt versioning less like editing marketing copy and more like changing a function whose output is probabilistic and model-dependent.",{},{"id":557,"data":1879,"type":42,"tunes":1881},{"text":1880,"level":253},"Context engineering becomes an operational concern",{},{"id":562,"data":1883,"type":218,"tunes":1885},{"text":1884},"The production model rarely receives only a static prompt. It may receive conversation history, retrieved documents, tool outputs, memory, current application state and policy instructions.",{},{"id":567,"data":1887,"type":218,"tunes":1889},{"text":1888},"LLMOps must therefore observe context assembly: which evidence was selected, which state version was current, whether truncation occurred and whether important instructions survived compaction.",{},{"id":572,"data":1891,"type":218,"tunes":1893},{"text":1892},"A model regression and a context regression can look identical at the final answer. Tracing the actual context path is what lets the team separate them.",{},{"id":577,"data":1895,"type":42,"tunes":1897},{"text":1896,"level":253},"RAG creates its own operational lifecycle",{},{"id":582,"data":1899,"type":218,"tunes":1901},{"text":1900},"A RAG system introduces a second production pipeline beside model inference: ingestion, extraction, chunking, metadata, embeddings, indexes, retrieval, reranking and context selection.",{},{"id":587,"data":1903,"type":218,"tunes":1905},{"text":1904},"The knowledge corpus can change every day even when the model and prompt do not. A stale index or broken metadata filter can therefore degrade answer quality without any model drift.",{},{"id":592,"data":1907,"type":218,"tunes":1909},{"text":1908},"LLMOps for RAG should track corpus\u002Findex version, embedding model, chunking policy, retrieval configuration, source freshness and retrieval metrics separately from generation quality.",{},{"id":597,"data":1911,"type":603,"tunes":1916},{"url":1912,"title":1913,"excerpt":1914,"ctaLabel":1915},"https:\u002F\u002Fstajic.de\u002Fblog\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method","RAG Failed — But Which Layer Actually Failed? A Diagnostic Method","A production LLM pipeline needs separate observability for source coverage, retrieval, ranking, context assembly and generation.","Read the RAG diagnostic method",{},{"id":606,"data":1918,"type":42,"tunes":1920},{"text":1919,"level":253},"Evals replace “looks good to me” with release evidence",{},{"id":611,"data":1922,"type":218,"tunes":1924},{"text":1923},"Generative outputs are often open-ended, so exact-match tests are insufficient for many tasks. LLMOps adds evaluation datasets and scorers that can measure task success, correctness, safety, groundedness, style or domain-specific acceptance criteria.",{},{"id":616,"data":1926,"type":218,"tunes":1928},{"text":1927},"MLflow's current GenAI evaluation stack supports versioned evaluation datasets, prompt\u002Fmodel comparisons, custom scorers and evaluation over complete traces.",{},{"id":621,"data":1930,"type":218,"tunes":1932},{"text":1931},"The strongest practice is evaluation-driven development: define representative cases and acceptance criteria before or alongside changes, then compare releases against the same evidence.",{},{"id":626,"data":1934,"type":226,"tunes":1937},{"body":1935,"title":1936,"variant":630},"A deployment should not be considered equivalent merely because the API contract still works. If the prompt, model, retrieval or tools changed, the behavioral regression suite should run again.","Behavioral changes need behavioral tests",{},{"id":633,"data":1939,"type":42,"tunes":1941},{"text":1940,"level":253},"LLM-as-a-judge is useful but not ground truth",{},{"id":638,"data":1943,"type":218,"tunes":1945},{"text":1944},"LLM judges can scale evaluation for qualities that are expensive to encode as deterministic assertions, such as relevance, tone or groundedness.",{},{"id":643,"data":1947,"type":218,"tunes":1949},{"text":1948},"However, the judge is another model with its own bias, version and prompt. Judge configuration should therefore be versioned and calibrated against human or deterministic reference cases where consequence matters.",{},{"id":648,"data":1951,"type":218,"tunes":1953},{"text":1952},"A production eval can mix deterministic checks, reference-based metrics, model judges and human review rather than asking one metric to represent every quality dimension.",{},{"id":653,"data":1955,"type":42,"tunes":1957},{"text":1956,"level":253},"Tracing becomes more important than endpoint logs",{},{"id":658,"data":1959,"type":218,"tunes":1961},{"text":1960},"Traditional API logs can tell you that a request took two seconds and returned HTTP 200. They cannot tell you which retrieved chunks were selected, which tool the agent called or which model span consumed most tokens.",{},{"id":663,"data":1963,"type":218,"tunes":1965},{"text":1964},"MLflow's current GenAI tracing captures prompts, retrievals, tool calls and application spans, and its production evaluation flow can score intermediate trajectory information rather than only final text.",{},{"id":668,"data":1967,"type":218,"tunes":1969},{"text":1968},"This is a major LLMOps shift: observability follows the behavioral graph of the application, not only the serving endpoint.",{},{"id":673,"data":1971,"type":42,"tunes":1973},{"text":1972,"level":253},"Agents expand LLMOps into runtime operations",{},{"id":678,"data":1975,"type":218,"tunes":1977},{"text":1976},"An agentic application can perform several model calls, tool invocations and state transitions before producing a result.",{},{"id":683,"data":1979,"type":218,"tunes":1981},{"text":1980},"Operating agents therefore requires step counts, tool-call traces, permission denials, retries, loop detection, human approvals and verified final state in addition to ordinary model latency and token metrics.",{},{"id":688,"data":1983,"type":218,"tunes":1985},{"text":1984},"A correct final answer can hide a bad trajectory, so agent evaluation must inspect the path as well as the result.",{},{"id":693,"data":1987,"type":603,"tunes":1992},{"url":1988,"title":1989,"excerpt":1990,"ctaLabel":1991},"https:\u002F\u002Fstajic.de\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough","AI Agent Reliability: Why the Final Answer Is Not Enough","Why agent production evaluation must include tool calls, state transitions, approvals and recoverability.","Read the agent reliability article",{},{"id":701,"data":1994,"type":42,"tunes":1996},{"text":1995,"level":253},"Tokens, model calls and context become cost variables",{},{"id":706,"data":1998,"type":218,"tunes":2000},{"text":1999},"Classical ML inference cost is often dominated by serving infrastructure or per-prediction compute. LLM applications can add provider token pricing, repeated agent calls, embedding calls, reranking and tool\u002Fruntime overhead.",{},{"id":711,"data":2002,"type":218,"tunes":2004},{"text":2003},"Cost therefore has to be attributed to task or trace, not only to one endpoint. A workflow that makes eight hidden model calls can be functionally correct but operationally unacceptable.",{},{"id":716,"data":2006,"type":218,"tunes":2008},{"text":2007},"Latency behaves the same way: model latency, retrieval, reranking and external tools compose into end-to-end user latency.",{},{"id":721,"data":2010,"type":42,"tunes":2012},{"text":2011,"level":253},"Caching becomes semantic, not only technical",{},{"id":726,"data":2014,"type":218,"tunes":2016},{"text":2015},"LLM systems can cache prompts, embeddings, retrieval results or full responses, but the cache key must reflect the semantics that can change the result.",{},{"id":731,"data":2018,"type":218,"tunes":2020},{"text":2019},"A response cache that ignores model version, tenant, permissions or source freshness can return a technically valid but semantically invalid answer.",{},{"id":736,"data":2022,"type":218,"tunes":2024},{"text":2023},"LLMOps therefore treats cache invalidation as part of model\u002Fcontext\u002Fdata versioning rather than only infrastructure optimization.",{},{"id":741,"data":2026,"type":42,"tunes":2028},{"text":2027,"level":253},"Safety and permissions become release criteria",{},{"id":746,"data":2030,"type":218,"tunes":2032},{"text":2031},"Generative systems can produce unbounded text and agents can trigger external actions. Safety testing therefore sits closer to ordinary CI\u002FCD than in many classical predictive ML systems.",{},{"id":751,"data":2034,"type":218,"tunes":2036},{"text":2035},"Permission checks, prompt-injection tests, tenant-isolation tests and side-effect approvals should be reproducible regression tests where those risks exist.",{},{"id":756,"data":2038,"type":218,"tunes":2040},{"text":2039},"The model may suggest an operation, but the runtime still has to enforce authorization. LLMOps owns the evidence that those controls continue to work after model, prompt or tool changes.",{},{"id":761,"data":2042,"type":42,"tunes":2044},{"text":2043,"level":253},"What CI looks like in LLMOps",{},{"id":766,"data":2046,"type":411,"tunes":2077},{"content":2047,"stretched":43,"withHeadings":14},[2048,2051,2054,2057,2060,2062,2065,2068,2071,2074],[2049,2050],"CI layer","Example checks",[2052,2053],"Code","Unit tests, type checks, schema validation",[2055,2056],"Prompts","Template rendering, required variables, policy text, snapshot review",[2058,2059],"Models\u002Fproviders","Compatibility, output schema, capability and regression tests",[782,2061],"Chunking fixtures, filter tests, Recall@k, reranker regression",[2063,2064],"Tools","Input\u002Foutput schema tests, permission tests, idempotency tests",[2066,2067],"Agents","Trajectory fixtures, loop limits, handoff\u002Ftool-selection tests",[2069,2070],"Security","Prompt injection, unauthorized tools, cross-tenant negative tests",[2072,2073],"Behavioral evals","Task success, correctness, grounding, safety, domain criteria",[2075,2076],"Operational","Latency, token\u002Fcost budgets, timeout\u002Ffallback behavior",{},{"id":801,"data":2079,"type":42,"tunes":2081},{"text":2080,"level":253},"What CD looks like in LLMOps",{},{"id":806,"data":2083,"type":218,"tunes":2085},{"text":2084},"A production release may deploy no new model artifact at all. It may simply ship a new prompt, retrieval configuration, tool set or provider mapping.",{},{"id":811,"data":2087,"type":218,"tunes":2089},{"text":2088},"The release bundle should therefore identify the complete behavior-defining configuration rather than only the application container image.",{},{"id":816,"data":2091,"type":218,"tunes":2093},{"text":2092},"Feature flags, staged rollout, shadow evaluation, canary traffic and rollback are useful because LLM behavior can regress in ways that static contract tests do not detect.",{},{"id":821,"data":2095,"type":42,"tunes":2097},{"text":2096,"level":253},"Continuous training becomes optional; continuous evaluation becomes central",{},{"id":826,"data":2099,"type":218,"tunes":2101},{"text":2100},"Traditional MLOps often emphasizes continuous training when new data or drift justifies retraining.",{},{"id":831,"data":2103,"type":218,"tunes":2105},{"text":2104},"Many LLM applications never train the foundation model. Their equivalent continuous loop is continuous evaluation: collect failures and representative production cases, add them to evaluation datasets, test candidate prompt\u002Fmodel\u002Fretrieval changes and redeploy only when evidence improves.",{},{"id":836,"data":2107,"type":218,"tunes":2109},{"text":2108},"Fine-tuning can reintroduce a training lifecycle, but it should sit inside the same broader evaluation and release process.",{},{"id":841,"data":2111,"type":42,"tunes":2113},{"text":2112,"level":253},"What should be monitored in production?",{},{"id":846,"data":2115,"type":411,"tunes":2147},{"content":2116,"stretched":43,"withHeadings":14},[2117,2120,2123,2126,2129,2132,2135,2137,2139,2141,2144],[2118,2119],"Signal class","Examples",[2121,2122],"System health","Errors, timeouts, endpoint availability",[2124,2125],"Model\u002Fprovider","Model ID, snapshot, rate limits, provider errors",[2127,2128],"Latency","End-to-end, model, retrieval, tool and reranker spans",[2130,2131],"Cost","Input\u002Foutput tokens, embeddings, tool\u002FAPI spend",[2133,2134],"Quality","Sampled task success, correctness, relevance, groundedness",[782,2136],"Retrieval recall proxies, empty retrieval, stale sources, citation coverage",[2066,2138],"Tool selection, retries, loops, handoffs, approval frequency",[2069,2140],"Denied actions, prompt-injection indicators, tenant-boundary failures",[2142,2143],"User feedback","Corrections, abandonment, escalation, explicit ratings",[2145,2146],"Change drift","Provider\u002Fmodel\u002Fconfig changes relative to approved release",{},{"id":880,"data":2149,"type":42,"tunes":2151},{"text":2150,"level":253},"Production traces can become evaluation data",{},{"id":885,"data":2153,"type":218,"tunes":2155},{"text":2154},"One of the most useful modern LLMOps patterns is to turn sampled production traces into evaluation records.",{},{"id":890,"data":2157,"type":218,"tunes":2159},{"text":2158},"MLflow currently supports retrieving production traces and scoring not only outputs but intermediate spans such as retrieval or tool-call trajectories.",{},{"id":895,"data":2161,"type":218,"tunes":2163},{"text":2162},"This closes the loop between observability and development: real failures can become regression cases in the next release rather than disappear inside logs.",{},{"id":900,"data":2165,"type":42,"tunes":2167},{"text":2166,"level":253},"Reproducibility becomes conditional rather than exact",{},{"id":905,"data":2169,"type":218,"tunes":2171},{"text":2170},"Classical ML reproducibility often aims to recreate a model from versioned code, data, environment and training parameters.",{},{"id":910,"data":2173,"type":218,"tunes":2175},{"text":2174},"Hosted LLM applications cannot always reproduce identical output token-for-token because generation is probabilistic and providers may control infrastructure.",{},{"id":915,"data":2177,"type":218,"tunes":2179},{"text":2178},"LLMOps therefore aims for behavioral reproducibility: record enough model\u002Fprovider\u002Fversion, prompt, context inputs, retrieval state and runtime configuration to reproduce the conditions and validate behavior within expected tolerances.",{},{"id":920,"data":2181,"type":42,"tunes":2183},{"text":2182,"level":253},"Lineage expands from model lineage to application lineage",{},{"id":925,"data":2185,"type":218,"tunes":2187},{"text":2186},"AWS's MLOps guidance treats model lineage as the history of code, data, model and infrastructure artifacts needed for diagnosis and reproducibility.",{},{"id":930,"data":2189,"type":218,"tunes":2191},{"text":2190},"For LLM applications, lineage should additionally connect prompts, eval datasets, retrieval\u002Findex versions, tool schemas, agent\u002Fruntime configuration and provider\u002Fmodel snapshots.",{},{"id":935,"data":2193,"type":218,"tunes":2195},{"text":2194},"The target question becomes: Which exact application configuration produced this trace?",{},{"id":940,"data":2197,"type":42,"tunes":2199},{"text":2198,"level":253},"Multi-provider and model routing create operational policy",{},{"id":945,"data":2201,"type":218,"tunes":2203},{"text":2202},"Once an application can use several providers or local models, routing becomes an operational policy rather than a simple model string.",{},{"id":950,"data":2205,"type":218,"tunes":2207},{"text":2206},"Routing may depend on capability, latency, cost, privacy, context length, availability, tool support or locality. A fallback can preserve uptime while changing answer quality or data-processing assumptions.",{},{"id":955,"data":2209,"type":218,"tunes":2211},{"text":2210},"LLMOps should therefore log which route was actually selected and evaluate routes independently rather than treat every compatible endpoint as behaviorally interchangeable.",{},{"id":960,"data":2213,"type":42,"tunes":2215},{"text":2214,"level":253},"Original implementation evidence",{},{"id":965,"data":2217,"type":42,"tunes":2219},{"text":2218,"level":252},"Aaasaasa AI Client: provider, model and runtime are separate operational objects",{},{"id":970,"data":2221,"type":218,"tunes":2223},{"text":2222},"Aaasaasa AI Client separates agent\u002Fclient, provider, model, runtime location and permissions. Its AI Hub supports Ollama, LM Studio\u002FOpenAI-compatible endpoints and other provider protocols rather than treating “the model” as one global setting.",{},{"id":975,"data":2225,"type":218,"tunes":2227},{"text":2226},"The implementation includes dynamic local model discovery, streaming, thinking output and explicit Ollama warm\u002Fload and unload controls. That is operational evidence that local LLM serving introduces resource lifecycle concerns beyond an API model name.",{},{"id":980,"data":2229,"type":218,"tunes":2231},{"text":2230},"Provider status is queried through provider adapters, and connection types distinguish local, cloud API, account-backed, remote-agent and web-client paths. These are concrete operational dimensions an LLM-aware platform has to surface.",{},{"id":985,"data":2233,"type":218,"tunes":2235},{"text":2234},"The repository also preserves an important boundary: a local runtime is not automatically local inference. Provider\u002Fmodel\u002Fruntime location are versioned or configurable concerns that affect privacy, latency, cost and availability.",{},{"id":990,"data":2237,"type":42,"tunes":2239},{"text":2238,"level":252},"Source of Truth Research Engine: LLM application state extends beyond the model",{},{"id":995,"data":2241,"type":218,"tunes":2243},{"text":2242},"The Source of Truth Research Engine combines lexical search, optional embeddings, source snapshots, SHA-256 identity, claims, provenance and contradiction tracking around local model-assisted research.",{},{"id":1000,"data":2245,"type":218,"tunes":2247},{"text":2246},"This is useful LLMOps evidence because changing the model alone does not define the research system. Retrieval, source acquisition, evidence classification and persistent provenance are independent operational artifacts.",{},{"id":1005,"data":2249,"type":218,"tunes":2251},{"text":2250},"The implementation deliberately treats semantic similarity as discovery rather than evidence, showing why LLMOps observability should distinguish retrieval behavior from claim validity.",{},{"id":1010,"data":2253,"type":411,"tunes":2282},{"content":2254,"stretched":43,"withHeadings":14},[2255,2258,2261,2264,2267,2270,2273,2276,2279],[2256,2257],"Observed implementation","LLMOps lesson",[2259,2260],"Multiple provider protocols","Provider identity is an operational dependency",[2262,2263],"Dynamic model discovery","Available models can change independently of application code",[2265,2266],"Ollama load\u002Funload controls","Local models have memory\u002Fresource lifecycle",[2268,2269],"Provider health\u002Fstatus adapters","Model availability needs runtime observability",[2271,2272],"Separate runtime and inference location","Deployment topology is not one boolean “local\u002Fcloud”",[2274,2275],"Central permissions","Model capability and tool authority must remain separate",[2277,2278],"Lexical + semantic retrieval pipeline","Retrieval configuration is part of application behavior",[2280,2281],"Source\u002Fprovenance persistence","Operational state and evidence live outside model weights",{},{"id":1042,"data":2284,"type":226,"tunes":2287},{"body":2285,"title":2286,"variant":240},"These projects demonstrate multi-provider\u002Flocal-model operations, permission separation, retrieval infrastructure and evidence persistence. They are not presented as a complete commercial LLMOps platform or proof of large-scale production traffic.","Evidence boundary",{},{"id":1048,"data":2289,"type":42,"tunes":2291},{"text":2290,"level":253},"Common LLMOps failure modes",{},{"id":1053,"data":2293,"type":411,"tunes":2334},{"content":2294,"stretched":43,"withHeadings":14},[2295,2298,2301,2304,2307,2310,2313,2316,2319,2322,2325,2328,2331],[2296,2297],"Failure mode","What actually went wrong",[2299,2300],"Model alias upgraded silently","Behavior changed without controlled release",[2302,2303],"Prompt changed without evals","Behavioral regression passed normal unit tests",[2305,2306],"RAG index stale","Generation model was blamed for retrieval\u002Fdata failure",[2308,2309],"Only final answer is logged","Root cause in retrieval\u002Ftool\u002Fcontext trajectory is invisible",[2311,2312],"Provider fallback is silent","Different model\u002Fdata path changes behavior without attribution",[2314,2315],"Token cost tracked globally","Expensive workflows cannot be localized",[2317,2318],"Judge model changed","Evaluation scores drift without application change",[2320,2321],"Production traces never become tests","Known failures repeatedly return",[2323,2324],"Local model stays loaded indefinitely","VRAM\u002Fresource pressure becomes operational instability",[2326,2327],"Permissions encoded only in prompt","Model behavior is mistaken for authorization",[2329,2330],"One eval score gates everything","Different quality dimensions are collapsed into a misleading number",[2332,2333],"Model registry exists but prompt\u002Findex versions do not","Application lineage remains incomplete",{},{"id":1097,"data":2336,"type":42,"tunes":2338},{"text":2337,"level":253},"Common misconceptions",{},{"id":1102,"data":2340,"type":411,"tunes":2375},{"content":2341,"stretched":43,"withHeadings":14},[2342,2345,2348,2351,2354,2357,2360,2363,2366,2369,2372],[2343,2344],"Misconception","Correction",[2346,2347],"“LLMOps replaces MLOps.”","LLMOps extends MLOps principles to LLM-specific application behavior.",[2349,2350],"“LLMOps is prompt engineering.”","Prompts are one artifact among models, providers, context, retrieval, tools, evals and runtime.",[2352,2353],"“Hosted APIs remove operations work.”","They remove some model-serving\u002Ftraining work but add provider lifecycle, version and dependency management.",[2355,2356],"“If the API is stable, the app is stable.”","Model behavior and provider\u002Fmodel snapshots can change independently of API schema.",[2358,2359],"“RAG is just data preprocessing.”","In production it has its own ingestion, index, retrieval and freshness lifecycle.",[2361,2362],"“LLM outputs cannot be tested.”","They can be evaluated with deterministic, reference, judge and human criteria.",[2364,2365],"“LLM judges are objective ground truth.”","They are model-based evaluators that also require calibration and version control.",[2367,2368],"“A local model eliminates LLMOps.”","Local serving adds model files, VRAM, load\u002Funload, runtime health and upgrade concerns.",[2370,2371],"“Observability means token counts.”","Useful observability follows prompts, retrievals, tools, model spans and outcomes.",[2373,2374],"“Continuous training is mandatory.”","Many LLM apps use continuous evaluation without training the foundation model.",{},{"id":1140,"data":2377,"type":42,"tunes":2379},{"text":2378,"level":253},"A practical LLMOps design sequence",{},{"id":1145,"data":2381,"type":346,"tunes":2420},{"steps":2382,"title":2419,"orientation":345},[2383,2386,2389,2392,2395,2398,2401,2404,2407,2410,2413,2416],{"label":2384,"description":2385},"1. Define the behavior unit","List every component that can materially change output: model, prompt, retrieval, tools, context and policy.",{"label":2387,"description":2388},"2. Establish application lineage","Version code, model\u002Fprovider, prompts, eval datasets, retrieval configuration and tool contracts.",{"label":2390,"description":2391},"3. Build representative eval datasets","Use expected success\u002Ffailure cases from design and production.",{"label":2393,"description":2394},"4. Separate deterministic and behavioral tests","Keep schema\u002Fsecurity assertions distinct from semantic output evaluation.",{"label":2396,"description":2397},"5. Trace end-to-end execution","Instrument model, retrieval, reranking, tools and agent\u002Fruntime spans.",{"label":2399,"description":2400},"6. Define release gates","Set quality, safety, latency and cost thresholds.",{"label":2402,"description":2403},"7. Pin or explicitly record model versions","Treat model\u002Fprovider changes as release events.",{"label":2405,"description":2406},"8. Deploy progressively","Use flags, canaries or staged rollout where consequence warrants it.",{"label":2408,"description":2409},"9. Evaluate production traces","Measure real task behavior and identify recurrent failures.",{"label":2411,"description":2412},"10. Feed failures back into eval datasets","Turn incidents and corrections into permanent regression coverage.",{"label":2414,"description":2415},"11. Monitor provider and data lifecycles","Track deprecations, index freshness, source changes and runtime availability.",{"label":2417,"description":2418},"12. Retire obsolete versions cleanly","Remove old prompts\u002Fmodels\u002Findexes\u002Fcredentials after migration and evidence retention decisions.","Operate the complete behavior-producing system",{},{"id":1187,"data":2422,"type":42,"tunes":2424},{"text":2423,"level":253},"LLMOps architecture checklist",{},{"id":1192,"data":2426,"type":411,"tunes":2473},{"content":2427,"stretched":43,"withHeadings":14},[2428,2431,2434,2437,2440,2443,2446,2449,2452,2455,2458,2461,2464,2467,2470],[2429,2430],"Question","Expected evidence",[2432,2433],"Which model\u002Fprovider\u002Fversion served the request?","Traceable model identity",[2435,2436],"Which prompt\u002Finstructions were active?","Versioned application code\u002Fconfig",[2438,2439],"Which context reached the model?","Context\u002Fretrieval trace",[2441,2442],"Which corpus\u002Findex version was used?","Retrieval lineage",[2444,2445],"Which tools were available and called?","Tool schema + trajectory trace",[2447,2448],"Which permissions applied?","Runtime authorization record",[2450,2451],"How is quality measured?","Versioned eval dataset + scorers",[2453,2454],"How are model upgrades tested?","Behavioral regression suite",[2456,2457],"How is production quality sampled?","Trace evaluation\u002Ffeedback process",[2459,2460],"Can one failure be reproduced approximately?","Model\u002Fcontext\u002Fprovider\u002Fapplication lineage",[2462,2463],"Where is cost spent?","Per-trace model\u002Ftool\u002Fretrieval attribution",[2465,2466],"What triggers rollback?","Defined quality\u002Fsafety\u002Fcost\u002Favailability threshold",[2468,2469],"How are provider deprecations handled?","Migration\u002Ffallback process",[2471,2472],"How are local models operated?","Health, resource, load\u002Funload and version controls",{},{"id":1242,"data":2475,"type":42,"tunes":2477},{"text":2476,"level":253},"Edge cases and limitations",{},{"id":1247,"data":2479,"type":218,"tunes":2481},{"text":2480},"A simple application that calls one fixed hosted model with no retrieval or tools may need only lightweight LLMOps: versioned prompt code, evals, model pinning, basic tracing and provider monitoring.",{},{"id":1252,"data":2483,"type":218,"tunes":2485},{"text":2484},"A self-hosted fine-tuned model may require nearly the full classical MLOps stack plus LLM-specific application evaluation, making the boundary between MLOps and LLMOps intentionally blurry.",{},{"id":1257,"data":2487,"type":218,"tunes":2489},{"text":2488},"An agent platform can have minimal model-training operations but substantial runtime operations because failures occur in tool selection, state and orchestration.",{},{"id":1262,"data":2491,"type":218,"tunes":2493},{"text":2492},"A RAG-heavy system can be operationally dominated by document ingestion and retrieval quality rather than model serving.",{},{"id":1267,"data":2495,"type":218,"tunes":2497},{"text":2496},"Terminology will continue to evolve. The durable architecture question is not which “Ops” label wins, but which artifacts produce behavior and therefore must be versioned, evaluated, observed and governed.",{},{"id":1272,"data":2499,"type":42,"tunes":2501},{"text":2500,"level":253},"What would change this answer?",{},{"id":1277,"data":2503,"type":218,"tunes":2505},{"text":2504},"If foundation-model providers standardize perfectly stable model behavior and long-term version support, provider\u002Fsnapshot management could become less operationally significant.",{},{"id":1282,"data":2507,"type":218,"tunes":2509},{"text":2508},"If applications increasingly own fine-tuning or training, classical MLOps concerns become more central again.",{},{"id":1287,"data":2511,"type":218,"tunes":2513},{"text":2512},"The operational principle would remain: every component that can materially change production behavior belongs in lineage, testing, observability and change control.",{},{"id":1292,"data":2515,"type":42,"tunes":2517},{"text":2516,"level":253},"Related canonical knowledge",{},{"id":1297,"data":2519,"type":218,"tunes":2521},{"text":2520},"LLMOps sits below AI Governance and Enterprise AI Architecture: governance defines which changes require evidence and approval, while LLMOps provides the operational machinery to version, evaluate, deploy and observe those changes.",{},{"id":1302,"data":2523,"type":218,"tunes":2525},{"text":2524},"Context Engineering and RAG are operational subdomains inside many LLM applications because context and retrieval can change behavior independently of the model.",{},{"id":1307,"data":2527,"type":218,"tunes":2529},{"text":2528},"Agentic AI extends LLMOps further into trajectory, permissions and tool-runtime operations.",{},{"id":1312,"data":2531,"type":603,"tunes":2536},{"url":2532,"title":2533,"excerpt":2534,"ctaLabel":2535},"https:\u002F\u002Fstajic.de\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context","AI Agent Memory Is Not RAG: How to Separate Memory, Retrieval, State and Context","Operational reliability improves when memory, retrieval, application state and model context remain separate lifecycle objects.","Read the architecture article",{},{"id":1320,"data":2538,"type":603,"tunes":2543},{"url":2539,"title":2540,"excerpt":2541,"ctaLabel":2542},"https:\u002F\u002Fstajic.de\u002Fblog\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers","The Answer Validity Boundary: The Missing Layer Between Relevance and Reliable AI Answers","LLMOps evaluation should preserve the version, scope and evidence conditions under which an answer remains supported.","Read the Answer Validity Boundary",{},{"id":1328,"data":2545,"type":42,"tunes":2547},{"text":2546,"level":253},"Frequently asked questions",{},{"id":1333,"data":2549,"type":1333,"tunes":2579},{"items":2550,"title":2578},[2551,2554,2557,2560,2563,2566,2569,2572,2575],{"id":1337,"answer":2552,"question":2553},"MLOps operates machine-learning systems across data, training, deployment and monitoring. LLMOps extends those practices to LLM applications where prompts, context, retrieval, providers, tools and evaluations also materially affect behavior.","What is the difference between MLOps and LLMOps?",{"id":1341,"answer":2555,"question":2556},"No. LLMOps reuses MLOps disciplines such as CI\u002FCD, lineage, evaluation, deployment and monitoring and adds LLM-specific operational concerns.","Does LLMOps replace MLOps?",{"id":1345,"answer":2558,"question":2559},"Not necessarily. Many use external foundation models and instead rely on continuous evaluation of prompts, models, retrieval and application behavior. Fine-tuned or self-trained systems can still require training pipelines.","Do LLM applications need continuous training?",{"id":1349,"answer":2561,"question":2562},"Generative outputs are open-ended and model behavior can change across prompts, snapshots and context. Evals provide repeatable evidence that a release still meets defined quality and safety criteria.","Why are evals so important in LLMOps?",{"id":1353,"answer":2564,"question":2565},"At minimum: application code, model\u002Fprovider\u002Fversion, prompts, eval datasets\u002Fscorers, retrieval configuration\u002Findexes, tool schemas, context rules and relevant safety\u002Fpermission configuration.","What should be versioned in LLMOps?",{"id":1357,"answer":2567,"question":2568},"No. The same prompt can behave differently with another model, retrieval set, context order, tool surface or provider.","Is prompt versioning enough?",{"id":1361,"answer":2570,"question":2571},"GenAIOps is another industry term for operating generative-AI applications. Some vendors use it interchangeably or as a broader label than LLMOps.","What is GenAIOps?",{"id":1365,"answer":2573,"question":2574},"Monitor end-to-end traces including model calls, prompts\u002Fcontext, retrieval, tools, latency, token\u002Fcost, quality samples, safety and final task outcomes.","How do you monitor an LLM application?",{"id":1369,"answer":2576,"question":2577},"Yes. Local models add their own operational concerns such as model files, hardware\u002FVRAM, load\u002Funload, runtime health, quantization and upgrade management.","Can local LLMs use LLMOps practices?","MLOps vs LLMOps FAQ",{},{"id":1375,"data":2581,"type":42,"tunes":2583},{"text":2582,"level":253},"Glossary",{},{"id":1380,"data":2585,"type":1380,"tunes":2619},{"title":2586,"entries":2587},"Key MLOps and LLMOps terms",[2588,2590,2592,2594,2596,2599,2602,2605,2608,2610,2613,2616],{"term":415,"anchor":414,"definition":2589},"Engineering practices for building, deploying, monitoring and maintaining machine-learning systems and their data\u002Fmodel lifecycle.",{"term":418,"anchor":417,"definition":2591},"Operational practices for production applications whose behavior materially depends on large language models and surrounding prompts, context, retrieval, tools and runtime.",{"term":1389,"anchor":1390,"definition":2593},"Operational discipline for generative-AI applications; often used as a broader or alternate label for LLMOps.",{"term":1748,"anchor":1393,"definition":2595},"Automated or repeated retraining and serving of ML models as data or implementations change.",{"term":2597,"anchor":1397,"definition":2598},"Continuous evaluation","Repeated evaluation of candidate and production AI behavior against versioned datasets and criteria.",{"term":2600,"anchor":1401,"definition":2601},"Model snapshot","A concrete version of a hosted or packaged model whose behavior can be tested and referenced.",{"term":2603,"anchor":1405,"definition":2604},"Application lineage","Traceable relationship among code, model\u002Fprovider, prompts, data\u002Fretrieval, tools, runtime and release configuration.",{"term":2606,"anchor":1409,"definition":2607},"Trace","Structured record of one application execution containing spans such as model calls, retrievals and tool operations.",{"term":1803,"anchor":1412,"definition":2609},"Versioned set of representative inputs, expectations and optionally traces\u002Foutputs used to measure behavior.",{"term":2611,"anchor":1416,"definition":2612},"LLM judge","A language model used as an evaluator for qualitative or semantic criteria; it is itself a versioned evaluation dependency.",{"term":2614,"anchor":1420,"definition":2615},"Behavioral regression","A degradation in application output or trajectory despite interfaces and code continuing to execute successfully.",{"term":2617,"anchor":1424,"definition":2618},"Provider routing","Policy for selecting among available model providers\u002Fendpoints according to capability, cost, latency, privacy or availability.",{},{"id":1428,"data":2621,"type":42,"tunes":2623},{"text":2622,"level":253},"Conclusion",{},{"id":1433,"data":2625,"type":218,"tunes":2627},{"text":2626},"MLOps and LLMOps share the same engineering objective: make AI systems reproducible enough, testable enough and observable enough to operate reliably in production.",{},{"id":1438,"data":2629,"type":218,"tunes":2631},{"text":2630},"The difference is the shape of the system. Classical MLOps often centers on training and serving model artifacts; LLMOps must operate a behavioral stack in which model snapshots, prompts, context, retrieval, tools, permissions and providers can change independently.",{},{"id":1443,"data":2633,"type":218,"tunes":2635},{"text":2634},"The shortest useful rule is: version, evaluate and observe everything that can materially change the LLM application's behavior — not only the model.",{},{"id":1448,"data":2637,"type":42,"tunes":2639},{"text":2638,"level":253},"Primary sources and current documentation",{},{"id":1453,"data":2641,"type":218,"tunes":2643},{"text":2642},"The sources below ground the MLOps baseline and the current operational patterns for LLM and agent applications. Project sections are original implementation evidence and are intentionally narrower than claims about a complete LLMOps platform.",{},{"id":1458,"data":2645,"type":1465,"tunes":2650},{"link":1460,"meta":2646},{"image":2647,"title":2648,"description":2649},{"url":381},"Google Cloud — MLOps: Continuous delivery and automation pipelines","Reference architecture describing CI, CD, continuous training, model registry, metadata, serving and monitoring for ML systems.",{},{"id":1468,"data":2652,"type":1465,"tunes":2657},{"link":1470,"meta":2653},{"image":2654,"title":2655,"description":2656},{"url":381},"AWS Machine Learning Lens — Model lineage","Current guidance for tracking code, data, models, environments and infrastructure across ML releases.",{},{"id":1477,"data":2659,"type":1465,"tunes":2664},{"link":1479,"meta":2660},{"image":2661,"title":2662,"description":2663},{"url":381},"AWS Machine Learning Lens — Model observability and tracking","Current guidance for production model monitoring, drift, endpoint health and lineage.",{},{"id":1486,"data":2666,"type":1465,"tunes":2671},{"link":1488,"meta":2667},{"image":2668,"title":2669,"description":2670},{"url":381},"Microsoft Azure — GenAIOps \u002F LLMOps lifecycle","Official guidance describing GenAIOps, sometimes called LLMOps, across initialization, experimentation, evaluation\u002Frefinement and deployment.",{},{"id":1495,"data":2673,"type":1465,"tunes":2678},{"link":1497,"meta":2674},{"image":2675,"title":2676,"description":2677},{"url":381},"MLflow — Agents and LLM applications","Current GenAI operations documentation covering tracing, evaluation, prompts and production observability for LLM applications and agents.",{},{"id":1504,"data":2680,"type":1465,"tunes":2685},{"link":1506,"meta":2681},{"image":2682,"title":2683,"description":2684},{"url":381},"MLflow — Evaluating production traces","Current guidance for evaluating complete LLM\u002Fagent traces, including retrieval and tool-call trajectories.",{},{"id":1513,"data":2687,"type":1465,"tunes":2692},{"link":1515,"meta":2688},{"image":2689,"title":2690,"description":2691},{"url":381},"MLflow — Evaluating prompts","Current prompt\u002Fmodel evaluation workflow using versioned prompts, datasets, scorers and traces.",{},{"id":1522,"data":2694,"type":1465,"tunes":2699},{"link":1524,"meta":2695},{"image":2696,"title":2697,"description":2698},{"url":381},"OpenAI API — Versioning and model snapshots","Current API guidance recommending pinned model versions and evals because prompting behavior can change between snapshots.",{},{"id":1531,"data":2701,"type":1465,"tunes":2706},{"link":1533,"meta":2702},{"image":2703,"title":2704,"description":2705},{"url":381},"OpenAI — Prompting","Current guidance to treat production prompts as application code, version them through source control and cover changes with tests and evaluation checks.",{},{"id":1540,"data":2708,"type":1465,"tunes":2713},{"link":1542,"meta":2709},{"image":2710,"title":2711,"description":2712},{"url":381},"OpenAI — Deprecations","Current provider lifecycle evidence showing model and platform-surface retirement as an operational dependency.",{},{"id":1549,"data":2715,"type":1465,"tunes":2720},{"link":1551,"meta":2716},{"image":2717,"title":2718,"description":2719},{"url":381},"OpenAI — Moving evaluation workflows to Promptfoo","Current 2026 migration guidance illustrating why evaluation assets should remain portable as provider tooling changes.",{},"2.31.6","MLOps operates machine-learning systems; LLMOps extends those practices to prompts, context, retrieval, providers, tools, evaluations and runtime behavior around large language models.",{"lang":7,"title":208,"content":210,"contentJson":2724,"excerpt":1558},{"time":212,"blocks":2725,"version":1557},[2726,2729,2732,2735,2738,2741,2744,2747,2750,2753,2756,2759,2762,2765,2768,2771,2774,2777,2780,2792,2795,2798,2801,2804,2807,2830,2833,2836,2839,2842,2845,2864,2867,2870,2873,2876,2879,2882,2885,2888,2891,2894,2897,2900,2903,2906,2909,2912,2915,2918,2921,2924,2927,2930,2933,2936,2939,2942,2945,2948,2951,2954,2957,2960,2963,2966,2969,2972,2975,2978,2981,2984,2987,2990,2993,2996,2999,3002,3005,3008,3011,3014,3017,3020,3034,3037,3040,3043,3046,3049,3052,3055,3058,3061,3076,3079,3082,3085,3088,3091,3094,3097,3100,3103,3106,3109,3112,3115,3118,3121,3124,3127,3130,3133,3136,3139,3142,3145,3148,3151,3154,3167,3170,3173,3190,3193,3208,3211,3227,3230,3249,3252,3255,3258,3261,3264,3267,3270,3273,3276,3279,3282,3285,3288,3291,3294,3297,3300,3313,3316,3332,3335,3338,3341,3344,3347,3350,3355,3360,3365,3370,3375,3380,3385,3390,3395,3400],{"id":215,"data":2727,"type":218,"tunes":2728},{"text":217},{},{"id":221,"data":2730,"type":226,"tunes":2731},{"body":223,"title":224,"variant":225},{},{"id":229,"data":2733,"type":226,"tunes":2734},{"body":231,"title":232,"variant":233},{},{"id":236,"data":2736,"type":226,"tunes":2737},{"body":238,"title":239,"variant":240},{},{"id":243,"data":2739,"type":226,"tunes":2740},{"body":245,"title":246,"variant":240},{},{"id":249,"data":2742,"type":254,"tunes":2743},{"title":251,"maxLevel":252,"minLevel":253},{},{"id":257,"data":2745,"type":42,"tunes":2746},{"text":259,"level":253},{},{"id":262,"data":2748,"type":218,"tunes":2749},{"text":264},{},{"id":267,"data":2751,"type":218,"tunes":2752},{"text":269},{},{"id":272,"data":2754,"type":218,"tunes":2755},{"text":274},{},{"id":277,"data":2757,"type":42,"tunes":2758},{"text":279,"level":253},{},{"id":282,"data":2760,"type":218,"tunes":2761},{"text":284},{},{"id":287,"data":2763,"type":218,"tunes":2764},{"text":289},{},{"id":292,"data":2766,"type":218,"tunes":2767},{"text":294},{},{"id":297,"data":2769,"type":42,"tunes":2770},{"text":299,"level":253},{},{"id":302,"data":2772,"type":218,"tunes":2773},{"text":304},{},{"id":307,"data":2775,"type":218,"tunes":2776},{"text":309},{},{"id":312,"data":2778,"type":218,"tunes":2779},{"text":314},{},{"id":317,"data":2781,"type":346,"tunes":2791},{"steps":2782,"title":344,"orientation":345},[2783,2784,2785,2786,2787,2788,2789,2790],{"label":321,"description":322},{"label":324,"description":325},{"label":327,"description":328},{"label":330,"description":331},{"label":333,"description":334},{"label":336,"description":337},{"label":339,"description":340},{"label":342,"description":343},{},{"id":349,"data":2793,"type":42,"tunes":2794},{"text":351,"level":253},{},{"id":354,"data":2796,"type":218,"tunes":2797},{"text":356},{},{"id":359,"data":2799,"type":218,"tunes":2800},{"text":361},{},{"id":364,"data":2802,"type":218,"tunes":2803},{"text":366},{},{"id":369,"data":2805,"type":42,"tunes":2806},{"text":371,"level":253},{},{"id":374,"data":2808,"type":419,"tunes":2829},{"rows":2809,"title":410,"layout":411,"columns":2826},[2810,2812,2814,2816,2818,2820,2822,2824],{"id":378,"label":379,"values":2811},[381,381],{"id":383,"label":384,"values":2813},[381,381],{"id":387,"label":388,"values":2815},[381,381],{"id":391,"label":392,"values":2817},[381,381],{"id":395,"label":396,"values":2819},[381,381],{"id":399,"label":400,"values":2821},[381,381],{"id":403,"label":404,"values":2823},[381,381],{"id":407,"label":408,"values":2825},[381,381],[2827,2828],{"id":414,"label":415},{"id":417,"label":418},{},{"id":422,"data":2831,"type":42,"tunes":2832},{"text":424,"level":253},{},{"id":427,"data":2834,"type":218,"tunes":2835},{"text":429},{},{"id":432,"data":2837,"type":218,"tunes":2838},{"text":434},{},{"id":437,"data":2840,"type":218,"tunes":2841},{"text":439},{},{"id":442,"data":2843,"type":42,"tunes":2844},{"text":444,"level":253},{},{"id":447,"data":2846,"type":411,"tunes":2863},{"content":2847,"stretched":43,"withHeadings":14},[2848,2849,2850,2851,2852,2853,2854,2855,2856,2857,2858,2859,2860,2861,2862],[451,452],[454,455],[457,458],[460,461],[463,464],[466,467],[469,470],[472,473],[475,476],[478,479],[481,482],[484,485],[487,488],[490,491],[493,494],{},{"id":497,"data":2865,"type":42,"tunes":2866},{"text":499,"level":253},{},{"id":502,"data":2868,"type":218,"tunes":2869},{"text":504},{},{"id":507,"data":2871,"type":218,"tunes":2872},{"text":509},{},{"id":512,"data":2874,"type":218,"tunes":2875},{"text":514},{},{"id":517,"data":2877,"type":42,"tunes":2878},{"text":519,"level":253},{},{"id":522,"data":2880,"type":218,"tunes":2881},{"text":524},{},{"id":527,"data":2883,"type":218,"tunes":2884},{"text":529},{},{"id":532,"data":2886,"type":218,"tunes":2887},{"text":534},{},{"id":537,"data":2889,"type":42,"tunes":2890},{"text":539,"level":253},{},{"id":542,"data":2892,"type":218,"tunes":2893},{"text":544},{},{"id":547,"data":2895,"type":218,"tunes":2896},{"text":549},{},{"id":552,"data":2898,"type":218,"tunes":2899},{"text":554},{},{"id":557,"data":2901,"type":42,"tunes":2902},{"text":559,"level":253},{},{"id":562,"data":2904,"type":218,"tunes":2905},{"text":564},{},{"id":567,"data":2907,"type":218,"tunes":2908},{"text":569},{},{"id":572,"data":2910,"type":218,"tunes":2911},{"text":574},{},{"id":577,"data":2913,"type":42,"tunes":2914},{"text":579,"level":253},{},{"id":582,"data":2916,"type":218,"tunes":2917},{"text":584},{},{"id":587,"data":2919,"type":218,"tunes":2920},{"text":589},{},{"id":592,"data":2922,"type":218,"tunes":2923},{"text":594},{},{"id":597,"data":2925,"type":603,"tunes":2926},{"url":599,"title":600,"excerpt":601,"ctaLabel":602},{},{"id":606,"data":2928,"type":42,"tunes":2929},{"text":608,"level":253},{},{"id":611,"data":2931,"type":218,"tunes":2932},{"text":613},{},{"id":616,"data":2934,"type":218,"tunes":2935},{"text":618},{},{"id":621,"data":2937,"type":218,"tunes":2938},{"text":623},{},{"id":626,"data":2940,"type":226,"tunes":2941},{"body":628,"title":629,"variant":630},{},{"id":633,"data":2943,"type":42,"tunes":2944},{"text":635,"level":253},{},{"id":638,"data":2946,"type":218,"tunes":2947},{"text":640},{},{"id":643,"data":2949,"type":218,"tunes":2950},{"text":645},{},{"id":648,"data":2952,"type":218,"tunes":2953},{"text":650},{},{"id":653,"data":2955,"type":42,"tunes":2956},{"text":655,"level":253},{},{"id":658,"data":2958,"type":218,"tunes":2959},{"text":660},{},{"id":663,"data":2961,"type":218,"tunes":2962},{"text":665},{},{"id":668,"data":2964,"type":218,"tunes":2965},{"text":670},{},{"id":673,"data":2967,"type":42,"tunes":2968},{"text":675,"level":253},{},{"id":678,"data":2970,"type":218,"tunes":2971},{"text":680},{},{"id":683,"data":2973,"type":218,"tunes":2974},{"text":685},{},{"id":688,"data":2976,"type":218,"tunes":2977},{"text":690},{},{"id":693,"data":2979,"type":603,"tunes":2980},{"url":695,"title":696,"excerpt":697,"ctaLabel":698},{},{"id":701,"data":2982,"type":42,"tunes":2983},{"text":703,"level":253},{},{"id":706,"data":2985,"type":218,"tunes":2986},{"text":708},{},{"id":711,"data":2988,"type":218,"tunes":2989},{"text":713},{},{"id":716,"data":2991,"type":218,"tunes":2992},{"text":718},{},{"id":721,"data":2994,"type":42,"tunes":2995},{"text":723,"level":253},{},{"id":726,"data":2997,"type":218,"tunes":2998},{"text":728},{},{"id":731,"data":3000,"type":218,"tunes":3001},{"text":733},{},{"id":736,"data":3003,"type":218,"tunes":3004},{"text":738},{},{"id":741,"data":3006,"type":42,"tunes":3007},{"text":743,"level":253},{},{"id":746,"data":3009,"type":218,"tunes":3010},{"text":748},{},{"id":751,"data":3012,"type":218,"tunes":3013},{"text":753},{},{"id":756,"data":3015,"type":218,"tunes":3016},{"text":758},{},{"id":761,"data":3018,"type":42,"tunes":3019},{"text":763,"level":253},{},{"id":766,"data":3021,"type":411,"tunes":3033},{"content":3022,"stretched":43,"withHeadings":14},[3023,3024,3025,3026,3027,3028,3029,3030,3031,3032],[770,771],[773,774],[776,777],[779,780],[782,783],[785,786],[788,789],[791,792],[794,795],[797,798],{},{"id":801,"data":3035,"type":42,"tunes":3036},{"text":803,"level":253},{},{"id":806,"data":3038,"type":218,"tunes":3039},{"text":808},{},{"id":811,"data":3041,"type":218,"tunes":3042},{"text":813},{},{"id":816,"data":3044,"type":218,"tunes":3045},{"text":818},{},{"id":821,"data":3047,"type":42,"tunes":3048},{"text":823,"level":253},{},{"id":826,"data":3050,"type":218,"tunes":3051},{"text":828},{},{"id":831,"data":3053,"type":218,"tunes":3054},{"text":833},{},{"id":836,"data":3056,"type":218,"tunes":3057},{"text":838},{},{"id":841,"data":3059,"type":42,"tunes":3060},{"text":843,"level":253},{},{"id":846,"data":3062,"type":411,"tunes":3075},{"content":3063,"stretched":43,"withHeadings":14},[3064,3065,3066,3067,3068,3069,3070,3071,3072,3073,3074],[850,851],[853,854],[779,856],[858,859],[861,862],[864,865],[782,867],[788,869],[791,871],[873,874],[876,877],{},{"id":880,"data":3077,"type":42,"tunes":3078},{"text":882,"level":253},{},{"id":885,"data":3080,"type":218,"tunes":3081},{"text":887},{},{"id":890,"data":3083,"type":218,"tunes":3084},{"text":892},{},{"id":895,"data":3086,"type":218,"tunes":3087},{"text":897},{},{"id":900,"data":3089,"type":42,"tunes":3090},{"text":902,"level":253},{},{"id":905,"data":3092,"type":218,"tunes":3093},{"text":907},{},{"id":910,"data":3095,"type":218,"tunes":3096},{"text":912},{},{"id":915,"data":3098,"type":218,"tunes":3099},{"text":917},{},{"id":920,"data":3101,"type":42,"tunes":3102},{"text":922,"level":253},{},{"id":925,"data":3104,"type":218,"tunes":3105},{"text":927},{},{"id":930,"data":3107,"type":218,"tunes":3108},{"text":932},{},{"id":935,"data":3110,"type":218,"tunes":3111},{"text":937},{},{"id":940,"data":3113,"type":42,"tunes":3114},{"text":942,"level":253},{},{"id":945,"data":3116,"type":218,"tunes":3117},{"text":947},{},{"id":950,"data":3119,"type":218,"tunes":3120},{"text":952},{},{"id":955,"data":3122,"type":218,"tunes":3123},{"text":957},{},{"id":960,"data":3125,"type":42,"tunes":3126},{"text":962,"level":253},{},{"id":965,"data":3128,"type":42,"tunes":3129},{"text":967,"level":252},{},{"id":970,"data":3131,"type":218,"tunes":3132},{"text":972},{},{"id":975,"data":3134,"type":218,"tunes":3135},{"text":977},{},{"id":980,"data":3137,"type":218,"tunes":3138},{"text":982},{},{"id":985,"data":3140,"type":218,"tunes":3141},{"text":987},{},{"id":990,"data":3143,"type":42,"tunes":3144},{"text":992,"level":252},{},{"id":995,"data":3146,"type":218,"tunes":3147},{"text":997},{},{"id":1000,"data":3149,"type":218,"tunes":3150},{"text":1002},{},{"id":1005,"data":3152,"type":218,"tunes":3153},{"text":1007},{},{"id":1010,"data":3155,"type":411,"tunes":3166},{"content":3156,"stretched":43,"withHeadings":14},[3157,3158,3159,3160,3161,3162,3163,3164,3165],[1014,1015],[1017,1018],[1020,1021],[1023,1024],[1026,1027],[1029,1030],[1032,1033],[1035,1036],[1038,1039],{},{"id":1042,"data":3168,"type":226,"tunes":3169},{"body":1044,"title":1045,"variant":240},{},{"id":1048,"data":3171,"type":42,"tunes":3172},{"text":1050,"level":253},{},{"id":1053,"data":3174,"type":411,"tunes":3189},{"content":3175,"stretched":43,"withHeadings":14},[3176,3177,3178,3179,3180,3181,3182,3183,3184,3185,3186,3187,3188],[1057,1058],[1060,1061],[1063,1064],[1066,1067],[1069,1070],[1072,1073],[1075,1076],[1078,1079],[1081,1082],[1084,1085],[1087,1088],[1090,1091],[1093,1094],{},{"id":1097,"data":3191,"type":42,"tunes":3192},{"text":1099,"level":253},{},{"id":1102,"data":3194,"type":411,"tunes":3207},{"content":3195,"stretched":43,"withHeadings":14},[3196,3197,3198,3199,3200,3201,3202,3203,3204,3205,3206],[1106,1107],[1109,1110],[1112,1113],[1115,1116],[1118,1119],[1121,1122],[1124,1125],[1127,1128],[1130,1131],[1133,1134],[1136,1137],{},{"id":1140,"data":3209,"type":42,"tunes":3210},{"text":1142,"level":253},{},{"id":1145,"data":3212,"type":346,"tunes":3226},{"steps":3213,"title":1184,"orientation":345},[3214,3215,3216,3217,3218,3219,3220,3221,3222,3223,3224,3225],{"label":1149,"description":1150},{"label":1152,"description":1153},{"label":1155,"description":1156},{"label":1158,"description":1159},{"label":1161,"description":1162},{"label":1164,"description":1165},{"label":1167,"description":1168},{"label":1170,"description":1171},{"label":1173,"description":1174},{"label":1176,"description":1177},{"label":1179,"description":1180},{"label":1182,"description":1183},{},{"id":1187,"data":3228,"type":42,"tunes":3229},{"text":1189,"level":253},{},{"id":1192,"data":3231,"type":411,"tunes":3248},{"content":3232,"stretched":43,"withHeadings":14},[3233,3234,3235,3236,3237,3238,3239,3240,3241,3242,3243,3244,3245,3246,3247],[1196,1197],[1199,1200],[1202,1203],[1205,1206],[1208,1209],[1211,1212],[1214,1215],[1217,1218],[1220,1221],[1223,1224],[1226,1227],[1229,1230],[1232,1233],[1235,1236],[1238,1239],{},{"id":1242,"data":3250,"type":42,"tunes":3251},{"text":1244,"level":253},{},{"id":1247,"data":3253,"type":218,"tunes":3254},{"text":1249},{},{"id":1252,"data":3256,"type":218,"tunes":3257},{"text":1254},{},{"id":1257,"data":3259,"type":218,"tunes":3260},{"text":1259},{},{"id":1262,"data":3262,"type":218,"tunes":3263},{"text":1264},{},{"id":1267,"data":3265,"type":218,"tunes":3266},{"text":1269},{},{"id":1272,"data":3268,"type":42,"tunes":3269},{"text":1274,"level":253},{},{"id":1277,"data":3271,"type":218,"tunes":3272},{"text":1279},{},{"id":1282,"data":3274,"type":218,"tunes":3275},{"text":1284},{},{"id":1287,"data":3277,"type":218,"tunes":3278},{"text":1289},{},{"id":1292,"data":3280,"type":42,"tunes":3281},{"text":1294,"level":253},{},{"id":1297,"data":3283,"type":218,"tunes":3284},{"text":1299},{},{"id":1302,"data":3286,"type":218,"tunes":3287},{"text":1304},{},{"id":1307,"data":3289,"type":218,"tunes":3290},{"text":1309},{},{"id":1312,"data":3292,"type":603,"tunes":3293},{"url":1314,"title":1315,"excerpt":1316,"ctaLabel":1317},{},{"id":1320,"data":3295,"type":603,"tunes":3296},{"url":1322,"title":1323,"excerpt":1324,"ctaLabel":1325},{},{"id":1328,"data":3298,"type":42,"tunes":3299},{"text":1330,"level":253},{},{"id":1333,"data":3301,"type":1333,"tunes":3312},{"items":3302,"title":1372},[3303,3304,3305,3306,3307,3308,3309,3310,3311],{"id":1337,"answer":1338,"question":1339},{"id":1341,"answer":1342,"question":1343},{"id":1345,"answer":1346,"question":1347},{"id":1349,"answer":1350,"question":1351},{"id":1353,"answer":1354,"question":1355},{"id":1357,"answer":1358,"question":1359},{"id":1361,"answer":1362,"question":1363},{"id":1365,"answer":1366,"question":1367},{"id":1369,"answer":1370,"question":1371},{},{"id":1375,"data":3314,"type":42,"tunes":3315},{"text":1377,"level":253},{},{"id":1380,"data":3317,"type":1380,"tunes":3331},{"title":1382,"entries":3318},[3319,3320,3321,3322,3323,3324,3325,3326,3327,3328,3329,3330],{"term":415,"anchor":414,"definition":1385},{"term":418,"anchor":417,"definition":1387},{"term":1389,"anchor":1390,"definition":1391},{"term":400,"anchor":1393,"definition":1394},{"term":1396,"anchor":1397,"definition":1398},{"term":1400,"anchor":1401,"definition":1402},{"term":1404,"anchor":1405,"definition":1406},{"term":1408,"anchor":1409,"definition":1410},{"term":469,"anchor":1412,"definition":1413},{"term":1415,"anchor":1416,"definition":1417},{"term":1419,"anchor":1420,"definition":1421},{"term":1423,"anchor":1424,"definition":1425},{},{"id":1428,"data":3333,"type":42,"tunes":3334},{"text":1430,"level":253},{},{"id":1433,"data":3336,"type":218,"tunes":3337},{"text":1435},{},{"id":1438,"data":3339,"type":218,"tunes":3340},{"text":1440},{},{"id":1443,"data":3342,"type":218,"tunes":3343},{"text":1445},{},{"id":1448,"data":3345,"type":42,"tunes":3346},{"text":1450,"level":253},{},{"id":1453,"data":3348,"type":218,"tunes":3349},{"text":1455},{},{"id":1458,"data":3351,"type":1465,"tunes":3354},{"link":1460,"meta":3352},{"image":3353,"title":1463,"description":1464},{"url":381},{},{"id":1468,"data":3356,"type":1465,"tunes":3359},{"link":1470,"meta":3357},{"image":3358,"title":1473,"description":1474},{"url":381},{},{"id":1477,"data":3361,"type":1465,"tunes":3364},{"link":1479,"meta":3362},{"image":3363,"title":1482,"description":1483},{"url":381},{},{"id":1486,"data":3366,"type":1465,"tunes":3369},{"link":1488,"meta":3367},{"image":3368,"title":1491,"description":1492},{"url":381},{},{"id":1495,"data":3371,"type":1465,"tunes":3374},{"link":1497,"meta":3372},{"image":3373,"title":1500,"description":1501},{"url":381},{},{"id":1504,"data":3376,"type":1465,"tunes":3379},{"link":1506,"meta":3377},{"image":3378,"title":1509,"description":1510},{"url":381},{},{"id":1513,"data":3381,"type":1465,"tunes":3384},{"link":1515,"meta":3382},{"image":3383,"title":1518,"description":1519},{"url":381},{},{"id":1522,"data":3386,"type":1465,"tunes":3389},{"link":1524,"meta":3387},{"image":3388,"title":1527,"description":1528},{"url":381},{},{"id":1531,"data":3391,"type":1465,"tunes":3394},{"link":1533,"meta":3392},{"image":3393,"title":1536,"description":1537},{"url":381},{},{"id":1540,"data":3396,"type":1465,"tunes":3399},{"link":1542,"meta":3397},{"image":3398,"title":1545,"description":1546},{"url":381},{},{"id":1549,"data":3401,"type":1465,"tunes":3404},{"link":1551,"meta":3402},{"image":3403,"title":1554,"description":1555},{"url":381},{},"Post erfolgreich abgerufen",{"items":3407,"source":3488,"manualIds":3489,"manualMatchedIds":3490},[3408,3414,3419,3426,3433,3440,3447,3454,3460,3467,3474,3481],{"id":3409,"slug":1586,"title":3410,"excerpt":3411,"featuredImage":3412,"publishedAt":3413},"434","全面评估指南：精通LLM性能评估","本指南详细介绍了评估工具（Evaluation Harness），这是一个在企业级LLMOps流程中严格评估大型语言模型（LLM）能力的关键框架。您将学习其设置方法、最佳实践以及高级技巧，以确保模型基准测试与优化的可靠性。","\u002Fuploads\u002F2026\u002F04\u002Fevaluation-harness-1775466944495-4s0xv2.webp","2026-03-01T17:50:00.000Z",{"id":3415,"slug":3416,"title":3416,"excerpt":10,"featuredImage":3417,"publishedAt":3418},"369","git-with-automatic-upload-and-synchronization-to-a-production-server","\u002Fuploads\u002F2024\u002F05\u002Fstep-by-step-guide-illustration-showing-the-process-of-setting-up-Git-with-auto-upload-and-synchronization-to-a-production-server-large.webp","2024-05-28T22:48:00.000Z",{"id":3420,"slug":3421,"title":3422,"excerpt":3423,"featuredImage":3424,"publishedAt":3425},"488","what-is-context-engineering-what-the-model-receives-before-it-answers","什么是上下文工程？模型在回答之前接收到了什么","上下文工程负责设计 AI 模型在推理前接收的信息，包括提示词、检索内容、记忆、应用状态、工具结果和对话历史。","\u002Fuploads\u002F2026\u002F10\u002Fwhat-is-context-engineering-what-the-model-receives-before-it-answers-1791480653258-018kcv.webp","2026-10-08T13:29:00.000Z",{"id":3427,"slug":3428,"title":3429,"excerpt":3430,"featuredImage":3431,"publishedAt":3432},"459","ollama-is-not-the-product-building-production-ready-open-llm-applications","Ollama 并非产品：构建可投入生产的开源大语言模型应用","使用Ollama运行本地模型很简单。但构建一个可用于生产环境的开源大语言模型（Open-LLM）应用则更具挑战性：它需要RAG（检索增强生成）、访问控制、供应商抽象、评估、日志记录、部署规范，以及围绕模型构建受控的应用层。","\u002Fuploads\u002F2026\u002F06\u002Follama-is-not-the-product-building-production-ready-open-llm-applications-1782679361640-h0usqf.webp","2026-06-28T16:39:00.000Z",{"id":3434,"slug":3435,"title":3436,"excerpt":3437,"featuredImage":3438,"publishedAt":3439},"479","where-does-an-llm-get-its-data-rag-data-sources-in-python","LLM从哪里获取数据？Python中的RAG数据源","LLM 并不会神奇地知道你的文件、数据库或 API。这个 RAG 系列的实用续篇用简单的 Python 展示了外部数据如何变成可检索的证据：从文本文件和 SQL 到全文搜索、嵌入、上下文组装以及最终的 LLM 调用。","\u002Fuploads\u002F2026\u002F09\u002Fwhere-does-an-llm-get-its-data-rag-data-sources-in-python-1790517200521-nfsi5i.webp","2026-09-27T05:51:00.000Z",{"id":3441,"slug":3442,"title":3443,"excerpt":3444,"featuredImage":3445,"publishedAt":3446},"364","tipps-fuer-die-verbesserung-der-seo-suchmaschinenoptimierung","Mastering the SEO Workflow: Essential Optimization Strategies for Organic Growth","A structured SEO workflow is crucial for sustainable organic growth. Learn the ten foundational strategies, from keyword research and technical optimization to content quality and performance analysis.","\u002Fuploads\u002F2026\u002F03\u002Ftipps-fuer-die-verbesserung-der-seo-suchmaschinenoptimierung-1774866098131-hwkzrg.webp","2024-01-26T06:35:00.000Z",{"id":3448,"slug":3449,"title":3450,"excerpt":3451,"featuredImage":3452,"publishedAt":3453},"384","new-qwen-3-5-plus","全新Qwen 3.5-Plus：开源AI迈入新纪元","探索阿里巴巴Qwen 3.5-Plus的革命性特性与优势，这款为开发者打造的颠覆性开源人工智能模型。","\u002Fuploads\u002F2026\u002F02\u002Fnew-qwen-3-5-plus-1771515512741-dcbi9p.webp","2026-02-19T10:23:00.000Z",{"id":3455,"slug":3456,"title":3457,"excerpt":3458,"featuredImage":3452,"publishedAt":3459},"445","qwen-3-6-in-production-release-runbook-ai-rollback-and-llmops-versioning","Qwen 3.6 生产环境部署：发布手册、AI 回滚与 LLMOps 版本管理","Qwen 3.6 不仅仅是一次模型升级。它同时是一个发布事件、一个回滚场景和一个版本管理问题。本文通过LLMOps规范、提示词与模型可追溯性、受控发布以及基于证据的回滚准备，阐述了在生产环境中应如何处理Qwen 3.6。","2026-05-04T02:49:00.000Z",{"id":3461,"slug":3462,"title":3463,"excerpt":3464,"featuredImage":3465,"publishedAt":3466},"477","computer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system","计算机使用代理：为什么成功的演示仍可能是一个不可靠的系统","计算机使用代理如今能够完成令人印象深刻的浏览器和桌面工作流程，但一次成功的运行证明的是能力——而非可靠性。本文展示了如何测试可重复性、环境鲁棒性、长时程控制、状态感知、结果验证以及安全的目标处理。","\u002Fuploads\u002F2026\u002F09\u002Fcomputer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system-1790352854690-75qnrg.webp","2026-09-25T12:13:00.000Z",{"id":3468,"slug":3469,"title":3470,"excerpt":3471,"featuredImage":3472,"publishedAt":3473},"460","ai-agent-reliability-why-the-final-answer-is-not-enough","AI Agent可靠性：为什么最终答案并不足够","正确的输出并不能证明推理的正确性、执行的安全性，或系统的可信赖性。","\u002Fuploads\u002F2026\u002F09\u002Fai-agent-reliability-why-the-final-answer-is-not-enough-1788955466306-pl0qhz.webp","2026-09-09T04:01:00.000Z",{"id":3475,"slug":3476,"title":3477,"excerpt":3478,"featuredImage":3479,"publishedAt":3480},"480","when-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger","人工智能何时应停止信任自身知识？——检索触发机制","AI 模型并非每个问题都需要检索。重要的问题在于知道何时其内部知识已不再足够。检索触发器是一个实用的决策边界，它决定 AI 系统何时应停止仅依赖模型知识，并在回答前获取外部证据。","\u002Fuploads\u002F2026\u002F09\u002Fwhen-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger-1790574991244-f4rpyg.webp","2026-09-28T01:49:00.000Z",{"id":3482,"slug":3483,"title":3484,"excerpt":3485,"featuredImage":3486,"publishedAt":3487},"478","what-is-rag-the-simplest-explanation-of-how-it-works","什么是RAG？对其工作原理的最简单解释","RAG听起来很复杂，但想法很简单：在AI回答之前，它先从知识源查找有用的信息，并将该信息提供给语言模型。本指南使用一个简单的思维模型来解释RAG、LLM、状态、记忆和工具。","\u002Fuploads\u002F2026\u002F09\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works-1790377492124-khjagt.webp","2026-09-25T19:03:00.000Z","fallback",[],[]]