[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"portal-settings:stajic:zh":3,"public-menus:all":38,"post:what-is-context-engineering-what-the-model-receives-before-it-answers:zh":205,"related:post:what-is-context-engineering-what-the-model-receives-before-it-answers:zh:1":3034},{"statusCode":4,"data":5,"message":37},200,{"tenantId":6,"lang":7,"defaultLang":8,"siteUrl":9,"contactEmail":10,"brandName":11,"logoUrl":12,"siteName":11,"siteDescription":13,"ogImage":10,"robotsIndex":14,"socialLinks":10,"reservedSlugs":10,"seoPolicy":15},"stajic","zh","de","https:\u002F\u002Fstajic.de",null,"Stajic Platform","\u002FLogo_Planet.svg","Stajic Portal",true,{"branding":16,"relatedContent":17,"crossDomainLinks":18},{"logoUrl":12},{"enabled":14},[19,22,25,28,31,34],{"url":20,"label":21,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Ffigure.rocks","figure.rocks",{"url":23,"label":24,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Floving.rocks","loving.rocks",{"url":26,"label":27,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.com","bazify.com",{"url":29,"label":30,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.de","bazify.de",{"url":32,"label":33,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.at","bazify.at",{"url":35,"label":36,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.ba","bazify.ba","Portal settings resolved",[39,45],{"id":40,"name":41,"location":42,"isActive":14,"isDefault":43,"items":44},1,"main-navigation","header",false,[],{"id":46,"name":47,"location":48,"isActive":14,"isDefault":14,"items":49},4,"main-menu","sidebar",[50,66,79,93,103,118,133],{"id":51,"title":52,"url":60,"target":61,"icon":62,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":64,"portfolioId":10,"children":65},"item-18",{"de":53,"en":54,"es":55,"fr":56,"it":54,"ru":57,"sr":58,"zh":59},"Startseite","Home","Inicio","Accueil","Главная","Почетна","首页","\u002Ffull-stack-web-developer-munich-performance-seo-and-maintainable-builds","_self","i-lucide-home","page",111,[],{"id":67,"title":68,"url":75,"target":61,"icon":76,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":77,"portfolioId":10,"children":78},"item-22",{"de":69,"en":69,"es":70,"fr":69,"it":71,"ru":72,"sr":73,"zh":74},"Vision","Visión","Visione","Видение","Визија","想象","\u002Fueber-uns-webdesign-muenchen-webaplikation","i-lucide-eye",113,[],{"id":80,"title":81,"url":89,"target":61,"icon":90,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":91,"portfolioId":10,"children":92},"item-19",{"de":82,"en":83,"es":84,"fr":83,"it":85,"ru":86,"sr":87,"zh":88},"Leistungen","Services","Servicios","Servizi","Услуги","Услуге","服务","\u002Fservices-dienstleistungen-muenchen","i-lucide-wrench",116,[],{"id":94,"title":95,"url":99,"target":61,"icon":100,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":101,"portfolioId":10,"children":102},"item-23",{"de":96,"en":96,"es":96,"fr":96,"it":96,"ru":97,"sr":97,"zh":98},"Blog","Блог","博客","\u002Fblog","i-lucide-book-open",112,[],{"id":104,"title":105,"url":114,"target":61,"icon":115,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":116,"portfolioId":10,"children":117},"item-32",{"de":106,"en":107,"es":108,"fr":109,"it":110,"ru":111,"sr":112,"zh":113},"Neue Technologien","New Technologies","Nuevas tecnologías","Nouvelles technologies","Nuove tecnologie","Новые технологии","Нове технологије","新技术！","\u002Fneue-webtechnologien","i-lucide-sparkles",122,[],{"id":119,"title":120,"url":129,"target":61,"icon":130,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":131,"portfolioId":10,"children":132},"item-20",{"de":121,"en":122,"es":123,"fr":124,"it":125,"ru":126,"sr":127,"zh":128},"Kontakt","Contact us!","Contacto","Contact","Contatto","Контакт","Контактирајте нас","联系我们！","\u002Fcontact","i-lucide-mail",115,[],{"id":134,"title":135,"url":144,"target":61,"icon":145,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":147},"item-21",{"de":136,"en":137,"es":138,"fr":139,"it":140,"ru":141,"sr":142,"zh":143},"Unsere Arbeit","Our Work","Nuestro trabajo","Nos réalisations","I nostri lavori","Наши работы","Наши радови","文件夹","\u002Fportfolio","i-lucide-briefcase",114,[148,161,175,181,193],{"id":149,"title":150,"url":144,"target":61,"icon":159,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":160},"item-24",{"de":151,"en":152,"es":153,"fr":154,"it":155,"ru":156,"sr":157,"zh":158},"Alle Projekte","All Projects","Todos los proyectos","Tous les projets","Tutti i progetti","Все проекты","Сви пројекти","所有项目","i-lucide-grid-3x3",[],{"id":162,"title":163,"url":171,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":174},"item-29",{"de":164,"en":165,"es":166,"fr":167,"it":168,"ru":169,"sr":170,"zh":143},"Local Roots, Global Reach","Local Roots - Global Reach","Empresa local ","Entreprise locale","Azienda locale","Местная компания","Локално предузеће глобално тржиште","\u002Fportfolio\u002Flocal-roots-global-reach-communication-media-systems-for-modern-business","i-lucide-folder","custom",[],{"id":176,"title":177,"url":179,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":180},"item-28",{"de":178,"en":178,"es":178,"fr":178,"it":178,"ru":178,"sr":178,"zh":178},"Solr Suggester","\u002Fportfolio\u002Fsolr-fuzzy-suggester-und-solr-infix-suggester-abfrage-ueber-ajax-und-filterung",[],{"id":182,"title":183,"url":191,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":192},"item-27",{"de":184,"en":185,"es":186,"fr":187,"it":188,"ru":189,"sr":190,"zh":185},"Firmenwebseite SEO","Company Website SEO","Sitio web corporativo SEO","Site web d’entreprise SEO","Sito web aziendale SEO","Корпоративный сайт SEO","Пословна веб-страница SEO","\u002Fportfolio\u002Fseo-sem-branding-mobile-webseite-muenchen",[],{"id":194,"title":195,"url":203,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":204},"item-31",{"de":196,"en":197,"es":198,"fr":199,"it":200,"ru":201,"sr":202,"zh":197},"Digitalisierungsportal","Digitalization Portal","Portal de digitalización","Portail de numérisation","Portale di digitalizzazione","Портал цифровизации","Портал за дигитализацију","\u002Fportfolio\u002Fdigitalisierungsportal-archiv-museum-bibliothek-ead-lido-mets-mods",[],{"statusCode":4,"data":206,"message":3033},{"id":207,"title":208,"slug":209,"content":210,"contentJson":211,"excerpt":1407,"featuredImage":1408,"featuredImageAlt":1409,"featuredImageCaption":10,"featuredImageTitle":10,"featuredImageCopyright":10,"featuredImageAuthor":10,"featuredImageSourceUrl":10,"featuredImageLicense":10,"featuredImageIsAiGenerated":43,"status":1410,"publishedAt":1411,"createdAt":1412,"updatedAt":1413,"seoLocalePaths":1414,"categories":1423,"author":1436,"translations":1441},"488","什么是上下文工程？模型在回答之前接收到了什么","what-is-context-engineering-what-the-model-receives-before-it-answers","\u003Cp>上下文工程是指设计语言模型在推理时接收哪些信息、以何种形式、按何种顺序以及保留多长时间。它比提示工程更广泛，因为模型上下文可以包括系统指令、用户消息、检索到的文档、工具结果、记忆、当前应用状态、示例、结构化数据和中间产物。目标不是最大化令牌数量，而是构建最小的有用上下文，保留当前任务所需的信息、约束和证据。\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--info my-6 rounded-xl border p-5 border-blue-300 bg-blue-50 dark:border-blue-900 dark:bg-blue-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">直接回答\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">提示工程问的是\u003Cstrong>我们应该如何指示模型？\u003C\u002Fstrong>上下文工程问的是\u003Cstrong>模型现在应该知道什么，以及这些信息应该如何组装？\u003C\u002Fstrong>\u003Cbr>\u003Cbr>因此，当检索、记忆、状态管理、工具设计、历史裁剪、压缩和排序决定了模型在产生下一个输出之前可用的令牌时，它们就是上下文工程机制。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Caside class=\"editorjs-callout editorjs-callout--warning my-6 rounded-xl border p-5 border-amber-300 bg-amber-50 dark:border-amber-900 dark:bg-amber-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">上下文不等同于知识或记忆\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">一个系统可以知道某件事，而不把它放入当前上下文。它可以在模型窗口之外记住某件事。它可以检索一个文档，但后来将其排除在最终提示之外。模型只能直接使用到达当前推理的上下文。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Caside class=\"editorjs-callout editorjs-callout--note my-6 rounded-xl border p-5 border-gray-300 bg-gray-50 dark:border-gray-700 dark:bg-gray-900\u002F40\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">当前来源说明——2026年10月8日\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">上下文工程现已成为主要AI工程指南中确立的实用术语，但它并不是一个具有单一强制架构的正式标准。Anthropic将其描述为为推理策划和维护最优令牌集；OpenAI当前的智能体指南将会话上下文、裁剪和压缩视为长期运行系统的明确工程关注点。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Cnav class=\"editorjs-toc\" data-editorjs-toc=\"true\" aria-label=\"目录\">\u003Cstrong class=\"editorjs-toc__title\">目录\u003C\u002Fstrong>\u003Col class=\"editorjs-toc__list editorjs-toc__list--depth-0\">\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-6\" class=\"editorjs-toc__link\">上下文工程真正意味着什么\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-10\" class=\"editorjs-toc__link\">最简单的例子\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-15\" class=\"editorjs-toc__link\">简单例子止步之处\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-18\" class=\"editorjs-toc__link\">什么可以进入模型上下文？\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-20\" class=\"editorjs-toc__link\">上下文工程与提示工程\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-23\" class=\"editorjs-toc__link\">上下文工程与检索\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-28\" class=\"editorjs-toc__link\">上下文工程与记忆\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-33\" class=\"editorjs-toc__link\">上下文工程与应用状态\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-37\" class=\"editorjs-toc__link\">工具设计是上下文工程的一部分\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-41\" class=\"editorjs-toc__link\">即时上下文与预加载上下文\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-44\" class=\"editorjs-toc__link\">上下文是预算，不是存储系统\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-48\" class=\"editorjs-toc__link\">为什么更多上下文可能更糟\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-52\" class=\"editorjs-toc__link\">上下文排序应当是有意为之\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-56\" class=\"editorjs-toc__link\">冲突的上下文需要明确的优先级\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-60\" class=\"editorjs-toc__link\">压缩是上下文转换，而非无损存储\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-65\" class=\"editorjs-toc__link\">保留有效性边界\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-69\" class=\"editorjs-toc__link\">上下文工程也是安全边界\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-73\" class=\"editorjs-toc__link\">一个实用的上下文工程架构\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-77\" class=\"editorjs-toc__link\">一个实用的上下文构建策略\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-79\" class=\"editorjs-toc__link\">如何评估上下文工程\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-81\" class=\"editorjs-toc__link\">上下文组装是一个独立的 RAG 故障层\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-85\" class=\"editorjs-toc__link\">原始实现证据\u003C\u002Fa>\u003Col class=\"editorjs-toc__list editorjs-toc__list--depth-1\">\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-86\" class=\"editorjs-toc__link\">真相来源研究引擎：有界研究而非无限上下文\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-90\" class=\"editorjs-toc__link\">Aaasaasa AI 客户端：运行时、权限和上下文是相互独立的关注点\u003C\u002Fa>\u003C\u002Fli>\u003C\u002Fol>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-96\" class=\"editorjs-toc__link\">常见的上下文工程失败模式\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-98\" class=\"editorjs-toc__link\">常见误解\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-100\" class=\"editorjs-toc__link\">一个实用的上下文工程序列\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-102\" class=\"editorjs-toc__link\">上下文工程检查清单\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-104\" class=\"editorjs-toc__link\">边缘情况和限制\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-110\" class=\"editorjs-toc__link\">什么会改变这个答案？\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-114\" class=\"editorjs-toc__link\">相关权威知识\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-119\" class=\"editorjs-toc__link\">常见问题\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-121\" class=\"editorjs-toc__link\">术语表\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-123\" class=\"editorjs-toc__link\">结论\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-127\" class=\"editorjs-toc__link\">主要来源与当前指南\u003C\u002Fa>\u003C\u002Fli>\u003C\u002Fol>\u003C\u002Fnav>\n\u003Ch2 id=\"section-6\">上下文工程真正意味着什么\u003C\u002Fh2>\n\u003Cp>每次模型调用都是在一个临时工作环境中进行的：当前指令、消息、检索到的证据、工具输出和状态，这些内容都适合放入活动上下文窗口。上下文工程就是有意构建该环境的学科。\u003C\u002Fp>\n\u003Cp>关键词是有意。一个朴素系统只是把它拥有的一切拼接起来：完整历史、所有检索到的文档、每个工具响应和大型系统提示。一个经过上下文工程的系统会决定当前决策需要哪些信息，以及哪些信息应留在窗口之外直到需要时再使用。\u003C\u002Fp>\n\u003Cp>这使上下文工程部分成为信息架构问题，部分成为运行时问题，部分成为评估问题。设计必须决定什么可以进入上下文、它来自哪里、哪个版本是当前的、冲突如何解决、保留多少细节以及结果如何测试。\u003C\u002Fp>\n\u003Ch2 id=\"section-10\">最简单的例子\u003C\u002Fh2>\n\u003Cp>想象一个内部支持助手。用户问：“这个客户可以免费取消吗？”\u003C\u002Fp>\n\u003Cp>模型可能需要五样东西：当前取消政策、客户当前合同类型、合同生效日期、相关例外规则以及用户的授权范围。\u003C\u002Fp>\n\u003Cp>它不一定需要整个客户数据库、完整政策档案、每一次之前的对话或每一张支持工单。上下文工程就是选择并组装这五个有用部分，同时排除无关信息的过程。\u003C\u002Fp>\n\u003Csection class=\"editorjs-process my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">从应用状态到模型上下文\u003C\u002Fh3>\u003Cdiv class=\"grid grid-cols-1 md:grid-cols-2 xl:grid-cols-3 gap-4\">\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">1\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">1. 理解任务\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">对当前问题需要什么以及哪些信息类型可能影响答案进行分类。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">2\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">2. 解析权威状态\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">读取不应从记忆中猜测的当前应用或业务状态。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">3\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">3. 检索支持性知识\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">找到与特定任务相关的政策、文档或外部证据。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">4\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">4. 应用资格和权限\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">排除当前用户或运行时不允许暴露给模型的数据。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">5\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">5. 精简和结构化\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">删除重复内容，选择有用摘录，并保留关键元数据、条件和例外。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">6\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">6. 排列上下文顺序\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">将指令、当前状态和决定性证据放在模型能够一致使用它们的位置。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">7\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">7. 运行推理\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">模型接收组装好的上下文，并产生下一个答案或行动建议。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-15\">简单例子止步之处\u003C\u002Fh2>\n\u003Cp>真实系统更困难，因为某一步所需的信息可能在执行开始前并不已知。智能体可以通过工具发现新事实、创建中间文件、接收不断变化的外部状态，或跨越比一个上下文窗口更长的任务。\u003C\u002Fp>\n\u003Cp>因此，上下文工程变得动态。第12步的上下文不应只是第1步上下文加上十一层累积输出。它应反映当前任务状态、仍然重要的决策以及下一步行动所需的证据。\u003C\u002Fp>\n\u003Ch2 id=\"section-18\">什么可以进入模型上下文？\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">上下文组件\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">用途\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">典型风险\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">系统\u002F开发者指令\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">定义角色、约束、策略和行为\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">过于模糊、自相矛盾或充斥着脆弱的逻辑\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">当前用户请求\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">定义即时任务和意图\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">存在歧义或与先前历史冲突\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">对话历史\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">保持跨轮次的连续性\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">过时的假设、重复和令牌增长\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">检索到的文档\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">提供外部知识\u002F证据\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">不相关、版本过时、权威性弱或重复\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">当前应用状态\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">提供易变的业务\u002F系统事实\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">使用缓存或记忆中的状态而非当前权威状态\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">工具定义\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">告知模型存在哪些能力以及如何调用它们\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">工具过多重叠或模式过于冗长\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">工具结果\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">将来自环境的观察带入循环\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">大量嘈杂输出、不可信内容或过时的观察\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">记忆\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">重新引入先前交互中的选定信息\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">过时、不正确的泛化或过度个性化\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">示例\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">展示期望的行为\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">过多的边缘案例可能挤占当前任务\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">中间产物\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">承载计划、摘要、代码、计算或笔记\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">旧的中间状态可能被误认为最终真相\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">策略\u002F护栏\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">定义禁止或受约束的行为\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">与业务逻辑冲突或存在隐藏的执行缺口\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-20\">上下文工程与提示工程\u003C\u002Fh2>\n\u003Csection class=\"editorjs-comparison my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">提示工程和上下文工程解决不同层面的问题\u003C\u002Fh3>\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left dark:border-gray-700 dark:bg-gray-900\">\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">提示工程\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">上下文工程\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">主要关注点\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">典型范围\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">何时变化\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">典型失败\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">关系\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Cp>Anthropic 明确将上下文工程描述为提示工程的自然演进，适用于模型必须与工具、外部数据、消息历史和长时间运行的代理状态协同工作的系统。这种实际区分很有用，因为写得再完美的提示也无法弥补缺失的权威数据或被矛盾状态污染的上下文。\u003C\u002Fp>\n\u003Ch2 id=\"section-23\">上下文工程与检索\u003C\u002Fh2>\n\u003Cp>检索从外部语料库或来源中选择候选信息。上下文工程决定检索之后以及围绕检索发生什么。\u003C\u002Fp>\n\u003Cp>检索器可能返回 30 个段落。重排序器可能将其减少到 10 个。上下文层可能选择四个段落，删除重复项，附加来源\u002F版本元数据，将它们与当前应用状态结合，并将它们放在系统指令之后。\u003C\u002Fp>\n\u003Cp>这就是为什么 RAG 系统可以检索到正确的段落却仍然回答得很差：失败可能发生在上下文组装期间，而不是检索期间。\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--success my-6 rounded-xl border p-5 border-emerald-300 bg-emerald-50 dark:border-emerald-900 dark:bg-emerald-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">检索找到候选；上下文工程构建模型输入\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">正确的检索结果只有在经过过滤、排序、压缩和令牌预算决策后仍然存在，并以可用的形式实际到达模型时才有用。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-28\">上下文工程与记忆\u003C\u002Fh2>\n\u003Cp>记忆是在即时模型调用之外保留的信息，以便以后再次使用。上下文是实际加载到当前调用中的信息。\u003C\u002Fp>\n\u003Cp>记忆系统可能包含数千个事实、笔记或先前的决策。上下文工程选择其中哪些应针对当前任务重新引入。在每一轮都加载所有记忆违背了拥有外部记忆层的目的。\u003C\u002Fp>\n\u003Cp>这种区分对于易变状态变得至关重要。记忆中的项目状态或用户偏好可能有用，但在做出重要决策之前，可能需要重新读取当前的权威状态。\u003C\u002Fp>\n\u003Caside class=\"editorjs-referral my-6\">\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context\" class=\"flex flex-col sm:flex-row gap-4 rounded-xl border border-gray-200 dark:border-gray-700 p-4 transition hover:border-primary-500\">\u003Cdiv class=\"min-w-0 flex-1\">\u003Cstrong class=\"block text-lg text-gray-900 dark:text-gray-100\">AI 代理记忆不是 RAG：如何区分记忆、检索、状态和上下文\u003C\u002Fstrong>\u003Cp class=\"mt-2 text-sm text-gray-600 dark:text-gray-300\">一种实用的架构，区分什么持续存在、什么是当前权威的、什么被检索以及模型实际接收什么。\u003C\u002Fp>\u003Cspan class=\"mt-3 inline-flex text-sm font-medium text-primary-600 dark:text-primary-400\">阅读记忆架构文章 →\u003C\u002Fspan>\u003C\u002Fdiv>\u003C\u002Fa>\u003C\u002Faside>\n\u003Ch2 id=\"section-33\">上下文工程与应用状态\u003C\u002Fh2>\n\u003Cp>应用状态是外部系统的当前状况：账户余额、工单状态、文件版本、工作流阶段、部署状态或任务进度。\u003C\u002Fp>\n\u003Cp>状态可以摘要到上下文中，但摘要不是状态本身。对于重要操作，运行时可能需要在操作前立即重新读取权威系统，而不是信任模型先前可见的快照。\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--warning my-6 rounded-xl border p-5 border-amber-300 bg-amber-50 dark:border-amber-900 dark:bg-amber-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">上下文是快照\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">一旦状态被复制到提示中，它就可能变得过时。上下文工程必须定义易变状态何时需要刷新，以及哪些操作需要新的权威读取。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-37\">工具设计是上下文工程的一部分\u003C\u002Fh2>\n\u003Cp>工具不仅仅是为智能体提供能力。工具名称、描述、模式和结果都会成为模型可见的信息，从而影响决策。\u003C\u002Fp>\n\u003Cp>Anthropic 当前的上下文工程指南强调使用 token 高效的工具，并警告不要使用功能重叠的臃肿工具集。一个连人类都难以区分的工具目录，模型也很难可靠地路由。\u003C\u002Fp>\n\u003Cp>工具输出也需要上下文纪律。当智能体只请求一个错误条件时，却返回整个 20,000 行日志，会消耗注意力，并可能埋没决定性证据。\u003C\u002Fp>\n\u003Ch2 id=\"section-41\">即时上下文与预加载上下文\u003C\u002Fh2>\n\u003Csection class=\"editorjs-comparison my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">提供信息的两种方式\u003C\u002Fh3>\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left dark:border-gray-700 dark:bg-gray-900\">\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">预加载上下文\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">即时上下文\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">方法\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">优势\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">风险\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">适用场景\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Cp>Anthropic 描述了一种混合模式：一些稳定的上下文被预加载，而智能体在运行时检索额外信息。这是一种有用的架构模式，因为并非每个重要事实都值得在上下文窗口中永久驻留。\u003C\u002Fp>\n\u003Ch2 id=\"section-44\">上下文是预算，不是存储系统\u003C\u002Fh2>\n\u003Cp>上下文窗口定义了容量。它并不保证每个 token 都会被同等有效地利用。模型必须在指令、历史、证据、工具和中间状态之间分配注意力。\u003C\u002Fp>\n\u003Cp>因此，实际目标不是“填满窗口”，而是最大化有限注意力预算的效用。\u003C\u002Fp>\n\u003Cp>Anthropic 将类似原则表述为：找到最小的、高信号的 token 集合，以最大化期望行为的概率。OpenAI 的上下文管理指南同样警告，未经整理的历史、冗余的工具结果和嘈杂的检索，即使对大型窗口也可能造成过载。\u003C\u002Fp>\n\u003Ch2 id=\"section-48\">为什么更多上下文可能更糟\u003C\u002Fh2>\n\u003Cp>额外的上下文可能引入无关信息、过时状态、重复证据、矛盾指令或位置竞争。它还可能导致压缩系统丢弃后来变得重要的细节。\u003C\u002Fp>\n\u003Cp>经典的“迷失在中间”研究表明，长上下文模型对信息的利用可能因相关内容出现的位置而异，当决定性信息被放在长输入的中间时，性能往往会下降。\u003C\u002Fp>\n\u003Cp>这并不意味着长上下文本身不好。它意味着窗口内的可用性并不等同于可靠利用。\u003C\u002Fp>\n\u003Ch2 id=\"section-52\">上下文排序应当是有意为之\u003C\u002Fh2>\n\u003Cp>上下文构建也是一个排序问题。关键指令、当前状态、决定性证据和任务特定约束不应被任意拼接。\u003C\u002Fp>\n\u003Cp>并不存在适用于每个模型和任务的通用完美排序。因此，架构应测试重新排序证据是否会改变正确性，以及重要信息在现实上下文变化中是否仍然稳健。\u003C\u002Fp>\n\u003Cp>一个稳定的答案在两条同等有效的段落交换位置后发生剧烈变化，这表明存在上下文敏感性，应当加以衡量而非忽视。\u003C\u002Fp>\n\u003Ch2 id=\"section-56\">冲突的上下文需要明确的优先级\u003C\u002Fh2>\n\u003Cp>模型可能同时接收到旧政策和新政策、记忆中的偏好和当前的明确指令，或缓存的实时状态和实时API结果。系统不应期望模型从行文风格中推断优先级。\u003C\u002Fp>\n\u003Cp>上下文工程应通过来源选择、元数据、排序或明确指令来编码优先级：当前权威状态覆盖过时副本；当前用户的明确指令覆盖较早的推断偏好；已批准的政策取代过时的草案。\u003C\u002Fp>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">冲突\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">首选的上下文规则\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">当前状态与记忆状态\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">刷新并优先使用权威的当前来源。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">当前政策与已被取代的政策\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">包含当前版本；仅在需要历史对比时保留旧版本。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">明确的用户指令与旧的推断偏好\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">优先使用当前的明确指令。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">一手来源与二手摘要\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">对于需要权威性的主张，使用一手来源；摘要可用于辅助解释。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">工具观察与模型先验\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">当工具对该事实具有权威性时，优先使用当前观察到的状态。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">两个未解决的权威来源\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">暴露冲突，而不是编造一个一致的答案。\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-60\">压缩是上下文转换，而非无损存储\u003C\u002Fh2>\n\u003Cp>长期运行的系统最终需要裁剪、总结或压缩历史记录。压缩会创建先前上下文的新表示，使智能体无需重放每个令牌即可继续运行。\u003C\u002Fp>\n\u003Cp>OpenAI的上下文管理示例使用裁剪和压缩来处理长时间运行的会话。Anthropic将压缩描述为在交互接近上下文限制时保持连贯性的主要技术。\u003C\u002Fp>\n\u003Cp>困难之处在于决定哪些内容不能安全移除：未解决的任务、标识符、用户约束、安全边界、架构决策、异常情况、来源出处，以及使先前结论有效的条件。\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--warning my-6 rounded-xl border p-5 border-amber-300 bg-amber-50 dark:border-amber-900 dark:bg-amber-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">摘要可以保留结论却摧毁理由\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">如果压缩保留了“使用方案X”，却丢弃了选择X的原因、测试过的版本，或会使该结论失效的条件，那么后续回答可能在内部保持一致的同时，在外部变得错误。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-65\">保留有效性边界\u003C\u002Fh2>\n\u003Cp>重要结论应附带其仍然成立的条件：版本、日期、范围、假设、来源权威性以及未解决的分歧。\u003C\u002Fp>\n\u003Cp>因此，上下文工程与答案有效性边界相关联。上下文组装器不应剥离决定证据是否仍然适用的元数据。\u003C\u002Fp>\n\u003Caside class=\"editorjs-referral my-6\">\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers\" class=\"flex flex-col sm:flex-row gap-4 rounded-xl border border-gray-200 dark:border-gray-700 p-4 transition hover:border-primary-500\">\u003Cdiv class=\"min-w-0 flex-1\">\u003Cstrong class=\"block text-lg text-gray-900 dark:text-gray-100\">答案有效性边界：相关性与可靠AI答案之间缺失的一层\u003C\u002Fstrong>\u003Cp class=\"mt-2 text-sm text-gray-600 dark:text-gray-300\">一个用于保留AI主张仍然成立时的范围、假设、版本和证据条件的框架。\u003C\u002Fp>\u003Cspan class=\"mt-3 inline-flex text-sm font-medium text-primary-600 dark:text-primary-400\">阅读答案有效性边界 →\u003C\u002Fspan>\u003C\u002Fdiv>\u003C\u002Fa>\u003C\u002Faside>\n\u003Ch2 id=\"section-69\">上下文工程也是安全边界\u003C\u002Fh2>\n\u003Cp>到达模型的数据已经跨越了一个重要的系统边界。因此，上下文组装必须遵守授权、租户隔离、保密性和数据最小化规则。\u003C\u002Fp>\n\u003Cp>检索器在技术上可能找到当前用户无权访问的段落。正确的设计是阻止该段落进入模型上下文，而不是依赖模型忽略它。\u003C\u002Fp>\n\u003Cp>工具输出也可能包含不可信的指令或对抗性内容。上下文工程应保留应用程序指令与外部数据之间的区别，使检索到的文本无法悄然获得指令权威。\u003C\u002Fp>\n\u003Ch2 id=\"section-73\">一个实用的上下文工程架构\u003C\u002Fh2>\n\u003Caside class=\"editorjs-callout editorjs-callout--note my-6 rounded-xl border p-5 border-gray-300 bg-gray-50 dark:border-gray-700 dark:bg-gray-900\u002F40\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">提议的架构模型\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">以下分层是针对生产系统的实用综合方案，并非正式的行业标准。其目的是将信息所有权与临时的面向模型的上下文分开。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">层\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">职责\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">权威系统\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">拥有当前业务\u002F系统状态和官方记录。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">知识来源\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">拥有文档、政策、规范、研究或外部证据。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">记忆存储\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">跨轮次或会话保留选定的信息。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">检索层\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">从外部来源定位与任务相关的候选内容。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">工具\u002F运行时层\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">读取状态、执行操作并返回观察结果。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">上下文组装器\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">选择、过滤、去重、排序并格式化模型可见的信息。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">在组装后的上下文上进行推理和生成。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">验证\u002F评估\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">检查所选上下文和生成的输出是否满足特定任务的要求。\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Cp>即使没有任何模块使用这个确切的名称，上下文组装器在概念上也很重要。在小型应用中，它可能只是普通的应用代码。在大型智能体平台中，它可能结合了会话管理、检索、记忆、工具中间件、压缩和策略执行。\u003C\u002Fp>\n\u003Ch2 id=\"section-77\">一个实用的上下文构建策略\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">规则\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">为什么重要\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">从当前任务出发\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">不要仅仅因为信息之前存在就携带它。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">重新读取易变状态\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">记忆和旧上下文可能已经过时。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">只检索刚好足够的证据\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">大量候选集会稀释决定性信息。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">保留来源元数据\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">版本、日期和权威性决定证据是否仍然适用。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">移除重复内容\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">冗余会消耗 token 而不增加信息。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">对大型工具输出优先使用结构化摘要\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">在保真度允许的情况下，暴露决定性字段而非原始噪声。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">规则与例外保持在一起\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">将规则与其例外分开会造成虚假的确定性。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">明确优先级\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">不要让模型去推断哪个冲突来源胜出。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">将持久状态保留在上下文之外\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">上下文是临时工作记忆，不是数据库。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">用保留测试进行压缩\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">验证标识符、约束、来源和未解决状态是否得以保留。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">测量顺序敏感性\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">正确性不应意外地依赖于任意的文档顺序。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">将上下文评估与模型质量分开\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">更强的模型无法可靠地弥补缺失或未经授权的证据。\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-79\">如何评估上下文工程\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">属性\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">问题\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">示例测试\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">充分性\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">上下文是否包含解决任务所需的一切？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">移除一个证据项，观察答案是否变得缺乏支持。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">相关性\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">有多少上下文对任务是不必要的？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">在添加或移除无关段落时测量质量。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">权威性\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">决定性主张是否基于正确的来源类别？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">注入一个更流畅但非权威的冲突来源。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">新鲜度\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">当前状态是否覆盖过时的副本？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">在上一轮之后更改权威状态并重新运行。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">位置稳健性\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">答案质量是否强烈依赖于证据位置？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">在重复试验中随机化候选顺序。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">冲突处理\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型是否遵循明确的优先级规则？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">将旧状态和新状态一起呈现。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">压缩保留\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">摘要是否保留了约束和有效性边界？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">比较压缩前后的任务表现。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Token 效率\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">额外上下文带来的质量提升是否足以证明延迟\u002F成本合理？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">运行受控的上下文大小消融实验。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">安全性\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">未经授权或对抗性内容能否进入模型上下文？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">测试租户、权限和提示注入边界。\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-81\">上下文组装是一个独立的 RAG 故障层\u003C\u002Fh2>\n\u003Cp>一个 RAG 流水线可能在检索上成功，却在下游失败。相关来源可能出现在第 2 位，但上下文组装器可能将其丢弃、截断、与过时的矛盾材料合并，或超出 token 预算。\u003C\u002Fp>\n\u003Cp>这就是为什么应将检索轨迹与实际发送给模型的上下文进行比较。没有这种比较，上下文故障很容易被误诊为嵌入或模型故障。\u003C\u002Fp>\n\u003Caside class=\"editorjs-referral my-6\">\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method\" class=\"flex flex-col sm:flex-row gap-4 rounded-xl border border-gray-200 dark:border-gray-700 p-4 transition hover:border-primary-500\">\u003Cdiv class=\"min-w-0 flex-1\">\u003Cstrong class=\"block text-lg text-gray-900 dark:text-gray-100\">RAG 失败了——但究竟是哪一层失败了？一种诊断方法\u003C\u002Fstrong>\u003Cp class=\"mt-2 text-sm text-gray-600 dark:text-gray-300\">一种逐层方法，用于区分来源覆盖、检索、排序、上下文组装、生成、证据归因和新鲜度故障。\u003C\u002Fp>\u003Cspan class=\"mt-3 inline-flex text-sm font-medium text-primary-600 dark:text-primary-400\">阅读 RAG 诊断方法 →\u003C\u002Fspan>\u003C\u002Fdiv>\u003C\u002Fa>\u003C\u002Faside>\n\u003Ch2 id=\"section-85\">原始实现证据\u003C\u002Fh2>\n\u003Ch3 id=\"section-86\">真相来源研究引擎：有界研究而非无限上下文\u003C\u002Fh3>\n\u003Cp>真相来源研究引擎将发现、获取、提取、验证、矛盾分析和综合分为有界的研究阶段，而不是将一个庞大的研究任务和所有累积材料发送到单次模型调用中。\u003C\u002Fp>\n\u003Cp>其证据模型将来源、工件、主张、关系、矛盾和来源信息存储在模型上下文之外。模型可以接收当前研究步骤所需的子集，而持久证据保留在外部存储中。\u003C\u002Fp>\n\u003Cp>这是一个具体的上下文工程模式：持久研究状态存在于模型窗口之外；活跃的模型上下文针对当前阶段重新构建。\u003C\u002Fp>\n\u003Ch3 id=\"section-90\">Aaasaasa AI 客户端：运行时、权限和上下文是相互独立的关注点\u003C\u002Fh3>\n\u003Cp>Aaasaasa AI Client 将提供商\u002F模型选择、运行时位置、工作区权限、本地资源和工具访问分离开来。这可以防止模型上下文成为授权或应用状态的所有者。\u003C\u002Fp>\n\u003Cp>直接聊天和代理运行时可以具有不同的工具能力。工作区权限配置文件由运行时强制执行，而不仅仅是在自然语言上下文中描述。这一区别很重要：上下文可以告诉模型它应该做什么，而运行时仍然必须强制执行它实际被允许做什么。\u003C\u002Fp>\n\u003Cp>这里的实现证据是架构分离，而不是声称本文中描述的每一种高级上下文管理技术都已经实现。\u003C\u002Fp>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">实现模式\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">上下文工程经验\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">外部证据存储\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">持久知识不需要保留在模型窗口中。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">有界研究阶段\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">不同步骤可以接收不同的上下文，而不是累积一个巨大的历史记录。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">上下文之外的声明 + 来源\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">证据身份在临时推理状态之外仍然存在。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">运行时强制执行的权限\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">安全授权不依赖于模型记住指令。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">分离本地\u002F提供商\u002F模型\u002F运行时概念\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">上下文只是更广泛 AI 应用架构中的一层。\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Caside class=\"editorjs-callout editorjs-callout--note my-6 rounded-xl border p-5 border-gray-300 bg-gray-50 dark:border-gray-700 dark:bg-gray-900\u002F40\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">证据边界\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">这些实现支持持久状态、检索、运行时控制和面向模型的上下文之间的架构分离。它们并不是作为基准证明来呈现，以表明某一种上下文策略普遍最优。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-96\">常见的上下文工程失败模式\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">失败模式\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">出了什么问题\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">永远重放整个对话\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">旧假设、重复和 token 增长会压过当前意图。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">把每个检索结果都放进提示词\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">噪声、重复和冲突版本会稀释决定性证据。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">把记忆当作当前状态\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">过时信息会悄悄取代权威的实时状态。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">返回原始工具输出\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">大型日志或响应会消耗注意力，却不增加决策价值。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">用模糊名称隐藏工具描述\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型无法可靠地决定应使用哪种能力。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">没有保留测试就进行压缩\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">关键约束、标识符或例外会消失。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">混合指令和不可信数据\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">外部内容可能被解释为更高权威的指令。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">对每个任务使用同一个静态上下文模板\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">不同任务会收到无关信息，并错过任务特定证据。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">忽略来源版本\u002F日期\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">过时但相关的证据可能压过当前权威状态。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">把更大的上下文窗口当作质量保证\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">容量增加了，但注意力和冲突问题仍然存在。\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-98\">常见误解\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">误解\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">纠正\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“上下文工程只是换了个名字的提示词工程。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">提示词只是一个组成部分；上下文工程还涵盖检索、记忆、状态、工具结果、历史和压缩。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“上下文就是聊天历史。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">历史只是可能的上下文来源之一。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“上下文越多总是越好。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">额外信息可能降低信号、引入冲突并增加成本。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“如果检索找到了它，模型就看到了它。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">检索到的候选内容可能在推理前被过滤、截断或省略。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“长上下文消除了对 RAG 的需求。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">大窗口增加了容量，但并不能解决新鲜度、权威性、权限或动态检索问题。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“记忆应该总是被加载。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">记忆应根据当前任务进行选择。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“摘要会保留所有重要内容。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">除非明确评估保留情况，否则压缩是有损的。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“指令可以强制执行权限。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">授权必须由运行时\u002F应用控制来强制执行，而不仅仅由上下文执行。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“一种上下文配方适用于所有模型。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">上下文敏感性因模型、任务、语料库和运行时而异。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“上下文工程只适用于代理。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">代理会放大这种需求，但普通 RAG 和对话应用也需要上下文构建。\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-100\">一个实用的上下文工程序列\u003C\u002Fh2>\n\u003Csection class=\"editorjs-process my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">从当前决策反向构建上下文\u003C\u002Fh3>\u003Cdiv class=\"grid grid-cols-1 md:grid-cols-2 xl:grid-cols-3 gap-4\">\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">1\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">1. 定义下一个模型决策\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">明确模型在这一步必须回答、分类、规划或选择什么。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">2\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">2. 识别所需事实和约束\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">列出能够实质性改变结果的最小状态、规则、证据和指令。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">3\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">3. 解决权威性和权限问题\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">确定哪些来源是当前的、权威的，并且当前主体可以访问。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">4\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">4. 按需检索或读取\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">获取必要证据和易变状态，而不是依赖过时上下文。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">5\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">5. 降低噪声\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">去重、摘要或选择段落，同时不丢弃决定性例外或来源。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">6\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">6. 结构化并排序\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">使指令、当前状态、证据和工具观察结果可以区分。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">7\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">7. 适配 token 预算\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">优先使用高信号上下文，并将持久信息移到窗口之外。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">8\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">8. 运行模型\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">在组装好的上下文上执行推理。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">9\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">9. 观察失败\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">记录问题来自缺失、过时、噪声、冲突还是排序不佳的上下文。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">10\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">10. 在模型\u002F运行时变更后重新评估\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">上下文策略只对它所测试过的模型、工具和工作负载有效。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-102\">上下文工程检查清单\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">问题\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">预期答案\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型下一步将做出什么确切决策？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">一个有边界的任务，而不是模糊的长期目标。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">哪些信息能够实质性改变该决策？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">明确的最小证据\u002F状态集合。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">现在哪些数据是权威的？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">当前来源\u002F版本和新鲜度规则。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">哪些数据是可选的背景信息？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">与决定性证据分开。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">什么内容不得进入上下文？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">未授权、不必要或过于敏感的数据。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">哪些记忆项是相关的？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">按任务选择，而不是自动重放。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">哪些工具输出应被精简？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">大型响应被转换为与决策相关的形式。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">哪些约束必须在压缩后保留？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">标识符、例外、义务、未解决状态和来源。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">优先级如何表示？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">当前\u002F权威信息可以可靠地覆盖过时或较弱来源。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">你如何知道上下文失败了？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">存在上下文特定的评估和追踪。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">答案可以复现吗？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">在适当情况下，模型输入或可重建的上下文追踪可用。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">更强或更大的模型会改变策略吗？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">上下文策略具备版本意识，并根据经验重新评估。\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-104\">边缘情况和限制\u003C\u002Fh2>\n\u003Cp>有些任务足够简单，以至于上下文工程简化为一个简短的系统提示词和一条用户消息。添加检索、记忆和压缩只会引入不必要的架构。\u003C\u002Fp>\n\u003Cp>有些任务需要高召回率，并且可能有意在后续综合之前包含更多上下文。研究、发现和法律审查可能更倾向于避免遗漏，而不是最小化 token 数量。\u003C\u002Fp>\n\u003Cp>有些信息在使用前绝不应被摘要。精确合同、代码、密码材料、数值记录和监管文本可能需要逐字或结构化检索，因为压缩可能会改变含义。\u003C\u002Fp>\n\u003Cp>长上下文行为在不同模型之间差异很大。在一个模型、上下文长度或工具框架上验证过的策略，不应自动转移到另一个上。\u003C\u002Fp>\n\u003Cp>模型仍然可能忽略或误解优秀的上下文。上下文工程改善的是信息环境；它并不保证推理的正确性。\u003C\u002Fp>\n\u003Ch2 id=\"section-110\">什么会改变这个答案？\u003C\u002Fh2>\n\u003Cp>未来的模型可能会对长上下文、位置效应和冲突信息变得更加稳健。这可能会减少所需的人工整理工作量。\u003C\u002Fp>\n\u003Cp>架构上的区分仍然有用，因为权限、时效性、记忆持久性、来源权威性和外部应用状态无论上下文窗口大小如何都存在于模型之外。\u003C\u002Fp>\n\u003Cp>预加载上下文与即时上下文之间的推荐平衡也会随延迟要求、工具可靠性、语料库规模、模型成本以及底层信息的动态程度而变化。\u003C\u002Fp>\n\u003Ch2 id=\"section-114\">相关权威知识\u003C\u002Fh2>\n\u003Cp>上下文工程介于检索与生成之间。RAG 解释了外部知识如何被检索；R01 区分了嵌入、向量搜索和重排序；上下文工程解释了最终到达模型的是什么。\u003C\u002Fp>\n\u003Caside class=\"editorjs-referral my-6\">\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works\" class=\"flex flex-col sm:flex-row gap-4 rounded-xl border border-gray-200 dark:border-gray-700 p-4 transition hover:border-primary-500\">\u003Cdiv class=\"min-w-0 flex-1\">\u003Cstrong class=\"block text-lg text-gray-900 dark:text-gray-100\">什么是 RAG？对其工作原理的最简单解释\u003C\u002Fstrong>\u003Cp class=\"mt-2 text-sm text-gray-600 dark:text-gray-300\">理解外部知识如何在生成之前被提供给模型的检索基础。\u003C\u002Fp>\u003Cspan class=\"mt-3 inline-flex text-sm font-medium text-primary-600 dark:text-primary-400\">阅读 RAG 基础 →\u003C\u002Fspan>\u003C\u002Fdiv>\u003C\u002Fa>\u003C\u002Faside>\n\u003Cp>真相来源架构回答的是另一个问题：不是上下文中存在哪些信息，而是哪个来源被授权来确立一项主张。\u003C\u002Fp>\n\u003Cp>现有文章《为什么更多上下文会让 AI 答案更糟》是这一权威定义的诊断性配套文章。它侧重于上下文污染、位置效应、top-k 增长、压缩损失和答案退化，而不是重新定义上下文工程本身。\u003C\u002Fp>\n\u003Ch2 id=\"section-119\">常见问题\u003C\u002Fh2>\n\u003Csection class=\"editorjs-faq my-6 rounded-xl border border-gray-200 p-5 dark:border-gray-700\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">上下文工程常见问题\u003C\u002Fh3>\u003Cdiv id=\"faq1\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">什么是上下文工程？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">上下文工程是对语言模型在推理时接收到的信息进行的设计和运行时管理，包括指令、历史记录、检索到的证据、记忆、状态、工具和工具结果。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq2\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">上下文工程与提示工程有何不同？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">提示工程侧重于指令和示例的编写方式。上下文工程包括提示，但也决定将哪些外部信息、状态、历史记录、记忆和工具观察结果置于提示周围。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq3\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">RAG 与上下文工程是一回事吗？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">不是。RAG 检索外部信息。上下文工程决定检索到的信息如何被过滤、如何与其他状态结合，以及如何实际传递给模型。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq4\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">记忆与上下文是一回事吗？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">不是。记忆将信息持久保存在当前模型调用之外。上下文是加载到当前推理中的那部分信息。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq5\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">为什么更多上下文会让答案更糟？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">额外的上下文可能引入噪声、过时状态、相互冲突的证据、重复内容和位置竞争。大的上下文容量并不保证每个 token 都能被同样可靠地利用。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq6\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">什么是上下文压缩？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">压缩是将累积的历史记录总结或转换为更小的表示形式，以便长时间运行的系统可以继续运行而无需重放此前的每一个 token。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq7\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">当前应用状态应该存储在上下文中吗？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">它可以被表示在上下文中用于推理，但具有后果的操作通常应重新读取权威来源，因为上下文快照可能会过时。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq8\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">上下文工程只对 AI 智能体有需要吗？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">不是。智能体使上下文管理更加动态，但 RAG 系统、助手、副驾驶和多轮应用也需要有意识地构建上下文。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-121\">术语表\u003C\u002Fh2>\n\u003Csection class=\"editorjs-glossary my-6 rounded-xl border border-gray-200 dark:border-gray-700 p-5\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">关键上下文工程术语\u003C\u002Fh3>\u003Cdl>\u003Cdiv id=\"context-engineering\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">上下文工程\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">为特定推理步骤向语言模型提供的信息进行的设计和运行时管理。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"context-window\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">上下文窗口\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">模型对输入的有限 token 容量，并且根据模型接口的不同，还包括相关的生成 token 或活动序列。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"prompt-engineering\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">提示工程\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">旨在引出有用模型行为的指令、示例和提示结构的设计。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"context-assembly\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">上下文组装\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">在推理之前选择、过滤、排序和格式化模型可见信息的过程。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"just-in-time-retrieval\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">即时检索\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">在当前任务需要时动态加载信息，而不是预加载所有可能相关的数据。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"compaction\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">压缩\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">将累积的上下文缩减为更小的表示形式，同时尝试保留未来步骤所需的信息。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"context-pollution\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">上下文污染\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">由无关、过时、矛盾或冗余信息占据模型工作上下文而导致的性能下降。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"application-state\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">应用状态\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">外部系统、工作流或领域独立于模型上下文而存在的当前权威状况。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"memory\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">记忆\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">存储在即时模型调用之外、可能用于后续轮次或会话的信息。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"retrieved-context\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">检索上下文\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">由检索系统选择并全部或部分提供给模型的外部信息。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"position-robustness\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">位置稳健性\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">当相关上下文的位置或顺序发生变化时，模型正确性保持稳定的程度。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"validity-boundary\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">有效性边界\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">结论仍然得到支持的范围、时间、假设、版本和证据条件。\u003C\u002Fdd>\u003C\u002Fdiv>\u003C\u002Fdl>\u003C\u002Fsection>\n\u003Ch2 id=\"section-123\">结论\u003C\u002Fh2>\n\u003Cp>上下文工程是决定模型在回答之前能看到什么的那一层。这使它比提示更广泛，并处于检索的下游，同时又有别于持久记忆和权威应用状态。\u003C\u002Fp>\n\u003Cp>强大的上下文架构不会将上下文窗口视为数据库。它将持久状态和知识保留在模型之外，加载当前决策所需的内容，保留权威性和来源，去除不必要的噪声，并在需要时刷新易变信息。\u003C\u002Fp>\n\u003Cp>因此，实际目标不是最大上下文。而是为下一次模型决策提供最小充分、高信号、正确授权且保持有效性的上下文。\u003C\u002Fp>\n\u003Ch2 id=\"section-127\">主要来源与当前指南\u003C\u002Fh2>\n\u003Cp>以下来源支持当前的上下文工程术语、长上下文行为以及可操作的上下文管理模式。项目部分明确属于实现证据，而非普遍性主张。\u003C\u002Fp>\n\u003Ca href=\"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Feffective-context-engineering-for-ai-agents\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">Anthropic — 面向 AI 智能体的有效上下文工程\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">官方工程指南，定义了上下文工程、即时检索、压缩、结构化记忆以及面向智能体的上下文策展。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fdevelopers.openai.com\u002Fcookbook\u002Fexamples\u002Fagents_sdk\u002Fsession_memory\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">OpenAI — 上下文工程：使用会话进行短期记忆管理\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">官方 cookbook 指南，涉及长时间运行的智能体会话的上下文管理、裁剪和压缩。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fagents\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">OpenAI — 智能体指南\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">OpenAI 当前面向开发者的指南，涉及智能体运行时、跨步骤上下文以及编排归属。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2307.03172\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">迷失在中间：语言模型如何使用长上下文\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">研究表明，长上下文模型的性能可能在很大程度上取决于相关信息在输入中的位置。\u003C\u002Fp>\u003C\u002Fa>",{"time":212,"blocks":213,"version":1406},1791481033118,[214,220,228,235,242,250,255,260,265,270,275,280,285,290,319,324,329,334,339,393,398,433,438,443,448,453,458,465,470,475,480,485,494,499,504,509,515,520,525,530,535,540,569,574,579,584,589,594,599,604,609,614,619,624,629,634,639,644,649,675,680,685,690,695,701,706,711,716,724,729,734,739,744,749,755,787,792,797,841,846,891,896,901,906,914,919,924,929,934,939,944,949,954,959,982,988,993,1031,1036,1074,1079,1115,1120,1163,1168,1173,1178,1183,1188,1193,1198,1203,1208,1213,1218,1223,1231,1236,1241,1246,1284,1289,1339,1344,1349,1354,1359,1364,1369,1379,1388,1397],{"id":215,"data":216,"type":218,"tunes":219},"intro",{"text":217},"上下文工程是指设计语言模型在推理时接收哪些信息、以何种形式、按何种顺序以及保留多长时间。它比提示工程更广泛，因为模型上下文可以包括系统指令、用户消息、检索到的文档、工具结果、记忆、当前应用状态、示例、结构化数据和中间产物。目标不是最大化令牌数量，而是构建最小的有用上下文，保留当前任务所需的信息、约束和证据。","paragraph",{},{"id":221,"data":222,"type":226,"tunes":227},"direct",{"body":223,"title":224,"variant":225},"提示工程问的是\u003Cstrong>我们应该如何指示模型？\u003C\u002Fstrong>上下文工程问的是\u003Cstrong>模型现在应该知道什么，以及这些信息应该如何组装？\u003C\u002Fstrong>\u003Cbr>\u003Cbr>因此，当检索、记忆、状态管理、工具设计、历史裁剪、压缩和排序决定了模型在产生下一个输出之前可用的令牌时，它们就是上下文工程机制。","直接回答","info","callout",{},{"id":229,"data":230,"type":226,"tunes":234},"boundary",{"body":231,"title":232,"variant":233},"一个系统可以知道某件事，而不把它放入当前上下文。它可以在模型窗口之外记住某件事。它可以检索一个文档，但后来将其排除在最终提示之外。模型只能直接使用到达当前推理的上下文。","上下文不等同于知识或记忆","warning",{},{"id":236,"data":237,"type":226,"tunes":241},"current",{"body":238,"title":239,"variant":240},"上下文工程现已成为主要AI工程指南中确立的实用术语，但它并不是一个具有单一强制架构的正式标准。Anthropic将其描述为为推理策划和维护最优令牌集；OpenAI当前的智能体指南将会话上下文、裁剪和压缩视为长期运行系统的明确工程关注点。","当前来源说明——2026年10月8日","note",{},{"id":243,"data":244,"type":248,"tunes":249},"toc",{"title":245,"maxLevel":246,"minLevel":247},"目录",3,2,"tableOfContents",{},{"id":251,"data":252,"type":42,"tunes":254},"h-meaning",{"text":253,"level":247},"上下文工程真正意味着什么",{},{"id":256,"data":257,"type":218,"tunes":259},"p-meaning-1",{"text":258},"每次模型调用都是在一个临时工作环境中进行的：当前指令、消息、检索到的证据、工具输出和状态，这些内容都适合放入活动上下文窗口。上下文工程就是有意构建该环境的学科。",{},{"id":261,"data":262,"type":218,"tunes":264},"p-meaning-2",{"text":263},"关键词是有意。一个朴素系统只是把它拥有的一切拼接起来：完整历史、所有检索到的文档、每个工具响应和大型系统提示。一个经过上下文工程的系统会决定当前决策需要哪些信息，以及哪些信息应留在窗口之外直到需要时再使用。",{},{"id":266,"data":267,"type":218,"tunes":269},"p-meaning-3",{"text":268},"这使上下文工程部分成为信息架构问题，部分成为运行时问题，部分成为评估问题。设计必须决定什么可以进入上下文、它来自哪里、哪个版本是当前的、冲突如何解决、保留多少细节以及结果如何测试。",{},{"id":271,"data":272,"type":42,"tunes":274},"h-simple",{"text":273,"level":247},"最简单的例子",{},{"id":276,"data":277,"type":218,"tunes":279},"p-simple-1",{"text":278},"想象一个内部支持助手。用户问：“这个客户可以免费取消吗？”",{},{"id":281,"data":282,"type":218,"tunes":284},"p-simple-2",{"text":283},"模型可能需要五样东西：当前取消政策、客户当前合同类型、合同生效日期、相关例外规则以及用户的授权范围。",{},{"id":286,"data":287,"type":218,"tunes":289},"p-simple-3",{"text":288},"它不一定需要整个客户数据库、完整政策档案、每一次之前的对话或每一张支持工单。上下文工程就是选择并组装这五个有用部分，同时排除无关信息的过程。",{},{"id":291,"data":292,"type":317,"tunes":318},"simple-flow",{"steps":293,"title":315,"orientation":316},[294,297,300,303,306,309,312],{"label":295,"description":296},"1. 理解任务","对当前问题需要什么以及哪些信息类型可能影响答案进行分类。",{"label":298,"description":299},"2. 解析权威状态","读取不应从记忆中猜测的当前应用或业务状态。",{"label":301,"description":302},"3. 检索支持性知识","找到与特定任务相关的政策、文档或外部证据。",{"label":304,"description":305},"4. 应用资格和权限","排除当前用户或运行时不允许暴露给模型的数据。",{"label":307,"description":308},"5. 精简和结构化","删除重复内容，选择有用摘录，并保留关键元数据、条件和例外。",{"label":310,"description":311},"6. 排列上下文顺序","将指令、当前状态和决定性证据放在模型能够一致使用它们的位置。",{"label":313,"description":314},"7. 运行推理","模型接收组装好的上下文，并产生下一个答案或行动建议。","从应用状态到模型上下文","auto","processFlow",{},{"id":320,"data":321,"type":42,"tunes":323},"h-stops",{"text":322,"level":247},"简单例子止步之处",{},{"id":325,"data":326,"type":218,"tunes":328},"p-stops-1",{"text":327},"真实系统更困难，因为某一步所需的信息可能在执行开始前并不已知。智能体可以通过工具发现新事实、创建中间文件、接收不断变化的外部状态，或跨越比一个上下文窗口更长的任务。",{},{"id":330,"data":331,"type":218,"tunes":333},"p-stops-2",{"text":332},"因此，上下文工程变得动态。第12步的上下文不应只是第1步上下文加上十一层累积输出。它应反映当前任务状态、仍然重要的决策以及下一步行动所需的证据。",{},{"id":335,"data":336,"type":42,"tunes":338},"h-anatomy",{"text":337,"level":247},"什么可以进入模型上下文？",{},{"id":340,"data":341,"type":391,"tunes":392},"anatomy-table",{"content":342,"stretched":43,"withHeadings":14},[343,347,351,355,359,363,367,371,375,379,383,387],[344,345,346],"上下文组件","用途","典型风险",[348,349,350],"系统\u002F开发者指令","定义角色、约束、策略和行为","过于模糊、自相矛盾或充斥着脆弱的逻辑",[352,353,354],"当前用户请求","定义即时任务和意图","存在歧义或与先前历史冲突",[356,357,358],"对话历史","保持跨轮次的连续性","过时的假设、重复和令牌增长",[360,361,362],"检索到的文档","提供外部知识\u002F证据","不相关、版本过时、权威性弱或重复",[364,365,366],"当前应用状态","提供易变的业务\u002F系统事实","使用缓存或记忆中的状态而非当前权威状态",[368,369,370],"工具定义","告知模型存在哪些能力以及如何调用它们","工具过多重叠或模式过于冗长",[372,373,374],"工具结果","将来自环境的观察带入循环","大量嘈杂输出、不可信内容或过时的观察",[376,377,378],"记忆","重新引入先前交互中的选定信息","过时、不正确的泛化或过度个性化",[380,381,382],"示例","展示期望的行为","过多的边缘案例可能挤占当前任务",[384,385,386],"中间产物","承载计划、摘要、代码、计算或笔记","旧的中间状态可能被误认为最终真相",[388,389,390],"策略\u002F护栏","定义禁止或受约束的行为","与业务逻辑冲突或存在隐藏的执行缺口","table",{},{"id":394,"data":395,"type":42,"tunes":397},"h-prompt",{"text":396,"level":247},"上下文工程与提示工程",{},{"id":399,"data":400,"type":431,"tunes":432},"prompt-comparison",{"rows":401,"title":423,"layout":391,"columns":424},[402,407,411,415,419],{"id":403,"label":404,"values":405},"focus","主要关注点",[406,406],"",{"id":408,"label":409,"values":410},"scope","典型范围",[406,406],{"id":412,"label":413,"values":414},"timing","何时变化",[406,406],{"id":416,"label":417,"values":418},"failure","典型失败",[406,406],{"id":420,"label":421,"values":422},"relationship","关系",[406,406],"提示工程和上下文工程解决不同层面的问题",[425,428],{"id":426,"label":427},"prompt","提示工程",{"id":429,"label":430},"context","上下文工程","comparison",{},{"id":434,"data":435,"type":218,"tunes":437},"p-prompt-1",{"text":436},"Anthropic 明确将上下文工程描述为提示工程的自然演进，适用于模型必须与工具、外部数据、消息历史和长时间运行的代理状态协同工作的系统。这种实际区分很有用，因为写得再完美的提示也无法弥补缺失的权威数据或被矛盾状态污染的上下文。",{},{"id":439,"data":440,"type":42,"tunes":442},"h-retrieval",{"text":441,"level":247},"上下文工程与检索",{},{"id":444,"data":445,"type":218,"tunes":447},"p-ret-1",{"text":446},"检索从外部语料库或来源中选择候选信息。上下文工程决定检索之后以及围绕检索发生什么。",{},{"id":449,"data":450,"type":218,"tunes":452},"p-ret-2",{"text":451},"检索器可能返回 30 个段落。重排序器可能将其减少到 10 个。上下文层可能选择四个段落，删除重复项，附加来源\u002F版本元数据，将它们与当前应用状态结合，并将它们放在系统指令之后。",{},{"id":454,"data":455,"type":218,"tunes":457},"p-ret-3",{"text":456},"这就是为什么 RAG 系统可以检索到正确的段落却仍然回答得很差：失败可能发生在上下文组装期间，而不是检索期间。",{},{"id":459,"data":460,"type":226,"tunes":464},"retrieval-boundary",{"body":461,"title":462,"variant":463},"正确的检索结果只有在经过过滤、排序、压缩和令牌预算决策后仍然存在，并以可用的形式实际到达模型时才有用。","检索找到候选；上下文工程构建模型输入","success",{},{"id":466,"data":467,"type":42,"tunes":469},"h-memory",{"text":468,"level":247},"上下文工程与记忆",{},{"id":471,"data":472,"type":218,"tunes":474},"p-memory-1",{"text":473},"记忆是在即时模型调用之外保留的信息，以便以后再次使用。上下文是实际加载到当前调用中的信息。",{},{"id":476,"data":477,"type":218,"tunes":479},"p-memory-2",{"text":478},"记忆系统可能包含数千个事实、笔记或先前的决策。上下文工程选择其中哪些应针对当前任务重新引入。在每一轮都加载所有记忆违背了拥有外部记忆层的目的。",{},{"id":481,"data":482,"type":218,"tunes":484},"p-memory-3",{"text":483},"这种区分对于易变状态变得至关重要。记忆中的项目状态或用户偏好可能有用，但在做出重要决策之前，可能需要重新读取当前的权威状态。",{},{"id":486,"data":487,"type":492,"tunes":493},"ref-memory",{"url":488,"title":489,"excerpt":490,"ctaLabel":491},"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context","AI 代理记忆不是 RAG：如何区分记忆、检索、状态和上下文","一种实用的架构，区分什么持续存在、什么是当前权威的、什么被检索以及模型实际接收什么。","阅读记忆架构文章","referralArticle",{},{"id":495,"data":496,"type":42,"tunes":498},"h-state",{"text":497,"level":247},"上下文工程与应用状态",{},{"id":500,"data":501,"type":218,"tunes":503},"p-state-1",{"text":502},"应用状态是外部系统的当前状况：账户余额、工单状态、文件版本、工作流阶段、部署状态或任务进度。",{},{"id":505,"data":506,"type":218,"tunes":508},"p-state-2",{"text":507},"状态可以摘要到上下文中，但摘要不是状态本身。对于重要操作，运行时可能需要在操作前立即重新读取权威系统，而不是信任模型先前可见的快照。",{},{"id":510,"data":511,"type":226,"tunes":514},"state-rule",{"body":512,"title":513,"variant":233},"一旦状态被复制到提示中，它就可能变得过时。上下文工程必须定义易变状态何时需要刷新，以及哪些操作需要新的权威读取。","上下文是快照",{},{"id":516,"data":517,"type":42,"tunes":519},"h-tools",{"text":518,"level":247},"工具设计是上下文工程的一部分",{},{"id":521,"data":522,"type":218,"tunes":524},"p-tools-1",{"text":523},"工具不仅仅是为智能体提供能力。工具名称、描述、模式和结果都会成为模型可见的信息，从而影响决策。",{},{"id":526,"data":527,"type":218,"tunes":529},"p-tools-2",{"text":528},"Anthropic 当前的上下文工程指南强调使用 token 高效的工具，并警告不要使用功能重叠的臃肿工具集。一个连人类都难以区分的工具目录，模型也很难可靠地路由。",{},{"id":531,"data":532,"type":218,"tunes":534},"p-tools-3",{"text":533},"工具输出也需要上下文纪律。当智能体只请求一个错误条件时，却返回整个 20,000 行日志，会消耗注意力，并可能埋没决定性证据。",{},{"id":536,"data":537,"type":42,"tunes":539},"h-jit",{"text":538,"level":247},"即时上下文与预加载上下文",{},{"id":541,"data":542,"type":431,"tunes":568},"jit-comparison",{"rows":543,"title":560,"layout":391,"columns":561},[544,548,552,556],{"id":545,"label":546,"values":547},"method","方法",[406,406],{"id":549,"label":550,"values":551},"strength","优势",[406,406],{"id":553,"label":554,"values":555},"risk","风险",[406,406],{"id":557,"label":558,"values":559},"best","适用场景",[406,406],"提供信息的两种方式",[562,565],{"id":563,"label":564},"preload","预加载上下文",{"id":566,"label":567},"jit","即时上下文",{},{"id":570,"data":571,"type":218,"tunes":573},"p-jit-1",{"text":572},"Anthropic 描述了一种混合模式：一些稳定的上下文被预加载，而智能体在运行时检索额外信息。这是一种有用的架构模式，因为并非每个重要事实都值得在上下文窗口中永久驻留。",{},{"id":575,"data":576,"type":42,"tunes":578},"h-budget",{"text":577,"level":247},"上下文是预算，不是存储系统",{},{"id":580,"data":581,"type":218,"tunes":583},"p-budget-1",{"text":582},"上下文窗口定义了容量。它并不保证每个 token 都会被同等有效地利用。模型必须在指令、历史、证据、工具和中间状态之间分配注意力。",{},{"id":585,"data":586,"type":218,"tunes":588},"p-budget-2",{"text":587},"因此，实际目标不是“填满窗口”，而是最大化有限注意力预算的效用。",{},{"id":590,"data":591,"type":218,"tunes":593},"p-budget-3",{"text":592},"Anthropic 将类似原则表述为：找到最小的、高信号的 token 集合，以最大化期望行为的概率。OpenAI 的上下文管理指南同样警告，未经整理的历史、冗余的工具结果和嘈杂的检索，即使对大型窗口也可能造成过载。",{},{"id":595,"data":596,"type":42,"tunes":598},"h-more",{"text":597,"level":247},"为什么更多上下文可能更糟",{},{"id":600,"data":601,"type":218,"tunes":603},"p-more-1",{"text":602},"额外的上下文可能引入无关信息、过时状态、重复证据、矛盾指令或位置竞争。它还可能导致压缩系统丢弃后来变得重要的细节。",{},{"id":605,"data":606,"type":218,"tunes":608},"p-more-2",{"text":607},"经典的“迷失在中间”研究表明，长上下文模型对信息的利用可能因相关内容出现的位置而异，当决定性信息被放在长输入的中间时，性能往往会下降。",{},{"id":610,"data":611,"type":218,"tunes":613},"p-more-3",{"text":612},"这并不意味着长上下文本身不好。它意味着窗口内的可用性并不等同于可靠利用。",{},{"id":615,"data":616,"type":42,"tunes":618},"h-order",{"text":617,"level":247},"上下文排序应当是有意为之",{},{"id":620,"data":621,"type":218,"tunes":623},"p-order-1",{"text":622},"上下文构建也是一个排序问题。关键指令、当前状态、决定性证据和任务特定约束不应被任意拼接。",{},{"id":625,"data":626,"type":218,"tunes":628},"p-order-2",{"text":627},"并不存在适用于每个模型和任务的通用完美排序。因此，架构应测试重新排序证据是否会改变正确性，以及重要信息在现实上下文变化中是否仍然稳健。",{},{"id":630,"data":631,"type":218,"tunes":633},"p-order-3",{"text":632},"一个稳定的答案在两条同等有效的段落交换位置后发生剧烈变化，这表明存在上下文敏感性，应当加以衡量而非忽视。",{},{"id":635,"data":636,"type":42,"tunes":638},"h-conflict",{"text":637,"level":247},"冲突的上下文需要明确的优先级",{},{"id":640,"data":641,"type":218,"tunes":643},"p-conflict-1",{"text":642},"模型可能同时接收到旧政策和新政策、记忆中的偏好和当前的明确指令，或缓存的实时状态和实时API结果。系统不应期望模型从行文风格中推断优先级。",{},{"id":645,"data":646,"type":218,"tunes":648},"p-conflict-2",{"text":647},"上下文工程应通过来源选择、元数据、排序或明确指令来编码优先级：当前权威状态覆盖过时副本；当前用户的明确指令覆盖较早的推断偏好；已批准的政策取代过时的草案。",{},{"id":650,"data":651,"type":391,"tunes":674},"conflict-table",{"content":652,"stretched":43,"withHeadings":14},[653,656,659,662,665,668,671],[654,655],"冲突","首选的上下文规则",[657,658],"当前状态与记忆状态","刷新并优先使用权威的当前来源。",[660,661],"当前政策与已被取代的政策","包含当前版本；仅在需要历史对比时保留旧版本。",[663,664],"明确的用户指令与旧的推断偏好","优先使用当前的明确指令。",[666,667],"一手来源与二手摘要","对于需要权威性的主张，使用一手来源；摘要可用于辅助解释。",[669,670],"工具观察与模型先验","当工具对该事实具有权威性时，优先使用当前观察到的状态。",[672,673],"两个未解决的权威来源","暴露冲突，而不是编造一个一致的答案。",{},{"id":676,"data":677,"type":42,"tunes":679},"h-compaction",{"text":678,"level":247},"压缩是上下文转换，而非无损存储",{},{"id":681,"data":682,"type":218,"tunes":684},"p-comp-1",{"text":683},"长期运行的系统最终需要裁剪、总结或压缩历史记录。压缩会创建先前上下文的新表示，使智能体无需重放每个令牌即可继续运行。",{},{"id":686,"data":687,"type":218,"tunes":689},"p-comp-2",{"text":688},"OpenAI的上下文管理示例使用裁剪和压缩来处理长时间运行的会话。Anthropic将压缩描述为在交互接近上下文限制时保持连贯性的主要技术。",{},{"id":691,"data":692,"type":218,"tunes":694},"p-comp-3",{"text":693},"困难之处在于决定哪些内容不能安全移除：未解决的任务、标识符、用户约束、安全边界、架构决策、异常情况、来源出处，以及使先前结论有效的条件。",{},{"id":696,"data":697,"type":226,"tunes":700},"compaction-rule",{"body":698,"title":699,"variant":233},"如果压缩保留了“使用方案X”，却丢弃了选择X的原因、测试过的版本，或会使该结论失效的条件，那么后续回答可能在内部保持一致的同时，在外部变得错误。","摘要可以保留结论却摧毁理由",{},{"id":702,"data":703,"type":42,"tunes":705},"h-validity",{"text":704,"level":247},"保留有效性边界",{},{"id":707,"data":708,"type":218,"tunes":710},"p-validity-1",{"text":709},"重要结论应附带其仍然成立的条件：版本、日期、范围、假设、来源权威性以及未解决的分歧。",{},{"id":712,"data":713,"type":218,"tunes":715},"p-validity-2",{"text":714},"因此，上下文工程与答案有效性边界相关联。上下文组装器不应剥离决定证据是否仍然适用的元数据。",{},{"id":717,"data":718,"type":492,"tunes":723},"ref-avb",{"url":719,"title":720,"excerpt":721,"ctaLabel":722},"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers","答案有效性边界：相关性与可靠AI答案之间缺失的一层","一个用于保留AI主张仍然成立时的范围、假设、版本和证据条件的框架。","阅读答案有效性边界",{},{"id":725,"data":726,"type":42,"tunes":728},"h-security",{"text":727,"level":247},"上下文工程也是安全边界",{},{"id":730,"data":731,"type":218,"tunes":733},"p-sec-1",{"text":732},"到达模型的数据已经跨越了一个重要的系统边界。因此，上下文组装必须遵守授权、租户隔离、保密性和数据最小化规则。",{},{"id":735,"data":736,"type":218,"tunes":738},"p-sec-2",{"text":737},"检索器在技术上可能找到当前用户无权访问的段落。正确的设计是阻止该段落进入模型上下文，而不是依赖模型忽略它。",{},{"id":740,"data":741,"type":218,"tunes":743},"p-sec-3",{"text":742},"工具输出也可能包含不可信的指令或对抗性内容。上下文工程应保留应用程序指令与外部数据之间的区别，使检索到的文本无法悄然获得指令权威。",{},{"id":745,"data":746,"type":42,"tunes":748},"h-architecture",{"text":747,"level":247},"一个实用的上下文工程架构",{},{"id":750,"data":751,"type":226,"tunes":754},"arch-note",{"body":752,"title":753,"variant":240},"以下分层是针对生产系统的实用综合方案，并非正式的行业标准。其目的是将信息所有权与临时的面向模型的上下文分开。","提议的架构模型",{},{"id":756,"data":757,"type":391,"tunes":786},"arch-table",{"content":758,"stretched":43,"withHeadings":14},[759,762,765,768,771,774,777,780,783],[760,761],"层","职责",[763,764],"权威系统","拥有当前业务\u002F系统状态和官方记录。",[766,767],"知识来源","拥有文档、政策、规范、研究或外部证据。",[769,770],"记忆存储","跨轮次或会话保留选定的信息。",[772,773],"检索层","从外部来源定位与任务相关的候选内容。",[775,776],"工具\u002F运行时层","读取状态、执行操作并返回观察结果。",[778,779],"上下文组装器","选择、过滤、去重、排序并格式化模型可见的信息。",[781,782],"模型","在组装后的上下文上进行推理和生成。",[784,785],"验证\u002F评估","检查所选上下文和生成的输出是否满足特定任务的要求。",{},{"id":788,"data":789,"type":218,"tunes":791},"p-arch-1",{"text":790},"即使没有任何模块使用这个确切的名称，上下文组装器在概念上也很重要。在小型应用中，它可能只是普通的应用代码。在大型智能体平台中，它可能结合了会话管理、检索、记忆、工具中间件、压缩和策略执行。",{},{"id":793,"data":794,"type":42,"tunes":796},"h-policy",{"text":795,"level":247},"一个实用的上下文构建策略",{},{"id":798,"data":799,"type":391,"tunes":840},"policy-table",{"content":800,"stretched":43,"withHeadings":14},[801,804,807,810,813,816,819,822,825,828,831,834,837],[802,803],"规则","为什么重要",[805,806],"从当前任务出发","不要仅仅因为信息之前存在就携带它。",[808,809],"重新读取易变状态","记忆和旧上下文可能已经过时。",[811,812],"只检索刚好足够的证据","大量候选集会稀释决定性信息。",[814,815],"保留来源元数据","版本、日期和权威性决定证据是否仍然适用。",[817,818],"移除重复内容","冗余会消耗 token 而不增加信息。",[820,821],"对大型工具输出优先使用结构化摘要","在保真度允许的情况下，暴露决定性字段而非原始噪声。",[823,824],"规则与例外保持在一起","将规则与其例外分开会造成虚假的确定性。",[826,827],"明确优先级","不要让模型去推断哪个冲突来源胜出。",[829,830],"将持久状态保留在上下文之外","上下文是临时工作记忆，不是数据库。",[832,833],"用保留测试进行压缩","验证标识符、约束、来源和未解决状态是否得以保留。",[835,836],"测量顺序敏感性","正确性不应意外地依赖于任意的文档顺序。",[838,839],"将上下文评估与模型质量分开","更强的模型无法可靠地弥补缺失或未经授权的证据。",{},{"id":842,"data":843,"type":42,"tunes":845},"h-eval",{"text":844,"level":247},"如何评估上下文工程",{},{"id":847,"data":848,"type":391,"tunes":890},"eval-table",{"content":849,"stretched":43,"withHeadings":14},[850,854,858,862,866,870,874,878,882,886],[851,852,853],"属性","问题","示例测试",[855,856,857],"充分性","上下文是否包含解决任务所需的一切？","移除一个证据项，观察答案是否变得缺乏支持。",[859,860,861],"相关性","有多少上下文对任务是不必要的？","在添加或移除无关段落时测量质量。",[863,864,865],"权威性","决定性主张是否基于正确的来源类别？","注入一个更流畅但非权威的冲突来源。",[867,868,869],"新鲜度","当前状态是否覆盖过时的副本？","在上一轮之后更改权威状态并重新运行。",[871,872,873],"位置稳健性","答案质量是否强烈依赖于证据位置？","在重复试验中随机化候选顺序。",[875,876,877],"冲突处理","模型是否遵循明确的优先级规则？","将旧状态和新状态一起呈现。",[879,880,881],"压缩保留","摘要是否保留了约束和有效性边界？","比较压缩前后的任务表现。",[883,884,885],"Token 效率","额外上下文带来的质量提升是否足以证明延迟\u002F成本合理？","运行受控的上下文大小消融实验。",[887,888,889],"安全性","未经授权或对抗性内容能否进入模型上下文？","测试租户、权限和提示注入边界。",{},{"id":892,"data":893,"type":42,"tunes":895},"h-rag-diagnostic",{"text":894,"level":247},"上下文组装是一个独立的 RAG 故障层",{},{"id":897,"data":898,"type":218,"tunes":900},"p-ragdiag-1",{"text":899},"一个 RAG 流水线可能在检索上成功，却在下游失败。相关来源可能出现在第 2 位，但上下文组装器可能将其丢弃、截断、与过时的矛盾材料合并，或超出 token 预算。",{},{"id":902,"data":903,"type":218,"tunes":905},"p-ragdiag-2",{"text":904},"这就是为什么应将检索轨迹与实际发送给模型的上下文进行比较。没有这种比较，上下文故障很容易被误诊为嵌入或模型故障。",{},{"id":907,"data":908,"type":492,"tunes":913},"ref-ragfail",{"url":909,"title":910,"excerpt":911,"ctaLabel":912},"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method","RAG 失败了——但究竟是哪一层失败了？一种诊断方法","一种逐层方法，用于区分来源覆盖、检索、排序、上下文组装、生成、证据归因和新鲜度故障。","阅读 RAG 诊断方法",{},{"id":915,"data":916,"type":42,"tunes":918},"h-implementation",{"text":917,"level":247},"原始实现证据",{},{"id":920,"data":921,"type":42,"tunes":923},"h-sot-engine",{"text":922,"level":246},"真相来源研究引擎：有界研究而非无限上下文",{},{"id":925,"data":926,"type":218,"tunes":928},"p-sot-1",{"text":927},"真相来源研究引擎将发现、获取、提取、验证、矛盾分析和综合分为有界的研究阶段，而不是将一个庞大的研究任务和所有累积材料发送到单次模型调用中。",{},{"id":930,"data":931,"type":218,"tunes":933},"p-sot-2",{"text":932},"其证据模型将来源、工件、主张、关系、矛盾和来源信息存储在模型上下文之外。模型可以接收当前研究步骤所需的子集，而持久证据保留在外部存储中。",{},{"id":935,"data":936,"type":218,"tunes":938},"p-sot-3",{"text":937},"这是一个具体的上下文工程模式：持久研究状态存在于模型窗口之外；活跃的模型上下文针对当前阶段重新构建。",{},{"id":940,"data":941,"type":42,"tunes":943},"h-ai-client",{"text":942,"level":246},"Aaasaasa AI 客户端：运行时、权限和上下文是相互独立的关注点",{},{"id":945,"data":946,"type":218,"tunes":948},"p-client-1",{"text":947},"Aaasaasa AI Client 将提供商\u002F模型选择、运行时位置、工作区权限、本地资源和工具访问分离开来。这可以防止模型上下文成为授权或应用状态的所有者。",{},{"id":950,"data":951,"type":218,"tunes":953},"p-client-2",{"text":952},"直接聊天和代理运行时可以具有不同的工具能力。工作区权限配置文件由运行时强制执行，而不仅仅是在自然语言上下文中描述。这一区别很重要：上下文可以告诉模型它应该做什么，而运行时仍然必须强制执行它实际被允许做什么。",{},{"id":955,"data":956,"type":218,"tunes":958},"p-client-3",{"text":957},"这里的实现证据是架构分离，而不是声称本文中描述的每一种高级上下文管理技术都已经实现。",{},{"id":960,"data":961,"type":391,"tunes":981},"impl-table",{"content":962,"stretched":43,"withHeadings":14},[963,966,969,972,975,978],[964,965],"实现模式","上下文工程经验",[967,968],"外部证据存储","持久知识不需要保留在模型窗口中。",[970,971],"有界研究阶段","不同步骤可以接收不同的上下文，而不是累积一个巨大的历史记录。",[973,974],"上下文之外的声明 + 来源","证据身份在临时推理状态之外仍然存在。",[976,977],"运行时强制执行的权限","安全授权不依赖于模型记住指令。",[979,980],"分离本地\u002F提供商\u002F模型\u002F运行时概念","上下文只是更广泛 AI 应用架构中的一层。",{},{"id":983,"data":984,"type":226,"tunes":987},"impl-boundary",{"body":985,"title":986,"variant":240},"这些实现支持持久状态、检索、运行时控制和面向模型的上下文之间的架构分离。它们并不是作为基准证明来呈现，以表明某一种上下文策略普遍最优。","证据边界",{},{"id":989,"data":990,"type":42,"tunes":992},"h-failures",{"text":991,"level":247},"常见的上下文工程失败模式",{},{"id":994,"data":995,"type":391,"tunes":1030},"failure-table",{"content":996,"stretched":43,"withHeadings":14},[997,1000,1003,1006,1009,1012,1015,1018,1021,1024,1027],[998,999],"失败模式","出了什么问题",[1001,1002],"永远重放整个对话","旧假设、重复和 token 增长会压过当前意图。",[1004,1005],"把每个检索结果都放进提示词","噪声、重复和冲突版本会稀释决定性证据。",[1007,1008],"把记忆当作当前状态","过时信息会悄悄取代权威的实时状态。",[1010,1011],"返回原始工具输出","大型日志或响应会消耗注意力，却不增加决策价值。",[1013,1014],"用模糊名称隐藏工具描述","模型无法可靠地决定应使用哪种能力。",[1016,1017],"没有保留测试就进行压缩","关键约束、标识符或例外会消失。",[1019,1020],"混合指令和不可信数据","外部内容可能被解释为更高权威的指令。",[1022,1023],"对每个任务使用同一个静态上下文模板","不同任务会收到无关信息，并错过任务特定证据。",[1025,1026],"忽略来源版本\u002F日期","过时但相关的证据可能压过当前权威状态。",[1028,1029],"把更大的上下文窗口当作质量保证","容量增加了，但注意力和冲突问题仍然存在。",{},{"id":1032,"data":1033,"type":42,"tunes":1035},"h-misconceptions",{"text":1034,"level":247},"常见误解",{},{"id":1037,"data":1038,"type":391,"tunes":1073},"misconceptions-table",{"content":1039,"stretched":43,"withHeadings":14},[1040,1043,1046,1049,1052,1055,1058,1061,1064,1067,1070],[1041,1042],"误解","纠正",[1044,1045],"“上下文工程只是换了个名字的提示词工程。”","提示词只是一个组成部分；上下文工程还涵盖检索、记忆、状态、工具结果、历史和压缩。",[1047,1048],"“上下文就是聊天历史。”","历史只是可能的上下文来源之一。",[1050,1051],"“上下文越多总是越好。”","额外信息可能降低信号、引入冲突并增加成本。",[1053,1054],"“如果检索找到了它，模型就看到了它。”","检索到的候选内容可能在推理前被过滤、截断或省略。",[1056,1057],"“长上下文消除了对 RAG 的需求。”","大窗口增加了容量，但并不能解决新鲜度、权威性、权限或动态检索问题。",[1059,1060],"“记忆应该总是被加载。”","记忆应根据当前任务进行选择。",[1062,1063],"“摘要会保留所有重要内容。”","除非明确评估保留情况，否则压缩是有损的。",[1065,1066],"“指令可以强制执行权限。”","授权必须由运行时\u002F应用控制来强制执行，而不仅仅由上下文执行。",[1068,1069],"“一种上下文配方适用于所有模型。”","上下文敏感性因模型、任务、语料库和运行时而异。",[1071,1072],"“上下文工程只适用于代理。”","代理会放大这种需求，但普通 RAG 和对话应用也需要上下文构建。",{},{"id":1075,"data":1076,"type":42,"tunes":1078},"h-sequence",{"text":1077,"level":247},"一个实用的上下文工程序列",{},{"id":1080,"data":1081,"type":317,"tunes":1114},"design-sequence",{"steps":1082,"title":1113,"orientation":316},[1083,1086,1089,1092,1095,1098,1101,1104,1107,1110],{"label":1084,"description":1085},"1. 定义下一个模型决策","明确模型在这一步必须回答、分类、规划或选择什么。",{"label":1087,"description":1088},"2. 识别所需事实和约束","列出能够实质性改变结果的最小状态、规则、证据和指令。",{"label":1090,"description":1091},"3. 解决权威性和权限问题","确定哪些来源是当前的、权威的，并且当前主体可以访问。",{"label":1093,"description":1094},"4. 按需检索或读取","获取必要证据和易变状态，而不是依赖过时上下文。",{"label":1096,"description":1097},"5. 降低噪声","去重、摘要或选择段落，同时不丢弃决定性例外或来源。",{"label":1099,"description":1100},"6. 结构化并排序","使指令、当前状态、证据和工具观察结果可以区分。",{"label":1102,"description":1103},"7. 适配 token 预算","优先使用高信号上下文，并将持久信息移到窗口之外。",{"label":1105,"description":1106},"8. 运行模型","在组装好的上下文上执行推理。",{"label":1108,"description":1109},"9. 观察失败","记录问题来自缺失、过时、噪声、冲突还是排序不佳的上下文。",{"label":1111,"description":1112},"10. 在模型\u002F运行时变更后重新评估","上下文策略只对它所测试过的模型、工具和工作负载有效。","从当前决策反向构建上下文",{},{"id":1116,"data":1117,"type":42,"tunes":1119},"h-checklist",{"text":1118,"level":247},"上下文工程检查清单",{},{"id":1121,"data":1122,"type":391,"tunes":1162},"checklist-table",{"content":1123,"stretched":43,"withHeadings":14},[1124,1126,1129,1132,1135,1138,1141,1144,1147,1150,1153,1156,1159],[852,1125],"预期答案",[1127,1128],"模型下一步将做出什么确切决策？","一个有边界的任务，而不是模糊的长期目标。",[1130,1131],"哪些信息能够实质性改变该决策？","明确的最小证据\u002F状态集合。",[1133,1134],"现在哪些数据是权威的？","当前来源\u002F版本和新鲜度规则。",[1136,1137],"哪些数据是可选的背景信息？","与决定性证据分开。",[1139,1140],"什么内容不得进入上下文？","未授权、不必要或过于敏感的数据。",[1142,1143],"哪些记忆项是相关的？","按任务选择，而不是自动重放。",[1145,1146],"哪些工具输出应被精简？","大型响应被转换为与决策相关的形式。",[1148,1149],"哪些约束必须在压缩后保留？","标识符、例外、义务、未解决状态和来源。",[1151,1152],"优先级如何表示？","当前\u002F权威信息可以可靠地覆盖过时或较弱来源。",[1154,1155],"你如何知道上下文失败了？","存在上下文特定的评估和追踪。",[1157,1158],"答案可以复现吗？","在适当情况下，模型输入或可重建的上下文追踪可用。",[1160,1161],"更强或更大的模型会改变策略吗？","上下文策略具备版本意识，并根据经验重新评估。",{},{"id":1164,"data":1165,"type":42,"tunes":1167},"h-edge",{"text":1166,"level":247},"边缘情况和限制",{},{"id":1169,"data":1170,"type":218,"tunes":1172},"p-edge-1",{"text":1171},"有些任务足够简单，以至于上下文工程简化为一个简短的系统提示词和一条用户消息。添加检索、记忆和压缩只会引入不必要的架构。",{},{"id":1174,"data":1175,"type":218,"tunes":1177},"p-edge-2",{"text":1176},"有些任务需要高召回率，并且可能有意在后续综合之前包含更多上下文。研究、发现和法律审查可能更倾向于避免遗漏，而不是最小化 token 数量。",{},{"id":1179,"data":1180,"type":218,"tunes":1182},"p-edge-3",{"text":1181},"有些信息在使用前绝不应被摘要。精确合同、代码、密码材料、数值记录和监管文本可能需要逐字或结构化检索，因为压缩可能会改变含义。",{},{"id":1184,"data":1185,"type":218,"tunes":1187},"p-edge-4",{"text":1186},"长上下文行为在不同模型之间差异很大。在一个模型、上下文长度或工具框架上验证过的策略，不应自动转移到另一个上。",{},{"id":1189,"data":1190,"type":218,"tunes":1192},"p-edge-5",{"text":1191},"模型仍然可能忽略或误解优秀的上下文。上下文工程改善的是信息环境；它并不保证推理的正确性。",{},{"id":1194,"data":1195,"type":42,"tunes":1197},"h-change",{"text":1196,"level":247},"什么会改变这个答案？",{},{"id":1199,"data":1200,"type":218,"tunes":1202},"p-change-1",{"text":1201},"未来的模型可能会对长上下文、位置效应和冲突信息变得更加稳健。这可能会减少所需的人工整理工作量。",{},{"id":1204,"data":1205,"type":218,"tunes":1207},"p-change-2",{"text":1206},"架构上的区分仍然有用，因为权限、时效性、记忆持久性、来源权威性和外部应用状态无论上下文窗口大小如何都存在于模型之外。",{},{"id":1209,"data":1210,"type":218,"tunes":1212},"p-change-3",{"text":1211},"预加载上下文与即时上下文之间的推荐平衡也会随延迟要求、工具可靠性、语料库规模、模型成本以及底层信息的动态程度而变化。",{},{"id":1214,"data":1215,"type":42,"tunes":1217},"h-related",{"text":1216,"level":247},"相关权威知识",{},{"id":1219,"data":1220,"type":218,"tunes":1222},"p-related-1",{"text":1221},"上下文工程介于检索与生成之间。RAG 解释了外部知识如何被检索；R01 区分了嵌入、向量搜索和重排序；上下文工程解释了最终到达模型的是什么。",{},{"id":1224,"data":1225,"type":492,"tunes":1230},"ref-rag",{"url":1226,"title":1227,"excerpt":1228,"ctaLabel":1229},"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works","什么是 RAG？对其工作原理的最简单解释","理解外部知识如何在生成之前被提供给模型的检索基础。","阅读 RAG 基础",{},{"id":1232,"data":1233,"type":218,"tunes":1235},"p-related-2",{"text":1234},"真相来源架构回答的是另一个问题：不是上下文中存在哪些信息，而是哪个来源被授权来确立一项主张。",{},{"id":1237,"data":1238,"type":218,"tunes":1240},"p-related-3",{"text":1239},"现有文章《为什么更多上下文会让 AI 答案更糟》是这一权威定义的诊断性配套文章。它侧重于上下文污染、位置效应、top-k 增长、压缩损失和答案退化，而不是重新定义上下文工程本身。",{},{"id":1242,"data":1243,"type":42,"tunes":1245},"h-faq",{"text":1244,"level":247},"常见问题",{},{"id":1247,"data":1248,"type":1247,"tunes":1283},"faq",{"items":1249,"title":1282},[1250,1254,1258,1262,1266,1270,1274,1278],{"id":1251,"answer":1252,"question":1253},"faq1","上下文工程是对语言模型在推理时接收到的信息进行的设计和运行时管理，包括指令、历史记录、检索到的证据、记忆、状态、工具和工具结果。","什么是上下文工程？",{"id":1255,"answer":1256,"question":1257},"faq2","提示工程侧重于指令和示例的编写方式。上下文工程包括提示，但也决定将哪些外部信息、状态、历史记录、记忆和工具观察结果置于提示周围。","上下文工程与提示工程有何不同？",{"id":1259,"answer":1260,"question":1261},"faq3","不是。RAG 检索外部信息。上下文工程决定检索到的信息如何被过滤、如何与其他状态结合，以及如何实际传递给模型。","RAG 与上下文工程是一回事吗？",{"id":1263,"answer":1264,"question":1265},"faq4","不是。记忆将信息持久保存在当前模型调用之外。上下文是加载到当前推理中的那部分信息。","记忆与上下文是一回事吗？",{"id":1267,"answer":1268,"question":1269},"faq5","额外的上下文可能引入噪声、过时状态、相互冲突的证据、重复内容和位置竞争。大的上下文容量并不保证每个 token 都能被同样可靠地利用。","为什么更多上下文会让答案更糟？",{"id":1271,"answer":1272,"question":1273},"faq6","压缩是将累积的历史记录总结或转换为更小的表示形式，以便长时间运行的系统可以继续运行而无需重放此前的每一个 token。","什么是上下文压缩？",{"id":1275,"answer":1276,"question":1277},"faq7","它可以被表示在上下文中用于推理，但具有后果的操作通常应重新读取权威来源，因为上下文快照可能会过时。","当前应用状态应该存储在上下文中吗？",{"id":1279,"answer":1280,"question":1281},"faq8","不是。智能体使上下文管理更加动态，但 RAG 系统、助手、副驾驶和多轮应用也需要有意识地构建上下文。","上下文工程只对 AI 智能体有需要吗？","上下文工程常见问题",{},{"id":1285,"data":1286,"type":42,"tunes":1288},"h-glossary",{"text":1287,"level":247},"术语表",{},{"id":1290,"data":1291,"type":1290,"tunes":1338},"glossary",{"title":1292,"entries":1293},"关键上下文工程术语",[1294,1297,1301,1304,1308,1312,1316,1320,1324,1327,1331,1334],{"term":430,"anchor":1295,"definition":1296},"context-engineering","为特定推理步骤向语言模型提供的信息进行的设计和运行时管理。",{"term":1298,"anchor":1299,"definition":1300},"上下文窗口","context-window","模型对输入的有限 token 容量，并且根据模型接口的不同，还包括相关的生成 token 或活动序列。",{"term":427,"anchor":1302,"definition":1303},"prompt-engineering","旨在引出有用模型行为的指令、示例和提示结构的设计。",{"term":1305,"anchor":1306,"definition":1307},"上下文组装","context-assembly","在推理之前选择、过滤、排序和格式化模型可见信息的过程。",{"term":1309,"anchor":1310,"definition":1311},"即时检索","just-in-time-retrieval","在当前任务需要时动态加载信息，而不是预加载所有可能相关的数据。",{"term":1313,"anchor":1314,"definition":1315},"压缩","compaction","将累积的上下文缩减为更小的表示形式，同时尝试保留未来步骤所需的信息。",{"term":1317,"anchor":1318,"definition":1319},"上下文污染","context-pollution","由无关、过时、矛盾或冗余信息占据模型工作上下文而导致的性能下降。",{"term":1321,"anchor":1322,"definition":1323},"应用状态","application-state","外部系统、工作流或领域独立于模型上下文而存在的当前权威状况。",{"term":376,"anchor":1325,"definition":1326},"memory","存储在即时模型调用之外、可能用于后续轮次或会话的信息。",{"term":1328,"anchor":1329,"definition":1330},"检索上下文","retrieved-context","由检索系统选择并全部或部分提供给模型的外部信息。",{"term":871,"anchor":1332,"definition":1333},"position-robustness","当相关上下文的位置或顺序发生变化时，模型正确性保持稳定的程度。",{"term":1335,"anchor":1336,"definition":1337},"有效性边界","validity-boundary","结论仍然得到支持的范围、时间、假设、版本和证据条件。",{},{"id":1340,"data":1341,"type":42,"tunes":1343},"h-conclusion",{"text":1342,"level":247},"结论",{},{"id":1345,"data":1346,"type":218,"tunes":1348},"p-conclusion-1",{"text":1347},"上下文工程是决定模型在回答之前能看到什么的那一层。这使它比提示更广泛，并处于检索的下游，同时又有别于持久记忆和权威应用状态。",{},{"id":1350,"data":1351,"type":218,"tunes":1353},"p-conclusion-2",{"text":1352},"强大的上下文架构不会将上下文窗口视为数据库。它将持久状态和知识保留在模型之外，加载当前决策所需的内容，保留权威性和来源，去除不必要的噪声，并在需要时刷新易变信息。",{},{"id":1355,"data":1356,"type":218,"tunes":1358},"p-conclusion-3",{"text":1357},"因此，实际目标不是最大上下文。而是为下一次模型决策提供最小充分、高信号、正确授权且保持有效性的上下文。",{},{"id":1360,"data":1361,"type":42,"tunes":1363},"h-sources",{"text":1362,"level":247},"主要来源与当前指南",{},{"id":1365,"data":1366,"type":218,"tunes":1368},"p-sources-note",{"text":1367},"以下来源支持当前的上下文工程术语、长上下文行为以及可操作的上下文管理模式。项目部分明确属于实现证据，而非普遍性主张。",{},{"id":1370,"data":1371,"type":1377,"tunes":1378},"src-anthropic",{"link":1372,"meta":1373},"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Feffective-context-engineering-for-ai-agents",{"image":1374,"title":1375,"description":1376},{"url":406},"Anthropic — 面向 AI 智能体的有效上下文工程","官方工程指南，定义了上下文工程、即时检索、压缩、结构化记忆以及面向智能体的上下文策展。","linkTool",{},{"id":1380,"data":1381,"type":1377,"tunes":1387},"src-openai-session",{"link":1382,"meta":1383},"https:\u002F\u002Fdevelopers.openai.com\u002Fcookbook\u002Fexamples\u002Fagents_sdk\u002Fsession_memory",{"image":1384,"title":1385,"description":1386},{"url":406},"OpenAI — 上下文工程：使用会话进行短期记忆管理","官方 cookbook 指南，涉及长时间运行的智能体会话的上下文管理、裁剪和压缩。",{},{"id":1389,"data":1390,"type":1377,"tunes":1396},"src-openai-agents",{"link":1391,"meta":1392},"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fagents",{"image":1393,"title":1394,"description":1395},{"url":406},"OpenAI — 智能体指南","OpenAI 当前面向开发者的指南，涉及智能体运行时、跨步骤上下文以及编排归属。",{},{"id":1398,"data":1399,"type":1377,"tunes":1405},"src-lost-middle",{"link":1400,"meta":1401},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2307.03172",{"image":1402,"title":1403,"description":1404},{"url":406},"迷失在中间：语言模型如何使用长上下文","研究表明，长上下文模型的性能可能在很大程度上取决于相关信息在输入中的位置。",{},"2.31","上下文工程负责设计 AI 模型在推理前接收的信息，包括提示词、检索内容、记忆、应用状态、工具结果和对话历史。","\u002Fuploads\u002F2026\u002F10\u002Fwhat-is-context-engineering-what-the-model-receives-before-it-answers-1791480653258-018kcv.webp","what-is-context-engineering-what-the-model-receives-before-it-answers-1791480653258-018kcv","PUBLISHED","2026-10-08T13:29:00.000Z","2026-10-08T17:29:05.600Z","2026-10-08T17:43:15.694Z",{"en":1415,"de":1416,"sr":1417,"es":1418,"fr":1419,"it":1420,"ru":1421,"zh":1422},"\u002Fblog\u002Fwhat-is-context-engineering-what-the-model-receives-before-it-answers","\u002Fde\u002Fblog\u002Fwhat-is-context-engineering-what-the-model-receives-before-it-answers","\u002Fsr\u002Fblog\u002Fwhat-is-context-engineering-what-the-model-receives-before-it-answers","\u002Fes\u002Fblog\u002Fwhat-is-context-engineering-what-the-model-receives-before-it-answers","\u002Ffr\u002Fblog\u002Fwhat-is-context-engineering-what-the-model-receives-before-it-answers","\u002Fit\u002Fblog\u002Fwhat-is-context-engineering-what-the-model-receives-before-it-answers","\u002Fru\u002Fblog\u002Fwhat-is-context-engineering-what-the-model-receives-before-it-answers","\u002Fzh\u002Fblog\u002Fwhat-is-context-engineering-what-the-model-receives-before-it-answers",[1424,1428,1432],{"id":1425,"name":1426,"slug":1427},55,"LLM能力参考模型","llm-capability",{"id":1429,"name":1430,"slug":1431},64,"信息架构","information-architecture",{"id":1433,"name":1434,"slug":1435},88,"版本管理（提示\u002F模型）","versioning",{"id":1437,"login":1438,"email":1439,"displayName":1440},"20","rooth8233","aleksandar@stajic.de","Aleksandar Stajić",[1442,2458],{"lang":1443,"title":1444,"content":1445,"contentJson":1446,"excerpt":2457},"en","What Is Context Engineering? What the Model Receives Before It Answers","{\"time\":1791480654232,\"blocks\":[{\"id\":\"intro\",\"type\":\"paragraph\",\"data\":{\"text\":\"Context engineering is the design of what information a language model receives at inference time, in what form, in what order and for how long. It is broader than prompt engineering because the model context can include system instructions, user messages, retrieved documents, tool results, memory, current application state, examples, structured data and intermediate artifacts. The goal is not to maximize the number of tokens, but to construct the smallest useful context that preserves the information, constraints and evidence needed for the current task.\"},\"tunes\":{}},{\"id\":\"direct\",\"type\":\"callout\",\"data\":{\"variant\":\"info\",\"title\":\"Direct answer\",\"body\":\"Prompt engineering asks \u003Cstrong>how should we instruct the model?\u003C\u002Fstrong> Context engineering asks \u003Cstrong>what should the model know right now, and how should that information be assembled?\u003C\u002Fstrong>\u003Cbr>\u003Cbr>Retrieval, memory, state management, tool design, history trimming, compaction and ordering are therefore context-engineering mechanisms when they determine the tokens available to the model before it produces the next output.\"},\"tunes\":{}},{\"id\":\"boundary\",\"type\":\"callout\",\"data\":{\"variant\":\"warning\",\"title\":\"Context is not the same as knowledge or memory\",\"body\":\"A system can know something without placing it in the current context. It can remember something outside the model window. It can retrieve a document but later exclude it from the final prompt. The model can only directly use the context that reaches the current inference.\"},\"tunes\":{}},{\"id\":\"current\",\"type\":\"callout\",\"data\":{\"variant\":\"note\",\"title\":\"Current-source note — 8 October 2026\",\"body\":\"Context engineering is now established practical terminology in major AI engineering guidance, but it is not a single formal standard with one mandatory architecture. Anthropic describes it as curating and maintaining the optimal set of tokens for inference; OpenAI's current agent guidance treats session context, trimming and compression as explicit engineering concerns for long-running systems.\"},\"tunes\":{}},{\"id\":\"toc\",\"type\":\"tableOfContents\",\"data\":{\"title\":\"Contents\",\"minLevel\":2,\"maxLevel\":3},\"tunes\":{}},{\"id\":\"h-meaning\",\"type\":\"header\",\"data\":{\"text\":\"What context engineering really means\",\"level\":2},\"tunes\":{}},{\"id\":\"p-meaning-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Every model call is made under a temporary working environment: the current instructions, messages, retrieved evidence, tool outputs and state that fit into the active context window. Context engineering is the discipline of constructing that environment deliberately.\"},\"tunes\":{}},{\"id\":\"p-meaning-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The key word is deliberately. A naive system simply concatenates everything it has: full history, all retrieved documents, every tool response and large system prompts. A context-engineered system decides which information is required for the current decision and which information should remain outside the window until needed.\"},\"tunes\":{}},{\"id\":\"p-meaning-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"This makes context engineering partly an information-architecture problem, partly a runtime problem and partly an evaluation problem. The design must decide what can enter context, where it comes from, which version is current, how conflicts are resolved, how much detail is retained and how the result is tested.\"},\"tunes\":{}},{\"id\":\"h-simple\",\"type\":\"header\",\"data\":{\"text\":\"The simplest example\",\"level\":2},\"tunes\":{}},{\"id\":\"p-simple-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Imagine an internal support assistant. A user asks: “Can this customer cancel without a fee?”\"},\"tunes\":{}},{\"id\":\"p-simple-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The model might need five things: the current cancellation policy, the customer's current contract type, the effective contract date, the relevant exception rules and the user's authorization scope.\"},\"tunes\":{}},{\"id\":\"p-simple-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"It does not necessarily need the entire customer database, the full policy archive, every previous conversation or every support ticket. Context engineering is the process that selects and assembles the five useful pieces while excluding unrelated information.\"},\"tunes\":{}},{\"id\":\"simple-flow\",\"type\":\"processFlow\",\"data\":{\"title\":\"From application state to model context\",\"orientation\":\"auto\",\"steps\":[{\"label\":\"1. Understand the task\",\"description\":\"Classify what the current question requires and which information types can affect the answer.\"},{\"label\":\"2. Resolve authoritative state\",\"description\":\"Read current application or business state that should not be guessed from memory.\"},{\"label\":\"3. Retrieve supporting knowledge\",\"description\":\"Find the policy, documents or external evidence relevant to the specific task.\"},{\"label\":\"4. Apply eligibility and permissions\",\"description\":\"Exclude data the current user or runtime is not allowed to expose to the model.\"},{\"label\":\"5. Reduce and structure\",\"description\":\"Remove duplication, select useful excerpts and preserve critical metadata, conditions and exceptions.\"},{\"label\":\"6. Order the context\",\"description\":\"Place instructions, current state and decisive evidence where the model can use them consistently.\"},{\"label\":\"7. Run inference\",\"description\":\"The model receives the assembled context and produces the next answer or action proposal.\"}]},\"tunes\":{}},{\"id\":\"h-stops\",\"type\":\"header\",\"data\":{\"text\":\"Where the simple example stops\",\"level\":2},\"tunes\":{}},{\"id\":\"p-stops-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Real systems are more difficult because the information needed for one step may not be known before execution begins. An agent can discover new facts through tools, create intermediate files, receive changing external state or span a task longer than one context window.\"},\"tunes\":{}},{\"id\":\"p-stops-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Context engineering therefore becomes dynamic. The context for step 12 should not simply be step 1 context plus eleven layers of accumulated output. It should reflect the current task state, the decisions that still matter and the evidence required for the next action.\"},\"tunes\":{}},{\"id\":\"h-anatomy\",\"type\":\"header\",\"data\":{\"text\":\"What can enter a model context?\",\"level\":2},\"tunes\":{}},{\"id\":\"anatomy-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Context component\",\"Purpose\",\"Typical risk\"],[\"System \u002F developer instructions\",\"Define role, constraints, policies and behavior\",\"Too vague, contradictory or overloaded with brittle logic\"],[\"Current user request\",\"Defines immediate task and intent\",\"Ambiguity or conflict with prior history\"],[\"Conversation history\",\"Preserves continuity across turns\",\"Stale assumptions, repetition and token growth\"],[\"Retrieved documents\",\"Provide external knowledge\u002Fevidence\",\"Irrelevance, stale versions, weak authority or duplication\"],[\"Current application state\",\"Supplies volatile business\u002Fsystem facts\",\"Using cached or remembered state instead of current authority\"],[\"Tool definitions\",\"Tell the model what capabilities exist and how to call them\",\"Too many overlapping tools or verbose schemas\"],[\"Tool results\",\"Bring observations from the environment into the loop\",\"Large noisy outputs, untrusted content or obsolete observations\"],[\"Memory\",\"Reintroduces selected information from previous interactions\",\"Staleness, incorrect generalization or over-personalization\"],[\"Examples\",\"Demonstrate desired behavior\",\"Too many edge cases can crowd out the current task\"],[\"Intermediate artifacts\",\"Carry plans, summaries, code, calculations or notes\",\"Old intermediate state may be mistaken for final truth\"],[\"Policies \u002F guardrails\",\"Define prohibited or constrained behavior\",\"Conflict with business logic or hidden enforcement gaps\"]]},\"tunes\":{}},{\"id\":\"h-prompt\",\"type\":\"header\",\"data\":{\"text\":\"Context engineering vs prompt engineering\",\"level\":2},\"tunes\":{}},{\"id\":\"prompt-comparison\",\"type\":\"comparison\",\"data\":{\"title\":\"Prompt engineering and context engineering solve different layers\",\"layout\":\"table\",\"columns\":[{\"id\":\"prompt\",\"label\":\"Prompt engineering\"},{\"id\":\"context\",\"label\":\"Context engineering\"}],\"rows\":[{\"id\":\"focus\",\"label\":\"Primary focus\",\"values\":[\"\",\"\"]},{\"id\":\"scope\",\"label\":\"Typical scope\",\"values\":[\"\",\"\"]},{\"id\":\"timing\",\"label\":\"When it changes\",\"values\":[\"\",\"\"]},{\"id\":\"failure\",\"label\":\"Typical failure\",\"values\":[\"\",\"\"]},{\"id\":\"relationship\",\"label\":\"Relationship\",\"values\":[\"\",\"\"]}]},\"tunes\":{}},{\"id\":\"p-prompt-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Anthropic explicitly describes context engineering as the natural progression of prompt engineering for systems in which the model must work with tools, external data, message history and long-running agent state. The practical distinction is useful because a perfectly written prompt cannot compensate for missing authoritative data or a context polluted by contradictory state.\"},\"tunes\":{}},{\"id\":\"h-retrieval\",\"type\":\"header\",\"data\":{\"text\":\"Context engineering vs retrieval\",\"level\":2},\"tunes\":{}},{\"id\":\"p-ret-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Retrieval selects candidate information from an external corpus or source. Context engineering decides what happens after and around that retrieval.\"},\"tunes\":{}},{\"id\":\"p-ret-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The retriever may return 30 passages. A reranker may reduce them to 10. The context layer may select four passages, remove duplicates, attach source\u002Fversion metadata, combine them with current application state and place them after the system instructions.\"},\"tunes\":{}},{\"id\":\"p-ret-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"This is why a RAG system can retrieve the correct passage and still answer badly: the failure may occur during context assembly rather than retrieval.\"},\"tunes\":{}},{\"id\":\"retrieval-boundary\",\"type\":\"callout\",\"data\":{\"variant\":\"success\",\"title\":\"Retrieval finds candidates; context engineering constructs the model input\",\"body\":\"The correct retrieval result is only useful if it survives filtering, ordering, compression and token-budget decisions and actually reaches the model in a usable form.\"},\"tunes\":{}},{\"id\":\"h-memory\",\"type\":\"header\",\"data\":{\"text\":\"Context engineering vs memory\",\"level\":2},\"tunes\":{}},{\"id\":\"p-memory-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Memory is information preserved outside the immediate model invocation so it can be used again later. Context is the information actually loaded into the current invocation.\"},\"tunes\":{}},{\"id\":\"p-memory-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"A memory system may contain thousands of facts, notes or prior decisions. Context engineering selects which of those should be reintroduced for the current task. Loading all memory on every turn defeats the purpose of having an external memory layer.\"},\"tunes\":{}},{\"id\":\"p-memory-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The distinction becomes crucial for volatile state. A remembered project status or user preference can be useful, but current authoritative state may need to be re-read before a consequential decision.\"},\"tunes\":{}},{\"id\":\"ref-memory\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context\",\"title\":\"AI Agent Memory Is Not RAG: How to Separate Memory, Retrieval, State and Context\",\"excerpt\":\"A practical architecture separating what persists, what is authoritative now, what is retrieved and what the model actually receives.\",\"ctaLabel\":\"Read the memory architecture article\"},\"tunes\":{}},{\"id\":\"h-state\",\"type\":\"header\",\"data\":{\"text\":\"Context engineering vs application state\",\"level\":2},\"tunes\":{}},{\"id\":\"p-state-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Application state is the current condition of the outside system: account balance, ticket status, file version, workflow stage, deployment state or task progress.\"},\"tunes\":{}},{\"id\":\"p-state-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"State can be summarized into context, but the summary is not the state itself. For consequential operations, the runtime may need to re-read the authoritative system immediately before the action rather than trust an earlier model-visible snapshot.\"},\"tunes\":{}},{\"id\":\"state-rule\",\"type\":\"callout\",\"data\":{\"variant\":\"warning\",\"title\":\"Context is a snapshot\",\"body\":\"Once state is copied into a prompt, it can become stale. Context engineering must define when volatile state needs refreshing and which operations require a new authoritative read.\"},\"tunes\":{}},{\"id\":\"h-tools\",\"type\":\"header\",\"data\":{\"text\":\"Tool design is part of context engineering\",\"level\":2},\"tunes\":{}},{\"id\":\"p-tools-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Tools do more than give agents capabilities. Tool names, descriptions, schemas and results become model-visible information that shapes decisions.\"},\"tunes\":{}},{\"id\":\"p-tools-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Anthropic's current context-engineering guidance emphasizes token-efficient tools and warns against bloated tool sets with overlapping functionality. A tool catalog that is difficult for a human to distinguish is also difficult for a model to route reliably.\"},\"tunes\":{}},{\"id\":\"p-tools-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"Tool outputs also need context discipline. Returning an entire 20,000-line log when the agent requested one error condition consumes attention and can bury the decisive evidence.\"},\"tunes\":{}},{\"id\":\"h-jit\",\"type\":\"header\",\"data\":{\"text\":\"Just-in-time context vs preloaded context\",\"level\":2},\"tunes\":{}},{\"id\":\"jit-comparison\",\"type\":\"comparison\",\"data\":{\"title\":\"Two ways to supply information\",\"layout\":\"table\",\"columns\":[{\"id\":\"preload\",\"label\":\"Preloaded context\"},{\"id\":\"jit\",\"label\":\"Just-in-time context\"}],\"rows\":[{\"id\":\"method\",\"label\":\"Method\",\"values\":[\"\",\"\"]},{\"id\":\"strength\",\"label\":\"Strength\",\"values\":[\"\",\"\"]},{\"id\":\"risk\",\"label\":\"Risk\",\"values\":[\"\",\"\"]},{\"id\":\"best\",\"label\":\"Useful when\",\"values\":[\"\",\"\"]}]},\"tunes\":{}},{\"id\":\"p-jit-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Anthropic describes a hybrid pattern in which some stable context is preloaded while agents retrieve additional information at runtime. This is a useful architecture pattern because not every important fact deserves permanent residency in the context window.\"},\"tunes\":{}},{\"id\":\"h-budget\",\"type\":\"header\",\"data\":{\"text\":\"Context is a budget, not a storage system\",\"level\":2},\"tunes\":{}},{\"id\":\"p-budget-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A context window defines capacity. It does not guarantee that every token will be used equally well. The model must distribute attention across instructions, history, evidence, tools and intermediate state.\"},\"tunes\":{}},{\"id\":\"p-budget-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The practical objective is therefore not “fill the window.” It is to maximize the utility of the limited attention budget.\"},\"tunes\":{}},{\"id\":\"p-budget-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"Anthropic formulates a similar principle as finding the smallest high-signal set of tokens that maximizes the probability of the desired behavior. OpenAI's context-management guidance likewise warns that uncurated history, redundant tool results and noisy retrieval can overwhelm even large windows.\"},\"tunes\":{}},{\"id\":\"h-more\",\"type\":\"header\",\"data\":{\"text\":\"Why more context can be worse\",\"level\":2},\"tunes\":{}},{\"id\":\"p-more-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Additional context can introduce irrelevant information, stale state, duplicate evidence, contradictory instructions or positional competition. It can also cause compaction systems to discard details that later become important.\"},\"tunes\":{}},{\"id\":\"p-more-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The classic Lost in the Middle study demonstrated that long-context models can use information differently depending on where relevant content appears, with performance often degrading when decisive information is placed in the middle of long inputs.\"},\"tunes\":{}},{\"id\":\"p-more-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"This does not mean long context is inherently bad. It means availability inside the window is not the same as reliable utilization.\"},\"tunes\":{}},{\"id\":\"h-order\",\"type\":\"header\",\"data\":{\"text\":\"Context ordering should be intentional\",\"level\":2},\"tunes\":{}},{\"id\":\"p-order-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Context construction is also an ordering problem. Critical instructions, current state, decisive evidence and task-specific constraints should not be concatenated arbitrarily.\"},\"tunes\":{}},{\"id\":\"p-order-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"There is no universal perfect ordering for every model and task. The architecture should therefore test whether reordering evidence changes correctness and whether important information remains robust across realistic context variations.\"},\"tunes\":{}},{\"id\":\"p-order-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"A stable answer that changes dramatically when two equally valid passages swap positions indicates context sensitivity that should be measured rather than ignored.\"},\"tunes\":{}},{\"id\":\"h-conflict\",\"type\":\"header\",\"data\":{\"text\":\"Conflicting context needs explicit precedence\",\"level\":2},\"tunes\":{}},{\"id\":\"p-conflict-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A model may receive an old policy and a new policy, a remembered preference and a current explicit instruction, or a cached status and a live API result. The system should not expect the model to infer precedence from prose style.\"},\"tunes\":{}},{\"id\":\"p-conflict-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Context engineering should encode precedence through source selection, metadata, ordering or explicit instructions: current authoritative state overrides stale copies; explicit current user instruction overrides older inferred preference; approved policy supersedes obsolete drafts.\"},\"tunes\":{}},{\"id\":\"conflict-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Conflict\",\"Preferred context rule\"],[\"Current state vs remembered state\",\"Refresh and prefer the authoritative current source.\"],[\"Current policy vs superseded policy\",\"Include current version; keep old version only when historical comparison is required.\"],[\"Explicit user instruction vs old inferred preference\",\"Prefer the current explicit instruction.\"],[\"Primary source vs secondary summary\",\"Use primary source for claims that require authority; summary may support explanation.\"],[\"Tool observation vs model prior\",\"Prefer current observed state when the tool is authoritative for that fact.\"],[\"Two unresolved authoritative sources\",\"Expose the conflict rather than fabricating one consistent answer.\"]]},\"tunes\":{}},{\"id\":\"h-compaction\",\"type\":\"header\",\"data\":{\"text\":\"Compaction is context transformation, not lossless storage\",\"level\":2},\"tunes\":{}},{\"id\":\"p-comp-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Long-running systems eventually need to trim, summarize or compact history. Compaction creates a new representation of prior context so the agent can continue without replaying every token.\"},\"tunes\":{}},{\"id\":\"p-comp-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"OpenAI's context-management examples use trimming and compression for long-running sessions. Anthropic describes compaction as a primary technique for maintaining coherence when an interaction approaches the context limit.\"},\"tunes\":{}},{\"id\":\"p-comp-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The difficult part is deciding what cannot be safely removed: unresolved tasks, identifiers, user constraints, security boundaries, architecture decisions, exceptions, source provenance and the conditions that make a previous conclusion valid.\"},\"tunes\":{}},{\"id\":\"compaction-rule\",\"type\":\"callout\",\"data\":{\"variant\":\"warning\",\"title\":\"A summary can preserve the conclusion and destroy the reason\",\"body\":\"If compaction keeps “use approach X” but discards why X was chosen, which version was tested or what condition would invalidate it, later responses can remain internally consistent while becoming externally wrong.\"},\"tunes\":{}},{\"id\":\"h-validity\",\"type\":\"header\",\"data\":{\"text\":\"Preserve validity boundaries\",\"level\":2},\"tunes\":{}},{\"id\":\"p-validity-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Important conclusions should carry the conditions under which they remain supported: version, date, scope, assumptions, source authority and unresolved disagreement.\"},\"tunes\":{}},{\"id\":\"p-validity-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Context engineering is therefore connected to the Answer Validity Boundary. The context assembler should not strip away the metadata that determines whether evidence still applies.\"},\"tunes\":{}},{\"id\":\"ref-avb\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers\",\"title\":\"The Answer Validity Boundary: The Missing Layer Between Relevance and Reliable AI Answers\",\"excerpt\":\"A framework for preserving the scope, assumptions, versions and evidence conditions under which an AI claim remains supported.\",\"ctaLabel\":\"Read the Answer Validity Boundary\"},\"tunes\":{}},{\"id\":\"h-security\",\"type\":\"header\",\"data\":{\"text\":\"Context engineering is also a security boundary\",\"level\":2},\"tunes\":{}},{\"id\":\"p-sec-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Data that reaches the model has crossed an important system boundary. Context assembly must therefore respect authorization, tenant isolation, confidentiality and data-minimization rules.\"},\"tunes\":{}},{\"id\":\"p-sec-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"A retriever may technically find a passage the current user cannot access. The correct design is to prevent that passage from entering model context rather than rely on the model to ignore it.\"},\"tunes\":{}},{\"id\":\"p-sec-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"Tool outputs can also contain untrusted instructions or adversarial content. Context engineering should preserve the distinction between application instructions and external data so retrieved text cannot silently acquire instruction authority.\"},\"tunes\":{}},{\"id\":\"h-architecture\",\"type\":\"header\",\"data\":{\"text\":\"A practical context-engineering architecture\",\"level\":2},\"tunes\":{}},{\"id\":\"arch-note\",\"type\":\"callout\",\"data\":{\"variant\":\"note\",\"title\":\"Proposed architecture model\",\"body\":\"The following layers are a practical synthesis for production systems, not a formal industry standard. The purpose is to keep information ownership separate from the temporary model-facing context.\"},\"tunes\":{}},{\"id\":\"arch-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Layer\",\"Responsibility\"],[\"Authoritative systems\",\"Own current business\u002Fsystem state and official records.\"],[\"Knowledge sources\",\"Own documents, policies, specifications, research or external evidence.\"],[\"Memory store\",\"Preserves selected information across turns or sessions.\"],[\"Retrieval layer\",\"Locates task-relevant candidates from external sources.\"],[\"Tool\u002Fruntime layer\",\"Reads state, performs actions and returns observations.\"],[\"Context assembler\",\"Selects, filters, deduplicates, orders and formats model-visible information.\"],[\"Model\",\"Reasons and generates over the assembled context.\"],[\"Validation\u002Fevaluation\",\"Checks whether selected context and resulting output satisfy task-specific requirements.\"]]},\"tunes\":{}},{\"id\":\"p-arch-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The context assembler is conceptually important even when no module has that exact name. In a small application it may be ordinary application code. In a large agent platform it may combine session management, retrieval, memory, tool middleware, compaction and policy enforcement.\"},\"tunes\":{}},{\"id\":\"h-policy\",\"type\":\"header\",\"data\":{\"text\":\"A practical context construction policy\",\"level\":2},\"tunes\":{}},{\"id\":\"policy-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Rule\",\"Why it matters\"],[\"Start from the current task\",\"Do not carry information merely because it existed earlier.\"],[\"Re-read volatile state\",\"Memory and old context can be stale.\"],[\"Retrieve just enough evidence\",\"Large candidate sets can dilute decisive information.\"],[\"Preserve source metadata\",\"Version, date and authority determine whether evidence still applies.\"],[\"Remove duplicate content\",\"Redundancy consumes tokens without adding information.\"],[\"Prefer structured summaries for large tool output\",\"Expose decisive fields instead of raw noise where fidelity permits.\"],[\"Keep rules with exceptions\",\"Separating a rule from its exception creates false certainty.\"],[\"Make precedence explicit\",\"Do not ask the model to infer which conflicting source wins.\"],[\"Keep durable state outside context\",\"Context is temporary working memory, not the database.\"],[\"Compact with retention tests\",\"Verify that identifiers, constraints, provenance and unresolved state survive.\"],[\"Measure order sensitivity\",\"Correctness should not depend accidentally on arbitrary document ordering.\"],[\"Evaluate context separately from model quality\",\"A stronger model cannot compensate reliably for missing or unauthorized evidence.\"]]},\"tunes\":{}},{\"id\":\"h-eval\",\"type\":\"header\",\"data\":{\"text\":\"How to evaluate context engineering\",\"level\":2},\"tunes\":{}},{\"id\":\"eval-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Property\",\"Question\",\"Example test\"],[\"Sufficiency\",\"Does the context contain everything required to solve the task?\",\"Remove one evidence item and observe whether the answer becomes unsupported.\"],[\"Relevance\",\"How much context is unnecessary for the task?\",\"Measure quality as irrelevant passages are added or removed.\"],[\"Authority\",\"Are decisive claims grounded in the correct source class?\",\"Inject a more fluent but non-authoritative conflicting source.\"],[\"Freshness\",\"Does current state override stale copies?\",\"Change authoritative state after a previous turn and rerun.\"],[\"Position robustness\",\"Does answer quality depend strongly on evidence position?\",\"Randomize candidate ordering across repeated trials.\"],[\"Conflict handling\",\"Does the model follow explicit precedence rules?\",\"Present old and new state together.\"],[\"Compaction retention\",\"Does summarization preserve constraints and validity boundaries?\",\"Compare pre\u002Fpost-compaction task performance.\"],[\"Token efficiency\",\"Does extra context improve quality enough to justify latency\u002Fcost?\",\"Run controlled context-size ablations.\"],[\"Security\",\"Can unauthorized or adversarial content enter model context?\",\"Test tenant, permission and prompt-injection boundaries.\"]]},\"tunes\":{}},{\"id\":\"h-rag-diagnostic\",\"type\":\"header\",\"data\":{\"text\":\"Context assembly is a distinct RAG failure layer\",\"level\":2},\"tunes\":{}},{\"id\":\"p-ragdiag-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A RAG pipeline can succeed at retrieval and still fail downstream. The relevant source may appear at rank 2, yet the context assembler can drop it, truncate it, combine it with stale contradictory material or exceed the token budget.\"},\"tunes\":{}},{\"id\":\"p-ragdiag-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"This is why retrieval traces should be compared with the actual context sent to the model. Without that comparison, context failures are easily misdiagnosed as embedding or model failures.\"},\"tunes\":{}},{\"id\":\"ref-ragfail\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method\",\"title\":\"RAG Failed — But Which Layer Actually Failed? A Diagnostic Method\",\"excerpt\":\"A layer-by-layer approach to separating source coverage, retrieval, ranking, context assembly, generation, evidence attribution and freshness failures.\",\"ctaLabel\":\"Read the RAG diagnostic method\"},\"tunes\":{}},{\"id\":\"h-implementation\",\"type\":\"header\",\"data\":{\"text\":\"Original implementation evidence\",\"level\":2},\"tunes\":{}},{\"id\":\"h-sot-engine\",\"type\":\"header\",\"data\":{\"text\":\"Source of Truth Research Engine: bounded research instead of unlimited context\",\"level\":3},\"tunes\":{}},{\"id\":\"p-sot-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The Source of Truth Research Engine separates discovery, acquisition, extraction, verification, contradiction analysis and synthesis into bounded research stages instead of sending one huge research task and all accumulated material into a single model call.\"},\"tunes\":{}},{\"id\":\"p-sot-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Its evidence model stores Sources, Artifacts, Claims, Relations, Contradictions and provenance outside the model context. The model can receive the subset needed for the current research step while durable evidence remains in the external store.\"},\"tunes\":{}},{\"id\":\"p-sot-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"That is a concrete context-engineering pattern: durable research state lives outside the model window; the active model context is reconstructed for the current stage.\"},\"tunes\":{}},{\"id\":\"h-ai-client\",\"type\":\"header\",\"data\":{\"text\":\"Aaasaasa AI Client: runtime, permissions and context are separate concerns\",\"level\":3},\"tunes\":{}},{\"id\":\"p-client-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Aaasaasa AI Client separates provider\u002Fmodel selection, runtime location, workspace permissions, local resources and tool access. This prevents the model context from becoming the owner of authorization or application state.\"},\"tunes\":{}},{\"id\":\"p-client-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Direct Chat and agentic runtimes can have different tool capabilities. Workspace permission profiles are enforced by the runtime rather than merely described in natural-language context. This distinction is important: context can tell a model what it should do, while the runtime must still enforce what it is actually allowed to do.\"},\"tunes\":{}},{\"id\":\"p-client-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The implementation evidence here is architectural separation, not a claim that every advanced context-management technique described in this article is already implemented.\"},\"tunes\":{}},{\"id\":\"impl-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Implementation pattern\",\"Context-engineering lesson\"],[\"External evidence store\",\"Durable knowledge does not need to remain in the model window.\"],[\"Bounded research stages\",\"Different steps can receive different context instead of accumulating one giant history.\"],[\"Claims + provenance outside context\",\"Evidence identity survives beyond temporary inference state.\"],[\"Runtime-enforced permissions\",\"Security authority does not depend on the model remembering an instruction.\"],[\"Separate local\u002Fprovider\u002Fmodel\u002Fruntime concepts\",\"Context is only one layer of the wider AI application architecture.\"]]},\"tunes\":{}},{\"id\":\"impl-boundary\",\"type\":\"callout\",\"data\":{\"variant\":\"note\",\"title\":\"Evidence boundary\",\"body\":\"These implementations support the architectural separation between durable state, retrieval, runtime controls and model-facing context. They are not presented as benchmark proof that one context strategy is universally optimal.\"},\"tunes\":{}},{\"id\":\"h-failures\",\"type\":\"header\",\"data\":{\"text\":\"Common context-engineering failure modes\",\"level\":2},\"tunes\":{}},{\"id\":\"failure-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Failure mode\",\"What goes wrong\"],[\"Replay the entire conversation forever\",\"Old assumptions, repetition and token growth overwhelm current intent.\"],[\"Put every retrieved result into the prompt\",\"Noise, duplication and conflicting versions dilute decisive evidence.\"],[\"Use memory as current state\",\"Stale information silently replaces authoritative live state.\"],[\"Return raw tool output\",\"Large logs or responses consume attention without adding decision value.\"],[\"Hide tool descriptions behind vague names\",\"The model cannot reliably decide which capability to use.\"],[\"Compact without retention tests\",\"Critical constraints, identifiers or exceptions disappear.\"],[\"Mix instructions and untrusted data\",\"External content can be interpreted as higher-authority instruction.\"],[\"Use one static context template for every task\",\"Different tasks receive irrelevant information and miss task-specific evidence.\"],[\"Ignore source version\u002Fdate\",\"Stale but relevant evidence can dominate current authoritative state.\"],[\"Treat a larger context window as a quality guarantee\",\"Capacity increases while attention and conflict problems remain.\"]]},\"tunes\":{}},{\"id\":\"h-misconceptions\",\"type\":\"header\",\"data\":{\"text\":\"Common misconceptions\",\"level\":2},\"tunes\":{}},{\"id\":\"misconceptions-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Misconception\",\"Correction\"],[\"“Context engineering is just prompt engineering with a new name.”\",\"Prompts are one component; context engineering also covers retrieval, memory, state, tool results, history and compaction.\"],[\"“Context means chat history.”\",\"History is only one possible context source.\"],[\"“More context is always better.”\",\"Additional information can reduce signal, introduce conflicts and increase cost.\"],[\"“If retrieval found it, the model saw it.”\",\"Retrieved candidates can be filtered, truncated or omitted before inference.\"],[\"“Long context removes the need for RAG.”\",\"Large windows increase capacity but do not solve freshness, authority, permissions or dynamic retrieval.\"],[\"“Memory should always be loaded.”\",\"Memory should be selected according to the current task.\"],[\"“A summary preserves everything important.”\",\"Compaction is lossy unless explicitly evaluated for retention.\"],[\"“Instructions can enforce permissions.”\",\"Authorization must be enforced by runtime\u002Fapplication controls, not only by context.\"],[\"“One context recipe works for every model.”\",\"Context sensitivity varies by model, task, corpus and runtime.\"],[\"“Context engineering is only for agents.”\",\"Agents amplify the need, but ordinary RAG and conversational applications also require context construction.\"]]},\"tunes\":{}},{\"id\":\"h-sequence\",\"type\":\"header\",\"data\":{\"text\":\"A practical context-engineering sequence\",\"level\":2},\"tunes\":{}},{\"id\":\"design-sequence\",\"type\":\"processFlow\",\"data\":{\"title\":\"Construct context from the current decision backward\",\"orientation\":\"auto\",\"steps\":[{\"label\":\"1. Define the next model decision\",\"description\":\"Specify what the model must answer, classify, plan or choose at this step.\"},{\"label\":\"2. Identify required facts and constraints\",\"description\":\"List the minimum state, rules, evidence and instructions that can materially change the result.\"},{\"label\":\"3. Resolve authority and permissions\",\"description\":\"Determine which sources are current, authoritative and accessible to the current principal.\"},{\"label\":\"4. Retrieve or read on demand\",\"description\":\"Acquire the necessary evidence and volatile state rather than relying on stale context.\"},{\"label\":\"5. Reduce noise\",\"description\":\"Deduplicate, summarize or select passages without discarding decisive exceptions or provenance.\"},{\"label\":\"6. Structure and order\",\"description\":\"Make instructions, current state, evidence and tool observations distinguishable.\"},{\"label\":\"7. Fit the token budget\",\"description\":\"Prefer high-signal context and move durable information outside the window.\"},{\"label\":\"8. Run the model\",\"description\":\"Execute inference over the assembled context.\"},{\"label\":\"9. Observe failures\",\"description\":\"Capture whether the problem came from missing, stale, noisy, conflicting or poorly ordered context.\"},{\"label\":\"10. Re-evaluate after model\u002Fruntime changes\",\"description\":\"A context strategy is only valid for the models, tools and workloads on which it was tested.\"}]},\"tunes\":{}},{\"id\":\"h-checklist\",\"type\":\"header\",\"data\":{\"text\":\"Context-engineering checklist\",\"level\":2},\"tunes\":{}},{\"id\":\"checklist-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Question\",\"Expected answer\"],[\"What exact decision will the model make next?\",\"A bounded task, not a vague long-term objective.\"],[\"Which information can materially change that decision?\",\"Explicit minimum evidence\u002Fstate set.\"],[\"Which data is authoritative now?\",\"Current source\u002Fversion and freshness rule.\"],[\"Which data is optional background?\",\"Separated from decisive evidence.\"],[\"What must not enter context?\",\"Unauthorized, unnecessary or overly sensitive data.\"],[\"Which memory items are relevant?\",\"Selected by task, not replayed automatically.\"],[\"Which tool outputs should be reduced?\",\"Large responses are transformed into decision-relevant form.\"],[\"Which constraints must survive compaction?\",\"Identifiers, exceptions, obligations, unresolved state and provenance.\"],[\"How is precedence represented?\",\"Current\u002Fauthoritative information can reliably override stale or weaker sources.\"],[\"How will you know context failed?\",\"Context-specific evals and traces exist.\"],[\"Can the answer be reproduced?\",\"Model input or reconstructable context trace is available where appropriate.\"],[\"Can a stronger or larger model change the strategy?\",\"Context policy is version-aware and reevaluated empirically.\"]]},\"tunes\":{}},{\"id\":\"h-edge\",\"type\":\"header\",\"data\":{\"text\":\"Edge cases and limitations\",\"level\":2},\"tunes\":{}},{\"id\":\"p-edge-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Some tasks are simple enough that context engineering reduces to a short system prompt and one user message. Adding retrieval, memory and compaction would only introduce unnecessary architecture.\"},\"tunes\":{}},{\"id\":\"p-edge-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Some tasks require high recall and may intentionally include more context before later synthesis. Research, discovery and legal review can prefer omission avoidance over minimal token count.\"},\"tunes\":{}},{\"id\":\"p-edge-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"Some information should never be summarized before use. Exact contracts, code, cryptographic material, numerical records and regulatory text may require verbatim or structured retrieval where compression could alter meaning.\"},\"tunes\":{}},{\"id\":\"p-edge-4\",\"type\":\"paragraph\",\"data\":{\"text\":\"Long-context behavior varies substantially between models. A strategy validated on one model, context length or tool harness should not automatically be transferred to another.\"},\"tunes\":{}},{\"id\":\"p-edge-5\",\"type\":\"paragraph\",\"data\":{\"text\":\"The model can still ignore or misinterpret excellent context. Context engineering improves the information environment; it does not guarantee reasoning correctness.\"},\"tunes\":{}},{\"id\":\"h-change\",\"type\":\"header\",\"data\":{\"text\":\"What would change this answer?\",\"level\":2},\"tunes\":{}},{\"id\":\"p-change-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Future models may become more robust to long context, positional effects and conflicting information. That could reduce the amount of manual curation required.\"},\"tunes\":{}},{\"id\":\"p-change-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The architectural distinction would still remain useful because permissions, freshness, memory persistence, source authority and external application state exist outside the model regardless of context-window size.\"},\"tunes\":{}},{\"id\":\"p-change-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The recommended balance between preloaded and just-in-time context also changes with latency requirements, tool reliability, corpus size, model cost and how dynamic the underlying information is.\"},\"tunes\":{}},{\"id\":\"h-related\",\"type\":\"header\",\"data\":{\"text\":\"Related canonical knowledge\",\"level\":2},\"tunes\":{}},{\"id\":\"p-related-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Context engineering sits between retrieval and generation. RAG explains how external knowledge is retrieved; R01 separates embeddings, vector search and reranking; context engineering explains what eventually reaches the model.\"},\"tunes\":{}},{\"id\":\"ref-rag\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works\",\"title\":\"What Is RAG? The Simplest Explanation of How It Works\",\"excerpt\":\"The retrieval foundation for understanding how external knowledge can be supplied to a model before generation.\",\"ctaLabel\":\"Read the RAG foundation\"},\"tunes\":{}},{\"id\":\"p-related-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Source-of-Truth architecture answers a different question: not which information is present in context, but which source is authorized to establish a claim.\"},\"tunes\":{}},{\"id\":\"p-related-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The existing article Why More Context Can Make AI Answers Worse is the diagnostic companion to this canonical definition. It focuses on context pollution, position effects, top-k growth, compaction loss and answer degradation rather than redefining context engineering itself.\"},\"tunes\":{}},{\"id\":\"h-faq\",\"type\":\"header\",\"data\":{\"text\":\"Frequently asked questions\",\"level\":2},\"tunes\":{}},{\"id\":\"faq\",\"type\":\"faq\",\"data\":{\"title\":\"Context engineering FAQ\",\"items\":[{\"id\":\"faq1\",\"question\":\"What is context engineering?\",\"answer\":\"Context engineering is the design and runtime management of what information a language model receives at inference time, including instructions, history, retrieved evidence, memory, state, tools and tool results.\"},{\"id\":\"faq2\",\"question\":\"How is context engineering different from prompt engineering?\",\"answer\":\"Prompt engineering focuses on how instructions and examples are written. Context engineering includes prompts but also decides which external information, state, history, memory and tool observations are placed around them.\"},{\"id\":\"faq3\",\"question\":\"Is RAG the same as context engineering?\",\"answer\":\"No. RAG retrieves external information. Context engineering decides how retrieved information is filtered, combined with other state and actually delivered to the model.\"},{\"id\":\"faq4\",\"question\":\"Is memory the same as context?\",\"answer\":\"No. Memory persists information outside the current model call. Context is the subset of information loaded into the current inference.\"},{\"id\":\"faq5\",\"question\":\"Why can more context make an answer worse?\",\"answer\":\"Additional context can introduce noise, stale state, conflicting evidence, duplication and positional competition. Large context capacity does not guarantee equally reliable use of every token.\"},{\"id\":\"faq6\",\"question\":\"What is context compaction?\",\"answer\":\"Compaction summarizes or transforms accumulated history into a smaller representation so a long-running system can continue without replaying every prior token.\"},{\"id\":\"faq7\",\"question\":\"Should current application state be stored in context?\",\"answer\":\"It can be represented in context for reasoning, but consequential operations should often re-read the authoritative source because context snapshots can become stale.\"},{\"id\":\"faq8\",\"question\":\"Is context engineering only needed for AI agents?\",\"answer\":\"No. Agents make context management more dynamic, but RAG systems, assistants, copilots and multi-turn applications also need deliberate context construction.\"}]},\"tunes\":{}},{\"id\":\"h-glossary\",\"type\":\"header\",\"data\":{\"text\":\"Glossary\",\"level\":2},\"tunes\":{}},{\"id\":\"glossary\",\"type\":\"glossary\",\"data\":{\"title\":\"Key context-engineering terms\",\"entries\":[{\"term\":\"Context engineering\",\"definition\":\"The design and runtime management of the information supplied to a language model for a particular inference step.\",\"anchor\":\"context-engineering\"},{\"term\":\"Context window\",\"definition\":\"The model's finite token capacity for the input and, depending on the model interface, associated generated tokens or active sequence.\",\"anchor\":\"context-window\"},{\"term\":\"Prompt engineering\",\"definition\":\"The design of instructions, examples and prompt structure intended to elicit useful model behavior.\",\"anchor\":\"prompt-engineering\"},{\"term\":\"Context assembly\",\"definition\":\"The process of selecting, filtering, ordering and formatting model-visible information before inference.\",\"anchor\":\"context-assembly\"},{\"term\":\"Just-in-time retrieval\",\"definition\":\"Loading information dynamically when the current task requires it instead of preloading all potentially relevant data.\",\"anchor\":\"just-in-time-retrieval\"},{\"term\":\"Compaction\",\"definition\":\"Reducing accumulated context into a smaller representation while attempting to preserve information needed for future steps.\",\"anchor\":\"compaction\"},{\"term\":\"Context pollution\",\"definition\":\"Degradation caused by irrelevant, stale, contradictory or redundant information occupying the model's working context.\",\"anchor\":\"context-pollution\"},{\"term\":\"Application state\",\"definition\":\"The current authoritative condition of the external system, workflow or domain that exists independently of the model context.\",\"anchor\":\"application-state\"},{\"term\":\"Memory\",\"definition\":\"Information stored outside the immediate model invocation for possible use in later turns or sessions.\",\"anchor\":\"memory\"},{\"term\":\"Retrieved context\",\"definition\":\"External information selected by a retrieval system and made available, wholly or partly, to the model.\",\"anchor\":\"retrieved-context\"},{\"term\":\"Position robustness\",\"definition\":\"The degree to which model correctness remains stable when the location or order of relevant context changes.\",\"anchor\":\"position-robustness\"},{\"term\":\"Validity boundary\",\"definition\":\"The scope, time, assumptions, versions and evidence conditions within which a conclusion remains supported.\",\"anchor\":\"validity-boundary\"}]},\"tunes\":{}},{\"id\":\"h-conclusion\",\"type\":\"header\",\"data\":{\"text\":\"Conclusion\",\"level\":2},\"tunes\":{}},{\"id\":\"p-conclusion-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Context engineering is the layer that decides what the model gets to see before it answers. That makes it broader than prompting and downstream of retrieval, while remaining distinct from durable memory and authoritative application state.\"},\"tunes\":{}},{\"id\":\"p-conclusion-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"A strong context architecture does not treat the context window as a database. It keeps durable state and knowledge outside the model, loads what is required for the current decision, preserves authority and provenance, removes unnecessary noise and refreshes volatile information when needed.\"},\"tunes\":{}},{\"id\":\"p-conclusion-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The practical objective is therefore not maximum context. It is minimum sufficient, high-signal, correctly authorized and validity-preserving context for the next model decision.\"},\"tunes\":{}},{\"id\":\"h-sources\",\"type\":\"header\",\"data\":{\"text\":\"Primary sources and current guidance\",\"level\":2},\"tunes\":{}},{\"id\":\"p-sources-note\",\"type\":\"paragraph\",\"data\":{\"text\":\"The sources below support the current context-engineering terminology, long-context behavior and operational context-management patterns. Project sections are explicitly implementation evidence rather than universal claims.\"},\"tunes\":{}},{\"id\":\"src-anthropic\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Feffective-context-engineering-for-ai-agents\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Anthropic — Effective context engineering for AI agents\",\"description\":\"Official engineering guidance defining context engineering, just-in-time retrieval, compaction, structured memory and context curation for agents.\"}},\"tunes\":{}},{\"id\":\"src-openai-session\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdevelopers.openai.com\u002Fcookbook\u002Fexamples\u002Fagents_sdk\u002Fsession_memory\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — Context Engineering: Short-Term Memory Management with Sessions\",\"description\":\"Official cookbook guidance on context management, trimming and compression for long-running agent sessions.\"}},\"tunes\":{}},{\"id\":\"src-openai-agents\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fagents\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — Agents guide\",\"description\":\"Current OpenAI developer guidance on agent runtimes, context across steps and orchestration ownership.\"}},\"tunes\":{}},{\"id\":\"src-lost-middle\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2307.03172\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Lost in the Middle: How Language Models Use Long Contexts\",\"description\":\"Research showing that long-context model performance can depend strongly on the position of relevant information in the input.\"}},\"tunes\":{}}],\"version\":\"2.31.6\"}",{"time":1447,"blocks":1448,"version":2456},1791480654232,[1449,1453,1458,1463,1468,1472,1476,1480,1484,1488,1492,1496,1500,1504,1530,1534,1538,1542,1546,1598,1602,1627,1631,1635,1639,1643,1647,1652,1656,1660,1664,1668,1675,1679,1683,1687,1692,1696,1700,1704,1708,1712,1734,1738,1742,1746,1750,1754,1758,1762,1766,1770,1774,1778,1782,1786,1790,1794,1798,1823,1827,1831,1835,1839,1844,1848,1852,1856,1863,1867,1871,1875,1879,1883,1888,1919,1923,1927,1970,1974,2018,2022,2026,2030,2037,2041,2045,2049,2053,2057,2061,2065,2069,2073,2095,2100,2104,2141,2145,2182,2186,2221,2225,2267,2271,2275,2279,2283,2287,2291,2295,2299,2303,2307,2311,2315,2322,2326,2330,2334,2363,2367,2404,2408,2412,2416,2420,2424,2428,2435,2442,2449],{"id":215,"data":1450,"type":218,"tunes":1452},{"text":1451},"Context engineering is the design of what information a language model receives at inference time, in what form, in what order and for how long. It is broader than prompt engineering because the model context can include system instructions, user messages, retrieved documents, tool results, memory, current application state, examples, structured data and intermediate artifacts. The goal is not to maximize the number of tokens, but to construct the smallest useful context that preserves the information, constraints and evidence needed for the current task.",{},{"id":221,"data":1454,"type":226,"tunes":1457},{"body":1455,"title":1456,"variant":225},"Prompt engineering asks \u003Cstrong>how should we instruct the model?\u003C\u002Fstrong> Context engineering asks \u003Cstrong>what should the model know right now, and how should that information be assembled?\u003C\u002Fstrong>\u003Cbr>\u003Cbr>Retrieval, memory, state management, tool design, history trimming, compaction and ordering are therefore context-engineering mechanisms when they determine the tokens available to the model before it produces the next output.","Direct answer",{},{"id":229,"data":1459,"type":226,"tunes":1462},{"body":1460,"title":1461,"variant":233},"A system can know something without placing it in the current context. It can remember something outside the model window. It can retrieve a document but later exclude it from the final prompt. The model can only directly use the context that reaches the current inference.","Context is not the same as knowledge or memory",{},{"id":236,"data":1464,"type":226,"tunes":1467},{"body":1465,"title":1466,"variant":240},"Context engineering is now established practical terminology in major AI engineering guidance, but it is not a single formal standard with one mandatory architecture. Anthropic describes it as curating and maintaining the optimal set of tokens for inference; OpenAI's current agent guidance treats session context, trimming and compression as explicit engineering concerns for long-running systems.","Current-source note — 8 October 2026",{},{"id":243,"data":1469,"type":248,"tunes":1471},{"title":1470,"maxLevel":246,"minLevel":247},"Contents",{},{"id":251,"data":1473,"type":42,"tunes":1475},{"text":1474,"level":247},"What context engineering really means",{},{"id":256,"data":1477,"type":218,"tunes":1479},{"text":1478},"Every model call is made under a temporary working environment: the current instructions, messages, retrieved evidence, tool outputs and state that fit into the active context window. Context engineering is the discipline of constructing that environment deliberately.",{},{"id":261,"data":1481,"type":218,"tunes":1483},{"text":1482},"The key word is deliberately. A naive system simply concatenates everything it has: full history, all retrieved documents, every tool response and large system prompts. A context-engineered system decides which information is required for the current decision and which information should remain outside the window until needed.",{},{"id":266,"data":1485,"type":218,"tunes":1487},{"text":1486},"This makes context engineering partly an information-architecture problem, partly a runtime problem and partly an evaluation problem. The design must decide what can enter context, where it comes from, which version is current, how conflicts are resolved, how much detail is retained and how the result is tested.",{},{"id":271,"data":1489,"type":42,"tunes":1491},{"text":1490,"level":247},"The simplest example",{},{"id":276,"data":1493,"type":218,"tunes":1495},{"text":1494},"Imagine an internal support assistant. A user asks: “Can this customer cancel without a fee?”",{},{"id":281,"data":1497,"type":218,"tunes":1499},{"text":1498},"The model might need five things: the current cancellation policy, the customer's current contract type, the effective contract date, the relevant exception rules and the user's authorization scope.",{},{"id":286,"data":1501,"type":218,"tunes":1503},{"text":1502},"It does not necessarily need the entire customer database, the full policy archive, every previous conversation or every support ticket. Context engineering is the process that selects and assembles the five useful pieces while excluding unrelated information.",{},{"id":291,"data":1505,"type":317,"tunes":1529},{"steps":1506,"title":1528,"orientation":316},[1507,1510,1513,1516,1519,1522,1525],{"label":1508,"description":1509},"1. Understand the task","Classify what the current question requires and which information types can affect the answer.",{"label":1511,"description":1512},"2. Resolve authoritative state","Read current application or business state that should not be guessed from memory.",{"label":1514,"description":1515},"3. Retrieve supporting knowledge","Find the policy, documents or external evidence relevant to the specific task.",{"label":1517,"description":1518},"4. Apply eligibility and permissions","Exclude data the current user or runtime is not allowed to expose to the model.",{"label":1520,"description":1521},"5. Reduce and structure","Remove duplication, select useful excerpts and preserve critical metadata, conditions and exceptions.",{"label":1523,"description":1524},"6. Order the context","Place instructions, current state and decisive evidence where the model can use them consistently.",{"label":1526,"description":1527},"7. Run inference","The model receives the assembled context and produces the next answer or action proposal.","From application state to model context",{},{"id":320,"data":1531,"type":42,"tunes":1533},{"text":1532,"level":247},"Where the simple example stops",{},{"id":325,"data":1535,"type":218,"tunes":1537},{"text":1536},"Real systems are more difficult because the information needed for one step may not be known before execution begins. An agent can discover new facts through tools, create intermediate files, receive changing external state or span a task longer than one context window.",{},{"id":330,"data":1539,"type":218,"tunes":1541},{"text":1540},"Context engineering therefore becomes dynamic. The context for step 12 should not simply be step 1 context plus eleven layers of accumulated output. It should reflect the current task state, the decisions that still matter and the evidence required for the next action.",{},{"id":335,"data":1543,"type":42,"tunes":1545},{"text":1544,"level":247},"What can enter a model context?",{},{"id":340,"data":1547,"type":391,"tunes":1597},{"content":1548,"stretched":43,"withHeadings":14},[1549,1553,1557,1561,1565,1569,1573,1577,1581,1585,1589,1593],[1550,1551,1552],"Context component","Purpose","Typical risk",[1554,1555,1556],"System \u002F developer instructions","Define role, constraints, policies and behavior","Too vague, contradictory or overloaded with brittle logic",[1558,1559,1560],"Current user request","Defines immediate task and intent","Ambiguity or conflict with prior history",[1562,1563,1564],"Conversation history","Preserves continuity across turns","Stale assumptions, repetition and token growth",[1566,1567,1568],"Retrieved documents","Provide external knowledge\u002Fevidence","Irrelevance, stale versions, weak authority or duplication",[1570,1571,1572],"Current application state","Supplies volatile business\u002Fsystem facts","Using cached or remembered state instead of current authority",[1574,1575,1576],"Tool definitions","Tell the model what capabilities exist and how to call them","Too many overlapping tools or verbose schemas",[1578,1579,1580],"Tool results","Bring observations from the environment into the loop","Large noisy outputs, untrusted content or obsolete observations",[1582,1583,1584],"Memory","Reintroduces selected information from previous interactions","Staleness, incorrect generalization or over-personalization",[1586,1587,1588],"Examples","Demonstrate desired behavior","Too many edge cases can crowd out the current task",[1590,1591,1592],"Intermediate artifacts","Carry plans, summaries, code, calculations or notes","Old intermediate state may be mistaken for final truth",[1594,1595,1596],"Policies \u002F guardrails","Define prohibited or constrained behavior","Conflict with business logic or hidden enforcement gaps",{},{"id":394,"data":1599,"type":42,"tunes":1601},{"text":1600,"level":247},"Context engineering vs prompt engineering",{},{"id":399,"data":1603,"type":431,"tunes":1626},{"rows":1604,"title":1620,"layout":391,"columns":1621},[1605,1608,1611,1614,1617],{"id":403,"label":1606,"values":1607},"Primary focus",[406,406],{"id":408,"label":1609,"values":1610},"Typical scope",[406,406],{"id":412,"label":1612,"values":1613},"When it changes",[406,406],{"id":416,"label":1615,"values":1616},"Typical failure",[406,406],{"id":420,"label":1618,"values":1619},"Relationship",[406,406],"Prompt engineering and context engineering solve different layers",[1622,1624],{"id":426,"label":1623},"Prompt engineering",{"id":429,"label":1625},"Context engineering",{},{"id":434,"data":1628,"type":218,"tunes":1630},{"text":1629},"Anthropic explicitly describes context engineering as the natural progression of prompt engineering for systems in which the model must work with tools, external data, message history and long-running agent state. The practical distinction is useful because a perfectly written prompt cannot compensate for missing authoritative data or a context polluted by contradictory state.",{},{"id":439,"data":1632,"type":42,"tunes":1634},{"text":1633,"level":247},"Context engineering vs retrieval",{},{"id":444,"data":1636,"type":218,"tunes":1638},{"text":1637},"Retrieval selects candidate information from an external corpus or source. Context engineering decides what happens after and around that retrieval.",{},{"id":449,"data":1640,"type":218,"tunes":1642},{"text":1641},"The retriever may return 30 passages. A reranker may reduce them to 10. The context layer may select four passages, remove duplicates, attach source\u002Fversion metadata, combine them with current application state and place them after the system instructions.",{},{"id":454,"data":1644,"type":218,"tunes":1646},{"text":1645},"This is why a RAG system can retrieve the correct passage and still answer badly: the failure may occur during context assembly rather than retrieval.",{},{"id":459,"data":1648,"type":226,"tunes":1651},{"body":1649,"title":1650,"variant":463},"The correct retrieval result is only useful if it survives filtering, ordering, compression and token-budget decisions and actually reaches the model in a usable form.","Retrieval finds candidates; context engineering constructs the model input",{},{"id":466,"data":1653,"type":42,"tunes":1655},{"text":1654,"level":247},"Context engineering vs memory",{},{"id":471,"data":1657,"type":218,"tunes":1659},{"text":1658},"Memory is information preserved outside the immediate model invocation so it can be used again later. Context is the information actually loaded into the current invocation.",{},{"id":476,"data":1661,"type":218,"tunes":1663},{"text":1662},"A memory system may contain thousands of facts, notes or prior decisions. Context engineering selects which of those should be reintroduced for the current task. Loading all memory on every turn defeats the purpose of having an external memory layer.",{},{"id":481,"data":1665,"type":218,"tunes":1667},{"text":1666},"The distinction becomes crucial for volatile state. A remembered project status or user preference can be useful, but current authoritative state may need to be re-read before a consequential decision.",{},{"id":486,"data":1669,"type":492,"tunes":1674},{"url":1670,"title":1671,"excerpt":1672,"ctaLabel":1673},"https:\u002F\u002Fstajic.de\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context","AI Agent Memory Is Not RAG: How to Separate Memory, Retrieval, State and Context","A practical architecture separating what persists, what is authoritative now, what is retrieved and what the model actually receives.","Read the memory architecture article",{},{"id":495,"data":1676,"type":42,"tunes":1678},{"text":1677,"level":247},"Context engineering vs application state",{},{"id":500,"data":1680,"type":218,"tunes":1682},{"text":1681},"Application state is the current condition of the outside system: account balance, ticket status, file version, workflow stage, deployment state or task progress.",{},{"id":505,"data":1684,"type":218,"tunes":1686},{"text":1685},"State can be summarized into context, but the summary is not the state itself. For consequential operations, the runtime may need to re-read the authoritative system immediately before the action rather than trust an earlier model-visible snapshot.",{},{"id":510,"data":1688,"type":226,"tunes":1691},{"body":1689,"title":1690,"variant":233},"Once state is copied into a prompt, it can become stale. Context engineering must define when volatile state needs refreshing and which operations require a new authoritative read.","Context is a snapshot",{},{"id":516,"data":1693,"type":42,"tunes":1695},{"text":1694,"level":247},"Tool design is part of context engineering",{},{"id":521,"data":1697,"type":218,"tunes":1699},{"text":1698},"Tools do more than give agents capabilities. Tool names, descriptions, schemas and results become model-visible information that shapes decisions.",{},{"id":526,"data":1701,"type":218,"tunes":1703},{"text":1702},"Anthropic's current context-engineering guidance emphasizes token-efficient tools and warns against bloated tool sets with overlapping functionality. A tool catalog that is difficult for a human to distinguish is also difficult for a model to route reliably.",{},{"id":531,"data":1705,"type":218,"tunes":1707},{"text":1706},"Tool outputs also need context discipline. Returning an entire 20,000-line log when the agent requested one error condition consumes attention and can bury the decisive evidence.",{},{"id":536,"data":1709,"type":42,"tunes":1711},{"text":1710,"level":247},"Just-in-time context vs preloaded context",{},{"id":541,"data":1713,"type":431,"tunes":1733},{"rows":1714,"title":1727,"layout":391,"columns":1728},[1715,1718,1721,1724],{"id":545,"label":1716,"values":1717},"Method",[406,406],{"id":549,"label":1719,"values":1720},"Strength",[406,406],{"id":553,"label":1722,"values":1723},"Risk",[406,406],{"id":557,"label":1725,"values":1726},"Useful when",[406,406],"Two ways to supply information",[1729,1731],{"id":563,"label":1730},"Preloaded context",{"id":566,"label":1732},"Just-in-time context",{},{"id":570,"data":1735,"type":218,"tunes":1737},{"text":1736},"Anthropic describes a hybrid pattern in which some stable context is preloaded while agents retrieve additional information at runtime. This is a useful architecture pattern because not every important fact deserves permanent residency in the context window.",{},{"id":575,"data":1739,"type":42,"tunes":1741},{"text":1740,"level":247},"Context is a budget, not a storage system",{},{"id":580,"data":1743,"type":218,"tunes":1745},{"text":1744},"A context window defines capacity. It does not guarantee that every token will be used equally well. The model must distribute attention across instructions, history, evidence, tools and intermediate state.",{},{"id":585,"data":1747,"type":218,"tunes":1749},{"text":1748},"The practical objective is therefore not “fill the window.” It is to maximize the utility of the limited attention budget.",{},{"id":590,"data":1751,"type":218,"tunes":1753},{"text":1752},"Anthropic formulates a similar principle as finding the smallest high-signal set of tokens that maximizes the probability of the desired behavior. OpenAI's context-management guidance likewise warns that uncurated history, redundant tool results and noisy retrieval can overwhelm even large windows.",{},{"id":595,"data":1755,"type":42,"tunes":1757},{"text":1756,"level":247},"Why more context can be worse",{},{"id":600,"data":1759,"type":218,"tunes":1761},{"text":1760},"Additional context can introduce irrelevant information, stale state, duplicate evidence, contradictory instructions or positional competition. It can also cause compaction systems to discard details that later become important.",{},{"id":605,"data":1763,"type":218,"tunes":1765},{"text":1764},"The classic Lost in the Middle study demonstrated that long-context models can use information differently depending on where relevant content appears, with performance often degrading when decisive information is placed in the middle of long inputs.",{},{"id":610,"data":1767,"type":218,"tunes":1769},{"text":1768},"This does not mean long context is inherently bad. It means availability inside the window is not the same as reliable utilization.",{},{"id":615,"data":1771,"type":42,"tunes":1773},{"text":1772,"level":247},"Context ordering should be intentional",{},{"id":620,"data":1775,"type":218,"tunes":1777},{"text":1776},"Context construction is also an ordering problem. Critical instructions, current state, decisive evidence and task-specific constraints should not be concatenated arbitrarily.",{},{"id":625,"data":1779,"type":218,"tunes":1781},{"text":1780},"There is no universal perfect ordering for every model and task. The architecture should therefore test whether reordering evidence changes correctness and whether important information remains robust across realistic context variations.",{},{"id":630,"data":1783,"type":218,"tunes":1785},{"text":1784},"A stable answer that changes dramatically when two equally valid passages swap positions indicates context sensitivity that should be measured rather than ignored.",{},{"id":635,"data":1787,"type":42,"tunes":1789},{"text":1788,"level":247},"Conflicting context needs explicit precedence",{},{"id":640,"data":1791,"type":218,"tunes":1793},{"text":1792},"A model may receive an old policy and a new policy, a remembered preference and a current explicit instruction, or a cached status and a live API result. The system should not expect the model to infer precedence from prose style.",{},{"id":645,"data":1795,"type":218,"tunes":1797},{"text":1796},"Context engineering should encode precedence through source selection, metadata, ordering or explicit instructions: current authoritative state overrides stale copies; explicit current user instruction overrides older inferred preference; approved policy supersedes obsolete drafts.",{},{"id":650,"data":1799,"type":391,"tunes":1822},{"content":1800,"stretched":43,"withHeadings":14},[1801,1804,1807,1810,1813,1816,1819],[1802,1803],"Conflict","Preferred context rule",[1805,1806],"Current state vs remembered state","Refresh and prefer the authoritative current source.",[1808,1809],"Current policy vs superseded policy","Include current version; keep old version only when historical comparison is required.",[1811,1812],"Explicit user instruction vs old inferred preference","Prefer the current explicit instruction.",[1814,1815],"Primary source vs secondary summary","Use primary source for claims that require authority; summary may support explanation.",[1817,1818],"Tool observation vs model prior","Prefer current observed state when the tool is authoritative for that fact.",[1820,1821],"Two unresolved authoritative sources","Expose the conflict rather than fabricating one consistent answer.",{},{"id":676,"data":1824,"type":42,"tunes":1826},{"text":1825,"level":247},"Compaction is context transformation, not lossless storage",{},{"id":681,"data":1828,"type":218,"tunes":1830},{"text":1829},"Long-running systems eventually need to trim, summarize or compact history. Compaction creates a new representation of prior context so the agent can continue without replaying every token.",{},{"id":686,"data":1832,"type":218,"tunes":1834},{"text":1833},"OpenAI's context-management examples use trimming and compression for long-running sessions. Anthropic describes compaction as a primary technique for maintaining coherence when an interaction approaches the context limit.",{},{"id":691,"data":1836,"type":218,"tunes":1838},{"text":1837},"The difficult part is deciding what cannot be safely removed: unresolved tasks, identifiers, user constraints, security boundaries, architecture decisions, exceptions, source provenance and the conditions that make a previous conclusion valid.",{},{"id":696,"data":1840,"type":226,"tunes":1843},{"body":1841,"title":1842,"variant":233},"If compaction keeps “use approach X” but discards why X was chosen, which version was tested or what condition would invalidate it, later responses can remain internally consistent while becoming externally wrong.","A summary can preserve the conclusion and destroy the reason",{},{"id":702,"data":1845,"type":42,"tunes":1847},{"text":1846,"level":247},"Preserve validity boundaries",{},{"id":707,"data":1849,"type":218,"tunes":1851},{"text":1850},"Important conclusions should carry the conditions under which they remain supported: version, date, scope, assumptions, source authority and unresolved disagreement.",{},{"id":712,"data":1853,"type":218,"tunes":1855},{"text":1854},"Context engineering is therefore connected to the Answer Validity Boundary. The context assembler should not strip away the metadata that determines whether evidence still applies.",{},{"id":717,"data":1857,"type":492,"tunes":1862},{"url":1858,"title":1859,"excerpt":1860,"ctaLabel":1861},"https:\u002F\u002Fstajic.de\u002Fblog\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers","The Answer Validity Boundary: The Missing Layer Between Relevance and Reliable AI Answers","A framework for preserving the scope, assumptions, versions and evidence conditions under which an AI claim remains supported.","Read the Answer Validity Boundary",{},{"id":725,"data":1864,"type":42,"tunes":1866},{"text":1865,"level":247},"Context engineering is also a security boundary",{},{"id":730,"data":1868,"type":218,"tunes":1870},{"text":1869},"Data that reaches the model has crossed an important system boundary. Context assembly must therefore respect authorization, tenant isolation, confidentiality and data-minimization rules.",{},{"id":735,"data":1872,"type":218,"tunes":1874},{"text":1873},"A retriever may technically find a passage the current user cannot access. The correct design is to prevent that passage from entering model context rather than rely on the model to ignore it.",{},{"id":740,"data":1876,"type":218,"tunes":1878},{"text":1877},"Tool outputs can also contain untrusted instructions or adversarial content. Context engineering should preserve the distinction between application instructions and external data so retrieved text cannot silently acquire instruction authority.",{},{"id":745,"data":1880,"type":42,"tunes":1882},{"text":1881,"level":247},"A practical context-engineering architecture",{},{"id":750,"data":1884,"type":226,"tunes":1887},{"body":1885,"title":1886,"variant":240},"The following layers are a practical synthesis for production systems, not a formal industry standard. The purpose is to keep information ownership separate from the temporary model-facing context.","Proposed architecture model",{},{"id":756,"data":1889,"type":391,"tunes":1918},{"content":1890,"stretched":43,"withHeadings":14},[1891,1894,1897,1900,1903,1906,1909,1912,1915],[1892,1893],"Layer","Responsibility",[1895,1896],"Authoritative systems","Own current business\u002Fsystem state and official records.",[1898,1899],"Knowledge sources","Own documents, policies, specifications, research or external evidence.",[1901,1902],"Memory store","Preserves selected information across turns or sessions.",[1904,1905],"Retrieval layer","Locates task-relevant candidates from external sources.",[1907,1908],"Tool\u002Fruntime layer","Reads state, performs actions and returns observations.",[1910,1911],"Context assembler","Selects, filters, deduplicates, orders and formats model-visible information.",[1913,1914],"Model","Reasons and generates over the assembled context.",[1916,1917],"Validation\u002Fevaluation","Checks whether selected context and resulting output satisfy task-specific requirements.",{},{"id":788,"data":1920,"type":218,"tunes":1922},{"text":1921},"The context assembler is conceptually important even when no module has that exact name. In a small application it may be ordinary application code. In a large agent platform it may combine session management, retrieval, memory, tool middleware, compaction and policy enforcement.",{},{"id":793,"data":1924,"type":42,"tunes":1926},{"text":1925,"level":247},"A practical context construction policy",{},{"id":798,"data":1928,"type":391,"tunes":1969},{"content":1929,"stretched":43,"withHeadings":14},[1930,1933,1936,1939,1942,1945,1948,1951,1954,1957,1960,1963,1966],[1931,1932],"Rule","Why it matters",[1934,1935],"Start from the current task","Do not carry information merely because it existed earlier.",[1937,1938],"Re-read volatile state","Memory and old context can be stale.",[1940,1941],"Retrieve just enough evidence","Large candidate sets can dilute decisive information.",[1943,1944],"Preserve source metadata","Version, date and authority determine whether evidence still applies.",[1946,1947],"Remove duplicate content","Redundancy consumes tokens without adding information.",[1949,1950],"Prefer structured summaries for large tool output","Expose decisive fields instead of raw noise where fidelity permits.",[1952,1953],"Keep rules with exceptions","Separating a rule from its exception creates false certainty.",[1955,1956],"Make precedence explicit","Do not ask the model to infer which conflicting source wins.",[1958,1959],"Keep durable state outside context","Context is temporary working memory, not the database.",[1961,1962],"Compact with retention tests","Verify that identifiers, constraints, provenance and unresolved state survive.",[1964,1965],"Measure order sensitivity","Correctness should not depend accidentally on arbitrary document ordering.",[1967,1968],"Evaluate context separately from model quality","A stronger model cannot compensate reliably for missing or unauthorized evidence.",{},{"id":842,"data":1971,"type":42,"tunes":1973},{"text":1972,"level":247},"How to evaluate context engineering",{},{"id":847,"data":1975,"type":391,"tunes":2017},{"content":1976,"stretched":43,"withHeadings":14},[1977,1981,1985,1989,1993,1997,2001,2005,2009,2013],[1978,1979,1980],"Property","Question","Example test",[1982,1983,1984],"Sufficiency","Does the context contain everything required to solve the task?","Remove one evidence item and observe whether the answer becomes unsupported.",[1986,1987,1988],"Relevance","How much context is unnecessary for the task?","Measure quality as irrelevant passages are added or removed.",[1990,1991,1992],"Authority","Are decisive claims grounded in the correct source class?","Inject a more fluent but non-authoritative conflicting source.",[1994,1995,1996],"Freshness","Does current state override stale copies?","Change authoritative state after a previous turn and rerun.",[1998,1999,2000],"Position robustness","Does answer quality depend strongly on evidence position?","Randomize candidate ordering across repeated trials.",[2002,2003,2004],"Conflict handling","Does the model follow explicit precedence rules?","Present old and new state together.",[2006,2007,2008],"Compaction retention","Does summarization preserve constraints and validity boundaries?","Compare pre\u002Fpost-compaction task performance.",[2010,2011,2012],"Token efficiency","Does extra context improve quality enough to justify latency\u002Fcost?","Run controlled context-size ablations.",[2014,2015,2016],"Security","Can unauthorized or adversarial content enter model context?","Test tenant, permission and prompt-injection boundaries.",{},{"id":892,"data":2019,"type":42,"tunes":2021},{"text":2020,"level":247},"Context assembly is a distinct RAG failure layer",{},{"id":897,"data":2023,"type":218,"tunes":2025},{"text":2024},"A RAG pipeline can succeed at retrieval and still fail downstream. The relevant source may appear at rank 2, yet the context assembler can drop it, truncate it, combine it with stale contradictory material or exceed the token budget.",{},{"id":902,"data":2027,"type":218,"tunes":2029},{"text":2028},"This is why retrieval traces should be compared with the actual context sent to the model. Without that comparison, context failures are easily misdiagnosed as embedding or model failures.",{},{"id":907,"data":2031,"type":492,"tunes":2036},{"url":2032,"title":2033,"excerpt":2034,"ctaLabel":2035},"https:\u002F\u002Fstajic.de\u002Fblog\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method","RAG Failed — But Which Layer Actually Failed? A Diagnostic Method","A layer-by-layer approach to separating source coverage, retrieval, ranking, context assembly, generation, evidence attribution and freshness failures.","Read the RAG diagnostic method",{},{"id":915,"data":2038,"type":42,"tunes":2040},{"text":2039,"level":247},"Original implementation evidence",{},{"id":920,"data":2042,"type":42,"tunes":2044},{"text":2043,"level":246},"Source of Truth Research Engine: bounded research instead of unlimited context",{},{"id":925,"data":2046,"type":218,"tunes":2048},{"text":2047},"The Source of Truth Research Engine separates discovery, acquisition, extraction, verification, contradiction analysis and synthesis into bounded research stages instead of sending one huge research task and all accumulated material into a single model call.",{},{"id":930,"data":2050,"type":218,"tunes":2052},{"text":2051},"Its evidence model stores Sources, Artifacts, Claims, Relations, Contradictions and provenance outside the model context. The model can receive the subset needed for the current research step while durable evidence remains in the external store.",{},{"id":935,"data":2054,"type":218,"tunes":2056},{"text":2055},"That is a concrete context-engineering pattern: durable research state lives outside the model window; the active model context is reconstructed for the current stage.",{},{"id":940,"data":2058,"type":42,"tunes":2060},{"text":2059,"level":246},"Aaasaasa AI Client: runtime, permissions and context are separate concerns",{},{"id":945,"data":2062,"type":218,"tunes":2064},{"text":2063},"Aaasaasa AI Client separates provider\u002Fmodel selection, runtime location, workspace permissions, local resources and tool access. This prevents the model context from becoming the owner of authorization or application state.",{},{"id":950,"data":2066,"type":218,"tunes":2068},{"text":2067},"Direct Chat and agentic runtimes can have different tool capabilities. Workspace permission profiles are enforced by the runtime rather than merely described in natural-language context. This distinction is important: context can tell a model what it should do, while the runtime must still enforce what it is actually allowed to do.",{},{"id":955,"data":2070,"type":218,"tunes":2072},{"text":2071},"The implementation evidence here is architectural separation, not a claim that every advanced context-management technique described in this article is already implemented.",{},{"id":960,"data":2074,"type":391,"tunes":2094},{"content":2075,"stretched":43,"withHeadings":14},[2076,2079,2082,2085,2088,2091],[2077,2078],"Implementation pattern","Context-engineering lesson",[2080,2081],"External evidence store","Durable knowledge does not need to remain in the model window.",[2083,2084],"Bounded research stages","Different steps can receive different context instead of accumulating one giant history.",[2086,2087],"Claims + provenance outside context","Evidence identity survives beyond temporary inference state.",[2089,2090],"Runtime-enforced permissions","Security authority does not depend on the model remembering an instruction.",[2092,2093],"Separate local\u002Fprovider\u002Fmodel\u002Fruntime concepts","Context is only one layer of the wider AI application architecture.",{},{"id":983,"data":2096,"type":226,"tunes":2099},{"body":2097,"title":2098,"variant":240},"These implementations support the architectural separation between durable state, retrieval, runtime controls and model-facing context. They are not presented as benchmark proof that one context strategy is universally optimal.","Evidence boundary",{},{"id":989,"data":2101,"type":42,"tunes":2103},{"text":2102,"level":247},"Common context-engineering failure modes",{},{"id":994,"data":2105,"type":391,"tunes":2140},{"content":2106,"stretched":43,"withHeadings":14},[2107,2110,2113,2116,2119,2122,2125,2128,2131,2134,2137],[2108,2109],"Failure mode","What goes wrong",[2111,2112],"Replay the entire conversation forever","Old assumptions, repetition and token growth overwhelm current intent.",[2114,2115],"Put every retrieved result into the prompt","Noise, duplication and conflicting versions dilute decisive evidence.",[2117,2118],"Use memory as current state","Stale information silently replaces authoritative live state.",[2120,2121],"Return raw tool output","Large logs or responses consume attention without adding decision value.",[2123,2124],"Hide tool descriptions behind vague names","The model cannot reliably decide which capability to use.",[2126,2127],"Compact without retention tests","Critical constraints, identifiers or exceptions disappear.",[2129,2130],"Mix instructions and untrusted data","External content can be interpreted as higher-authority instruction.",[2132,2133],"Use one static context template for every task","Different tasks receive irrelevant information and miss task-specific evidence.",[2135,2136],"Ignore source version\u002Fdate","Stale but relevant evidence can dominate current authoritative state.",[2138,2139],"Treat a larger context window as a quality guarantee","Capacity increases while attention and conflict problems remain.",{},{"id":1032,"data":2142,"type":42,"tunes":2144},{"text":2143,"level":247},"Common misconceptions",{},{"id":1037,"data":2146,"type":391,"tunes":2181},{"content":2147,"stretched":43,"withHeadings":14},[2148,2151,2154,2157,2160,2163,2166,2169,2172,2175,2178],[2149,2150],"Misconception","Correction",[2152,2153],"“Context engineering is just prompt engineering with a new name.”","Prompts are one component; context engineering also covers retrieval, memory, state, tool results, history and compaction.",[2155,2156],"“Context means chat history.”","History is only one possible context source.",[2158,2159],"“More context is always better.”","Additional information can reduce signal, introduce conflicts and increase cost.",[2161,2162],"“If retrieval found it, the model saw it.”","Retrieved candidates can be filtered, truncated or omitted before inference.",[2164,2165],"“Long context removes the need for RAG.”","Large windows increase capacity but do not solve freshness, authority, permissions or dynamic retrieval.",[2167,2168],"“Memory should always be loaded.”","Memory should be selected according to the current task.",[2170,2171],"“A summary preserves everything important.”","Compaction is lossy unless explicitly evaluated for retention.",[2173,2174],"“Instructions can enforce permissions.”","Authorization must be enforced by runtime\u002Fapplication controls, not only by context.",[2176,2177],"“One context recipe works for every model.”","Context sensitivity varies by model, task, corpus and runtime.",[2179,2180],"“Context engineering is only for agents.”","Agents amplify the need, but ordinary RAG and conversational applications also require context construction.",{},{"id":1075,"data":2183,"type":42,"tunes":2185},{"text":2184,"level":247},"A practical context-engineering sequence",{},{"id":1080,"data":2187,"type":317,"tunes":2220},{"steps":2188,"title":2219,"orientation":316},[2189,2192,2195,2198,2201,2204,2207,2210,2213,2216],{"label":2190,"description":2191},"1. Define the next model decision","Specify what the model must answer, classify, plan or choose at this step.",{"label":2193,"description":2194},"2. Identify required facts and constraints","List the minimum state, rules, evidence and instructions that can materially change the result.",{"label":2196,"description":2197},"3. Resolve authority and permissions","Determine which sources are current, authoritative and accessible to the current principal.",{"label":2199,"description":2200},"4. Retrieve or read on demand","Acquire the necessary evidence and volatile state rather than relying on stale context.",{"label":2202,"description":2203},"5. Reduce noise","Deduplicate, summarize or select passages without discarding decisive exceptions or provenance.",{"label":2205,"description":2206},"6. Structure and order","Make instructions, current state, evidence and tool observations distinguishable.",{"label":2208,"description":2209},"7. Fit the token budget","Prefer high-signal context and move durable information outside the window.",{"label":2211,"description":2212},"8. Run the model","Execute inference over the assembled context.",{"label":2214,"description":2215},"9. Observe failures","Capture whether the problem came from missing, stale, noisy, conflicting or poorly ordered context.",{"label":2217,"description":2218},"10. Re-evaluate after model\u002Fruntime changes","A context strategy is only valid for the models, tools and workloads on which it was tested.","Construct context from the current decision backward",{},{"id":1116,"data":2222,"type":42,"tunes":2224},{"text":2223,"level":247},"Context-engineering checklist",{},{"id":1121,"data":2226,"type":391,"tunes":2266},{"content":2227,"stretched":43,"withHeadings":14},[2228,2230,2233,2236,2239,2242,2245,2248,2251,2254,2257,2260,2263],[1979,2229],"Expected answer",[2231,2232],"What exact decision will the model make next?","A bounded task, not a vague long-term objective.",[2234,2235],"Which information can materially change that decision?","Explicit minimum evidence\u002Fstate set.",[2237,2238],"Which data is authoritative now?","Current source\u002Fversion and freshness rule.",[2240,2241],"Which data is optional background?","Separated from decisive evidence.",[2243,2244],"What must not enter context?","Unauthorized, unnecessary or overly sensitive data.",[2246,2247],"Which memory items are relevant?","Selected by task, not replayed automatically.",[2249,2250],"Which tool outputs should be reduced?","Large responses are transformed into decision-relevant form.",[2252,2253],"Which constraints must survive compaction?","Identifiers, exceptions, obligations, unresolved state and provenance.",[2255,2256],"How is precedence represented?","Current\u002Fauthoritative information can reliably override stale or weaker sources.",[2258,2259],"How will you know context failed?","Context-specific evals and traces exist.",[2261,2262],"Can the answer be reproduced?","Model input or reconstructable context trace is available where appropriate.",[2264,2265],"Can a stronger or larger model change the strategy?","Context policy is version-aware and reevaluated empirically.",{},{"id":1164,"data":2268,"type":42,"tunes":2270},{"text":2269,"level":247},"Edge cases and limitations",{},{"id":1169,"data":2272,"type":218,"tunes":2274},{"text":2273},"Some tasks are simple enough that context engineering reduces to a short system prompt and one user message. Adding retrieval, memory and compaction would only introduce unnecessary architecture.",{},{"id":1174,"data":2276,"type":218,"tunes":2278},{"text":2277},"Some tasks require high recall and may intentionally include more context before later synthesis. Research, discovery and legal review can prefer omission avoidance over minimal token count.",{},{"id":1179,"data":2280,"type":218,"tunes":2282},{"text":2281},"Some information should never be summarized before use. Exact contracts, code, cryptographic material, numerical records and regulatory text may require verbatim or structured retrieval where compression could alter meaning.",{},{"id":1184,"data":2284,"type":218,"tunes":2286},{"text":2285},"Long-context behavior varies substantially between models. A strategy validated on one model, context length or tool harness should not automatically be transferred to another.",{},{"id":1189,"data":2288,"type":218,"tunes":2290},{"text":2289},"The model can still ignore or misinterpret excellent context. Context engineering improves the information environment; it does not guarantee reasoning correctness.",{},{"id":1194,"data":2292,"type":42,"tunes":2294},{"text":2293,"level":247},"What would change this answer?",{},{"id":1199,"data":2296,"type":218,"tunes":2298},{"text":2297},"Future models may become more robust to long context, positional effects and conflicting information. That could reduce the amount of manual curation required.",{},{"id":1204,"data":2300,"type":218,"tunes":2302},{"text":2301},"The architectural distinction would still remain useful because permissions, freshness, memory persistence, source authority and external application state exist outside the model regardless of context-window size.",{},{"id":1209,"data":2304,"type":218,"tunes":2306},{"text":2305},"The recommended balance between preloaded and just-in-time context also changes with latency requirements, tool reliability, corpus size, model cost and how dynamic the underlying information is.",{},{"id":1214,"data":2308,"type":42,"tunes":2310},{"text":2309,"level":247},"Related canonical knowledge",{},{"id":1219,"data":2312,"type":218,"tunes":2314},{"text":2313},"Context engineering sits between retrieval and generation. RAG explains how external knowledge is retrieved; R01 separates embeddings, vector search and reranking; context engineering explains what eventually reaches the model.",{},{"id":1224,"data":2316,"type":492,"tunes":2321},{"url":2317,"title":2318,"excerpt":2319,"ctaLabel":2320},"https:\u002F\u002Fstajic.de\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works","What Is RAG? The Simplest Explanation of How It Works","The retrieval foundation for understanding how external knowledge can be supplied to a model before generation.","Read the RAG foundation",{},{"id":1232,"data":2323,"type":218,"tunes":2325},{"text":2324},"Source-of-Truth architecture answers a different question: not which information is present in context, but which source is authorized to establish a claim.",{},{"id":1237,"data":2327,"type":218,"tunes":2329},{"text":2328},"The existing article Why More Context Can Make AI Answers Worse is the diagnostic companion to this canonical definition. It focuses on context pollution, position effects, top-k growth, compaction loss and answer degradation rather than redefining context engineering itself.",{},{"id":1242,"data":2331,"type":42,"tunes":2333},{"text":2332,"level":247},"Frequently asked questions",{},{"id":1247,"data":2335,"type":1247,"tunes":2362},{"items":2336,"title":2361},[2337,2340,2343,2346,2349,2352,2355,2358],{"id":1251,"answer":2338,"question":2339},"Context engineering is the design and runtime management of what information a language model receives at inference time, including instructions, history, retrieved evidence, memory, state, tools and tool results.","What is context engineering?",{"id":1255,"answer":2341,"question":2342},"Prompt engineering focuses on how instructions and examples are written. Context engineering includes prompts but also decides which external information, state, history, memory and tool observations are placed around them.","How is context engineering different from prompt engineering?",{"id":1259,"answer":2344,"question":2345},"No. RAG retrieves external information. Context engineering decides how retrieved information is filtered, combined with other state and actually delivered to the model.","Is RAG the same as context engineering?",{"id":1263,"answer":2347,"question":2348},"No. Memory persists information outside the current model call. Context is the subset of information loaded into the current inference.","Is memory the same as context?",{"id":1267,"answer":2350,"question":2351},"Additional context can introduce noise, stale state, conflicting evidence, duplication and positional competition. Large context capacity does not guarantee equally reliable use of every token.","Why can more context make an answer worse?",{"id":1271,"answer":2353,"question":2354},"Compaction summarizes or transforms accumulated history into a smaller representation so a long-running system can continue without replaying every prior token.","What is context compaction?",{"id":1275,"answer":2356,"question":2357},"It can be represented in context for reasoning, but consequential operations should often re-read the authoritative source because context snapshots can become stale.","Should current application state be stored in context?",{"id":1279,"answer":2359,"question":2360},"No. Agents make context management more dynamic, but RAG systems, assistants, copilots and multi-turn applications also need deliberate context construction.","Is context engineering only needed for AI agents?","Context engineering FAQ",{},{"id":1285,"data":2364,"type":42,"tunes":2366},{"text":2365,"level":247},"Glossary",{},{"id":1290,"data":2368,"type":1290,"tunes":2403},{"title":2369,"entries":2370},"Key context-engineering terms",[2371,2373,2376,2378,2381,2384,2387,2390,2393,2395,2398,2400],{"term":1625,"anchor":1295,"definition":2372},"The design and runtime management of the information supplied to a language model for a particular inference step.",{"term":2374,"anchor":1299,"definition":2375},"Context window","The model's finite token capacity for the input and, depending on the model interface, associated generated tokens or active sequence.",{"term":1623,"anchor":1302,"definition":2377},"The design of instructions, examples and prompt structure intended to elicit useful model behavior.",{"term":2379,"anchor":1306,"definition":2380},"Context assembly","The process of selecting, filtering, ordering and formatting model-visible information before inference.",{"term":2382,"anchor":1310,"definition":2383},"Just-in-time retrieval","Loading information dynamically when the current task requires it instead of preloading all potentially relevant data.",{"term":2385,"anchor":1314,"definition":2386},"Compaction","Reducing accumulated context into a smaller representation while attempting to preserve information needed for future steps.",{"term":2388,"anchor":1318,"definition":2389},"Context pollution","Degradation caused by irrelevant, stale, contradictory or redundant information occupying the model's working context.",{"term":2391,"anchor":1322,"definition":2392},"Application state","The current authoritative condition of the external system, workflow or domain that exists independently of the model context.",{"term":1582,"anchor":1325,"definition":2394},"Information stored outside the immediate model invocation for possible use in later turns or sessions.",{"term":2396,"anchor":1329,"definition":2397},"Retrieved context","External information selected by a retrieval system and made available, wholly or partly, to the model.",{"term":1998,"anchor":1332,"definition":2399},"The degree to which model correctness remains stable when the location or order of relevant context changes.",{"term":2401,"anchor":1336,"definition":2402},"Validity boundary","The scope, time, assumptions, versions and evidence conditions within which a conclusion remains supported.",{},{"id":1340,"data":2405,"type":42,"tunes":2407},{"text":2406,"level":247},"Conclusion",{},{"id":1345,"data":2409,"type":218,"tunes":2411},{"text":2410},"Context engineering is the layer that decides what the model gets to see before it answers. That makes it broader than prompting and downstream of retrieval, while remaining distinct from durable memory and authoritative application state.",{},{"id":1350,"data":2413,"type":218,"tunes":2415},{"text":2414},"A strong context architecture does not treat the context window as a database. It keeps durable state and knowledge outside the model, loads what is required for the current decision, preserves authority and provenance, removes unnecessary noise and refreshes volatile information when needed.",{},{"id":1355,"data":2417,"type":218,"tunes":2419},{"text":2418},"The practical objective is therefore not maximum context. It is minimum sufficient, high-signal, correctly authorized and validity-preserving context for the next model decision.",{},{"id":1360,"data":2421,"type":42,"tunes":2423},{"text":2422,"level":247},"Primary sources and current guidance",{},{"id":1365,"data":2425,"type":218,"tunes":2427},{"text":2426},"The sources below support the current context-engineering terminology, long-context behavior and operational context-management patterns. Project sections are explicitly implementation evidence rather than universal claims.",{},{"id":1370,"data":2429,"type":1377,"tunes":2434},{"link":1372,"meta":2430},{"image":2431,"title":2432,"description":2433},{"url":406},"Anthropic — Effective context engineering for AI agents","Official engineering guidance defining context engineering, just-in-time retrieval, compaction, structured memory and context curation for agents.",{},{"id":1380,"data":2436,"type":1377,"tunes":2441},{"link":1382,"meta":2437},{"image":2438,"title":2439,"description":2440},{"url":406},"OpenAI — Context Engineering: Short-Term Memory Management with Sessions","Official cookbook guidance on context management, trimming and compression for long-running agent sessions.",{},{"id":1389,"data":2443,"type":1377,"tunes":2448},{"link":1391,"meta":2444},{"image":2445,"title":2446,"description":2447},{"url":406},"OpenAI — Agents guide","Current OpenAI developer guidance on agent runtimes, context across steps and orchestration ownership.",{},{"id":1398,"data":2450,"type":1377,"tunes":2455},{"link":1400,"meta":2451},{"image":2452,"title":2453,"description":2454},{"url":406},"Lost in the Middle: How Language Models Use Long Contexts","Research showing that long-context model performance can depend strongly on the position of relevant information in the input.",{},"2.31.6","Context engineering designs what information an AI model receives before inference, including prompts, retrieval, memory, application state, tool results and conversation history.",{"lang":7,"title":208,"content":210,"contentJson":2459,"excerpt":1407},{"time":212,"blocks":2460,"version":1406},[2461,2464,2467,2470,2473,2476,2479,2482,2485,2488,2491,2494,2497,2500,2511,2514,2517,2520,2523,2539,2542,2559,2562,2565,2568,2571,2574,2577,2580,2583,2586,2589,2592,2595,2598,2601,2604,2607,2610,2613,2616,2619,2634,2637,2640,2643,2646,2649,2652,2655,2658,2661,2664,2667,2670,2673,2676,2679,2682,2693,2696,2699,2702,2705,2708,2711,2714,2717,2720,2723,2726,2729,2732,2735,2738,2751,2754,2757,2774,2777,2791,2794,2797,2800,2803,2806,2809,2812,2815,2818,2821,2824,2827,2830,2840,2843,2846,2861,2864,2879,2882,2896,2899,2916,2919,2922,2925,2928,2931,2934,2937,2940,2943,2946,2949,2952,2955,2958,2961,2964,2976,2979,2995,2998,3001,3004,3007,3010,3013,3018,3023,3028],{"id":215,"data":2462,"type":218,"tunes":2463},{"text":217},{},{"id":221,"data":2465,"type":226,"tunes":2466},{"body":223,"title":224,"variant":225},{},{"id":229,"data":2468,"type":226,"tunes":2469},{"body":231,"title":232,"variant":233},{},{"id":236,"data":2471,"type":226,"tunes":2472},{"body":238,"title":239,"variant":240},{},{"id":243,"data":2474,"type":248,"tunes":2475},{"title":245,"maxLevel":246,"minLevel":247},{},{"id":251,"data":2477,"type":42,"tunes":2478},{"text":253,"level":247},{},{"id":256,"data":2480,"type":218,"tunes":2481},{"text":258},{},{"id":261,"data":2483,"type":218,"tunes":2484},{"text":263},{},{"id":266,"data":2486,"type":218,"tunes":2487},{"text":268},{},{"id":271,"data":2489,"type":42,"tunes":2490},{"text":273,"level":247},{},{"id":276,"data":2492,"type":218,"tunes":2493},{"text":278},{},{"id":281,"data":2495,"type":218,"tunes":2496},{"text":283},{},{"id":286,"data":2498,"type":218,"tunes":2499},{"text":288},{},{"id":291,"data":2501,"type":317,"tunes":2510},{"steps":2502,"title":315,"orientation":316},[2503,2504,2505,2506,2507,2508,2509],{"label":295,"description":296},{"label":298,"description":299},{"label":301,"description":302},{"label":304,"description":305},{"label":307,"description":308},{"label":310,"description":311},{"label":313,"description":314},{},{"id":320,"data":2512,"type":42,"tunes":2513},{"text":322,"level":247},{},{"id":325,"data":2515,"type":218,"tunes":2516},{"text":327},{},{"id":330,"data":2518,"type":218,"tunes":2519},{"text":332},{},{"id":335,"data":2521,"type":42,"tunes":2522},{"text":337,"level":247},{},{"id":340,"data":2524,"type":391,"tunes":2538},{"content":2525,"stretched":43,"withHeadings":14},[2526,2527,2528,2529,2530,2531,2532,2533,2534,2535,2536,2537],[344,345,346],[348,349,350],[352,353,354],[356,357,358],[360,361,362],[364,365,366],[368,369,370],[372,373,374],[376,377,378],[380,381,382],[384,385,386],[388,389,390],{},{"id":394,"data":2540,"type":42,"tunes":2541},{"text":396,"level":247},{},{"id":399,"data":2543,"type":431,"tunes":2558},{"rows":2544,"title":423,"layout":391,"columns":2555},[2545,2547,2549,2551,2553],{"id":403,"label":404,"values":2546},[406,406],{"id":408,"label":409,"values":2548},[406,406],{"id":412,"label":413,"values":2550},[406,406],{"id":416,"label":417,"values":2552},[406,406],{"id":420,"label":421,"values":2554},[406,406],[2556,2557],{"id":426,"label":427},{"id":429,"label":430},{},{"id":434,"data":2560,"type":218,"tunes":2561},{"text":436},{},{"id":439,"data":2563,"type":42,"tunes":2564},{"text":441,"level":247},{},{"id":444,"data":2566,"type":218,"tunes":2567},{"text":446},{},{"id":449,"data":2569,"type":218,"tunes":2570},{"text":451},{},{"id":454,"data":2572,"type":218,"tunes":2573},{"text":456},{},{"id":459,"data":2575,"type":226,"tunes":2576},{"body":461,"title":462,"variant":463},{},{"id":466,"data":2578,"type":42,"tunes":2579},{"text":468,"level":247},{},{"id":471,"data":2581,"type":218,"tunes":2582},{"text":473},{},{"id":476,"data":2584,"type":218,"tunes":2585},{"text":478},{},{"id":481,"data":2587,"type":218,"tunes":2588},{"text":483},{},{"id":486,"data":2590,"type":492,"tunes":2591},{"url":488,"title":489,"excerpt":490,"ctaLabel":491},{},{"id":495,"data":2593,"type":42,"tunes":2594},{"text":497,"level":247},{},{"id":500,"data":2596,"type":218,"tunes":2597},{"text":502},{},{"id":505,"data":2599,"type":218,"tunes":2600},{"text":507},{},{"id":510,"data":2602,"type":226,"tunes":2603},{"body":512,"title":513,"variant":233},{},{"id":516,"data":2605,"type":42,"tunes":2606},{"text":518,"level":247},{},{"id":521,"data":2608,"type":218,"tunes":2609},{"text":523},{},{"id":526,"data":2611,"type":218,"tunes":2612},{"text":528},{},{"id":531,"data":2614,"type":218,"tunes":2615},{"text":533},{},{"id":536,"data":2617,"type":42,"tunes":2618},{"text":538,"level":247},{},{"id":541,"data":2620,"type":431,"tunes":2633},{"rows":2621,"title":560,"layout":391,"columns":2630},[2622,2624,2626,2628],{"id":545,"label":546,"values":2623},[406,406],{"id":549,"label":550,"values":2625},[406,406],{"id":553,"label":554,"values":2627},[406,406],{"id":557,"label":558,"values":2629},[406,406],[2631,2632],{"id":563,"label":564},{"id":566,"label":567},{},{"id":570,"data":2635,"type":218,"tunes":2636},{"text":572},{},{"id":575,"data":2638,"type":42,"tunes":2639},{"text":577,"level":247},{},{"id":580,"data":2641,"type":218,"tunes":2642},{"text":582},{},{"id":585,"data":2644,"type":218,"tunes":2645},{"text":587},{},{"id":590,"data":2647,"type":218,"tunes":2648},{"text":592},{},{"id":595,"data":2650,"type":42,"tunes":2651},{"text":597,"level":247},{},{"id":600,"data":2653,"type":218,"tunes":2654},{"text":602},{},{"id":605,"data":2656,"type":218,"tunes":2657},{"text":607},{},{"id":610,"data":2659,"type":218,"tunes":2660},{"text":612},{},{"id":615,"data":2662,"type":42,"tunes":2663},{"text":617,"level":247},{},{"id":620,"data":2665,"type":218,"tunes":2666},{"text":622},{},{"id":625,"data":2668,"type":218,"tunes":2669},{"text":627},{},{"id":630,"data":2671,"type":218,"tunes":2672},{"text":632},{},{"id":635,"data":2674,"type":42,"tunes":2675},{"text":637,"level":247},{},{"id":640,"data":2677,"type":218,"tunes":2678},{"text":642},{},{"id":645,"data":2680,"type":218,"tunes":2681},{"text":647},{},{"id":650,"data":2683,"type":391,"tunes":2692},{"content":2684,"stretched":43,"withHeadings":14},[2685,2686,2687,2688,2689,2690,2691],[654,655],[657,658],[660,661],[663,664],[666,667],[669,670],[672,673],{},{"id":676,"data":2694,"type":42,"tunes":2695},{"text":678,"level":247},{},{"id":681,"data":2697,"type":218,"tunes":2698},{"text":683},{},{"id":686,"data":2700,"type":218,"tunes":2701},{"text":688},{},{"id":691,"data":2703,"type":218,"tunes":2704},{"text":693},{},{"id":696,"data":2706,"type":226,"tunes":2707},{"body":698,"title":699,"variant":233},{},{"id":702,"data":2709,"type":42,"tunes":2710},{"text":704,"level":247},{},{"id":707,"data":2712,"type":218,"tunes":2713},{"text":709},{},{"id":712,"data":2715,"type":218,"tunes":2716},{"text":714},{},{"id":717,"data":2718,"type":492,"tunes":2719},{"url":719,"title":720,"excerpt":721,"ctaLabel":722},{},{"id":725,"data":2721,"type":42,"tunes":2722},{"text":727,"level":247},{},{"id":730,"data":2724,"type":218,"tunes":2725},{"text":732},{},{"id":735,"data":2727,"type":218,"tunes":2728},{"text":737},{},{"id":740,"data":2730,"type":218,"tunes":2731},{"text":742},{},{"id":745,"data":2733,"type":42,"tunes":2734},{"text":747,"level":247},{},{"id":750,"data":2736,"type":226,"tunes":2737},{"body":752,"title":753,"variant":240},{},{"id":756,"data":2739,"type":391,"tunes":2750},{"content":2740,"stretched":43,"withHeadings":14},[2741,2742,2743,2744,2745,2746,2747,2748,2749],[760,761],[763,764],[766,767],[769,770],[772,773],[775,776],[778,779],[781,782],[784,785],{},{"id":788,"data":2752,"type":218,"tunes":2753},{"text":790},{},{"id":793,"data":2755,"type":42,"tunes":2756},{"text":795,"level":247},{},{"id":798,"data":2758,"type":391,"tunes":2773},{"content":2759,"stretched":43,"withHeadings":14},[2760,2761,2762,2763,2764,2765,2766,2767,2768,2769,2770,2771,2772],[802,803],[805,806],[808,809],[811,812],[814,815],[817,818],[820,821],[823,824],[826,827],[829,830],[832,833],[835,836],[838,839],{},{"id":842,"data":2775,"type":42,"tunes":2776},{"text":844,"level":247},{},{"id":847,"data":2778,"type":391,"tunes":2790},{"content":2779,"stretched":43,"withHeadings":14},[2780,2781,2782,2783,2784,2785,2786,2787,2788,2789],[851,852,853],[855,856,857],[859,860,861],[863,864,865],[867,868,869],[871,872,873],[875,876,877],[879,880,881],[883,884,885],[887,888,889],{},{"id":892,"data":2792,"type":42,"tunes":2793},{"text":894,"level":247},{},{"id":897,"data":2795,"type":218,"tunes":2796},{"text":899},{},{"id":902,"data":2798,"type":218,"tunes":2799},{"text":904},{},{"id":907,"data":2801,"type":492,"tunes":2802},{"url":909,"title":910,"excerpt":911,"ctaLabel":912},{},{"id":915,"data":2804,"type":42,"tunes":2805},{"text":917,"level":247},{},{"id":920,"data":2807,"type":42,"tunes":2808},{"text":922,"level":246},{},{"id":925,"data":2810,"type":218,"tunes":2811},{"text":927},{},{"id":930,"data":2813,"type":218,"tunes":2814},{"text":932},{},{"id":935,"data":2816,"type":218,"tunes":2817},{"text":937},{},{"id":940,"data":2819,"type":42,"tunes":2820},{"text":942,"level":246},{},{"id":945,"data":2822,"type":218,"tunes":2823},{"text":947},{},{"id":950,"data":2825,"type":218,"tunes":2826},{"text":952},{},{"id":955,"data":2828,"type":218,"tunes":2829},{"text":957},{},{"id":960,"data":2831,"type":391,"tunes":2839},{"content":2832,"stretched":43,"withHeadings":14},[2833,2834,2835,2836,2837,2838],[964,965],[967,968],[970,971],[973,974],[976,977],[979,980],{},{"id":983,"data":2841,"type":226,"tunes":2842},{"body":985,"title":986,"variant":240},{},{"id":989,"data":2844,"type":42,"tunes":2845},{"text":991,"level":247},{},{"id":994,"data":2847,"type":391,"tunes":2860},{"content":2848,"stretched":43,"withHeadings":14},[2849,2850,2851,2852,2853,2854,2855,2856,2857,2858,2859],[998,999],[1001,1002],[1004,1005],[1007,1008],[1010,1011],[1013,1014],[1016,1017],[1019,1020],[1022,1023],[1025,1026],[1028,1029],{},{"id":1032,"data":2862,"type":42,"tunes":2863},{"text":1034,"level":247},{},{"id":1037,"data":2865,"type":391,"tunes":2878},{"content":2866,"stretched":43,"withHeadings":14},[2867,2868,2869,2870,2871,2872,2873,2874,2875,2876,2877],[1041,1042],[1044,1045],[1047,1048],[1050,1051],[1053,1054],[1056,1057],[1059,1060],[1062,1063],[1065,1066],[1068,1069],[1071,1072],{},{"id":1075,"data":2880,"type":42,"tunes":2881},{"text":1077,"level":247},{},{"id":1080,"data":2883,"type":317,"tunes":2895},{"steps":2884,"title":1113,"orientation":316},[2885,2886,2887,2888,2889,2890,2891,2892,2893,2894],{"label":1084,"description":1085},{"label":1087,"description":1088},{"label":1090,"description":1091},{"label":1093,"description":1094},{"label":1096,"description":1097},{"label":1099,"description":1100},{"label":1102,"description":1103},{"label":1105,"description":1106},{"label":1108,"description":1109},{"label":1111,"description":1112},{},{"id":1116,"data":2897,"type":42,"tunes":2898},{"text":1118,"level":247},{},{"id":1121,"data":2900,"type":391,"tunes":2915},{"content":2901,"stretched":43,"withHeadings":14},[2902,2903,2904,2905,2906,2907,2908,2909,2910,2911,2912,2913,2914],[852,1125],[1127,1128],[1130,1131],[1133,1134],[1136,1137],[1139,1140],[1142,1143],[1145,1146],[1148,1149],[1151,1152],[1154,1155],[1157,1158],[1160,1161],{},{"id":1164,"data":2917,"type":42,"tunes":2918},{"text":1166,"level":247},{},{"id":1169,"data":2920,"type":218,"tunes":2921},{"text":1171},{},{"id":1174,"data":2923,"type":218,"tunes":2924},{"text":1176},{},{"id":1179,"data":2926,"type":218,"tunes":2927},{"text":1181},{},{"id":1184,"data":2929,"type":218,"tunes":2930},{"text":1186},{},{"id":1189,"data":2932,"type":218,"tunes":2933},{"text":1191},{},{"id":1194,"data":2935,"type":42,"tunes":2936},{"text":1196,"level":247},{},{"id":1199,"data":2938,"type":218,"tunes":2939},{"text":1201},{},{"id":1204,"data":2941,"type":218,"tunes":2942},{"text":1206},{},{"id":1209,"data":2944,"type":218,"tunes":2945},{"text":1211},{},{"id":1214,"data":2947,"type":42,"tunes":2948},{"text":1216,"level":247},{},{"id":1219,"data":2950,"type":218,"tunes":2951},{"text":1221},{},{"id":1224,"data":2953,"type":492,"tunes":2954},{"url":1226,"title":1227,"excerpt":1228,"ctaLabel":1229},{},{"id":1232,"data":2956,"type":218,"tunes":2957},{"text":1234},{},{"id":1237,"data":2959,"type":218,"tunes":2960},{"text":1239},{},{"id":1242,"data":2962,"type":42,"tunes":2963},{"text":1244,"level":247},{},{"id":1247,"data":2965,"type":1247,"tunes":2975},{"items":2966,"title":1282},[2967,2968,2969,2970,2971,2972,2973,2974],{"id":1251,"answer":1252,"question":1253},{"id":1255,"answer":1256,"question":1257},{"id":1259,"answer":1260,"question":1261},{"id":1263,"answer":1264,"question":1265},{"id":1267,"answer":1268,"question":1269},{"id":1271,"answer":1272,"question":1273},{"id":1275,"answer":1276,"question":1277},{"id":1279,"answer":1280,"question":1281},{},{"id":1285,"data":2977,"type":42,"tunes":2978},{"text":1287,"level":247},{},{"id":1290,"data":2980,"type":1290,"tunes":2994},{"title":1292,"entries":2981},[2982,2983,2984,2985,2986,2987,2988,2989,2990,2991,2992,2993],{"term":430,"anchor":1295,"definition":1296},{"term":1298,"anchor":1299,"definition":1300},{"term":427,"anchor":1302,"definition":1303},{"term":1305,"anchor":1306,"definition":1307},{"term":1309,"anchor":1310,"definition":1311},{"term":1313,"anchor":1314,"definition":1315},{"term":1317,"anchor":1318,"definition":1319},{"term":1321,"anchor":1322,"definition":1323},{"term":376,"anchor":1325,"definition":1326},{"term":1328,"anchor":1329,"definition":1330},{"term":871,"anchor":1332,"definition":1333},{"term":1335,"anchor":1336,"definition":1337},{},{"id":1340,"data":2996,"type":42,"tunes":2997},{"text":1342,"level":247},{},{"id":1345,"data":2999,"type":218,"tunes":3000},{"text":1347},{},{"id":1350,"data":3002,"type":218,"tunes":3003},{"text":1352},{},{"id":1355,"data":3005,"type":218,"tunes":3006},{"text":1357},{},{"id":1360,"data":3008,"type":42,"tunes":3009},{"text":1362,"level":247},{},{"id":1365,"data":3011,"type":218,"tunes":3012},{"text":1367},{},{"id":1370,"data":3014,"type":1377,"tunes":3017},{"link":1372,"meta":3015},{"image":3016,"title":1375,"description":1376},{"url":406},{},{"id":1380,"data":3019,"type":1377,"tunes":3022},{"link":1382,"meta":3020},{"image":3021,"title":1385,"description":1386},{"url":406},{},{"id":1389,"data":3024,"type":1377,"tunes":3027},{"link":1391,"meta":3025},{"image":3026,"title":1394,"description":1395},{"url":406},{},{"id":1398,"data":3029,"type":1377,"tunes":3032},{"link":1400,"meta":3030},{"image":3031,"title":1403,"description":1404},{"url":406},{},"Post erfolgreich abgerufen",{"items":3035,"source":3117,"manualIds":3118,"manualMatchedIds":3119},[3036,3043,3048,3055,3062,3069,3076,3083,3089,3096,3103,3110],{"id":3037,"slug":3038,"title":3039,"excerpt":3040,"featuredImage":3041,"publishedAt":3042},"363","front-und-backend-entwicklung","前端与后端开发","前端和后端开发是网络开发的重要组成部分，涉及创建网络应用程序和网站。前端开发专注于用户界面，而后端开发则负责编程和管理服务器端。","\u002Fuploads\u002F2026\u002F03\u002Ffront-und-backend-entwicklung-1774872219531-wyu4i1.webp","2023-04-12T11:11:00.000Z",{"id":3044,"slug":3045,"title":3045,"excerpt":10,"featuredImage":3046,"publishedAt":3047},"369","git-with-automatic-upload-and-synchronization-to-a-production-server","\u002Fuploads\u002F2024\u002F05\u002Fstep-by-step-guide-illustration-showing-the-process-of-setting-up-Git-with-auto-upload-and-synchronization-to-a-production-server-large.webp","2024-05-28T22:48:00.000Z",{"id":3049,"slug":3050,"title":3051,"excerpt":3052,"featuredImage":3053,"publishedAt":3054},"364","tipps-fuer-die-verbesserung-der-seo-suchmaschinenoptimierung","Mastering the SEO Workflow: Essential Optimization Strategies for Organic Growth","A structured SEO workflow is crucial for sustainable organic growth. Learn the ten foundational strategies, from keyword research and technical optimization to content quality and performance analysis.","\u002Fuploads\u002F2026\u002F03\u002Ftipps-fuer-die-verbesserung-der-seo-suchmaschinenoptimierung-1774866098131-hwkzrg.webp","2024-01-26T06:35:00.000Z",{"id":3056,"slug":3057,"title":3058,"excerpt":3059,"featuredImage":3060,"publishedAt":3061},"487","vector-databases-embeddings-and-reranking-three-different-parts-of-retrieval","向量数据库、嵌入和重排序：检索的三个不同部分","嵌入表示含义，向量数据库检索候选结果，重排序器则精炼结果。了解这三个检索层在RAG中如何不同并协同工作。","\u002Fuploads\u002F2026\u002F10\u002Fvector-databases-embeddings-and-reranking-three-different-parts-of-retrieval-1791480129884-9dtasz.webp","2026-10-08T11:21:00.000Z",{"id":3063,"slug":3064,"title":3065,"excerpt":3066,"featuredImage":3067,"publishedAt":3068},"466","the-gpu-is-not-the-product-future-proof-private-ai-architecture","GPU 不是产品：面向未来的私有 AI 架构","私有 AI 基础设施不应围绕单一 GPU 或单一模型来设计。更具韧性的做法是将快速推理 GPU、内存充裕的 AI 系统、物理 AI 节点以及可选的前沿云模型，统一置于一个具备能力感知的路由层之后。","\u002Fuploads\u002F2026\u002F09\u002Fthe-gpu-is-not-the-product-future-proof-private-ai-architecture-1790140878812-8hsl39.webp","2026-09-23T01:19:00.000Z",{"id":3070,"slug":3071,"title":3072,"excerpt":3073,"featuredImage":3074,"publishedAt":3075},"384","new-qwen-3-5-plus","全新Qwen 3.5-Plus：开源AI迈入新纪元","探索阿里巴巴Qwen 3.5-Plus的革命性特性与优势，这款为开发者打造的颠覆性开源人工智能模型。","\u002Fuploads\u002F2026\u002F02\u002Fnew-qwen-3-5-plus-1771515512741-dcbi9p.webp","2026-02-19T10:23:00.000Z",{"id":3077,"slug":3078,"title":3079,"excerpt":3080,"featuredImage":3081,"publishedAt":3082},"467","the-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers","答案有效性边界：相关性到可靠AI答案之间缺失的层级","一个来源可能相关、权威，但对于所提出的问题仍然是错误的。缺失的层次是适用性：答案成立的条件，以及迫使其被重新考虑的变化。本文介绍了“答案有效性边界”这一面向人类、AI搜索和RAG系统的来源设计模式。","\u002Fuploads\u002F2026\u002F09\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers-1790272901306-1g5jly.webp","2026-09-24T11:59:00.000Z",{"id":3084,"slug":3085,"title":3086,"excerpt":3087,"featuredImage":3074,"publishedAt":3088},"445","qwen-3-6-in-production-release-runbook-ai-rollback-and-llmops-versioning","Qwen 3.6 生产环境部署：发布手册、AI 回滚与 LLMOps 版本管理","Qwen 3.6 不仅仅是一次模型升级。它同时是一个发布事件、一个回滚场景和一个版本管理问题。本文通过LLMOps规范、提示词与模型可追溯性、受控发布以及基于证据的回滚准备，阐述了在生产环境中应如何处理Qwen 3.6。","2026-05-04T02:49:00.000Z",{"id":3090,"slug":3091,"title":3092,"excerpt":3093,"featuredImage":3094,"publishedAt":3095},"493","mlops-vs-llmops-what-changes-when-the-model-is-an-llm","MLOps 与 LLMOps：当模型是 LLM 时，会发生哪些变化","MLOps 运维机器学习系统；LLMOps 将这些实践扩展到围绕大型语言模型的提示、上下文、检索、提供商、工具、评估和运行时行为。","\u002Fuploads\u002F2026\u002F10\u002Fmlops-vs-llmops-what-changes-when-the-model-is-an-llm-1791487319869-2v7hxo.webp","2026-10-08T15:20:00.000Z",{"id":3097,"slug":3098,"title":3099,"excerpt":3100,"featuredImage":3101,"publishedAt":3102},"472","why-more-context-can-make-ai-answers-worse","为什么更多上下文会让AI的回答更糟","更大的上下文窗口并不保证更好的答案。本文解释了信号稀释、证据冲突、状态过时、位置敏感性和有损压缩如何降低AI可靠性——并介绍了一种实用的上下文压力测试。","\u002Fuploads\u002F2026\u002F09\u002Fwhy-more-context-can-make-ai-answers-worse-1790351615793-2ntv2v.webp","2026-09-25T11:51:00.000Z",{"id":3104,"slug":3105,"title":3106,"excerpt":3107,"featuredImage":3108,"publishedAt":3109},"468","ai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context","AI代理记忆不是RAG：如何区分记忆、检索、状态和上下文","代理记忆、RAG、状态和上下文经常被当作可以互换的概念来使用。它们并不是。这个实用的架构模型将这四个层次区分开来，展示了每一层各自应处的位置，并解释了当系统将它们合并为一层时会出现什么问题。","\u002Fuploads\u002F2026\u002F09\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context-1790350560308-np0xy6.webp","2026-09-25T11:34:00.000Z",{"id":3111,"slug":3112,"title":3113,"excerpt":3114,"featuredImage":3115,"publishedAt":3116},"490","rbac-vs-tenant-isolation-two-different-security-boundaries","RBAC与租户隔离：两种不同的安全边界","RBAC 控制用户可以做什么；租户隔离控制该操作可以触及哪个租户的资源。了解为什么多租户 SaaS 安全需要这两道边界。","\u002Fuploads\u002F2026\u002F10\u002Frbac-vs-tenant-isolation-two-different-security-boundaries-1791485111528-qqtzby.webp","2026-10-08T14:43:00.000Z","fallback",[],[]]