[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"portal-settings:stajic:zh":3,"public-menus:all":38,"post:why-more-context-can-make-ai-answers-worse:zh":205,"related:post:why-more-context-can-make-ai-answers-worse:zh:1":1598},{"statusCode":4,"data":5,"message":37},200,{"tenantId":6,"lang":7,"defaultLang":8,"siteUrl":9,"contactEmail":10,"brandName":11,"logoUrl":12,"siteName":11,"siteDescription":13,"ogImage":10,"robotsIndex":14,"socialLinks":10,"reservedSlugs":10,"seoPolicy":15},"stajic","zh","de","https:\u002F\u002Fstajic.de",null,"Stajic Platform","\u002FLogo_Planet.svg","Stajic Portal",true,{"branding":16,"relatedContent":17,"crossDomainLinks":18},{"logoUrl":12},{"enabled":14},[19,22,25,28,31,34],{"url":20,"label":21,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Ffigure.rocks","figure.rocks",{"url":23,"label":24,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Floving.rocks","loving.rocks",{"url":26,"label":27,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.com","bazify.com",{"url":29,"label":30,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.de","bazify.de",{"url":32,"label":33,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.at","bazify.at",{"url":35,"label":36,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.ba","bazify.ba","Portal settings resolved",[39,45],{"id":40,"name":41,"location":42,"isActive":14,"isDefault":43,"items":44},1,"main-navigation","header",false,[],{"id":46,"name":47,"location":48,"isActive":14,"isDefault":14,"items":49},4,"main-menu","sidebar",[50,66,79,93,103,118,133],{"id":51,"title":52,"url":60,"target":61,"icon":62,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":64,"portfolioId":10,"children":65},"item-18",{"de":53,"en":54,"es":55,"fr":56,"it":54,"ru":57,"sr":58,"zh":59},"Startseite","Home","Inicio","Accueil","Главная","Почетна","首页","\u002Ffull-stack-web-developer-munich-performance-seo-and-maintainable-builds","_self","i-lucide-home","page",111,[],{"id":67,"title":68,"url":75,"target":61,"icon":76,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":77,"portfolioId":10,"children":78},"item-22",{"de":69,"en":69,"es":70,"fr":69,"it":71,"ru":72,"sr":73,"zh":74},"Vision","Visión","Visione","Видение","Визија","想象","\u002Fueber-uns-webdesign-muenchen-webaplikation","i-lucide-eye",113,[],{"id":80,"title":81,"url":89,"target":61,"icon":90,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":91,"portfolioId":10,"children":92},"item-19",{"de":82,"en":83,"es":84,"fr":83,"it":85,"ru":86,"sr":87,"zh":88},"Leistungen","Services","Servicios","Servizi","Услуги","Услуге","服务","\u002Fservices-dienstleistungen-muenchen","i-lucide-wrench",116,[],{"id":94,"title":95,"url":99,"target":61,"icon":100,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":101,"portfolioId":10,"children":102},"item-23",{"de":96,"en":96,"es":96,"fr":96,"it":96,"ru":97,"sr":97,"zh":98},"Blog","Блог","博客","\u002Fblog","i-lucide-book-open",112,[],{"id":104,"title":105,"url":114,"target":61,"icon":115,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":116,"portfolioId":10,"children":117},"item-32",{"de":106,"en":107,"es":108,"fr":109,"it":110,"ru":111,"sr":112,"zh":113},"Neue Technologien","New Technologies","Nuevas tecnologías","Nouvelles technologies","Nuove tecnologie","Новые технологии","Нове технологије","新技术！","\u002Fneue-webtechnologien","i-lucide-sparkles",122,[],{"id":119,"title":120,"url":129,"target":61,"icon":130,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":131,"portfolioId":10,"children":132},"item-20",{"de":121,"en":122,"es":123,"fr":124,"it":125,"ru":126,"sr":127,"zh":128},"Kontakt","Contact us!","Contacto","Contact","Contatto","Контакт","Контактирајте нас","联系我们！","\u002Fcontact","i-lucide-mail",115,[],{"id":134,"title":135,"url":144,"target":61,"icon":145,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":147},"item-21",{"de":136,"en":137,"es":138,"fr":139,"it":140,"ru":141,"sr":142,"zh":143},"Unsere Arbeit","Our Work","Nuestro trabajo","Nos réalisations","I nostri lavori","Наши работы","Наши радови","文件夹","\u002Fportfolio","i-lucide-briefcase",114,[148,161,175,181,193],{"id":149,"title":150,"url":144,"target":61,"icon":159,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":160},"item-24",{"de":151,"en":152,"es":153,"fr":154,"it":155,"ru":156,"sr":157,"zh":158},"Alle Projekte","All Projects","Todos los proyectos","Tous les projets","Tutti i progetti","Все проекты","Сви пројекти","所有项目","i-lucide-grid-3x3",[],{"id":162,"title":163,"url":171,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":174},"item-29",{"de":164,"en":165,"es":166,"fr":167,"it":168,"ru":169,"sr":170,"zh":143},"Local Roots, Global Reach","Local Roots - Global Reach","Empresa local ","Entreprise locale","Azienda locale","Местная компания","Локално предузеће глобално тржиште","\u002Fportfolio\u002Flocal-roots-global-reach-communication-media-systems-for-modern-business","i-lucide-folder","custom",[],{"id":176,"title":177,"url":179,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":180},"item-28",{"de":178,"en":178,"es":178,"fr":178,"it":178,"ru":178,"sr":178,"zh":178},"Solr Suggester","\u002Fportfolio\u002Fsolr-fuzzy-suggester-und-solr-infix-suggester-abfrage-ueber-ajax-und-filterung",[],{"id":182,"title":183,"url":191,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":192},"item-27",{"de":184,"en":185,"es":186,"fr":187,"it":188,"ru":189,"sr":190,"zh":185},"Firmenwebseite SEO","Company Website SEO","Sitio web corporativo SEO","Site web d’entreprise SEO","Sito web aziendale SEO","Корпоративный сайт SEO","Пословна веб-страница SEO","\u002Fportfolio\u002Fseo-sem-branding-mobile-webseite-muenchen",[],{"id":194,"title":195,"url":203,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":204},"item-31",{"de":196,"en":197,"es":198,"fr":199,"it":200,"ru":201,"sr":202,"zh":197},"Digitalisierungsportal","Digitalization Portal","Portal de digitalización","Portail de numérisation","Portale di digitalizzazione","Портал цифровизации","Портал за дигитализацију","\u002Fportfolio\u002Fdigitalisierungsportal-archiv-museum-bibliothek-ead-lido-mets-mods",[],{"statusCode":4,"data":206,"message":1597},{"id":207,"title":208,"slug":209,"content":210,"contentJson":211,"excerpt":797,"featuredImage":798,"featuredImageAlt":799,"featuredImageCaption":10,"featuredImageTitle":10,"featuredImageCopyright":10,"featuredImageAuthor":10,"featuredImageSourceUrl":10,"featuredImageLicense":10,"featuredImageIsAiGenerated":43,"status":800,"publishedAt":801,"createdAt":802,"updatedAt":803,"seoLocalePaths":804,"categories":813,"author":826,"translations":831},"472","为什么更多上下文会让AI的回答更糟","why-more-context-can-make-ai-answers-worse","\u003Cnav class=\"editorjs-toc\" data-editorjs-toc=\"true\" aria-label=\"目录\">\u003Cstrong class=\"editorjs-toc__title\">目录\u003C\u002Fstrong>\u003Col class=\"editorjs-toc__list editorjs-toc__list--depth-0\">\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-5\" class=\"editorjs-toc__link\">上下文容量不等于上下文可用性\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-9\" class=\"editorjs-toc__link\">额外上下文降低答案质量的五种方式\u003C\u002Fa>\u003Col class=\"editorjs-toc__list editorjs-toc__list--depth-1\">\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-11\" class=\"editorjs-toc__link\">1. 信号稀释：相关证据与其他所有内容竞争\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-15\" class=\"editorjs-toc__link\">2. 证据冲突：更多来源可能意味着更多版本的现实\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-19\" class=\"editorjs-toc__link\">3. 位置敏感性：证据出现的位置会改变结果\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-22\" class=\"editorjs-toc__link\">4. 陈旧上下文持续存在：模型同时看到真相和历史\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-25\" class=\"editorjs-toc__link\">5. 压缩损失：更小的上下文也可能变成更差的上下文\u003C\u002Fa>\u003C\u002Fli>\u003C\u002Fol>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-28\" class=\"editorjs-toc__link\">上下文质量模型\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-32\" class=\"editorjs-toc__link\">上下文压力测试\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-35\" class=\"editorjs-toc__link\">应衡量什么而不是令牌数量\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-37\" class=\"editorjs-toc__link\">RAG：为什么增加 top-k 可能适得其反\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-41\" class=\"editorjs-toc__link\">长时间运行的智能体：连续性不等于累积\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-45\" class=\"editorjs-toc__link\">上下文顺序应当是有意设计的\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-48\" class=\"editorjs-toc__link\">在压缩过程中保留决策边界\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-52\" class=\"editorjs-toc__link\">一种实用的上下文构建策略\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-54\" class=\"editorjs-toc__link\">什么会改变这个答案？\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-58\" class=\"editorjs-toc__link\">局限性\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-61\" class=\"editorjs-toc__link\">结论\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-64\" class=\"editorjs-toc__link\">常见问题\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-66\" class=\"editorjs-toc__link\">术语表\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-68\" class=\"editorjs-toc__link\">主要来源与延伸阅读\u003C\u002Fa>\u003C\u002Fli>\u003C\u002Fol>\u003C\u002Fnav>\n\u003Cp>更大的上下文窗口为 AI 系统提供了更多容量。它并不保证模型会很好地利用这些容量。在长对话、RAG 流水线、研究智能体和工具密集型工作流中，添加更多历史记录、更多文档、更多工具输出或更多记忆，可能会让回答变得更不可靠，而不是更有依据。\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--info my-6 rounded-xl border p-5 border-blue-300 bg-blue-50 dark:border-blue-900 dark:bg-blue-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">直接回答\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">&lt;strong&gt;当额外信息降低了信噪比、引入冲突、掩盖决定性证据、保留过时状态或压缩掉重要条件时，更多上下文反而会让 AI 回答变差。&lt;\u002Fstrong&gt; 因此，相关的工程目标不是最大化上下文，而是&lt;strong&gt;在保留证据和决策边界的前提下，使用最小充分上下文&lt;\u002Fstrong&gt;。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Caside class=\"editorjs-callout editorjs-callout--note my-6 rounded-xl border p-5 border-gray-300 bg-gray-50 dark:border-gray-700 dark:bg-gray-900\u002F40\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">关于本文所用模型\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">下文中的上下文质量模型和上下文压力测试是本文提出的实用架构方法，并非正式的行业标准。它们综合了关于长上下文位置效应、上下文污染、压缩、检索和上下文工程的既有研究成果。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-5\">上下文容量不等于上下文可用性\u003C\u002Fh2>\n\u003Cp>模型宣称的上下文窗口描述的是它能接受多少输入。这并不意味着该窗口内的每个 token 都会获得同等关注，或对最终答案有同等贡献。这一区别很重要，因为生产系统越来越多地用对话历史、检索文档、工具结果、记忆、结构化状态、指令和中间产物来填充上下文。\u003C\u002Fp>\n\u003Cp>经典的“迷失在中间”研究表明，当相关证据出现在长输入的中部时，长上下文模型的表现可能比证据出现在开头或结尾附近时更差。更广泛的工程启示并不是长上下文不好，而是上下文中的可用性并不等同于可靠使用。\u003C\u002Fp>\n\u003Cp>OpenAI 的上下文管理指南从另一个方向得出了相同的操作结论：即使是非常大的上下文窗口，也可能被未经筛选的历史记录、冗余的工具输出和嘈杂的检索结果淹没。Anthropic 同样将上下文视为一种有限资源，需要主动工程化，而不是被动累积。\u003C\u002Fp>\n\u003Ch2 id=\"section-9\">额外上下文降低答案质量的五种方式\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">失败模式\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">增加更多上下文时发生的变化\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">典型症状\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">信号稀释\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">相关证据在总输入中所占比例变小\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型给出泛泛的回答，或错过决定性段落\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">证据冲突\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">不同文档、版本或记忆之间相互矛盾\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">答案混合了不兼容的说法，或选择了错误版本\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">位置敏感性\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">决定性信息移动到上下文中使用可靠性较低的位置\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">同一证据在一种排序下有效，在另一种排序下失效\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">过时上下文残留\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">现实变化后，旧状态或先前结论仍然存在\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型不断重复曾经正确但现已过时的答案\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">压缩损失\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">压缩或摘要去掉了限定条件、例外、出处或未解决的不确定性\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">摘要连贯，但由此产生的答案变得过度自信或过度概括\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch3 id=\"section-11\">1. 信号稀释：相关证据与其他所有内容竞争\u003C\u002Fh3>\n\u003Cp>假设一个问题可以用两段短文回答。一个 RAG 系统“为了保险”检索了这两段，再加上十八段关系不大的内容。检索召回率可能提高了，但生成器现在必须从背景材料中区分出决定性证据。如果相似措辞出现在多个文档中，额外上下文可能会让答案变得不够精确。\u003C\u002Fp>\n\u003Cp>这就在检索召回率和上下文效用之间形成了一个重要区别。更多检索材料可能提高答案存在于上下文中某处的概率，同时降低模型给予正确证据足够权重的概率。\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--tip my-6 rounded-xl border p-5 border-violet-300 bg-violet-50 dark:border-violet-900 dark:bg-violet-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">工程规则\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">不要孤立地优化 top-k。要衡量添加文档是否改善最终结论、保留证据归属，并能在反复试验中保持稳定。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch3 id=\"section-15\">2. 证据冲突：更多来源可能意味着更多版本的现实\u003C\u002Fh3>\n\u003Cp>长上下文常常包含相互不一致的信息：新旧 API 文档、两个政策版本、先前和当前用户偏好、相互竞争的网页来源、缓存状态，或不再与来源匹配的模型生成摘要。\u003C\u002Fp>\n\u003Cp>这种失败不一定是幻觉。模型可能是在忠实地组合相互矛盾的证据。因此，架构需要优先级规则：来源权威性、版本、时间戳、司法管辖区、租户、产品修订版、用户状态或显式取代元数据。\u003C\u002Fp>\n\u003Cp>如果没有这些规则，增加上下文可能会让矛盾增加得比知识更快。\u003C\u002Fp>\n\u003Ch3 id=\"section-19\">3. 位置敏感性：证据出现的位置会改变结果\u003C\u002Fh3>\n\u003Cp>“迷失在中间”的结果表明，仅改变相关信息的位置就可能实质性改变模型性能。这一发现对于按固定顺序拼接大量检索段落或长历史记录的系统尤为重要。\u003C\u002Fp>\n\u003Cp>因此，生产测试应改变文档顺序，而不仅仅是测试一个标准提示。如果系统只有在决定性证据位于最前或最后时才能正确回答，那么该应用比单一基准分数所显示的更为脆弱。\u003C\u002Fp>\n\u003Ch3 id=\"section-22\">4. 陈旧上下文持续存在：模型同时看到真相和历史\u003C\u002Fh3>\n\u003Cp>长时间运行的智能体经常将先前的结论延续下去。这种连续性在事实发生变化之前是有用的。如果昨天的工具结果显示某个部署是健康的，而当前的工具结果显示它已降级，除非系统明确替换或限定旧状态的范围，否则两者可能都保留在上下文中。\u003C\u002Fp>\n\u003Cp>这就是为什么当前操作状态通常应来自权威来源，而记忆则保留持久性上下文，如决策、偏好或流程。更多的对话历史并不能替代重新读取当前状态。\u003C\u002Fp>\n\u003Ch3 id=\"section-25\">5. 压缩损失：更小的上下文也可能变成更差的上下文\u003C\u002Fh3>\n\u003Cp>相反的操作——压缩上下文——也有失败模式。摘要可能会丢失例外情况、未解决的问题、来源、精确标识符、负面证据，或结论有效的条件。\u003C\u002Fp>\n\u003Cp>微软研究院的智能体上下文工程工作将相关问题描述为简洁性偏差和上下文崩溃：迭代重写可能会移除有用的领域细节。因此，目标不是“尽可能压缩”。而是在减少上下文的同时，保留那些会改变决策的信息。\u003C\u002Fp>\n\u003Ch2 id=\"section-28\">上下文质量模型\u003C\u002Fh2>\n\u003Cp>有用的上下文可以从六个维度进行评估。这些维度都不是简单的令牌数量。\u003C\u002Fp>\n\u003Csection class=\"editorjs-comparison my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">上下文质量的六个维度\u003C\u002Fh3>\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left dark:border-gray-700 dark:bg-gray-900\">\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">维度\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">问题\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">如果薄弱\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">相关性\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">权威性\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">新鲜度\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">一致性\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">决策完整性\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">可追溯性\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Caside class=\"editorjs-callout editorjs-callout--success my-6 rounded-xl border p-5 border-emerald-300 bg-emerald-50 dark:border-emerald-900 dark:bg-emerald-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">目标状态\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">最好的上下文不是最大的上下文。而是&lt;strong&gt;仍然保留可靠答案或行动所需的证据、约束、状态、例外情况和来源的最小上下文&lt;\u002Fstrong&gt;。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-32\">上下文压力测试\u003C\u002Fh2>\n\u003Cp>要确定应用是否能从更多上下文中受益，应将上下文大小作为实验变量进行测试，而不是假设越大越好。\u003C\u002Fp>\n\u003Csection class=\"editorjs-process my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">上下文压力测试\u003C\u002Fh3>\u003Cdiv class=\"grid grid-cols-1 md:grid-cols-2 xl:grid-cols-3 gap-4\">\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">1\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">1. 定义黄金案例\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">选择一个具有已知答案和已知最小证据集的任务。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">2\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">2. 运行最小充分上下文\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">仅提供答案所需的指令、当前状态和证据。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">3\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">3. 添加相关背景\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">添加有用但非决定性的上下文，并衡量质量是提高、保持稳定还是下降。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">4\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">4. 添加现实噪声\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">添加生产系统可能包含的松散相关历史、工具输出或检索段落。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">5\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">5. 添加受控冲突\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">引入带有明确版本元数据的陈旧或矛盾证据，并验证正确来源仍然胜出。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">6\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">6. 重新排序决定性证据\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">将关键信息放在开头、中间和结尾附近，以测试位置敏感性。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">7\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">7. 测试压缩\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">用摘要替换较旧的上下文，并验证限定词、来源、未解决问题和决策边界是否保留。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">8\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">8. 比较质量曲线\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">随着上下文变化，衡量正确性、证据使用、一致性、延迟、成本和方差。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-35\">应衡量什么而不是令牌数量\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">指标\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">它揭示了什么\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">答案正确性\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">最终结果是否正确\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">声明级证据支持\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">随着上下文变化，重要声明是否仍然有依据\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">证据利用率\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">答案是否遵循决定性证据，而不是先前的模型知识\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">冲突解决准确性\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">当前\u002F权威证据是否胜过陈旧或较弱的来源\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">位置鲁棒性\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">重新排序证据是否会改变正确性\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">压缩保留\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">摘要是否保留约束、例外、标识符、来源和未解决状态\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">多次试验的输出方差\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">额外的上下文是否使系统更不稳定\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">延迟和令牌成本\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">添加的信息是否产生足够质量以证明其运营成本合理\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-37\">RAG：为什么增加 top-k 可能适得其反\u003C\u002Fh2>\n\u003Cp>一种常见的 RAG 调优模式是：当系统漏掉答案时，就增加 top-k。这可以提高候选召回率，但也会增加无关上下文、重复证据、过时段落以及相互冲突的文档。\u003C\u002Fp>\n\u003Cp>更好的问题是：决定性证据究竟是缺失于检索阶段，还是仅仅在上下文组装后失去了影响力。如果正确的段落已经出现在候选集中，那么增加 top-k 可能是在解决错误的问题。\u003C\u002Fp>\n\u003Caside class=\"editorjs-referral my-6\">\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method\" class=\"flex flex-col sm:flex-row gap-4 rounded-xl border border-gray-200 dark:border-gray-700 p-4 transition hover:border-primary-500\">\u003Cdiv class=\"min-w-0 flex-1\">\u003Cstrong class=\"block text-lg text-gray-900 dark:text-gray-100\">RAG 失败了——但究竟是哪一层失败了？一种诊断方法\u003C\u002Fstrong>\u003Cp class=\"mt-2 text-sm text-gray-600 dark:text-gray-300\">一种逐层隔离的方法，用于定位来源覆盖、检索、排序、上下文组装、生成、证据归因和时效性方面的失败。\u003C\u002Fp>\u003Cspan class=\"mt-3 inline-flex text-sm font-medium text-primary-600 dark:text-primary-400\">阅读 RAG 诊断方法 →\u003C\u002Fspan>\u003C\u002Fdiv>\u003C\u002Fa>\u003C\u002Faside>\n\u003Ch2 id=\"section-41\">长时间运行的智能体：连续性不等于累积\u003C\u002Fh2>\n\u003Cp>智能体需要跨步骤的连续性，但连续性并不要求重放此前的每一个 token。OpenAI 展示了针对长时间运行会话上下文的裁剪和压缩方法。Anthropic 推荐压缩、结构化笔记以及其他技术，以在控制上下文污染的同时保留有用信息。\u003C\u002Fp>\n\u003Cp>一个强大的长时间运行架构通常会将持久记忆、当前状态、外部工件、检索以及面向模型的上下文分离开来。这样系统就能保留重要内容，而不必把每一个历史细节都塞进每一次推理中。\u003C\u002Fp>\n\u003Caside class=\"editorjs-referral my-6\">\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context\" class=\"flex flex-col sm:flex-row gap-4 rounded-xl border border-gray-200 dark:border-gray-700 p-4 transition hover:border-primary-500\">\u003Cdiv class=\"min-w-0 flex-1\">\u003Cstrong class=\"block text-lg text-gray-900 dark:text-gray-100\">AI 智能体记忆不是 RAG：如何分离记忆、检索、状态与上下文\u003C\u002Fstrong>\u003Cp class=\"mt-2 text-sm text-gray-600 dark:text-gray-300\">一种实用的四层架构，用于分离哪些内容持久存在、哪些内容当前具有权威性、哪些内容被检索，以及模型实际接收到什么。\u003C\u002Fp>\u003Cspan class=\"mt-3 inline-flex text-sm font-medium text-primary-600 dark:text-primary-400\">阅读记忆架构文章 →\u003C\u002Fspan>\u003C\u002Fdiv>\u003C\u002Fa>\u003C\u002Faside>\n\u003Ch2 id=\"section-45\">上下文顺序应当是有意设计的\u003C\u002Fh2>\n\u003Cp>上下文构建是一个信息架构问题。关键指令、当前状态、决定性证据以及任务特定约束不应被随意放置。当系统机械地拼接来源时，它们实际上是把优先级隐式交给了位置效应和模型注意力。\u003C\u002Fp>\n\u003Cp>并不存在适用于所有模型和任务的通用最佳顺序，因此顺序应通过实证评估。一个有用的测试套件会随机化或系统地改变文档位置，并衡量同一主张是否保持稳定。\u003C\u002Fp>\n\u003Ch2 id=\"section-48\">在压缩过程中保留决策边界\u003C\u002Fh2>\n\u003Cp>一个只说“采用方案 X”的摘要，弱于一个保留了为什么选择 X 以及什么会使该决策失效的摘要。上下文压缩应保留那些可能改变答案的变量：版本、日期、假设、状态、权威性、未解决的分歧以及证据来源。\u003C\u002Fp>\n\u003Cp>这将上下文工程直接与答案有效性联系起来。如果压缩保留了一个结论，却移除了它的有效性边界，那么未来的回答可能仍然内部一致，却在外部变得错误。\u003C\u002Fp>\n\u003Caside class=\"editorjs-referral my-6\">\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers\" class=\"flex flex-col sm:flex-row gap-4 rounded-xl border border-gray-200 dark:border-gray-700 p-4 transition hover:border-primary-500\">\u003Cdiv class=\"min-w-0 flex-1\">\u003Cstrong class=\"block text-lg text-gray-900 dark:text-gray-100\">答案有效性边界：相关性与可靠 AI 答案之间缺失的一层\u003C\u002Fstrong>\u003Cp class=\"mt-2 text-sm text-gray-600 dark:text-gray-300\">一个框架，用于明确 AI 主张在什么条件下适用，以及哪些变化需要限制、重新计算或放弃。\u003C\u002Fp>\u003Cspan class=\"mt-3 inline-flex text-sm font-medium text-primary-600 dark:text-primary-400\">阅读答案有效性边界 →\u003C\u002Fspan>\u003C\u002Fdiv>\u003C\u002Fa>\u003C\u002Faside>\n\u003Ch2 id=\"section-52\">一种实用的上下文构建策略\u003C\u002Fh2>\n\u003Cul>\u003Cli>从当前任务出发，而不是从系统所知的一切出发。\u003C\u002Fli>\u003Cli>在做出重大决策之前，从权威系统重新读取易变状态。\u003C\u002Fli>\u003Cli>针对当前问题检索证据，而不是把庞大的静态语料库一直带下去。\u003C\u002Fli>\u003Cli>移除重复或低价值的工具输出。\u003C\u002Fli>\u003Cli>为重要证据保留来源版本、时间戳、权威性和出处。\u003C\u002Fli>\u003Cli>当当前信息与历史信息冲突时，明确优先级。\u003C\u002Fli>\u003Cli>将规则与其例外和前提条件一起保留。\u003C\u002Fli>\u003Cli>当持久决策和可复用流程不需要逐字重放时，将其存储在即时上下文之外。\u003C\u002Fli>\u003Cli>只有在测试了约束、标识符、例外和出处保留情况后，才压缩历史。\u003C\u002Fli>\u003Cli>通过重复试验而非单次提示来评估上下文大小、顺序和噪声。\u003C\u002Fli>\u003C\u002Ful>\n\u003Ch2 id=\"section-54\">什么会改变这个答案？\u003C\u002Fh2>\n\u003Cp>这种权衡会随模型架构、训练方式、任务类型和上下文长度而变化。未来的模型可能会对位置、噪声和冲突信息变得显著更加稳健。对于拥有小型干净语料库的任务，直接提供完整来源，而不是构建复杂的检索流水线，也可能同样有益。\u003C\u002Fp>\n\u003Cp>当遗漏比噪声更危险时，这一建议也会改变。在高召回率的研究或发现任务中，在后续的过滤或综合阶段之前，使用更大的候选上下文可能是合理的。在对延迟敏感的生产系统中，更严格的上下文选择可能更可取。\u003C\u002Fp>\n\u003Cp>只有当模型能够可靠地对无关信息、位置、矛盾和过时证据保持不变时，核心原则才会改变。在此之前，上下文应被视为一种经过精心策划的执行资源，而不是被动存储。\u003C\u002Fp>\n\u003Ch2 id=\"section-58\">局限性\u003C\u002Fh2>\n\u003Cp>长上下文行为在不同模型和工作负载之间差异很大。最初的“迷失在中间”实验使用的是早期几代模型，因此不应假设其确切的效应量能够代表当前系统。这一发现作为需要测试的失败模式仍然有用，但不应被视为普遍固定的性能曲线。\u003C\u002Fp>\n\u003Cp>同样，减少上下文可能会移除必要的证据。压缩会引入摘要风险，而激进的检索过滤可能会降低召回率。目标不是不惜一切代价追求最少 token，而是为正在做出的决策提供充分、最新、可追溯的上下文。\u003C\u002Fp>\n\u003Ch2 id=\"section-61\">结论\u003C\u002Fh2>\n\u003Cp>“模型能接受多少上下文？”这个问题，不如“这些上下文中有多少能改善决策？”更有用。更多 token 可以增加证据，但也可能增加干扰、矛盾、过时状态、位置脆弱性和压缩债务。\u003C\u002Fp>\n\u003Cp>将上下文视为经过工程化设计的工作集。从最小充分证据开始。只有当信息能改善可衡量的性能时才添加它。明确测试噪声、冲突、排序和压缩。大上下文窗口是一种容量；上下文质量则是一种架构。\u003C\u002Fp>\n\u003Ch2 id=\"section-64\">常见问题\u003C\u002Fh2>\n\u003Csection class=\"editorjs-faq my-6 rounded-xl border border-gray-200 p-5 dark:border-gray-700\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">长上下文与 AI 回答质量\u003C\u002Fh3>\u003Cdiv id=\"faq1\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">给 AI 模型更多上下文会让它的回答更差吗？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">会。额外的上下文可能稀释相关证据、引入矛盾或过时信息、将决定性证据移到不够稳健的位置，并增加模型使用弱信号而非决定性信号的可能性。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq2\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">更大的上下文窗口会消除对 RAG 的需求吗？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">通常不会。更大的上下文窗口增加了容量，但检索仍然有助于选择最新且相关的信息、控制成本、保留来源边界，并避免在每次请求中发送大量无关数据。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq3\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">什么是“迷失在中间”问题？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">它描述了观察到的情况：当相关信息位于长上下文中间时，语言模型对该信息的使用不如它出现在开头或结尾附近时可靠。确切效应因模型和任务而异，应在当前系统上进行测试。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq4\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">我应该总是降低 RAG 的 top-k 吗？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">不应该。如果候选集中缺少相关证据，更大的 top-k 可能会提高召回率。如果证据已经存在，但被额外材料稀释，那么增加 top-k 可能会使上下文更差。应分别诊断检索和上下文组装。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq5\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">上下文摘要应保留什么？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">保留持久决策、当前目标、未解决问题、标识符、约束、例外情况、证据来源，以及会改变先前结论的条件。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-66\">术语表\u003C\u002Fh2>\n\u003Csection class=\"editorjs-glossary my-6 rounded-xl border border-gray-200 dark:border-gray-700 p-5\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">关键上下文工程术语\u003C\u002Fh3>\u003Cdl>\u003Cdiv id=\"context-window\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">上下文窗口\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">模型在一次推理序列中能够关注到的输入和输出 token 信息量。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"context-pollution\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">上下文污染\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">由于无关、过时、冗余、冲突或其他低价值信息占用模型上下文而导致的性能下降。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"signal-dilution\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">信号稀释\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">随着额外低价值或竞争性信息被加入，决定性证据的相对突出程度下降。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"context-compaction\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">上下文压缩\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">通过摘要、重构、外置或以其他方式将必要信息保留在更小的工作表示中，从而减少累积的上下文。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"position-robustness\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">位置稳健性\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">当相关信息出现在上下文中的不同位置时，模型性能保持稳定的程度。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"minimum-sufficient-context\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">最小充分上下文\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">仍然保留可靠执行所需的证据、状态、约束、例外和来源的最小实用工作上下文。\u003C\u002Fdd>\u003C\u002Fdiv>\u003C\u002Fdl>\u003C\u002Fsection>\n\u003Ch2 id=\"section-68\">主要来源与延伸阅读\u003C\u002Fh2>\n\u003Ca href=\"https:\u002F\u002Fdevelopers.openai.com\u002Fcookbook\u002Fexamples\u002Fagents_sdk\u002Fsession_memory\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">OpenAI — 上下文工程：使用会话进行短期记忆管理\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">关于裁剪和压缩的指导，并讨论了干扰、低效、过时上下文、噪声检索和长时间运行的会话。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Feffective-context-engineering-for-ai-agents\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">Anthropic — 面向 AI 智能体的有效上下文工程\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">关于上下文污染、压缩、结构化笔记和长时程智能体上下文管理的工程指导。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Faclanthology.org\u002F2024.tacl-1.9\u002F\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">Liu 等 — 迷失在中间：语言模型如何使用长上下文\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">TACL 论文展示了长上下文中相关信息使用对位置的敏感性，并推动了对长上下文稳健性的明确测试。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Fpublication\u002Fagentic-context-engineering-evolving-contexts-for-self-improving-language-models\u002F\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">Microsoft Research — 智能体上下文工程（ACE）\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">关于演化结构化上下文，同时应对简洁性偏差和上下文崩溃的研究。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fevaluation-best-practices\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">OpenAI — 评估最佳实践\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">关于测试边缘情况的指导，包括使用明确、可重复的评估来测试长上下文和长时间运行的对话。\u003C\u002Fp>\u003C\u002Fa>",{"time":212,"blocks":213,"version":796},1790366477042,[214,222,228,236,243,248,253,258,263,268,298,303,308,313,320,325,330,335,340,345,350,355,360,365,370,375,380,385,390,395,437,444,449,454,485,490,522,527,532,537,546,551,556,561,569,574,579,584,589,594,599,607,612,630,635,640,645,650,655,660,665,670,675,680,685,711,716,745,750,760,769,778,787],{"id":215,"data":216,"type":220,"tunes":221},"Qfxj3iD3g1",{"title":217,"maxLevel":218,"minLevel":219},"目录",3,2,"tableOfContents",{},{"id":223,"data":224,"type":226,"tunes":227},"intro",{"text":225},"更大的上下文窗口为 AI 系统提供了更多容量。它并不保证模型会很好地利用这些容量。在长对话、RAG 流水线、研究智能体和工具密集型工作流中，添加更多历史记录、更多文档、更多工具输出或更多记忆，可能会让回答变得更不可靠，而不是更有依据。","paragraph",{},{"id":229,"data":230,"type":234,"tunes":235},"direct",{"body":231,"title":232,"variant":233},"\u003Cstrong>当额外信息降低了信噪比、引入冲突、掩盖决定性证据、保留过时状态或压缩掉重要条件时，更多上下文反而会让 AI 回答变差。\u003C\u002Fstrong> 因此，相关的工程目标不是最大化上下文，而是\u003Cstrong>在保留证据和决策边界的前提下，使用最小充分上下文\u003C\u002Fstrong>。","直接回答","info","callout",{},{"id":237,"data":238,"type":234,"tunes":242},"model-note",{"body":239,"title":240,"variant":241},"下文中的上下文质量模型和上下文压力测试是本文提出的实用架构方法，并非正式的行业标准。它们综合了关于长上下文位置效应、上下文污染、压缩、检索和上下文工程的既有研究成果。","关于本文所用模型","note",{},{"id":244,"data":245,"type":42,"tunes":247},"h-capacity",{"text":246,"level":219},"上下文容量不等于上下文可用性",{},{"id":249,"data":250,"type":226,"tunes":252},"p-capacity-1",{"text":251},"模型宣称的上下文窗口描述的是它能接受多少输入。这并不意味着该窗口内的每个 token 都会获得同等关注，或对最终答案有同等贡献。这一区别很重要，因为生产系统越来越多地用对话历史、检索文档、工具结果、记忆、结构化状态、指令和中间产物来填充上下文。",{},{"id":254,"data":255,"type":226,"tunes":257},"p-capacity-2",{"text":256},"经典的“迷失在中间”研究表明，当相关证据出现在长输入的中部时，长上下文模型的表现可能比证据出现在开头或结尾附近时更差。更广泛的工程启示并不是长上下文不好，而是上下文中的可用性并不等同于可靠使用。",{},{"id":259,"data":260,"type":226,"tunes":262},"p-capacity-3",{"text":261},"OpenAI 的上下文管理指南从另一个方向得出了相同的操作结论：即使是非常大的上下文窗口，也可能被未经筛选的历史记录、冗余的工具输出和嘈杂的检索结果淹没。Anthropic 同样将上下文视为一种有限资源，需要主动工程化，而不是被动累积。",{},{"id":264,"data":265,"type":42,"tunes":267},"h-five",{"text":266,"level":219},"额外上下文降低答案质量的五种方式",{},{"id":269,"data":270,"type":296,"tunes":297},"five-table",{"content":271,"stretched":43,"withHeadings":14},[272,276,280,284,288,292],[273,274,275],"失败模式","增加更多上下文时发生的变化","典型症状",[277,278,279],"信号稀释","相关证据在总输入中所占比例变小","模型给出泛泛的回答，或错过决定性段落",[281,282,283],"证据冲突","不同文档、版本或记忆之间相互矛盾","答案混合了不兼容的说法，或选择了错误版本",[285,286,287],"位置敏感性","决定性信息移动到上下文中使用可靠性较低的位置","同一证据在一种排序下有效，在另一种排序下失效",[289,290,291],"过时上下文残留","现实变化后，旧状态或先前结论仍然存在","模型不断重复曾经正确但现已过时的答案",[293,294,295],"压缩损失","压缩或摘要去掉了限定条件、例外、出处或未解决的不确定性","摘要连贯，但由此产生的答案变得过度自信或过度概括","table",{},{"id":299,"data":300,"type":42,"tunes":302},"h-dilution",{"text":301,"level":218},"1. 信号稀释：相关证据与其他所有内容竞争",{},{"id":304,"data":305,"type":226,"tunes":307},"p-dilution-1",{"text":306},"假设一个问题可以用两段短文回答。一个 RAG 系统“为了保险”检索了这两段，再加上十八段关系不大的内容。检索召回率可能提高了，但生成器现在必须从背景材料中区分出决定性证据。如果相似措辞出现在多个文档中，额外上下文可能会让答案变得不够精确。",{},{"id":309,"data":310,"type":226,"tunes":312},"p-dilution-2",{"text":311},"这就在检索召回率和上下文效用之间形成了一个重要区别。更多检索材料可能提高答案存在于上下文中某处的概率，同时降低模型给予正确证据足够权重的概率。",{},{"id":314,"data":315,"type":234,"tunes":319},"dilution-tip",{"body":316,"title":317,"variant":318},"不要孤立地优化 top-k。要衡量添加文档是否改善最终结论、保留证据归属，并能在反复试验中保持稳定。","工程规则","tip",{},{"id":321,"data":322,"type":42,"tunes":324},"h-conflict",{"text":323,"level":218},"2. 证据冲突：更多来源可能意味着更多版本的现实",{},{"id":326,"data":327,"type":226,"tunes":329},"p-conflict-1",{"text":328},"长上下文常常包含相互不一致的信息：新旧 API 文档、两个政策版本、先前和当前用户偏好、相互竞争的网页来源、缓存状态，或不再与来源匹配的模型生成摘要。",{},{"id":331,"data":332,"type":226,"tunes":334},"p-conflict-2",{"text":333},"这种失败不一定是幻觉。模型可能是在忠实地组合相互矛盾的证据。因此，架构需要优先级规则：来源权威性、版本、时间戳、司法管辖区、租户、产品修订版、用户状态或显式取代元数据。",{},{"id":336,"data":337,"type":226,"tunes":339},"p-conflict-3",{"text":338},"如果没有这些规则，增加上下文可能会让矛盾增加得比知识更快。",{},{"id":341,"data":342,"type":42,"tunes":344},"h-position",{"text":343,"level":218},"3. 位置敏感性：证据出现的位置会改变结果",{},{"id":346,"data":347,"type":226,"tunes":349},"p-position-1",{"text":348},"“迷失在中间”的结果表明，仅改变相关信息的位置就可能实质性改变模型性能。这一发现对于按固定顺序拼接大量检索段落或长历史记录的系统尤为重要。",{},{"id":351,"data":352,"type":226,"tunes":354},"p-position-2",{"text":353},"因此，生产测试应改变文档顺序，而不仅仅是测试一个标准提示。如果系统只有在决定性证据位于最前或最后时才能正确回答，那么该应用比单一基准分数所显示的更为脆弱。",{},{"id":356,"data":357,"type":42,"tunes":359},"h-stale",{"text":358,"level":218},"4. 陈旧上下文持续存在：模型同时看到真相和历史",{},{"id":361,"data":362,"type":226,"tunes":364},"p-stale-1",{"text":363},"长时间运行的智能体经常将先前的结论延续下去。这种连续性在事实发生变化之前是有用的。如果昨天的工具结果显示某个部署是健康的，而当前的工具结果显示它已降级，除非系统明确替换或限定旧状态的范围，否则两者可能都保留在上下文中。",{},{"id":366,"data":367,"type":226,"tunes":369},"p-stale-2",{"text":368},"这就是为什么当前操作状态通常应来自权威来源，而记忆则保留持久性上下文，如决策、偏好或流程。更多的对话历史并不能替代重新读取当前状态。",{},{"id":371,"data":372,"type":42,"tunes":374},"h-compression",{"text":373,"level":218},"5. 压缩损失：更小的上下文也可能变成更差的上下文",{},{"id":376,"data":377,"type":226,"tunes":379},"p-compression-1",{"text":378},"相反的操作——压缩上下文——也有失败模式。摘要可能会丢失例外情况、未解决的问题、来源、精确标识符、负面证据，或结论有效的条件。",{},{"id":381,"data":382,"type":226,"tunes":384},"p-compression-2",{"text":383},"微软研究院的智能体上下文工程工作将相关问题描述为简洁性偏差和上下文崩溃：迭代重写可能会移除有用的领域细节。因此，目标不是“尽可能压缩”。而是在减少上下文的同时，保留那些会改变决策的信息。",{},{"id":386,"data":387,"type":42,"tunes":389},"h-quality",{"text":388,"level":219},"上下文质量模型",{},{"id":391,"data":392,"type":226,"tunes":394},"p-quality-intro",{"text":393},"有用的上下文可以从六个维度进行评估。这些维度都不是简单的令牌数量。",{},{"id":396,"data":397,"type":435,"tunes":436},"quality-comparison",{"rows":398,"title":424,"layout":296,"columns":425},[399,404,408,412,416,420],{"id":400,"label":401,"values":402},"relevance","相关性",[403,403,403],"",{"id":405,"label":406,"values":407},"authority","权威性",[403,403,403],{"id":409,"label":410,"values":411},"freshness","新鲜度",[403,403,403],{"id":413,"label":414,"values":415},"consistency","一致性",[403,403,403],{"id":417,"label":418,"values":419},"completeness","决策完整性",[403,403,403],{"id":421,"label":422,"values":423},"traceability","可追溯性",[403,403,403],"上下文质量的六个维度",[426,429,432],{"id":427,"label":428},"dimension","维度",{"id":430,"label":431},"question","问题",{"id":433,"label":434},"failure","如果薄弱","comparison",{},{"id":438,"data":439,"type":234,"tunes":443},"target-state",{"body":440,"title":441,"variant":442},"最好的上下文不是最大的上下文。而是\u003Cstrong>仍然保留可靠答案或行动所需的证据、约束、状态、例外情况和来源的最小上下文\u003C\u002Fstrong>。","目标状态","success",{},{"id":445,"data":446,"type":42,"tunes":448},"h-pressure",{"text":447,"level":219},"上下文压力测试",{},{"id":450,"data":451,"type":226,"tunes":453},"p-pressure-intro",{"text":452},"要确定应用是否能从更多上下文中受益，应将上下文大小作为实验变量进行测试，而不是假设越大越好。",{},{"id":455,"data":456,"type":483,"tunes":484},"pressure-flow",{"steps":457,"title":447,"orientation":482},[458,461,464,467,470,473,476,479],{"label":459,"description":460},"1. 定义黄金案例","选择一个具有已知答案和已知最小证据集的任务。",{"label":462,"description":463},"2. 运行最小充分上下文","仅提供答案所需的指令、当前状态和证据。",{"label":465,"description":466},"3. 添加相关背景","添加有用但非决定性的上下文，并衡量质量是提高、保持稳定还是下降。",{"label":468,"description":469},"4. 添加现实噪声","添加生产系统可能包含的松散相关历史、工具输出或检索段落。",{"label":471,"description":472},"5. 添加受控冲突","引入带有明确版本元数据的陈旧或矛盾证据，并验证正确来源仍然胜出。",{"label":474,"description":475},"6. 重新排序决定性证据","将关键信息放在开头、中间和结尾附近，以测试位置敏感性。",{"label":477,"description":478},"7. 测试压缩","用摘要替换较旧的上下文，并验证限定词、来源、未解决问题和决策边界是否保留。",{"label":480,"description":481},"8. 比较质量曲线","随着上下文变化，衡量正确性、证据使用、一致性、延迟、成本和方差。","auto","processFlow",{},{"id":486,"data":487,"type":42,"tunes":489},"h-measure",{"text":488,"level":219},"应衡量什么而不是令牌数量",{},{"id":491,"data":492,"type":296,"tunes":521},"measure-table",{"content":493,"stretched":43,"withHeadings":14},[494,497,500,503,506,509,512,515,518],[495,496],"指标","它揭示了什么",[498,499],"答案正确性","最终结果是否正确",[501,502],"声明级证据支持","随着上下文变化，重要声明是否仍然有依据",[504,505],"证据利用率","答案是否遵循决定性证据，而不是先前的模型知识",[507,508],"冲突解决准确性","当前\u002F权威证据是否胜过陈旧或较弱的来源",[510,511],"位置鲁棒性","重新排序证据是否会改变正确性",[513,514],"压缩保留","摘要是否保留约束、例外、标识符、来源和未解决状态",[516,517],"多次试验的输出方差","额外的上下文是否使系统更不稳定",[519,520],"延迟和令牌成本","添加的信息是否产生足够质量以证明其运营成本合理",{},{"id":523,"data":524,"type":42,"tunes":526},"h-topk",{"text":525,"level":219},"RAG：为什么增加 top-k 可能适得其反",{},{"id":528,"data":529,"type":226,"tunes":531},"p-topk-1",{"text":530},"一种常见的 RAG 调优模式是：当系统漏掉答案时，就增加 top-k。这可以提高候选召回率，但也会增加无关上下文、重复证据、过时段落以及相互冲突的文档。",{},{"id":533,"data":534,"type":226,"tunes":536},"p-topk-2",{"text":535},"更好的问题是：决定性证据究竟是缺失于检索阶段，还是仅仅在上下文组装后失去了影响力。如果正确的段落已经出现在候选集中，那么增加 top-k 可能是在解决错误的问题。",{},{"id":538,"data":539,"type":544,"tunes":545},"internal-rag",{"url":540,"title":541,"excerpt":542,"ctaLabel":543},"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method","RAG 失败了——但究竟是哪一层失败了？一种诊断方法","一种逐层隔离的方法，用于定位来源覆盖、检索、排序、上下文组装、生成、证据归因和时效性方面的失败。","阅读 RAG 诊断方法","referralArticle",{},{"id":547,"data":548,"type":42,"tunes":550},"h-agents",{"text":549,"level":219},"长时间运行的智能体：连续性不等于累积",{},{"id":552,"data":553,"type":226,"tunes":555},"p-agents-1",{"text":554},"智能体需要跨步骤的连续性，但连续性并不要求重放此前的每一个 token。OpenAI 展示了针对长时间运行会话上下文的裁剪和压缩方法。Anthropic 推荐压缩、结构化笔记以及其他技术，以在控制上下文污染的同时保留有用信息。",{},{"id":557,"data":558,"type":226,"tunes":560},"p-agents-2",{"text":559},"一个强大的长时间运行架构通常会将持久记忆、当前状态、外部工件、检索以及面向模型的上下文分离开来。这样系统就能保留重要内容，而不必把每一个历史细节都塞进每一次推理中。",{},{"id":562,"data":563,"type":544,"tunes":568},"internal-memory",{"url":564,"title":565,"excerpt":566,"ctaLabel":567},"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context","AI 智能体记忆不是 RAG：如何分离记忆、检索、状态与上下文","一种实用的四层架构，用于分离哪些内容持久存在、哪些内容当前具有权威性、哪些内容被检索，以及模型实际接收到什么。","阅读记忆架构文章",{},{"id":570,"data":571,"type":42,"tunes":573},"h-order",{"text":572,"level":219},"上下文顺序应当是有意设计的",{},{"id":575,"data":576,"type":226,"tunes":578},"p-order-1",{"text":577},"上下文构建是一个信息架构问题。关键指令、当前状态、决定性证据以及任务特定约束不应被随意放置。当系统机械地拼接来源时，它们实际上是把优先级隐式交给了位置效应和模型注意力。",{},{"id":580,"data":581,"type":226,"tunes":583},"p-order-2",{"text":582},"并不存在适用于所有模型和任务的通用最佳顺序，因此顺序应通过实证评估。一个有用的测试套件会随机化或系统地改变文档位置，并衡量同一主张是否保持稳定。",{},{"id":585,"data":586,"type":42,"tunes":588},"h-boundaries",{"text":587,"level":219},"在压缩过程中保留决策边界",{},{"id":590,"data":591,"type":226,"tunes":593},"p-bound-1",{"text":592},"一个只说“采用方案 X”的摘要，弱于一个保留了为什么选择 X 以及什么会使该决策失效的摘要。上下文压缩应保留那些可能改变答案的变量：版本、日期、假设、状态、权威性、未解决的分歧以及证据来源。",{},{"id":595,"data":596,"type":226,"tunes":598},"p-bound-2",{"text":597},"这将上下文工程直接与答案有效性联系起来。如果压缩保留了一个结论，却移除了它的有效性边界，那么未来的回答可能仍然内部一致，却在外部变得错误。",{},{"id":600,"data":601,"type":544,"tunes":606},"internal-avb",{"url":602,"title":603,"excerpt":604,"ctaLabel":605},"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers","答案有效性边界：相关性与可靠 AI 答案之间缺失的一层","一个框架，用于明确 AI 主张在什么条件下适用，以及哪些变化需要限制、重新计算或放弃。","阅读答案有效性边界",{},{"id":608,"data":609,"type":42,"tunes":611},"h-policy",{"text":610,"level":219},"一种实用的上下文构建策略",{},{"id":613,"data":614,"type":628,"tunes":629},"policy-list",{"meta":615,"items":616,"style":627},{},[617,618,619,620,621,622,623,624,625,626],"从当前任务出发，而不是从系统所知的一切出发。","在做出重大决策之前，从权威系统重新读取易变状态。","针对当前问题检索证据，而不是把庞大的静态语料库一直带下去。","移除重复或低价值的工具输出。","为重要证据保留来源版本、时间戳、权威性和出处。","当当前信息与历史信息冲突时，明确优先级。","将规则与其例外和前提条件一起保留。","当持久决策和可复用流程不需要逐字重放时，将其存储在即时上下文之外。","只有在测试了约束、标识符、例外和出处保留情况后，才压缩历史。","通过重复试验而非单次提示来评估上下文大小、顺序和噪声。","unordered","list",{},{"id":631,"data":632,"type":42,"tunes":634},"h-change",{"text":633,"level":219},"什么会改变这个答案？",{},{"id":636,"data":637,"type":226,"tunes":639},"p-change-1",{"text":638},"这种权衡会随模型架构、训练方式、任务类型和上下文长度而变化。未来的模型可能会对位置、噪声和冲突信息变得显著更加稳健。对于拥有小型干净语料库的任务，直接提供完整来源，而不是构建复杂的检索流水线，也可能同样有益。",{},{"id":641,"data":642,"type":226,"tunes":644},"p-change-2",{"text":643},"当遗漏比噪声更危险时，这一建议也会改变。在高召回率的研究或发现任务中，在后续的过滤或综合阶段之前，使用更大的候选上下文可能是合理的。在对延迟敏感的生产系统中，更严格的上下文选择可能更可取。",{},{"id":646,"data":647,"type":226,"tunes":649},"p-change-3",{"text":648},"只有当模型能够可靠地对无关信息、位置、矛盾和过时证据保持不变时，核心原则才会改变。在此之前，上下文应被视为一种经过精心策划的执行资源，而不是被动存储。",{},{"id":651,"data":652,"type":42,"tunes":654},"h-limitations",{"text":653,"level":219},"局限性",{},{"id":656,"data":657,"type":226,"tunes":659},"p-limit-1",{"text":658},"长上下文行为在不同模型和工作负载之间差异很大。最初的“迷失在中间”实验使用的是早期几代模型，因此不应假设其确切的效应量能够代表当前系统。这一发现作为需要测试的失败模式仍然有用，但不应被视为普遍固定的性能曲线。",{},{"id":661,"data":662,"type":226,"tunes":664},"p-limit-2",{"text":663},"同样，减少上下文可能会移除必要的证据。压缩会引入摘要风险，而激进的检索过滤可能会降低召回率。目标不是不惜一切代价追求最少 token，而是为正在做出的决策提供充分、最新、可追溯的上下文。",{},{"id":666,"data":667,"type":42,"tunes":669},"h-conclusion",{"text":668,"level":219},"结论",{},{"id":671,"data":672,"type":226,"tunes":674},"p-conclusion-1",{"text":673},"“模型能接受多少上下文？”这个问题，不如“这些上下文中有多少能改善决策？”更有用。更多 token 可以增加证据，但也可能增加干扰、矛盾、过时状态、位置脆弱性和压缩债务。",{},{"id":676,"data":677,"type":226,"tunes":679},"p-conclusion-2",{"text":678},"将上下文视为经过工程化设计的工作集。从最小充分证据开始。只有当信息能改善可衡量的性能时才添加它。明确测试噪声、冲突、排序和压缩。大上下文窗口是一种容量；上下文质量则是一种架构。",{},{"id":681,"data":682,"type":42,"tunes":684},"h-faq",{"text":683,"level":219},"常见问题",{},{"id":686,"data":687,"type":686,"tunes":710},"faq",{"items":688,"title":709},[689,693,697,701,705],{"id":690,"answer":691,"question":692},"faq1","会。额外的上下文可能稀释相关证据、引入矛盾或过时信息、将决定性证据移到不够稳健的位置，并增加模型使用弱信号而非决定性信号的可能性。","给 AI 模型更多上下文会让它的回答更差吗？",{"id":694,"answer":695,"question":696},"faq2","通常不会。更大的上下文窗口增加了容量，但检索仍然有助于选择最新且相关的信息、控制成本、保留来源边界，并避免在每次请求中发送大量无关数据。","更大的上下文窗口会消除对 RAG 的需求吗？",{"id":698,"answer":699,"question":700},"faq3","它描述了观察到的情况：当相关信息位于长上下文中间时，语言模型对该信息的使用不如它出现在开头或结尾附近时可靠。确切效应因模型和任务而异，应在当前系统上进行测试。","什么是“迷失在中间”问题？",{"id":702,"answer":703,"question":704},"faq4","不应该。如果候选集中缺少相关证据，更大的 top-k 可能会提高召回率。如果证据已经存在，但被额外材料稀释，那么增加 top-k 可能会使上下文更差。应分别诊断检索和上下文组装。","我应该总是降低 RAG 的 top-k 吗？",{"id":706,"answer":707,"question":708},"faq5","保留持久决策、当前目标、未解决问题、标识符、约束、例外情况、证据来源，以及会改变先前结论的条件。","上下文摘要应保留什么？","长上下文与 AI 回答质量",{},{"id":712,"data":713,"type":42,"tunes":715},"h-glossary",{"text":714,"level":219},"术语表",{},{"id":717,"data":718,"type":717,"tunes":744},"glossary",{"title":719,"entries":720},"关键上下文工程术语",[721,725,729,732,736,740],{"term":722,"anchor":723,"definition":724},"上下文窗口","context-window","模型在一次推理序列中能够关注到的输入和输出 token 信息量。",{"term":726,"anchor":727,"definition":728},"上下文污染","context-pollution","由于无关、过时、冗余、冲突或其他低价值信息占用模型上下文而导致的性能下降。",{"term":277,"anchor":730,"definition":731},"signal-dilution","随着额外低价值或竞争性信息被加入，决定性证据的相对突出程度下降。",{"term":733,"anchor":734,"definition":735},"上下文压缩","context-compaction","通过摘要、重构、外置或以其他方式将必要信息保留在更小的工作表示中，从而减少累积的上下文。",{"term":737,"anchor":738,"definition":739},"位置稳健性","position-robustness","当相关信息出现在上下文中的不同位置时，模型性能保持稳定的程度。",{"term":741,"anchor":742,"definition":743},"最小充分上下文","minimum-sufficient-context","仍然保留可靠执行所需的证据、状态、约束、例外和来源的最小实用工作上下文。",{},{"id":746,"data":747,"type":42,"tunes":749},"h-sources",{"text":748,"level":219},"主要来源与延伸阅读",{},{"id":751,"data":752,"type":758,"tunes":759},"src-openai-session",{"link":753,"meta":754},"https:\u002F\u002Fdevelopers.openai.com\u002Fcookbook\u002Fexamples\u002Fagents_sdk\u002Fsession_memory",{"image":755,"title":756,"description":757},{"url":403},"OpenAI — 上下文工程：使用会话进行短期记忆管理","关于裁剪和压缩的指导，并讨论了干扰、低效、过时上下文、噪声检索和长时间运行的会话。","linkTool",{},{"id":761,"data":762,"type":758,"tunes":768},"src-anthropic-context",{"link":763,"meta":764},"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Feffective-context-engineering-for-ai-agents",{"image":765,"title":766,"description":767},{"url":403},"Anthropic — 面向 AI 智能体的有效上下文工程","关于上下文污染、压缩、结构化笔记和长时程智能体上下文管理的工程指导。",{},{"id":770,"data":771,"type":758,"tunes":777},"src-lost-middle",{"link":772,"meta":773},"https:\u002F\u002Faclanthology.org\u002F2024.tacl-1.9\u002F",{"image":774,"title":775,"description":776},{"url":403},"Liu 等 — 迷失在中间：语言模型如何使用长上下文","TACL 论文展示了长上下文中相关信息使用对位置的敏感性，并推动了对长上下文稳健性的明确测试。",{},{"id":779,"data":780,"type":758,"tunes":786},"src-ms-ace",{"link":781,"meta":782},"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Fpublication\u002Fagentic-context-engineering-evolving-contexts-for-self-improving-language-models\u002F",{"image":783,"title":784,"description":785},{"url":403},"Microsoft Research — 智能体上下文工程（ACE）","关于演化结构化上下文，同时应对简洁性偏差和上下文崩溃的研究。",{},{"id":788,"data":789,"type":758,"tunes":795},"src-openai-evals",{"link":790,"meta":791},"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fevaluation-best-practices",{"image":792,"title":793,"description":794},{"url":403},"OpenAI — 评估最佳实践","关于测试边缘情况的指导，包括使用明确、可重复的评估来测试长上下文和长时间运行的对话。",{},"2.31","更大的上下文窗口并不保证更好的答案。本文解释了信号稀释、证据冲突、状态过时、位置敏感性和有损压缩如何降低AI可靠性——并介绍了一种实用的上下文压力测试。","\u002Fuploads\u002F2026\u002F09\u002Fwhy-more-context-can-make-ai-answers-worse-1790351615793-2ntv2v.webp","why-more-context-can-make-ai-answers-worse-1790351615793-2ntv2v","PUBLISHED","2026-09-25T11:51:00.000Z","2026-09-25T15:51:57.195Z","2026-09-25T20:09:28.015Z",{"en":805,"de":806,"sr":807,"es":808,"fr":809,"it":810,"ru":811,"zh":812},"\u002Fblog\u002Fwhy-more-context-can-make-ai-answers-worse","\u002Fde\u002Fblog\u002Fwhy-more-context-can-make-ai-answers-worse","\u002Fsr\u002Fblog\u002Fwhy-more-context-can-make-ai-answers-worse","\u002Fes\u002Fblog\u002Fwhy-more-context-can-make-ai-answers-worse","\u002Ffr\u002Fblog\u002Fwhy-more-context-can-make-ai-answers-worse","\u002Fit\u002Fblog\u002Fwhy-more-context-can-make-ai-answers-worse","\u002Fru\u002Fblog\u002Fwhy-more-context-can-make-ai-answers-worse","\u002Fzh\u002Fblog\u002Fwhy-more-context-can-make-ai-answers-worse",[814,818,822],{"id":815,"name":816,"slug":817},64,"信息架构","information-architecture",{"id":819,"name":820,"slug":821},60,"成本与延迟控制","cost-and-latency",{"id":823,"name":824,"slug":825},97,"在测试集上验证","verification",{"id":827,"login":828,"email":829,"displayName":830},"20","rooth8233","aleksandar@stajic.de","Aleksandar Stajić",[832,1307],{"lang":833,"title":834,"content":835,"contentJson":836,"excerpt":1306},"en","Why More Context Can Make AI Answers Worse","{\"time\":1790351629251,\"blocks\":[{\"id\":\"Qfxj3iD3g1\",\"type\":\"tableOfContents\",\"data\":{\"title\":\"Contents\",\"minLevel\":2,\"maxLevel\":3},\"tunes\":{}},{\"id\":\"intro\",\"type\":\"paragraph\",\"data\":{\"text\":\"A larger context window gives an AI system more capacity. It does not guarantee that the model will use that capacity well. In long conversations, RAG pipelines, research agents, and tool-heavy workflows, adding more history, more documents, more tool output, or more memory can make a response less reliable rather than more informed.\"},\"tunes\":{}},{\"id\":\"direct\",\"type\":\"callout\",\"data\":{\"variant\":\"info\",\"title\":\"Direct answer\",\"body\":\"\u003Cstrong>More context can make an AI answer worse when the additional information lowers the signal-to-noise ratio, introduces conflicts, hides decisive evidence, preserves stale state, or compresses away important conditions.\u003C\u002Fstrong> The relevant engineering objective is therefore not maximum context. It is \u003Cstrong>minimum sufficient context with preserved evidence and decision boundaries\u003C\u002Fstrong>.\"},\"tunes\":{}},{\"id\":\"model-note\",\"type\":\"callout\",\"data\":{\"variant\":\"note\",\"title\":\"About the model used in this article\",\"body\":\"The Context Quality model and Context Pressure Test below are practical architecture methods proposed in this article, not formal industry standards. They synthesize established findings on long-context position effects, context pollution, compaction, retrieval, and context engineering.\"},\"tunes\":{}},{\"id\":\"h-capacity\",\"type\":\"header\",\"data\":{\"text\":\"Context capacity is not context usability\",\"level\":2},\"tunes\":{}},{\"id\":\"p-capacity-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A model's advertised context window describes how much input it can accept. It does not imply that every token inside that window receives equal attention or contributes equally to the final answer. The distinction matters because production systems increasingly fill context with conversation history, retrieved documents, tool results, memory, structured state, instructions, and intermediate artifacts.\"},\"tunes\":{}},{\"id\":\"p-capacity-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The classic “Lost in the Middle” study showed that long-context models can perform worse when relevant evidence appears in the middle of a long input than when it appears near the beginning or end. The broader engineering lesson is not that long context is bad. It is that availability inside the context is not equivalent to reliable use.\"},\"tunes\":{}},{\"id\":\"p-capacity-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"OpenAI's context-management guidance reaches the same operational conclusion from another direction: even very large context windows can be overwhelmed by uncurated history, redundant tool output, and noisy retrieval. Anthropic likewise treats context as a finite resource that requires active engineering rather than passive accumulation.\"},\"tunes\":{}},{\"id\":\"h-five\",\"type\":\"header\",\"data\":{\"text\":\"Five ways additional context can reduce answer quality\",\"level\":2},\"tunes\":{}},{\"id\":\"five-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Failure mode\",\"What changes when more context is added\",\"Typical symptom\"],[\"Signal dilution\",\"Relevant evidence becomes a smaller fraction of the total input\",\"The model gives a generic answer or misses the decisive passage\"],[\"Evidence conflict\",\"Different documents, versions, or memories disagree\",\"The answer blends incompatible claims or chooses the wrong version\"],[\"Position sensitivity\",\"Decisive information moves into a less reliably used part of the context\",\"The same evidence works in one ordering but fails in another\"],[\"Stale-context persistence\",\"Old state or prior conclusions remain present after reality changes\",\"The model keeps repeating a formerly correct answer\"],[\"Compression loss\",\"Compaction or summarization removes qualifiers, exceptions, provenance, or unresolved uncertainty\",\"The summary is coherent but the resulting answer becomes overconfident or overgeneralized\"]]},\"tunes\":{}},{\"id\":\"h-dilution\",\"type\":\"header\",\"data\":{\"text\":\"1. Signal dilution: relevant evidence competes with everything else\",\"level\":3},\"tunes\":{}},{\"id\":\"p-dilution-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Suppose a question can be answered from two short passages. A RAG system retrieves those passages plus eighteen loosely related ones “for safety.” Retrieval recall may improve, but the generator must now distinguish decisive evidence from background material. If similar phrases appear across several documents, the additional context can make the answer less precise.\"},\"tunes\":{}},{\"id\":\"p-dilution-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"This creates an important distinction between retrieval recall and context utility. More retrieved material can increase the probability that the answer exists somewhere in the context while simultaneously reducing the probability that the model gives the right evidence enough weight.\"},\"tunes\":{}},{\"id\":\"dilution-tip\",\"type\":\"callout\",\"data\":{\"variant\":\"tip\",\"title\":\"Engineering rule\",\"body\":\"Do not optimize top-k in isolation. Measure whether adding documents improves the final claim, preserves evidence attribution, and survives repeated trials.\"},\"tunes\":{}},{\"id\":\"h-conflict\",\"type\":\"header\",\"data\":{\"text\":\"2. Evidence conflict: more sources can mean more versions of reality\",\"level\":3},\"tunes\":{}},{\"id\":\"p-conflict-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Long contexts often contain mutually inconsistent information: old and new API documentation, two policy versions, previous and current user preferences, competing web sources, cached state, or a model-generated summary that no longer matches the source.\"},\"tunes\":{}},{\"id\":\"p-conflict-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The failure is not necessarily hallucination. The model may be faithfully combining contradictory evidence. The architecture therefore needs precedence rules: source authority, version, timestamp, jurisdiction, tenant, product revision, user state, or explicit supersession metadata.\"},\"tunes\":{}},{\"id\":\"p-conflict-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"Without those rules, increasing context can increase contradiction faster than it increases knowledge.\"},\"tunes\":{}},{\"id\":\"h-position\",\"type\":\"header\",\"data\":{\"text\":\"3. Position sensitivity: where evidence appears can change the result\",\"level\":3},\"tunes\":{}},{\"id\":\"p-position-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The “Lost in the Middle” results demonstrated that changing only the position of relevant information can materially change model performance. That finding is especially important for systems that concatenate many retrieved passages or long histories in a fixed order.\"},\"tunes\":{}},{\"id\":\"p-position-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"A production test should therefore vary document order, not merely test one canonical prompt. If the system answers correctly only when the decisive evidence is first or last, the application is more fragile than a single benchmark score suggests.\"},\"tunes\":{}},{\"id\":\"h-stale\",\"type\":\"header\",\"data\":{\"text\":\"4. Stale-context persistence: the model sees truth and history together\",\"level\":3},\"tunes\":{}},{\"id\":\"p-stale-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Long-running agents frequently carry earlier conclusions forward. That continuity is useful until a fact changes. If a tool result from yesterday says a deployment is healthy and a current tool result says it is degraded, both may remain in context unless the system explicitly replaces or scopes old state.\"},\"tunes\":{}},{\"id\":\"p-stale-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"This is why current operational state should normally come from an authoritative source, while memory preserves durable context such as decisions, preferences, or procedures. More conversation history is not a substitute for re-reading the present.\"},\"tunes\":{}},{\"id\":\"h-compression\",\"type\":\"header\",\"data\":{\"text\":\"5. Compression loss: smaller context can also become worse context\",\"level\":3},\"tunes\":{}},{\"id\":\"p-compression-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The opposite intervention — compressing context — also has failure modes. Summaries can drop exceptions, unresolved questions, provenance, precise identifiers, negative evidence, or the conditions under which a conclusion was valid.\"},\"tunes\":{}},{\"id\":\"p-compression-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Microsoft Research's Agentic Context Engineering work describes a related problem as brevity bias and context collapse: iterative rewriting can remove useful domain detail. The objective is therefore not “compress as much as possible.” It is to reduce context while preserving the information that changes decisions.\"},\"tunes\":{}},{\"id\":\"h-quality\",\"type\":\"header\",\"data\":{\"text\":\"The Context Quality model\",\"level\":2},\"tunes\":{}},{\"id\":\"p-quality-intro\",\"type\":\"paragraph\",\"data\":{\"text\":\"A useful context can be evaluated across six dimensions. None of them is simply token count.\"},\"tunes\":{}},{\"id\":\"quality-comparison\",\"type\":\"comparison\",\"data\":{\"title\":\"Six dimensions of context quality\",\"layout\":\"table\",\"columns\":[{\"id\":\"dimension\",\"label\":\"Dimension\"},{\"id\":\"question\",\"label\":\"Question\"},{\"id\":\"failure\",\"label\":\"If weak\"}],\"rows\":[{\"id\":\"relevance\",\"label\":\"Relevance\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"authority\",\"label\":\"Authority\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"freshness\",\"label\":\"Freshness\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"consistency\",\"label\":\"Consistency\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"completeness\",\"label\":\"Decision completeness\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"traceability\",\"label\":\"Traceability\",\"values\":[\"\",\"\",\"\"]}]},\"tunes\":{}},{\"id\":\"target-state\",\"type\":\"callout\",\"data\":{\"variant\":\"success\",\"title\":\"Target state\",\"body\":\"The best context is not the largest context. It is the \u003Cstrong>smallest context that still preserves the evidence, constraints, state, exceptions, and provenance required for a reliable answer or action\u003C\u002Fstrong>.\"},\"tunes\":{}},{\"id\":\"h-pressure\",\"type\":\"header\",\"data\":{\"text\":\"The Context Pressure Test\",\"level\":2},\"tunes\":{}},{\"id\":\"p-pressure-intro\",\"type\":\"paragraph\",\"data\":{\"text\":\"To determine whether an application benefits from more context, test context size as an experimental variable instead of assuming that larger is better.\"},\"tunes\":{}},{\"id\":\"pressure-flow\",\"type\":\"processFlow\",\"data\":{\"title\":\"Context Pressure Test\",\"orientation\":\"auto\",\"steps\":[{\"label\":\"1. Define a gold case\",\"description\":\"Choose a task with a known answer and a known minimal evidence set.\"},{\"label\":\"2. Run minimal sufficient context\",\"description\":\"Provide only the instructions, current state, and evidence necessary for the answer.\"},{\"label\":\"3. Add relevant background\",\"description\":\"Add useful but non-decisive context and measure whether quality improves, stays stable, or falls.\"},{\"label\":\"4. Add realistic noise\",\"description\":\"Add loosely related history, tool output, or retrieved passages that a production system might include.\"},{\"label\":\"5. Add controlled conflicts\",\"description\":\"Introduce stale or contradictory evidence with clear version metadata and verify that the correct source still wins.\"},{\"label\":\"6. Reorder decisive evidence\",\"description\":\"Place the key information near the beginning, middle, and end to test position sensitivity.\"},{\"label\":\"7. Test compaction\",\"description\":\"Replace older context with a summary and verify that qualifiers, provenance, unresolved issues, and decision boundaries survive.\"},{\"label\":\"8. Compare the quality curve\",\"description\":\"Measure correctness, evidence use, consistency, latency, cost, and variance as context changes.\"}]},\"tunes\":{}},{\"id\":\"h-measure\",\"type\":\"header\",\"data\":{\"text\":\"What to measure instead of token count\",\"level\":2},\"tunes\":{}},{\"id\":\"measure-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Metric\",\"What it reveals\"],[\"Answer correctness\",\"Whether the final result is right\"],[\"Claim-level evidence support\",\"Whether material claims remain grounded as context changes\"],[\"Evidence utilization\",\"Whether the answer follows the decisive evidence instead of prior model knowledge\"],[\"Conflict resolution accuracy\",\"Whether current \u002F authoritative evidence wins over stale or weaker sources\"],[\"Position robustness\",\"Whether reordering evidence changes correctness\"],[\"Compaction retention\",\"Whether summaries preserve constraints, exceptions, identifiers, provenance, and unresolved state\"],[\"Output variance across trials\",\"Whether additional context makes the system less stable\"],[\"Latency and token cost\",\"Whether the added information produces enough quality to justify its operational cost\"]]},\"tunes\":{}},{\"id\":\"h-topk\",\"type\":\"header\",\"data\":{\"text\":\"RAG: why increasing top-k can hurt\",\"level\":2},\"tunes\":{}},{\"id\":\"p-topk-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A common RAG tuning pattern is to increase top-k when the system misses an answer. This can improve candidate recall but also increase irrelevant context, duplicate evidence, outdated passages, and conflicting documents.\"},\"tunes\":{}},{\"id\":\"p-topk-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The better question is whether the decisive evidence is missing from retrieval or merely losing influence after context assembly. If the correct passage already appears in the candidate set, increasing top-k may solve the wrong problem.\"},\"tunes\":{}},{\"id\":\"internal-rag\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method\",\"title\":\"RAG Failed — But Which Layer Actually Failed? A Diagnostic Method\",\"excerpt\":\"A layer-by-layer method for isolating source coverage, retrieval, ranking, context assembly, generation, evidence attribution, and freshness failures.\",\"ctaLabel\":\"Read the RAG diagnostic method\"},\"tunes\":{}},{\"id\":\"h-agents\",\"type\":\"header\",\"data\":{\"text\":\"Long-running agents: continuity is not accumulation\",\"level\":2},\"tunes\":{}},{\"id\":\"p-agents-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"An agent needs continuity across steps, but continuity does not require replaying every prior token. OpenAI demonstrates trimming and compression for long-running session context. Anthropic recommends compaction, structured note-taking, and other techniques to preserve useful information while controlling context pollution.\"},\"tunes\":{}},{\"id\":\"p-agents-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"A strong long-running architecture usually separates durable memory, current state, external artifacts, retrieval, and model-facing context. That allows the system to preserve what matters without forcing every historical detail into every inference.\"},\"tunes\":{}},{\"id\":\"internal-memory\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context\",\"title\":\"AI Agent Memory Is Not RAG: How to Separate Memory, Retrieval, State and Context\",\"excerpt\":\"A practical four-layer architecture for separating what persists, what is authoritative now, what is retrieved, and what the model actually receives.\",\"ctaLabel\":\"Read the memory architecture article\"},\"tunes\":{}},{\"id\":\"h-order\",\"type\":\"header\",\"data\":{\"text\":\"Context order should be intentional\",\"level\":2},\"tunes\":{}},{\"id\":\"p-order-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Context construction is an information architecture problem. Critical instructions, current state, decisive evidence, and task-specific constraints should not be placed arbitrarily. When systems concatenate sources mechanically, they implicitly delegate prioritization to positional effects and model attention.\"},\"tunes\":{}},{\"id\":\"p-order-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"There is no universal best ordering for every model and task, so ordering should be evaluated empirically. A useful test suite randomizes or systematically varies document position and measures whether the same claim remains stable.\"},\"tunes\":{}},{\"id\":\"h-boundaries\",\"type\":\"header\",\"data\":{\"text\":\"Preserve decision boundaries during compaction\",\"level\":2},\"tunes\":{}},{\"id\":\"p-bound-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A summary that says “use approach X” is weaker than a summary that preserves why X was chosen and what would invalidate the decision. Context compaction should retain the variables that can change the answer: version, date, assumptions, state, authority, unresolved disagreement, and evidence provenance.\"},\"tunes\":{}},{\"id\":\"p-bound-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"This connects context engineering directly to answer validity. If compaction preserves a conclusion but removes its validity boundary, future responses can remain internally consistent while becoming externally wrong.\"},\"tunes\":{}},{\"id\":\"internal-avb\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers\",\"title\":\"The Answer Validity Boundary: The Missing Layer Between Relevance and Reliable AI Answers\",\"excerpt\":\"A framework for making explicit the conditions under which an AI claim applies and what changes require restriction, recalculation, or abandonment.\",\"ctaLabel\":\"Read the Answer Validity Boundary\"},\"tunes\":{}},{\"id\":\"h-policy\",\"type\":\"header\",\"data\":{\"text\":\"A practical context construction policy\",\"level\":2},\"tunes\":{}},{\"id\":\"policy-list\",\"type\":\"list\",\"data\":{\"style\":\"unordered\",\"meta\":{},\"items\":[\"Start from the current task, not from everything the system knows.\",\"Re-read volatile state from authoritative systems before consequential decisions.\",\"Retrieve evidence for the current question instead of carrying large static corpora forward.\",\"Remove duplicate or low-value tool output.\",\"Keep source version, timestamp, authority, and provenance with important evidence.\",\"Make precedence explicit when current and historical information conflict.\",\"Preserve rules together with their exceptions and prerequisites.\",\"Store durable decisions and reusable procedures outside the immediate context when they do not need verbatim replay.\",\"Compact history only with tests for constraint, identifier, exception, and provenance retention.\",\"Evaluate context size, ordering, and noise with repeated trials rather than a single prompt.\"]},\"tunes\":{}},{\"id\":\"h-change\",\"type\":\"header\",\"data\":{\"text\":\"What would change this answer?\",\"level\":2},\"tunes\":{}},{\"id\":\"p-change-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The trade-off changes with model architecture, training, task type, and context length. Future models may become substantially more robust to position, noise, and conflicting information. A task with a small clean corpus can also benefit from simply providing the complete source rather than building an elaborate retrieval pipeline.\"},\"tunes\":{}},{\"id\":\"p-change-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The recommendation also changes when omission is more dangerous than noise. In high-recall research or discovery tasks, a larger candidate context may be justified before a later filtering or synthesis stage. In latency-sensitive production systems, stricter context selection may be preferable.\"},\"tunes\":{}},{\"id\":\"p-change-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The core principle would change only if models became reliably invariant to irrelevant information, position, contradiction, and stale evidence. Until then, context should be treated as a curated execution resource rather than passive storage.\"},\"tunes\":{}},{\"id\":\"h-limitations\",\"type\":\"header\",\"data\":{\"text\":\"Limitations\",\"level\":2},\"tunes\":{}},{\"id\":\"p-limit-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Long-context behaviour varies considerably across models and workloads. The original “Lost in the Middle” experiments used earlier generations of models, so their exact effect sizes should not be assumed to represent current systems. The finding remains useful as a failure pattern to test, not as a universal fixed performance curve.\"},\"tunes\":{}},{\"id\":\"p-limit-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Likewise, reducing context can remove necessary evidence. Compaction introduces summarization risk, and aggressive retrieval filtering can reduce recall. The objective is not minimal tokens at any cost; it is sufficient, current, traceable context for the decision being made.\"},\"tunes\":{}},{\"id\":\"h-conclusion\",\"type\":\"header\",\"data\":{\"text\":\"Conclusion\",\"level\":2},\"tunes\":{}},{\"id\":\"p-conclusion-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The question “How much context can the model accept?” is less useful than “How much of this context improves the decision?” More tokens can add evidence, but they can also add distraction, contradiction, stale state, positional fragility, and compression debt.\"},\"tunes\":{}},{\"id\":\"p-conclusion-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Treat context as an engineered working set. Start with minimum sufficient evidence. Add information only when it improves measured performance. Test noise, conflict, ordering, and compaction explicitly. A large context window is capacity; context quality is architecture.\"},\"tunes\":{}},{\"id\":\"h-faq\",\"type\":\"header\",\"data\":{\"text\":\"FAQ\",\"level\":2},\"tunes\":{}},{\"id\":\"faq\",\"type\":\"faq\",\"data\":{\"title\":\"Long context and AI answer quality\",\"items\":[{\"id\":\"faq1\",\"question\":\"Can giving an AI model more context make its answer worse?\",\"answer\":\"Yes. Additional context can dilute relevant evidence, introduce contradictory or stale information, move decisive evidence into less robust positions, and increase the chance that the model uses weak rather than decisive signals.\"},{\"id\":\"faq2\",\"question\":\"Does a larger context window eliminate the need for RAG?\",\"answer\":\"Not generally. A larger context window increases capacity, but retrieval still helps select current and relevant information, control cost, preserve source boundaries, and avoid sending large amounts of unrelated data into every request.\"},{\"id\":\"faq3\",\"question\":\"What is the Lost in the Middle problem?\",\"answer\":\"It describes observed cases where language models use relevant information less reliably when that information is located in the middle of a long context than when it appears near the beginning or end. The exact effect varies by model and task and should be tested on current systems.\"},{\"id\":\"faq4\",\"question\":\"Should I always reduce RAG top-k?\",\"answer\":\"No. If relevant evidence is missing from the candidate set, a larger top-k may improve recall. If the evidence is already present but gets diluted by additional material, increasing top-k can make the context worse. Diagnose retrieval and context assembly separately.\"},{\"id\":\"faq5\",\"question\":\"What should a context summary preserve?\",\"answer\":\"Preserve durable decisions, current goals, unresolved issues, identifiers, constraints, exceptions, evidence provenance, and the conditions that would change an earlier conclusion.\"}]},\"tunes\":{}},{\"id\":\"h-glossary\",\"type\":\"header\",\"data\":{\"text\":\"Glossary\",\"level\":2},\"tunes\":{}},{\"id\":\"glossary\",\"type\":\"glossary\",\"data\":{\"title\":\"Key context-engineering terms\",\"entries\":[{\"term\":\"Context window\",\"definition\":\"The amount of input and output token information a model can attend to within one inference sequence.\",\"anchor\":\"context-window\"},{\"term\":\"Context pollution\",\"definition\":\"Degradation caused by irrelevant, stale, redundant, conflicting, or otherwise low-value information occupying model context.\",\"anchor\":\"context-pollution\"},{\"term\":\"Signal dilution\",\"definition\":\"A reduction in the relative prominence of decisive evidence as additional low-value or competing information is added.\",\"anchor\":\"signal-dilution\"},{\"term\":\"Context compaction\",\"definition\":\"Reducing an accumulated context by summarizing, restructuring, externalizing, or otherwise preserving essential information in a smaller working representation.\",\"anchor\":\"context-compaction\"},{\"term\":\"Position robustness\",\"definition\":\"The degree to which model performance remains stable when relevant information appears in different positions inside the context.\",\"anchor\":\"position-robustness\"},{\"term\":\"Minimum sufficient context\",\"definition\":\"The smallest practical working context that still preserves the evidence, state, constraints, exceptions, and provenance required for reliable execution.\",\"anchor\":\"minimum-sufficient-context\"}]},\"tunes\":{}},{\"id\":\"h-sources\",\"type\":\"header\",\"data\":{\"text\":\"Primary sources and further reading\",\"level\":2},\"tunes\":{}},{\"id\":\"src-openai-session\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdevelopers.openai.com\u002Fcookbook\u002Fexamples\u002Fagents_sdk\u002Fsession_memory\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — Context Engineering: Short-Term Memory Management with Sessions\",\"description\":\"Guidance on trimming and compression, with discussion of distraction, inefficiency, stale context, noisy retrieval, and long-running sessions.\"}},\"tunes\":{}},{\"id\":\"src-anthropic-context\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Feffective-context-engineering-for-ai-agents\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Anthropic — Effective Context Engineering for AI Agents\",\"description\":\"Engineering guidance on context pollution, compaction, structured note-taking, and long-horizon agent context management.\"}},\"tunes\":{}},{\"id\":\"src-lost-middle\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Faclanthology.org\u002F2024.tacl-1.9\u002F\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Liu et al. — Lost in the Middle: How Language Models Use Long Contexts\",\"description\":\"TACL paper showing position-sensitive use of relevant information in long contexts and motivating explicit long-context robustness tests.\"}},\"tunes\":{}},{\"id\":\"src-ms-ace\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Fpublication\u002Fagentic-context-engineering-evolving-contexts-for-self-improving-language-models\u002F\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Microsoft Research — Agentic Context Engineering (ACE)\",\"description\":\"Research on evolving structured contexts while addressing brevity bias and context collapse.\"}},\"tunes\":{}},{\"id\":\"src-openai-evals\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fevaluation-best-practices\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — Evaluation Best Practices\",\"description\":\"Guidance on testing edge cases including long context and long-running conversations using explicit, repeatable evals.\"}},\"tunes\":{}}],\"version\":\"2.31.6\"}",{"time":837,"blocks":838,"version":1305},1790351629251,[839,843,847,852,857,861,865,869,873,877,905,909,913,917,922,926,930,934,938,942,946,950,954,958,962,966,970,974,978,982,1012,1017,1021,1025,1054,1058,1089,1093,1097,1101,1108,1112,1116,1120,1127,1131,1135,1139,1143,1147,1151,1158,1162,1177,1181,1185,1189,1193,1197,1201,1205,1209,1213,1217,1221,1241,1245,1266,1270,1277,1284,1291,1298],{"id":215,"data":840,"type":220,"tunes":842},{"title":841,"maxLevel":218,"minLevel":219},"Contents",{},{"id":223,"data":844,"type":226,"tunes":846},{"text":845},"A larger context window gives an AI system more capacity. It does not guarantee that the model will use that capacity well. In long conversations, RAG pipelines, research agents, and tool-heavy workflows, adding more history, more documents, more tool output, or more memory can make a response less reliable rather than more informed.",{},{"id":229,"data":848,"type":234,"tunes":851},{"body":849,"title":850,"variant":233},"\u003Cstrong>More context can make an AI answer worse when the additional information lowers the signal-to-noise ratio, introduces conflicts, hides decisive evidence, preserves stale state, or compresses away important conditions.\u003C\u002Fstrong> The relevant engineering objective is therefore not maximum context. It is \u003Cstrong>minimum sufficient context with preserved evidence and decision boundaries\u003C\u002Fstrong>.","Direct answer",{},{"id":237,"data":853,"type":234,"tunes":856},{"body":854,"title":855,"variant":241},"The Context Quality model and Context Pressure Test below are practical architecture methods proposed in this article, not formal industry standards. They synthesize established findings on long-context position effects, context pollution, compaction, retrieval, and context engineering.","About the model used in this article",{},{"id":244,"data":858,"type":42,"tunes":860},{"text":859,"level":219},"Context capacity is not context usability",{},{"id":249,"data":862,"type":226,"tunes":864},{"text":863},"A model's advertised context window describes how much input it can accept. It does not imply that every token inside that window receives equal attention or contributes equally to the final answer. The distinction matters because production systems increasingly fill context with conversation history, retrieved documents, tool results, memory, structured state, instructions, and intermediate artifacts.",{},{"id":254,"data":866,"type":226,"tunes":868},{"text":867},"The classic “Lost in the Middle” study showed that long-context models can perform worse when relevant evidence appears in the middle of a long input than when it appears near the beginning or end. The broader engineering lesson is not that long context is bad. It is that availability inside the context is not equivalent to reliable use.",{},{"id":259,"data":870,"type":226,"tunes":872},{"text":871},"OpenAI's context-management guidance reaches the same operational conclusion from another direction: even very large context windows can be overwhelmed by uncurated history, redundant tool output, and noisy retrieval. Anthropic likewise treats context as a finite resource that requires active engineering rather than passive accumulation.",{},{"id":264,"data":874,"type":42,"tunes":876},{"text":875,"level":219},"Five ways additional context can reduce answer quality",{},{"id":269,"data":878,"type":296,"tunes":904},{"content":879,"stretched":43,"withHeadings":14},[880,884,888,892,896,900],[881,882,883],"Failure mode","What changes when more context is added","Typical symptom",[885,886,887],"Signal dilution","Relevant evidence becomes a smaller fraction of the total input","The model gives a generic answer or misses the decisive passage",[889,890,891],"Evidence conflict","Different documents, versions, or memories disagree","The answer blends incompatible claims or chooses the wrong version",[893,894,895],"Position sensitivity","Decisive information moves into a less reliably used part of the context","The same evidence works in one ordering but fails in another",[897,898,899],"Stale-context persistence","Old state or prior conclusions remain present after reality changes","The model keeps repeating a formerly correct answer",[901,902,903],"Compression loss","Compaction or summarization removes qualifiers, exceptions, provenance, or unresolved uncertainty","The summary is coherent but the resulting answer becomes overconfident or overgeneralized",{},{"id":299,"data":906,"type":42,"tunes":908},{"text":907,"level":218},"1. Signal dilution: relevant evidence competes with everything else",{},{"id":304,"data":910,"type":226,"tunes":912},{"text":911},"Suppose a question can be answered from two short passages. A RAG system retrieves those passages plus eighteen loosely related ones “for safety.” Retrieval recall may improve, but the generator must now distinguish decisive evidence from background material. If similar phrases appear across several documents, the additional context can make the answer less precise.",{},{"id":309,"data":914,"type":226,"tunes":916},{"text":915},"This creates an important distinction between retrieval recall and context utility. More retrieved material can increase the probability that the answer exists somewhere in the context while simultaneously reducing the probability that the model gives the right evidence enough weight.",{},{"id":314,"data":918,"type":234,"tunes":921},{"body":919,"title":920,"variant":318},"Do not optimize top-k in isolation. Measure whether adding documents improves the final claim, preserves evidence attribution, and survives repeated trials.","Engineering rule",{},{"id":321,"data":923,"type":42,"tunes":925},{"text":924,"level":218},"2. Evidence conflict: more sources can mean more versions of reality",{},{"id":326,"data":927,"type":226,"tunes":929},{"text":928},"Long contexts often contain mutually inconsistent information: old and new API documentation, two policy versions, previous and current user preferences, competing web sources, cached state, or a model-generated summary that no longer matches the source.",{},{"id":331,"data":931,"type":226,"tunes":933},{"text":932},"The failure is not necessarily hallucination. The model may be faithfully combining contradictory evidence. The architecture therefore needs precedence rules: source authority, version, timestamp, jurisdiction, tenant, product revision, user state, or explicit supersession metadata.",{},{"id":336,"data":935,"type":226,"tunes":937},{"text":936},"Without those rules, increasing context can increase contradiction faster than it increases knowledge.",{},{"id":341,"data":939,"type":42,"tunes":941},{"text":940,"level":218},"3. Position sensitivity: where evidence appears can change the result",{},{"id":346,"data":943,"type":226,"tunes":945},{"text":944},"The “Lost in the Middle” results demonstrated that changing only the position of relevant information can materially change model performance. That finding is especially important for systems that concatenate many retrieved passages or long histories in a fixed order.",{},{"id":351,"data":947,"type":226,"tunes":949},{"text":948},"A production test should therefore vary document order, not merely test one canonical prompt. If the system answers correctly only when the decisive evidence is first or last, the application is more fragile than a single benchmark score suggests.",{},{"id":356,"data":951,"type":42,"tunes":953},{"text":952,"level":218},"4. Stale-context persistence: the model sees truth and history together",{},{"id":361,"data":955,"type":226,"tunes":957},{"text":956},"Long-running agents frequently carry earlier conclusions forward. That continuity is useful until a fact changes. If a tool result from yesterday says a deployment is healthy and a current tool result says it is degraded, both may remain in context unless the system explicitly replaces or scopes old state.",{},{"id":366,"data":959,"type":226,"tunes":961},{"text":960},"This is why current operational state should normally come from an authoritative source, while memory preserves durable context such as decisions, preferences, or procedures. More conversation history is not a substitute for re-reading the present.",{},{"id":371,"data":963,"type":42,"tunes":965},{"text":964,"level":218},"5. Compression loss: smaller context can also become worse context",{},{"id":376,"data":967,"type":226,"tunes":969},{"text":968},"The opposite intervention — compressing context — also has failure modes. Summaries can drop exceptions, unresolved questions, provenance, precise identifiers, negative evidence, or the conditions under which a conclusion was valid.",{},{"id":381,"data":971,"type":226,"tunes":973},{"text":972},"Microsoft Research's Agentic Context Engineering work describes a related problem as brevity bias and context collapse: iterative rewriting can remove useful domain detail. The objective is therefore not “compress as much as possible.” It is to reduce context while preserving the information that changes decisions.",{},{"id":386,"data":975,"type":42,"tunes":977},{"text":976,"level":219},"The Context Quality model",{},{"id":391,"data":979,"type":226,"tunes":981},{"text":980},"A useful context can be evaluated across six dimensions. None of them is simply token count.",{},{"id":396,"data":983,"type":435,"tunes":1011},{"rows":984,"title":1003,"layout":296,"columns":1004},[985,988,991,994,997,1000],{"id":400,"label":986,"values":987},"Relevance",[403,403,403],{"id":405,"label":989,"values":990},"Authority",[403,403,403],{"id":409,"label":992,"values":993},"Freshness",[403,403,403],{"id":413,"label":995,"values":996},"Consistency",[403,403,403],{"id":417,"label":998,"values":999},"Decision completeness",[403,403,403],{"id":421,"label":1001,"values":1002},"Traceability",[403,403,403],"Six dimensions of context quality",[1005,1007,1009],{"id":427,"label":1006},"Dimension",{"id":430,"label":1008},"Question",{"id":433,"label":1010},"If weak",{},{"id":438,"data":1013,"type":234,"tunes":1016},{"body":1014,"title":1015,"variant":442},"The best context is not the largest context. It is the \u003Cstrong>smallest context that still preserves the evidence, constraints, state, exceptions, and provenance required for a reliable answer or action\u003C\u002Fstrong>.","Target state",{},{"id":445,"data":1018,"type":42,"tunes":1020},{"text":1019,"level":219},"The Context Pressure Test",{},{"id":450,"data":1022,"type":226,"tunes":1024},{"text":1023},"To determine whether an application benefits from more context, test context size as an experimental variable instead of assuming that larger is better.",{},{"id":455,"data":1026,"type":483,"tunes":1053},{"steps":1027,"title":1052,"orientation":482},[1028,1031,1034,1037,1040,1043,1046,1049],{"label":1029,"description":1030},"1. Define a gold case","Choose a task with a known answer and a known minimal evidence set.",{"label":1032,"description":1033},"2. Run minimal sufficient context","Provide only the instructions, current state, and evidence necessary for the answer.",{"label":1035,"description":1036},"3. Add relevant background","Add useful but non-decisive context and measure whether quality improves, stays stable, or falls.",{"label":1038,"description":1039},"4. Add realistic noise","Add loosely related history, tool output, or retrieved passages that a production system might include.",{"label":1041,"description":1042},"5. Add controlled conflicts","Introduce stale or contradictory evidence with clear version metadata and verify that the correct source still wins.",{"label":1044,"description":1045},"6. Reorder decisive evidence","Place the key information near the beginning, middle, and end to test position sensitivity.",{"label":1047,"description":1048},"7. Test compaction","Replace older context with a summary and verify that qualifiers, provenance, unresolved issues, and decision boundaries survive.",{"label":1050,"description":1051},"8. Compare the quality curve","Measure correctness, evidence use, consistency, latency, cost, and variance as context changes.","Context Pressure Test",{},{"id":486,"data":1055,"type":42,"tunes":1057},{"text":1056,"level":219},"What to measure instead of token count",{},{"id":491,"data":1059,"type":296,"tunes":1088},{"content":1060,"stretched":43,"withHeadings":14},[1061,1064,1067,1070,1073,1076,1079,1082,1085],[1062,1063],"Metric","What it reveals",[1065,1066],"Answer correctness","Whether the final result is right",[1068,1069],"Claim-level evidence support","Whether material claims remain grounded as context changes",[1071,1072],"Evidence utilization","Whether the answer follows the decisive evidence instead of prior model knowledge",[1074,1075],"Conflict resolution accuracy","Whether current \u002F authoritative evidence wins over stale or weaker sources",[1077,1078],"Position robustness","Whether reordering evidence changes correctness",[1080,1081],"Compaction retention","Whether summaries preserve constraints, exceptions, identifiers, provenance, and unresolved state",[1083,1084],"Output variance across trials","Whether additional context makes the system less stable",[1086,1087],"Latency and token cost","Whether the added information produces enough quality to justify its operational cost",{},{"id":523,"data":1090,"type":42,"tunes":1092},{"text":1091,"level":219},"RAG: why increasing top-k can hurt",{},{"id":528,"data":1094,"type":226,"tunes":1096},{"text":1095},"A common RAG tuning pattern is to increase top-k when the system misses an answer. This can improve candidate recall but also increase irrelevant context, duplicate evidence, outdated passages, and conflicting documents.",{},{"id":533,"data":1098,"type":226,"tunes":1100},{"text":1099},"The better question is whether the decisive evidence is missing from retrieval or merely losing influence after context assembly. If the correct passage already appears in the candidate set, increasing top-k may solve the wrong problem.",{},{"id":538,"data":1102,"type":544,"tunes":1107},{"url":1103,"title":1104,"excerpt":1105,"ctaLabel":1106},"https:\u002F\u002Fstajic.de\u002Fblog\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method","RAG Failed — But Which Layer Actually Failed? A Diagnostic Method","A layer-by-layer method for isolating source coverage, retrieval, ranking, context assembly, generation, evidence attribution, and freshness failures.","Read the RAG diagnostic method",{},{"id":547,"data":1109,"type":42,"tunes":1111},{"text":1110,"level":219},"Long-running agents: continuity is not accumulation",{},{"id":552,"data":1113,"type":226,"tunes":1115},{"text":1114},"An agent needs continuity across steps, but continuity does not require replaying every prior token. OpenAI demonstrates trimming and compression for long-running session context. Anthropic recommends compaction, structured note-taking, and other techniques to preserve useful information while controlling context pollution.",{},{"id":557,"data":1117,"type":226,"tunes":1119},{"text":1118},"A strong long-running architecture usually separates durable memory, current state, external artifacts, retrieval, and model-facing context. That allows the system to preserve what matters without forcing every historical detail into every inference.",{},{"id":562,"data":1121,"type":544,"tunes":1126},{"url":1122,"title":1123,"excerpt":1124,"ctaLabel":1125},"https:\u002F\u002Fstajic.de\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context","AI Agent Memory Is Not RAG: How to Separate Memory, Retrieval, State and Context","A practical four-layer architecture for separating what persists, what is authoritative now, what is retrieved, and what the model actually receives.","Read the memory architecture article",{},{"id":570,"data":1128,"type":42,"tunes":1130},{"text":1129,"level":219},"Context order should be intentional",{},{"id":575,"data":1132,"type":226,"tunes":1134},{"text":1133},"Context construction is an information architecture problem. Critical instructions, current state, decisive evidence, and task-specific constraints should not be placed arbitrarily. When systems concatenate sources mechanically, they implicitly delegate prioritization to positional effects and model attention.",{},{"id":580,"data":1136,"type":226,"tunes":1138},{"text":1137},"There is no universal best ordering for every model and task, so ordering should be evaluated empirically. A useful test suite randomizes or systematically varies document position and measures whether the same claim remains stable.",{},{"id":585,"data":1140,"type":42,"tunes":1142},{"text":1141,"level":219},"Preserve decision boundaries during compaction",{},{"id":590,"data":1144,"type":226,"tunes":1146},{"text":1145},"A summary that says “use approach X” is weaker than a summary that preserves why X was chosen and what would invalidate the decision. Context compaction should retain the variables that can change the answer: version, date, assumptions, state, authority, unresolved disagreement, and evidence provenance.",{},{"id":595,"data":1148,"type":226,"tunes":1150},{"text":1149},"This connects context engineering directly to answer validity. If compaction preserves a conclusion but removes its validity boundary, future responses can remain internally consistent while becoming externally wrong.",{},{"id":600,"data":1152,"type":544,"tunes":1157},{"url":1153,"title":1154,"excerpt":1155,"ctaLabel":1156},"https:\u002F\u002Fstajic.de\u002Fblog\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers","The Answer Validity Boundary: The Missing Layer Between Relevance and Reliable AI Answers","A framework for making explicit the conditions under which an AI claim applies and what changes require restriction, recalculation, or abandonment.","Read the Answer Validity Boundary",{},{"id":608,"data":1159,"type":42,"tunes":1161},{"text":1160,"level":219},"A practical context construction policy",{},{"id":613,"data":1163,"type":628,"tunes":1176},{"meta":1164,"items":1165,"style":627},{},[1166,1167,1168,1169,1170,1171,1172,1173,1174,1175],"Start from the current task, not from everything the system knows.","Re-read volatile state from authoritative systems before consequential decisions.","Retrieve evidence for the current question instead of carrying large static corpora forward.","Remove duplicate or low-value tool output.","Keep source version, timestamp, authority, and provenance with important evidence.","Make precedence explicit when current and historical information conflict.","Preserve rules together with their exceptions and prerequisites.","Store durable decisions and reusable procedures outside the immediate context when they do not need verbatim replay.","Compact history only with tests for constraint, identifier, exception, and provenance retention.","Evaluate context size, ordering, and noise with repeated trials rather than a single prompt.",{},{"id":631,"data":1178,"type":42,"tunes":1180},{"text":1179,"level":219},"What would change this answer?",{},{"id":636,"data":1182,"type":226,"tunes":1184},{"text":1183},"The trade-off changes with model architecture, training, task type, and context length. Future models may become substantially more robust to position, noise, and conflicting information. A task with a small clean corpus can also benefit from simply providing the complete source rather than building an elaborate retrieval pipeline.",{},{"id":641,"data":1186,"type":226,"tunes":1188},{"text":1187},"The recommendation also changes when omission is more dangerous than noise. In high-recall research or discovery tasks, a larger candidate context may be justified before a later filtering or synthesis stage. In latency-sensitive production systems, stricter context selection may be preferable.",{},{"id":646,"data":1190,"type":226,"tunes":1192},{"text":1191},"The core principle would change only if models became reliably invariant to irrelevant information, position, contradiction, and stale evidence. Until then, context should be treated as a curated execution resource rather than passive storage.",{},{"id":651,"data":1194,"type":42,"tunes":1196},{"text":1195,"level":219},"Limitations",{},{"id":656,"data":1198,"type":226,"tunes":1200},{"text":1199},"Long-context behaviour varies considerably across models and workloads. The original “Lost in the Middle” experiments used earlier generations of models, so their exact effect sizes should not be assumed to represent current systems. The finding remains useful as a failure pattern to test, not as a universal fixed performance curve.",{},{"id":661,"data":1202,"type":226,"tunes":1204},{"text":1203},"Likewise, reducing context can remove necessary evidence. Compaction introduces summarization risk, and aggressive retrieval filtering can reduce recall. The objective is not minimal tokens at any cost; it is sufficient, current, traceable context for the decision being made.",{},{"id":666,"data":1206,"type":42,"tunes":1208},{"text":1207,"level":219},"Conclusion",{},{"id":671,"data":1210,"type":226,"tunes":1212},{"text":1211},"The question “How much context can the model accept?” is less useful than “How much of this context improves the decision?” More tokens can add evidence, but they can also add distraction, contradiction, stale state, positional fragility, and compression debt.",{},{"id":676,"data":1214,"type":226,"tunes":1216},{"text":1215},"Treat context as an engineered working set. Start with minimum sufficient evidence. Add information only when it improves measured performance. Test noise, conflict, ordering, and compaction explicitly. A large context window is capacity; context quality is architecture.",{},{"id":681,"data":1218,"type":42,"tunes":1220},{"text":1219,"level":219},"FAQ",{},{"id":686,"data":1222,"type":686,"tunes":1240},{"items":1223,"title":1239},[1224,1227,1230,1233,1236],{"id":690,"answer":1225,"question":1226},"Yes. Additional context can dilute relevant evidence, introduce contradictory or stale information, move decisive evidence into less robust positions, and increase the chance that the model uses weak rather than decisive signals.","Can giving an AI model more context make its answer worse?",{"id":694,"answer":1228,"question":1229},"Not generally. A larger context window increases capacity, but retrieval still helps select current and relevant information, control cost, preserve source boundaries, and avoid sending large amounts of unrelated data into every request.","Does a larger context window eliminate the need for RAG?",{"id":698,"answer":1231,"question":1232},"It describes observed cases where language models use relevant information less reliably when that information is located in the middle of a long context than when it appears near the beginning or end. The exact effect varies by model and task and should be tested on current systems.","What is the Lost in the Middle problem?",{"id":702,"answer":1234,"question":1235},"No. If relevant evidence is missing from the candidate set, a larger top-k may improve recall. If the evidence is already present but gets diluted by additional material, increasing top-k can make the context worse. Diagnose retrieval and context assembly separately.","Should I always reduce RAG top-k?",{"id":706,"answer":1237,"question":1238},"Preserve durable decisions, current goals, unresolved issues, identifiers, constraints, exceptions, evidence provenance, and the conditions that would change an earlier conclusion.","What should a context summary preserve?","Long context and AI answer quality",{},{"id":712,"data":1242,"type":42,"tunes":1244},{"text":1243,"level":219},"Glossary",{},{"id":717,"data":1246,"type":717,"tunes":1265},{"title":1247,"entries":1248},"Key context-engineering terms",[1249,1252,1255,1257,1260,1262],{"term":1250,"anchor":723,"definition":1251},"Context window","The amount of input and output token information a model can attend to within one inference sequence.",{"term":1253,"anchor":727,"definition":1254},"Context pollution","Degradation caused by irrelevant, stale, redundant, conflicting, or otherwise low-value information occupying model context.",{"term":885,"anchor":730,"definition":1256},"A reduction in the relative prominence of decisive evidence as additional low-value or competing information is added.",{"term":1258,"anchor":734,"definition":1259},"Context compaction","Reducing an accumulated context by summarizing, restructuring, externalizing, or otherwise preserving essential information in a smaller working representation.",{"term":1077,"anchor":738,"definition":1261},"The degree to which model performance remains stable when relevant information appears in different positions inside the context.",{"term":1263,"anchor":742,"definition":1264},"Minimum sufficient context","The smallest practical working context that still preserves the evidence, state, constraints, exceptions, and provenance required for reliable execution.",{},{"id":746,"data":1267,"type":42,"tunes":1269},{"text":1268,"level":219},"Primary sources and further reading",{},{"id":751,"data":1271,"type":758,"tunes":1276},{"link":753,"meta":1272},{"image":1273,"title":1274,"description":1275},{"url":403},"OpenAI — Context Engineering: Short-Term Memory Management with Sessions","Guidance on trimming and compression, with discussion of distraction, inefficiency, stale context, noisy retrieval, and long-running sessions.",{},{"id":761,"data":1278,"type":758,"tunes":1283},{"link":763,"meta":1279},{"image":1280,"title":1281,"description":1282},{"url":403},"Anthropic — Effective Context Engineering for AI Agents","Engineering guidance on context pollution, compaction, structured note-taking, and long-horizon agent context management.",{},{"id":770,"data":1285,"type":758,"tunes":1290},{"link":772,"meta":1286},{"image":1287,"title":1288,"description":1289},{"url":403},"Liu et al. — Lost in the Middle: How Language Models Use Long Contexts","TACL paper showing position-sensitive use of relevant information in long contexts and motivating explicit long-context robustness tests.",{},{"id":779,"data":1292,"type":758,"tunes":1297},{"link":781,"meta":1293},{"image":1294,"title":1295,"description":1296},{"url":403},"Microsoft Research — Agentic Context Engineering (ACE)","Research on evolving structured contexts while addressing brevity bias and context collapse.",{},{"id":788,"data":1299,"type":758,"tunes":1304},{"link":790,"meta":1300},{"image":1301,"title":1302,"description":1303},{"url":403},"OpenAI — Evaluation Best Practices","Guidance on testing edge cases including long context and long-running conversations using explicit, repeatable evals.",{},"2.31.6","A larger context window does not guarantee a better answer. This article explains how signal dilution, conflicting evidence, stale state, position sensitivity, and lossy compression can reduce AI reliability—and introduces a practical Context Pressure Test.",{"lang":7,"title":208,"content":210,"contentJson":1308,"excerpt":797},{"time":212,"blocks":1309,"version":796},[1310,1313,1316,1319,1322,1325,1328,1331,1334,1337,1347,1350,1353,1356,1359,1362,1365,1368,1371,1374,1377,1380,1383,1386,1389,1392,1395,1398,1401,1404,1424,1427,1430,1433,1445,1448,1461,1464,1467,1470,1473,1476,1479,1482,1485,1488,1491,1494,1497,1500,1503,1506,1509,1514,1517,1520,1523,1526,1529,1532,1535,1538,1541,1544,1547,1556,1559,1569,1572,1577,1582,1587,1592],{"id":215,"data":1311,"type":220,"tunes":1312},{"title":217,"maxLevel":218,"minLevel":219},{},{"id":223,"data":1314,"type":226,"tunes":1315},{"text":225},{},{"id":229,"data":1317,"type":234,"tunes":1318},{"body":231,"title":232,"variant":233},{},{"id":237,"data":1320,"type":234,"tunes":1321},{"body":239,"title":240,"variant":241},{},{"id":244,"data":1323,"type":42,"tunes":1324},{"text":246,"level":219},{},{"id":249,"data":1326,"type":226,"tunes":1327},{"text":251},{},{"id":254,"data":1329,"type":226,"tunes":1330},{"text":256},{},{"id":259,"data":1332,"type":226,"tunes":1333},{"text":261},{},{"id":264,"data":1335,"type":42,"tunes":1336},{"text":266,"level":219},{},{"id":269,"data":1338,"type":296,"tunes":1346},{"content":1339,"stretched":43,"withHeadings":14},[1340,1341,1342,1343,1344,1345],[273,274,275],[277,278,279],[281,282,283],[285,286,287],[289,290,291],[293,294,295],{},{"id":299,"data":1348,"type":42,"tunes":1349},{"text":301,"level":218},{},{"id":304,"data":1351,"type":226,"tunes":1352},{"text":306},{},{"id":309,"data":1354,"type":226,"tunes":1355},{"text":311},{},{"id":314,"data":1357,"type":234,"tunes":1358},{"body":316,"title":317,"variant":318},{},{"id":321,"data":1360,"type":42,"tunes":1361},{"text":323,"level":218},{},{"id":326,"data":1363,"type":226,"tunes":1364},{"text":328},{},{"id":331,"data":1366,"type":226,"tunes":1367},{"text":333},{},{"id":336,"data":1369,"type":226,"tunes":1370},{"text":338},{},{"id":341,"data":1372,"type":42,"tunes":1373},{"text":343,"level":218},{},{"id":346,"data":1375,"type":226,"tunes":1376},{"text":348},{},{"id":351,"data":1378,"type":226,"tunes":1379},{"text":353},{},{"id":356,"data":1381,"type":42,"tunes":1382},{"text":358,"level":218},{},{"id":361,"data":1384,"type":226,"tunes":1385},{"text":363},{},{"id":366,"data":1387,"type":226,"tunes":1388},{"text":368},{},{"id":371,"data":1390,"type":42,"tunes":1391},{"text":373,"level":218},{},{"id":376,"data":1393,"type":226,"tunes":1394},{"text":378},{},{"id":381,"data":1396,"type":226,"tunes":1397},{"text":383},{},{"id":386,"data":1399,"type":42,"tunes":1400},{"text":388,"level":219},{},{"id":391,"data":1402,"type":226,"tunes":1403},{"text":393},{},{"id":396,"data":1405,"type":435,"tunes":1423},{"rows":1406,"title":424,"layout":296,"columns":1419},[1407,1409,1411,1413,1415,1417],{"id":400,"label":401,"values":1408},[403,403,403],{"id":405,"label":406,"values":1410},[403,403,403],{"id":409,"label":410,"values":1412},[403,403,403],{"id":413,"label":414,"values":1414},[403,403,403],{"id":417,"label":418,"values":1416},[403,403,403],{"id":421,"label":422,"values":1418},[403,403,403],[1420,1421,1422],{"id":427,"label":428},{"id":430,"label":431},{"id":433,"label":434},{},{"id":438,"data":1425,"type":234,"tunes":1426},{"body":440,"title":441,"variant":442},{},{"id":445,"data":1428,"type":42,"tunes":1429},{"text":447,"level":219},{},{"id":450,"data":1431,"type":226,"tunes":1432},{"text":452},{},{"id":455,"data":1434,"type":483,"tunes":1444},{"steps":1435,"title":447,"orientation":482},[1436,1437,1438,1439,1440,1441,1442,1443],{"label":459,"description":460},{"label":462,"description":463},{"label":465,"description":466},{"label":468,"description":469},{"label":471,"description":472},{"label":474,"description":475},{"label":477,"description":478},{"label":480,"description":481},{},{"id":486,"data":1446,"type":42,"tunes":1447},{"text":488,"level":219},{},{"id":491,"data":1449,"type":296,"tunes":1460},{"content":1450,"stretched":43,"withHeadings":14},[1451,1452,1453,1454,1455,1456,1457,1458,1459],[495,496],[498,499],[501,502],[504,505],[507,508],[510,511],[513,514],[516,517],[519,520],{},{"id":523,"data":1462,"type":42,"tunes":1463},{"text":525,"level":219},{},{"id":528,"data":1465,"type":226,"tunes":1466},{"text":530},{},{"id":533,"data":1468,"type":226,"tunes":1469},{"text":535},{},{"id":538,"data":1471,"type":544,"tunes":1472},{"url":540,"title":541,"excerpt":542,"ctaLabel":543},{},{"id":547,"data":1474,"type":42,"tunes":1475},{"text":549,"level":219},{},{"id":552,"data":1477,"type":226,"tunes":1478},{"text":554},{},{"id":557,"data":1480,"type":226,"tunes":1481},{"text":559},{},{"id":562,"data":1483,"type":544,"tunes":1484},{"url":564,"title":565,"excerpt":566,"ctaLabel":567},{},{"id":570,"data":1486,"type":42,"tunes":1487},{"text":572,"level":219},{},{"id":575,"data":1489,"type":226,"tunes":1490},{"text":577},{},{"id":580,"data":1492,"type":226,"tunes":1493},{"text":582},{},{"id":585,"data":1495,"type":42,"tunes":1496},{"text":587,"level":219},{},{"id":590,"data":1498,"type":226,"tunes":1499},{"text":592},{},{"id":595,"data":1501,"type":226,"tunes":1502},{"text":597},{},{"id":600,"data":1504,"type":544,"tunes":1505},{"url":602,"title":603,"excerpt":604,"ctaLabel":605},{},{"id":608,"data":1507,"type":42,"tunes":1508},{"text":610,"level":219},{},{"id":613,"data":1510,"type":628,"tunes":1513},{"meta":1511,"items":1512,"style":627},{},[617,618,619,620,621,622,623,624,625,626],{},{"id":631,"data":1515,"type":42,"tunes":1516},{"text":633,"level":219},{},{"id":636,"data":1518,"type":226,"tunes":1519},{"text":638},{},{"id":641,"data":1521,"type":226,"tunes":1522},{"text":643},{},{"id":646,"data":1524,"type":226,"tunes":1525},{"text":648},{},{"id":651,"data":1527,"type":42,"tunes":1528},{"text":653,"level":219},{},{"id":656,"data":1530,"type":226,"tunes":1531},{"text":658},{},{"id":661,"data":1533,"type":226,"tunes":1534},{"text":663},{},{"id":666,"data":1536,"type":42,"tunes":1537},{"text":668,"level":219},{},{"id":671,"data":1539,"type":226,"tunes":1540},{"text":673},{},{"id":676,"data":1542,"type":226,"tunes":1543},{"text":678},{},{"id":681,"data":1545,"type":42,"tunes":1546},{"text":683,"level":219},{},{"id":686,"data":1548,"type":686,"tunes":1555},{"items":1549,"title":709},[1550,1551,1552,1553,1554],{"id":690,"answer":691,"question":692},{"id":694,"answer":695,"question":696},{"id":698,"answer":699,"question":700},{"id":702,"answer":703,"question":704},{"id":706,"answer":707,"question":708},{},{"id":712,"data":1557,"type":42,"tunes":1558},{"text":714,"level":219},{},{"id":717,"data":1560,"type":717,"tunes":1568},{"title":719,"entries":1561},[1562,1563,1564,1565,1566,1567],{"term":722,"anchor":723,"definition":724},{"term":726,"anchor":727,"definition":728},{"term":277,"anchor":730,"definition":731},{"term":733,"anchor":734,"definition":735},{"term":737,"anchor":738,"definition":739},{"term":741,"anchor":742,"definition":743},{},{"id":746,"data":1570,"type":42,"tunes":1571},{"text":748,"level":219},{},{"id":751,"data":1573,"type":758,"tunes":1576},{"link":753,"meta":1574},{"image":1575,"title":756,"description":757},{"url":403},{},{"id":761,"data":1578,"type":758,"tunes":1581},{"link":763,"meta":1579},{"image":1580,"title":766,"description":767},{"url":403},{},{"id":770,"data":1583,"type":758,"tunes":1586},{"link":772,"meta":1584},{"image":1585,"title":775,"description":776},{"url":403},{},{"id":779,"data":1588,"type":758,"tunes":1591},{"link":781,"meta":1589},{"image":1590,"title":784,"description":785},{"url":403},{},{"id":788,"data":1593,"type":758,"tunes":1596},{"link":790,"meta":1594},{"image":1595,"title":793,"description":794},{"url":403},{},"Post erfolgreich abgerufen",{"items":1599,"source":1663,"manualIds":1664,"manualMatchedIds":1665},[1600,1607,1614,1621,1628,1635,1642,1649,1656],{"id":1601,"slug":1602,"title":1603,"excerpt":1604,"featuredImage":1605,"publishedAt":1606},"477","computer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system","计算机使用代理：为什么成功的演示仍可能是一个不可靠的系统","计算机使用代理如今能够完成令人印象深刻的浏览器和桌面工作流程，但一次成功的运行证明的是能力——而非可靠性。本文展示了如何测试可重复性、环境鲁棒性、长时程控制、状态感知、结果验证以及安全的目标处理。","\u002Fuploads\u002F2026\u002F09\u002Fcomputer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system-1790352854690-75qnrg.webp","2026-09-25T12:13:00.000Z",{"id":1608,"slug":1609,"title":1610,"excerpt":1611,"featuredImage":1612,"publishedAt":1613},"466","the-gpu-is-not-the-product-future-proof-private-ai-architecture","GPU 不是产品：面向未来的私有 AI 架构","私有 AI 基础设施不应围绕单一 GPU 或单一模型来设计。更具韧性的做法是将快速推理 GPU、内存充裕的 AI 系统、物理 AI 节点以及可选的前沿云模型，统一置于一个具备能力感知的路由层之后。","\u002Fuploads\u002F2026\u002F09\u002Fthe-gpu-is-not-the-product-future-proof-private-ai-architecture-1790140878812-8hsl39.webp","2026-09-23T01:19:00.000Z",{"id":1615,"slug":1616,"title":1617,"excerpt":1618,"featuredImage":1619,"publishedAt":1620},"457","should-you-buy-5g-openwrt-router-old-firmware","你应该购买带有旧固件的5G OpenWrt路由器吗？以ZBT Z8102AX为例","购买搭载旧版固件的5G OpenWrt路由器在特定条件下是合理的。ZBT Z8102AX型号清晰展现了利弊两面：硬件实用、调制解调器工作正常，测试中路由器保持稳定，但OpenWrt 21.02版本、简陋的包装以及不明确的升级路径，要求消费者在购买时需审慎决策。","\u002Fuploads\u002F2026\u002F06\u002Fopenwrt-router-review-dual-sim-05-1781620596218-5ldld4.webp","2026-06-16T10:41:00.000Z",{"id":1622,"slug":1623,"title":1624,"excerpt":1625,"featuredImage":1626,"publishedAt":1627},"364","tipps-fuer-die-verbesserung-der-seo-suchmaschinenoptimierung","Mastering the SEO Workflow: Essential Optimization Strategies for Organic Growth","A structured SEO workflow is crucial for sustainable organic growth. Learn the ten foundational strategies, from keyword research and technical optimization to content quality and performance analysis.","\u002Fuploads\u002F2026\u002F03\u002Ftipps-fuer-die-verbesserung-der-seo-suchmaschinenoptimierung-1774866098131-hwkzrg.webp","2024-01-26T06:35:00.000Z",{"id":1629,"slug":1630,"title":1631,"excerpt":1632,"featuredImage":1633,"publishedAt":1634},"363","front-und-backend-entwicklung","前端与后端开发","前端和后端开发是网络开发的重要组成部分，涉及创建网络应用程序和网站。前端开发专注于用户界面，而后端开发则负责编程和管理服务器端。","\u002Fuploads\u002F2026\u002F03\u002Ffront-und-backend-entwicklung-1774872219531-wyu4i1.webp","2023-04-12T11:11:00.000Z",{"id":1636,"slug":1637,"title":1638,"excerpt":1639,"featuredImage":1640,"publishedAt":1641},"471","how-to-know-whether-an-ai-agent-actually-used-the-right-evidence","如何判断一个AI智能体是否真正使用了正确的证据","AI代理可以引用来源，却仍然使用错误的证据。本文介绍一种实用方法，用于核查主张支持、来源权威性、适用性、出处，以及证据是否实际影响了答案。","\u002Fuploads\u002F2026\u002F09\u002Fhow-to-know-whether-an-ai-agent-actually-used-the-right-evidence-1790351317188-o5z9ve.webp","2026-09-25T11:47:00.000Z",{"id":1643,"slug":1644,"title":1645,"excerpt":1646,"featuredImage":1647,"publishedAt":1648},"478","what-is-rag-the-simplest-explanation-of-how-it-works","什么是RAG？对其工作原理的最简单解释","RAG听起来很复杂，但想法很简单：在AI回答之前，它先从知识源查找有用的信息，并将该信息提供给语言模型。本指南使用一个简单的思维模型来解释RAG、LLM、状态、记忆和工具。","\u002Fuploads\u002F2026\u002F09\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works-1790377492124-khjagt.webp","2026-09-25T19:03:00.000Z",{"id":1650,"slug":1651,"title":1652,"excerpt":1653,"featuredImage":1654,"publishedAt":1655},"467","the-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers","答案有效性边界：相关性到可靠AI答案之间缺失的层级","一个来源可能相关、权威，但对于所提出的问题仍然是错误的。缺失的层次是适用性：答案成立的条件，以及迫使其被重新考虑的变化。本文介绍了“答案有效性边界”这一面向人类、AI搜索和RAG系统的来源设计模式。","\u002Fuploads\u002F2026\u002F09\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers-1790272901306-1g5jly.webp","2026-09-24T11:59:00.000Z",{"id":1657,"slug":1658,"title":1659,"excerpt":1660,"featuredImage":1661,"publishedAt":1662},"468","ai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context","AI代理记忆不是RAG：如何区分记忆、检索、状态和上下文","代理记忆、RAG、状态和上下文经常被当作可以互换的概念来使用。它们并不是。这个实用的架构模型将这四个层次区分开来，展示了每一层各自应处的位置，并解释了当系统将它们合并为一层时会出现什么问题。","\u002Fuploads\u002F2026\u002F09\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context-1790350560308-np0xy6.webp","2026-09-25T11:34:00.000Z","fallback",[],[]]