[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"portal-settings:stajic:zh":3,"post:ai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context:zh":38,"related:post:ai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context:zh:1":1333,"public-menus:all":1408},{"statusCode":4,"data":5,"message":37},200,{"tenantId":6,"lang":7,"defaultLang":8,"siteUrl":9,"contactEmail":10,"brandName":11,"logoUrl":12,"siteName":11,"siteDescription":13,"ogImage":10,"robotsIndex":14,"socialLinks":10,"reservedSlugs":10,"seoPolicy":15},"stajic","zh","de","https:\u002F\u002Fstajic.de",null,"Stajic Platform","\u002FLogo_Planet.svg","Stajic Portal",true,{"branding":16,"relatedContent":17,"crossDomainLinks":18},{"logoUrl":12},{"enabled":14},[19,22,25,28,31,34],{"url":20,"label":21,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Ffigure.rocks","figure.rocks",{"url":23,"label":24,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Floving.rocks","loving.rocks",{"url":26,"label":27,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.com","bazify.com",{"url":29,"label":30,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.de","bazify.de",{"url":32,"label":33,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.at","bazify.at",{"url":35,"label":36,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.ba","bazify.ba","Portal settings resolved",{"statusCode":4,"data":39,"message":1332},{"id":40,"title":41,"slug":42,"content":43,"contentJson":44,"excerpt":588,"featuredImage":589,"featuredImageAlt":590,"featuredImageCaption":10,"featuredImageTitle":10,"featuredImageCopyright":10,"featuredImageAuthor":10,"featuredImageSourceUrl":10,"featuredImageLicense":10,"featuredImageIsAiGenerated":126,"status":591,"publishedAt":592,"createdAt":593,"updatedAt":594,"seoLocalePaths":595,"categories":604,"author":617,"translations":622},"468","AI代理记忆不是RAG：如何区分记忆、检索、状态和上下文","ai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context","\u003Cnav class=\"editorjs-toc\" data-editorjs-toc=\"true\" aria-label=\"目录\">\u003Cstrong class=\"editorjs-toc__title\">目录\u003C\u002Fstrong>\u003Col class=\"editorjs-toc__list editorjs-toc__list--depth-0\">\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-5\" class=\"editorjs-toc__link\">类别错误：把所有看似持久的东西都当作记忆\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-8\" class=\"editorjs-toc__link\">四层架构：状态、记忆、检索、上下文\u003C\u002Fa>\u003Col class=\"editorjs-toc__list editorjs-toc__list--depth-1\">\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-10\" class=\"editorjs-toc__link\">1. 状态：现在什么是真的\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-13\" class=\"editorjs-toc__link\">2. 记忆：过去哪些内容应该持久保留\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-16\" class=\"editorjs-toc__link\">3. 检索：现在应该选择什么\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-19\" class=\"editorjs-toc__link\">4. 上下文：模型当前实际可以使用的内容\u003C\u002Fa>\u003C\u002Fli>\u003C\u002Fol>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-22\" class=\"editorjs-toc__link\">各层如何交互\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-24\" class=\"editorjs-toc__link\">为什么 RAG 不是记忆\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-28\" class=\"editorjs-toc__link\">四层分离测试\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-32\" class=\"editorjs-toc__link\">层被混同导致的失败模式\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-34\" class=\"editorjs-toc__link\">什么应该被记住、检索、重新计算或重新读取？\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-36\" class=\"editorjs-toc__link\">记忆系统需要写入策略，而不仅仅是检索策略\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-40\" class=\"editorjs-toc__link\">来源是记忆与可靠证据之间的桥梁\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-43\" class=\"editorjs-toc__link\">更多记忆并不意味着更多上下文\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-46\" class=\"editorjs-toc__link\">生产设计检查清单\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-48\" class=\"editorjs-toc__link\">什么会改变这个答案？\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-51\" class=\"editorjs-toc__link\">局限性\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-53\" class=\"editorjs-toc__link\">结论\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-56\" class=\"editorjs-toc__link\">常见问题\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-58\" class=\"editorjs-toc__link\">术语表\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-60\" class=\"editorjs-toc__link\">主要来源与延伸阅读\u003C\u002Fa>\u003C\u002Fli>\u003C\u002Fol>\u003C\u002Fnav>\n\u003Cp>AI智能体记忆、检索增强生成（RAG）、运行时状态和模型上下文经常被当作可以互换的概念来讨论。但它们并非如此。将它们混为一谈会使智能体系统更难推理、更难调试，也更容易变得过时或不安全。\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--info my-6 rounded-xl border p-5 border-blue-300 bg-blue-50 dark:border-blue-900 dark:bg-blue-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">直接回答\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">&lt;strong&gt;RAG不是智能体记忆。&lt;\u002Fstrong&gt;RAG是一种检索模式：它为当前模型调用选择可能有用的信息。记忆是从先前交互或经验中派生并在时间维度上被管理的持久信息。状态表示当前任务或环境的真实情况。上下文是实际提供给模型用于当前推理的信息。生产级智能体可能同时使用这四者，但它们解决的是不同的问题。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Caside class=\"editorjs-callout editorjs-callout--note my-6 rounded-xl border p-5 border-gray-300 bg-gray-50 dark:border-gray-700 dark:bg-gray-900\u002F40\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">关于本文所用模型\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">下文的四层分离是一种实用架构模型，而非正式的行业标准。供应商和研究论文使用的术语存在重叠。其目的是操作性的：使设计决策、归属划分、故障分析和测试更加清晰。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-5\">类别错误：把所有看似持久的东西都当作记忆\u003C\u002Fh2>\n\u003Cp>向量数据库可以存储对话片段。会话对象可以携带最近的轮次。数据库行可以保存当前工作流状态。摘要器可以压缩先前的工作。检索器可以获取旧证据。所有这些都可以让智能体看起来像是在“记住”，但它们并不具有相同的语义。\u003C\u002Fp>\n\u003Cp>这种区分很重要，因为所需的正确性规则不同。当前状态必须是权威且新鲜的。记忆需要写入、修订、遗忘和冲突处理的生命周期规则。检索需要相关性和证据选择质量。上下文需要令牌预算纪律以及对无关或冲突材料的防护。\u003C\u002Fp>\n\u003Ch2 id=\"section-8\">四层架构：状态、记忆、检索、上下文\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">层级\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">核心问题\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">典型示例\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">主要正确性关注点\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">状态\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">现在什么是真的？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">任务状态、购物车内容、工作流步骤、活动权限、当前游戏状态\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">新鲜度和权威性\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">记忆\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">过去哪些内容应该持久保留？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">用户偏好、先前决策、学到的约束、已解决的故障、持久项目事实\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">生命周期、修订、来源、遗忘\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">检索\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">现在应该选择哪些信息？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">向量搜索、关键词搜索、图查找、重排序、文档搜索\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">相关性和证据选择\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">上下文\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型在这次调用中看到什么？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">系统指令、当前请求、检索到的段落、工具结果、摘要\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">每令牌效用、排序、一致性、噪声\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch3 id=\"section-10\">1. 状态：现在什么是真的\u003C\u002Fh3>\n\u003Cp>状态属于运行中的系统，而不是模型的回忆。如果订单被取消、部署被暂停、用户失去权限，或任务从“进行中”变为“已批准”，权威值应来自拥有该事实的系统。\u003C\u002Fp>\n\u003Cp>一种危险的设计是让旧的对话摘要成为当前状态的替代品。智能体可能准确记得订单昨天是活跃的，但今天仍然是错的。因此，状态需要明确的所有权、相关情况下的版本控制或时间戳，以及在采取重大行动前重新读取真相来源的路径。\u003C\u002Fp>\n\u003Ch3 id=\"section-13\">2. 记忆：过去哪些内容应该持久保留\u003C\u002Fh3>\n\u003Cp>记忆不仅仅是“我们能存储的一切”。有用的记忆层决定什么值得持久保留、以什么形式、保留多久、具有什么来源，以及在什么条件下必须修订或删除。\u003C\u002Fp>\n\u003Cp>最近的智能体记忆研究越来越认为原始对话记录存储是不够的。微软的PlugMem工作专注于将原始交互历史转化为结构化、可复用的知识。Memora将丰富的存储内容与更轻量的抽象和检索线索分离，使长时程系统不必在细节和可扩展访问之间做出选择。\u003C\u002Fp>\n\u003Ch3 id=\"section-16\">3. 检索：现在应该选择什么\u003C\u002Fh3>\n\u003Cp>检索是一种选择机制。它可以搜索外部文档、内部知识库、存储的记忆、日志、图、数据库或混合来源。RAG通常位于此处：检索证据，将选定的材料放入模型的工作输入中，然后生成答案。\u003C\u002Fp>\n\u003Cp>仅仅因为检索语料库包含过去的交互，该机制并不会变成记忆。同一个检索器可以搜索智能体从未经历过的策略文档、来自另一个系统的产品数据，或用户的先前决策。检索描述的是信息如何被选择；记忆描述的是为什么某些信息会跨时间持久保留，以及这种持久性如何被管理。\u003C\u002Fp>\n\u003Ch3 id=\"section-19\">4. 上下文：模型当前实际可以使用的内容\u003C\u002Fh3>\n\u003Cp>上下文是面向模型的层。Anthropic 将上下文工程描述为决定何种上下文配置最有可能产生期望行为，其中上下文是模型在生成期间可用的 token。OpenAI 的会话记忆指南同样将裁剪和压缩视为面向长时间运行的智能体交互的上下文管理技术。\u003C\u002Fp>\n\u003Cp>这就是为什么一个系统可以拥有出色的记忆却仍然失败。相关记忆可能存在但未被检索到。它可能被检索到，但在上下文中被放置在更强的冲突文本旁边。它可能被压缩，直到决定性细节消失。或者模型可能接收到过多材料，以至于有用证据被噪声稀释。\u003C\u002Fp>\n\u003Ch2 id=\"section-22\">各层如何交互\u003C\u002Fh2>\n\u003Csection class=\"editorjs-process my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">一种可能的生产流程\u003C\u002Fh3>\u003Cdiv class=\"grid grid-cols-1 md:grid-cols-2 xl:grid-cols-3 gap-4\">\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">1\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">1. 读取权威状态\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">从拥有这些事实的系统中加载当前任务、用户、系统或环境事实。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">2\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">2. 识别记忆需求\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">确定先前的决策、偏好、经验教训或长期约束是否相关。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">3\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">3. 检索证据\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">使用语义、词汇、图、结构化或混合检索来搜索记忆和外部知识。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">4\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">4. 构建上下文\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">在模型可用上下文内组装指令、当前状态、选定证据和压缩后的历史。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">5\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">5. 生成或行动\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">模型基于组装后的上下文进行推理，并生成答案、计划或工具调用。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">6\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">6. 验证并写回\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">验证重要输出，在允许的情况下更新权威状态，并且只持久化通过写入策略的记忆。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-24\">为什么 RAG 不是记忆\u003C\u002Fh2>\n\u003Cp>最简单的测试是：RAG 系统可以检索智能体从未见过的信息。仅这一点就表明，检索和记忆是不同的抽象。\u003C\u002Fp>\n\u003Cp>RAG 回答的是：“我应该获取哪些证据？”记忆系统还必须额外回答诸如：“这个事件是否应该成为持久知识？”、“这个新信息是否取代了较旧的记忆？”、“这个记忆是否仍然可信？”、“谁被允许读取它？”以及“它应该在什么时候被遗忘？”\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--warning my-6 rounded-xl border p-5 border-amber-300 bg-amber-50 dark:border-amber-900 dark:bg-amber-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">一个常见的设计陷阱\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">如果每一轮对话都被嵌入向量存储，并在之后通过相似度检索，那么系统拥有的是持久查找，但不一定拥有治理良好的记忆架构。仅凭持久化并不能定义记忆质量。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-28\">四层分离测试\u003C\u002Fh2>\n\u003Cp>当某个功能被称为“记忆”时，请问以下四个问题。答案通常会揭示实际涉及的是哪一层。\u003C\u002Fp>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">问题\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">如果答案是肯定的，你主要处理的是\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">这是否代表任务或环境的当前权威状况？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">状态\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">这些信息是否必须跨越当前运行而保留，因为它捕捉了有用的先前经验、偏好或决策？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">记忆\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">主要问题是否是决定哪些已存储或外部信息与当前请求相关？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">检索\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">主要问题是否是决定将哪些信息放入当前模型调用中？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">上下文\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Cp>单个组件可以参与不止一层。数据库可以同时存储状态和记忆。向量索引可以同时检索外部知识和记忆。这种分离是语义上的，不一定是物理上的。\u003C\u002Fp>\n\u003Ch2 id=\"section-32\">层被混同导致的失败模式\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">失败模式\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">发生了什么\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">结果\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">陈旧状态伪装成记忆\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">信任旧摘要，而不是重新读取权威系统\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">智能体基于曾经为真的事实行动\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">记忆被当作不可变事实\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">先前的偏好或决策被存储，却没有修订规则\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">已被取代的信息继续影响未来答案\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">检索命中被当作真相\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">高相似度被误认为事实权威\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">看起来相关但不正确的证据占据主导\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">上下文过载\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">注入过多检索段落、记忆、日志和指令\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">决定性证据被稀释或被矛盾信息抵消\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">不受控的记忆写入\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型生成的解释被自动存储为持久记忆\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">错误变得持久并自我强化\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">没有来源边界\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">系统无法区分用户陈述、来源事实、模型推断和生成摘要\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">后续检索丢失信息的证据地位\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-34\">什么应该被记住、检索、重新计算或重新读取？\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">信息类型\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">首选处理方式\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">原因\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">当前权限、订单状态、库存、工作流状态\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">重新读取权威状态\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">新鲜度比回忆更重要\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">用户明确提供的稳定用户偏好\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">记忆，并具备编辑\u002F删除语义\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">跨会话有用，且归用户所有\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">长期项目期间做出的决策\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">带时间戳、来源和取代规则的记忆\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">历史很重要，但决策可能改变\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">产品规格或公共政策文档\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">从来源检索\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">外部知识应与其证据保持关联\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">可以低成本重新计算的派生指标\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">重新计算\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">避免持久化陈旧的派生值\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">冗长的原始工具输出\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">外部存储；需要时检索或摘要\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">不要永久占用上下文\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型假设或不确定解释\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">不要自动提升为持久记忆\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">推断不等同于事实\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-36\">记忆系统需要写入策略，而不仅仅是检索策略\u003C\u002Fh2>\n\u003Cp>RAG 架构的讨论通常聚焦于检索质量：分块、嵌入、重排序、混合搜索和接地。长期记忆引入了问题的另一面：首先，什么被允许进入持久化存储？\u003C\u002Fp>\n\u003Cp>对于持久的智能体记忆，实用的写入策略应当对候选记忆进行分类、保留来源、检测与现有条目的冲突、区分观察与推断、定义敏感性和访问范围，并决定该信息应当过期、修订还是需要用户确认。\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--tip my-6 rounded-xl border p-5 border-violet-300 bg-violet-50 dark:border-violet-900 dark:bg-violet-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">设计原则\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">错误记忆随时间推移变得越昂贵，写入策略就应当越严格。一次糟糕的检索只影响一个答案。一条糟糕的持久记忆可能影响未来每一个检索到它的答案。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-40\">来源是记忆与可靠证据之间的桥梁\u003C\u002Fh2>\n\u003Cp>一条记忆条目理想情况下应保留足够的来源信息，以回答：这来自哪里、何时被观察到、是谁或什么断言了它、它是用户提供的还是模型推断的、有什么来源支持它，以及是否有任何东西取代了它？\u003C\u002Fp>\n\u003Cp>没有来源，压缩后的记忆可能变得比创造它的证据更具权威性。这在长期运行的智能体中尤其危险，因为摘要和抽象会被反复复用。系统可能保留了结论，却丢失了结论成立的条件。\u003C\u002Fp>\n\u003Ch2 id=\"section-43\">更多记忆并不意味着更多上下文\u003C\u002Fh2>\n\u003Cp>一个长期存在的智能体可能积累数 GB 的状态、历史、文档和学习到的信息。模型不需要——通常也不应该——在每一步都接收全部内容。检索、摘要、压缩和结构化记忆的目的是将庞大的持久信息空间转换为小而相关的工作上下文。\u003C\u002Fp>\n\u003Cp>这也是为什么更大的上下文窗口并不能消除记忆架构。容量减轻了一些压力，但它并不能解决新鲜度、权威性、冲突证据、隐私范围、写入质量、修订或决定什么值得关注的问题。\u003C\u002Fp>\n\u003Ch2 id=\"section-46\">生产设计检查清单\u003C\u002Fh2>\n\u003Cul>\u003Cli>定义哪些系统拥有权威的运行时状态。\u003C\u002Fli>\u003Cli>定义哪些信息有资格成为持久记忆。\u003C\u002Fli>\u003Cli>保持用户提供的事实、外部证据和模型推断可区分。\u003C\u002Fli>\u003Cli>为重要记忆附加时间戳、来源、范围和修订语义。\u003C\u002Fli>\u003Cli>将检索相关性与事实权威性视为不同的事物。\u003C\u002Fli>\u003Cli>有意识地构建上下文，而不是注入所有检索到的材料。\u003C\u002Fli>\u003Cli>重新读取易变的事实，而不是信任旧记忆。\u003C\u002Fli>\u003Cli>当陈旧代价高昂时，重新计算廉价的派生值。\u003C\u002Fli>\u003Cli>像测试记忆读取一样仔细地测试记忆写入。\u003C\u002Fli>\u003Cli>分别衡量失败：状态错误、记忆错误、检索错误、上下文构建错误、推理错误和行动错误。\u003C\u002Fli>\u003C\u002Ful>\n\u003Ch2 id=\"section-48\">什么会改变这个答案？\u003C\u002Fh2>\n\u003Cp>随着智能体平台的发展，这些层之间的边界可能会移动。供应商可能提供托管记忆服务，在内部执行存储、修订、检索、摘要和上下文构建。这可以合并实现组件，但并不能消除架构问题。你仍然需要知道返回的条目是当前状态、持久记忆、检索到的证据，还是仅仅放入上下文的文本。\u003C\u002Fp>\n\u003Cp>对于没有跨会话连续性的系统、每个任务都从干净不可变语料库开始的系统，或者所有相关状态都能安全地放入一次调用的严格受限工作流，建议也会改变。在这些情况下，专用的长期记忆层可能增加复杂性而没有足够价值。\u003C\u002Fp>\n\u003Ch2 id=\"section-51\">局限性\u003C\u002Fh2>\n\u003Cp>智能体系统中的术语仍在快速演变。一些框架将对话历史称为“记忆”，另一些使用“会话”、“检查点”、“存储”、“上下文”或“状态”。研究系统也在不同层面定义记忆，从持久查找表到学习到的内部适应。本文中的模型有意分离操作职责，而不是试图强加一套通用词汇。\u003C\u002Fp>\n\u003Ch2 id=\"section-53\">结论\u003C\u002Fh2>\n\u003Cp>有用的问题不是“这个智能体有记忆吗？”而是：什么是状态，从经验中持久化了什么，相关信息如何被检索，以及最终什么作为上下文到达模型？\u003C\u002Fp>\n\u003Cp>一旦这些职责被分离，设计选择就变得更容易测试。过时的事实可以追溯到状态所有权。糟糕的回忆可以追溯到记忆生命周期或检索。过载的提示可以追溯到上下文构建。持续性的幻觉可以追溯到写入策略和来源。RAG 仍然是一个重要的工具，但它只是可靠的长时运行智能体架构的一部分。\u003C\u002Fp>\n\u003Ch2 id=\"section-56\">常见问题\u003C\u002Fh2>\n\u003Csection class=\"editorjs-faq my-6 rounded-xl border border-gray-200 p-5 dark:border-gray-700\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">AI 智能体记忆、RAG、状态与上下文\u003C\u002Fh3>\u003Cdiv id=\"faq1\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">RAG 和 AI 智能体记忆是一回事吗？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">不是。RAG 主要是一种检索模式，为模型调用选择信息。记忆关注的是来自先前交互或经验的信息如何随时间持续存在，以及这些信息如何被管理。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq2\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">向量数据库是智能体记忆吗？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">它可以是记忆的一部分，但向量数据库本身只是一个存储和检索组件。生产级记忆架构还需要决定存储什么、来源、修订、冲突、访问、过期和遗忘。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq3\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">更大的上下文窗口能消除对记忆的需求吗？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">不一定。更大的上下文有助于提升容量，但它不能解决跨会话的持久知识、新鲜度、来源、隐私范围、修订，或决定哪些内容应在以后重用的问题。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq4\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">当前应用状态应该作为记忆存储吗？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">通常，权威的应用或领域系统应继续作为易变状态的真相来源。记忆可以记录状态变更的历史或重要性，但重要操作应重新读取当前的权威值。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-58\">术语表\u003C\u002Fh2>\n\u003Csection class=\"editorjs-glossary my-6 rounded-xl border border-gray-200 dark:border-gray-700 p-5\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">关键术语\u003C\u002Fh3>\u003Cdl>\u003Cdiv id=\"state\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">状态\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">任务、应用、用户、工作流或环境的当前权威状况。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"memory\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">记忆\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">来自先前经验或交互的信息，因其以后可能有用而持续存在，并受生命周期规则约束。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"retrieval\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">检索\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">用于从记忆、外部知识、数据库、图或其他存储中选择可能相关信息的机制。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"context\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">上下文\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">在特定推理或生成步骤中，语言模型实际可用的信息。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"rag\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">RAG\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">检索增强生成：一种模式，其中检索外部或存储的信息并提供给生成模型，以改进当前输出。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"provenance\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">来源\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">描述信息来自何处、何时被观察到、由谁或什么断言，以及如何被转换的元数据。\u003C\u002Fdd>\u003C\u002Fdiv>\u003C\u002Fdl>\u003C\u002Fsection>\n\u003Ch2 id=\"section-60\">主要来源与延伸阅读\u003C\u002Fh2>\n\u003Ca href=\"https:\u002F\u002Fdevelopers.openai.com\u002Fcookbook\u002Fexamples\u002Fagents_sdk\u002Fsession_memory\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">OpenAI — 上下文工程：使用会话进行短期记忆管理\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">OpenAI 关于为长时运行智能体上下文进行裁剪和压缩的指导。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fagents\u002Fsandboxes\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">OpenAI — 沙盒智能体\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">文档展示了持久记忆作为一种能力，具有渐进式披露和读\u002F写行为。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Feffective-context-engineering-for-ai-agents\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">Anthropic — 面向 AI 智能体的有效上下文工程\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">关于为可靠智能体行为策划有限模型上下文的工程指导。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Fblog\u002Fmemora-a-harmonic-memory-representation-balancing-abstraction-and-specificity\u002F\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">Microsoft Research — Memora\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">关于在长时程智能体记忆中平衡抽象与具体性的研究。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Fblog\u002Ffrom-raw-interaction-to-reusable-knowledge-rethinking-memory-for-ai-agents\u002F\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">Microsoft Research — PlugMem\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">关于将原始智能体交互历史转换为可重用结构化知识的研究。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Fpublication\u002Fagentic-context-engineering-evolving-contexts-for-self-improving-language-models\u002F\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">Microsoft Research — 智能体上下文工程（ACE）\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">关于将上下文作为结构化手册演进，而不是反复重写或压缩一切的研究。\u003C\u002Fp>\u003C\u002Fa>",{"time":45,"blocks":46,"version":587},1790369267412,[47,55,61,69,76,82,87,92,97,129,134,139,144,149,154,159,164,169,174,179,184,189,194,220,225,230,235,242,247,252,268,273,278,311,316,353,358,363,368,375,380,385,390,395,400,405,410,428,433,438,443,448,453,458,463,468,473,495,500,526,531,542,551,560,569,578],{"id":48,"data":49,"type":53,"tunes":54},"_4kVYTpqbe",{"title":50,"maxLevel":51,"minLevel":52},"目录",3,2,"tableOfContents",{},{"id":56,"data":57,"type":59,"tunes":60},"intro",{"text":58},"AI智能体记忆、检索增强生成（RAG）、运行时状态和模型上下文经常被当作可以互换的概念来讨论。但它们并非如此。将它们混为一谈会使智能体系统更难推理、更难调试，也更容易变得过时或不安全。","paragraph",{},{"id":62,"data":63,"type":67,"tunes":68},"direct",{"body":64,"title":65,"variant":66},"\u003Cstrong>RAG不是智能体记忆。\u003C\u002Fstrong>RAG是一种检索模式：它为当前模型调用选择可能有用的信息。记忆是从先前交互或经验中派生并在时间维度上被管理的持久信息。状态表示当前任务或环境的真实情况。上下文是实际提供给模型用于当前推理的信息。生产级智能体可能同时使用这四者，但它们解决的是不同的问题。","直接回答","info","callout",{},{"id":70,"data":71,"type":67,"tunes":75},"model-note",{"body":72,"title":73,"variant":74},"下文的四层分离是一种实用架构模型，而非正式的行业标准。供应商和研究论文使用的术语存在重叠。其目的是操作性的：使设计决策、归属划分、故障分析和测试更加清晰。","关于本文所用模型","note",{},{"id":77,"data":78,"type":80,"tunes":81},"h-category",{"text":79,"level":52},"类别错误：把所有看似持久的东西都当作记忆","header",{},{"id":83,"data":84,"type":59,"tunes":86},"p-cat-1",{"text":85},"向量数据库可以存储对话片段。会话对象可以携带最近的轮次。数据库行可以保存当前工作流状态。摘要器可以压缩先前的工作。检索器可以获取旧证据。所有这些都可以让智能体看起来像是在“记住”，但它们并不具有相同的语义。",{},{"id":88,"data":89,"type":59,"tunes":91},"p-cat-2",{"text":90},"这种区分很重要，因为所需的正确性规则不同。当前状态必须是权威且新鲜的。记忆需要写入、修订、遗忘和冲突处理的生命周期规则。检索需要相关性和证据选择质量。上下文需要令牌预算纪律以及对无关或冲突材料的防护。",{},{"id":93,"data":94,"type":80,"tunes":96},"h-layers",{"text":95,"level":52},"四层架构：状态、记忆、检索、上下文",{},{"id":98,"data":99,"type":127,"tunes":128},"table-layers",{"content":100,"stretched":126,"withHeadings":14},[101,106,111,116,121],[102,103,104,105],"层级","核心问题","典型示例","主要正确性关注点",[107,108,109,110],"状态","现在什么是真的？","任务状态、购物车内容、工作流步骤、活动权限、当前游戏状态","新鲜度和权威性",[112,113,114,115],"记忆","过去哪些内容应该持久保留？","用户偏好、先前决策、学到的约束、已解决的故障、持久项目事实","生命周期、修订、来源、遗忘",[117,118,119,120],"检索","现在应该选择哪些信息？","向量搜索、关键词搜索、图查找、重排序、文档搜索","相关性和证据选择",[122,123,124,125],"上下文","模型在这次调用中看到什么？","系统指令、当前请求、检索到的段落、工具结果、摘要","每令牌效用、排序、一致性、噪声",false,"table",{},{"id":130,"data":131,"type":80,"tunes":133},"h-state",{"text":132,"level":51},"1. 状态：现在什么是真的",{},{"id":135,"data":136,"type":59,"tunes":138},"p-state-1",{"text":137},"状态属于运行中的系统，而不是模型的回忆。如果订单被取消、部署被暂停、用户失去权限，或任务从“进行中”变为“已批准”，权威值应来自拥有该事实的系统。",{},{"id":140,"data":141,"type":59,"tunes":143},"p-state-2",{"text":142},"一种危险的设计是让旧的对话摘要成为当前状态的替代品。智能体可能准确记得订单昨天是活跃的，但今天仍然是错的。因此，状态需要明确的所有权、相关情况下的版本控制或时间戳，以及在采取重大行动前重新读取真相来源的路径。",{},{"id":145,"data":146,"type":80,"tunes":148},"h-memory",{"text":147,"level":51},"2. 记忆：过去哪些内容应该持久保留",{},{"id":150,"data":151,"type":59,"tunes":153},"p-memory-1",{"text":152},"记忆不仅仅是“我们能存储的一切”。有用的记忆层决定什么值得持久保留、以什么形式、保留多久、具有什么来源，以及在什么条件下必须修订或删除。",{},{"id":155,"data":156,"type":59,"tunes":158},"p-memory-2",{"text":157},"最近的智能体记忆研究越来越认为原始对话记录存储是不够的。微软的PlugMem工作专注于将原始交互历史转化为结构化、可复用的知识。Memora将丰富的存储内容与更轻量的抽象和检索线索分离，使长时程系统不必在细节和可扩展访问之间做出选择。",{},{"id":160,"data":161,"type":80,"tunes":163},"h-retrieval",{"text":162,"level":51},"3. 检索：现在应该选择什么",{},{"id":165,"data":166,"type":59,"tunes":168},"p-ret-1",{"text":167},"检索是一种选择机制。它可以搜索外部文档、内部知识库、存储的记忆、日志、图、数据库或混合来源。RAG通常位于此处：检索证据，将选定的材料放入模型的工作输入中，然后生成答案。",{},{"id":170,"data":171,"type":59,"tunes":173},"p-ret-2",{"text":172},"仅仅因为检索语料库包含过去的交互，该机制并不会变成记忆。同一个检索器可以搜索智能体从未经历过的策略文档、来自另一个系统的产品数据，或用户的先前决策。检索描述的是信息如何被选择；记忆描述的是为什么某些信息会跨时间持久保留，以及这种持久性如何被管理。",{},{"id":175,"data":176,"type":80,"tunes":178},"h-context",{"text":177,"level":51},"4. 上下文：模型当前实际可以使用的内容",{},{"id":180,"data":181,"type":59,"tunes":183},"p-ctx-1",{"text":182},"上下文是面向模型的层。Anthropic 将上下文工程描述为决定何种上下文配置最有可能产生期望行为，其中上下文是模型在生成期间可用的 token。OpenAI 的会话记忆指南同样将裁剪和压缩视为面向长时间运行的智能体交互的上下文管理技术。",{},{"id":185,"data":186,"type":59,"tunes":188},"p-ctx-2",{"text":187},"这就是为什么一个系统可以拥有出色的记忆却仍然失败。相关记忆可能存在但未被检索到。它可能被检索到，但在上下文中被放置在更强的冲突文本旁边。它可能被压缩，直到决定性细节消失。或者模型可能接收到过多材料，以至于有用证据被噪声稀释。",{},{"id":190,"data":191,"type":80,"tunes":193},"h-flow",{"text":192,"level":52},"各层如何交互",{},{"id":195,"data":196,"type":218,"tunes":219},"flow",{"steps":197,"title":216,"orientation":217},[198,201,204,207,210,213],{"label":199,"description":200},"1. 读取权威状态","从拥有这些事实的系统中加载当前任务、用户、系统或环境事实。",{"label":202,"description":203},"2. 识别记忆需求","确定先前的决策、偏好、经验教训或长期约束是否相关。",{"label":205,"description":206},"3. 检索证据","使用语义、词汇、图、结构化或混合检索来搜索记忆和外部知识。",{"label":208,"description":209},"4. 构建上下文","在模型可用上下文内组装指令、当前状态、选定证据和压缩后的历史。",{"label":211,"description":212},"5. 生成或行动","模型基于组装后的上下文进行推理，并生成答案、计划或工具调用。",{"label":214,"description":215},"6. 验证并写回","验证重要输出，在允许的情况下更新权威状态，并且只持久化通过写入策略的记忆。","一种可能的生产流程","auto","processFlow",{},{"id":221,"data":222,"type":80,"tunes":224},"h-rag",{"text":223,"level":52},"为什么 RAG 不是记忆",{},{"id":226,"data":227,"type":59,"tunes":229},"p-rag-1",{"text":228},"最简单的测试是：RAG 系统可以检索智能体从未见过的信息。仅这一点就表明，检索和记忆是不同的抽象。",{},{"id":231,"data":232,"type":59,"tunes":234},"p-rag-2",{"text":233},"RAG 回答的是：“我应该获取哪些证据？”记忆系统还必须额外回答诸如：“这个事件是否应该成为持久知识？”、“这个新信息是否取代了较旧的记忆？”、“这个记忆是否仍然可信？”、“谁被允许读取它？”以及“它应该在什么时候被遗忘？”",{},{"id":236,"data":237,"type":67,"tunes":241},"rag-trap",{"body":238,"title":239,"variant":240},"如果每一轮对话都被嵌入向量存储，并在之后通过相似度检索，那么系统拥有的是持久查找，但不一定拥有治理良好的记忆架构。仅凭持久化并不能定义记忆质量。","一个常见的设计陷阱","warning",{},{"id":243,"data":244,"type":80,"tunes":246},"h-test",{"text":245,"level":52},"四层分离测试",{},{"id":248,"data":249,"type":59,"tunes":251},"p-test",{"text":250},"当某个功能被称为“记忆”时，请问以下四个问题。答案通常会揭示实际涉及的是哪一层。",{},{"id":253,"data":254,"type":127,"tunes":267},"table-test",{"content":255,"stretched":126,"withHeadings":14},[256,259,261,263,265],[257,258],"问题","如果答案是肯定的，你主要处理的是",[260,107],"这是否代表任务或环境的当前权威状况？",[262,112],"这些信息是否必须跨越当前运行而保留，因为它捕捉了有用的先前经验、偏好或决策？",[264,117],"主要问题是否是决定哪些已存储或外部信息与当前请求相关？",[266,122],"主要问题是否是决定将哪些信息放入当前模型调用中？",{},{"id":269,"data":270,"type":59,"tunes":272},"p-test-note",{"text":271},"单个组件可以参与不止一层。数据库可以同时存储状态和记忆。向量索引可以同时检索外部知识和记忆。这种分离是语义上的，不一定是物理上的。",{},{"id":274,"data":275,"type":80,"tunes":277},"h-fail",{"text":276,"level":52},"层被混同导致的失败模式",{},{"id":279,"data":280,"type":127,"tunes":310},"table-fail",{"content":281,"stretched":126,"withHeadings":14},[282,286,290,294,298,302,306],[283,284,285],"失败模式","发生了什么","结果",[287,288,289],"陈旧状态伪装成记忆","信任旧摘要，而不是重新读取权威系统","智能体基于曾经为真的事实行动",[291,292,293],"记忆被当作不可变事实","先前的偏好或决策被存储，却没有修订规则","已被取代的信息继续影响未来答案",[295,296,297],"检索命中被当作真相","高相似度被误认为事实权威","看起来相关但不正确的证据占据主导",[299,300,301],"上下文过载","注入过多检索段落、记忆、日志和指令","决定性证据被稀释或被矛盾信息抵消",[303,304,305],"不受控的记忆写入","模型生成的解释被自动存储为持久记忆","错误变得持久并自我强化",[307,308,309],"没有来源边界","系统无法区分用户陈述、来源事实、模型推断和生成摘要","后续检索丢失信息的证据地位",{},{"id":312,"data":313,"type":80,"tunes":315},"h-decision",{"text":314,"level":52},"什么应该被记住、检索、重新计算或重新读取？",{},{"id":317,"data":318,"type":127,"tunes":352},"table-decision",{"content":319,"stretched":126,"withHeadings":14},[320,324,328,332,336,340,344,348],[321,322,323],"信息类型","首选处理方式","原因",[325,326,327],"当前权限、订单状态、库存、工作流状态","重新读取权威状态","新鲜度比回忆更重要",[329,330,331],"用户明确提供的稳定用户偏好","记忆，并具备编辑\u002F删除语义","跨会话有用，且归用户所有",[333,334,335],"长期项目期间做出的决策","带时间戳、来源和取代规则的记忆","历史很重要，但决策可能改变",[337,338,339],"产品规格或公共政策文档","从来源检索","外部知识应与其证据保持关联",[341,342,343],"可以低成本重新计算的派生指标","重新计算","避免持久化陈旧的派生值",[345,346,347],"冗长的原始工具输出","外部存储；需要时检索或摘要","不要永久占用上下文",[349,350,351],"模型假设或不确定解释","不要自动提升为持久记忆","推断不等同于事实",{},{"id":354,"data":355,"type":80,"tunes":357},"h-write",{"text":356,"level":52},"记忆系统需要写入策略，而不仅仅是检索策略",{},{"id":359,"data":360,"type":59,"tunes":362},"p-write-1",{"text":361},"RAG 架构的讨论通常聚焦于检索质量：分块、嵌入、重排序、混合搜索和接地。长期记忆引入了问题的另一面：首先，什么被允许进入持久化存储？",{},{"id":364,"data":365,"type":59,"tunes":367},"p-write-2",{"text":366},"对于持久的智能体记忆，实用的写入策略应当对候选记忆进行分类、保留来源、检测与现有条目的冲突、区分观察与推断、定义敏感性和访问范围，并决定该信息应当过期、修订还是需要用户确认。",{},{"id":369,"data":370,"type":67,"tunes":374},"write-tip",{"body":371,"title":372,"variant":373},"错误记忆随时间推移变得越昂贵，写入策略就应当越严格。一次糟糕的检索只影响一个答案。一条糟糕的持久记忆可能影响未来每一个检索到它的答案。","设计原则","tip",{},{"id":376,"data":377,"type":80,"tunes":379},"h-prov",{"text":378,"level":52},"来源是记忆与可靠证据之间的桥梁",{},{"id":381,"data":382,"type":59,"tunes":384},"p-prov-1",{"text":383},"一条记忆条目理想情况下应保留足够的来源信息，以回答：这来自哪里、何时被观察到、是谁或什么断言了它、它是用户提供的还是模型推断的、有什么来源支持它，以及是否有任何东西取代了它？",{},{"id":386,"data":387,"type":59,"tunes":389},"p-prov-2",{"text":388},"没有来源，压缩后的记忆可能变得比创造它的证据更具权威性。这在长期运行的智能体中尤其危险，因为摘要和抽象会被反复复用。系统可能保留了结论，却丢失了结论成立的条件。",{},{"id":391,"data":392,"type":80,"tunes":394},"h-budget",{"text":393,"level":52},"更多记忆并不意味着更多上下文",{},{"id":396,"data":397,"type":59,"tunes":399},"p-budget-1",{"text":398},"一个长期存在的智能体可能积累数 GB 的状态、历史、文档和学习到的信息。模型不需要——通常也不应该——在每一步都接收全部内容。检索、摘要、压缩和结构化记忆的目的是将庞大的持久信息空间转换为小而相关的工作上下文。",{},{"id":401,"data":402,"type":59,"tunes":404},"p-budget-2",{"text":403},"这也是为什么更大的上下文窗口并不能消除记忆架构。容量减轻了一些压力，但它并不能解决新鲜度、权威性、冲突证据、隐私范围、写入质量、修订或决定什么值得关注的问题。",{},{"id":406,"data":407,"type":80,"tunes":409},"h-check",{"text":408,"level":52},"生产设计检查清单",{},{"id":411,"data":412,"type":426,"tunes":427},"checklist",{"meta":413,"items":414,"style":425},{},[415,416,417,418,419,420,421,422,423,424],"定义哪些系统拥有权威的运行时状态。","定义哪些信息有资格成为持久记忆。","保持用户提供的事实、外部证据和模型推断可区分。","为重要记忆附加时间戳、来源、范围和修订语义。","将检索相关性与事实权威性视为不同的事物。","有意识地构建上下文，而不是注入所有检索到的材料。","重新读取易变的事实，而不是信任旧记忆。","当陈旧代价高昂时，重新计算廉价的派生值。","像测试记忆读取一样仔细地测试记忆写入。","分别衡量失败：状态错误、记忆错误、检索错误、上下文构建错误、推理错误和行动错误。","unordered","list",{},{"id":429,"data":430,"type":80,"tunes":432},"h-change",{"text":431,"level":52},"什么会改变这个答案？",{},{"id":434,"data":435,"type":59,"tunes":437},"p-change-1",{"text":436},"随着智能体平台的发展，这些层之间的边界可能会移动。供应商可能提供托管记忆服务，在内部执行存储、修订、检索、摘要和上下文构建。这可以合并实现组件，但并不能消除架构问题。你仍然需要知道返回的条目是当前状态、持久记忆、检索到的证据，还是仅仅放入上下文的文本。",{},{"id":439,"data":440,"type":59,"tunes":442},"p-change-2",{"text":441},"对于没有跨会话连续性的系统、每个任务都从干净不可变语料库开始的系统，或者所有相关状态都能安全地放入一次调用的严格受限工作流，建议也会改变。在这些情况下，专用的长期记忆层可能增加复杂性而没有足够价值。",{},{"id":444,"data":445,"type":80,"tunes":447},"h-limit",{"text":446,"level":52},"局限性",{},{"id":449,"data":450,"type":59,"tunes":452},"p-limit",{"text":451},"智能体系统中的术语仍在快速演变。一些框架将对话历史称为“记忆”，另一些使用“会话”、“检查点”、“存储”、“上下文”或“状态”。研究系统也在不同层面定义记忆，从持久查找表到学习到的内部适应。本文中的模型有意分离操作职责，而不是试图强加一套通用词汇。",{},{"id":454,"data":455,"type":80,"tunes":457},"h-conclusion",{"text":456,"level":52},"结论",{},{"id":459,"data":460,"type":59,"tunes":462},"p-conclusion-1",{"text":461},"有用的问题不是“这个智能体有记忆吗？”而是：什么是状态，从经验中持久化了什么，相关信息如何被检索，以及最终什么作为上下文到达模型？",{},{"id":464,"data":465,"type":59,"tunes":467},"p-conclusion-2",{"text":466},"一旦这些职责被分离，设计选择就变得更容易测试。过时的事实可以追溯到状态所有权。糟糕的回忆可以追溯到记忆生命周期或检索。过载的提示可以追溯到上下文构建。持续性的幻觉可以追溯到写入策略和来源。RAG 仍然是一个重要的工具，但它只是可靠的长时运行智能体架构的一部分。",{},{"id":469,"data":470,"type":80,"tunes":472},"h-faq",{"text":471,"level":52},"常见问题",{},{"id":474,"data":475,"type":474,"tunes":494},"faq",{"items":476,"title":493},[477,481,485,489],{"id":478,"answer":479,"question":480},"faq1","不是。RAG 主要是一种检索模式，为模型调用选择信息。记忆关注的是来自先前交互或经验的信息如何随时间持续存在，以及这些信息如何被管理。","RAG 和 AI 智能体记忆是一回事吗？",{"id":482,"answer":483,"question":484},"faq2","它可以是记忆的一部分，但向量数据库本身只是一个存储和检索组件。生产级记忆架构还需要决定存储什么、来源、修订、冲突、访问、过期和遗忘。","向量数据库是智能体记忆吗？",{"id":486,"answer":487,"question":488},"faq3","不一定。更大的上下文有助于提升容量，但它不能解决跨会话的持久知识、新鲜度、来源、隐私范围、修订，或决定哪些内容应在以后重用的问题。","更大的上下文窗口能消除对记忆的需求吗？",{"id":490,"answer":491,"question":492},"faq4","通常，权威的应用或领域系统应继续作为易变状态的真相来源。记忆可以记录状态变更的历史或重要性，但重要操作应重新读取当前的权威值。","当前应用状态应该作为记忆存储吗？","AI 智能体记忆、RAG、状态与上下文",{},{"id":496,"data":497,"type":80,"tunes":499},"h-glossary",{"text":498,"level":52},"术语表",{},{"id":501,"data":502,"type":501,"tunes":525},"glossary",{"title":503,"entries":504},"关键术语",[505,508,511,514,517,521],{"term":107,"anchor":506,"definition":507},"state","任务、应用、用户、工作流或环境的当前权威状况。",{"term":112,"anchor":509,"definition":510},"memory","来自先前经验或交互的信息，因其以后可能有用而持续存在，并受生命周期规则约束。",{"term":117,"anchor":512,"definition":513},"retrieval","用于从记忆、外部知识、数据库、图或其他存储中选择可能相关信息的机制。",{"term":122,"anchor":515,"definition":516},"context","在特定推理或生成步骤中，语言模型实际可用的信息。",{"term":518,"anchor":519,"definition":520},"RAG","rag","检索增强生成：一种模式，其中检索外部或存储的信息并提供给生成模型，以改进当前输出。",{"term":522,"anchor":523,"definition":524},"来源","provenance","描述信息来自何处、何时被观察到、由谁或什么断言，以及如何被转换的元数据。",{},{"id":527,"data":528,"type":80,"tunes":530},"h-sources",{"text":529,"level":52},"主要来源与延伸阅读",{},{"id":532,"data":533,"type":540,"tunes":541},"openai-session",{"link":534,"meta":535},"https:\u002F\u002Fdevelopers.openai.com\u002Fcookbook\u002Fexamples\u002Fagents_sdk\u002Fsession_memory",{"image":536,"title":538,"description":539},{"url":537},"","OpenAI — 上下文工程：使用会话进行短期记忆管理","OpenAI 关于为长时运行智能体上下文进行裁剪和压缩的指导。","linkTool",{},{"id":543,"data":544,"type":540,"tunes":550},"openai-sandbox",{"link":545,"meta":546},"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fagents\u002Fsandboxes",{"image":547,"title":548,"description":549},{"url":537},"OpenAI — 沙盒智能体","文档展示了持久记忆作为一种能力，具有渐进式披露和读\u002F写行为。",{},{"id":552,"data":553,"type":540,"tunes":559},"anthropic-context",{"link":554,"meta":555},"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Feffective-context-engineering-for-ai-agents",{"image":556,"title":557,"description":558},{"url":537},"Anthropic — 面向 AI 智能体的有效上下文工程","关于为可靠智能体行为策划有限模型上下文的工程指导。",{},{"id":561,"data":562,"type":540,"tunes":568},"ms-memora",{"link":563,"meta":564},"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Fblog\u002Fmemora-a-harmonic-memory-representation-balancing-abstraction-and-specificity\u002F",{"image":565,"title":566,"description":567},{"url":537},"Microsoft Research — Memora","关于在长时程智能体记忆中平衡抽象与具体性的研究。",{},{"id":570,"data":571,"type":540,"tunes":577},"ms-plugmem",{"link":572,"meta":573},"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Fblog\u002Ffrom-raw-interaction-to-reusable-knowledge-rethinking-memory-for-ai-agents\u002F",{"image":574,"title":575,"description":576},{"url":537},"Microsoft Research — PlugMem","关于将原始智能体交互历史转换为可重用结构化知识的研究。",{},{"id":579,"data":580,"type":540,"tunes":586},"ms-ace",{"link":581,"meta":582},"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Fpublication\u002Fagentic-context-engineering-evolving-contexts-for-self-improving-language-models\u002F",{"image":583,"title":584,"description":585},{"url":537},"Microsoft Research — 智能体上下文工程（ACE）","关于将上下文作为结构化手册演进，而不是反复重写或压缩一切的研究。",{},"2.31","代理记忆、RAG、状态和上下文经常被当作可以互换的概念来使用。它们并不是。这个实用的架构模型将这四个层次区分开来，展示了每一层各自应处的位置，并解释了当系统将它们合并为一层时会出现什么问题。","\u002Fuploads\u002F2026\u002F09\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context-1790350560308-np0xy6.webp","ai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context-1790350560308-np0xy6","PUBLISHED","2026-09-25T11:34:00.000Z","2026-09-25T15:34:35.975Z","2026-09-25T21:01:33.061Z",{"en":596,"de":597,"sr":598,"es":599,"fr":600,"it":601,"ru":602,"zh":603},"\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context","\u002Fde\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context","\u002Fsr\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context","\u002Fes\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context","\u002Ffr\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context","\u002Fit\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context","\u002Fru\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context","\u002Fzh\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context",[605,609,613],{"id":606,"name":607,"slug":608},64,"信息架构","information-architecture",{"id":610,"name":611,"slug":612},57,"数据边界","data-boundaries",{"id":614,"name":615,"slug":616},85,"质量门槛","quality-gates",{"id":618,"login":619,"email":620,"displayName":621},"20","rooth8233","aleksandar@stajic.de","Aleksandar Stajić",[623,1069],{"lang":624,"title":625,"content":626,"contentJson":627,"excerpt":1068},"en","AI Agent Memory Is Not RAG: How to Separate Memory, Retrieval, State and Context","{\"time\":1790350647507,\"blocks\":[{\"id\":\"_4kVYTpqbe\",\"type\":\"tableOfContents\",\"data\":{\"title\":\"Contents\",\"minLevel\":2,\"maxLevel\":3},\"tunes\":{}},{\"id\":\"intro\",\"type\":\"paragraph\",\"data\":{\"text\":\"AI agent memory, retrieval-augmented generation (RAG), runtime state, and model context are often discussed as if they were interchangeable. They are not. Collapsing them into one concept makes agent systems harder to reason about, harder to debug, and easier to make stale or unsafe.\"},\"tunes\":{}},{\"id\":\"direct\",\"type\":\"callout\",\"data\":{\"variant\":\"info\",\"title\":\"Direct answer\",\"body\":\"\u003Cstrong>RAG is not agent memory.\u003C\u002Fstrong> RAG is a retrieval pattern: it selects information that may be useful for the current model call. Memory is persistent information derived from prior interaction or experience and managed across time. State represents what is currently true about the running task or environment. Context is the information actually made available to the model for the current inference. A production agent may use all four, but they solve different problems.\"},\"tunes\":{}},{\"id\":\"model-note\",\"type\":\"callout\",\"data\":{\"variant\":\"note\",\"title\":\"About the model used in this article\",\"body\":\"The four-layer separation below is a practical architecture model, not a formal industry standard. Vendors and research papers use overlapping terminology. The purpose is operational: to make design decisions, ownership, failure analysis, and testing clearer.\"},\"tunes\":{}},{\"id\":\"h-category\",\"type\":\"header\",\"data\":{\"text\":\"The category error: treating every persistent-looking thing as memory\",\"level\":2},\"tunes\":{}},{\"id\":\"p-cat-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A vector database can store conversation fragments. A session object can carry recent turns. A database row can hold the current workflow status. A summarizer can compress previous work. A retriever can fetch old evidence. All of these can make an agent appear to “remember,” but they do not have the same semantics.\"},\"tunes\":{}},{\"id\":\"p-cat-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The distinction matters because the required correctness rules are different. Current state must be authoritative and fresh. Memory needs lifecycle rules for writing, revising, forgetting, and conflict handling. Retrieval needs relevance and evidence-selection quality. Context needs token-budget discipline and protection against irrelevant or conflicting material.\"},\"tunes\":{}},{\"id\":\"h-layers\",\"type\":\"header\",\"data\":{\"text\":\"A four-layer architecture: state, memory, retrieval, context\",\"level\":2},\"tunes\":{}},{\"id\":\"table-layers\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Layer\",\"Core question\",\"Typical examples\",\"Primary correctness concern\"],[\"State\",\"What is true now?\",\"Task status, cart contents, workflow step, active permissions, current game state\",\"Freshness and authority\"],[\"Memory\",\"What from the past should persist?\",\"User preference, prior decision, learned constraint, resolved failure, durable project fact\",\"Lifecycle, revision, provenance, forgetting\"],[\"Retrieval\",\"What information should be selected now?\",\"Vector search, keyword search, graph lookup, reranking, document search\",\"Relevance and evidence selection\"],[\"Context\",\"What does the model see for this call?\",\"System instructions, current request, retrieved passages, tool results, summaries\",\"Utility per token, ordering, consistency, noise\"]]},\"tunes\":{}},{\"id\":\"h-state\",\"type\":\"header\",\"data\":{\"text\":\"1. State: what is true now\",\"level\":3},\"tunes\":{}},{\"id\":\"p-state-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"State belongs to the running system, not to the model's recollection. If an order is cancelled, a deployment is paused, a user loses a permission, or a task moves from “in progress” to “approved,” the authoritative value should come from the system that owns that fact.\"},\"tunes\":{}},{\"id\":\"p-state-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"A dangerous design is to let an old conversation summary become a substitute for current state. The agent may accurately remember that the order was active yesterday and still be wrong today. State therefore needs explicit ownership, versioning or timestamps where relevant, and a path to re-read the source of truth before consequential actions.\"},\"tunes\":{}},{\"id\":\"h-memory\",\"type\":\"header\",\"data\":{\"text\":\"2. Memory: what from the past should persist\",\"level\":3},\"tunes\":{}},{\"id\":\"p-memory-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Memory is not simply “everything we can store.” A useful memory layer decides what deserves persistence, in what form, for how long, with what provenance, and under what conditions it must be revised or removed.\"},\"tunes\":{}},{\"id\":\"p-memory-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Recent agent-memory research increasingly treats raw transcript storage as insufficient. Microsoft's PlugMem work focuses on transforming raw interaction histories into structured reusable knowledge. Memora separates rich stored content from lighter abstractions and retrieval cues so that long-horizon systems do not have to choose between detail and scalable access.\"},\"tunes\":{}},{\"id\":\"h-retrieval\",\"type\":\"header\",\"data\":{\"text\":\"3. Retrieval: what should be selected now\",\"level\":3},\"tunes\":{}},{\"id\":\"p-ret-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Retrieval is a selection mechanism. It can search external documents, internal knowledge bases, stored memories, logs, graphs, databases, or mixed sources. RAG normally sits here: retrieve evidence, place selected material into the model's working input, then generate an answer.\"},\"tunes\":{}},{\"id\":\"p-ret-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"That mechanism does not become memory merely because the retrieved corpus contains past interactions. The same retriever can search policy documents that the agent never experienced, product data from another system, or a user's prior decisions. Retrieval describes how information is selected; memory describes why some information persists across time and how that persistence is governed.\"},\"tunes\":{}},{\"id\":\"h-context\",\"type\":\"header\",\"data\":{\"text\":\"4. Context: what the model can actually use right now\",\"level\":3},\"tunes\":{}},{\"id\":\"p-ctx-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Context is the model-facing layer. Anthropic describes context engineering as deciding what configuration of context is most likely to produce the desired behaviour, with context being the tokens available to the model during generation. OpenAI's session-memory guidance similarly treats trimming and compression as context-management techniques for long-running agent interactions.\"},\"tunes\":{}},{\"id\":\"p-ctx-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"This is why a system can have excellent memory and still fail. The relevant memory may exist but not be retrieved. It may be retrieved but placed into context next to stronger conflicting text. It may be compressed until the decisive detail disappears. Or the model may receive so much material that useful evidence is diluted by noise.\"},\"tunes\":{}},{\"id\":\"h-flow\",\"type\":\"header\",\"data\":{\"text\":\"How the layers interact\",\"level\":2},\"tunes\":{}},{\"id\":\"flow\",\"type\":\"processFlow\",\"data\":{\"title\":\"One possible production flow\",\"orientation\":\"auto\",\"steps\":[{\"label\":\"1. Read authoritative state\",\"description\":\"Load current task, user, system, or environment facts from the systems that own them.\"},{\"label\":\"2. Identify memory needs\",\"description\":\"Determine whether prior decisions, preferences, lessons, or long-term constraints are relevant.\"},{\"label\":\"3. Retrieve evidence\",\"description\":\"Search memory and external knowledge using semantic, lexical, graph, structured, or hybrid retrieval.\"},{\"label\":\"4. Build context\",\"description\":\"Assemble instructions, current state, selected evidence, and compacted history within the model's usable context.\"},{\"label\":\"5. Generate or act\",\"description\":\"The model reasons over the assembled context and produces an answer, plan, or tool call.\"},{\"label\":\"6. Validate and write back\",\"description\":\"Validate consequential outputs, update authoritative state where permitted, and persist only memories that pass the write policy.\"}]},\"tunes\":{}},{\"id\":\"h-rag\",\"type\":\"header\",\"data\":{\"text\":\"Why RAG is not memory\",\"level\":2},\"tunes\":{}},{\"id\":\"p-rag-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The simplest test is this: a RAG system can retrieve information the agent has never seen before. That alone shows that retrieval and memory are different abstractions.\"},\"tunes\":{}},{\"id\":\"p-rag-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"RAG answers: “Which evidence should I fetch?” A memory system must additionally answer questions such as: “Should this event become durable knowledge?”, “Does this new information supersede an older memory?”, “Can this memory still be trusted?”, “Who is allowed to read it?”, and “When should it be forgotten?”\"},\"tunes\":{}},{\"id\":\"rag-trap\",\"type\":\"callout\",\"data\":{\"variant\":\"warning\",\"title\":\"A common design trap\",\"body\":\"If every conversation turn is embedded into a vector store and later retrieved by similarity, the system has persistent lookup, but not necessarily a well-governed memory architecture. Persistence alone does not define memory quality.\"},\"tunes\":{}},{\"id\":\"h-test\",\"type\":\"header\",\"data\":{\"text\":\"The four-layer separation test\",\"level\":2},\"tunes\":{}},{\"id\":\"p-test\",\"type\":\"paragraph\",\"data\":{\"text\":\"When a feature is called “memory,” ask the following four questions. The answers usually reveal which layer is actually involved.\"},\"tunes\":{}},{\"id\":\"table-test\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Question\",\"If yes, you are primarily dealing with\"],[\"Does this represent the current authoritative condition of the task or environment?\",\"State\"],[\"Must this information survive the current run because it captures useful prior experience, preference, or decision?\",\"Memory\"],[\"Is the main problem deciding which stored or external information is relevant to the current request?\",\"Retrieval\"],[\"Is the main problem deciding what information to place inside the current model call?\",\"Context\"]]},\"tunes\":{}},{\"id\":\"p-test-note\",\"type\":\"paragraph\",\"data\":{\"text\":\"A single component can participate in more than one layer. A database may store both state and memory. A vector index may retrieve both external knowledge and memories. The separation is semantic, not necessarily physical.\"},\"tunes\":{}},{\"id\":\"h-fail\",\"type\":\"header\",\"data\":{\"text\":\"Failure modes caused by collapsing the layers\",\"level\":2},\"tunes\":{}},{\"id\":\"table-fail\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Failure mode\",\"What happened\",\"Result\"],[\"Stale state disguised as memory\",\"An old summary is trusted instead of re-reading the authoritative system\",\"The agent acts on facts that were once true\"],[\"Memory treated as immutable fact\",\"A prior preference or decision is stored without revision rules\",\"Superseded information keeps influencing future answers\"],[\"Retrieval hit treated as truth\",\"High similarity is mistaken for factual authority\",\"Relevant-looking but incorrect evidence dominates\"],[\"Context overload\",\"Too many retrieved passages, memories, logs, and instructions are injected\",\"The decisive evidence is diluted or contradicted\"],[\"Uncontrolled memory write\",\"Model-generated interpretations are stored automatically as durable memory\",\"Errors become persistent and self-reinforcing\"],[\"No provenance boundary\",\"The system cannot distinguish user statement, source fact, model inference, and generated summary\",\"Later retrieval loses the evidential status of the information\"]]},\"tunes\":{}},{\"id\":\"h-decision\",\"type\":\"header\",\"data\":{\"text\":\"What should be remembered, retrieved, recomputed, or re-read?\",\"level\":2},\"tunes\":{}},{\"id\":\"table-decision\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Information type\",\"Preferred treatment\",\"Reason\"],[\"Current permission, order status, inventory, workflow status\",\"Re-read authoritative state\",\"Freshness matters more than recollection\"],[\"Stable user preference explicitly provided by the user\",\"Memory, with edit\u002Fdelete semantics\",\"Useful across sessions and owned by the user\"],[\"Decision made during a long-running project\",\"Memory with timestamp, provenance, and supersession rules\",\"The history matters, but decisions can change\"],[\"Product specification or public policy document\",\"Retrieve from source\",\"External knowledge should remain tied to its evidence\"],[\"Derived metric that can be cheaply recalculated\",\"Recompute\",\"Avoid persisting stale derived values\"],[\"Long raw tool output\",\"Store externally; retrieve or summarize when needed\",\"Do not consume context permanently\"],[\"Model hypothesis or uncertain interpretation\",\"Do not promote automatically to durable memory\",\"Inference is not equivalent to fact\"]]},\"tunes\":{}},{\"id\":\"h-write\",\"type\":\"header\",\"data\":{\"text\":\"A memory system needs a write policy, not only a retrieval policy\",\"level\":2},\"tunes\":{}},{\"id\":\"p-write-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"RAG architecture discussions often focus on retrieval quality: chunking, embeddings, reranking, hybrid search, and grounding. Long-term memory introduces another side of the problem: what is allowed to enter the persistent store in the first place?\"},\"tunes\":{}},{\"id\":\"p-write-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"For durable agent memory, a practical write policy should classify the candidate memory, preserve provenance, detect conflicts with existing entries, distinguish observation from inference, define sensitivity and access scope, and decide whether the information should expire, be revised, or require user confirmation.\"},\"tunes\":{}},{\"id\":\"write-tip\",\"type\":\"callout\",\"data\":{\"variant\":\"tip\",\"title\":\"Design principle\",\"body\":\"The more expensive a wrong memory becomes over time, the stricter the write policy should be. A bad retrieval affects one answer. A bad durable memory can affect every future answer that retrieves it.\"},\"tunes\":{}},{\"id\":\"h-prov\",\"type\":\"header\",\"data\":{\"text\":\"Provenance is the bridge between memory and reliable evidence\",\"level\":2},\"tunes\":{}},{\"id\":\"p-prov-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A memory entry should ideally retain enough provenance to answer: where did this come from, when was it observed, who or what asserted it, was it user-provided or model-inferred, what source supported it, and has anything superseded it?\"},\"tunes\":{}},{\"id\":\"p-prov-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Without provenance, a compressed memory can become more authoritative than the evidence that created it. This is especially risky in long-running agents where summaries and abstractions are repeatedly reused. The system may preserve the conclusion while losing the conditions under which the conclusion was valid.\"},\"tunes\":{}},{\"id\":\"h-budget\",\"type\":\"header\",\"data\":{\"text\":\"More memory does not mean more context\",\"level\":2},\"tunes\":{}},{\"id\":\"p-budget-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A long-lived agent may accumulate gigabytes of state, history, documents, and learned information. The model does not need — and usually should not receive — all of it for each step. The purpose of retrieval, summarization, compaction, and structured memory is to convert a large persistent information space into a small, relevant working context.\"},\"tunes\":{}},{\"id\":\"p-budget-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"This is also why larger context windows do not eliminate memory architecture. Capacity reduces some pressure, but it does not solve freshness, authority, conflicting evidence, privacy scope, write quality, revision, or deciding what deserves attention.\"},\"tunes\":{}},{\"id\":\"h-check\",\"type\":\"header\",\"data\":{\"text\":\"Production design checklist\",\"level\":2},\"tunes\":{}},{\"id\":\"checklist\",\"type\":\"list\",\"data\":{\"style\":\"unordered\",\"meta\":{},\"items\":[\"Define which systems own authoritative runtime state.\",\"Define which information is eligible to become durable memory.\",\"Keep user-provided facts, external evidence, and model inference distinguishable.\",\"Attach timestamps, provenance, scope, and revision semantics to important memories.\",\"Treat retrieval relevance as different from factual authority.\",\"Build context intentionally instead of injecting all retrieved material.\",\"Re-read volatile facts instead of trusting old memories.\",\"Recompute cheap derived values when staleness would be costly.\",\"Test memory writes as carefully as memory reads.\",\"Measure failures separately: state error, memory error, retrieval error, context-construction error, reasoning error, and action error.\"]},\"tunes\":{}},{\"id\":\"h-change\",\"type\":\"header\",\"data\":{\"text\":\"What would change this answer?\",\"level\":2},\"tunes\":{}},{\"id\":\"p-change-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The boundary between these layers can move as agent platforms evolve. A vendor may offer a managed memory service that internally performs storage, revision, retrieval, summarization, and context construction. That can collapse implementation components, but it does not eliminate the architectural questions. You still need to know whether a returned item is current state, persistent memory, retrieved evidence, or simply text placed into context.\"},\"tunes\":{}},{\"id\":\"p-change-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The recommendation would also change for systems with no cross-session continuity, systems where every task starts from a clean immutable corpus, or tightly bounded workflows where all relevant state fits safely inside one call. In those cases, a dedicated long-term memory layer may add complexity without enough value.\"},\"tunes\":{}},{\"id\":\"h-limit\",\"type\":\"header\",\"data\":{\"text\":\"Limitations\",\"level\":2},\"tunes\":{}},{\"id\":\"p-limit\",\"type\":\"paragraph\",\"data\":{\"text\":\"Terminology in agent systems is still moving quickly. Some frameworks call conversation history “memory,” others use “session,” “checkpoint,” “store,” “context,” or “state.” Research systems also define memory at different levels, from persistent lookup to learned internal adaptation. The model in this article deliberately separates operational responsibilities rather than trying to impose one universal vocabulary.\"},\"tunes\":{}},{\"id\":\"h-conclusion\",\"type\":\"header\",\"data\":{\"text\":\"Conclusion\",\"level\":2},\"tunes\":{}},{\"id\":\"p-conclusion-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The useful question is not “Does this agent have memory?” It is: What is state, what is persisted from experience, how is relevant information retrieved, and what finally reaches the model as context?\"},\"tunes\":{}},{\"id\":\"p-conclusion-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Once those responsibilities are separated, design choices become easier to test. Stale facts can be traced to state ownership. Bad recall can be traced to memory lifecycle or retrieval. Overloaded prompts can be traced to context construction. Persistent hallucinations can be traced to write policy and provenance. RAG remains an important tool, but it is only one part of a reliable long-running agent architecture.\"},\"tunes\":{}},{\"id\":\"h-faq\",\"type\":\"header\",\"data\":{\"text\":\"FAQ\",\"level\":2},\"tunes\":{}},{\"id\":\"faq\",\"type\":\"faq\",\"data\":{\"title\":\"AI agent memory, RAG, state and context\",\"items\":[{\"id\":\"faq1\",\"question\":\"Is RAG the same as AI agent memory?\",\"answer\":\"No. RAG is primarily a retrieval pattern that selects information for a model call. Memory concerns what information from prior interactions or experience persists across time and how that information is governed.\"},{\"id\":\"faq2\",\"question\":\"Is a vector database an agent memory?\",\"answer\":\"It can be part of one, but a vector database by itself is a storage and retrieval component. A production memory architecture also needs decisions about what to store, provenance, revision, conflicts, access, expiration, and forgetting.\"},{\"id\":\"faq3\",\"question\":\"Does a larger context window remove the need for memory?\",\"answer\":\"Not necessarily. Larger context helps with capacity, but it does not solve persistent knowledge across sessions, freshness, provenance, privacy scope, revision, or deciding what should be reused later.\"},{\"id\":\"faq4\",\"question\":\"Should current application state be stored as memory?\",\"answer\":\"Usually the authoritative application or domain system should remain the source of truth for volatile state. Memory may record the history or significance of state changes, but consequential actions should re-read current authoritative values.\"}]},\"tunes\":{}},{\"id\":\"h-glossary\",\"type\":\"header\",\"data\":{\"text\":\"Glossary\",\"level\":2},\"tunes\":{}},{\"id\":\"glossary\",\"type\":\"glossary\",\"data\":{\"title\":\"Key terms\",\"entries\":[{\"term\":\"State\",\"definition\":\"The current authoritative condition of a task, application, user, workflow, or environment.\",\"anchor\":\"state\"},{\"term\":\"Memory\",\"definition\":\"Information from prior experience or interaction that persists because it may be useful later and is subject to lifecycle rules.\",\"anchor\":\"memory\"},{\"term\":\"Retrieval\",\"definition\":\"The mechanism used to select potentially relevant information from memory, external knowledge, databases, graphs, or other stores.\",\"anchor\":\"retrieval\"},{\"term\":\"Context\",\"definition\":\"The information actually available to the language model during a particular inference or generation step.\",\"anchor\":\"context\"},{\"term\":\"RAG\",\"definition\":\"Retrieval-augmented generation: a pattern in which external or stored information is retrieved and supplied to a generative model to improve the current output.\",\"anchor\":\"rag\"},{\"term\":\"Provenance\",\"definition\":\"Metadata describing where information came from, when it was observed, who or what asserted it, and how it was transformed.\",\"anchor\":\"provenance\"}]},\"tunes\":{}},{\"id\":\"h-sources\",\"type\":\"header\",\"data\":{\"text\":\"Primary sources and further reading\",\"level\":2},\"tunes\":{}},{\"id\":\"openai-session\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdevelopers.openai.com\u002Fcookbook\u002Fexamples\u002Fagents_sdk\u002Fsession_memory\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — Context Engineering: Short-Term Memory Management with Sessions\",\"description\":\"OpenAI guidance on trimming and compression for long-running agent context.\"}},\"tunes\":{}},{\"id\":\"openai-sandbox\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fagents\u002Fsandboxes\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — Sandbox Agents\",\"description\":\"Documentation showing persistent memory as a capability with progressive disclosure and read\u002Fwrite behaviour.\"}},\"tunes\":{}},{\"id\":\"anthropic-context\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Feffective-context-engineering-for-ai-agents\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Anthropic — Effective Context Engineering for AI Agents\",\"description\":\"Engineering guidance on curating finite model context for reliable agent behaviour.\"}},\"tunes\":{}},{\"id\":\"ms-memora\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Fblog\u002Fmemora-a-harmonic-memory-representation-balancing-abstraction-and-specificity\u002F\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Microsoft Research — Memora\",\"description\":\"Research on balancing abstraction and specificity in long-horizon agent memory.\"}},\"tunes\":{}},{\"id\":\"ms-plugmem\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Fblog\u002Ffrom-raw-interaction-to-reusable-knowledge-rethinking-memory-for-ai-agents\u002F\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Microsoft Research — PlugMem\",\"description\":\"Research on converting raw agent interaction histories into reusable structured knowledge.\"}},\"tunes\":{}},{\"id\":\"ms-ace\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Fpublication\u002Fagentic-context-engineering-evolving-contexts-for-self-improving-language-models\u002F\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Microsoft Research — Agentic Context Engineering (ACE)\",\"description\":\"Research on evolving context as structured playbooks rather than repeatedly rewriting or compressing everything.\"}},\"tunes\":{}}],\"version\":\"2.31.6\"}",{"time":628,"blocks":629,"version":1067},1790350647507,[630,634,638,643,648,652,656,660,664,693,697,701,705,709,713,717,721,725,729,733,737,741,745,768,772,776,780,785,789,793,808,812,816,848,852,888,892,896,900,905,909,913,917,921,925,929,933,948,952,956,960,964,968,972,976,980,984,1001,1005,1023,1027,1034,1041,1048,1054,1060],{"id":48,"data":631,"type":53,"tunes":633},{"title":632,"maxLevel":51,"minLevel":52},"Contents",{},{"id":56,"data":635,"type":59,"tunes":637},{"text":636},"AI agent memory, retrieval-augmented generation (RAG), runtime state, and model context are often discussed as if they were interchangeable. They are not. Collapsing them into one concept makes agent systems harder to reason about, harder to debug, and easier to make stale or unsafe.",{},{"id":62,"data":639,"type":67,"tunes":642},{"body":640,"title":641,"variant":66},"\u003Cstrong>RAG is not agent memory.\u003C\u002Fstrong> RAG is a retrieval pattern: it selects information that may be useful for the current model call. Memory is persistent information derived from prior interaction or experience and managed across time. State represents what is currently true about the running task or environment. Context is the information actually made available to the model for the current inference. A production agent may use all four, but they solve different problems.","Direct answer",{},{"id":70,"data":644,"type":67,"tunes":647},{"body":645,"title":646,"variant":74},"The four-layer separation below is a practical architecture model, not a formal industry standard. Vendors and research papers use overlapping terminology. The purpose is operational: to make design decisions, ownership, failure analysis, and testing clearer.","About the model used in this article",{},{"id":77,"data":649,"type":80,"tunes":651},{"text":650,"level":52},"The category error: treating every persistent-looking thing as memory",{},{"id":83,"data":653,"type":59,"tunes":655},{"text":654},"A vector database can store conversation fragments. A session object can carry recent turns. A database row can hold the current workflow status. A summarizer can compress previous work. A retriever can fetch old evidence. All of these can make an agent appear to “remember,” but they do not have the same semantics.",{},{"id":88,"data":657,"type":59,"tunes":659},{"text":658},"The distinction matters because the required correctness rules are different. Current state must be authoritative and fresh. Memory needs lifecycle rules for writing, revising, forgetting, and conflict handling. Retrieval needs relevance and evidence-selection quality. Context needs token-budget discipline and protection against irrelevant or conflicting material.",{},{"id":93,"data":661,"type":80,"tunes":663},{"text":662,"level":52},"A four-layer architecture: state, memory, retrieval, context",{},{"id":98,"data":665,"type":127,"tunes":692},{"content":666,"stretched":126,"withHeadings":14},[667,672,677,682,687],[668,669,670,671],"Layer","Core question","Typical examples","Primary correctness concern",[673,674,675,676],"State","What is true now?","Task status, cart contents, workflow step, active permissions, current game state","Freshness and authority",[678,679,680,681],"Memory","What from the past should persist?","User preference, prior decision, learned constraint, resolved failure, durable project fact","Lifecycle, revision, provenance, forgetting",[683,684,685,686],"Retrieval","What information should be selected now?","Vector search, keyword search, graph lookup, reranking, document search","Relevance and evidence selection",[688,689,690,691],"Context","What does the model see for this call?","System instructions, current request, retrieved passages, tool results, summaries","Utility per token, ordering, consistency, noise",{},{"id":130,"data":694,"type":80,"tunes":696},{"text":695,"level":51},"1. State: what is true now",{},{"id":135,"data":698,"type":59,"tunes":700},{"text":699},"State belongs to the running system, not to the model's recollection. If an order is cancelled, a deployment is paused, a user loses a permission, or a task moves from “in progress” to “approved,” the authoritative value should come from the system that owns that fact.",{},{"id":140,"data":702,"type":59,"tunes":704},{"text":703},"A dangerous design is to let an old conversation summary become a substitute for current state. The agent may accurately remember that the order was active yesterday and still be wrong today. State therefore needs explicit ownership, versioning or timestamps where relevant, and a path to re-read the source of truth before consequential actions.",{},{"id":145,"data":706,"type":80,"tunes":708},{"text":707,"level":51},"2. Memory: what from the past should persist",{},{"id":150,"data":710,"type":59,"tunes":712},{"text":711},"Memory is not simply “everything we can store.” A useful memory layer decides what deserves persistence, in what form, for how long, with what provenance, and under what conditions it must be revised or removed.",{},{"id":155,"data":714,"type":59,"tunes":716},{"text":715},"Recent agent-memory research increasingly treats raw transcript storage as insufficient. Microsoft's PlugMem work focuses on transforming raw interaction histories into structured reusable knowledge. Memora separates rich stored content from lighter abstractions and retrieval cues so that long-horizon systems do not have to choose between detail and scalable access.",{},{"id":160,"data":718,"type":80,"tunes":720},{"text":719,"level":51},"3. Retrieval: what should be selected now",{},{"id":165,"data":722,"type":59,"tunes":724},{"text":723},"Retrieval is a selection mechanism. It can search external documents, internal knowledge bases, stored memories, logs, graphs, databases, or mixed sources. RAG normally sits here: retrieve evidence, place selected material into the model's working input, then generate an answer.",{},{"id":170,"data":726,"type":59,"tunes":728},{"text":727},"That mechanism does not become memory merely because the retrieved corpus contains past interactions. The same retriever can search policy documents that the agent never experienced, product data from another system, or a user's prior decisions. Retrieval describes how information is selected; memory describes why some information persists across time and how that persistence is governed.",{},{"id":175,"data":730,"type":80,"tunes":732},{"text":731,"level":51},"4. Context: what the model can actually use right now",{},{"id":180,"data":734,"type":59,"tunes":736},{"text":735},"Context is the model-facing layer. Anthropic describes context engineering as deciding what configuration of context is most likely to produce the desired behaviour, with context being the tokens available to the model during generation. OpenAI's session-memory guidance similarly treats trimming and compression as context-management techniques for long-running agent interactions.",{},{"id":185,"data":738,"type":59,"tunes":740},{"text":739},"This is why a system can have excellent memory and still fail. The relevant memory may exist but not be retrieved. It may be retrieved but placed into context next to stronger conflicting text. It may be compressed until the decisive detail disappears. Or the model may receive so much material that useful evidence is diluted by noise.",{},{"id":190,"data":742,"type":80,"tunes":744},{"text":743,"level":52},"How the layers interact",{},{"id":195,"data":746,"type":218,"tunes":767},{"steps":747,"title":766,"orientation":217},[748,751,754,757,760,763],{"label":749,"description":750},"1. Read authoritative state","Load current task, user, system, or environment facts from the systems that own them.",{"label":752,"description":753},"2. Identify memory needs","Determine whether prior decisions, preferences, lessons, or long-term constraints are relevant.",{"label":755,"description":756},"3. Retrieve evidence","Search memory and external knowledge using semantic, lexical, graph, structured, or hybrid retrieval.",{"label":758,"description":759},"4. Build context","Assemble instructions, current state, selected evidence, and compacted history within the model's usable context.",{"label":761,"description":762},"5. Generate or act","The model reasons over the assembled context and produces an answer, plan, or tool call.",{"label":764,"description":765},"6. Validate and write back","Validate consequential outputs, update authoritative state where permitted, and persist only memories that pass the write policy.","One possible production flow",{},{"id":221,"data":769,"type":80,"tunes":771},{"text":770,"level":52},"Why RAG is not memory",{},{"id":226,"data":773,"type":59,"tunes":775},{"text":774},"The simplest test is this: a RAG system can retrieve information the agent has never seen before. That alone shows that retrieval and memory are different abstractions.",{},{"id":231,"data":777,"type":59,"tunes":779},{"text":778},"RAG answers: “Which evidence should I fetch?” A memory system must additionally answer questions such as: “Should this event become durable knowledge?”, “Does this new information supersede an older memory?”, “Can this memory still be trusted?”, “Who is allowed to read it?”, and “When should it be forgotten?”",{},{"id":236,"data":781,"type":67,"tunes":784},{"body":782,"title":783,"variant":240},"If every conversation turn is embedded into a vector store and later retrieved by similarity, the system has persistent lookup, but not necessarily a well-governed memory architecture. Persistence alone does not define memory quality.","A common design trap",{},{"id":243,"data":786,"type":80,"tunes":788},{"text":787,"level":52},"The four-layer separation test",{},{"id":248,"data":790,"type":59,"tunes":792},{"text":791},"When a feature is called “memory,” ask the following four questions. The answers usually reveal which layer is actually involved.",{},{"id":253,"data":794,"type":127,"tunes":807},{"content":795,"stretched":126,"withHeadings":14},[796,799,801,803,805],[797,798],"Question","If yes, you are primarily dealing with",[800,673],"Does this represent the current authoritative condition of the task or environment?",[802,678],"Must this information survive the current run because it captures useful prior experience, preference, or decision?",[804,683],"Is the main problem deciding which stored or external information is relevant to the current request?",[806,688],"Is the main problem deciding what information to place inside the current model call?",{},{"id":269,"data":809,"type":59,"tunes":811},{"text":810},"A single component can participate in more than one layer. A database may store both state and memory. A vector index may retrieve both external knowledge and memories. The separation is semantic, not necessarily physical.",{},{"id":274,"data":813,"type":80,"tunes":815},{"text":814,"level":52},"Failure modes caused by collapsing the layers",{},{"id":279,"data":817,"type":127,"tunes":847},{"content":818,"stretched":126,"withHeadings":14},[819,823,827,831,835,839,843],[820,821,822],"Failure mode","What happened","Result",[824,825,826],"Stale state disguised as memory","An old summary is trusted instead of re-reading the authoritative system","The agent acts on facts that were once true",[828,829,830],"Memory treated as immutable fact","A prior preference or decision is stored without revision rules","Superseded information keeps influencing future answers",[832,833,834],"Retrieval hit treated as truth","High similarity is mistaken for factual authority","Relevant-looking but incorrect evidence dominates",[836,837,838],"Context overload","Too many retrieved passages, memories, logs, and instructions are injected","The decisive evidence is diluted or contradicted",[840,841,842],"Uncontrolled memory write","Model-generated interpretations are stored automatically as durable memory","Errors become persistent and self-reinforcing",[844,845,846],"No provenance boundary","The system cannot distinguish user statement, source fact, model inference, and generated summary","Later retrieval loses the evidential status of the information",{},{"id":312,"data":849,"type":80,"tunes":851},{"text":850,"level":52},"What should be remembered, retrieved, recomputed, or re-read?",{},{"id":317,"data":853,"type":127,"tunes":887},{"content":854,"stretched":126,"withHeadings":14},[855,859,863,867,871,875,879,883],[856,857,858],"Information type","Preferred treatment","Reason",[860,861,862],"Current permission, order status, inventory, workflow status","Re-read authoritative state","Freshness matters more than recollection",[864,865,866],"Stable user preference explicitly provided by the user","Memory, with edit\u002Fdelete semantics","Useful across sessions and owned by the user",[868,869,870],"Decision made during a long-running project","Memory with timestamp, provenance, and supersession rules","The history matters, but decisions can change",[872,873,874],"Product specification or public policy document","Retrieve from source","External knowledge should remain tied to its evidence",[876,877,878],"Derived metric that can be cheaply recalculated","Recompute","Avoid persisting stale derived values",[880,881,882],"Long raw tool output","Store externally; retrieve or summarize when needed","Do not consume context permanently",[884,885,886],"Model hypothesis or uncertain interpretation","Do not promote automatically to durable memory","Inference is not equivalent to fact",{},{"id":354,"data":889,"type":80,"tunes":891},{"text":890,"level":52},"A memory system needs a write policy, not only a retrieval policy",{},{"id":359,"data":893,"type":59,"tunes":895},{"text":894},"RAG architecture discussions often focus on retrieval quality: chunking, embeddings, reranking, hybrid search, and grounding. Long-term memory introduces another side of the problem: what is allowed to enter the persistent store in the first place?",{},{"id":364,"data":897,"type":59,"tunes":899},{"text":898},"For durable agent memory, a practical write policy should classify the candidate memory, preserve provenance, detect conflicts with existing entries, distinguish observation from inference, define sensitivity and access scope, and decide whether the information should expire, be revised, or require user confirmation.",{},{"id":369,"data":901,"type":67,"tunes":904},{"body":902,"title":903,"variant":373},"The more expensive a wrong memory becomes over time, the stricter the write policy should be. A bad retrieval affects one answer. A bad durable memory can affect every future answer that retrieves it.","Design principle",{},{"id":376,"data":906,"type":80,"tunes":908},{"text":907,"level":52},"Provenance is the bridge between memory and reliable evidence",{},{"id":381,"data":910,"type":59,"tunes":912},{"text":911},"A memory entry should ideally retain enough provenance to answer: where did this come from, when was it observed, who or what asserted it, was it user-provided or model-inferred, what source supported it, and has anything superseded it?",{},{"id":386,"data":914,"type":59,"tunes":916},{"text":915},"Without provenance, a compressed memory can become more authoritative than the evidence that created it. This is especially risky in long-running agents where summaries and abstractions are repeatedly reused. The system may preserve the conclusion while losing the conditions under which the conclusion was valid.",{},{"id":391,"data":918,"type":80,"tunes":920},{"text":919,"level":52},"More memory does not mean more context",{},{"id":396,"data":922,"type":59,"tunes":924},{"text":923},"A long-lived agent may accumulate gigabytes of state, history, documents, and learned information. The model does not need — and usually should not receive — all of it for each step. The purpose of retrieval, summarization, compaction, and structured memory is to convert a large persistent information space into a small, relevant working context.",{},{"id":401,"data":926,"type":59,"tunes":928},{"text":927},"This is also why larger context windows do not eliminate memory architecture. Capacity reduces some pressure, but it does not solve freshness, authority, conflicting evidence, privacy scope, write quality, revision, or deciding what deserves attention.",{},{"id":406,"data":930,"type":80,"tunes":932},{"text":931,"level":52},"Production design checklist",{},{"id":411,"data":934,"type":426,"tunes":947},{"meta":935,"items":936,"style":425},{},[937,938,939,940,941,942,943,944,945,946],"Define which systems own authoritative runtime state.","Define which information is eligible to become durable memory.","Keep user-provided facts, external evidence, and model inference distinguishable.","Attach timestamps, provenance, scope, and revision semantics to important memories.","Treat retrieval relevance as different from factual authority.","Build context intentionally instead of injecting all retrieved material.","Re-read volatile facts instead of trusting old memories.","Recompute cheap derived values when staleness would be costly.","Test memory writes as carefully as memory reads.","Measure failures separately: state error, memory error, retrieval error, context-construction error, reasoning error, and action error.",{},{"id":429,"data":949,"type":80,"tunes":951},{"text":950,"level":52},"What would change this answer?",{},{"id":434,"data":953,"type":59,"tunes":955},{"text":954},"The boundary between these layers can move as agent platforms evolve. A vendor may offer a managed memory service that internally performs storage, revision, retrieval, summarization, and context construction. That can collapse implementation components, but it does not eliminate the architectural questions. You still need to know whether a returned item is current state, persistent memory, retrieved evidence, or simply text placed into context.",{},{"id":439,"data":957,"type":59,"tunes":959},{"text":958},"The recommendation would also change for systems with no cross-session continuity, systems where every task starts from a clean immutable corpus, or tightly bounded workflows where all relevant state fits safely inside one call. In those cases, a dedicated long-term memory layer may add complexity without enough value.",{},{"id":444,"data":961,"type":80,"tunes":963},{"text":962,"level":52},"Limitations",{},{"id":449,"data":965,"type":59,"tunes":967},{"text":966},"Terminology in agent systems is still moving quickly. Some frameworks call conversation history “memory,” others use “session,” “checkpoint,” “store,” “context,” or “state.” Research systems also define memory at different levels, from persistent lookup to learned internal adaptation. The model in this article deliberately separates operational responsibilities rather than trying to impose one universal vocabulary.",{},{"id":454,"data":969,"type":80,"tunes":971},{"text":970,"level":52},"Conclusion",{},{"id":459,"data":973,"type":59,"tunes":975},{"text":974},"The useful question is not “Does this agent have memory?” It is: What is state, what is persisted from experience, how is relevant information retrieved, and what finally reaches the model as context?",{},{"id":464,"data":977,"type":59,"tunes":979},{"text":978},"Once those responsibilities are separated, design choices become easier to test. Stale facts can be traced to state ownership. Bad recall can be traced to memory lifecycle or retrieval. Overloaded prompts can be traced to context construction. Persistent hallucinations can be traced to write policy and provenance. RAG remains an important tool, but it is only one part of a reliable long-running agent architecture.",{},{"id":469,"data":981,"type":80,"tunes":983},{"text":982,"level":52},"FAQ",{},{"id":474,"data":985,"type":474,"tunes":1000},{"items":986,"title":999},[987,990,993,996],{"id":478,"answer":988,"question":989},"No. RAG is primarily a retrieval pattern that selects information for a model call. Memory concerns what information from prior interactions or experience persists across time and how that information is governed.","Is RAG the same as AI agent memory?",{"id":482,"answer":991,"question":992},"It can be part of one, but a vector database by itself is a storage and retrieval component. A production memory architecture also needs decisions about what to store, provenance, revision, conflicts, access, expiration, and forgetting.","Is a vector database an agent memory?",{"id":486,"answer":994,"question":995},"Not necessarily. Larger context helps with capacity, but it does not solve persistent knowledge across sessions, freshness, provenance, privacy scope, revision, or deciding what should be reused later.","Does a larger context window remove the need for memory?",{"id":490,"answer":997,"question":998},"Usually the authoritative application or domain system should remain the source of truth for volatile state. Memory may record the history or significance of state changes, but consequential actions should re-read current authoritative values.","Should current application state be stored as memory?","AI agent memory, RAG, state and context",{},{"id":496,"data":1002,"type":80,"tunes":1004},{"text":1003,"level":52},"Glossary",{},{"id":501,"data":1006,"type":501,"tunes":1022},{"title":1007,"entries":1008},"Key terms",[1009,1011,1013,1015,1017,1019],{"term":673,"anchor":506,"definition":1010},"The current authoritative condition of a task, application, user, workflow, or environment.",{"term":678,"anchor":509,"definition":1012},"Information from prior experience or interaction that persists because it may be useful later and is subject to lifecycle rules.",{"term":683,"anchor":512,"definition":1014},"The mechanism used to select potentially relevant information from memory, external knowledge, databases, graphs, or other stores.",{"term":688,"anchor":515,"definition":1016},"The information actually available to the language model during a particular inference or generation step.",{"term":518,"anchor":519,"definition":1018},"Retrieval-augmented generation: a pattern in which external or stored information is retrieved and supplied to a generative model to improve the current output.",{"term":1020,"anchor":523,"definition":1021},"Provenance","Metadata describing where information came from, when it was observed, who or what asserted it, and how it was transformed.",{},{"id":527,"data":1024,"type":80,"tunes":1026},{"text":1025,"level":52},"Primary sources and further reading",{},{"id":532,"data":1028,"type":540,"tunes":1033},{"link":534,"meta":1029},{"image":1030,"title":1031,"description":1032},{"url":537},"OpenAI — Context Engineering: Short-Term Memory Management with Sessions","OpenAI guidance on trimming and compression for long-running agent context.",{},{"id":543,"data":1035,"type":540,"tunes":1040},{"link":545,"meta":1036},{"image":1037,"title":1038,"description":1039},{"url":537},"OpenAI — Sandbox Agents","Documentation showing persistent memory as a capability with progressive disclosure and read\u002Fwrite behaviour.",{},{"id":552,"data":1042,"type":540,"tunes":1047},{"link":554,"meta":1043},{"image":1044,"title":1045,"description":1046},{"url":537},"Anthropic — Effective Context Engineering for AI Agents","Engineering guidance on curating finite model context for reliable agent behaviour.",{},{"id":561,"data":1049,"type":540,"tunes":1053},{"link":563,"meta":1050},{"image":1051,"title":566,"description":1052},{"url":537},"Research on balancing abstraction and specificity in long-horizon agent memory.",{},{"id":570,"data":1055,"type":540,"tunes":1059},{"link":572,"meta":1056},{"image":1057,"title":575,"description":1058},{"url":537},"Research on converting raw agent interaction histories into reusable structured knowledge.",{},{"id":579,"data":1061,"type":540,"tunes":1066},{"link":581,"meta":1062},{"image":1063,"title":1064,"description":1065},{"url":537},"Microsoft Research — Agentic Context Engineering (ACE)","Research on evolving context as structured playbooks rather than repeatedly rewriting or compressing everything.",{},"2.31.6","Agent memory, RAG, state, and context are often used as if they were interchangeable. They are not. This practical architecture model separates the four layers, shows where each belongs, and explains what breaks when systems collapse them into one.",{"lang":7,"title":41,"content":43,"contentJson":1070,"excerpt":588},{"time":45,"blocks":1071,"version":587},[1072,1075,1078,1081,1084,1087,1090,1093,1096,1105,1108,1111,1114,1117,1120,1123,1126,1129,1132,1135,1138,1141,1144,1154,1157,1160,1163,1166,1169,1172,1181,1184,1187,1198,1201,1213,1216,1219,1222,1225,1228,1231,1234,1237,1240,1243,1246,1251,1254,1257,1260,1263,1266,1269,1272,1275,1278,1286,1289,1299,1302,1307,1312,1317,1322,1327],{"id":48,"data":1073,"type":53,"tunes":1074},{"title":50,"maxLevel":51,"minLevel":52},{},{"id":56,"data":1076,"type":59,"tunes":1077},{"text":58},{},{"id":62,"data":1079,"type":67,"tunes":1080},{"body":64,"title":65,"variant":66},{},{"id":70,"data":1082,"type":67,"tunes":1083},{"body":72,"title":73,"variant":74},{},{"id":77,"data":1085,"type":80,"tunes":1086},{"text":79,"level":52},{},{"id":83,"data":1088,"type":59,"tunes":1089},{"text":85},{},{"id":88,"data":1091,"type":59,"tunes":1092},{"text":90},{},{"id":93,"data":1094,"type":80,"tunes":1095},{"text":95,"level":52},{},{"id":98,"data":1097,"type":127,"tunes":1104},{"content":1098,"stretched":126,"withHeadings":14},[1099,1100,1101,1102,1103],[102,103,104,105],[107,108,109,110],[112,113,114,115],[117,118,119,120],[122,123,124,125],{},{"id":130,"data":1106,"type":80,"tunes":1107},{"text":132,"level":51},{},{"id":135,"data":1109,"type":59,"tunes":1110},{"text":137},{},{"id":140,"data":1112,"type":59,"tunes":1113},{"text":142},{},{"id":145,"data":1115,"type":80,"tunes":1116},{"text":147,"level":51},{},{"id":150,"data":1118,"type":59,"tunes":1119},{"text":152},{},{"id":155,"data":1121,"type":59,"tunes":1122},{"text":157},{},{"id":160,"data":1124,"type":80,"tunes":1125},{"text":162,"level":51},{},{"id":165,"data":1127,"type":59,"tunes":1128},{"text":167},{},{"id":170,"data":1130,"type":59,"tunes":1131},{"text":172},{},{"id":175,"data":1133,"type":80,"tunes":1134},{"text":177,"level":51},{},{"id":180,"data":1136,"type":59,"tunes":1137},{"text":182},{},{"id":185,"data":1139,"type":59,"tunes":1140},{"text":187},{},{"id":190,"data":1142,"type":80,"tunes":1143},{"text":192,"level":52},{},{"id":195,"data":1145,"type":218,"tunes":1153},{"steps":1146,"title":216,"orientation":217},[1147,1148,1149,1150,1151,1152],{"label":199,"description":200},{"label":202,"description":203},{"label":205,"description":206},{"label":208,"description":209},{"label":211,"description":212},{"label":214,"description":215},{},{"id":221,"data":1155,"type":80,"tunes":1156},{"text":223,"level":52},{},{"id":226,"data":1158,"type":59,"tunes":1159},{"text":228},{},{"id":231,"data":1161,"type":59,"tunes":1162},{"text":233},{},{"id":236,"data":1164,"type":67,"tunes":1165},{"body":238,"title":239,"variant":240},{},{"id":243,"data":1167,"type":80,"tunes":1168},{"text":245,"level":52},{},{"id":248,"data":1170,"type":59,"tunes":1171},{"text":250},{},{"id":253,"data":1173,"type":127,"tunes":1180},{"content":1174,"stretched":126,"withHeadings":14},[1175,1176,1177,1178,1179],[257,258],[260,107],[262,112],[264,117],[266,122],{},{"id":269,"data":1182,"type":59,"tunes":1183},{"text":271},{},{"id":274,"data":1185,"type":80,"tunes":1186},{"text":276,"level":52},{},{"id":279,"data":1188,"type":127,"tunes":1197},{"content":1189,"stretched":126,"withHeadings":14},[1190,1191,1192,1193,1194,1195,1196],[283,284,285],[287,288,289],[291,292,293],[295,296,297],[299,300,301],[303,304,305],[307,308,309],{},{"id":312,"data":1199,"type":80,"tunes":1200},{"text":314,"level":52},{},{"id":317,"data":1202,"type":127,"tunes":1212},{"content":1203,"stretched":126,"withHeadings":14},[1204,1205,1206,1207,1208,1209,1210,1211],[321,322,323],[325,326,327],[329,330,331],[333,334,335],[337,338,339],[341,342,343],[345,346,347],[349,350,351],{},{"id":354,"data":1214,"type":80,"tunes":1215},{"text":356,"level":52},{},{"id":359,"data":1217,"type":59,"tunes":1218},{"text":361},{},{"id":364,"data":1220,"type":59,"tunes":1221},{"text":366},{},{"id":369,"data":1223,"type":67,"tunes":1224},{"body":371,"title":372,"variant":373},{},{"id":376,"data":1226,"type":80,"tunes":1227},{"text":378,"level":52},{},{"id":381,"data":1229,"type":59,"tunes":1230},{"text":383},{},{"id":386,"data":1232,"type":59,"tunes":1233},{"text":388},{},{"id":391,"data":1235,"type":80,"tunes":1236},{"text":393,"level":52},{},{"id":396,"data":1238,"type":59,"tunes":1239},{"text":398},{},{"id":401,"data":1241,"type":59,"tunes":1242},{"text":403},{},{"id":406,"data":1244,"type":80,"tunes":1245},{"text":408,"level":52},{},{"id":411,"data":1247,"type":426,"tunes":1250},{"meta":1248,"items":1249,"style":425},{},[415,416,417,418,419,420,421,422,423,424],{},{"id":429,"data":1252,"type":80,"tunes":1253},{"text":431,"level":52},{},{"id":434,"data":1255,"type":59,"tunes":1256},{"text":436},{},{"id":439,"data":1258,"type":59,"tunes":1259},{"text":441},{},{"id":444,"data":1261,"type":80,"tunes":1262},{"text":446,"level":52},{},{"id":449,"data":1264,"type":59,"tunes":1265},{"text":451},{},{"id":454,"data":1267,"type":80,"tunes":1268},{"text":456,"level":52},{},{"id":459,"data":1270,"type":59,"tunes":1271},{"text":461},{},{"id":464,"data":1273,"type":59,"tunes":1274},{"text":466},{},{"id":469,"data":1276,"type":80,"tunes":1277},{"text":471,"level":52},{},{"id":474,"data":1279,"type":474,"tunes":1285},{"items":1280,"title":493},[1281,1282,1283,1284],{"id":478,"answer":479,"question":480},{"id":482,"answer":483,"question":484},{"id":486,"answer":487,"question":488},{"id":490,"answer":491,"question":492},{},{"id":496,"data":1287,"type":80,"tunes":1288},{"text":498,"level":52},{},{"id":501,"data":1290,"type":501,"tunes":1298},{"title":503,"entries":1291},[1292,1293,1294,1295,1296,1297],{"term":107,"anchor":506,"definition":507},{"term":112,"anchor":509,"definition":510},{"term":117,"anchor":512,"definition":513},{"term":122,"anchor":515,"definition":516},{"term":518,"anchor":519,"definition":520},{"term":522,"anchor":523,"definition":524},{},{"id":527,"data":1300,"type":80,"tunes":1301},{"text":529,"level":52},{},{"id":532,"data":1303,"type":540,"tunes":1306},{"link":534,"meta":1304},{"image":1305,"title":538,"description":539},{"url":537},{},{"id":543,"data":1308,"type":540,"tunes":1311},{"link":545,"meta":1309},{"image":1310,"title":548,"description":549},{"url":537},{},{"id":552,"data":1313,"type":540,"tunes":1316},{"link":554,"meta":1314},{"image":1315,"title":557,"description":558},{"url":537},{},{"id":561,"data":1318,"type":540,"tunes":1321},{"link":563,"meta":1319},{"image":1320,"title":566,"description":567},{"url":537},{},{"id":570,"data":1323,"type":540,"tunes":1326},{"link":572,"meta":1324},{"image":1325,"title":575,"description":576},{"url":537},{},{"id":579,"data":1328,"type":540,"tunes":1331},{"link":581,"meta":1329},{"image":1330,"title":584,"description":585},{"url":537},{},"Post erfolgreich abgerufen",{"items":1334,"source":1405,"manualIds":1406,"manualMatchedIds":1407},[1335,1342,1349,1356,1363,1370,1377,1384,1391,1398],{"id":1336,"slug":1337,"title":1338,"excerpt":1339,"featuredImage":1340,"publishedAt":1341},"476","mcp-vs-a2a-vs-ucp-vs-ap2-vs-a2ui-the-agent-protocol-stack-explained","MCP vs A2A vs UCP vs AP2 vs A2UI：智能体协议栈详解","MCP、A2A、UCP、AP2 和 A2UI 常被描述为相互竞争的智能体标准。它们大多解决的是不同的互操作性问题。本指南将每个协议映射到其实际标准化的边界，并展示它们如何在同一个生产系统中协同工作。","\u002Fuploads\u002F2026\u002F09\u002Fmcp-vs-a2a-vs-ucp-vs-ap2-vs-a2ui-the-agent-protocol-stack-explained-1790352625869-2ezle0.webp","2026-09-25T12:09:00.000Z",{"id":1343,"slug":1344,"title":1345,"excerpt":1346,"featuredImage":1347,"publishedAt":1348},"478","what-is-rag-the-simplest-explanation-of-how-it-works","什么是RAG？对其工作原理的最简单解释","RAG听起来很复杂，但想法很简单：在AI回答之前，它先从知识源查找有用的信息，并将该信息提供给语言模型。本指南使用一个简单的思维模型来解释RAG、LLM、状态、记忆和工具。","\u002Fuploads\u002F2026\u002F09\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works-1790377492124-khjagt.webp","2026-09-25T19:03:00.000Z",{"id":1350,"slug":1351,"title":1352,"excerpt":1353,"featuredImage":1354,"publishedAt":1355},"469","rag-failed-but-which-layer-actually-failed-a-diagnostic-method","RAG失败了——但究竟是哪一层真正失败了？一种诊断方法","当RAG答案出错时，将问题归咎于检索或模型过于笼统。这种诊断方法将来源覆盖、查询构建、检索、排序、上下文组装、生成、证据归因和时效性逐一隔离，从而使实际故障能够被复现并修复。","\u002Fuploads\u002F2026\u002F09\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method-1790350847177-pior4c.webp","2026-09-24T19:39:00.000Z",{"id":1357,"slug":1358,"title":1359,"excerpt":1360,"featuredImage":1361,"publishedAt":1362},"472","why-more-context-can-make-ai-answers-worse","为什么更多上下文会让AI的回答更糟","更大的上下文窗口并不保证更好的答案。本文解释了信号稀释、证据冲突、状态过时、位置敏感性和有损压缩如何降低AI可靠性——并介绍了一种实用的上下文压力测试。","\u002Fuploads\u002F2026\u002F09\u002Fwhy-more-context-can-make-ai-answers-worse-1790351615793-2ntv2v.webp","2026-09-25T11:51:00.000Z",{"id":1364,"slug":1365,"title":1366,"excerpt":1367,"featuredImage":1368,"publishedAt":1369},"467","the-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers","答案有效性边界：相关性到可靠AI答案之间缺失的层级","一个来源可能相关、权威，但对于所提出的问题仍然是错误的。缺失的层次是适用性：答案成立的条件，以及迫使其被重新考虑的变化。本文介绍了“答案有效性边界”这一面向人类、AI搜索和RAG系统的来源设计模式。","\u002Fuploads\u002F2026\u002F09\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers-1790272901306-1g5jly.webp","2026-09-24T11:59:00.000Z",{"id":1371,"slug":1372,"title":1373,"excerpt":1374,"featuredImage":1375,"publishedAt":1376},"466","the-gpu-is-not-the-product-future-proof-private-ai-architecture","GPU 不是产品：面向未来的私有 AI 架构","私有 AI 基础设施不应围绕单一 GPU 或单一模型来设计。更具韧性的做法是将快速推理 GPU、内存充裕的 AI 系统、物理 AI 节点以及可选的前沿云模型，统一置于一个具备能力感知的路由层之后。","\u002Fuploads\u002F2026\u002F09\u002Fthe-gpu-is-not-the-product-future-proof-private-ai-architecture-1790140878812-8hsl39.webp","2026-09-23T01:19:00.000Z",{"id":1378,"slug":1379,"title":1380,"excerpt":1381,"featuredImage":1382,"publishedAt":1383},"364","tipps-fuer-die-verbesserung-der-seo-suchmaschinenoptimierung","Mastering the SEO Workflow: Essential Optimization Strategies for Organic Growth","A structured SEO workflow is crucial for sustainable organic growth. Learn the ten foundational strategies, from keyword research and technical optimization to content quality and performance analysis.","\u002Fuploads\u002F2026\u002F03\u002Ftipps-fuer-die-verbesserung-der-seo-suchmaschinenoptimierung-1774866098131-hwkzrg.webp","2024-01-26T06:35:00.000Z",{"id":1385,"slug":1386,"title":1387,"excerpt":1388,"featuredImage":1389,"publishedAt":1390},"459","ollama-is-not-the-product-building-production-ready-open-llm-applications","Ollama 并非产品：构建可投入生产的开源大语言模型应用","使用Ollama运行本地模型很简单。但构建一个可用于生产环境的开源大语言模型（Open-LLM）应用则更具挑战性：它需要RAG（检索增强生成）、访问控制、供应商抽象、评估、日志记录、部署规范，以及围绕模型构建受控的应用层。","\u002Fuploads\u002F2026\u002F06\u002Follama-is-not-the-product-building-production-ready-open-llm-applications-1782679361640-h0usqf.webp","2026-06-28T16:39:00.000Z",{"id":1392,"slug":1393,"title":1394,"excerpt":1395,"featuredImage":1396,"publishedAt":1397},"363","front-und-backend-entwicklung","前端与后端开发","前端和后端开发是网络开发的重要组成部分，涉及创建网络应用程序和网站。前端开发专注于用户界面，而后端开发则负责编程和管理服务器端。","\u002Fuploads\u002F2026\u002F03\u002Ffront-und-backend-entwicklung-1774872219531-wyu4i1.webp","2023-04-12T11:11:00.000Z",{"id":1399,"slug":1400,"title":1401,"excerpt":1402,"featuredImage":1403,"publishedAt":1404},"470","what-should-an-ai-agent-remember-forget-recompute-or-retrieve-again","AI代理应该记住、遗忘、重新计算还是再次检索什么？","长时间运行的代理不应记住所有内容。本文提供了一个实用的生命周期模型，用于决定哪些内容应属于持久记忆、哪些内容应重新检索、哪些内容重新计算更安全，以及哪些内容应过期或被取代。","\u002Fuploads\u002F2026\u002F09\u002Fwhat-should-an-ai-agent-remember-forget-recompute-or-retrieve-again-1790351131087-iehz28.webp","2026-09-25T09:43:00.000Z","fallback",[],[],[1409,1413],{"id":1410,"name":1411,"location":80,"isActive":14,"isDefault":126,"items":1412},1,"main-navigation",[],{"id":1414,"name":1415,"location":1416,"isActive":14,"isDefault":14,"items":1417},4,"main-menu","sidebar",[1418,1434,1447,1461,1471,1486,1501],{"id":1419,"title":1420,"url":1428,"target":1429,"icon":1430,"isActive":14,"type":1431,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":1432,"portfolioId":10,"children":1433},"item-18",{"de":1421,"en":1422,"es":1423,"fr":1424,"it":1422,"ru":1425,"sr":1426,"zh":1427},"Startseite","Home","Inicio","Accueil","Главная","Почетна","首页","\u002Ffull-stack-web-developer-munich-performance-seo-and-maintainable-builds","_self","i-lucide-home","page",111,[],{"id":1435,"title":1436,"url":1443,"target":1429,"icon":1444,"isActive":14,"type":1431,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":1445,"portfolioId":10,"children":1446},"item-22",{"de":1437,"en":1437,"es":1438,"fr":1437,"it":1439,"ru":1440,"sr":1441,"zh":1442},"Vision","Visión","Visione","Видение","Визија","想象","\u002Fueber-uns-webdesign-muenchen-webaplikation","i-lucide-eye",113,[],{"id":1448,"title":1449,"url":1457,"target":1429,"icon":1458,"isActive":14,"type":1431,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":1459,"portfolioId":10,"children":1460},"item-19",{"de":1450,"en":1451,"es":1452,"fr":1451,"it":1453,"ru":1454,"sr":1455,"zh":1456},"Leistungen","Services","Servicios","Servizi","Услуги","Услуге","服务","\u002Fservices-dienstleistungen-muenchen","i-lucide-wrench",116,[],{"id":1462,"title":1463,"url":1467,"target":1429,"icon":1468,"isActive":14,"type":1431,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":1469,"portfolioId":10,"children":1470},"item-23",{"de":1464,"en":1464,"es":1464,"fr":1464,"it":1464,"ru":1465,"sr":1465,"zh":1466},"Blog","Блог","博客","\u002Fblog","i-lucide-book-open",112,[],{"id":1472,"title":1473,"url":1482,"target":1429,"icon":1483,"isActive":14,"type":1431,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":1484,"portfolioId":10,"children":1485},"item-32",{"de":1474,"en":1475,"es":1476,"fr":1477,"it":1478,"ru":1479,"sr":1480,"zh":1481},"Neue Technologien","New Technologies","Nuevas tecnologías","Nouvelles technologies","Nuove tecnologie","Новые технологии","Нове технологије","新技术！","\u002Fneue-webtechnologien","i-lucide-sparkles",122,[],{"id":1487,"title":1488,"url":1497,"target":1429,"icon":1498,"isActive":14,"type":1431,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":1499,"portfolioId":10,"children":1500},"item-20",{"de":1489,"en":1490,"es":1491,"fr":1492,"it":1493,"ru":1494,"sr":1495,"zh":1496},"Kontakt","Contact us!","Contacto","Contact","Contatto","Контакт","Контактирајте нас","联系我们！","\u002Fcontact","i-lucide-mail",115,[],{"id":1502,"title":1503,"url":1512,"target":1429,"icon":1513,"isActive":14,"type":1431,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":1514,"portfolioId":10,"children":1515},"item-21",{"de":1504,"en":1505,"es":1506,"fr":1507,"it":1508,"ru":1509,"sr":1510,"zh":1511},"Unsere Arbeit","Our Work","Nuestro trabajo","Nos réalisations","I nostri lavori","Наши работы","Наши радови","文件夹","\u002Fportfolio","i-lucide-briefcase",114,[1516,1529,1543,1549,1561],{"id":1517,"title":1518,"url":1512,"target":1429,"icon":1527,"isActive":14,"type":1431,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":1514,"portfolioId":10,"children":1528},"item-24",{"de":1519,"en":1520,"es":1521,"fr":1522,"it":1523,"ru":1524,"sr":1525,"zh":1526},"Alle Projekte","All Projects","Todos los proyectos","Tous les projets","Tutti i progetti","Все проекты","Сви пројекти","所有项目","i-lucide-grid-3x3",[],{"id":1530,"title":1531,"url":1539,"target":1429,"icon":1540,"isActive":14,"type":1541,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":1542},"item-29",{"de":1532,"en":1533,"es":1534,"fr":1535,"it":1536,"ru":1537,"sr":1538,"zh":1511},"Local Roots, Global Reach","Local Roots - Global Reach","Empresa local ","Entreprise locale","Azienda locale","Местная компания","Локално предузеће глобално тржиште","\u002Fportfolio\u002Flocal-roots-global-reach-communication-media-systems-for-modern-business","i-lucide-folder","custom",[],{"id":1544,"title":1545,"url":1547,"target":1429,"icon":1540,"isActive":14,"type":1541,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":1548},"item-28",{"de":1546,"en":1546,"es":1546,"fr":1546,"it":1546,"ru":1546,"sr":1546,"zh":1546},"Solr Suggester","\u002Fportfolio\u002Fsolr-fuzzy-suggester-und-solr-infix-suggester-abfrage-ueber-ajax-und-filterung",[],{"id":1550,"title":1551,"url":1559,"target":1429,"icon":1540,"isActive":14,"type":1541,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":1560},"item-27",{"de":1552,"en":1553,"es":1554,"fr":1555,"it":1556,"ru":1557,"sr":1558,"zh":1553},"Firmenwebseite SEO","Company Website SEO","Sitio web corporativo SEO","Site web d’entreprise SEO","Sito web aziendale SEO","Корпоративный сайт SEO","Пословна веб-страница SEO","\u002Fportfolio\u002Fseo-sem-branding-mobile-webseite-muenchen",[],{"id":1562,"title":1563,"url":1571,"target":1429,"icon":1540,"isActive":14,"type":1541,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":1572},"item-31",{"de":1564,"en":1565,"es":1566,"fr":1567,"it":1568,"ru":1569,"sr":1570,"zh":1565},"Digitalisierungsportal","Digitalization Portal","Portal de digitalización","Portail de numérisation","Portale di digitalizzazione","Портал цифровизации","Портал за дигитализацију","\u002Fportfolio\u002Fdigitalisierungsportal-archiv-museum-bibliothek-ead-lido-mets-mods",[]]