[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"portal-settings:stajic:en":3,"public-menus:all":38,"post:ai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context:en":205,"related:post:ai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context:en:1":1052},{"statusCode":4,"data":5,"message":37},200,{"tenantId":6,"lang":7,"defaultLang":8,"siteUrl":9,"contactEmail":10,"brandName":11,"logoUrl":12,"siteName":11,"siteDescription":13,"ogImage":10,"robotsIndex":14,"socialLinks":10,"reservedSlugs":10,"seoPolicy":15},"stajic","en","de","https:\u002F\u002Fstajic.de",null,"Stajic Platform","\u002FLogo_Planet.svg","Stajic Portal",true,{"branding":16,"relatedContent":17,"crossDomainLinks":18},{"logoUrl":12},{"enabled":14},[19,22,25,28,31,34],{"url":20,"label":21,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Ffigure.rocks","figure.rocks",{"url":23,"label":24,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Floving.rocks","loving.rocks",{"url":26,"label":27,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.com","bazify.com",{"url":29,"label":30,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.de","bazify.de",{"url":32,"label":33,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.at","bazify.at",{"url":35,"label":36,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.ba","bazify.ba","Portal settings resolved",[39,45],{"id":40,"name":41,"location":42,"isActive":14,"isDefault":43,"items":44},1,"main-navigation","header",false,[],{"id":46,"name":47,"location":48,"isActive":14,"isDefault":14,"items":49},4,"main-menu","sidebar",[50,66,79,93,103,118,133],{"id":51,"title":52,"url":60,"target":61,"icon":62,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":64,"portfolioId":10,"children":65},"item-18",{"de":53,"en":54,"es":55,"fr":56,"it":54,"ru":57,"sr":58,"zh":59},"Startseite","Home","Inicio","Accueil","Главная","Почетна","首页","\u002Ffull-stack-web-developer-munich-performance-seo-and-maintainable-builds","_self","i-lucide-home","page",111,[],{"id":67,"title":68,"url":75,"target":61,"icon":76,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":77,"portfolioId":10,"children":78},"item-22",{"de":69,"en":69,"es":70,"fr":69,"it":71,"ru":72,"sr":73,"zh":74},"Vision","Visión","Visione","Видение","Визија","想象","\u002Fueber-uns-webdesign-muenchen-webaplikation","i-lucide-eye",113,[],{"id":80,"title":81,"url":89,"target":61,"icon":90,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":91,"portfolioId":10,"children":92},"item-19",{"de":82,"en":83,"es":84,"fr":83,"it":85,"ru":86,"sr":87,"zh":88},"Leistungen","Services","Servicios","Servizi","Услуги","Услуге","服务","\u002Fservices-dienstleistungen-muenchen","i-lucide-wrench",116,[],{"id":94,"title":95,"url":99,"target":61,"icon":100,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":101,"portfolioId":10,"children":102},"item-23",{"de":96,"en":96,"es":96,"fr":96,"it":96,"ru":97,"sr":97,"zh":98},"Blog","Блог","博客","\u002Fblog","i-lucide-book-open",112,[],{"id":104,"title":105,"url":114,"target":61,"icon":115,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":116,"portfolioId":10,"children":117},"item-32",{"de":106,"en":107,"es":108,"fr":109,"it":110,"ru":111,"sr":112,"zh":113},"Neue Technologien","New Technologies","Nuevas tecnologías","Nouvelles technologies","Nuove tecnologie","Новые технологии","Нове технологије","新技术！","\u002Fneue-webtechnologien","i-lucide-sparkles",122,[],{"id":119,"title":120,"url":129,"target":61,"icon":130,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":131,"portfolioId":10,"children":132},"item-20",{"de":121,"en":122,"es":123,"fr":124,"it":125,"ru":126,"sr":127,"zh":128},"Kontakt","Contact us!","Contacto","Contact","Contatto","Контакт","Контактирајте нас","联系我们！","\u002Fcontact","i-lucide-mail",115,[],{"id":134,"title":135,"url":144,"target":61,"icon":145,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":147},"item-21",{"de":136,"en":137,"es":138,"fr":139,"it":140,"ru":141,"sr":142,"zh":143},"Unsere Arbeit","Our Work","Nuestro trabajo","Nos réalisations","I nostri lavori","Наши работы","Наши радови","文件夹","\u002Fportfolio","i-lucide-briefcase",114,[148,161,175,181,193],{"id":149,"title":150,"url":144,"target":61,"icon":159,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":160},"item-24",{"de":151,"en":152,"es":153,"fr":154,"it":155,"ru":156,"sr":157,"zh":158},"Alle Projekte","All Projects","Todos los proyectos","Tous les projets","Tutti i progetti","Все проекты","Сви пројекти","所有项目","i-lucide-grid-3x3",[],{"id":162,"title":163,"url":171,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":174},"item-29",{"de":164,"en":165,"es":166,"fr":167,"it":168,"ru":169,"sr":170,"zh":143},"Local Roots, Global Reach","Local Roots - Global Reach","Empresa local ","Entreprise locale","Azienda locale","Местная компания","Локално предузеће глобално тржиште","\u002Fportfolio\u002Flocal-roots-global-reach-communication-media-systems-for-modern-business","i-lucide-folder","custom",[],{"id":176,"title":177,"url":179,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":180},"item-28",{"de":178,"en":178,"es":178,"fr":178,"it":178,"ru":178,"sr":178,"zh":178},"Solr Suggester","\u002Fportfolio\u002Fsolr-fuzzy-suggester-und-solr-infix-suggester-abfrage-ueber-ajax-und-filterung",[],{"id":182,"title":183,"url":191,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":192},"item-27",{"de":184,"en":185,"es":186,"fr":187,"it":188,"ru":189,"sr":190,"zh":185},"Firmenwebseite SEO","Company Website SEO","Sitio web corporativo SEO","Site web d’entreprise SEO","Sito web aziendale SEO","Корпоративный сайт SEO","Пословна веб-страница SEO","\u002Fportfolio\u002Fseo-sem-branding-mobile-webseite-muenchen",[],{"id":194,"title":195,"url":203,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":204},"item-31",{"de":196,"en":197,"es":198,"fr":199,"it":200,"ru":201,"sr":202,"zh":197},"Digitalisierungsportal","Digitalization Portal","Portal de digitalización","Portail de numérisation","Portale di digitalizzazione","Портал цифровизации","Портал за дигитализацију","\u002Fportfolio\u002Fdigitalisierungsportal-archiv-museum-bibliothek-ead-lido-mets-mods",[],{"statusCode":4,"data":206,"message":1051},{"id":207,"title":208,"slug":209,"content":210,"contentJson":211,"excerpt":753,"featuredImage":754,"featuredImageAlt":755,"featuredImageCaption":10,"featuredImageTitle":10,"featuredImageCopyright":10,"featuredImageAuthor":10,"featuredImageSourceUrl":10,"featuredImageLicense":10,"featuredImageIsAiGenerated":43,"status":756,"publishedAt":757,"createdAt":758,"updatedAt":759,"seoLocalePaths":760,"categories":769,"author":782,"translations":787},"468","AI Agent Memory Is Not RAG: How to Separate Memory, Retrieval, State and Context","ai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context","{\"time\":1790350647507,\"blocks\":[{\"id\":\"_4kVYTpqbe\",\"type\":\"tableOfContents\",\"data\":{\"title\":\"Contents\",\"minLevel\":2,\"maxLevel\":3},\"tunes\":{}},{\"id\":\"intro\",\"type\":\"paragraph\",\"data\":{\"text\":\"AI agent memory, retrieval-augmented generation (RAG), runtime state, and model context are often discussed as if they were interchangeable. They are not. Collapsing them into one concept makes agent systems harder to reason about, harder to debug, and easier to make stale or unsafe.\"},\"tunes\":{}},{\"id\":\"direct\",\"type\":\"callout\",\"data\":{\"variant\":\"info\",\"title\":\"Direct answer\",\"body\":\"\u003Cstrong>RAG is not agent memory.\u003C\u002Fstrong> RAG is a retrieval pattern: it selects information that may be useful for the current model call. Memory is persistent information derived from prior interaction or experience and managed across time. State represents what is currently true about the running task or environment. Context is the information actually made available to the model for the current inference. A production agent may use all four, but they solve different problems.\"},\"tunes\":{}},{\"id\":\"model-note\",\"type\":\"callout\",\"data\":{\"variant\":\"note\",\"title\":\"About the model used in this article\",\"body\":\"The four-layer separation below is a practical architecture model, not a formal industry standard. Vendors and research papers use overlapping terminology. The purpose is operational: to make design decisions, ownership, failure analysis, and testing clearer.\"},\"tunes\":{}},{\"id\":\"h-category\",\"type\":\"header\",\"data\":{\"text\":\"The category error: treating every persistent-looking thing as memory\",\"level\":2},\"tunes\":{}},{\"id\":\"p-cat-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A vector database can store conversation fragments. A session object can carry recent turns. A database row can hold the current workflow status. A summarizer can compress previous work. A retriever can fetch old evidence. All of these can make an agent appear to “remember,” but they do not have the same semantics.\"},\"tunes\":{}},{\"id\":\"p-cat-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The distinction matters because the required correctness rules are different. Current state must be authoritative and fresh. Memory needs lifecycle rules for writing, revising, forgetting, and conflict handling. Retrieval needs relevance and evidence-selection quality. Context needs token-budget discipline and protection against irrelevant or conflicting material.\"},\"tunes\":{}},{\"id\":\"h-layers\",\"type\":\"header\",\"data\":{\"text\":\"A four-layer architecture: state, memory, retrieval, context\",\"level\":2},\"tunes\":{}},{\"id\":\"table-layers\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Layer\",\"Core question\",\"Typical examples\",\"Primary correctness concern\"],[\"State\",\"What is true now?\",\"Task status, cart contents, workflow step, active permissions, current game state\",\"Freshness and authority\"],[\"Memory\",\"What from the past should persist?\",\"User preference, prior decision, learned constraint, resolved failure, durable project fact\",\"Lifecycle, revision, provenance, forgetting\"],[\"Retrieval\",\"What information should be selected now?\",\"Vector search, keyword search, graph lookup, reranking, document search\",\"Relevance and evidence selection\"],[\"Context\",\"What does the model see for this call?\",\"System instructions, current request, retrieved passages, tool results, summaries\",\"Utility per token, ordering, consistency, noise\"]]},\"tunes\":{}},{\"id\":\"h-state\",\"type\":\"header\",\"data\":{\"text\":\"1. State: what is true now\",\"level\":3},\"tunes\":{}},{\"id\":\"p-state-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"State belongs to the running system, not to the model's recollection. If an order is cancelled, a deployment is paused, a user loses a permission, or a task moves from “in progress” to “approved,” the authoritative value should come from the system that owns that fact.\"},\"tunes\":{}},{\"id\":\"p-state-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"A dangerous design is to let an old conversation summary become a substitute for current state. The agent may accurately remember that the order was active yesterday and still be wrong today. State therefore needs explicit ownership, versioning or timestamps where relevant, and a path to re-read the source of truth before consequential actions.\"},\"tunes\":{}},{\"id\":\"h-memory\",\"type\":\"header\",\"data\":{\"text\":\"2. Memory: what from the past should persist\",\"level\":3},\"tunes\":{}},{\"id\":\"p-memory-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Memory is not simply “everything we can store.” A useful memory layer decides what deserves persistence, in what form, for how long, with what provenance, and under what conditions it must be revised or removed.\"},\"tunes\":{}},{\"id\":\"p-memory-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Recent agent-memory research increasingly treats raw transcript storage as insufficient. Microsoft's PlugMem work focuses on transforming raw interaction histories into structured reusable knowledge. Memora separates rich stored content from lighter abstractions and retrieval cues so that long-horizon systems do not have to choose between detail and scalable access.\"},\"tunes\":{}},{\"id\":\"h-retrieval\",\"type\":\"header\",\"data\":{\"text\":\"3. Retrieval: what should be selected now\",\"level\":3},\"tunes\":{}},{\"id\":\"p-ret-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Retrieval is a selection mechanism. It can search external documents, internal knowledge bases, stored memories, logs, graphs, databases, or mixed sources. RAG normally sits here: retrieve evidence, place selected material into the model's working input, then generate an answer.\"},\"tunes\":{}},{\"id\":\"p-ret-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"That mechanism does not become memory merely because the retrieved corpus contains past interactions. The same retriever can search policy documents that the agent never experienced, product data from another system, or a user's prior decisions. Retrieval describes how information is selected; memory describes why some information persists across time and how that persistence is governed.\"},\"tunes\":{}},{\"id\":\"h-context\",\"type\":\"header\",\"data\":{\"text\":\"4. Context: what the model can actually use right now\",\"level\":3},\"tunes\":{}},{\"id\":\"p-ctx-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Context is the model-facing layer. Anthropic describes context engineering as deciding what configuration of context is most likely to produce the desired behaviour, with context being the tokens available to the model during generation. OpenAI's session-memory guidance similarly treats trimming and compression as context-management techniques for long-running agent interactions.\"},\"tunes\":{}},{\"id\":\"p-ctx-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"This is why a system can have excellent memory and still fail. The relevant memory may exist but not be retrieved. It may be retrieved but placed into context next to stronger conflicting text. It may be compressed until the decisive detail disappears. Or the model may receive so much material that useful evidence is diluted by noise.\"},\"tunes\":{}},{\"id\":\"h-flow\",\"type\":\"header\",\"data\":{\"text\":\"How the layers interact\",\"level\":2},\"tunes\":{}},{\"id\":\"flow\",\"type\":\"processFlow\",\"data\":{\"title\":\"One possible production flow\",\"orientation\":\"auto\",\"steps\":[{\"label\":\"1. Read authoritative state\",\"description\":\"Load current task, user, system, or environment facts from the systems that own them.\"},{\"label\":\"2. Identify memory needs\",\"description\":\"Determine whether prior decisions, preferences, lessons, or long-term constraints are relevant.\"},{\"label\":\"3. Retrieve evidence\",\"description\":\"Search memory and external knowledge using semantic, lexical, graph, structured, or hybrid retrieval.\"},{\"label\":\"4. Build context\",\"description\":\"Assemble instructions, current state, selected evidence, and compacted history within the model's usable context.\"},{\"label\":\"5. Generate or act\",\"description\":\"The model reasons over the assembled context and produces an answer, plan, or tool call.\"},{\"label\":\"6. Validate and write back\",\"description\":\"Validate consequential outputs, update authoritative state where permitted, and persist only memories that pass the write policy.\"}]},\"tunes\":{}},{\"id\":\"h-rag\",\"type\":\"header\",\"data\":{\"text\":\"Why RAG is not memory\",\"level\":2},\"tunes\":{}},{\"id\":\"p-rag-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The simplest test is this: a RAG system can retrieve information the agent has never seen before. That alone shows that retrieval and memory are different abstractions.\"},\"tunes\":{}},{\"id\":\"p-rag-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"RAG answers: “Which evidence should I fetch?” A memory system must additionally answer questions such as: “Should this event become durable knowledge?”, “Does this new information supersede an older memory?”, “Can this memory still be trusted?”, “Who is allowed to read it?”, and “When should it be forgotten?”\"},\"tunes\":{}},{\"id\":\"rag-trap\",\"type\":\"callout\",\"data\":{\"variant\":\"warning\",\"title\":\"A common design trap\",\"body\":\"If every conversation turn is embedded into a vector store and later retrieved by similarity, the system has persistent lookup, but not necessarily a well-governed memory architecture. Persistence alone does not define memory quality.\"},\"tunes\":{}},{\"id\":\"h-test\",\"type\":\"header\",\"data\":{\"text\":\"The four-layer separation test\",\"level\":2},\"tunes\":{}},{\"id\":\"p-test\",\"type\":\"paragraph\",\"data\":{\"text\":\"When a feature is called “memory,” ask the following four questions. The answers usually reveal which layer is actually involved.\"},\"tunes\":{}},{\"id\":\"table-test\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Question\",\"If yes, you are primarily dealing with\"],[\"Does this represent the current authoritative condition of the task or environment?\",\"State\"],[\"Must this information survive the current run because it captures useful prior experience, preference, or decision?\",\"Memory\"],[\"Is the main problem deciding which stored or external information is relevant to the current request?\",\"Retrieval\"],[\"Is the main problem deciding what information to place inside the current model call?\",\"Context\"]]},\"tunes\":{}},{\"id\":\"p-test-note\",\"type\":\"paragraph\",\"data\":{\"text\":\"A single component can participate in more than one layer. A database may store both state and memory. A vector index may retrieve both external knowledge and memories. The separation is semantic, not necessarily physical.\"},\"tunes\":{}},{\"id\":\"h-fail\",\"type\":\"header\",\"data\":{\"text\":\"Failure modes caused by collapsing the layers\",\"level\":2},\"tunes\":{}},{\"id\":\"table-fail\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Failure mode\",\"What happened\",\"Result\"],[\"Stale state disguised as memory\",\"An old summary is trusted instead of re-reading the authoritative system\",\"The agent acts on facts that were once true\"],[\"Memory treated as immutable fact\",\"A prior preference or decision is stored without revision rules\",\"Superseded information keeps influencing future answers\"],[\"Retrieval hit treated as truth\",\"High similarity is mistaken for factual authority\",\"Relevant-looking but incorrect evidence dominates\"],[\"Context overload\",\"Too many retrieved passages, memories, logs, and instructions are injected\",\"The decisive evidence is diluted or contradicted\"],[\"Uncontrolled memory write\",\"Model-generated interpretations are stored automatically as durable memory\",\"Errors become persistent and self-reinforcing\"],[\"No provenance boundary\",\"The system cannot distinguish user statement, source fact, model inference, and generated summary\",\"Later retrieval loses the evidential status of the information\"]]},\"tunes\":{}},{\"id\":\"h-decision\",\"type\":\"header\",\"data\":{\"text\":\"What should be remembered, retrieved, recomputed, or re-read?\",\"level\":2},\"tunes\":{}},{\"id\":\"table-decision\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Information type\",\"Preferred treatment\",\"Reason\"],[\"Current permission, order status, inventory, workflow status\",\"Re-read authoritative state\",\"Freshness matters more than recollection\"],[\"Stable user preference explicitly provided by the user\",\"Memory, with edit\u002Fdelete semantics\",\"Useful across sessions and owned by the user\"],[\"Decision made during a long-running project\",\"Memory with timestamp, provenance, and supersession rules\",\"The history matters, but decisions can change\"],[\"Product specification or public policy document\",\"Retrieve from source\",\"External knowledge should remain tied to its evidence\"],[\"Derived metric that can be cheaply recalculated\",\"Recompute\",\"Avoid persisting stale derived values\"],[\"Long raw tool output\",\"Store externally; retrieve or summarize when needed\",\"Do not consume context permanently\"],[\"Model hypothesis or uncertain interpretation\",\"Do not promote automatically to durable memory\",\"Inference is not equivalent to fact\"]]},\"tunes\":{}},{\"id\":\"h-write\",\"type\":\"header\",\"data\":{\"text\":\"A memory system needs a write policy, not only a retrieval policy\",\"level\":2},\"tunes\":{}},{\"id\":\"p-write-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"RAG architecture discussions often focus on retrieval quality: chunking, embeddings, reranking, hybrid search, and grounding. Long-term memory introduces another side of the problem: what is allowed to enter the persistent store in the first place?\"},\"tunes\":{}},{\"id\":\"p-write-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"For durable agent memory, a practical write policy should classify the candidate memory, preserve provenance, detect conflicts with existing entries, distinguish observation from inference, define sensitivity and access scope, and decide whether the information should expire, be revised, or require user confirmation.\"},\"tunes\":{}},{\"id\":\"write-tip\",\"type\":\"callout\",\"data\":{\"variant\":\"tip\",\"title\":\"Design principle\",\"body\":\"The more expensive a wrong memory becomes over time, the stricter the write policy should be. A bad retrieval affects one answer. A bad durable memory can affect every future answer that retrieves it.\"},\"tunes\":{}},{\"id\":\"h-prov\",\"type\":\"header\",\"data\":{\"text\":\"Provenance is the bridge between memory and reliable evidence\",\"level\":2},\"tunes\":{}},{\"id\":\"p-prov-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A memory entry should ideally retain enough provenance to answer: where did this come from, when was it observed, who or what asserted it, was it user-provided or model-inferred, what source supported it, and has anything superseded it?\"},\"tunes\":{}},{\"id\":\"p-prov-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Without provenance, a compressed memory can become more authoritative than the evidence that created it. This is especially risky in long-running agents where summaries and abstractions are repeatedly reused. The system may preserve the conclusion while losing the conditions under which the conclusion was valid.\"},\"tunes\":{}},{\"id\":\"h-budget\",\"type\":\"header\",\"data\":{\"text\":\"More memory does not mean more context\",\"level\":2},\"tunes\":{}},{\"id\":\"p-budget-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A long-lived agent may accumulate gigabytes of state, history, documents, and learned information. The model does not need — and usually should not receive — all of it for each step. The purpose of retrieval, summarization, compaction, and structured memory is to convert a large persistent information space into a small, relevant working context.\"},\"tunes\":{}},{\"id\":\"p-budget-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"This is also why larger context windows do not eliminate memory architecture. Capacity reduces some pressure, but it does not solve freshness, authority, conflicting evidence, privacy scope, write quality, revision, or deciding what deserves attention.\"},\"tunes\":{}},{\"id\":\"h-check\",\"type\":\"header\",\"data\":{\"text\":\"Production design checklist\",\"level\":2},\"tunes\":{}},{\"id\":\"checklist\",\"type\":\"list\",\"data\":{\"style\":\"unordered\",\"meta\":{},\"items\":[\"Define which systems own authoritative runtime state.\",\"Define which information is eligible to become durable memory.\",\"Keep user-provided facts, external evidence, and model inference distinguishable.\",\"Attach timestamps, provenance, scope, and revision semantics to important memories.\",\"Treat retrieval relevance as different from factual authority.\",\"Build context intentionally instead of injecting all retrieved material.\",\"Re-read volatile facts instead of trusting old memories.\",\"Recompute cheap derived values when staleness would be costly.\",\"Test memory writes as carefully as memory reads.\",\"Measure failures separately: state error, memory error, retrieval error, context-construction error, reasoning error, and action error.\"]},\"tunes\":{}},{\"id\":\"h-change\",\"type\":\"header\",\"data\":{\"text\":\"What would change this answer?\",\"level\":2},\"tunes\":{}},{\"id\":\"p-change-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The boundary between these layers can move as agent platforms evolve. A vendor may offer a managed memory service that internally performs storage, revision, retrieval, summarization, and context construction. That can collapse implementation components, but it does not eliminate the architectural questions. You still need to know whether a returned item is current state, persistent memory, retrieved evidence, or simply text placed into context.\"},\"tunes\":{}},{\"id\":\"p-change-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The recommendation would also change for systems with no cross-session continuity, systems where every task starts from a clean immutable corpus, or tightly bounded workflows where all relevant state fits safely inside one call. In those cases, a dedicated long-term memory layer may add complexity without enough value.\"},\"tunes\":{}},{\"id\":\"h-limit\",\"type\":\"header\",\"data\":{\"text\":\"Limitations\",\"level\":2},\"tunes\":{}},{\"id\":\"p-limit\",\"type\":\"paragraph\",\"data\":{\"text\":\"Terminology in agent systems is still moving quickly. Some frameworks call conversation history “memory,” others use “session,” “checkpoint,” “store,” “context,” or “state.” Research systems also define memory at different levels, from persistent lookup to learned internal adaptation. The model in this article deliberately separates operational responsibilities rather than trying to impose one universal vocabulary.\"},\"tunes\":{}},{\"id\":\"h-conclusion\",\"type\":\"header\",\"data\":{\"text\":\"Conclusion\",\"level\":2},\"tunes\":{}},{\"id\":\"p-conclusion-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The useful question is not “Does this agent have memory?” It is: What is state, what is persisted from experience, how is relevant information retrieved, and what finally reaches the model as context?\"},\"tunes\":{}},{\"id\":\"p-conclusion-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Once those responsibilities are separated, design choices become easier to test. Stale facts can be traced to state ownership. Bad recall can be traced to memory lifecycle or retrieval. Overloaded prompts can be traced to context construction. Persistent hallucinations can be traced to write policy and provenance. RAG remains an important tool, but it is only one part of a reliable long-running agent architecture.\"},\"tunes\":{}},{\"id\":\"h-faq\",\"type\":\"header\",\"data\":{\"text\":\"FAQ\",\"level\":2},\"tunes\":{}},{\"id\":\"faq\",\"type\":\"faq\",\"data\":{\"title\":\"AI agent memory, RAG, state and context\",\"items\":[{\"id\":\"faq1\",\"question\":\"Is RAG the same as AI agent memory?\",\"answer\":\"No. RAG is primarily a retrieval pattern that selects information for a model call. Memory concerns what information from prior interactions or experience persists across time and how that information is governed.\"},{\"id\":\"faq2\",\"question\":\"Is a vector database an agent memory?\",\"answer\":\"It can be part of one, but a vector database by itself is a storage and retrieval component. A production memory architecture also needs decisions about what to store, provenance, revision, conflicts, access, expiration, and forgetting.\"},{\"id\":\"faq3\",\"question\":\"Does a larger context window remove the need for memory?\",\"answer\":\"Not necessarily. Larger context helps with capacity, but it does not solve persistent knowledge across sessions, freshness, provenance, privacy scope, revision, or deciding what should be reused later.\"},{\"id\":\"faq4\",\"question\":\"Should current application state be stored as memory?\",\"answer\":\"Usually the authoritative application or domain system should remain the source of truth for volatile state. Memory may record the history or significance of state changes, but consequential actions should re-read current authoritative values.\"}]},\"tunes\":{}},{\"id\":\"h-glossary\",\"type\":\"header\",\"data\":{\"text\":\"Glossary\",\"level\":2},\"tunes\":{}},{\"id\":\"glossary\",\"type\":\"glossary\",\"data\":{\"title\":\"Key terms\",\"entries\":[{\"term\":\"State\",\"definition\":\"The current authoritative condition of a task, application, user, workflow, or environment.\",\"anchor\":\"state\"},{\"term\":\"Memory\",\"definition\":\"Information from prior experience or interaction that persists because it may be useful later and is subject to lifecycle rules.\",\"anchor\":\"memory\"},{\"term\":\"Retrieval\",\"definition\":\"The mechanism used to select potentially relevant information from memory, external knowledge, databases, graphs, or other stores.\",\"anchor\":\"retrieval\"},{\"term\":\"Context\",\"definition\":\"The information actually available to the language model during a particular inference or generation step.\",\"anchor\":\"context\"},{\"term\":\"RAG\",\"definition\":\"Retrieval-augmented generation: a pattern in which external or stored information is retrieved and supplied to a generative model to improve the current output.\",\"anchor\":\"rag\"},{\"term\":\"Provenance\",\"definition\":\"Metadata describing where information came from, when it was observed, who or what asserted it, and how it was transformed.\",\"anchor\":\"provenance\"}]},\"tunes\":{}},{\"id\":\"h-sources\",\"type\":\"header\",\"data\":{\"text\":\"Primary sources and further reading\",\"level\":2},\"tunes\":{}},{\"id\":\"openai-session\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdevelopers.openai.com\u002Fcookbook\u002Fexamples\u002Fagents_sdk\u002Fsession_memory\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — Context Engineering: Short-Term Memory Management with Sessions\",\"description\":\"OpenAI guidance on trimming and compression for long-running agent context.\"}},\"tunes\":{}},{\"id\":\"openai-sandbox\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fagents\u002Fsandboxes\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — Sandbox Agents\",\"description\":\"Documentation showing persistent memory as a capability with progressive disclosure and read\u002Fwrite behaviour.\"}},\"tunes\":{}},{\"id\":\"anthropic-context\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Feffective-context-engineering-for-ai-agents\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Anthropic — Effective Context Engineering for AI Agents\",\"description\":\"Engineering guidance on curating finite model context for reliable agent behaviour.\"}},\"tunes\":{}},{\"id\":\"ms-memora\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Fblog\u002Fmemora-a-harmonic-memory-representation-balancing-abstraction-and-specificity\u002F\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Microsoft Research — Memora\",\"description\":\"Research on balancing abstraction and specificity in long-horizon agent memory.\"}},\"tunes\":{}},{\"id\":\"ms-plugmem\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Fblog\u002Ffrom-raw-interaction-to-reusable-knowledge-rethinking-memory-for-ai-agents\u002F\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Microsoft Research — PlugMem\",\"description\":\"Research on converting raw agent interaction histories into reusable structured knowledge.\"}},\"tunes\":{}},{\"id\":\"ms-ace\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Fpublication\u002Fagentic-context-engineering-evolving-contexts-for-self-improving-language-models\u002F\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Microsoft Research — Agentic Context Engineering (ACE)\",\"description\":\"Research on evolving context as structured playbooks rather than repeatedly rewriting or compressing everything.\"}},\"tunes\":{}}],\"version\":\"2.31.6\"}",{"time":212,"blocks":213,"version":752},1790350647507,[214,222,228,236,243,248,253,258,263,294,299,304,309,314,319,324,329,334,339,344,349,354,359,385,390,395,400,407,412,417,433,438,443,476,481,518,523,528,533,540,545,550,555,560,565,570,575,593,598,603,608,613,618,623,628,633,638,660,665,691,696,707,716,725,734,743],{"id":215,"data":216,"type":220,"tunes":221},"_4kVYTpqbe",{"title":217,"maxLevel":218,"minLevel":219},"Contents",3,2,"tableOfContents",{},{"id":223,"data":224,"type":226,"tunes":227},"intro",{"text":225},"AI agent memory, retrieval-augmented generation (RAG), runtime state, and model context are often discussed as if they were interchangeable. They are not. Collapsing them into one concept makes agent systems harder to reason about, harder to debug, and easier to make stale or unsafe.","paragraph",{},{"id":229,"data":230,"type":234,"tunes":235},"direct",{"body":231,"title":232,"variant":233},"\u003Cstrong>RAG is not agent memory.\u003C\u002Fstrong> RAG is a retrieval pattern: it selects information that may be useful for the current model call. Memory is persistent information derived from prior interaction or experience and managed across time. State represents what is currently true about the running task or environment. Context is the information actually made available to the model for the current inference. A production agent may use all four, but they solve different problems.","Direct answer","info","callout",{},{"id":237,"data":238,"type":234,"tunes":242},"model-note",{"body":239,"title":240,"variant":241},"The four-layer separation below is a practical architecture model, not a formal industry standard. Vendors and research papers use overlapping terminology. The purpose is operational: to make design decisions, ownership, failure analysis, and testing clearer.","About the model used in this article","note",{},{"id":244,"data":245,"type":42,"tunes":247},"h-category",{"text":246,"level":219},"The category error: treating every persistent-looking thing as memory",{},{"id":249,"data":250,"type":226,"tunes":252},"p-cat-1",{"text":251},"A vector database can store conversation fragments. A session object can carry recent turns. A database row can hold the current workflow status. A summarizer can compress previous work. A retriever can fetch old evidence. All of these can make an agent appear to “remember,” but they do not have the same semantics.",{},{"id":254,"data":255,"type":226,"tunes":257},"p-cat-2",{"text":256},"The distinction matters because the required correctness rules are different. Current state must be authoritative and fresh. Memory needs lifecycle rules for writing, revising, forgetting, and conflict handling. Retrieval needs relevance and evidence-selection quality. Context needs token-budget discipline and protection against irrelevant or conflicting material.",{},{"id":259,"data":260,"type":42,"tunes":262},"h-layers",{"text":261,"level":219},"A four-layer architecture: state, memory, retrieval, context",{},{"id":264,"data":265,"type":292,"tunes":293},"table-layers",{"content":266,"stretched":43,"withHeadings":14},[267,272,277,282,287],[268,269,270,271],"Layer","Core question","Typical examples","Primary correctness concern",[273,274,275,276],"State","What is true now?","Task status, cart contents, workflow step, active permissions, current game state","Freshness and authority",[278,279,280,281],"Memory","What from the past should persist?","User preference, prior decision, learned constraint, resolved failure, durable project fact","Lifecycle, revision, provenance, forgetting",[283,284,285,286],"Retrieval","What information should be selected now?","Vector search, keyword search, graph lookup, reranking, document search","Relevance and evidence selection",[288,289,290,291],"Context","What does the model see for this call?","System instructions, current request, retrieved passages, tool results, summaries","Utility per token, ordering, consistency, noise","table",{},{"id":295,"data":296,"type":42,"tunes":298},"h-state",{"text":297,"level":218},"1. State: what is true now",{},{"id":300,"data":301,"type":226,"tunes":303},"p-state-1",{"text":302},"State belongs to the running system, not to the model's recollection. If an order is cancelled, a deployment is paused, a user loses a permission, or a task moves from “in progress” to “approved,” the authoritative value should come from the system that owns that fact.",{},{"id":305,"data":306,"type":226,"tunes":308},"p-state-2",{"text":307},"A dangerous design is to let an old conversation summary become a substitute for current state. The agent may accurately remember that the order was active yesterday and still be wrong today. State therefore needs explicit ownership, versioning or timestamps where relevant, and a path to re-read the source of truth before consequential actions.",{},{"id":310,"data":311,"type":42,"tunes":313},"h-memory",{"text":312,"level":218},"2. Memory: what from the past should persist",{},{"id":315,"data":316,"type":226,"tunes":318},"p-memory-1",{"text":317},"Memory is not simply “everything we can store.” A useful memory layer decides what deserves persistence, in what form, for how long, with what provenance, and under what conditions it must be revised or removed.",{},{"id":320,"data":321,"type":226,"tunes":323},"p-memory-2",{"text":322},"Recent agent-memory research increasingly treats raw transcript storage as insufficient. Microsoft's PlugMem work focuses on transforming raw interaction histories into structured reusable knowledge. Memora separates rich stored content from lighter abstractions and retrieval cues so that long-horizon systems do not have to choose between detail and scalable access.",{},{"id":325,"data":326,"type":42,"tunes":328},"h-retrieval",{"text":327,"level":218},"3. Retrieval: what should be selected now",{},{"id":330,"data":331,"type":226,"tunes":333},"p-ret-1",{"text":332},"Retrieval is a selection mechanism. It can search external documents, internal knowledge bases, stored memories, logs, graphs, databases, or mixed sources. RAG normally sits here: retrieve evidence, place selected material into the model's working input, then generate an answer.",{},{"id":335,"data":336,"type":226,"tunes":338},"p-ret-2",{"text":337},"That mechanism does not become memory merely because the retrieved corpus contains past interactions. The same retriever can search policy documents that the agent never experienced, product data from another system, or a user's prior decisions. Retrieval describes how information is selected; memory describes why some information persists across time and how that persistence is governed.",{},{"id":340,"data":341,"type":42,"tunes":343},"h-context",{"text":342,"level":218},"4. Context: what the model can actually use right now",{},{"id":345,"data":346,"type":226,"tunes":348},"p-ctx-1",{"text":347},"Context is the model-facing layer. Anthropic describes context engineering as deciding what configuration of context is most likely to produce the desired behaviour, with context being the tokens available to the model during generation. OpenAI's session-memory guidance similarly treats trimming and compression as context-management techniques for long-running agent interactions.",{},{"id":350,"data":351,"type":226,"tunes":353},"p-ctx-2",{"text":352},"This is why a system can have excellent memory and still fail. The relevant memory may exist but not be retrieved. It may be retrieved but placed into context next to stronger conflicting text. It may be compressed until the decisive detail disappears. Or the model may receive so much material that useful evidence is diluted by noise.",{},{"id":355,"data":356,"type":42,"tunes":358},"h-flow",{"text":357,"level":219},"How the layers interact",{},{"id":360,"data":361,"type":383,"tunes":384},"flow",{"steps":362,"title":381,"orientation":382},[363,366,369,372,375,378],{"label":364,"description":365},"1. Read authoritative state","Load current task, user, system, or environment facts from the systems that own them.",{"label":367,"description":368},"2. Identify memory needs","Determine whether prior decisions, preferences, lessons, or long-term constraints are relevant.",{"label":370,"description":371},"3. Retrieve evidence","Search memory and external knowledge using semantic, lexical, graph, structured, or hybrid retrieval.",{"label":373,"description":374},"4. Build context","Assemble instructions, current state, selected evidence, and compacted history within the model's usable context.",{"label":376,"description":377},"5. Generate or act","The model reasons over the assembled context and produces an answer, plan, or tool call.",{"label":379,"description":380},"6. Validate and write back","Validate consequential outputs, update authoritative state where permitted, and persist only memories that pass the write policy.","One possible production flow","auto","processFlow",{},{"id":386,"data":387,"type":42,"tunes":389},"h-rag",{"text":388,"level":219},"Why RAG is not memory",{},{"id":391,"data":392,"type":226,"tunes":394},"p-rag-1",{"text":393},"The simplest test is this: a RAG system can retrieve information the agent has never seen before. That alone shows that retrieval and memory are different abstractions.",{},{"id":396,"data":397,"type":226,"tunes":399},"p-rag-2",{"text":398},"RAG answers: “Which evidence should I fetch?” A memory system must additionally answer questions such as: “Should this event become durable knowledge?”, “Does this new information supersede an older memory?”, “Can this memory still be trusted?”, “Who is allowed to read it?”, and “When should it be forgotten?”",{},{"id":401,"data":402,"type":234,"tunes":406},"rag-trap",{"body":403,"title":404,"variant":405},"If every conversation turn is embedded into a vector store and later retrieved by similarity, the system has persistent lookup, but not necessarily a well-governed memory architecture. Persistence alone does not define memory quality.","A common design trap","warning",{},{"id":408,"data":409,"type":42,"tunes":411},"h-test",{"text":410,"level":219},"The four-layer separation test",{},{"id":413,"data":414,"type":226,"tunes":416},"p-test",{"text":415},"When a feature is called “memory,” ask the following four questions. The answers usually reveal which layer is actually involved.",{},{"id":418,"data":419,"type":292,"tunes":432},"table-test",{"content":420,"stretched":43,"withHeadings":14},[421,424,426,428,430],[422,423],"Question","If yes, you are primarily dealing with",[425,273],"Does this represent the current authoritative condition of the task or environment?",[427,278],"Must this information survive the current run because it captures useful prior experience, preference, or decision?",[429,283],"Is the main problem deciding which stored or external information is relevant to the current request?",[431,288],"Is the main problem deciding what information to place inside the current model call?",{},{"id":434,"data":435,"type":226,"tunes":437},"p-test-note",{"text":436},"A single component can participate in more than one layer. A database may store both state and memory. A vector index may retrieve both external knowledge and memories. The separation is semantic, not necessarily physical.",{},{"id":439,"data":440,"type":42,"tunes":442},"h-fail",{"text":441,"level":219},"Failure modes caused by collapsing the layers",{},{"id":444,"data":445,"type":292,"tunes":475},"table-fail",{"content":446,"stretched":43,"withHeadings":14},[447,451,455,459,463,467,471],[448,449,450],"Failure mode","What happened","Result",[452,453,454],"Stale state disguised as memory","An old summary is trusted instead of re-reading the authoritative system","The agent acts on facts that were once true",[456,457,458],"Memory treated as immutable fact","A prior preference or decision is stored without revision rules","Superseded information keeps influencing future answers",[460,461,462],"Retrieval hit treated as truth","High similarity is mistaken for factual authority","Relevant-looking but incorrect evidence dominates",[464,465,466],"Context overload","Too many retrieved passages, memories, logs, and instructions are injected","The decisive evidence is diluted or contradicted",[468,469,470],"Uncontrolled memory write","Model-generated interpretations are stored automatically as durable memory","Errors become persistent and self-reinforcing",[472,473,474],"No provenance boundary","The system cannot distinguish user statement, source fact, model inference, and generated summary","Later retrieval loses the evidential status of the information",{},{"id":477,"data":478,"type":42,"tunes":480},"h-decision",{"text":479,"level":219},"What should be remembered, retrieved, recomputed, or re-read?",{},{"id":482,"data":483,"type":292,"tunes":517},"table-decision",{"content":484,"stretched":43,"withHeadings":14},[485,489,493,497,501,505,509,513],[486,487,488],"Information type","Preferred treatment","Reason",[490,491,492],"Current permission, order status, inventory, workflow status","Re-read authoritative state","Freshness matters more than recollection",[494,495,496],"Stable user preference explicitly provided by the user","Memory, with edit\u002Fdelete semantics","Useful across sessions and owned by the user",[498,499,500],"Decision made during a long-running project","Memory with timestamp, provenance, and supersession rules","The history matters, but decisions can change",[502,503,504],"Product specification or public policy document","Retrieve from source","External knowledge should remain tied to its evidence",[506,507,508],"Derived metric that can be cheaply recalculated","Recompute","Avoid persisting stale derived values",[510,511,512],"Long raw tool output","Store externally; retrieve or summarize when needed","Do not consume context permanently",[514,515,516],"Model hypothesis or uncertain interpretation","Do not promote automatically to durable memory","Inference is not equivalent to fact",{},{"id":519,"data":520,"type":42,"tunes":522},"h-write",{"text":521,"level":219},"A memory system needs a write policy, not only a retrieval policy",{},{"id":524,"data":525,"type":226,"tunes":527},"p-write-1",{"text":526},"RAG architecture discussions often focus on retrieval quality: chunking, embeddings, reranking, hybrid search, and grounding. Long-term memory introduces another side of the problem: what is allowed to enter the persistent store in the first place?",{},{"id":529,"data":530,"type":226,"tunes":532},"p-write-2",{"text":531},"For durable agent memory, a practical write policy should classify the candidate memory, preserve provenance, detect conflicts with existing entries, distinguish observation from inference, define sensitivity and access scope, and decide whether the information should expire, be revised, or require user confirmation.",{},{"id":534,"data":535,"type":234,"tunes":539},"write-tip",{"body":536,"title":537,"variant":538},"The more expensive a wrong memory becomes over time, the stricter the write policy should be. A bad retrieval affects one answer. A bad durable memory can affect every future answer that retrieves it.","Design principle","tip",{},{"id":541,"data":542,"type":42,"tunes":544},"h-prov",{"text":543,"level":219},"Provenance is the bridge between memory and reliable evidence",{},{"id":546,"data":547,"type":226,"tunes":549},"p-prov-1",{"text":548},"A memory entry should ideally retain enough provenance to answer: where did this come from, when was it observed, who or what asserted it, was it user-provided or model-inferred, what source supported it, and has anything superseded it?",{},{"id":551,"data":552,"type":226,"tunes":554},"p-prov-2",{"text":553},"Without provenance, a compressed memory can become more authoritative than the evidence that created it. This is especially risky in long-running agents where summaries and abstractions are repeatedly reused. The system may preserve the conclusion while losing the conditions under which the conclusion was valid.",{},{"id":556,"data":557,"type":42,"tunes":559},"h-budget",{"text":558,"level":219},"More memory does not mean more context",{},{"id":561,"data":562,"type":226,"tunes":564},"p-budget-1",{"text":563},"A long-lived agent may accumulate gigabytes of state, history, documents, and learned information. The model does not need — and usually should not receive — all of it for each step. The purpose of retrieval, summarization, compaction, and structured memory is to convert a large persistent information space into a small, relevant working context.",{},{"id":566,"data":567,"type":226,"tunes":569},"p-budget-2",{"text":568},"This is also why larger context windows do not eliminate memory architecture. Capacity reduces some pressure, but it does not solve freshness, authority, conflicting evidence, privacy scope, write quality, revision, or deciding what deserves attention.",{},{"id":571,"data":572,"type":42,"tunes":574},"h-check",{"text":573,"level":219},"Production design checklist",{},{"id":576,"data":577,"type":591,"tunes":592},"checklist",{"meta":578,"items":579,"style":590},{},[580,581,582,583,584,585,586,587,588,589],"Define which systems own authoritative runtime state.","Define which information is eligible to become durable memory.","Keep user-provided facts, external evidence, and model inference distinguishable.","Attach timestamps, provenance, scope, and revision semantics to important memories.","Treat retrieval relevance as different from factual authority.","Build context intentionally instead of injecting all retrieved material.","Re-read volatile facts instead of trusting old memories.","Recompute cheap derived values when staleness would be costly.","Test memory writes as carefully as memory reads.","Measure failures separately: state error, memory error, retrieval error, context-construction error, reasoning error, and action error.","unordered","list",{},{"id":594,"data":595,"type":42,"tunes":597},"h-change",{"text":596,"level":219},"What would change this answer?",{},{"id":599,"data":600,"type":226,"tunes":602},"p-change-1",{"text":601},"The boundary between these layers can move as agent platforms evolve. A vendor may offer a managed memory service that internally performs storage, revision, retrieval, summarization, and context construction. That can collapse implementation components, but it does not eliminate the architectural questions. You still need to know whether a returned item is current state, persistent memory, retrieved evidence, or simply text placed into context.",{},{"id":604,"data":605,"type":226,"tunes":607},"p-change-2",{"text":606},"The recommendation would also change for systems with no cross-session continuity, systems where every task starts from a clean immutable corpus, or tightly bounded workflows where all relevant state fits safely inside one call. In those cases, a dedicated long-term memory layer may add complexity without enough value.",{},{"id":609,"data":610,"type":42,"tunes":612},"h-limit",{"text":611,"level":219},"Limitations",{},{"id":614,"data":615,"type":226,"tunes":617},"p-limit",{"text":616},"Terminology in agent systems is still moving quickly. Some frameworks call conversation history “memory,” others use “session,” “checkpoint,” “store,” “context,” or “state.” Research systems also define memory at different levels, from persistent lookup to learned internal adaptation. The model in this article deliberately separates operational responsibilities rather than trying to impose one universal vocabulary.",{},{"id":619,"data":620,"type":42,"tunes":622},"h-conclusion",{"text":621,"level":219},"Conclusion",{},{"id":624,"data":625,"type":226,"tunes":627},"p-conclusion-1",{"text":626},"The useful question is not “Does this agent have memory?” It is: What is state, what is persisted from experience, how is relevant information retrieved, and what finally reaches the model as context?",{},{"id":629,"data":630,"type":226,"tunes":632},"p-conclusion-2",{"text":631},"Once those responsibilities are separated, design choices become easier to test. Stale facts can be traced to state ownership. Bad recall can be traced to memory lifecycle or retrieval. Overloaded prompts can be traced to context construction. Persistent hallucinations can be traced to write policy and provenance. RAG remains an important tool, but it is only one part of a reliable long-running agent architecture.",{},{"id":634,"data":635,"type":42,"tunes":637},"h-faq",{"text":636,"level":219},"FAQ",{},{"id":639,"data":640,"type":639,"tunes":659},"faq",{"items":641,"title":658},[642,646,650,654],{"id":643,"answer":644,"question":645},"faq1","No. RAG is primarily a retrieval pattern that selects information for a model call. Memory concerns what information from prior interactions or experience persists across time and how that information is governed.","Is RAG the same as AI agent memory?",{"id":647,"answer":648,"question":649},"faq2","It can be part of one, but a vector database by itself is a storage and retrieval component. A production memory architecture also needs decisions about what to store, provenance, revision, conflicts, access, expiration, and forgetting.","Is a vector database an agent memory?",{"id":651,"answer":652,"question":653},"faq3","Not necessarily. Larger context helps with capacity, but it does not solve persistent knowledge across sessions, freshness, provenance, privacy scope, revision, or deciding what should be reused later.","Does a larger context window remove the need for memory?",{"id":655,"answer":656,"question":657},"faq4","Usually the authoritative application or domain system should remain the source of truth for volatile state. Memory may record the history or significance of state changes, but consequential actions should re-read current authoritative values.","Should current application state be stored as memory?","AI agent memory, RAG, state and context",{},{"id":661,"data":662,"type":42,"tunes":664},"h-glossary",{"text":663,"level":219},"Glossary",{},{"id":666,"data":667,"type":666,"tunes":690},"glossary",{"title":668,"entries":669},"Key terms",[670,673,676,679,682,686],{"term":273,"anchor":671,"definition":672},"state","The current authoritative condition of a task, application, user, workflow, or environment.",{"term":278,"anchor":674,"definition":675},"memory","Information from prior experience or interaction that persists because it may be useful later and is subject to lifecycle rules.",{"term":283,"anchor":677,"definition":678},"retrieval","The mechanism used to select potentially relevant information from memory, external knowledge, databases, graphs, or other stores.",{"term":288,"anchor":680,"definition":681},"context","The information actually available to the language model during a particular inference or generation step.",{"term":683,"anchor":684,"definition":685},"RAG","rag","Retrieval-augmented generation: a pattern in which external or stored information is retrieved and supplied to a generative model to improve the current output.",{"term":687,"anchor":688,"definition":689},"Provenance","provenance","Metadata describing where information came from, when it was observed, who or what asserted it, and how it was transformed.",{},{"id":692,"data":693,"type":42,"tunes":695},"h-sources",{"text":694,"level":219},"Primary sources and further reading",{},{"id":697,"data":698,"type":705,"tunes":706},"openai-session",{"link":699,"meta":700},"https:\u002F\u002Fdevelopers.openai.com\u002Fcookbook\u002Fexamples\u002Fagents_sdk\u002Fsession_memory",{"image":701,"title":703,"description":704},{"url":702},"","OpenAI — Context Engineering: Short-Term Memory Management with Sessions","OpenAI guidance on trimming and compression for long-running agent context.","linkTool",{},{"id":708,"data":709,"type":705,"tunes":715},"openai-sandbox",{"link":710,"meta":711},"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fagents\u002Fsandboxes",{"image":712,"title":713,"description":714},{"url":702},"OpenAI — Sandbox Agents","Documentation showing persistent memory as a capability with progressive disclosure and read\u002Fwrite behaviour.",{},{"id":717,"data":718,"type":705,"tunes":724},"anthropic-context",{"link":719,"meta":720},"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Feffective-context-engineering-for-ai-agents",{"image":721,"title":722,"description":723},{"url":702},"Anthropic — Effective Context Engineering for AI Agents","Engineering guidance on curating finite model context for reliable agent behaviour.",{},{"id":726,"data":727,"type":705,"tunes":733},"ms-memora",{"link":728,"meta":729},"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Fblog\u002Fmemora-a-harmonic-memory-representation-balancing-abstraction-and-specificity\u002F",{"image":730,"title":731,"description":732},{"url":702},"Microsoft Research — Memora","Research on balancing abstraction and specificity in long-horizon agent memory.",{},{"id":735,"data":736,"type":705,"tunes":742},"ms-plugmem",{"link":737,"meta":738},"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Fblog\u002Ffrom-raw-interaction-to-reusable-knowledge-rethinking-memory-for-ai-agents\u002F",{"image":739,"title":740,"description":741},{"url":702},"Microsoft Research — PlugMem","Research on converting raw agent interaction histories into reusable structured knowledge.",{},{"id":744,"data":745,"type":705,"tunes":751},"ms-ace",{"link":746,"meta":747},"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Fpublication\u002Fagentic-context-engineering-evolving-contexts-for-self-improving-language-models\u002F",{"image":748,"title":749,"description":750},{"url":702},"Microsoft Research — Agentic Context Engineering (ACE)","Research on evolving context as structured playbooks rather than repeatedly rewriting or compressing everything.",{},"2.31.6","Agent memory, RAG, state, and context are often used as if they were interchangeable. They are not. This practical architecture model separates the four layers, shows where each belongs, and explains what breaks when systems collapse them into one.","\u002Fuploads\u002F2026\u002F09\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context-1790350560308-np0xy6.webp","ai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context-1790350560308-np0xy6","PUBLISHED","2026-09-25T11:34:00.000Z","2026-09-25T15:34:35.975Z","2026-09-25T21:01:33.061Z",{"en":761,"de":762,"sr":763,"es":764,"fr":765,"it":766,"ru":767,"zh":768},"\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context","\u002Fde\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context","\u002Fsr\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context","\u002Fes\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context","\u002Ffr\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context","\u002Fit\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context","\u002Fru\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context","\u002Fzh\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context",[770,774,778],{"id":771,"name":772,"slug":773},64,"Information Architecture","information-architecture",{"id":775,"name":776,"slug":777},57,"Data Boundaries","data-boundaries",{"id":779,"name":780,"slug":781},85,"Quality Gates","quality-gates",{"id":783,"login":784,"email":785,"displayName":786},"20","rooth8233","aleksandar@stajic.de","Aleksandar Stajić",[788],{"lang":7,"title":208,"content":210,"contentJson":789,"excerpt":753},{"time":212,"blocks":790,"version":752},[791,794,797,800,803,806,809,812,815,824,827,830,833,836,839,842,845,848,851,854,857,860,863,873,876,879,882,885,888,891,900,903,906,917,920,932,935,938,941,944,947,950,953,956,959,962,965,970,973,976,979,982,985,988,991,994,997,1005,1008,1018,1021,1026,1031,1036,1041,1046],{"id":215,"data":792,"type":220,"tunes":793},{"title":217,"maxLevel":218,"minLevel":219},{},{"id":223,"data":795,"type":226,"tunes":796},{"text":225},{},{"id":229,"data":798,"type":234,"tunes":799},{"body":231,"title":232,"variant":233},{},{"id":237,"data":801,"type":234,"tunes":802},{"body":239,"title":240,"variant":241},{},{"id":244,"data":804,"type":42,"tunes":805},{"text":246,"level":219},{},{"id":249,"data":807,"type":226,"tunes":808},{"text":251},{},{"id":254,"data":810,"type":226,"tunes":811},{"text":256},{},{"id":259,"data":813,"type":42,"tunes":814},{"text":261,"level":219},{},{"id":264,"data":816,"type":292,"tunes":823},{"content":817,"stretched":43,"withHeadings":14},[818,819,820,821,822],[268,269,270,271],[273,274,275,276],[278,279,280,281],[283,284,285,286],[288,289,290,291],{},{"id":295,"data":825,"type":42,"tunes":826},{"text":297,"level":218},{},{"id":300,"data":828,"type":226,"tunes":829},{"text":302},{},{"id":305,"data":831,"type":226,"tunes":832},{"text":307},{},{"id":310,"data":834,"type":42,"tunes":835},{"text":312,"level":218},{},{"id":315,"data":837,"type":226,"tunes":838},{"text":317},{},{"id":320,"data":840,"type":226,"tunes":841},{"text":322},{},{"id":325,"data":843,"type":42,"tunes":844},{"text":327,"level":218},{},{"id":330,"data":846,"type":226,"tunes":847},{"text":332},{},{"id":335,"data":849,"type":226,"tunes":850},{"text":337},{},{"id":340,"data":852,"type":42,"tunes":853},{"text":342,"level":218},{},{"id":345,"data":855,"type":226,"tunes":856},{"text":347},{},{"id":350,"data":858,"type":226,"tunes":859},{"text":352},{},{"id":355,"data":861,"type":42,"tunes":862},{"text":357,"level":219},{},{"id":360,"data":864,"type":383,"tunes":872},{"steps":865,"title":381,"orientation":382},[866,867,868,869,870,871],{"label":364,"description":365},{"label":367,"description":368},{"label":370,"description":371},{"label":373,"description":374},{"label":376,"description":377},{"label":379,"description":380},{},{"id":386,"data":874,"type":42,"tunes":875},{"text":388,"level":219},{},{"id":391,"data":877,"type":226,"tunes":878},{"text":393},{},{"id":396,"data":880,"type":226,"tunes":881},{"text":398},{},{"id":401,"data":883,"type":234,"tunes":884},{"body":403,"title":404,"variant":405},{},{"id":408,"data":886,"type":42,"tunes":887},{"text":410,"level":219},{},{"id":413,"data":889,"type":226,"tunes":890},{"text":415},{},{"id":418,"data":892,"type":292,"tunes":899},{"content":893,"stretched":43,"withHeadings":14},[894,895,896,897,898],[422,423],[425,273],[427,278],[429,283],[431,288],{},{"id":434,"data":901,"type":226,"tunes":902},{"text":436},{},{"id":439,"data":904,"type":42,"tunes":905},{"text":441,"level":219},{},{"id":444,"data":907,"type":292,"tunes":916},{"content":908,"stretched":43,"withHeadings":14},[909,910,911,912,913,914,915],[448,449,450],[452,453,454],[456,457,458],[460,461,462],[464,465,466],[468,469,470],[472,473,474],{},{"id":477,"data":918,"type":42,"tunes":919},{"text":479,"level":219},{},{"id":482,"data":921,"type":292,"tunes":931},{"content":922,"stretched":43,"withHeadings":14},[923,924,925,926,927,928,929,930],[486,487,488],[490,491,492],[494,495,496],[498,499,500],[502,503,504],[506,507,508],[510,511,512],[514,515,516],{},{"id":519,"data":933,"type":42,"tunes":934},{"text":521,"level":219},{},{"id":524,"data":936,"type":226,"tunes":937},{"text":526},{},{"id":529,"data":939,"type":226,"tunes":940},{"text":531},{},{"id":534,"data":942,"type":234,"tunes":943},{"body":536,"title":537,"variant":538},{},{"id":541,"data":945,"type":42,"tunes":946},{"text":543,"level":219},{},{"id":546,"data":948,"type":226,"tunes":949},{"text":548},{},{"id":551,"data":951,"type":226,"tunes":952},{"text":553},{},{"id":556,"data":954,"type":42,"tunes":955},{"text":558,"level":219},{},{"id":561,"data":957,"type":226,"tunes":958},{"text":563},{},{"id":566,"data":960,"type":226,"tunes":961},{"text":568},{},{"id":571,"data":963,"type":42,"tunes":964},{"text":573,"level":219},{},{"id":576,"data":966,"type":591,"tunes":969},{"meta":967,"items":968,"style":590},{},[580,581,582,583,584,585,586,587,588,589],{},{"id":594,"data":971,"type":42,"tunes":972},{"text":596,"level":219},{},{"id":599,"data":974,"type":226,"tunes":975},{"text":601},{},{"id":604,"data":977,"type":226,"tunes":978},{"text":606},{},{"id":609,"data":980,"type":42,"tunes":981},{"text":611,"level":219},{},{"id":614,"data":983,"type":226,"tunes":984},{"text":616},{},{"id":619,"data":986,"type":42,"tunes":987},{"text":621,"level":219},{},{"id":624,"data":989,"type":226,"tunes":990},{"text":626},{},{"id":629,"data":992,"type":226,"tunes":993},{"text":631},{},{"id":634,"data":995,"type":42,"tunes":996},{"text":636,"level":219},{},{"id":639,"data":998,"type":639,"tunes":1004},{"items":999,"title":658},[1000,1001,1002,1003],{"id":643,"answer":644,"question":645},{"id":647,"answer":648,"question":649},{"id":651,"answer":652,"question":653},{"id":655,"answer":656,"question":657},{},{"id":661,"data":1006,"type":42,"tunes":1007},{"text":663,"level":219},{},{"id":666,"data":1009,"type":666,"tunes":1017},{"title":668,"entries":1010},[1011,1012,1013,1014,1015,1016],{"term":273,"anchor":671,"definition":672},{"term":278,"anchor":674,"definition":675},{"term":283,"anchor":677,"definition":678},{"term":288,"anchor":680,"definition":681},{"term":683,"anchor":684,"definition":685},{"term":687,"anchor":688,"definition":689},{},{"id":692,"data":1019,"type":42,"tunes":1020},{"text":694,"level":219},{},{"id":697,"data":1022,"type":705,"tunes":1025},{"link":699,"meta":1023},{"image":1024,"title":703,"description":704},{"url":702},{},{"id":708,"data":1027,"type":705,"tunes":1030},{"link":710,"meta":1028},{"image":1029,"title":713,"description":714},{"url":702},{},{"id":717,"data":1032,"type":705,"tunes":1035},{"link":719,"meta":1033},{"image":1034,"title":722,"description":723},{"url":702},{},{"id":726,"data":1037,"type":705,"tunes":1040},{"link":728,"meta":1038},{"image":1039,"title":731,"description":732},{"url":702},{},{"id":735,"data":1042,"type":705,"tunes":1045},{"link":737,"meta":1043},{"image":1044,"title":740,"description":741},{"url":702},{},{"id":744,"data":1047,"type":705,"tunes":1050},{"link":746,"meta":1048},{"image":1049,"title":749,"description":750},{"url":702},{},"Post erfolgreich abgerufen",{"items":1053,"source":1124,"manualIds":1125,"manualMatchedIds":1126},[1054,1061,1068,1075,1082,1089,1096,1103,1110,1117],{"id":1055,"slug":1056,"title":1057,"excerpt":1058,"featuredImage":1059,"publishedAt":1060},"363","front-und-backend-entwicklung","Front- and Backend Development","Front-end and back-end development is an essential part of web development and involves the creation of web applications and websites. Front-end development focuses on the user interface, while back-end development is responsible for programming and managing the server side.","\u002Fuploads\u002F2026\u002F03\u002Ffront-und-backend-entwicklung-1774872219531-wyu4i1.webp","2023-04-12T11:11:00.000Z",{"id":1062,"slug":1063,"title":1064,"excerpt":1065,"featuredImage":1066,"publishedAt":1067},"466","the-gpu-is-not-the-product-future-proof-private-ai-architecture","The GPU Is Not the Product: Future-Proof Private AI Architecture","Private AI infrastructure should not be designed around one GPU or one model. A more resilient approach combines fast inference GPUs, memory-rich AI systems, physical-AI nodes and optional frontier cloud models behind a capability-aware routing layer.","\u002Fuploads\u002F2026\u002F09\u002Fthe-gpu-is-not-the-product-future-proof-private-ai-architecture-1790140878812-8hsl39.webp","2026-09-23T01:19:00.000Z",{"id":1069,"slug":1070,"title":1071,"excerpt":1072,"featuredImage":1073,"publishedAt":1074},"472","why-more-context-can-make-ai-answers-worse","Why More Context Can Make AI Answers Worse","A larger context window does not guarantee a better answer. This article explains how signal dilution, conflicting evidence, stale state, position sensitivity, and lossy compression can reduce AI reliability—and introduces a practical Context Pressure Test.","\u002Fuploads\u002F2026\u002F09\u002Fwhy-more-context-can-make-ai-answers-worse-1790351615793-2ntv2v.webp","2026-09-25T11:51:00.000Z",{"id":1076,"slug":1077,"title":1078,"excerpt":1079,"featuredImage":1080,"publishedAt":1081},"459","ollama-is-not-the-product-building-production-ready-open-llm-applications","Ollama Is Not the Product: Building Production-Ready Open-LLM Applications","Running a local model with Ollama is easy. Building a production-ready Open-LLM application is harder: it requires RAG, access control, provider abstraction, evaluation, logging, deployment discipline and a controlled application layer around the model.\n","\u002Fuploads\u002F2026\u002F06\u002Follama-is-not-the-product-building-production-ready-open-llm-applications-1782679361640-h0usqf.webp","2026-06-28T16:39:00.000Z",{"id":1083,"slug":1084,"title":1085,"excerpt":1086,"featuredImage":1087,"publishedAt":1088},"478","what-is-rag-the-simplest-explanation-of-how-it-works","What Is RAG? The Simplest Explanation of How It Works","RAG sounds complicated, but the idea is simple: before an AI answers, it first looks up useful information from a knowledge source and gives that information to the language model. This guide explains RAG, LLMs, state, memory and tools using one simple mental model.","\u002Fuploads\u002F2026\u002F09\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works-1790377492124-khjagt.webp","2026-09-25T19:03:00.000Z",{"id":1090,"slug":1091,"title":1092,"excerpt":1093,"featuredImage":1094,"publishedAt":1095},"364","tipps-fuer-die-verbesserung-der-seo-suchmaschinenoptimierung","Mastering the SEO Workflow: Essential Optimization Strategies for Organic Growth","A structured SEO workflow is crucial for sustainable organic growth. Learn the ten foundational strategies, from keyword research and technical optimization to content quality and performance analysis.","\u002Fuploads\u002F2026\u002F03\u002Ftipps-fuer-die-verbesserung-der-seo-suchmaschinenoptimierung-1774866098131-hwkzrg.webp","2024-01-26T06:35:00.000Z",{"id":1097,"slug":1098,"title":1099,"excerpt":1100,"featuredImage":1101,"publishedAt":1102},"469","rag-failed-but-which-layer-actually-failed-a-diagnostic-method","RAG Failed — But Which Layer Actually Failed? A Diagnostic Method","When a RAG answer is wrong, blaming retrieval or the model is too vague. This diagnostic method isolates source coverage, query construction, retrieval, ranking, context assembly, generation, evidence attribution, and freshness—so the actual failure can be reproduced and fixed.","\u002Fuploads\u002F2026\u002F09\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method-1790350847177-pior4c.webp","2026-09-24T19:39:00.000Z",{"id":1104,"slug":1105,"title":1106,"excerpt":1107,"featuredImage":1108,"publishedAt":1109},"470","what-should-an-ai-agent-remember-forget-recompute-or-retrieve-again","What Should an AI Agent Remember, Forget, Recompute or Retrieve Again?","Long-running agents should not remember everything. This article provides a practical lifecycle model for deciding what belongs in durable memory, what should be retrieved again, what is safer to recompute, and what should expire or be superseded.","\u002Fuploads\u002F2026\u002F09\u002Fwhat-should-an-ai-agent-remember-forget-recompute-or-retrieve-again-1790351131087-iehz28.webp","2026-09-25T09:43:00.000Z",{"id":1111,"slug":1112,"title":1113,"excerpt":1114,"featuredImage":1115,"publishedAt":1116},"476","mcp-vs-a2a-vs-ucp-vs-ap2-vs-a2ui-the-agent-protocol-stack-explained","MCP vs A2A vs UCP vs AP2 vs A2UI: The Agent Protocol Stack Explained","MCP, A2A, UCP, AP2 and A2UI are often presented as competing agent standards. They mostly solve different interoperability problems. This guide maps each protocol to the boundary it actually standardizes—and shows how they can work together in one production system.","\u002Fuploads\u002F2026\u002F09\u002Fmcp-vs-a2a-vs-ucp-vs-ap2-vs-a2ui-the-agent-protocol-stack-explained-1790352625869-2ezle0.webp","2026-09-25T12:09:00.000Z",{"id":1118,"slug":1119,"title":1120,"excerpt":1121,"featuredImage":1122,"publishedAt":1123},"467","the-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers","The Answer Validity Boundary: The Missing Layer Between Relevance and Reliable AI Answers","A source can be relevant, authoritative and still be wrong for the question being asked. The missing layer is applicability: the conditions under which an answer holds, and the changes that force it to be reconsidered. This article introduces the Answer Validity Boundary as a source-design pattern for humans, AI search and RAG systems.","\u002Fuploads\u002F2026\u002F09\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers-1790272901306-1g5jly.webp","2026-09-24T11:59:00.000Z","fallback",[],[]]