[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"portal-settings:stajic:en":3,"public-menus:all":38,"post:how-to-know-whether-an-ai-agent-actually-used-the-right-evidence:en":205,"related:post:how-to-know-whether-an-ai-agent-actually-used-the-right-evidence:en:1":1192},{"statusCode":4,"data":5,"message":37},200,{"tenantId":6,"lang":7,"defaultLang":8,"siteUrl":9,"contactEmail":10,"brandName":11,"logoUrl":12,"siteName":11,"siteDescription":13,"ogImage":10,"robotsIndex":14,"socialLinks":10,"reservedSlugs":10,"seoPolicy":15},"stajic","en","de","https:\u002F\u002Fstajic.de",null,"Stajic Platform","\u002FLogo_Planet.svg","Stajic Portal",true,{"branding":16,"relatedContent":17,"crossDomainLinks":18},{"logoUrl":12},{"enabled":14},[19,22,25,28,31,34],{"url":20,"label":21,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Ffigure.rocks","figure.rocks",{"url":23,"label":24,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Floving.rocks","loving.rocks",{"url":26,"label":27,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.com","bazify.com",{"url":29,"label":30,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.de","bazify.de",{"url":32,"label":33,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.at","bazify.at",{"url":35,"label":36,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.ba","bazify.ba","Portal settings resolved",[39,45],{"id":40,"name":41,"location":42,"isActive":14,"isDefault":43,"items":44},1,"main-navigation","header",false,[],{"id":46,"name":47,"location":48,"isActive":14,"isDefault":14,"items":49},4,"main-menu","sidebar",[50,66,79,93,103,118,133],{"id":51,"title":52,"url":60,"target":61,"icon":62,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":64,"portfolioId":10,"children":65},"item-18",{"de":53,"en":54,"es":55,"fr":56,"it":54,"ru":57,"sr":58,"zh":59},"Startseite","Home","Inicio","Accueil","Главная","Почетна","首页","\u002Ffull-stack-web-developer-munich-performance-seo-and-maintainable-builds","_self","i-lucide-home","page",111,[],{"id":67,"title":68,"url":75,"target":61,"icon":76,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":77,"portfolioId":10,"children":78},"item-22",{"de":69,"en":69,"es":70,"fr":69,"it":71,"ru":72,"sr":73,"zh":74},"Vision","Visión","Visione","Видение","Визија","想象","\u002Fueber-uns-webdesign-muenchen-webaplikation","i-lucide-eye",113,[],{"id":80,"title":81,"url":89,"target":61,"icon":90,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":91,"portfolioId":10,"children":92},"item-19",{"de":82,"en":83,"es":84,"fr":83,"it":85,"ru":86,"sr":87,"zh":88},"Leistungen","Services","Servicios","Servizi","Услуги","Услуге","服务","\u002Fservices-dienstleistungen-muenchen","i-lucide-wrench",116,[],{"id":94,"title":95,"url":99,"target":61,"icon":100,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":101,"portfolioId":10,"children":102},"item-23",{"de":96,"en":96,"es":96,"fr":96,"it":96,"ru":97,"sr":97,"zh":98},"Blog","Блог","博客","\u002Fblog","i-lucide-book-open",112,[],{"id":104,"title":105,"url":114,"target":61,"icon":115,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":116,"portfolioId":10,"children":117},"item-32",{"de":106,"en":107,"es":108,"fr":109,"it":110,"ru":111,"sr":112,"zh":113},"Neue Technologien","New Technologies","Nuevas tecnologías","Nouvelles technologies","Nuove tecnologie","Новые технологии","Нове технологије","新技术！","\u002Fneue-webtechnologien","i-lucide-sparkles",122,[],{"id":119,"title":120,"url":129,"target":61,"icon":130,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":131,"portfolioId":10,"children":132},"item-20",{"de":121,"en":122,"es":123,"fr":124,"it":125,"ru":126,"sr":127,"zh":128},"Kontakt","Contact us!","Contacto","Contact","Contatto","Контакт","Контактирајте нас","联系我们！","\u002Fcontact","i-lucide-mail",115,[],{"id":134,"title":135,"url":144,"target":61,"icon":145,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":147},"item-21",{"de":136,"en":137,"es":138,"fr":139,"it":140,"ru":141,"sr":142,"zh":143},"Unsere Arbeit","Our Work","Nuestro trabajo","Nos réalisations","I nostri lavori","Наши работы","Наши радови","文件夹","\u002Fportfolio","i-lucide-briefcase",114,[148,161,175,181,193],{"id":149,"title":150,"url":144,"target":61,"icon":159,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":160},"item-24",{"de":151,"en":152,"es":153,"fr":154,"it":155,"ru":156,"sr":157,"zh":158},"Alle Projekte","All Projects","Todos los proyectos","Tous les projets","Tutti i progetti","Все проекты","Сви пројекти","所有项目","i-lucide-grid-3x3",[],{"id":162,"title":163,"url":171,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":174},"item-29",{"de":164,"en":165,"es":166,"fr":167,"it":168,"ru":169,"sr":170,"zh":143},"Local Roots, Global Reach","Local Roots - Global Reach","Empresa local ","Entreprise locale","Azienda locale","Местная компания","Локално предузеће глобално тржиште","\u002Fportfolio\u002Flocal-roots-global-reach-communication-media-systems-for-modern-business","i-lucide-folder","custom",[],{"id":176,"title":177,"url":179,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":180},"item-28",{"de":178,"en":178,"es":178,"fr":178,"it":178,"ru":178,"sr":178,"zh":178},"Solr Suggester","\u002Fportfolio\u002Fsolr-fuzzy-suggester-und-solr-infix-suggester-abfrage-ueber-ajax-und-filterung",[],{"id":182,"title":183,"url":191,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":192},"item-27",{"de":184,"en":185,"es":186,"fr":187,"it":188,"ru":189,"sr":190,"zh":185},"Firmenwebseite SEO","Company Website SEO","Sitio web corporativo SEO","Site web d’entreprise SEO","Sito web aziendale SEO","Корпоративный сайт SEO","Пословна веб-страница SEO","\u002Fportfolio\u002Fseo-sem-branding-mobile-webseite-muenchen",[],{"id":194,"title":195,"url":203,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":204},"item-31",{"de":196,"en":197,"es":198,"fr":199,"it":200,"ru":201,"sr":202,"zh":197},"Digitalisierungsportal","Digitalization Portal","Portal de digitalización","Portail de numérisation","Portale di digitalizzazione","Портал цифровизации","Портал за дигитализацију","\u002Fportfolio\u002Fdigitalisierungsportal-archiv-museum-bibliothek-ead-lido-mets-mods",[],{"statusCode":4,"data":206,"message":1191},{"id":207,"title":208,"slug":209,"content":210,"contentJson":211,"excerpt":852,"featuredImage":853,"featuredImageAlt":854,"featuredImageCaption":10,"featuredImageTitle":10,"featuredImageCopyright":10,"featuredImageAuthor":10,"featuredImageSourceUrl":10,"featuredImageLicense":10,"featuredImageIsAiGenerated":43,"status":855,"publishedAt":856,"createdAt":857,"updatedAt":858,"seoLocalePaths":859,"categories":868,"author":881,"translations":886},"471","How to Know Whether an AI Agent Actually Used the Right Evidence","how-to-know-whether-an-ai-agent-actually-used-the-right-evidence","{\"time\":1790351390243,\"blocks\":[{\"id\":\"0hk9UtwqZf\",\"type\":\"tableOfContents\",\"data\":{\"title\":\"Contents\",\"minLevel\":2,\"maxLevel\":3},\"tunes\":{}},{\"id\":\"intro\",\"type\":\"paragraph\",\"data\":{\"text\":\"An AI agent can cite sources, retrieve documents, and still use the wrong evidence. A source may be authoritative but irrelevant to the exact claim. A retrieved passage may support only part of an answer. A correct source can be stale, superseded, or valid for the wrong jurisdiction, product version, user, or system state. This creates a harder evaluation problem than simple citation checking: did the agent actually use the right evidence for the claim it made?\"},\"tunes\":{}},{\"id\":\"direct\",\"type\":\"callout\",\"data\":{\"variant\":\"info\",\"title\":\"Direct answer\",\"body\":\"To know whether an AI agent used the right evidence, evaluate the chain \u003Cstrong>Claim → Evidence → Applicability → Use\u003C\u002Fstrong>. For each material claim, verify that the evidence directly supports it, comes from an appropriate authority, applies to the current conditions, and was actually available to the agent before the claim was produced. A citation alone proves none of those things.\"},\"tunes\":{}},{\"id\":\"method-note\",\"type\":\"callout\",\"data\":{\"variant\":\"note\",\"title\":\"About the method\",\"body\":\"The Claim–Evidence–Applicability–Use model and Evidence Utilization Test in this article are practical evaluation methods, not formal industry standards. They build on established ideas such as groundedness, source quality, citation coverage, trace evaluation, and task-specific evals.\"},\"tunes\":{}},{\"id\":\"h-citations\",\"type\":\"header\",\"data\":{\"text\":\"Why citations are not enough\",\"level\":2},\"tunes\":{}},{\"id\":\"p-citations-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A citation answers only a narrow question: the system associated a claim or response with a source. It does not automatically establish that the source supports the specific claim, that the source is authoritative enough for the task, that the cited passage contains the necessary condition or exception, or that the model relied on that evidence rather than producing the answer from prior model knowledge.\"},\"tunes\":{}},{\"id\":\"p-citations-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Anthropic's guidance for research-agent evaluation explicitly separates groundedness, coverage, and source quality. OpenAI's agent-evaluation guidance similarly emphasizes traces because a final output does not reveal whether the agent selected the right tools or followed the intended workflow. Those ideas point to a broader conclusion: evidence quality is a property of the execution path, not just the final prose.\"},\"tunes\":{}},{\"id\":\"h-four\",\"type\":\"header\",\"data\":{\"text\":\"The four questions every material claim should pass\",\"level\":2},\"tunes\":{}},{\"id\":\"table-four\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Dimension\",\"Question\",\"Typical failure\"],[\"Claim support\",\"Does the evidence directly support this exact claim?\",\"The source is topically related but does not establish the statement\"],[\"Evidence authority\",\"Is this an appropriate source for this kind of claim?\",\"A secondary summary is used where a primary source or live system is required\"],[\"Applicability\",\"Does the evidence apply to this time, version, jurisdiction, user, state, or population?\",\"A true statement is applied outside its valid conditions\"],[\"Evidence use\",\"Was this evidence actually available and used in the agent's execution path?\",\"The final answer is correct, but the retrieved evidence was irrelevant or unused\"]]},\"tunes\":{}},{\"id\":\"h-support\",\"type\":\"header\",\"data\":{\"text\":\"1. Claim support: does the source establish what the agent says?\",\"level\":3},\"tunes\":{}},{\"id\":\"p-support-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Evidence should be evaluated at claim level. A document can be relevant to the subject and still fail to support a specific statement. If a source says that a feature is available in selected regions, the answer “the feature is available globally” is unsupported even though the citation looks plausible.\"},\"tunes\":{}},{\"id\":\"p-support-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"This is where broad “grounded \u002F not grounded” judgments are often too coarse. Split the response into material claims, map each claim to the smallest evidence span that supports it, and classify the relationship: direct support, partial support, contradiction, or no support.\"},\"tunes\":{}},{\"id\":\"h-authority\",\"type\":\"header\",\"data\":{\"text\":\"2. Evidence authority: is this the right kind of source?\",\"level\":3},\"tunes\":{}},{\"id\":\"p-authority-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Correct evidence selection is not only semantic relevance. The source must be suitable for the decision. Current account status should come from the account system, not an old email. An API behaviour claim should preferably be checked against current vendor documentation or reproducible behaviour. A legal requirement may need the applicable law, regulator, or authoritative guidance rather than a generic blog post.\"},\"tunes\":{}},{\"id\":\"p-authority-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Source authority is task-specific. A community report can be the best evidence for a real-world bug that vendor documentation does not acknowledge. A vendor announcement can be authoritative for what the vendor claims but weak evidence for independent performance. The evaluator therefore needs an explicit source hierarchy for the task rather than one universal authority score.\"},\"tunes\":{}},{\"id\":\"h-applicability\",\"type\":\"header\",\"data\":{\"text\":\"3. Applicability: right evidence, wrong conditions\",\"level\":3},\"tunes\":{}},{\"id\":\"p-apply-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The most dangerous evidence errors are often not fabricated sources but valid sources used outside their boundary. A recommendation can change with software version, date, jurisdiction, hardware revision, user permissions, product availability, current game state, tenant configuration, or other environmental variables.\"},\"tunes\":{}},{\"id\":\"p-apply-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"For each material source, preserve the conditions that determine whether it still applies. This is especially important after summarization: a compressed memory or citation may preserve the conclusion while dropping the exception, date, or prerequisite that made the conclusion valid.\"},\"tunes\":{}},{\"id\":\"apply-warning\",\"type\":\"callout\",\"data\":{\"variant\":\"warning\",\"title\":\"Evidence can be authentic and still be wrong for the answer\",\"body\":\"A real source, quoted accurately, can still produce a wrong answer when its time, version, population, jurisdiction, or state does not match the current question.\"},\"tunes\":{}},{\"id\":\"h-use\",\"type\":\"header\",\"data\":{\"text\":\"4. Evidence use: did the agent actually rely on the evidence?\",\"level\":3},\"tunes\":{}},{\"id\":\"p-use-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"An answer can be correct even when retrieval failed. The model may already know the answer, infer it from unrelated context, or simply guess correctly. If the evaluation checks only final correctness, the system may appear well-grounded while the evidence path is broken.\"},\"tunes\":{}},{\"id\":\"p-use-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"To evaluate evidence use, inspect the trace. Confirm which sources were retrieved, which passages reached the model context, when they became available, and whether the final claim can be explained by those inputs. OpenAI's current agent-evaluation tooling emphasizes trace grading precisely because workflow-level behaviour cannot be reconstructed reliably from the final answer alone.\"},\"tunes\":{}},{\"id\":\"h-eut\",\"type\":\"header\",\"data\":{\"text\":\"The Evidence Utilization Test\",\"level\":2},\"tunes\":{}},{\"id\":\"p-eut-intro\",\"type\":\"paragraph\",\"data\":{\"text\":\"A practical evaluation can be built as a controlled counterfactual. Instead of asking only whether the answer is correct, change the evidence and observe whether the claim changes in the expected direction.\"},\"tunes\":{}},{\"id\":\"eut-flow\",\"type\":\"processFlow\",\"data\":{\"title\":\"Evidence Utilization Test\",\"orientation\":\"auto\",\"steps\":[{\"label\":\"1. Select one material claim\",\"description\":\"Choose a claim whose correctness matters and define the expected answer precisely.\"},{\"label\":\"2. Identify gold evidence\",\"description\":\"Provide the smallest authoritative evidence set sufficient to support the claim.\"},{\"label\":\"3. Run with gold evidence\",\"description\":\"Verify that the agent produces the supported answer when the correct evidence is available.\"},{\"label\":\"4. Remove the decisive evidence\",\"description\":\"Run the same task without the key supporting passage while keeping other inputs stable.\"},{\"label\":\"5. Replace it with contradictory or superseding evidence\",\"description\":\"Where safe, provide controlled evidence that changes the correct conclusion.\"},{\"label\":\"6. Compare the claims\",\"description\":\"Check whether the answer tracks the evidence change or remains anchored to prior model knowledge.\"},{\"label\":\"7. Inspect the trace\",\"description\":\"Confirm what was retrieved, what reached context, and what source or tool result preceded the claim.\"}]},\"tunes\":{}},{\"id\":\"eut-tip\",\"type\":\"callout\",\"data\":{\"variant\":\"tip\",\"title\":\"The key signal\",\"body\":\"If the decisive evidence changes but the agent's claim does not, you have evidence that the system may not be using retrieval as intended — even when the original answer happened to be correct.\"},\"tunes\":{}},{\"id\":\"h-matrix\",\"type\":\"header\",\"data\":{\"text\":\"A claim–evidence matrix is more useful than a source list\",\"level\":2},\"tunes\":{}},{\"id\":\"claim-matrix\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Claim\",\"Evidence\",\"Support\",\"Authority\",\"Applicability\",\"Used in trace\"],[\"Feature X is available\",\"Vendor documentation\",\"Direct\",\"High for availability claim\",\"Current version and region must match\",\"Yes \u002F No\"],[\"Configuration Y is faster\",\"Vendor benchmark\",\"Partial\",\"High for vendor's test, not independent performance\",\"Hardware and workload must match\",\"Yes \u002F No\"],[\"Policy applies to this user\",\"Current policy + account state\",\"Direct only when combined\",\"High\",\"Jurisdiction, date, role and account state must match\",\"Yes \u002F No\"],[\"A product is in stock\",\"Live inventory API\",\"Direct\",\"Authoritative for current stock\",\"Expires quickly\",\"Yes \u002F No\"]]},\"tunes\":{}},{\"id\":\"p-matrix-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"This matrix forces several questions that conventional citation checking hides. One claim can require multiple sources. One source can support only part of a claim. An authoritative source can have a short validity window. And a perfectly good source can be irrelevant if it never entered the execution path.\"},\"tunes\":{}},{\"id\":\"h-retrieval-evidence\",\"type\":\"header\",\"data\":{\"text\":\"Separate retrieval quality from evidence quality\",\"level\":2},\"tunes\":{}},{\"id\":\"p-re-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Retrieval metrics ask whether relevant material was found and ranked. Evidence evaluation asks whether that material justifies the resulting claims. The two are related but not identical.\"},\"tunes\":{}},{\"id\":\"retrieval-comparison\",\"type\":\"comparison\",\"data\":{\"title\":\"Retrieval success is not evidence success\",\"layout\":\"table\",\"columns\":[{\"id\":\"situation\",\"label\":\"Situation\"},{\"id\":\"retrieval\",\"label\":\"Retrieval\"},{\"id\":\"evidence\",\"label\":\"Evidence quality\"}],\"rows\":[{\"id\":\"a\",\"label\":\"Right document, wrong claim\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"b\",\"label\":\"Right fact, stale source\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"c\",\"label\":\"Weak source, correct answer\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"d\",\"label\":\"Multiple sources required\",\"values\":[\"\",\"\",\"\"]}]},\"tunes\":{}},{\"id\":\"h-source-quality\",\"type\":\"header\",\"data\":{\"text\":\"Evaluate source quality as a rubric, not a domain whitelist\",\"level\":2},\"tunes\":{}},{\"id\":\"p-source-quality-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Hard-coded lists of “trusted domains” are tempting but often brittle. Source quality should instead reflect the claim type. Useful dimensions include primary versus secondary status, recency, directness, reproducibility, independence, domain expertise, data provenance, update cadence, and whether the source has an incentive to overstate the claim.\"},\"tunes\":{}},{\"id\":\"p-source-quality-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Anthropic's research-agent evaluation guidance explicitly calls out source-quality checks alongside groundedness and coverage. The practical implementation should therefore grade both what the source says and whether this source is appropriate for this kind of statement.\"},\"tunes\":{}},{\"id\":\"h-coverage\",\"type\":\"header\",\"data\":{\"text\":\"Evidence coverage: every important claim needs support, not every sentence\",\"level\":2},\"tunes\":{}},{\"id\":\"p-coverage-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Not every sentence needs a citation. Transitional language, arithmetic derived transparently from cited values, or clearly marked interpretation may not require a separate source. But every material externally verifiable claim should have enough support that an evaluator can reconstruct why the agent was allowed to say it.\"},\"tunes\":{}},{\"id\":\"p-coverage-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Coverage should therefore be weighted by claim importance. Missing support for a decorative detail is not equivalent to missing support for a price, eligibility decision, safety instruction, legal requirement, technical compatibility statement, or recommendation-driving fact.\"},\"tunes\":{}},{\"id\":\"h-provenance\",\"type\":\"header\",\"data\":{\"text\":\"Evidence provenance must survive summarization and memory\",\"level\":2},\"tunes\":{}},{\"id\":\"p-prov-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Long-running agents often summarize previous work or write durable memories. If evidence provenance is stripped during that transformation, future agents may retrieve a clean conclusion without knowing whether it came from a user statement, a live API, an old document, a model inference, or an unverified web result.\"},\"tunes\":{}},{\"id\":\"p-prov-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"For important facts, preserve at least the source identity, retrieval or observation time, evidence type, relevant version or state, and whether the stored text is quoted, summarized, inferred, or derived. Provenance is what allows a later agent to decide whether the evidence should be trusted, refreshed, restricted, or discarded.\"},\"tunes\":{}},{\"id\":\"h-record\",\"type\":\"header\",\"data\":{\"text\":\"A practical evidence record\",\"level\":2},\"tunes\":{}},{\"id\":\"evidence-record\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Field\",\"Purpose\"],[\"claim_id\",\"Identifies the material claim being supported\"],[\"source_id \u002F source_url \u002F system\",\"Identifies where the evidence came from\"],[\"evidence_span\",\"Preserves the smallest passage, record, or tool result that supports the claim\"],[\"retrieved_at \u002F observed_at\",\"Allows freshness and timeline checks\"],[\"source_version \u002F object_version\",\"Allows supersession and reproducibility checks\"],[\"authority_role\",\"Explains why this source is suitable for this claim\"],[\"applicability\",\"Stores relevant date, jurisdiction, product version, user, tenant, state, or other conditions\"],[\"transformation\",\"Marks whether evidence is raw, quoted, summarized, normalized, or derived\"],[\"trace_step\",\"Shows when the evidence became available to the agent\"],[\"support_status\",\"Direct, partial, contradictory, unsupported, or uncertain\"]]},\"tunes\":{}},{\"id\":\"h-failures\",\"type\":\"header\",\"data\":{\"text\":\"Failure modes that look grounded but are not\",\"level\":2},\"tunes\":{}},{\"id\":\"failure-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Failure mode\",\"Why it fools evaluators\",\"What to test\"],[\"Citation decoration\",\"The answer contains sources, so it looks researched\",\"Map each material claim to an exact supporting span\"],[\"Authority mismatch\",\"The source is reputable but not authoritative for the specific fact\",\"Define claim-specific source hierarchy\"],[\"Temporal mismatch\",\"The source was correct when published\",\"Check retrieval time, source date, and superseding evidence\"],[\"Condition stripping\",\"A summary keeps the conclusion but drops exceptions\",\"Compare generated claim with full local source context\"],[\"Post-hoc citation\",\"A plausible source is attached after the answer is generated\",\"Inspect trace ordering and whether evidence preceded the claim\"],[\"Parametric override\",\"The model ignores retrieved evidence and answers from prior knowledge\",\"Run counterfactual evidence-utilization tests\"],[\"Evidence laundering\",\"Model inference is summarized and later stored as if it were a source fact\",\"Preserve transformation type and provenance across memory writes\"],[\"Source majority fallacy\",\"Several secondary pages repeat the same unsupported statement\",\"Trace claims back to independent or primary evidence\"]]},\"tunes\":{}},{\"id\":\"h-production\",\"type\":\"header\",\"data\":{\"text\":\"How to evaluate the agent in production\",\"level\":2},\"tunes\":{}},{\"id\":\"production-flow\",\"type\":\"processFlow\",\"data\":{\"title\":\"Evidence evaluation pipeline\",\"orientation\":\"auto\",\"steps\":[{\"label\":\"1. Define material claims\",\"description\":\"Identify the facts, recommendations, or decisions whose correctness matters to the task.\"},{\"label\":\"2. Build gold evidence\",\"description\":\"Create reference evidence and source-quality expectations for representative cases.\"},{\"label\":\"3. Capture traces\",\"description\":\"Log retrieval queries, tool calls, returned evidence, context construction, model output, and citations.\"},{\"label\":\"4. Grade support\",\"description\":\"Check whether each material claim is directly, partially, contradictorily, or not supported.\"},{\"label\":\"5. Grade authority and applicability\",\"description\":\"Evaluate whether the source is appropriate and whether its conditions match the current task.\"},{\"label\":\"6. Run counterfactuals\",\"description\":\"Remove, replace, or supersede decisive evidence and test whether the answer follows the change.\"},{\"label\":\"7. Review high-impact failures\",\"description\":\"Use human or domain-expert review where automated grading is not reliable enough.\"},{\"label\":\"8. Convert failures into eval cases\",\"description\":\"Add production failures and edge cases to a repeatable regression dataset.\"}]},\"tunes\":{}},{\"id\":\"p-production-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"OpenAI's current evaluation guidance recommends task-specific evals, continuous evaluation, production-derived datasets, and traces for debugging agent behaviour. Anthropic likewise recommends combining grader types for research agents because correctness, source quality, coverage, and groundedness are separate dimensions. Evidence evaluation should follow the same pattern: several narrow graders are more diagnostic than one opaque “quality” score.\"},\"tunes\":{}},{\"id\":\"h-judge\",\"type\":\"header\",\"data\":{\"text\":\"Do not let an LLM judge become the only evidence judge\",\"level\":2},\"tunes\":{}},{\"id\":\"p-judge-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"LLM graders are useful for scalable claim classification, relevance checks, and pairwise comparisons, but they can share the same blind spots as the system they evaluate. A grader may accept a plausible but unsupported claim, miss a subtle version boundary, or overrate a polished source.\"},\"tunes\":{}},{\"id\":\"p-judge-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"OpenAI's evaluation guidance recommends calibrating automated graders against human judgment and using clear, scoped criteria. For evidence-heavy systems, deterministic checks should be used wherever possible: timestamps, object versions, permission scope, exact source IDs, retrieval order, document hashes, and whether the evidence was present before the model generated the claim.\"},\"tunes\":{}},{\"id\":\"h-change\",\"type\":\"header\",\"data\":{\"text\":\"What would change this answer?\",\"level\":2},\"tunes\":{}},{\"id\":\"p-change-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The evaluation can be simpler when the agent operates over a small, immutable, authoritative corpus and every answer is strictly extractive. In that environment, source authority and applicability are mostly fixed, and claim-to-span support may be enough.\"},\"tunes\":{}},{\"id\":\"p-change-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The evaluation must become stricter when the agent mixes web search, long-term memory, live tools, multiple jurisdictions, rapidly changing information, user-specific state, or autonomous actions. In those systems, evidence validity depends not only on the source text but also on when and how the evidence was obtained.\"},\"tunes\":{}},{\"id\":\"p-change-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"Future models may become better at internally tracking provenance and uncertainty, but that would not remove the need for external evidence records in systems that require auditability. A system should not depend on the model's self-report of what influenced it when traces and source metadata can provide stronger evidence.\"},\"tunes\":{}},{\"id\":\"h-limitations\",\"type\":\"header\",\"data\":{\"text\":\"Limitations\",\"level\":2},\"tunes\":{}},{\"id\":\"p-limit-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"It is not always possible to prove causal evidence use from traces alone. A source can be present in context without influencing the answer, and a model may independently know the same fact. Counterfactual tests strengthen the inference but can themselves change the task distribution.\"},\"tunes\":{}},{\"id\":\"p-limit-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Source authority can also be contested or domain-dependent. Some questions have no single authoritative source, and experts may disagree about which evidence deserves more weight. In those cases the evaluator should preserve disagreement and score transparency, coverage, and reasoning against an explicit rubric rather than pretending there is one unquestioned source of truth.\"},\"tunes\":{}},{\"id\":\"h-conclusion\",\"type\":\"header\",\"data\":{\"text\":\"Conclusion\",\"level\":2},\"tunes\":{}},{\"id\":\"p-conclusion-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The question “Did the agent cite a source?” is too weak for production AI. The stronger question is: Did each important claim come from evidence that actually supports it, has the right authority, still applies to the current conditions, and was available in the execution path before the claim was made?\"},\"tunes\":{}},{\"id\":\"p-conclusion-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"That turns evidence from decoration into an evaluable system property. Capture the trace. Map claims to evidence. Check authority and applicability. Run counterfactual evidence tests. Preserve provenance through summaries and memory. Then a correct answer is not only plausible — it has an evidence path you can inspect.\"},\"tunes\":{}},{\"id\":\"internal-reliability\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough\",\"title\":\"AI Agent Reliability: Why the Final Answer Is Not Enough\",\"excerpt\":\"Outcome correctness alone cannot prove that an agent's execution path was safe or reliable. This related article explains why trajectories, tools and intermediate decisions matter.\",\"ctaLabel\":\"Read the related article\"},\"tunes\":{}},{\"id\":\"internal-reasoning\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Ffrom-research-protocol-to-a-general-ai-reasoning-framework\",\"title\":\"From Research Protocol to a General AI Reasoning Framework\",\"excerpt\":\"A practical reasoning method for separating evidence, assumptions, competing hypotheses and domain-specific validation.\",\"ctaLabel\":\"Read the reasoning framework\"},\"tunes\":{}},{\"id\":\"h-faq\",\"type\":\"header\",\"data\":{\"text\":\"FAQ\",\"level\":2},\"tunes\":{}},{\"id\":\"faq\",\"type\":\"faq\",\"data\":{\"title\":\"Evaluating evidence use in AI agents\",\"items\":[{\"id\":\"faq1\",\"question\":\"Does a citation prove that an AI answer is grounded?\",\"answer\":\"No. A citation may be relevant to the topic without supporting the exact claim, may come from the wrong authority, may no longer apply, or may have been attached without materially influencing the generated answer.\"},{\"id\":\"faq2\",\"question\":\"How can I test whether an AI agent actually used retrieved evidence?\",\"answer\":\"Use a counterfactual evidence-utilization test: run the task with known-correct evidence, then remove or replace the decisive evidence while keeping other inputs stable. If the answer does not respond to the evidence change, inspect whether the model is relying on prior knowledge or another source.\"},{\"id\":\"faq3\",\"question\":\"What is the difference between groundedness and source quality?\",\"answer\":\"Groundedness asks whether claims are supported by the supplied evidence. Source quality asks whether the evidence itself is appropriate and authoritative enough for the type of claim being made.\"},{\"id\":\"faq4\",\"question\":\"Why can a real source still produce a wrong AI answer?\",\"answer\":\"The source may be stale, superseded, valid for another version, jurisdiction, user, population, or system state, or may contain conditions that were lost during retrieval or summarization.\"},{\"id\":\"faq5\",\"question\":\"What should I log for evidence evaluation?\",\"answer\":\"Log the retrieval query, returned sources, exact evidence spans, timestamps and versions, filters, final context, model output, citations, and trace ordering so evaluators can reconstruct what evidence was available before each material claim.\"}]},\"tunes\":{}},{\"id\":\"h-glossary\",\"type\":\"header\",\"data\":{\"text\":\"Glossary\",\"level\":2},\"tunes\":{}},{\"id\":\"glossary\",\"type\":\"glossary\",\"data\":{\"title\":\"Key evidence-evaluation terms\",\"entries\":[{\"term\":\"Claim support\",\"definition\":\"The degree to which a specific evidence span directly establishes a generated claim.\",\"anchor\":\"claim-support\"},{\"term\":\"Evidence authority\",\"definition\":\"How appropriate a source is for establishing a particular type of claim, given its role, provenance and relationship to the underlying fact.\",\"anchor\":\"evidence-authority\"},{\"term\":\"Applicability\",\"definition\":\"The conditions under which evidence remains valid for a claim, including time, version, jurisdiction, user, population, system state or other boundaries.\",\"anchor\":\"applicability\"},{\"term\":\"Evidence utilization\",\"definition\":\"Whether the agent's output actually responds to and depends on the evidence made available in its execution path.\",\"anchor\":\"evidence-utilization\"},{\"term\":\"Counterfactual evidence test\",\"definition\":\"An evaluation that removes, replaces or changes decisive evidence to test whether the agent's claim changes appropriately.\",\"anchor\":\"counterfactual-evidence-test\"},{\"term\":\"Provenance\",\"definition\":\"Metadata that records where evidence came from, when it was obtained, how it was transformed and what version or state it represented.\",\"anchor\":\"provenance\"}]},\"tunes\":{}},{\"id\":\"h-sources\",\"type\":\"header\",\"data\":{\"text\":\"Primary sources and further reading\",\"level\":2},\"tunes\":{}},{\"id\":\"src-openai-agent-evals\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fagent-evals\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — Evaluate Agent Workflows\",\"description\":\"Guidance on trace grading, workflow-level evaluation, datasets and repeatable eval runs for agents.\"}},\"tunes\":{}},{\"id\":\"src-openai-eval-best\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fevaluation-best-practices\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — Evaluation Best Practices\",\"description\":\"Guidance on task-specific evals, production-derived datasets, scoped metrics, continuous evaluation and grader calibration.\"}},\"tunes\":{}},{\"id\":\"src-anthropic-evals\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Fdemystifying-evals-for-ai-agents\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Anthropic — Demystifying Evals for AI Agents\",\"description\":\"Agent-evaluation guidance including groundedness, coverage and source-quality checks for research agents.\"}},\"tunes\":{}},{\"id\":\"src-openai-thirdparty\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fopenai.com\u002Findex\u002Ftrustworthy-third-party-evaluations-foundations\u002F\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — A Shared Playbook for Trustworthy Third-Party Evaluations\",\"description\":\"Evaluation guidance emphasizing that modern agent performance depends on workflow and environment, not only final model output.\"}},\"tunes\":{}}],\"version\":\"2.31.6\"}",{"time":212,"blocks":213,"version":851},1790351390243,[214,222,228,236,243,248,253,258,263,289,294,299,304,309,314,319,324,329,334,341,346,351,356,361,366,395,402,407,442,447,452,457,491,496,501,506,511,516,521,526,531,536,541,579,584,625,630,660,665,670,675,680,685,690,695,700,705,710,715,720,725,730,739,747,752,778,783,809,814,824,833,842],{"id":215,"data":216,"type":220,"tunes":221},"0hk9UtwqZf",{"title":217,"maxLevel":218,"minLevel":219},"Contents",3,2,"tableOfContents",{},{"id":223,"data":224,"type":226,"tunes":227},"intro",{"text":225},"An AI agent can cite sources, retrieve documents, and still use the wrong evidence. A source may be authoritative but irrelevant to the exact claim. A retrieved passage may support only part of an answer. A correct source can be stale, superseded, or valid for the wrong jurisdiction, product version, user, or system state. This creates a harder evaluation problem than simple citation checking: did the agent actually use the right evidence for the claim it made?","paragraph",{},{"id":229,"data":230,"type":234,"tunes":235},"direct",{"body":231,"title":232,"variant":233},"To know whether an AI agent used the right evidence, evaluate the chain \u003Cstrong>Claim → Evidence → Applicability → Use\u003C\u002Fstrong>. For each material claim, verify that the evidence directly supports it, comes from an appropriate authority, applies to the current conditions, and was actually available to the agent before the claim was produced. A citation alone proves none of those things.","Direct answer","info","callout",{},{"id":237,"data":238,"type":234,"tunes":242},"method-note",{"body":239,"title":240,"variant":241},"The Claim–Evidence–Applicability–Use model and Evidence Utilization Test in this article are practical evaluation methods, not formal industry standards. They build on established ideas such as groundedness, source quality, citation coverage, trace evaluation, and task-specific evals.","About the method","note",{},{"id":244,"data":245,"type":42,"tunes":247},"h-citations",{"text":246,"level":219},"Why citations are not enough",{},{"id":249,"data":250,"type":226,"tunes":252},"p-citations-1",{"text":251},"A citation answers only a narrow question: the system associated a claim or response with a source. It does not automatically establish that the source supports the specific claim, that the source is authoritative enough for the task, that the cited passage contains the necessary condition or exception, or that the model relied on that evidence rather than producing the answer from prior model knowledge.",{},{"id":254,"data":255,"type":226,"tunes":257},"p-citations-2",{"text":256},"Anthropic's guidance for research-agent evaluation explicitly separates groundedness, coverage, and source quality. OpenAI's agent-evaluation guidance similarly emphasizes traces because a final output does not reveal whether the agent selected the right tools or followed the intended workflow. Those ideas point to a broader conclusion: evidence quality is a property of the execution path, not just the final prose.",{},{"id":259,"data":260,"type":42,"tunes":262},"h-four",{"text":261,"level":219},"The four questions every material claim should pass",{},{"id":264,"data":265,"type":287,"tunes":288},"table-four",{"content":266,"stretched":43,"withHeadings":14},[267,271,275,279,283],[268,269,270],"Dimension","Question","Typical failure",[272,273,274],"Claim support","Does the evidence directly support this exact claim?","The source is topically related but does not establish the statement",[276,277,278],"Evidence authority","Is this an appropriate source for this kind of claim?","A secondary summary is used where a primary source or live system is required",[280,281,282],"Applicability","Does the evidence apply to this time, version, jurisdiction, user, state, or population?","A true statement is applied outside its valid conditions",[284,285,286],"Evidence use","Was this evidence actually available and used in the agent's execution path?","The final answer is correct, but the retrieved evidence was irrelevant or unused","table",{},{"id":290,"data":291,"type":42,"tunes":293},"h-support",{"text":292,"level":218},"1. Claim support: does the source establish what the agent says?",{},{"id":295,"data":296,"type":226,"tunes":298},"p-support-1",{"text":297},"Evidence should be evaluated at claim level. A document can be relevant to the subject and still fail to support a specific statement. If a source says that a feature is available in selected regions, the answer “the feature is available globally” is unsupported even though the citation looks plausible.",{},{"id":300,"data":301,"type":226,"tunes":303},"p-support-2",{"text":302},"This is where broad “grounded \u002F not grounded” judgments are often too coarse. Split the response into material claims, map each claim to the smallest evidence span that supports it, and classify the relationship: direct support, partial support, contradiction, or no support.",{},{"id":305,"data":306,"type":42,"tunes":308},"h-authority",{"text":307,"level":218},"2. Evidence authority: is this the right kind of source?",{},{"id":310,"data":311,"type":226,"tunes":313},"p-authority-1",{"text":312},"Correct evidence selection is not only semantic relevance. The source must be suitable for the decision. Current account status should come from the account system, not an old email. An API behaviour claim should preferably be checked against current vendor documentation or reproducible behaviour. A legal requirement may need the applicable law, regulator, or authoritative guidance rather than a generic blog post.",{},{"id":315,"data":316,"type":226,"tunes":318},"p-authority-2",{"text":317},"Source authority is task-specific. A community report can be the best evidence for a real-world bug that vendor documentation does not acknowledge. A vendor announcement can be authoritative for what the vendor claims but weak evidence for independent performance. The evaluator therefore needs an explicit source hierarchy for the task rather than one universal authority score.",{},{"id":320,"data":321,"type":42,"tunes":323},"h-applicability",{"text":322,"level":218},"3. Applicability: right evidence, wrong conditions",{},{"id":325,"data":326,"type":226,"tunes":328},"p-apply-1",{"text":327},"The most dangerous evidence errors are often not fabricated sources but valid sources used outside their boundary. A recommendation can change with software version, date, jurisdiction, hardware revision, user permissions, product availability, current game state, tenant configuration, or other environmental variables.",{},{"id":330,"data":331,"type":226,"tunes":333},"p-apply-2",{"text":332},"For each material source, preserve the conditions that determine whether it still applies. This is especially important after summarization: a compressed memory or citation may preserve the conclusion while dropping the exception, date, or prerequisite that made the conclusion valid.",{},{"id":335,"data":336,"type":234,"tunes":340},"apply-warning",{"body":337,"title":338,"variant":339},"A real source, quoted accurately, can still produce a wrong answer when its time, version, population, jurisdiction, or state does not match the current question.","Evidence can be authentic and still be wrong for the answer","warning",{},{"id":342,"data":343,"type":42,"tunes":345},"h-use",{"text":344,"level":218},"4. Evidence use: did the agent actually rely on the evidence?",{},{"id":347,"data":348,"type":226,"tunes":350},"p-use-1",{"text":349},"An answer can be correct even when retrieval failed. The model may already know the answer, infer it from unrelated context, or simply guess correctly. If the evaluation checks only final correctness, the system may appear well-grounded while the evidence path is broken.",{},{"id":352,"data":353,"type":226,"tunes":355},"p-use-2",{"text":354},"To evaluate evidence use, inspect the trace. Confirm which sources were retrieved, which passages reached the model context, when they became available, and whether the final claim can be explained by those inputs. OpenAI's current agent-evaluation tooling emphasizes trace grading precisely because workflow-level behaviour cannot be reconstructed reliably from the final answer alone.",{},{"id":357,"data":358,"type":42,"tunes":360},"h-eut",{"text":359,"level":219},"The Evidence Utilization Test",{},{"id":362,"data":363,"type":226,"tunes":365},"p-eut-intro",{"text":364},"A practical evaluation can be built as a controlled counterfactual. Instead of asking only whether the answer is correct, change the evidence and observe whether the claim changes in the expected direction.",{},{"id":367,"data":368,"type":393,"tunes":394},"eut-flow",{"steps":369,"title":391,"orientation":392},[370,373,376,379,382,385,388],{"label":371,"description":372},"1. Select one material claim","Choose a claim whose correctness matters and define the expected answer precisely.",{"label":374,"description":375},"2. Identify gold evidence","Provide the smallest authoritative evidence set sufficient to support the claim.",{"label":377,"description":378},"3. Run with gold evidence","Verify that the agent produces the supported answer when the correct evidence is available.",{"label":380,"description":381},"4. Remove the decisive evidence","Run the same task without the key supporting passage while keeping other inputs stable.",{"label":383,"description":384},"5. Replace it with contradictory or superseding evidence","Where safe, provide controlled evidence that changes the correct conclusion.",{"label":386,"description":387},"6. Compare the claims","Check whether the answer tracks the evidence change or remains anchored to prior model knowledge.",{"label":389,"description":390},"7. Inspect the trace","Confirm what was retrieved, what reached context, and what source or tool result preceded the claim.","Evidence Utilization Test","auto","processFlow",{},{"id":396,"data":397,"type":234,"tunes":401},"eut-tip",{"body":398,"title":399,"variant":400},"If the decisive evidence changes but the agent's claim does not, you have evidence that the system may not be using retrieval as intended — even when the original answer happened to be correct.","The key signal","tip",{},{"id":403,"data":404,"type":42,"tunes":406},"h-matrix",{"text":405,"level":219},"A claim–evidence matrix is more useful than a source list",{},{"id":408,"data":409,"type":287,"tunes":441},"claim-matrix",{"content":410,"stretched":43,"withHeadings":14},[411,417,424,430,436],[412,413,414,415,280,416],"Claim","Evidence","Support","Authority","Used in trace",[418,419,420,421,422,423],"Feature X is available","Vendor documentation","Direct","High for availability claim","Current version and region must match","Yes \u002F No",[425,426,427,428,429,423],"Configuration Y is faster","Vendor benchmark","Partial","High for vendor's test, not independent performance","Hardware and workload must match",[431,432,433,434,435,423],"Policy applies to this user","Current policy + account state","Direct only when combined","High","Jurisdiction, date, role and account state must match",[437,438,420,439,440,423],"A product is in stock","Live inventory API","Authoritative for current stock","Expires quickly",{},{"id":443,"data":444,"type":226,"tunes":446},"p-matrix-1",{"text":445},"This matrix forces several questions that conventional citation checking hides. One claim can require multiple sources. One source can support only part of a claim. An authoritative source can have a short validity window. And a perfectly good source can be irrelevant if it never entered the execution path.",{},{"id":448,"data":449,"type":42,"tunes":451},"h-retrieval-evidence",{"text":450,"level":219},"Separate retrieval quality from evidence quality",{},{"id":453,"data":454,"type":226,"tunes":456},"p-re-1",{"text":455},"Retrieval metrics ask whether relevant material was found and ranked. Evidence evaluation asks whether that material justifies the resulting claims. The two are related but not identical.",{},{"id":458,"data":459,"type":489,"tunes":490},"retrieval-comparison",{"rows":460,"title":478,"layout":287,"columns":479},[461,466,470,474],{"id":462,"label":463,"values":464},"a","Right document, wrong claim",[465,465,465],"",{"id":467,"label":468,"values":469},"b","Right fact, stale source",[465,465,465],{"id":471,"label":472,"values":473},"c","Weak source, correct answer",[465,465,465],{"id":475,"label":476,"values":477},"d","Multiple sources required",[465,465,465],"Retrieval success is not evidence success",[480,483,486],{"id":481,"label":482},"situation","Situation",{"id":484,"label":485},"retrieval","Retrieval",{"id":487,"label":488},"evidence","Evidence quality","comparison",{},{"id":492,"data":493,"type":42,"tunes":495},"h-source-quality",{"text":494,"level":219},"Evaluate source quality as a rubric, not a domain whitelist",{},{"id":497,"data":498,"type":226,"tunes":500},"p-source-quality-1",{"text":499},"Hard-coded lists of “trusted domains” are tempting but often brittle. Source quality should instead reflect the claim type. Useful dimensions include primary versus secondary status, recency, directness, reproducibility, independence, domain expertise, data provenance, update cadence, and whether the source has an incentive to overstate the claim.",{},{"id":502,"data":503,"type":226,"tunes":505},"p-source-quality-2",{"text":504},"Anthropic's research-agent evaluation guidance explicitly calls out source-quality checks alongside groundedness and coverage. The practical implementation should therefore grade both what the source says and whether this source is appropriate for this kind of statement.",{},{"id":507,"data":508,"type":42,"tunes":510},"h-coverage",{"text":509,"level":219},"Evidence coverage: every important claim needs support, not every sentence",{},{"id":512,"data":513,"type":226,"tunes":515},"p-coverage-1",{"text":514},"Not every sentence needs a citation. Transitional language, arithmetic derived transparently from cited values, or clearly marked interpretation may not require a separate source. But every material externally verifiable claim should have enough support that an evaluator can reconstruct why the agent was allowed to say it.",{},{"id":517,"data":518,"type":226,"tunes":520},"p-coverage-2",{"text":519},"Coverage should therefore be weighted by claim importance. Missing support for a decorative detail is not equivalent to missing support for a price, eligibility decision, safety instruction, legal requirement, technical compatibility statement, or recommendation-driving fact.",{},{"id":522,"data":523,"type":42,"tunes":525},"h-provenance",{"text":524,"level":219},"Evidence provenance must survive summarization and memory",{},{"id":527,"data":528,"type":226,"tunes":530},"p-prov-1",{"text":529},"Long-running agents often summarize previous work or write durable memories. If evidence provenance is stripped during that transformation, future agents may retrieve a clean conclusion without knowing whether it came from a user statement, a live API, an old document, a model inference, or an unverified web result.",{},{"id":532,"data":533,"type":226,"tunes":535},"p-prov-2",{"text":534},"For important facts, preserve at least the source identity, retrieval or observation time, evidence type, relevant version or state, and whether the stored text is quoted, summarized, inferred, or derived. Provenance is what allows a later agent to decide whether the evidence should be trusted, refreshed, restricted, or discarded.",{},{"id":537,"data":538,"type":42,"tunes":540},"h-record",{"text":539,"level":219},"A practical evidence record",{},{"id":542,"data":543,"type":287,"tunes":578},"evidence-record",{"content":544,"stretched":43,"withHeadings":14},[545,548,551,554,557,560,563,566,569,572,575],[546,547],"Field","Purpose",[549,550],"claim_id","Identifies the material claim being supported",[552,553],"source_id \u002F source_url \u002F system","Identifies where the evidence came from",[555,556],"evidence_span","Preserves the smallest passage, record, or tool result that supports the claim",[558,559],"retrieved_at \u002F observed_at","Allows freshness and timeline checks",[561,562],"source_version \u002F object_version","Allows supersession and reproducibility checks",[564,565],"authority_role","Explains why this source is suitable for this claim",[567,568],"applicability","Stores relevant date, jurisdiction, product version, user, tenant, state, or other conditions",[570,571],"transformation","Marks whether evidence is raw, quoted, summarized, normalized, or derived",[573,574],"trace_step","Shows when the evidence became available to the agent",[576,577],"support_status","Direct, partial, contradictory, unsupported, or uncertain",{},{"id":580,"data":581,"type":42,"tunes":583},"h-failures",{"text":582,"level":219},"Failure modes that look grounded but are not",{},{"id":585,"data":586,"type":287,"tunes":624},"failure-table",{"content":587,"stretched":43,"withHeadings":14},[588,592,596,600,604,608,612,616,620],[589,590,591],"Failure mode","Why it fools evaluators","What to test",[593,594,595],"Citation decoration","The answer contains sources, so it looks researched","Map each material claim to an exact supporting span",[597,598,599],"Authority mismatch","The source is reputable but not authoritative for the specific fact","Define claim-specific source hierarchy",[601,602,603],"Temporal mismatch","The source was correct when published","Check retrieval time, source date, and superseding evidence",[605,606,607],"Condition stripping","A summary keeps the conclusion but drops exceptions","Compare generated claim with full local source context",[609,610,611],"Post-hoc citation","A plausible source is attached after the answer is generated","Inspect trace ordering and whether evidence preceded the claim",[613,614,615],"Parametric override","The model ignores retrieved evidence and answers from prior knowledge","Run counterfactual evidence-utilization tests",[617,618,619],"Evidence laundering","Model inference is summarized and later stored as if it were a source fact","Preserve transformation type and provenance across memory writes",[621,622,623],"Source majority fallacy","Several secondary pages repeat the same unsupported statement","Trace claims back to independent or primary evidence",{},{"id":626,"data":627,"type":42,"tunes":629},"h-production",{"text":628,"level":219},"How to evaluate the agent in production",{},{"id":631,"data":632,"type":393,"tunes":659},"production-flow",{"steps":633,"title":658,"orientation":392},[634,637,640,643,646,649,652,655],{"label":635,"description":636},"1. Define material claims","Identify the facts, recommendations, or decisions whose correctness matters to the task.",{"label":638,"description":639},"2. Build gold evidence","Create reference evidence and source-quality expectations for representative cases.",{"label":641,"description":642},"3. Capture traces","Log retrieval queries, tool calls, returned evidence, context construction, model output, and citations.",{"label":644,"description":645},"4. Grade support","Check whether each material claim is directly, partially, contradictorily, or not supported.",{"label":647,"description":648},"5. Grade authority and applicability","Evaluate whether the source is appropriate and whether its conditions match the current task.",{"label":650,"description":651},"6. Run counterfactuals","Remove, replace, or supersede decisive evidence and test whether the answer follows the change.",{"label":653,"description":654},"7. Review high-impact failures","Use human or domain-expert review where automated grading is not reliable enough.",{"label":656,"description":657},"8. Convert failures into eval cases","Add production failures and edge cases to a repeatable regression dataset.","Evidence evaluation pipeline",{},{"id":661,"data":662,"type":226,"tunes":664},"p-production-1",{"text":663},"OpenAI's current evaluation guidance recommends task-specific evals, continuous evaluation, production-derived datasets, and traces for debugging agent behaviour. Anthropic likewise recommends combining grader types for research agents because correctness, source quality, coverage, and groundedness are separate dimensions. Evidence evaluation should follow the same pattern: several narrow graders are more diagnostic than one opaque “quality” score.",{},{"id":666,"data":667,"type":42,"tunes":669},"h-judge",{"text":668,"level":219},"Do not let an LLM judge become the only evidence judge",{},{"id":671,"data":672,"type":226,"tunes":674},"p-judge-1",{"text":673},"LLM graders are useful for scalable claim classification, relevance checks, and pairwise comparisons, but they can share the same blind spots as the system they evaluate. A grader may accept a plausible but unsupported claim, miss a subtle version boundary, or overrate a polished source.",{},{"id":676,"data":677,"type":226,"tunes":679},"p-judge-2",{"text":678},"OpenAI's evaluation guidance recommends calibrating automated graders against human judgment and using clear, scoped criteria. For evidence-heavy systems, deterministic checks should be used wherever possible: timestamps, object versions, permission scope, exact source IDs, retrieval order, document hashes, and whether the evidence was present before the model generated the claim.",{},{"id":681,"data":682,"type":42,"tunes":684},"h-change",{"text":683,"level":219},"What would change this answer?",{},{"id":686,"data":687,"type":226,"tunes":689},"p-change-1",{"text":688},"The evaluation can be simpler when the agent operates over a small, immutable, authoritative corpus and every answer is strictly extractive. In that environment, source authority and applicability are mostly fixed, and claim-to-span support may be enough.",{},{"id":691,"data":692,"type":226,"tunes":694},"p-change-2",{"text":693},"The evaluation must become stricter when the agent mixes web search, long-term memory, live tools, multiple jurisdictions, rapidly changing information, user-specific state, or autonomous actions. In those systems, evidence validity depends not only on the source text but also on when and how the evidence was obtained.",{},{"id":696,"data":697,"type":226,"tunes":699},"p-change-3",{"text":698},"Future models may become better at internally tracking provenance and uncertainty, but that would not remove the need for external evidence records in systems that require auditability. A system should not depend on the model's self-report of what influenced it when traces and source metadata can provide stronger evidence.",{},{"id":701,"data":702,"type":42,"tunes":704},"h-limitations",{"text":703,"level":219},"Limitations",{},{"id":706,"data":707,"type":226,"tunes":709},"p-limit-1",{"text":708},"It is not always possible to prove causal evidence use from traces alone. A source can be present in context without influencing the answer, and a model may independently know the same fact. Counterfactual tests strengthen the inference but can themselves change the task distribution.",{},{"id":711,"data":712,"type":226,"tunes":714},"p-limit-2",{"text":713},"Source authority can also be contested or domain-dependent. Some questions have no single authoritative source, and experts may disagree about which evidence deserves more weight. In those cases the evaluator should preserve disagreement and score transparency, coverage, and reasoning against an explicit rubric rather than pretending there is one unquestioned source of truth.",{},{"id":716,"data":717,"type":42,"tunes":719},"h-conclusion",{"text":718,"level":219},"Conclusion",{},{"id":721,"data":722,"type":226,"tunes":724},"p-conclusion-1",{"text":723},"The question “Did the agent cite a source?” is too weak for production AI. The stronger question is: Did each important claim come from evidence that actually supports it, has the right authority, still applies to the current conditions, and was available in the execution path before the claim was made?",{},{"id":726,"data":727,"type":226,"tunes":729},"p-conclusion-2",{"text":728},"That turns evidence from decoration into an evaluable system property. Capture the trace. Map claims to evidence. Check authority and applicability. Run counterfactual evidence tests. Preserve provenance through summaries and memory. Then a correct answer is not only plausible — it has an evidence path you can inspect.",{},{"id":731,"data":732,"type":737,"tunes":738},"internal-reliability",{"url":733,"title":734,"excerpt":735,"ctaLabel":736},"https:\u002F\u002Fstajic.de\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough","AI Agent Reliability: Why the Final Answer Is Not Enough","Outcome correctness alone cannot prove that an agent's execution path was safe or reliable. This related article explains why trajectories, tools and intermediate decisions matter.","Read the related article","referralArticle",{},{"id":740,"data":741,"type":737,"tunes":746},"internal-reasoning",{"url":742,"title":743,"excerpt":744,"ctaLabel":745},"https:\u002F\u002Fstajic.de\u002Fblog\u002Ffrom-research-protocol-to-a-general-ai-reasoning-framework","From Research Protocol to a General AI Reasoning Framework","A practical reasoning method for separating evidence, assumptions, competing hypotheses and domain-specific validation.","Read the reasoning framework",{},{"id":748,"data":749,"type":42,"tunes":751},"h-faq",{"text":750,"level":219},"FAQ",{},{"id":753,"data":754,"type":753,"tunes":777},"faq",{"items":755,"title":776},[756,760,764,768,772],{"id":757,"answer":758,"question":759},"faq1","No. A citation may be relevant to the topic without supporting the exact claim, may come from the wrong authority, may no longer apply, or may have been attached without materially influencing the generated answer.","Does a citation prove that an AI answer is grounded?",{"id":761,"answer":762,"question":763},"faq2","Use a counterfactual evidence-utilization test: run the task with known-correct evidence, then remove or replace the decisive evidence while keeping other inputs stable. If the answer does not respond to the evidence change, inspect whether the model is relying on prior knowledge or another source.","How can I test whether an AI agent actually used retrieved evidence?",{"id":765,"answer":766,"question":767},"faq3","Groundedness asks whether claims are supported by the supplied evidence. Source quality asks whether the evidence itself is appropriate and authoritative enough for the type of claim being made.","What is the difference between groundedness and source quality?",{"id":769,"answer":770,"question":771},"faq4","The source may be stale, superseded, valid for another version, jurisdiction, user, population, or system state, or may contain conditions that were lost during retrieval or summarization.","Why can a real source still produce a wrong AI answer?",{"id":773,"answer":774,"question":775},"faq5","Log the retrieval query, returned sources, exact evidence spans, timestamps and versions, filters, final context, model output, citations, and trace ordering so evaluators can reconstruct what evidence was available before each material claim.","What should I log for evidence evaluation?","Evaluating evidence use in AI agents",{},{"id":779,"data":780,"type":42,"tunes":782},"h-glossary",{"text":781,"level":219},"Glossary",{},{"id":784,"data":785,"type":784,"tunes":808},"glossary",{"title":786,"entries":787},"Key evidence-evaluation terms",[788,791,794,796,800,804],{"term":272,"anchor":789,"definition":790},"claim-support","The degree to which a specific evidence span directly establishes a generated claim.",{"term":276,"anchor":792,"definition":793},"evidence-authority","How appropriate a source is for establishing a particular type of claim, given its role, provenance and relationship to the underlying fact.",{"term":280,"anchor":567,"definition":795},"The conditions under which evidence remains valid for a claim, including time, version, jurisdiction, user, population, system state or other boundaries.",{"term":797,"anchor":798,"definition":799},"Evidence utilization","evidence-utilization","Whether the agent's output actually responds to and depends on the evidence made available in its execution path.",{"term":801,"anchor":802,"definition":803},"Counterfactual evidence test","counterfactual-evidence-test","An evaluation that removes, replaces or changes decisive evidence to test whether the agent's claim changes appropriately.",{"term":805,"anchor":806,"definition":807},"Provenance","provenance","Metadata that records where evidence came from, when it was obtained, how it was transformed and what version or state it represented.",{},{"id":810,"data":811,"type":42,"tunes":813},"h-sources",{"text":812,"level":219},"Primary sources and further reading",{},{"id":815,"data":816,"type":822,"tunes":823},"src-openai-agent-evals",{"link":817,"meta":818},"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fagent-evals",{"image":819,"title":820,"description":821},{"url":465},"OpenAI — Evaluate Agent Workflows","Guidance on trace grading, workflow-level evaluation, datasets and repeatable eval runs for agents.","linkTool",{},{"id":825,"data":826,"type":822,"tunes":832},"src-openai-eval-best",{"link":827,"meta":828},"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fevaluation-best-practices",{"image":829,"title":830,"description":831},{"url":465},"OpenAI — Evaluation Best Practices","Guidance on task-specific evals, production-derived datasets, scoped metrics, continuous evaluation and grader calibration.",{},{"id":834,"data":835,"type":822,"tunes":841},"src-anthropic-evals",{"link":836,"meta":837},"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Fdemystifying-evals-for-ai-agents",{"image":838,"title":839,"description":840},{"url":465},"Anthropic — Demystifying Evals for AI Agents","Agent-evaluation guidance including groundedness, coverage and source-quality checks for research agents.",{},{"id":843,"data":844,"type":822,"tunes":850},"src-openai-thirdparty",{"link":845,"meta":846},"https:\u002F\u002Fopenai.com\u002Findex\u002Ftrustworthy-third-party-evaluations-foundations\u002F",{"image":847,"title":848,"description":849},{"url":465},"OpenAI — A Shared Playbook for Trustworthy Third-Party Evaluations","Evaluation guidance emphasizing that modern agent performance depends on workflow and environment, not only final model output.",{},"2.31.6","An AI agent can cite sources and still use the wrong evidence. This article introduces a practical method for checking claim support, source authority, applicability, provenance, and whether the evidence actually influenced the answer.","\u002Fuploads\u002F2026\u002F09\u002Fhow-to-know-whether-an-ai-agent-actually-used-the-right-evidence-1790351317188-o5z9ve.webp","how-to-know-whether-an-ai-agent-actually-used-the-right-evidence-1790351317188-o5z9ve","PUBLISHED","2026-09-25T11:47:00.000Z","2026-09-25T15:47:02.186Z","2026-09-25T20:33:01.720Z",{"en":860,"de":861,"sr":862,"es":863,"fr":864,"it":865,"ru":866,"zh":867},"\u002Fblog\u002Fhow-to-know-whether-an-ai-agent-actually-used-the-right-evidence","\u002Fde\u002Fblog\u002Fhow-to-know-whether-an-ai-agent-actually-used-the-right-evidence","\u002Fsr\u002Fblog\u002Fhow-to-know-whether-an-ai-agent-actually-used-the-right-evidence","\u002Fes\u002Fblog\u002Fhow-to-know-whether-an-ai-agent-actually-used-the-right-evidence","\u002Ffr\u002Fblog\u002Fhow-to-know-whether-an-ai-agent-actually-used-the-right-evidence","\u002Fit\u002Fblog\u002Fhow-to-know-whether-an-ai-agent-actually-used-the-right-evidence","\u002Fru\u002Fblog\u002Fhow-to-know-whether-an-ai-agent-actually-used-the-right-evidence","\u002Fzh\u002Fblog\u002Fhow-to-know-whether-an-ai-agent-actually-used-the-right-evidence",[869,873,877],{"id":870,"name":871,"slug":872},84,"Policy & Data Boundaries","policy-and-data",{"id":874,"name":875,"slug":876},97,"Verification on Test Set","verification",{"id":878,"name":879,"slug":880},66,"Content Operations","content-ops",{"id":882,"login":883,"email":884,"displayName":885},"20","rooth8233","aleksandar@stajic.de","Aleksandar Stajić",[887],{"lang":7,"title":208,"content":210,"contentJson":888,"excerpt":852},{"time":212,"blocks":889,"version":851},[890,893,896,899,902,905,908,911,914,923,926,929,932,935,938,941,944,947,950,953,956,959,962,965,968,979,982,985,994,997,1000,1003,1019,1022,1025,1028,1031,1034,1037,1040,1043,1046,1049,1064,1067,1080,1083,1095,1098,1101,1104,1107,1110,1113,1116,1119,1122,1125,1128,1131,1134,1137,1140,1143,1146,1155,1158,1168,1171,1176,1181,1186],{"id":215,"data":891,"type":220,"tunes":892},{"title":217,"maxLevel":218,"minLevel":219},{},{"id":223,"data":894,"type":226,"tunes":895},{"text":225},{},{"id":229,"data":897,"type":234,"tunes":898},{"body":231,"title":232,"variant":233},{},{"id":237,"data":900,"type":234,"tunes":901},{"body":239,"title":240,"variant":241},{},{"id":244,"data":903,"type":42,"tunes":904},{"text":246,"level":219},{},{"id":249,"data":906,"type":226,"tunes":907},{"text":251},{},{"id":254,"data":909,"type":226,"tunes":910},{"text":256},{},{"id":259,"data":912,"type":42,"tunes":913},{"text":261,"level":219},{},{"id":264,"data":915,"type":287,"tunes":922},{"content":916,"stretched":43,"withHeadings":14},[917,918,919,920,921],[268,269,270],[272,273,274],[276,277,278],[280,281,282],[284,285,286],{},{"id":290,"data":924,"type":42,"tunes":925},{"text":292,"level":218},{},{"id":295,"data":927,"type":226,"tunes":928},{"text":297},{},{"id":300,"data":930,"type":226,"tunes":931},{"text":302},{},{"id":305,"data":933,"type":42,"tunes":934},{"text":307,"level":218},{},{"id":310,"data":936,"type":226,"tunes":937},{"text":312},{},{"id":315,"data":939,"type":226,"tunes":940},{"text":317},{},{"id":320,"data":942,"type":42,"tunes":943},{"text":322,"level":218},{},{"id":325,"data":945,"type":226,"tunes":946},{"text":327},{},{"id":330,"data":948,"type":226,"tunes":949},{"text":332},{},{"id":335,"data":951,"type":234,"tunes":952},{"body":337,"title":338,"variant":339},{},{"id":342,"data":954,"type":42,"tunes":955},{"text":344,"level":218},{},{"id":347,"data":957,"type":226,"tunes":958},{"text":349},{},{"id":352,"data":960,"type":226,"tunes":961},{"text":354},{},{"id":357,"data":963,"type":42,"tunes":964},{"text":359,"level":219},{},{"id":362,"data":966,"type":226,"tunes":967},{"text":364},{},{"id":367,"data":969,"type":393,"tunes":978},{"steps":970,"title":391,"orientation":392},[971,972,973,974,975,976,977],{"label":371,"description":372},{"label":374,"description":375},{"label":377,"description":378},{"label":380,"description":381},{"label":383,"description":384},{"label":386,"description":387},{"label":389,"description":390},{},{"id":396,"data":980,"type":234,"tunes":981},{"body":398,"title":399,"variant":400},{},{"id":403,"data":983,"type":42,"tunes":984},{"text":405,"level":219},{},{"id":408,"data":986,"type":287,"tunes":993},{"content":987,"stretched":43,"withHeadings":14},[988,989,990,991,992],[412,413,414,415,280,416],[418,419,420,421,422,423],[425,426,427,428,429,423],[431,432,433,434,435,423],[437,438,420,439,440,423],{},{"id":443,"data":995,"type":226,"tunes":996},{"text":445},{},{"id":448,"data":998,"type":42,"tunes":999},{"text":450,"level":219},{},{"id":453,"data":1001,"type":226,"tunes":1002},{"text":455},{},{"id":458,"data":1004,"type":489,"tunes":1018},{"rows":1005,"title":478,"layout":287,"columns":1014},[1006,1008,1010,1012],{"id":462,"label":463,"values":1007},[465,465,465],{"id":467,"label":468,"values":1009},[465,465,465],{"id":471,"label":472,"values":1011},[465,465,465],{"id":475,"label":476,"values":1013},[465,465,465],[1015,1016,1017],{"id":481,"label":482},{"id":484,"label":485},{"id":487,"label":488},{},{"id":492,"data":1020,"type":42,"tunes":1021},{"text":494,"level":219},{},{"id":497,"data":1023,"type":226,"tunes":1024},{"text":499},{},{"id":502,"data":1026,"type":226,"tunes":1027},{"text":504},{},{"id":507,"data":1029,"type":42,"tunes":1030},{"text":509,"level":219},{},{"id":512,"data":1032,"type":226,"tunes":1033},{"text":514},{},{"id":517,"data":1035,"type":226,"tunes":1036},{"text":519},{},{"id":522,"data":1038,"type":42,"tunes":1039},{"text":524,"level":219},{},{"id":527,"data":1041,"type":226,"tunes":1042},{"text":529},{},{"id":532,"data":1044,"type":226,"tunes":1045},{"text":534},{},{"id":537,"data":1047,"type":42,"tunes":1048},{"text":539,"level":219},{},{"id":542,"data":1050,"type":287,"tunes":1063},{"content":1051,"stretched":43,"withHeadings":14},[1052,1053,1054,1055,1056,1057,1058,1059,1060,1061,1062],[546,547],[549,550],[552,553],[555,556],[558,559],[561,562],[564,565],[567,568],[570,571],[573,574],[576,577],{},{"id":580,"data":1065,"type":42,"tunes":1066},{"text":582,"level":219},{},{"id":585,"data":1068,"type":287,"tunes":1079},{"content":1069,"stretched":43,"withHeadings":14},[1070,1071,1072,1073,1074,1075,1076,1077,1078],[589,590,591],[593,594,595],[597,598,599],[601,602,603],[605,606,607],[609,610,611],[613,614,615],[617,618,619],[621,622,623],{},{"id":626,"data":1081,"type":42,"tunes":1082},{"text":628,"level":219},{},{"id":631,"data":1084,"type":393,"tunes":1094},{"steps":1085,"title":658,"orientation":392},[1086,1087,1088,1089,1090,1091,1092,1093],{"label":635,"description":636},{"label":638,"description":639},{"label":641,"description":642},{"label":644,"description":645},{"label":647,"description":648},{"label":650,"description":651},{"label":653,"description":654},{"label":656,"description":657},{},{"id":661,"data":1096,"type":226,"tunes":1097},{"text":663},{},{"id":666,"data":1099,"type":42,"tunes":1100},{"text":668,"level":219},{},{"id":671,"data":1102,"type":226,"tunes":1103},{"text":673},{},{"id":676,"data":1105,"type":226,"tunes":1106},{"text":678},{},{"id":681,"data":1108,"type":42,"tunes":1109},{"text":683,"level":219},{},{"id":686,"data":1111,"type":226,"tunes":1112},{"text":688},{},{"id":691,"data":1114,"type":226,"tunes":1115},{"text":693},{},{"id":696,"data":1117,"type":226,"tunes":1118},{"text":698},{},{"id":701,"data":1120,"type":42,"tunes":1121},{"text":703,"level":219},{},{"id":706,"data":1123,"type":226,"tunes":1124},{"text":708},{},{"id":711,"data":1126,"type":226,"tunes":1127},{"text":713},{},{"id":716,"data":1129,"type":42,"tunes":1130},{"text":718,"level":219},{},{"id":721,"data":1132,"type":226,"tunes":1133},{"text":723},{},{"id":726,"data":1135,"type":226,"tunes":1136},{"text":728},{},{"id":731,"data":1138,"type":737,"tunes":1139},{"url":733,"title":734,"excerpt":735,"ctaLabel":736},{},{"id":740,"data":1141,"type":737,"tunes":1142},{"url":742,"title":743,"excerpt":744,"ctaLabel":745},{},{"id":748,"data":1144,"type":42,"tunes":1145},{"text":750,"level":219},{},{"id":753,"data":1147,"type":753,"tunes":1154},{"items":1148,"title":776},[1149,1150,1151,1152,1153],{"id":757,"answer":758,"question":759},{"id":761,"answer":762,"question":763},{"id":765,"answer":766,"question":767},{"id":769,"answer":770,"question":771},{"id":773,"answer":774,"question":775},{},{"id":779,"data":1156,"type":42,"tunes":1157},{"text":781,"level":219},{},{"id":784,"data":1159,"type":784,"tunes":1167},{"title":786,"entries":1160},[1161,1162,1163,1164,1165,1166],{"term":272,"anchor":789,"definition":790},{"term":276,"anchor":792,"definition":793},{"term":280,"anchor":567,"definition":795},{"term":797,"anchor":798,"definition":799},{"term":801,"anchor":802,"definition":803},{"term":805,"anchor":806,"definition":807},{},{"id":810,"data":1169,"type":42,"tunes":1170},{"text":812,"level":219},{},{"id":815,"data":1172,"type":822,"tunes":1175},{"link":817,"meta":1173},{"image":1174,"title":820,"description":821},{"url":465},{},{"id":825,"data":1177,"type":822,"tunes":1180},{"link":827,"meta":1178},{"image":1179,"title":830,"description":831},{"url":465},{},{"id":834,"data":1182,"type":822,"tunes":1185},{"link":836,"meta":1183},{"image":1184,"title":839,"description":840},{"url":465},{},{"id":843,"data":1187,"type":822,"tunes":1190},{"link":845,"meta":1188},{"image":1189,"title":848,"description":849},{"url":465},{},"Post erfolgreich abgerufen",{"items":1193,"source":1257,"manualIds":1258,"manualMatchedIds":1259},[1194,1201,1208,1215,1222,1229,1236,1243,1250],{"id":1195,"slug":1196,"title":1197,"excerpt":1198,"featuredImage":1199,"publishedAt":1200},"383","canonical-architecture-url-design-resolver-logic-api-scalability-specification","Canonical Architecture, URL Design, Resolver Logic, API & Scalability Specification","Geo-based discovery architecture for multi-tenant portals. Defines canonical URLs, resolver logic, caching strategy, and a geo read-model without CMS coupling or database refactoring. Designed for SEO stability, scalability, and future extensions like booking and maps.","\u002Fuploads\u002F2026\u002F01\u002Fcanonical-architecture-url-design-resolver-logic-api-scalability-specification-1769890763607-7rghbp.webp","2026-01-31T06:12:00.000Z",{"id":1202,"slug":1203,"title":1204,"excerpt":1205,"featuredImage":1206,"publishedAt":1207},"476","mcp-vs-a2a-vs-ucp-vs-ap2-vs-a2ui-the-agent-protocol-stack-explained","MCP vs A2A vs UCP vs AP2 vs A2UI: The Agent Protocol Stack Explained","MCP, A2A, UCP, AP2 and A2UI are often presented as competing agent standards. They mostly solve different interoperability problems. This guide maps each protocol to the boundary it actually standardizes—and shows how they can work together in one production system.","\u002Fuploads\u002F2026\u002F09\u002Fmcp-vs-a2a-vs-ucp-vs-ap2-vs-a2ui-the-agent-protocol-stack-explained-1790352625869-2ezle0.webp","2026-09-25T12:09:00.000Z",{"id":1209,"slug":1210,"title":1211,"excerpt":1212,"featuredImage":1213,"publishedAt":1214},"467","the-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers","The Answer Validity Boundary: The Missing Layer Between Relevance and Reliable AI Answers","A source can be relevant, authoritative and still be wrong for the question being asked. The missing layer is applicability: the conditions under which an answer holds, and the changes that force it to be reconsidered. This article introduces the Answer Validity Boundary as a source-design pattern for humans, AI search and RAG systems.","\u002Fuploads\u002F2026\u002F09\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers-1790272901306-1g5jly.webp","2026-09-24T11:59:00.000Z",{"id":1216,"slug":1217,"title":1218,"excerpt":1219,"featuredImage":1220,"publishedAt":1221},"457","should-you-buy-5g-openwrt-router-old-firmware","Should You Buy a 5G OpenWrt Router with Old Firmware? ZBT Z8102AX as a Practical Example","Buying a 5G OpenWrt router with older firmware can make sense, but only under the right conditions. The ZBT Z8102AX shows both sides clearly: the hardware is useful, the modem works, and the router stayed stable in testing, but OpenWrt 21.02, weak packaging and unclear upgrade paths require a careful buying decision.","\u002Fuploads\u002F2026\u002F06\u002Fopenwrt-router-review-dual-sim-05-1781620596218-5ldld4.webp","2026-06-16T10:41:00.000Z",{"id":1223,"slug":1224,"title":1225,"excerpt":1226,"featuredImage":1227,"publishedAt":1228},"472","why-more-context-can-make-ai-answers-worse","Why More Context Can Make AI Answers Worse","A larger context window does not guarantee a better answer. This article explains how signal dilution, conflicting evidence, stale state, position sensitivity, and lossy compression can reduce AI reliability—and introduces a practical Context Pressure Test.","\u002Fuploads\u002F2026\u002F09\u002Fwhy-more-context-can-make-ai-answers-worse-1790351615793-2ntv2v.webp","2026-09-25T11:51:00.000Z",{"id":1230,"slug":1231,"title":1232,"excerpt":1233,"featuredImage":1234,"publishedAt":1235},"454","zbt-z8102ax-rm500u-ea-5g-modem-test","Quectel RM500U-EA in the ZBT Z8102AX: 5G Bands, o2 Germany and Real-World Signal Behavior","The ZBT Z8102AX uses a Quectel RM500U-EA modem for 4G and 5G connectivity. In the first practical test, the router connected successfully to o2 Germany with LTE Band 3 and NR n28. The modem works, but deeper diagnostics such as RSRP, RSRQ, SINR, band locking and cell behavior still need proper testing.","\u002Fuploads\u002F2026\u002F06\u002Fopenwrt-router-review-dual-sim-06-1781620597879-qay2sx.webp","2026-06-16T08:39:00.000Z",{"id":1237,"slug":1238,"title":1239,"excerpt":1240,"featuredImage":1241,"publishedAt":1242},"470","what-should-an-ai-agent-remember-forget-recompute-or-retrieve-again","What Should an AI Agent Remember, Forget, Recompute or Retrieve Again?","Long-running agents should not remember everything. This article provides a practical lifecycle model for deciding what belongs in durable memory, what should be retrieved again, what is safer to recompute, and what should expire or be superseded.","\u002Fuploads\u002F2026\u002F09\u002Fwhat-should-an-ai-agent-remember-forget-recompute-or-retrieve-again-1790351131087-iehz28.webp","2026-09-25T09:43:00.000Z",{"id":1244,"slug":1245,"title":1246,"excerpt":1247,"featuredImage":1248,"publishedAt":1249},"456","zbt-z8102ax-hardware-packaging-review","ZBT Z8102AX Hardware and Packaging Review: Strong Router, Weak Box","The ZBT Z8102AX makes a solid first impression as a slim black metal 5G OpenWrt router with multiple antenna connectors, dual-SIM slots, USB, LAN\u002FWAN ports and a practical accessory set. The hardware feels useful and serious, but the packaging is clearly the weak point.","\u002Fuploads\u002F2026\u002F06\u002Fopenwrt-router-review-dual-sim-02-1781620590938-y33j4b.webp","2026-06-16T04:40:00.000Z",{"id":1251,"slug":1252,"title":1253,"excerpt":1254,"featuredImage":1255,"publishedAt":1256},"477","computer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system","Computer-Use Agents: Why a Successful Demo Can Still Be an Unreliable System","Computer-use agents can now complete impressive browser and desktop workflows, but one successful run proves capability—not reliability. This article shows how to test repeatability, environmental robustness, long-horizon control, state awareness, outcome verification, and safe goal handling.","\u002Fuploads\u002F2026\u002F09\u002Fcomputer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system-1790352854690-75qnrg.webp","2026-09-25T12:13:00.000Z","fallback",[],[]]