[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"portal-settings:stajic:en":3,"public-menus:all":38,"post:computer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system:en":205,"related:post:computer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system:en:1":1329},{"statusCode":4,"data":5,"message":37},200,{"tenantId":6,"lang":7,"defaultLang":8,"siteUrl":9,"contactEmail":10,"brandName":11,"logoUrl":12,"siteName":11,"siteDescription":13,"ogImage":10,"robotsIndex":14,"socialLinks":10,"reservedSlugs":10,"seoPolicy":15},"stajic","en","de","https:\u002F\u002Fstajic.de",null,"Stajic Platform","\u002FLogo_Planet.svg","Stajic Portal",true,{"branding":16,"relatedContent":17,"crossDomainLinks":18},{"logoUrl":12},{"enabled":14},[19,22,25,28,31,34],{"url":20,"label":21,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Ffigure.rocks","figure.rocks",{"url":23,"label":24,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Floving.rocks","loving.rocks",{"url":26,"label":27,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.com","bazify.com",{"url":29,"label":30,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.de","bazify.de",{"url":32,"label":33,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.at","bazify.at",{"url":35,"label":36,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.ba","bazify.ba","Portal settings resolved",[39,45],{"id":40,"name":41,"location":42,"isActive":14,"isDefault":43,"items":44},1,"main-navigation","header",false,[],{"id":46,"name":47,"location":48,"isActive":14,"isDefault":14,"items":49},4,"main-menu","sidebar",[50,66,79,93,103,118,133],{"id":51,"title":52,"url":60,"target":61,"icon":62,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":64,"portfolioId":10,"children":65},"item-18",{"de":53,"en":54,"es":55,"fr":56,"it":54,"ru":57,"sr":58,"zh":59},"Startseite","Home","Inicio","Accueil","Главная","Почетна","首页","\u002Ffull-stack-web-developer-munich-performance-seo-and-maintainable-builds","_self","i-lucide-home","page",111,[],{"id":67,"title":68,"url":75,"target":61,"icon":76,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":77,"portfolioId":10,"children":78},"item-22",{"de":69,"en":69,"es":70,"fr":69,"it":71,"ru":72,"sr":73,"zh":74},"Vision","Visión","Visione","Видение","Визија","想象","\u002Fueber-uns-webdesign-muenchen-webaplikation","i-lucide-eye",113,[],{"id":80,"title":81,"url":89,"target":61,"icon":90,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":91,"portfolioId":10,"children":92},"item-19",{"de":82,"en":83,"es":84,"fr":83,"it":85,"ru":86,"sr":87,"zh":88},"Leistungen","Services","Servicios","Servizi","Услуги","Услуге","服务","\u002Fservices-dienstleistungen-muenchen","i-lucide-wrench",116,[],{"id":94,"title":95,"url":99,"target":61,"icon":100,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":101,"portfolioId":10,"children":102},"item-23",{"de":96,"en":96,"es":96,"fr":96,"it":96,"ru":97,"sr":97,"zh":98},"Blog","Блог","博客","\u002Fblog","i-lucide-book-open",112,[],{"id":104,"title":105,"url":114,"target":61,"icon":115,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":116,"portfolioId":10,"children":117},"item-32",{"de":106,"en":107,"es":108,"fr":109,"it":110,"ru":111,"sr":112,"zh":113},"Neue Technologien","New Technologies","Nuevas tecnologías","Nouvelles technologies","Nuove tecnologie","Новые технологии","Нове технологије","新技术！","\u002Fneue-webtechnologien","i-lucide-sparkles",122,[],{"id":119,"title":120,"url":129,"target":61,"icon":130,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":131,"portfolioId":10,"children":132},"item-20",{"de":121,"en":122,"es":123,"fr":124,"it":125,"ru":126,"sr":127,"zh":128},"Kontakt","Contact us!","Contacto","Contact","Contatto","Контакт","Контактирајте нас","联系我们！","\u002Fcontact","i-lucide-mail",115,[],{"id":134,"title":135,"url":144,"target":61,"icon":145,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":147},"item-21",{"de":136,"en":137,"es":138,"fr":139,"it":140,"ru":141,"sr":142,"zh":143},"Unsere Arbeit","Our Work","Nuestro trabajo","Nos réalisations","I nostri lavori","Наши работы","Наши радови","文件夹","\u002Fportfolio","i-lucide-briefcase",114,[148,161,175,181,193],{"id":149,"title":150,"url":144,"target":61,"icon":159,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":160},"item-24",{"de":151,"en":152,"es":153,"fr":154,"it":155,"ru":156,"sr":157,"zh":158},"Alle Projekte","All Projects","Todos los proyectos","Tous les projets","Tutti i progetti","Все проекты","Сви пројекти","所有项目","i-lucide-grid-3x3",[],{"id":162,"title":163,"url":171,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":174},"item-29",{"de":164,"en":165,"es":166,"fr":167,"it":168,"ru":169,"sr":170,"zh":143},"Local Roots, Global Reach","Local Roots - Global Reach","Empresa local ","Entreprise locale","Azienda locale","Местная компания","Локално предузеће глобално тржиште","\u002Fportfolio\u002Flocal-roots-global-reach-communication-media-systems-for-modern-business","i-lucide-folder","custom",[],{"id":176,"title":177,"url":179,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":180},"item-28",{"de":178,"en":178,"es":178,"fr":178,"it":178,"ru":178,"sr":178,"zh":178},"Solr Suggester","\u002Fportfolio\u002Fsolr-fuzzy-suggester-und-solr-infix-suggester-abfrage-ueber-ajax-und-filterung",[],{"id":182,"title":183,"url":191,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":192},"item-27",{"de":184,"en":185,"es":186,"fr":187,"it":188,"ru":189,"sr":190,"zh":185},"Firmenwebseite SEO","Company Website SEO","Sitio web corporativo SEO","Site web d’entreprise SEO","Sito web aziendale SEO","Корпоративный сайт SEO","Пословна веб-страница SEO","\u002Fportfolio\u002Fseo-sem-branding-mobile-webseite-muenchen",[],{"id":194,"title":195,"url":203,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":204},"item-31",{"de":196,"en":197,"es":198,"fr":199,"it":200,"ru":201,"sr":202,"zh":197},"Digitalisierungsportal","Digitalization Portal","Portal de digitalización","Portail de numérisation","Portale di digitalizzazione","Портал цифровизации","Портал за дигитализацију","\u002Fportfolio\u002Fdigitalisierungsportal-archiv-museum-bibliothek-ead-lido-mets-mods",[],{"statusCode":4,"data":206,"message":1328},{"id":207,"title":208,"slug":209,"content":210,"contentJson":211,"excerpt":939,"featuredImage":940,"featuredImageAlt":941,"featuredImageCaption":10,"featuredImageTitle":10,"featuredImageCopyright":10,"featuredImageAuthor":10,"featuredImageSourceUrl":10,"featuredImageLicense":10,"featuredImageIsAiGenerated":43,"status":942,"publishedAt":943,"createdAt":944,"updatedAt":945,"seoLocalePaths":946,"categories":955,"author":968,"translations":973},"477","Computer-Use Agents: Why a Successful Demo Can Still Be an Unreliable System","computer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system","{\"time\":1790352872794,\"blocks\":[{\"id\":\"IYG9UPcPY0\",\"type\":\"tableOfContents\",\"data\":{\"title\":\"Contents\",\"minLevel\":2,\"maxLevel\":3},\"tunes\":{}},{\"id\":\"intro\",\"type\":\"paragraph\",\"data\":{\"text\":\"Computer-use agents can now click, type, browse, edit files, operate desktop applications, and complete impressive multi-step tasks. That makes successful demos easy to understand and easy to overinterpret. A single completed workflow shows that the agent can succeed under those conditions. It does not show how often it succeeds, how it behaves when the environment changes, whether it verifies the result, or how safely it acts when the goal becomes ambiguous.\"},\"tunes\":{}},{\"id\":\"direct\",\"type\":\"callout\",\"data\":{\"variant\":\"info\",\"title\":\"Direct answer\",\"body\":\"\u003Cstrong>A successful computer-use demo proves capability, not reliability.\u003C\u002Fstrong> Production reliability requires the agent to succeed repeatedly across environmental variation, recover from transient failures, preserve constraints over long horizons, detect hidden or changing state, verify the actual outcome, and stop or ask when the goal becomes ambiguous or unsafe. The correct production question is not “Can the agent do this task?” but “Under which conditions can we trust it to do this task repeatedly?”\"},\"tunes\":{}},{\"id\":\"freshness\",\"type\":\"callout\",\"data\":{\"variant\":\"warning\",\"title\":\"Fast-moving field\",\"body\":\"This article reflects computer-use agent research and platform guidance available on \u003Cstrong>25 September 2026\u003C\u002Fstrong>. Benchmark results are not directly comparable across different task sets, environments, models, step limits, judges, or harnesses. Treat every benchmark number together with its evaluation conditions.\"},\"tunes\":{}},{\"id\":\"model-note\",\"type\":\"callout\",\"data\":{\"variant\":\"note\",\"title\":\"The model used in this article\",\"body\":\"The Computer-Use Reliability Ladder and Demo-to-Production Stress Test below are practical evaluation models proposed here. They are not formal industry standards.\"},\"tunes\":{}},{\"id\":\"h-demo\",\"type\":\"header\",\"data\":{\"text\":\"Why the demo is the easiest possible reliability test\",\"level\":2},\"tunes\":{}},{\"id\":\"p-demo-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A demo normally shows one trajectory that worked. The environment is known, the task is selected in advance, the operator can restart after a failure, and the audience sees the successful path. Production systems face a distribution instead: different pages, network conditions, account states, pop-ups, latency, UI changes, hidden state, permissions, interruptions, and users who describe goals imperfectly.\"},\"tunes\":{}},{\"id\":\"p-demo-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"That distinction matters because computer-use agents operate through interfaces designed for humans rather than deterministic APIs. Their action loop depends on perception, state interpretation, planning, interaction timing, and environment response. Small changes can alter the trajectory even when the user goal is unchanged.\"},\"tunes\":{}},{\"id\":\"p-demo-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"Microsoft Research's WAREX work makes the problem explicit: benchmark agents that look capable in controlled settings lose substantial task success when realistic web instability is introduced. The failure is not necessarily “the model became less intelligent.” The environment stopped being deterministic.\"},\"tunes\":{}},{\"id\":\"h-claims\",\"type\":\"header\",\"data\":{\"text\":\"Capability, success rate, reliability, and safety are different claims\",\"level\":2},\"tunes\":{}},{\"id\":\"claims-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Claim\",\"What it actually establishes\",\"What it does not establish\"],[\"The agent completed the task once\",\"Capability under one observed trajectory\",\"Repeatability, robustness, safety, or generalization\"],[\"The agent scores highly on a benchmark\",\"Performance under that benchmark's task and evaluation conditions\",\"Equivalent production performance on different environments\"],[\"The agent usually reaches the goal\",\"Outcome success frequency\",\"Correct process, safe behaviour, or evidence that the result was verified\"],[\"The agent follows the intended process\",\"Trajectory quality under the evaluated rubric\",\"That the external environment actually accepted the final outcome\"],[\"The agent avoids unsafe actions in a test set\",\"Performance on represented safety cases\",\"Safety under every novel ambiguity, injection, or side effect\"]]},\"tunes\":{}},{\"id\":\"h-ladder\",\"type\":\"header\",\"data\":{\"text\":\"The Computer-Use Reliability Ladder\",\"level\":2},\"tunes\":{}},{\"id\":\"p-ladder-intro\",\"type\":\"paragraph\",\"data\":{\"text\":\"A useful way to evaluate computer-use systems is to move from one-off capability toward progressively harder reliability properties. Higher levels assume the lower levels but do not follow automatically from them.\"},\"tunes\":{}},{\"id\":\"ladder-flow\",\"type\":\"processFlow\",\"data\":{\"title\":\"Computer-Use Reliability Ladder\",\"orientation\":\"auto\",\"steps\":[{\"label\":\"1. Capability\",\"description\":\"Can the agent complete the task at least once under known conditions?\"},{\"label\":\"2. Repeatability\",\"description\":\"Can it complete the same task consistently across repeated trials?\"},{\"label\":\"3. Environmental robustness\",\"description\":\"Does it survive timing changes, network issues, pop-ups, UI variation, and small environmental perturbations?\"},{\"label\":\"4. Long-horizon control\",\"description\":\"Can it preserve goals, constraints, and progress across many steps, applications, and delayed events?\"},{\"label\":\"5. State awareness\",\"description\":\"Can it detect when the environment changed, when hidden state matters, or when an assumption is no longer valid?\"},{\"label\":\"6. Outcome verification\",\"description\":\"Does it verify that the intended result actually happened instead of trusting its own action sequence?\"},{\"label\":\"7. Safe goal handling\",\"description\":\"Can it stop, ask, refuse, or hand control back when the goal is ambiguous, infeasible, contradictory, or high impact?\"}]},\"tunes\":{}},{\"id\":\"h-capability\",\"type\":\"header\",\"data\":{\"text\":\"Level 1 — Capability: the demo question\",\"level\":3},\"tunes\":{}},{\"id\":\"p-capability-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Capability asks whether an agent can perform the task at all. This is valuable. Computer-use systems have advanced rapidly, and modern agents can complete workflows that older systems could not execute reliably.\"},\"tunes\":{}},{\"id\":\"p-capability-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"But capability is a weak deployment criterion. One successful run does not tell you whether the agent succeeds 95% of the time or 30% of the time, whether failures are harmless or destructive, or whether success depends on a lucky page state.\"},\"tunes\":{}},{\"id\":\"h-repeatability\",\"type\":\"header\",\"data\":{\"text\":\"Level 2 — Repeatability: does the same task stay solved?\",\"level\":3},\"tunes\":{}},{\"id\":\"p-repeat-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Computer-use trajectories are stochastic. Model outputs vary, pages load at different speeds, visual states change, and long workflows create many branching opportunities. A production test should therefore run the same task multiple times rather than treating one passing trace as representative.\"},\"tunes\":{}},{\"id\":\"p-repeat-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Measure not only the average success rate but also the distribution of failure modes: wrong click, premature termination, missed confirmation, incorrect field, duplicate action, navigation loop, stale-state assumption, and false success report.\"},\"tunes\":{}},{\"id\":\"h-robustness\",\"type\":\"header\",\"data\":{\"text\":\"Level 3 — Environmental robustness: what happens when the web behaves like the web?\",\"level\":3},\"tunes\":{}},{\"id\":\"p-robust-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Real websites are not benchmark fixtures. Requests fail, elements load late, sessions expire, pages change, consent banners appear, servers return errors, and network conditions fluctuate.\"},\"tunes\":{}},{\"id\":\"p-robust-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"WAREX evaluates this gap by injecting realistic web unreliability into existing benchmark environments and reports significant drops in task success. This is a critical production insight: a benchmark can measure task competence while under-measuring recovery from environmental instability.\"},\"tunes\":{}},{\"id\":\"robust-tip\",\"type\":\"callout\",\"data\":{\"variant\":\"tip\",\"title\":\"Reliability test\",\"body\":\"Inject delays, transient HTTP failures, stale page state, modal dialogs, session expiration, duplicate responses, and controlled UI variation. If the agent only works on the clean path, it is a demo-capable system, not a production-reliable one.\"},\"tunes\":{}},{\"id\":\"h-long\",\"type\":\"header\",\"data\":{\"text\":\"Level 4 — Long-horizon control: success changes when the task becomes real work\",\"level\":3},\"tunes\":{}},{\"id\":\"p-long-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Short tasks hide a class of failures that appear only after dozens or hundreds of actions: forgotten constraints, duplicated work, premature completion, missed state changes, cross-application inconsistencies, and accumulated small errors.\"},\"tunes\":{}},{\"id\":\"p-long-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"OSWorld 2.0 was designed specifically around long-horizon real-world workflows. Its tasks take human users a median of roughly 1.6 hours and require many more tool calls than earlier computer-use benchmarks. Under its primary completion metric, even the strongest evaluated systems remain far from complete task reliability.\"},\"tunes\":{}},{\"id\":\"p-long-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"WeaveBench reaches a similar conclusion from another angle. It evaluates hybrid GUI, CLI and code workflows and reports that the best evaluated model-runtime pairing passes only 41.2% of tasks. The important result is not one leaderboard number; it is that realistic cross-interface orchestration exposes failures hidden by simpler single-interface tasks.\"},\"tunes\":{}},{\"id\":\"h-state\",\"type\":\"header\",\"data\":{\"text\":\"Level 5 — State awareness: the environment can change underneath the plan\",\"level\":3},\"tunes\":{}},{\"id\":\"p-state-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Long-running tasks often depend on hidden or changing state: an email arrives, a calendar changes, a form is submitted, a background process finishes, a browser session expires, a user modifies a file, or an external system changes availability.\"},\"tunes\":{}},{\"id\":\"p-state-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Microsoft's SentinelBench argues that many long-running tasks should not be solved through continuous action at all. The correct behaviour may be to monitor, wait for an external event, then act when the state changes. This is a different capability from clicking faster or planning more steps.\"},\"tunes\":{}},{\"id\":\"p-state-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"A reliable computer-use agent therefore needs to distinguish actionable now, waiting for state, state changed, and assumption invalidated.\"},\"tunes\":{}},{\"id\":\"h-verify\",\"type\":\"header\",\"data\":{\"text\":\"Level 6 — Outcome verification: did the action actually work?\",\"level\":3},\"tunes\":{}},{\"id\":\"p-verify-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"An agent can execute an apparently correct sequence and still fail the task. A button click may not register. A form may reject hidden validation. A file may save to the wrong directory. A purchase may remain unconfirmed. A site may display a success-looking screen while the underlying operation failed.\"},\"tunes\":{}},{\"id\":\"p-verify-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"OpenAI's current computer-use guidance explicitly recommends bounding and verifying the run instead of relying only on the model's final answer. Microsoft Research's work on computer-use verifiers reaches the same conclusion from evaluation: process and outcome need to be judged separately.\"},\"tunes\":{}},{\"id\":\"p-verify-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The Universal Verifier research reports that earlier verifier setups can produce high false-positive rates, while stronger rubric design and explicit separation of process, outcome, controllable failures, and uncontrollable failures substantially improve agreement with human labels.\"},\"tunes\":{}},{\"id\":\"h-safe-goal\",\"type\":\"header\",\"data\":{\"text\":\"Level 7 — Safe goal handling: the agent must know when not to continue\",\"level\":3},\"tunes\":{}},{\"id\":\"p-safe-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Computer-use agents are optimized to complete goals, but goal persistence can itself become a failure mode. An ambiguous request, impossible condition, contradictory instruction, suspicious webpage, or changed environment may require clarification or stopping rather than more action.\"},\"tunes\":{}},{\"id\":\"p-safe-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The BLIND-ACT benchmark studies this problem as Blind Goal-Directedness. Across the systems evaluated in that work, agents frequently continued pursuing tasks despite ambiguity, infeasibility, conflicting context, or other reasons to reconsider. The authors identify patterns such as execution-first bias and request primacy.\"},\"tunes\":{}},{\"id\":\"p-safe-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"This failure class matters because a highly capable agent can make a bad situation worse faster. Reliability therefore includes a policy for when not to act.\"},\"tunes\":{}},{\"id\":\"h-stress\",\"type\":\"header\",\"data\":{\"text\":\"The Demo-to-Production Stress Test\",\"level\":2},\"tunes\":{}},{\"id\":\"p-stress-intro\",\"type\":\"paragraph\",\"data\":{\"text\":\"Before deploying a computer-use workflow, take the successful demo and systematically remove the assumptions that made it easy.\"},\"tunes\":{}},{\"id\":\"stress-flow\",\"type\":\"processFlow\",\"data\":{\"title\":\"Demo-to-Production Stress Test\",\"orientation\":\"auto\",\"steps\":[{\"label\":\"1. Re-run the clean task\",\"description\":\"Establish repeatability over multiple trials before adding complexity.\"},{\"label\":\"2. Perturb the environment\",\"description\":\"Add latency, retries, pop-ups, page variation, stale sessions and temporary failures.\"},{\"label\":\"3. Extend the horizon\",\"description\":\"Turn the short demo into the full real workflow with intermediate state, multiple applications and delayed steps.\"},{\"label\":\"4. Change hidden state\",\"description\":\"Modify account, file, task or external state after the agent has formed a plan and test whether it detects the change.\"},{\"label\":\"5. Inject ambiguity\",\"description\":\"Remove one important assumption and test whether the agent asks instead of guessing.\"},{\"label\":\"6. Inject a controlled contradiction\",\"description\":\"Present old and new state together and verify that authoritative current state wins.\"},{\"label\":\"7. Require outcome proof\",\"description\":\"Make task completion depend on verifiable final state, not the model's self-report.\"},{\"label\":\"8. Test consequential boundaries\",\"description\":\"Confirm that irreversible or sensitive actions trigger the expected approval, refusal or handoff.\"},{\"label\":\"9. Repeat after harness or model changes\",\"description\":\"Treat runtime upgrades as reliability changes that need regression testing.\"}]},\"tunes\":{}},{\"id\":\"h-benchmark-boundary\",\"type\":\"header\",\"data\":{\"text\":\"Benchmark success has a validity boundary\",\"level\":2},\"tunes\":{}},{\"id\":\"p-boundary-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A benchmark score is a conditional statement. It is valid for a particular model, harness, environment, task set, judge, tool interface, step budget, retry policy, date and evaluation method.\"},\"tunes\":{}},{\"id\":\"p-boundary-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The number becomes misleading when those conditions disappear from the claim. “Agent X scores 80%” is weaker than “Agent X scored 80% on benchmark Y under environment Z with judge J and step budget N.” The second statement preserves the boundary that tells you whether the number transfers to your application.\"},\"tunes\":{}},{\"id\":\"ref-avb\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers\",\"title\":\"The Answer Validity Boundary: The Missing Layer Between Relevance and Reliable AI Answers\",\"excerpt\":\"A framework for making explicit the conditions under which an AI claim remains valid and what changes require restriction, recalculation, or abandonment.\",\"ctaLabel\":\"Read the Answer Validity Boundary\"},\"tunes\":{}},{\"id\":\"h-process-outcome\",\"type\":\"header\",\"data\":{\"text\":\"Process success and outcome success must be scored separately\",\"level\":2},\"tunes\":{}},{\"id\":\"process-outcome-comparison\",\"type\":\"comparison\",\"data\":{\"title\":\"Four possible outcomes of one computer-use run\",\"layout\":\"table\",\"columns\":[{\"id\":\"process\",\"label\":\"Process\"},{\"id\":\"outcome\",\"label\":\"Outcome\"},{\"id\":\"interpretation\",\"label\":\"Interpretation\"}],\"rows\":[{\"id\":\"good-good\",\"label\":\"Correct process \u002F correct outcome\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"bad-good\",\"label\":\"Wrong process \u002F correct outcome\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"good-bad\",\"label\":\"Correct process \u002F wrong outcome\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"bad-bad\",\"label\":\"Wrong process \u002F wrong outcome\",\"values\":[\"\",\"\",\"\"]}]},\"tunes\":{}},{\"id\":\"p-process-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"WeaveBench reports that outcome-only grading can materially overestimate computer-use performance because an agent may produce an apparently successful artifact through a shortcut or fabricated evidence. The verifier must inspect the trajectory and deliverables, not merely the final claim.\"},\"tunes\":{}},{\"id\":\"h-distribution\",\"type\":\"header\",\"data\":{\"text\":\"Production reliability is a distribution, not a single pass rate\",\"level\":2},\"tunes\":{}},{\"id\":\"p-dist-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A useful production evaluation samples the dimensions that actually vary in your environment. For a browser workflow, that might include account age, locale, viewport, page version, network quality, authentication state, existing cart state, cookies, pop-ups, user permissions and whether a human interrupts the run.\"},\"tunes\":{}},{\"id\":\"distribution-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Dimension\",\"Example variation\",\"Why it matters\"],[\"Environment\",\"Fast vs slow network, transient failures, page timing\",\"Tests recovery and waiting behaviour\"],[\"UI\",\"Different viewport, modal, reordered element, minor redesign\",\"Tests brittle visual\u002Faction assumptions\"],[\"State\",\"Logged in\u002Fout, empty\u002Fnon-empty cart, existing file, changed permissions\",\"Tests hidden-state reasoning\"],[\"Task horizon\",\"5 steps vs 50+ steps, one app vs several apps\",\"Tests accumulated trajectory error\"],[\"Ambiguity\",\"Missing preference or incomplete user instruction\",\"Tests whether the agent asks instead of guesses\"],[\"Consequence\",\"Read-only vs purchase\u002Fsend\u002Fdelete\u002Fchange\",\"Tests confirmation and authorization controls\"],[\"Adversarial content\",\"Prompt injection or misleading page text\",\"Tests instruction hierarchy and containment\"],[\"Model \u002F harness version\",\"Runtime upgrade\",\"Tests regression from system-level changes\"]]},\"tunes\":{}},{\"id\":\"h-budget\",\"type\":\"header\",\"data\":{\"text\":\"Reliability needs a failure budget, not perfection\",\"level\":2},\"tunes\":{}},{\"id\":\"p-budget-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"No production system is perfectly reliable. The useful engineering question is which failures are acceptable, detectable and recoverable. A failed attempt to sort a local folder is not equivalent to sending the wrong email, purchasing the wrong product or changing an account setting.\"},\"tunes\":{}},{\"id\":\"p-budget-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Classify actions by consequence and reversibility. Low-impact reversible actions can tolerate more autonomy. High-impact, externally visible or hard-to-reverse actions need stronger confirmation, state verification, authorization and post-action checks.\"},\"tunes\":{}},{\"id\":\"h-matrix\",\"type\":\"header\",\"data\":{\"text\":\"A practical computer-use reliability matrix\",\"level\":2},\"tunes\":{}},{\"id\":\"control-matrix\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Action class\",\"Example\",\"Recommended control\"],[\"Read \u002F inspect\",\"Open pages, read files, gather information\",\"Bound scope, log sources, tolerate recoverable navigation errors\"],[\"Reversible local change\",\"Edit draft file, reorganize temporary workspace\",\"Checkpoint or version before change; verify result\"],[\"External communication\",\"Send email, publish content, submit form\",\"User confirmation or explicit delegated authority; verify accepted state\"],[\"Financial \u002F transactional\",\"Purchase, checkout, paid subscription\",\"Strict mandate, amount\u002Fmerchant constraints, final confirmation and receipt verification\"],[\"Destructive \u002F privilege-changing\",\"Delete data, change permissions, revoke access\",\"Narrow authorization, explicit confirmation, reversible path where possible, post-action audit\"]]},\"tunes\":{}},{\"id\":\"h-log\",\"type\":\"header\",\"data\":{\"text\":\"What to log for a computer-use failure\",\"level\":2},\"tunes\":{}},{\"id\":\"log-list\",\"type\":\"list\",\"data\":{\"style\":\"unordered\",\"meta\":{},\"items\":[\"User goal and explicit constraints.\",\"Model and harness version.\",\"Environment and application versions.\",\"Screenshots or structured observations relevant to the failure.\",\"Actions taken with timestamps.\",\"Tool, click, keyboard and navigation results.\",\"State transitions and waiting periods.\",\"Approval, refusal or handoff events.\",\"External errors and network failures.\",\"Final observable environment state.\",\"The agent's reported outcome.\",\"Verifier result and whether the failure was controllable by the agent.\"]},\"tunes\":{}},{\"id\":\"p-log-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The crucial comparison is between reported success and observable success. A system that cannot distinguish those two will eventually accumulate false positives in production.\"},\"tunes\":{}},{\"id\":\"ref-reliability\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough\",\"title\":\"AI Agent Reliability: Why the Final Answer Is Not Enough\",\"excerpt\":\"A broader reliability model for evaluating agent trajectories, tool use and intermediate decisions instead of accepting the final answer as proof that the system worked correctly.\",\"ctaLabel\":\"Read the agent reliability article\"},\"tunes\":{}},{\"id\":\"h-security\",\"type\":\"header\",\"data\":{\"text\":\"Security is part of reliability for computer-use agents\",\"level\":2},\"tunes\":{}},{\"id\":\"p-sec-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Computer-use agents do not merely read untrusted content; they can act after reading it. That turns prompt injection, malicious page content and phishing into execution-path risks.\"},\"tunes\":{}},{\"id\":\"p-sec-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"OpenAI's current computer-use guidance recommends isolating the environment, allow-listing sites and actions, treating screen content as untrusted, confirming consequential actions, bounding the run and verifying the actual outcome. ChatGPT agent similarly uses confirmations, prompt-injection monitoring and supervised modes for sensitive contexts.\"},\"tunes\":{}},{\"id\":\"p-sec-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The architecture principle is broader than any one provider: content observed by the agent must not be allowed to redefine the user's authority. A webpage can provide data. It cannot grant permission to send data elsewhere, purchase something, change credentials or override the task boundary.\"},\"tunes\":{}},{\"id\":\"h-change\",\"type\":\"header\",\"data\":{\"text\":\"What would change this answer?\",\"level\":2},\"tunes\":{}},{\"id\":\"p-change-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The reliability gap would narrow if computer-use models became robust to long horizons, dynamic state, UI variation, environmental failures and ambiguous goals across representative production distributions. Better native state APIs, standardized machine-readable interfaces and stronger verifier infrastructure could also reduce the amount of fragile GUI interaction required.\"},\"tunes\":{}},{\"id\":\"p-change-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The deployment threshold also changes with task consequence. A 70% success rate can be useful for a supervised low-risk research task and unacceptable for an autonomous financial or destructive workflow. Reliability must therefore be evaluated against the cost of each failure class, not one universal pass-rate threshold.\"},\"tunes\":{}},{\"id\":\"h-limitations\",\"type\":\"header\",\"data\":{\"text\":\"Limitations\",\"level\":2},\"tunes\":{}},{\"id\":\"p-limit-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The cited benchmarks evaluate different environments and should not be ranked against one another as if they measured the same thing. WAREX stresses web unreliability; WeaveBench targets hybrid long-horizon work; OSWorld 2.0 targets realistic long workflows; BLIND-ACT focuses on goal handling under ambiguity and infeasibility.\"},\"tunes\":{}},{\"id\":\"p-limit-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Benchmark results also age quickly. Model, harness and verifier improvements can materially change scores within months. The durable lesson is therefore the evaluation method: vary conditions, separate process from outcome, verify external state, and preserve the boundary around each performance claim.\"},\"tunes\":{}},{\"id\":\"h-conclusion\",\"type\":\"header\",\"data\":{\"text\":\"Conclusion\",\"level\":2},\"tunes\":{}},{\"id\":\"p-conclusion-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Computer-use agents are already capable enough to be useful. That is exactly why the evaluation question has changed. The challenge is no longer only whether an agent can click through a workflow. It is whether the system remains dependable when the clean demo conditions disappear.\"},\"tunes\":{}},{\"id\":\"p-conclusion-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Treat one successful run as evidence of capability. Then test repeatability, environmental robustness, long-horizon control, state awareness, outcome verification and safe goal handling. A production computer-use agent is not the one that can complete the demo. It is the one whose failure boundaries are known, measured and controlled.\"},\"tunes\":{}},{\"id\":\"h-faq\",\"type\":\"header\",\"data\":{\"text\":\"FAQ\",\"level\":2},\"tunes\":{}},{\"id\":\"faq\",\"type\":\"faq\",\"data\":{\"title\":\"Computer-use agent reliability\",\"items\":[{\"id\":\"faq1\",\"question\":\"Does a successful computer-use agent demo prove production reliability?\",\"answer\":\"No. It proves capability under one observed trajectory. Production reliability requires repeated success across environmental variation, long-running tasks, changing state, ambiguity, recovery conditions and consequential actions.\"},{\"id\":\"faq2\",\"question\":\"Why can computer-use benchmarks look much better than real-world performance?\",\"answer\":\"Benchmarks can use more controlled environments, shorter tasks, stable network conditions, simpler application combinations or outcome criteria that do not capture all process failures. The exact validity boundary depends on each benchmark.\"},{\"id\":\"faq3\",\"question\":\"What is the most important reliability check after a computer-use action?\",\"answer\":\"Verify the actual external outcome. Do not treat the agent's final statement or intended click sequence as proof that the target system accepted the operation.\"},{\"id\":\"faq4\",\"question\":\"Why do long-horizon computer tasks remain difficult?\",\"answer\":\"Errors accumulate across many actions, constraints are forgotten, external state changes, work spans multiple applications, hidden state matters, and the agent must decide when to wait, ask, verify or recover rather than simply continue acting.\"},{\"id\":\"faq5\",\"question\":\"How should I test a browser or desktop agent before deployment?\",\"answer\":\"Repeat clean tasks, inject realistic environmental failures, vary UI and state, extend the workflow horizon, introduce ambiguity, require observable outcome proof, test high-impact action controls and rerun the suite after model or harness changes.\"},{\"id\":\"faq6\",\"question\":\"Should computer-use agents always require human confirmation?\",\"answer\":\"Not for every low-risk action. Confirmation requirements should scale with consequence, reversibility, authority and uncertainty. High-impact, externally visible or difficult-to-reverse actions need stronger controls.\"}]},\"tunes\":{}},{\"id\":\"h-glossary\",\"type\":\"header\",\"data\":{\"text\":\"Glossary\",\"level\":2},\"tunes\":{}},{\"id\":\"glossary\",\"type\":\"glossary\",\"data\":{\"title\":\"Key reliability terms\",\"entries\":[{\"term\":\"Computer-use agent\",\"definition\":\"An AI agent that interacts with graphical user interfaces or computer environments through observations and actions such as clicking, typing, scrolling, file operations or cross-application workflows.\",\"anchor\":\"computer-use-agent\"},{\"term\":\"Repeatability\",\"definition\":\"The degree to which an agent can complete the same task consistently across repeated runs rather than succeeding only on selected trajectories.\",\"anchor\":\"repeatability\"},{\"term\":\"Environmental robustness\",\"definition\":\"The ability to preserve correct behaviour despite realistic variation such as latency, transient errors, UI changes, session state and unexpected page conditions.\",\"anchor\":\"environmental-robustness\"},{\"term\":\"Outcome verification\",\"definition\":\"Checking the actual external state after an action to confirm that the intended result occurred instead of relying on the agent's self-report.\",\"anchor\":\"outcome-verification\"},{\"term\":\"Blind Goal-Directedness\",\"definition\":\"A failure pattern in which a computer-use agent continues pursuing a goal despite ambiguity, infeasibility, contradictory conditions or reasons to stop and reassess.\",\"anchor\":\"blind-goal-directedness\"},{\"term\":\"Reliability boundary\",\"definition\":\"The set of conditions under which an observed success rate or capability claim remains representative enough for a specific deployment decision.\",\"anchor\":\"reliability-boundary\"}]},\"tunes\":{}},{\"id\":\"h-sources\",\"type\":\"header\",\"data\":{\"text\":\"Primary sources and further reading\",\"level\":2},\"tunes\":{}},{\"id\":\"src-openai-computer\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Ftools-computer-use\",\"meta\":{\"title\":\"OpenAI — Computer use\",\"description\":\"Current developer guidance on isolating environments, treating screen content as untrusted, confirming consequential actions, bounding runs and verifying outcomes.\",\"image\":{\"url\":\"\"}}},\"tunes\":{}},{\"id\":\"src-openai-safety\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fopenai.com\u002Findex\u002Frunning-codex-safely\u002F\",\"meta\":{\"title\":\"OpenAI — Running Codex safely at OpenAI\",\"description\":\"Current production guidance on technical boundaries, human approval, telemetry and control for agents that act on real systems.\",\"image\":{\"url\":\"\"}}},\"tunes\":{}},{\"id\":\"src-ms-warex\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Fpublication\u002Fwarex-web-agent-reliability-evaluation-on-existing-benchmarks\u002F\",\"meta\":{\"title\":\"Microsoft Research — WAREX\",\"description\":\"2026 evaluation showing that realistic web unreliability causes significant drops in browser-agent task success on existing benchmarks.\",\"image\":{\"url\":\"\"}}},\"tunes\":{}},{\"id\":\"src-ms-verifier\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Farticles\u002Fthe-art-of-building-verifiers-for-computer-use-agents\u002F\",\"meta\":{\"title\":\"Microsoft Research — The Art of Building Verifiers for Computer Use Agents\",\"description\":\"2026 work on process versus outcome evaluation, controllable versus uncontrollable failures and reliable trajectory verification.\",\"image\":{\"url\":\"\"}}},\"tunes\":{}},{\"id\":\"src-ms-weavebench\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Fpublication\u002Fweavebench-a-long-horizon-real-world-benchmark-for-computer-use-agents-with-hybrid-interfaces\u002F\",\"meta\":{\"title\":\"Microsoft Research — WeaveBench\",\"description\":\"2026 long-horizon benchmark combining GUI, CLI and code workflows and showing a substantial gap between current agents and reliable real-world completion.\",\"image\":{\"url\":\"\"}}},\"tunes\":{}},{\"id\":\"src-osworld2\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2606.29537\",\"meta\":{\"title\":\"OSWorld 2.0 — Benchmarking Computer Use Agents on Long-Horizon Real-World Tasks\",\"description\":\"2026 benchmark focused on realistic long-horizon computer-use workflows, hidden state and cross-source reasoning.\",\"image\":{\"url\":\"\"}}},\"tunes\":{}},{\"id\":\"src-ms-sentinel\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Fpublication\u002Fsentinelbench-a-benchmark-for-long-running-monitoring-agents\u002F\",\"meta\":{\"title\":\"Microsoft Research — SentinelBench\",\"description\":\"2026 benchmark for time-evolving tasks where agents must monitor environments and respond to state changes rather than continuously act.\",\"image\":{\"url\":\"\"}}},\"tunes\":{}},{\"id\":\"src-ms-blind\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Fpublication\u002Fjust-do-it-computer-use-agents-exhibit-blind-goal-directedness\u002F\",\"meta\":{\"title\":\"Microsoft Research — Just Do It!? Computer-Use Agents Exhibit Blind Goal-Directedness\",\"description\":\"ICLR 2026 research on agents continuing to pursue ambiguous, contradictory or infeasible goals.\",\"image\":{\"url\":\"\"}}},\"tunes\":{}}],\"version\":\"2.31.6\"}",{"time":212,"blocks":213,"version":938},1790352872794,[214,222,228,236,243,250,255,260,265,270,275,305,310,315,344,349,354,359,364,369,374,379,384,389,396,401,406,411,416,421,426,431,436,441,446,451,456,461,466,471,476,481,486,519,524,529,534,543,548,582,587,592,597,638,643,648,653,658,687,692,712,717,725,730,735,740,745,750,755,760,765,770,775,780,785,790,795,825,830,860,865,875,884,893,902,911,920,929],{"id":215,"data":216,"type":220,"tunes":221},"IYG9UPcPY0",{"title":217,"maxLevel":218,"minLevel":219},"Contents",3,2,"tableOfContents",{},{"id":223,"data":224,"type":226,"tunes":227},"intro",{"text":225},"Computer-use agents can now click, type, browse, edit files, operate desktop applications, and complete impressive multi-step tasks. That makes successful demos easy to understand and easy to overinterpret. A single completed workflow shows that the agent can succeed under those conditions. It does not show how often it succeeds, how it behaves when the environment changes, whether it verifies the result, or how safely it acts when the goal becomes ambiguous.","paragraph",{},{"id":229,"data":230,"type":234,"tunes":235},"direct",{"body":231,"title":232,"variant":233},"\u003Cstrong>A successful computer-use demo proves capability, not reliability.\u003C\u002Fstrong> Production reliability requires the agent to succeed repeatedly across environmental variation, recover from transient failures, preserve constraints over long horizons, detect hidden or changing state, verify the actual outcome, and stop or ask when the goal becomes ambiguous or unsafe. The correct production question is not “Can the agent do this task?” but “Under which conditions can we trust it to do this task repeatedly?”","Direct answer","info","callout",{},{"id":237,"data":238,"type":234,"tunes":242},"freshness",{"body":239,"title":240,"variant":241},"This article reflects computer-use agent research and platform guidance available on \u003Cstrong>25 September 2026\u003C\u002Fstrong>. Benchmark results are not directly comparable across different task sets, environments, models, step limits, judges, or harnesses. Treat every benchmark number together with its evaluation conditions.","Fast-moving field","warning",{},{"id":244,"data":245,"type":234,"tunes":249},"model-note",{"body":246,"title":247,"variant":248},"The Computer-Use Reliability Ladder and Demo-to-Production Stress Test below are practical evaluation models proposed here. They are not formal industry standards.","The model used in this article","note",{},{"id":251,"data":252,"type":42,"tunes":254},"h-demo",{"text":253,"level":219},"Why the demo is the easiest possible reliability test",{},{"id":256,"data":257,"type":226,"tunes":259},"p-demo-1",{"text":258},"A demo normally shows one trajectory that worked. The environment is known, the task is selected in advance, the operator can restart after a failure, and the audience sees the successful path. Production systems face a distribution instead: different pages, network conditions, account states, pop-ups, latency, UI changes, hidden state, permissions, interruptions, and users who describe goals imperfectly.",{},{"id":261,"data":262,"type":226,"tunes":264},"p-demo-2",{"text":263},"That distinction matters because computer-use agents operate through interfaces designed for humans rather than deterministic APIs. Their action loop depends on perception, state interpretation, planning, interaction timing, and environment response. Small changes can alter the trajectory even when the user goal is unchanged.",{},{"id":266,"data":267,"type":226,"tunes":269},"p-demo-3",{"text":268},"Microsoft Research's WAREX work makes the problem explicit: benchmark agents that look capable in controlled settings lose substantial task success when realistic web instability is introduced. The failure is not necessarily “the model became less intelligent.” The environment stopped being deterministic.",{},{"id":271,"data":272,"type":42,"tunes":274},"h-claims",{"text":273,"level":219},"Capability, success rate, reliability, and safety are different claims",{},{"id":276,"data":277,"type":303,"tunes":304},"claims-table",{"content":278,"stretched":43,"withHeadings":14},[279,283,287,291,295,299],[280,281,282],"Claim","What it actually establishes","What it does not establish",[284,285,286],"The agent completed the task once","Capability under one observed trajectory","Repeatability, robustness, safety, or generalization",[288,289,290],"The agent scores highly on a benchmark","Performance under that benchmark's task and evaluation conditions","Equivalent production performance on different environments",[292,293,294],"The agent usually reaches the goal","Outcome success frequency","Correct process, safe behaviour, or evidence that the result was verified",[296,297,298],"The agent follows the intended process","Trajectory quality under the evaluated rubric","That the external environment actually accepted the final outcome",[300,301,302],"The agent avoids unsafe actions in a test set","Performance on represented safety cases","Safety under every novel ambiguity, injection, or side effect","table",{},{"id":306,"data":307,"type":42,"tunes":309},"h-ladder",{"text":308,"level":219},"The Computer-Use Reliability Ladder",{},{"id":311,"data":312,"type":226,"tunes":314},"p-ladder-intro",{"text":313},"A useful way to evaluate computer-use systems is to move from one-off capability toward progressively harder reliability properties. Higher levels assume the lower levels but do not follow automatically from them.",{},{"id":316,"data":317,"type":342,"tunes":343},"ladder-flow",{"steps":318,"title":340,"orientation":341},[319,322,325,328,331,334,337],{"label":320,"description":321},"1. Capability","Can the agent complete the task at least once under known conditions?",{"label":323,"description":324},"2. Repeatability","Can it complete the same task consistently across repeated trials?",{"label":326,"description":327},"3. Environmental robustness","Does it survive timing changes, network issues, pop-ups, UI variation, and small environmental perturbations?",{"label":329,"description":330},"4. Long-horizon control","Can it preserve goals, constraints, and progress across many steps, applications, and delayed events?",{"label":332,"description":333},"5. State awareness","Can it detect when the environment changed, when hidden state matters, or when an assumption is no longer valid?",{"label":335,"description":336},"6. Outcome verification","Does it verify that the intended result actually happened instead of trusting its own action sequence?",{"label":338,"description":339},"7. Safe goal handling","Can it stop, ask, refuse, or hand control back when the goal is ambiguous, infeasible, contradictory, or high impact?","Computer-Use Reliability Ladder","auto","processFlow",{},{"id":345,"data":346,"type":42,"tunes":348},"h-capability",{"text":347,"level":218},"Level 1 — Capability: the demo question",{},{"id":350,"data":351,"type":226,"tunes":353},"p-capability-1",{"text":352},"Capability asks whether an agent can perform the task at all. This is valuable. Computer-use systems have advanced rapidly, and modern agents can complete workflows that older systems could not execute reliably.",{},{"id":355,"data":356,"type":226,"tunes":358},"p-capability-2",{"text":357},"But capability is a weak deployment criterion. One successful run does not tell you whether the agent succeeds 95% of the time or 30% of the time, whether failures are harmless or destructive, or whether success depends on a lucky page state.",{},{"id":360,"data":361,"type":42,"tunes":363},"h-repeatability",{"text":362,"level":218},"Level 2 — Repeatability: does the same task stay solved?",{},{"id":365,"data":366,"type":226,"tunes":368},"p-repeat-1",{"text":367},"Computer-use trajectories are stochastic. Model outputs vary, pages load at different speeds, visual states change, and long workflows create many branching opportunities. A production test should therefore run the same task multiple times rather than treating one passing trace as representative.",{},{"id":370,"data":371,"type":226,"tunes":373},"p-repeat-2",{"text":372},"Measure not only the average success rate but also the distribution of failure modes: wrong click, premature termination, missed confirmation, incorrect field, duplicate action, navigation loop, stale-state assumption, and false success report.",{},{"id":375,"data":376,"type":42,"tunes":378},"h-robustness",{"text":377,"level":218},"Level 3 — Environmental robustness: what happens when the web behaves like the web?",{},{"id":380,"data":381,"type":226,"tunes":383},"p-robust-1",{"text":382},"Real websites are not benchmark fixtures. Requests fail, elements load late, sessions expire, pages change, consent banners appear, servers return errors, and network conditions fluctuate.",{},{"id":385,"data":386,"type":226,"tunes":388},"p-robust-2",{"text":387},"WAREX evaluates this gap by injecting realistic web unreliability into existing benchmark environments and reports significant drops in task success. This is a critical production insight: a benchmark can measure task competence while under-measuring recovery from environmental instability.",{},{"id":390,"data":391,"type":234,"tunes":395},"robust-tip",{"body":392,"title":393,"variant":394},"Inject delays, transient HTTP failures, stale page state, modal dialogs, session expiration, duplicate responses, and controlled UI variation. If the agent only works on the clean path, it is a demo-capable system, not a production-reliable one.","Reliability test","tip",{},{"id":397,"data":398,"type":42,"tunes":400},"h-long",{"text":399,"level":218},"Level 4 — Long-horizon control: success changes when the task becomes real work",{},{"id":402,"data":403,"type":226,"tunes":405},"p-long-1",{"text":404},"Short tasks hide a class of failures that appear only after dozens or hundreds of actions: forgotten constraints, duplicated work, premature completion, missed state changes, cross-application inconsistencies, and accumulated small errors.",{},{"id":407,"data":408,"type":226,"tunes":410},"p-long-2",{"text":409},"OSWorld 2.0 was designed specifically around long-horizon real-world workflows. Its tasks take human users a median of roughly 1.6 hours and require many more tool calls than earlier computer-use benchmarks. Under its primary completion metric, even the strongest evaluated systems remain far from complete task reliability.",{},{"id":412,"data":413,"type":226,"tunes":415},"p-long-3",{"text":414},"WeaveBench reaches a similar conclusion from another angle. It evaluates hybrid GUI, CLI and code workflows and reports that the best evaluated model-runtime pairing passes only 41.2% of tasks. The important result is not one leaderboard number; it is that realistic cross-interface orchestration exposes failures hidden by simpler single-interface tasks.",{},{"id":417,"data":418,"type":42,"tunes":420},"h-state",{"text":419,"level":218},"Level 5 — State awareness: the environment can change underneath the plan",{},{"id":422,"data":423,"type":226,"tunes":425},"p-state-1",{"text":424},"Long-running tasks often depend on hidden or changing state: an email arrives, a calendar changes, a form is submitted, a background process finishes, a browser session expires, a user modifies a file, or an external system changes availability.",{},{"id":427,"data":428,"type":226,"tunes":430},"p-state-2",{"text":429},"Microsoft's SentinelBench argues that many long-running tasks should not be solved through continuous action at all. The correct behaviour may be to monitor, wait for an external event, then act when the state changes. This is a different capability from clicking faster or planning more steps.",{},{"id":432,"data":433,"type":226,"tunes":435},"p-state-3",{"text":434},"A reliable computer-use agent therefore needs to distinguish actionable now, waiting for state, state changed, and assumption invalidated.",{},{"id":437,"data":438,"type":42,"tunes":440},"h-verify",{"text":439,"level":218},"Level 6 — Outcome verification: did the action actually work?",{},{"id":442,"data":443,"type":226,"tunes":445},"p-verify-1",{"text":444},"An agent can execute an apparently correct sequence and still fail the task. A button click may not register. A form may reject hidden validation. A file may save to the wrong directory. A purchase may remain unconfirmed. A site may display a success-looking screen while the underlying operation failed.",{},{"id":447,"data":448,"type":226,"tunes":450},"p-verify-2",{"text":449},"OpenAI's current computer-use guidance explicitly recommends bounding and verifying the run instead of relying only on the model's final answer. Microsoft Research's work on computer-use verifiers reaches the same conclusion from evaluation: process and outcome need to be judged separately.",{},{"id":452,"data":453,"type":226,"tunes":455},"p-verify-3",{"text":454},"The Universal Verifier research reports that earlier verifier setups can produce high false-positive rates, while stronger rubric design and explicit separation of process, outcome, controllable failures, and uncontrollable failures substantially improve agreement with human labels.",{},{"id":457,"data":458,"type":42,"tunes":460},"h-safe-goal",{"text":459,"level":218},"Level 7 — Safe goal handling: the agent must know when not to continue",{},{"id":462,"data":463,"type":226,"tunes":465},"p-safe-1",{"text":464},"Computer-use agents are optimized to complete goals, but goal persistence can itself become a failure mode. An ambiguous request, impossible condition, contradictory instruction, suspicious webpage, or changed environment may require clarification or stopping rather than more action.",{},{"id":467,"data":468,"type":226,"tunes":470},"p-safe-2",{"text":469},"The BLIND-ACT benchmark studies this problem as Blind Goal-Directedness. Across the systems evaluated in that work, agents frequently continued pursuing tasks despite ambiguity, infeasibility, conflicting context, or other reasons to reconsider. The authors identify patterns such as execution-first bias and request primacy.",{},{"id":472,"data":473,"type":226,"tunes":475},"p-safe-3",{"text":474},"This failure class matters because a highly capable agent can make a bad situation worse faster. Reliability therefore includes a policy for when not to act.",{},{"id":477,"data":478,"type":42,"tunes":480},"h-stress",{"text":479,"level":219},"The Demo-to-Production Stress Test",{},{"id":482,"data":483,"type":226,"tunes":485},"p-stress-intro",{"text":484},"Before deploying a computer-use workflow, take the successful demo and systematically remove the assumptions that made it easy.",{},{"id":487,"data":488,"type":342,"tunes":518},"stress-flow",{"steps":489,"title":517,"orientation":341},[490,493,496,499,502,505,508,511,514],{"label":491,"description":492},"1. Re-run the clean task","Establish repeatability over multiple trials before adding complexity.",{"label":494,"description":495},"2. Perturb the environment","Add latency, retries, pop-ups, page variation, stale sessions and temporary failures.",{"label":497,"description":498},"3. Extend the horizon","Turn the short demo into the full real workflow with intermediate state, multiple applications and delayed steps.",{"label":500,"description":501},"4. Change hidden state","Modify account, file, task or external state after the agent has formed a plan and test whether it detects the change.",{"label":503,"description":504},"5. Inject ambiguity","Remove one important assumption and test whether the agent asks instead of guessing.",{"label":506,"description":507},"6. Inject a controlled contradiction","Present old and new state together and verify that authoritative current state wins.",{"label":509,"description":510},"7. Require outcome proof","Make task completion depend on verifiable final state, not the model's self-report.",{"label":512,"description":513},"8. Test consequential boundaries","Confirm that irreversible or sensitive actions trigger the expected approval, refusal or handoff.",{"label":515,"description":516},"9. Repeat after harness or model changes","Treat runtime upgrades as reliability changes that need regression testing.","Demo-to-Production Stress Test",{},{"id":520,"data":521,"type":42,"tunes":523},"h-benchmark-boundary",{"text":522,"level":219},"Benchmark success has a validity boundary",{},{"id":525,"data":526,"type":226,"tunes":528},"p-boundary-1",{"text":527},"A benchmark score is a conditional statement. It is valid for a particular model, harness, environment, task set, judge, tool interface, step budget, retry policy, date and evaluation method.",{},{"id":530,"data":531,"type":226,"tunes":533},"p-boundary-2",{"text":532},"The number becomes misleading when those conditions disappear from the claim. “Agent X scores 80%” is weaker than “Agent X scored 80% on benchmark Y under environment Z with judge J and step budget N.” The second statement preserves the boundary that tells you whether the number transfers to your application.",{},{"id":535,"data":536,"type":541,"tunes":542},"ref-avb",{"url":537,"title":538,"excerpt":539,"ctaLabel":540},"https:\u002F\u002Fstajic.de\u002Fblog\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers","The Answer Validity Boundary: The Missing Layer Between Relevance and Reliable AI Answers","A framework for making explicit the conditions under which an AI claim remains valid and what changes require restriction, recalculation, or abandonment.","Read the Answer Validity Boundary","referralArticle",{},{"id":544,"data":545,"type":42,"tunes":547},"h-process-outcome",{"text":546,"level":219},"Process success and outcome success must be scored separately",{},{"id":549,"data":550,"type":580,"tunes":581},"process-outcome-comparison",{"rows":551,"title":569,"layout":303,"columns":570},[552,557,561,565],{"id":553,"label":554,"values":555},"good-good","Correct process \u002F correct outcome",[556,556,556],"",{"id":558,"label":559,"values":560},"bad-good","Wrong process \u002F correct outcome",[556,556,556],{"id":562,"label":563,"values":564},"good-bad","Correct process \u002F wrong outcome",[556,556,556],{"id":566,"label":567,"values":568},"bad-bad","Wrong process \u002F wrong outcome",[556,556,556],"Four possible outcomes of one computer-use run",[571,574,577],{"id":572,"label":573},"process","Process",{"id":575,"label":576},"outcome","Outcome",{"id":578,"label":579},"interpretation","Interpretation","comparison",{},{"id":583,"data":584,"type":226,"tunes":586},"p-process-1",{"text":585},"WeaveBench reports that outcome-only grading can materially overestimate computer-use performance because an agent may produce an apparently successful artifact through a shortcut or fabricated evidence. The verifier must inspect the trajectory and deliverables, not merely the final claim.",{},{"id":588,"data":589,"type":42,"tunes":591},"h-distribution",{"text":590,"level":219},"Production reliability is a distribution, not a single pass rate",{},{"id":593,"data":594,"type":226,"tunes":596},"p-dist-1",{"text":595},"A useful production evaluation samples the dimensions that actually vary in your environment. For a browser workflow, that might include account age, locale, viewport, page version, network quality, authentication state, existing cart state, cookies, pop-ups, user permissions and whether a human interrupts the run.",{},{"id":598,"data":599,"type":303,"tunes":637},"distribution-table",{"content":600,"stretched":43,"withHeadings":14},[601,605,609,613,617,621,625,629,633],[602,603,604],"Dimension","Example variation","Why it matters",[606,607,608],"Environment","Fast vs slow network, transient failures, page timing","Tests recovery and waiting behaviour",[610,611,612],"UI","Different viewport, modal, reordered element, minor redesign","Tests brittle visual\u002Faction assumptions",[614,615,616],"State","Logged in\u002Fout, empty\u002Fnon-empty cart, existing file, changed permissions","Tests hidden-state reasoning",[618,619,620],"Task horizon","5 steps vs 50+ steps, one app vs several apps","Tests accumulated trajectory error",[622,623,624],"Ambiguity","Missing preference or incomplete user instruction","Tests whether the agent asks instead of guesses",[626,627,628],"Consequence","Read-only vs purchase\u002Fsend\u002Fdelete\u002Fchange","Tests confirmation and authorization controls",[630,631,632],"Adversarial content","Prompt injection or misleading page text","Tests instruction hierarchy and containment",[634,635,636],"Model \u002F harness version","Runtime upgrade","Tests regression from system-level changes",{},{"id":639,"data":640,"type":42,"tunes":642},"h-budget",{"text":641,"level":219},"Reliability needs a failure budget, not perfection",{},{"id":644,"data":645,"type":226,"tunes":647},"p-budget-1",{"text":646},"No production system is perfectly reliable. The useful engineering question is which failures are acceptable, detectable and recoverable. A failed attempt to sort a local folder is not equivalent to sending the wrong email, purchasing the wrong product or changing an account setting.",{},{"id":649,"data":650,"type":226,"tunes":652},"p-budget-2",{"text":651},"Classify actions by consequence and reversibility. Low-impact reversible actions can tolerate more autonomy. High-impact, externally visible or hard-to-reverse actions need stronger confirmation, state verification, authorization and post-action checks.",{},{"id":654,"data":655,"type":42,"tunes":657},"h-matrix",{"text":656,"level":219},"A practical computer-use reliability matrix",{},{"id":659,"data":660,"type":303,"tunes":686},"control-matrix",{"content":661,"stretched":43,"withHeadings":14},[662,666,670,674,678,682],[663,664,665],"Action class","Example","Recommended control",[667,668,669],"Read \u002F inspect","Open pages, read files, gather information","Bound scope, log sources, tolerate recoverable navigation errors",[671,672,673],"Reversible local change","Edit draft file, reorganize temporary workspace","Checkpoint or version before change; verify result",[675,676,677],"External communication","Send email, publish content, submit form","User confirmation or explicit delegated authority; verify accepted state",[679,680,681],"Financial \u002F transactional","Purchase, checkout, paid subscription","Strict mandate, amount\u002Fmerchant constraints, final confirmation and receipt verification",[683,684,685],"Destructive \u002F privilege-changing","Delete data, change permissions, revoke access","Narrow authorization, explicit confirmation, reversible path where possible, post-action audit",{},{"id":688,"data":689,"type":42,"tunes":691},"h-log",{"text":690,"level":219},"What to log for a computer-use failure",{},{"id":693,"data":694,"type":710,"tunes":711},"log-list",{"meta":695,"items":696,"style":709},{},[697,698,699,700,701,702,703,704,705,706,707,708],"User goal and explicit constraints.","Model and harness version.","Environment and application versions.","Screenshots or structured observations relevant to the failure.","Actions taken with timestamps.","Tool, click, keyboard and navigation results.","State transitions and waiting periods.","Approval, refusal or handoff events.","External errors and network failures.","Final observable environment state.","The agent's reported outcome.","Verifier result and whether the failure was controllable by the agent.","unordered","list",{},{"id":713,"data":714,"type":226,"tunes":716},"p-log-1",{"text":715},"The crucial comparison is between reported success and observable success. A system that cannot distinguish those two will eventually accumulate false positives in production.",{},{"id":718,"data":719,"type":541,"tunes":724},"ref-reliability",{"url":720,"title":721,"excerpt":722,"ctaLabel":723},"https:\u002F\u002Fstajic.de\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough","AI Agent Reliability: Why the Final Answer Is Not Enough","A broader reliability model for evaluating agent trajectories, tool use and intermediate decisions instead of accepting the final answer as proof that the system worked correctly.","Read the agent reliability article",{},{"id":726,"data":727,"type":42,"tunes":729},"h-security",{"text":728,"level":219},"Security is part of reliability for computer-use agents",{},{"id":731,"data":732,"type":226,"tunes":734},"p-sec-1",{"text":733},"Computer-use agents do not merely read untrusted content; they can act after reading it. That turns prompt injection, malicious page content and phishing into execution-path risks.",{},{"id":736,"data":737,"type":226,"tunes":739},"p-sec-2",{"text":738},"OpenAI's current computer-use guidance recommends isolating the environment, allow-listing sites and actions, treating screen content as untrusted, confirming consequential actions, bounding the run and verifying the actual outcome. ChatGPT agent similarly uses confirmations, prompt-injection monitoring and supervised modes for sensitive contexts.",{},{"id":741,"data":742,"type":226,"tunes":744},"p-sec-3",{"text":743},"The architecture principle is broader than any one provider: content observed by the agent must not be allowed to redefine the user's authority. A webpage can provide data. It cannot grant permission to send data elsewhere, purchase something, change credentials or override the task boundary.",{},{"id":746,"data":747,"type":42,"tunes":749},"h-change",{"text":748,"level":219},"What would change this answer?",{},{"id":751,"data":752,"type":226,"tunes":754},"p-change-1",{"text":753},"The reliability gap would narrow if computer-use models became robust to long horizons, dynamic state, UI variation, environmental failures and ambiguous goals across representative production distributions. Better native state APIs, standardized machine-readable interfaces and stronger verifier infrastructure could also reduce the amount of fragile GUI interaction required.",{},{"id":756,"data":757,"type":226,"tunes":759},"p-change-2",{"text":758},"The deployment threshold also changes with task consequence. A 70% success rate can be useful for a supervised low-risk research task and unacceptable for an autonomous financial or destructive workflow. Reliability must therefore be evaluated against the cost of each failure class, not one universal pass-rate threshold.",{},{"id":761,"data":762,"type":42,"tunes":764},"h-limitations",{"text":763,"level":219},"Limitations",{},{"id":766,"data":767,"type":226,"tunes":769},"p-limit-1",{"text":768},"The cited benchmarks evaluate different environments and should not be ranked against one another as if they measured the same thing. WAREX stresses web unreliability; WeaveBench targets hybrid long-horizon work; OSWorld 2.0 targets realistic long workflows; BLIND-ACT focuses on goal handling under ambiguity and infeasibility.",{},{"id":771,"data":772,"type":226,"tunes":774},"p-limit-2",{"text":773},"Benchmark results also age quickly. Model, harness and verifier improvements can materially change scores within months. The durable lesson is therefore the evaluation method: vary conditions, separate process from outcome, verify external state, and preserve the boundary around each performance claim.",{},{"id":776,"data":777,"type":42,"tunes":779},"h-conclusion",{"text":778,"level":219},"Conclusion",{},{"id":781,"data":782,"type":226,"tunes":784},"p-conclusion-1",{"text":783},"Computer-use agents are already capable enough to be useful. That is exactly why the evaluation question has changed. The challenge is no longer only whether an agent can click through a workflow. It is whether the system remains dependable when the clean demo conditions disappear.",{},{"id":786,"data":787,"type":226,"tunes":789},"p-conclusion-2",{"text":788},"Treat one successful run as evidence of capability. Then test repeatability, environmental robustness, long-horizon control, state awareness, outcome verification and safe goal handling. A production computer-use agent is not the one that can complete the demo. It is the one whose failure boundaries are known, measured and controlled.",{},{"id":791,"data":792,"type":42,"tunes":794},"h-faq",{"text":793,"level":219},"FAQ",{},{"id":796,"data":797,"type":796,"tunes":824},"faq",{"items":798,"title":823},[799,803,807,811,815,819],{"id":800,"answer":801,"question":802},"faq1","No. It proves capability under one observed trajectory. Production reliability requires repeated success across environmental variation, long-running tasks, changing state, ambiguity, recovery conditions and consequential actions.","Does a successful computer-use agent demo prove production reliability?",{"id":804,"answer":805,"question":806},"faq2","Benchmarks can use more controlled environments, shorter tasks, stable network conditions, simpler application combinations or outcome criteria that do not capture all process failures. The exact validity boundary depends on each benchmark.","Why can computer-use benchmarks look much better than real-world performance?",{"id":808,"answer":809,"question":810},"faq3","Verify the actual external outcome. Do not treat the agent's final statement or intended click sequence as proof that the target system accepted the operation.","What is the most important reliability check after a computer-use action?",{"id":812,"answer":813,"question":814},"faq4","Errors accumulate across many actions, constraints are forgotten, external state changes, work spans multiple applications, hidden state matters, and the agent must decide when to wait, ask, verify or recover rather than simply continue acting.","Why do long-horizon computer tasks remain difficult?",{"id":816,"answer":817,"question":818},"faq5","Repeat clean tasks, inject realistic environmental failures, vary UI and state, extend the workflow horizon, introduce ambiguity, require observable outcome proof, test high-impact action controls and rerun the suite after model or harness changes.","How should I test a browser or desktop agent before deployment?",{"id":820,"answer":821,"question":822},"faq6","Not for every low-risk action. Confirmation requirements should scale with consequence, reversibility, authority and uncertainty. High-impact, externally visible or difficult-to-reverse actions need stronger controls.","Should computer-use agents always require human confirmation?","Computer-use agent reliability",{},{"id":826,"data":827,"type":42,"tunes":829},"h-glossary",{"text":828,"level":219},"Glossary",{},{"id":831,"data":832,"type":831,"tunes":859},"glossary",{"title":833,"entries":834},"Key reliability terms",[835,839,843,847,851,855],{"term":836,"anchor":837,"definition":838},"Computer-use agent","computer-use-agent","An AI agent that interacts with graphical user interfaces or computer environments through observations and actions such as clicking, typing, scrolling, file operations or cross-application workflows.",{"term":840,"anchor":841,"definition":842},"Repeatability","repeatability","The degree to which an agent can complete the same task consistently across repeated runs rather than succeeding only on selected trajectories.",{"term":844,"anchor":845,"definition":846},"Environmental robustness","environmental-robustness","The ability to preserve correct behaviour despite realistic variation such as latency, transient errors, UI changes, session state and unexpected page conditions.",{"term":848,"anchor":849,"definition":850},"Outcome verification","outcome-verification","Checking the actual external state after an action to confirm that the intended result occurred instead of relying on the agent's self-report.",{"term":852,"anchor":853,"definition":854},"Blind Goal-Directedness","blind-goal-directedness","A failure pattern in which a computer-use agent continues pursuing a goal despite ambiguity, infeasibility, contradictory conditions or reasons to stop and reassess.",{"term":856,"anchor":857,"definition":858},"Reliability boundary","reliability-boundary","The set of conditions under which an observed success rate or capability claim remains representative enough for a specific deployment decision.",{},{"id":861,"data":862,"type":42,"tunes":864},"h-sources",{"text":863,"level":219},"Primary sources and further reading",{},{"id":866,"data":867,"type":873,"tunes":874},"src-openai-computer",{"link":868,"meta":869},"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Ftools-computer-use",{"image":870,"title":871,"description":872},{"url":556},"OpenAI — Computer use","Current developer guidance on isolating environments, treating screen content as untrusted, confirming consequential actions, bounding runs and verifying outcomes.","linkTool",{},{"id":876,"data":877,"type":873,"tunes":883},"src-openai-safety",{"link":878,"meta":879},"https:\u002F\u002Fopenai.com\u002Findex\u002Frunning-codex-safely\u002F",{"image":880,"title":881,"description":882},{"url":556},"OpenAI — Running Codex safely at OpenAI","Current production guidance on technical boundaries, human approval, telemetry and control for agents that act on real systems.",{},{"id":885,"data":886,"type":873,"tunes":892},"src-ms-warex",{"link":887,"meta":888},"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Fpublication\u002Fwarex-web-agent-reliability-evaluation-on-existing-benchmarks\u002F",{"image":889,"title":890,"description":891},{"url":556},"Microsoft Research — WAREX","2026 evaluation showing that realistic web unreliability causes significant drops in browser-agent task success on existing benchmarks.",{},{"id":894,"data":895,"type":873,"tunes":901},"src-ms-verifier",{"link":896,"meta":897},"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Farticles\u002Fthe-art-of-building-verifiers-for-computer-use-agents\u002F",{"image":898,"title":899,"description":900},{"url":556},"Microsoft Research — The Art of Building Verifiers for Computer Use Agents","2026 work on process versus outcome evaluation, controllable versus uncontrollable failures and reliable trajectory verification.",{},{"id":903,"data":904,"type":873,"tunes":910},"src-ms-weavebench",{"link":905,"meta":906},"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Fpublication\u002Fweavebench-a-long-horizon-real-world-benchmark-for-computer-use-agents-with-hybrid-interfaces\u002F",{"image":907,"title":908,"description":909},{"url":556},"Microsoft Research — WeaveBench","2026 long-horizon benchmark combining GUI, CLI and code workflows and showing a substantial gap between current agents and reliable real-world completion.",{},{"id":912,"data":913,"type":873,"tunes":919},"src-osworld2",{"link":914,"meta":915},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2606.29537",{"image":916,"title":917,"description":918},{"url":556},"OSWorld 2.0 — Benchmarking Computer Use Agents on Long-Horizon Real-World Tasks","2026 benchmark focused on realistic long-horizon computer-use workflows, hidden state and cross-source reasoning.",{},{"id":921,"data":922,"type":873,"tunes":928},"src-ms-sentinel",{"link":923,"meta":924},"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Fpublication\u002Fsentinelbench-a-benchmark-for-long-running-monitoring-agents\u002F",{"image":925,"title":926,"description":927},{"url":556},"Microsoft Research — SentinelBench","2026 benchmark for time-evolving tasks where agents must monitor environments and respond to state changes rather than continuously act.",{},{"id":930,"data":931,"type":873,"tunes":937},"src-ms-blind",{"link":932,"meta":933},"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Fpublication\u002Fjust-do-it-computer-use-agents-exhibit-blind-goal-directedness\u002F",{"image":934,"title":935,"description":936},{"url":556},"Microsoft Research — Just Do It!? Computer-Use Agents Exhibit Blind Goal-Directedness","ICLR 2026 research on agents continuing to pursue ambiguous, contradictory or infeasible goals.",{},"2.31.6","Computer-use agents can now complete impressive browser and desktop workflows, but one successful run proves capability—not reliability. This article shows how to test repeatability, environmental robustness, long-horizon control, state awareness, outcome verification, and safe goal handling.","\u002Fuploads\u002F2026\u002F09\u002Fcomputer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system-1790352854690-75qnrg.webp","computer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system-1790352854690-75qnrg","PUBLISHED","2026-09-25T12:13:00.000Z","2026-09-25T16:13:28.344Z","2026-09-25T19:19:11.089Z",{"en":947,"de":948,"sr":949,"es":950,"fr":951,"it":952,"ru":953,"zh":954},"\u002Fblog\u002Fcomputer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system","\u002Fde\u002Fblog\u002Fcomputer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system","\u002Fsr\u002Fblog\u002Fcomputer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system","\u002Fes\u002Fblog\u002Fcomputer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system","\u002Ffr\u002Fblog\u002Fcomputer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system","\u002Fit\u002Fblog\u002Fcomputer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system","\u002Fru\u002Fblog\u002Fcomputer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system","\u002Fzh\u002Fblog\u002Fcomputer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system",[956,960,964],{"id":957,"name":958,"slug":959},58,"Evaluation & Quality Gates","evaluation",{"id":961,"name":962,"slug":963},97,"Verification on Test Set","verification",{"id":965,"name":966,"slug":967},73,"Verification & Diffing","verification-and-diffing",{"id":969,"login":970,"email":971,"displayName":972},"20","rooth8233","aleksandar@stajic.de","Aleksandar Stajić",[974],{"lang":7,"title":208,"content":210,"contentJson":975,"excerpt":939},{"time":212,"blocks":976,"version":938},[977,980,983,986,989,992,995,998,1001,1004,1007,1017,1020,1023,1034,1037,1040,1043,1046,1049,1052,1055,1058,1061,1064,1067,1070,1073,1076,1079,1082,1085,1088,1091,1094,1097,1100,1103,1106,1109,1112,1115,1118,1131,1134,1137,1140,1143,1146,1162,1165,1168,1171,1184,1187,1190,1193,1196,1206,1209,1214,1217,1220,1223,1226,1229,1232,1235,1238,1241,1244,1247,1250,1253,1256,1259,1262,1272,1275,1285,1288,1293,1298,1303,1308,1313,1318,1323],{"id":215,"data":978,"type":220,"tunes":979},{"title":217,"maxLevel":218,"minLevel":219},{},{"id":223,"data":981,"type":226,"tunes":982},{"text":225},{},{"id":229,"data":984,"type":234,"tunes":985},{"body":231,"title":232,"variant":233},{},{"id":237,"data":987,"type":234,"tunes":988},{"body":239,"title":240,"variant":241},{},{"id":244,"data":990,"type":234,"tunes":991},{"body":246,"title":247,"variant":248},{},{"id":251,"data":993,"type":42,"tunes":994},{"text":253,"level":219},{},{"id":256,"data":996,"type":226,"tunes":997},{"text":258},{},{"id":261,"data":999,"type":226,"tunes":1000},{"text":263},{},{"id":266,"data":1002,"type":226,"tunes":1003},{"text":268},{},{"id":271,"data":1005,"type":42,"tunes":1006},{"text":273,"level":219},{},{"id":276,"data":1008,"type":303,"tunes":1016},{"content":1009,"stretched":43,"withHeadings":14},[1010,1011,1012,1013,1014,1015],[280,281,282],[284,285,286],[288,289,290],[292,293,294],[296,297,298],[300,301,302],{},{"id":306,"data":1018,"type":42,"tunes":1019},{"text":308,"level":219},{},{"id":311,"data":1021,"type":226,"tunes":1022},{"text":313},{},{"id":316,"data":1024,"type":342,"tunes":1033},{"steps":1025,"title":340,"orientation":341},[1026,1027,1028,1029,1030,1031,1032],{"label":320,"description":321},{"label":323,"description":324},{"label":326,"description":327},{"label":329,"description":330},{"label":332,"description":333},{"label":335,"description":336},{"label":338,"description":339},{},{"id":345,"data":1035,"type":42,"tunes":1036},{"text":347,"level":218},{},{"id":350,"data":1038,"type":226,"tunes":1039},{"text":352},{},{"id":355,"data":1041,"type":226,"tunes":1042},{"text":357},{},{"id":360,"data":1044,"type":42,"tunes":1045},{"text":362,"level":218},{},{"id":365,"data":1047,"type":226,"tunes":1048},{"text":367},{},{"id":370,"data":1050,"type":226,"tunes":1051},{"text":372},{},{"id":375,"data":1053,"type":42,"tunes":1054},{"text":377,"level":218},{},{"id":380,"data":1056,"type":226,"tunes":1057},{"text":382},{},{"id":385,"data":1059,"type":226,"tunes":1060},{"text":387},{},{"id":390,"data":1062,"type":234,"tunes":1063},{"body":392,"title":393,"variant":394},{},{"id":397,"data":1065,"type":42,"tunes":1066},{"text":399,"level":218},{},{"id":402,"data":1068,"type":226,"tunes":1069},{"text":404},{},{"id":407,"data":1071,"type":226,"tunes":1072},{"text":409},{},{"id":412,"data":1074,"type":226,"tunes":1075},{"text":414},{},{"id":417,"data":1077,"type":42,"tunes":1078},{"text":419,"level":218},{},{"id":422,"data":1080,"type":226,"tunes":1081},{"text":424},{},{"id":427,"data":1083,"type":226,"tunes":1084},{"text":429},{},{"id":432,"data":1086,"type":226,"tunes":1087},{"text":434},{},{"id":437,"data":1089,"type":42,"tunes":1090},{"text":439,"level":218},{},{"id":442,"data":1092,"type":226,"tunes":1093},{"text":444},{},{"id":447,"data":1095,"type":226,"tunes":1096},{"text":449},{},{"id":452,"data":1098,"type":226,"tunes":1099},{"text":454},{},{"id":457,"data":1101,"type":42,"tunes":1102},{"text":459,"level":218},{},{"id":462,"data":1104,"type":226,"tunes":1105},{"text":464},{},{"id":467,"data":1107,"type":226,"tunes":1108},{"text":469},{},{"id":472,"data":1110,"type":226,"tunes":1111},{"text":474},{},{"id":477,"data":1113,"type":42,"tunes":1114},{"text":479,"level":219},{},{"id":482,"data":1116,"type":226,"tunes":1117},{"text":484},{},{"id":487,"data":1119,"type":342,"tunes":1130},{"steps":1120,"title":517,"orientation":341},[1121,1122,1123,1124,1125,1126,1127,1128,1129],{"label":491,"description":492},{"label":494,"description":495},{"label":497,"description":498},{"label":500,"description":501},{"label":503,"description":504},{"label":506,"description":507},{"label":509,"description":510},{"label":512,"description":513},{"label":515,"description":516},{},{"id":520,"data":1132,"type":42,"tunes":1133},{"text":522,"level":219},{},{"id":525,"data":1135,"type":226,"tunes":1136},{"text":527},{},{"id":530,"data":1138,"type":226,"tunes":1139},{"text":532},{},{"id":535,"data":1141,"type":541,"tunes":1142},{"url":537,"title":538,"excerpt":539,"ctaLabel":540},{},{"id":544,"data":1144,"type":42,"tunes":1145},{"text":546,"level":219},{},{"id":549,"data":1147,"type":580,"tunes":1161},{"rows":1148,"title":569,"layout":303,"columns":1157},[1149,1151,1153,1155],{"id":553,"label":554,"values":1150},[556,556,556],{"id":558,"label":559,"values":1152},[556,556,556],{"id":562,"label":563,"values":1154},[556,556,556],{"id":566,"label":567,"values":1156},[556,556,556],[1158,1159,1160],{"id":572,"label":573},{"id":575,"label":576},{"id":578,"label":579},{},{"id":583,"data":1163,"type":226,"tunes":1164},{"text":585},{},{"id":588,"data":1166,"type":42,"tunes":1167},{"text":590,"level":219},{},{"id":593,"data":1169,"type":226,"tunes":1170},{"text":595},{},{"id":598,"data":1172,"type":303,"tunes":1183},{"content":1173,"stretched":43,"withHeadings":14},[1174,1175,1176,1177,1178,1179,1180,1181,1182],[602,603,604],[606,607,608],[610,611,612],[614,615,616],[618,619,620],[622,623,624],[626,627,628],[630,631,632],[634,635,636],{},{"id":639,"data":1185,"type":42,"tunes":1186},{"text":641,"level":219},{},{"id":644,"data":1188,"type":226,"tunes":1189},{"text":646},{},{"id":649,"data":1191,"type":226,"tunes":1192},{"text":651},{},{"id":654,"data":1194,"type":42,"tunes":1195},{"text":656,"level":219},{},{"id":659,"data":1197,"type":303,"tunes":1205},{"content":1198,"stretched":43,"withHeadings":14},[1199,1200,1201,1202,1203,1204],[663,664,665],[667,668,669],[671,672,673],[675,676,677],[679,680,681],[683,684,685],{},{"id":688,"data":1207,"type":42,"tunes":1208},{"text":690,"level":219},{},{"id":693,"data":1210,"type":710,"tunes":1213},{"meta":1211,"items":1212,"style":709},{},[697,698,699,700,701,702,703,704,705,706,707,708],{},{"id":713,"data":1215,"type":226,"tunes":1216},{"text":715},{},{"id":718,"data":1218,"type":541,"tunes":1219},{"url":720,"title":721,"excerpt":722,"ctaLabel":723},{},{"id":726,"data":1221,"type":42,"tunes":1222},{"text":728,"level":219},{},{"id":731,"data":1224,"type":226,"tunes":1225},{"text":733},{},{"id":736,"data":1227,"type":226,"tunes":1228},{"text":738},{},{"id":741,"data":1230,"type":226,"tunes":1231},{"text":743},{},{"id":746,"data":1233,"type":42,"tunes":1234},{"text":748,"level":219},{},{"id":751,"data":1236,"type":226,"tunes":1237},{"text":753},{},{"id":756,"data":1239,"type":226,"tunes":1240},{"text":758},{},{"id":761,"data":1242,"type":42,"tunes":1243},{"text":763,"level":219},{},{"id":766,"data":1245,"type":226,"tunes":1246},{"text":768},{},{"id":771,"data":1248,"type":226,"tunes":1249},{"text":773},{},{"id":776,"data":1251,"type":42,"tunes":1252},{"text":778,"level":219},{},{"id":781,"data":1254,"type":226,"tunes":1255},{"text":783},{},{"id":786,"data":1257,"type":226,"tunes":1258},{"text":788},{},{"id":791,"data":1260,"type":42,"tunes":1261},{"text":793,"level":219},{},{"id":796,"data":1263,"type":796,"tunes":1271},{"items":1264,"title":823},[1265,1266,1267,1268,1269,1270],{"id":800,"answer":801,"question":802},{"id":804,"answer":805,"question":806},{"id":808,"answer":809,"question":810},{"id":812,"answer":813,"question":814},{"id":816,"answer":817,"question":818},{"id":820,"answer":821,"question":822},{},{"id":826,"data":1273,"type":42,"tunes":1274},{"text":828,"level":219},{},{"id":831,"data":1276,"type":831,"tunes":1284},{"title":833,"entries":1277},[1278,1279,1280,1281,1282,1283],{"term":836,"anchor":837,"definition":838},{"term":840,"anchor":841,"definition":842},{"term":844,"anchor":845,"definition":846},{"term":848,"anchor":849,"definition":850},{"term":852,"anchor":853,"definition":854},{"term":856,"anchor":857,"definition":858},{},{"id":861,"data":1286,"type":42,"tunes":1287},{"text":863,"level":219},{},{"id":866,"data":1289,"type":873,"tunes":1292},{"link":868,"meta":1290},{"image":1291,"title":871,"description":872},{"url":556},{},{"id":876,"data":1294,"type":873,"tunes":1297},{"link":878,"meta":1295},{"image":1296,"title":881,"description":882},{"url":556},{},{"id":885,"data":1299,"type":873,"tunes":1302},{"link":887,"meta":1300},{"image":1301,"title":890,"description":891},{"url":556},{},{"id":894,"data":1304,"type":873,"tunes":1307},{"link":896,"meta":1305},{"image":1306,"title":899,"description":900},{"url":556},{},{"id":903,"data":1309,"type":873,"tunes":1312},{"link":905,"meta":1310},{"image":1311,"title":908,"description":909},{"url":556},{},{"id":912,"data":1314,"type":873,"tunes":1317},{"link":914,"meta":1315},{"image":1316,"title":917,"description":918},{"url":556},{},{"id":921,"data":1319,"type":873,"tunes":1322},{"link":923,"meta":1320},{"image":1321,"title":926,"description":927},{"url":556},{},{"id":930,"data":1324,"type":873,"tunes":1327},{"link":932,"meta":1325},{"image":1326,"title":935,"description":936},{"url":556},{},"Post erfolgreich abgerufen",{"items":1330,"source":1393,"manualIds":1394,"manualMatchedIds":1395},[1331,1338,1345,1351,1358,1365,1372,1379,1386],{"id":1332,"slug":1333,"title":1334,"excerpt":1335,"featuredImage":1336,"publishedAt":1337},"459","ollama-is-not-the-product-building-production-ready-open-llm-applications","Ollama Is Not the Product: Building Production-Ready Open-LLM Applications","Running a local model with Ollama is easy. Building a production-ready Open-LLM application is harder: it requires RAG, access control, provider abstraction, evaluation, logging, deployment discipline and a controlled application layer around the model.\n","\u002Fuploads\u002F2026\u002F06\u002Follama-is-not-the-product-building-production-ready-open-llm-applications-1782679361640-h0usqf.webp","2026-06-28T16:39:00.000Z",{"id":1339,"slug":1340,"title":1341,"excerpt":1342,"featuredImage":1343,"publishedAt":1344},"472","why-more-context-can-make-ai-answers-worse","Why More Context Can Make AI Answers Worse","A larger context window does not guarantee a better answer. This article explains how signal dilution, conflicting evidence, stale state, position sensitivity, and lossy compression can reduce AI reliability—and introduces a practical Context Pressure Test.","\u002Fuploads\u002F2026\u002F09\u002Fwhy-more-context-can-make-ai-answers-worse-1790351615793-2ntv2v.webp","2026-09-25T11:51:00.000Z",{"id":1346,"slug":1347,"title":721,"excerpt":1348,"featuredImage":1349,"publishedAt":1350},"460","ai-agent-reliability-why-the-final-answer-is-not-enough","Correct output does not prove correct reasoning, safe execution, or a trustworthy system.","\u002Fuploads\u002F2026\u002F09\u002Fai-agent-reliability-why-the-final-answer-is-not-enough-1788955466306-pl0qhz.webp","2026-09-09T04:01:00.000Z",{"id":1352,"slug":1353,"title":1354,"excerpt":1355,"featuredImage":1356,"publishedAt":1357},"364","tipps-fuer-die-verbesserung-der-seo-suchmaschinenoptimierung","Mastering the SEO Workflow: Essential Optimization Strategies for Organic Growth","A structured SEO workflow is crucial for sustainable organic growth. Learn the ten foundational strategies, from keyword research and technical optimization to content quality and performance analysis.","\u002Fuploads\u002F2026\u002F03\u002Ftipps-fuer-die-verbesserung-der-seo-suchmaschinenoptimierung-1774866098131-hwkzrg.webp","2024-01-26T06:35:00.000Z",{"id":1359,"slug":1360,"title":1361,"excerpt":1362,"featuredImage":1363,"publishedAt":1364},"469","rag-failed-but-which-layer-actually-failed-a-diagnostic-method","RAG Failed — But Which Layer Actually Failed? A Diagnostic Method","When a RAG answer is wrong, blaming retrieval or the model is too vague. This diagnostic method isolates source coverage, query construction, retrieval, ranking, context assembly, generation, evidence attribution, and freshness—so the actual failure can be reproduced and fixed.","\u002Fuploads\u002F2026\u002F09\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method-1790350847177-pior4c.webp","2026-09-24T19:39:00.000Z",{"id":1366,"slug":1367,"title":1368,"excerpt":1369,"featuredImage":1370,"publishedAt":1371},"474","migrating-from-openai-agents-sdk-to-the-agents-api-what-actually-changes-architecturally","Migrating from OpenAI Agents SDK to the Agents API: What Actually Changes Architecturally?","Migrating from the OpenAI Agents SDK to the new Agents API is not an import rename. The runtime boundary changes: the agent loop, durable session, orchestration, context compaction and recovery move toward a managed harness. This guide shows what should move, what should stay in your application, and how to prove the migration before cutover.","\u002Fuploads\u002F2026\u002F09\u002Fmigrating-from-openai-agents-sdk-to-the-agents-api-what-actually-changes-architecturally-1790352171968-ienxr9.webp","2026-09-25T12:01:00.000Z",{"id":1373,"slug":1374,"title":1375,"excerpt":1376,"featuredImage":1377,"publishedAt":1378},"478","what-is-rag-the-simplest-explanation-of-how-it-works","What Is RAG? The Simplest Explanation of How It Works","RAG sounds complicated, but the idea is simple: before an AI answers, it first looks up useful information from a knowledge source and gives that information to the language model. This guide explains RAG, LLMs, state, memory and tools using one simple mental model.","\u002Fuploads\u002F2026\u002F09\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works-1790377492124-khjagt.webp","2026-09-25T19:03:00.000Z",{"id":1380,"slug":1381,"title":1382,"excerpt":1383,"featuredImage":1384,"publishedAt":1385},"457","should-you-buy-5g-openwrt-router-old-firmware","Should You Buy a 5G OpenWrt Router with Old Firmware? ZBT Z8102AX as a Practical Example","Buying a 5G OpenWrt router with older firmware can make sense, but only under the right conditions. The ZBT Z8102AX shows both sides clearly: the hardware is useful, the modem works, and the router stayed stable in testing, but OpenWrt 21.02, weak packaging and unclear upgrade paths require a careful buying decision.","\u002Fuploads\u002F2026\u002F06\u002Fopenwrt-router-review-dual-sim-05-1781620596218-5ldld4.webp","2026-06-16T10:41:00.000Z",{"id":1387,"slug":1388,"title":1389,"excerpt":1390,"featuredImage":1391,"publishedAt":1392},"471","how-to-know-whether-an-ai-agent-actually-used-the-right-evidence","How to Know Whether an AI Agent Actually Used the Right Evidence","An AI agent can cite sources and still use the wrong evidence. This article introduces a practical method for checking claim support, source authority, applicability, provenance, and whether the evidence actually influenced the answer.","\u002Fuploads\u002F2026\u002F09\u002Fhow-to-know-whether-an-ai-agent-actually-used-the-right-evidence-1790351317188-o5z9ve.webp","2026-09-25T11:47:00.000Z","fallback",[],[]]