[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"portal-settings:stajic:zh":3,"public-menus:all":38,"post:computer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system:zh":205,"related:post:computer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system:zh:1":1926},{"statusCode":4,"data":5,"message":37},200,{"tenantId":6,"lang":7,"defaultLang":8,"siteUrl":9,"contactEmail":10,"brandName":11,"logoUrl":12,"siteName":11,"siteDescription":13,"ogImage":10,"robotsIndex":14,"socialLinks":10,"reservedSlugs":10,"seoPolicy":15},"stajic","zh","de","https:\u002F\u002Fstajic.de",null,"Stajic Platform","\u002FLogo_Planet.svg","Stajic Portal",true,{"branding":16,"relatedContent":17,"crossDomainLinks":18},{"logoUrl":12},{"enabled":14},[19,22,25,28,31,34],{"url":20,"label":21,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Ffigure.rocks","figure.rocks",{"url":23,"label":24,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Floving.rocks","loving.rocks",{"url":26,"label":27,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.com","bazify.com",{"url":29,"label":30,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.de","bazify.de",{"url":32,"label":33,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.at","bazify.at",{"url":35,"label":36,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.ba","bazify.ba","Portal settings resolved",[39,45],{"id":40,"name":41,"location":42,"isActive":14,"isDefault":43,"items":44},1,"main-navigation","header",false,[],{"id":46,"name":47,"location":48,"isActive":14,"isDefault":14,"items":49},4,"main-menu","sidebar",[50,66,79,93,103,118,133],{"id":51,"title":52,"url":60,"target":61,"icon":62,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":64,"portfolioId":10,"children":65},"item-18",{"de":53,"en":54,"es":55,"fr":56,"it":54,"ru":57,"sr":58,"zh":59},"Startseite","Home","Inicio","Accueil","Главная","Почетна","首页","\u002Ffull-stack-web-developer-munich-performance-seo-and-maintainable-builds","_self","i-lucide-home","page",111,[],{"id":67,"title":68,"url":75,"target":61,"icon":76,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":77,"portfolioId":10,"children":78},"item-22",{"de":69,"en":69,"es":70,"fr":69,"it":71,"ru":72,"sr":73,"zh":74},"Vision","Visión","Visione","Видение","Визија","想象","\u002Fueber-uns-webdesign-muenchen-webaplikation","i-lucide-eye",113,[],{"id":80,"title":81,"url":89,"target":61,"icon":90,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":91,"portfolioId":10,"children":92},"item-19",{"de":82,"en":83,"es":84,"fr":83,"it":85,"ru":86,"sr":87,"zh":88},"Leistungen","Services","Servicios","Servizi","Услуги","Услуге","服务","\u002Fservices-dienstleistungen-muenchen","i-lucide-wrench",116,[],{"id":94,"title":95,"url":99,"target":61,"icon":100,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":101,"portfolioId":10,"children":102},"item-23",{"de":96,"en":96,"es":96,"fr":96,"it":96,"ru":97,"sr":97,"zh":98},"Blog","Блог","博客","\u002Fblog","i-lucide-book-open",112,[],{"id":104,"title":105,"url":114,"target":61,"icon":115,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":116,"portfolioId":10,"children":117},"item-32",{"de":106,"en":107,"es":108,"fr":109,"it":110,"ru":111,"sr":112,"zh":113},"Neue Technologien","New Technologies","Nuevas tecnologías","Nouvelles technologies","Nuove tecnologie","Новые технологии","Нове технологије","新技术！","\u002Fneue-webtechnologien","i-lucide-sparkles",122,[],{"id":119,"title":120,"url":129,"target":61,"icon":130,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":131,"portfolioId":10,"children":132},"item-20",{"de":121,"en":122,"es":123,"fr":124,"it":125,"ru":126,"sr":127,"zh":128},"Kontakt","Contact us!","Contacto","Contact","Contatto","Контакт","Контактирајте нас","联系我们！","\u002Fcontact","i-lucide-mail",115,[],{"id":134,"title":135,"url":144,"target":61,"icon":145,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":147},"item-21",{"de":136,"en":137,"es":138,"fr":139,"it":140,"ru":141,"sr":142,"zh":143},"Unsere Arbeit","Our Work","Nuestro trabajo","Nos réalisations","I nostri lavori","Наши работы","Наши радови","文件夹","\u002Fportfolio","i-lucide-briefcase",114,[148,161,175,181,193],{"id":149,"title":150,"url":144,"target":61,"icon":159,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":160},"item-24",{"de":151,"en":152,"es":153,"fr":154,"it":155,"ru":156,"sr":157,"zh":158},"Alle Projekte","All Projects","Todos los proyectos","Tous les projets","Tutti i progetti","Все проекты","Сви пројекти","所有项目","i-lucide-grid-3x3",[],{"id":162,"title":163,"url":171,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":174},"item-29",{"de":164,"en":165,"es":166,"fr":167,"it":168,"ru":169,"sr":170,"zh":143},"Local Roots, Global Reach","Local Roots - Global Reach","Empresa local ","Entreprise locale","Azienda locale","Местная компания","Локално предузеће глобално тржиште","\u002Fportfolio\u002Flocal-roots-global-reach-communication-media-systems-for-modern-business","i-lucide-folder","custom",[],{"id":176,"title":177,"url":179,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":180},"item-28",{"de":178,"en":178,"es":178,"fr":178,"it":178,"ru":178,"sr":178,"zh":178},"Solr Suggester","\u002Fportfolio\u002Fsolr-fuzzy-suggester-und-solr-infix-suggester-abfrage-ueber-ajax-und-filterung",[],{"id":182,"title":183,"url":191,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":192},"item-27",{"de":184,"en":185,"es":186,"fr":187,"it":188,"ru":189,"sr":190,"zh":185},"Firmenwebseite SEO","Company Website SEO","Sitio web corporativo SEO","Site web d’entreprise SEO","Sito web aziendale SEO","Корпоративный сайт SEO","Пословна веб-страница SEO","\u002Fportfolio\u002Fseo-sem-branding-mobile-webseite-muenchen",[],{"id":194,"title":195,"url":203,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":204},"item-31",{"de":196,"en":197,"es":198,"fr":199,"it":200,"ru":201,"sr":202,"zh":197},"Digitalisierungsportal","Digitalization Portal","Portal de digitalización","Portail de numérisation","Portale di digitalizzazione","Портал цифровизации","Портал за дигитализацију","\u002Fportfolio\u002Fdigitalisierungsportal-archiv-museum-bibliothek-ead-lido-mets-mods",[],{"statusCode":4,"data":206,"message":1925},{"id":207,"title":208,"slug":209,"content":210,"contentJson":211,"excerpt":937,"featuredImage":938,"featuredImageAlt":939,"featuredImageCaption":10,"featuredImageTitle":10,"featuredImageCopyright":10,"featuredImageAuthor":10,"featuredImageSourceUrl":10,"featuredImageLicense":10,"featuredImageIsAiGenerated":43,"status":940,"publishedAt":941,"createdAt":942,"updatedAt":943,"seoLocalePaths":944,"categories":953,"author":966,"translations":971},"477","计算机使用代理：为什么成功的演示仍可能是一个不可靠的系统","computer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system","\u003Cnav class=\"editorjs-toc\" data-editorjs-toc=\"true\" aria-label=\"目录\">\u003Cstrong class=\"editorjs-toc__title\">目录\u003C\u002Fstrong>\u003Col class=\"editorjs-toc__list editorjs-toc__list--depth-0\">\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-6\" class=\"editorjs-toc__link\">为什么演示是最简单的可靠性测试\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-10\" class=\"editorjs-toc__link\">能力、成功率、可靠性和安全性是不同的主张\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-12\" class=\"editorjs-toc__link\">计算机使用可靠性阶梯\u003C\u002Fa>\u003Col class=\"editorjs-toc__list editorjs-toc__list--depth-1\">\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-15\" class=\"editorjs-toc__link\">第1级——能力：演示问题\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-18\" class=\"editorjs-toc__link\">第2级——可重复性：同一任务能否保持解决？\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-21\" class=\"editorjs-toc__link\">第3级——环境鲁棒性：当网络表现得像真实网络时会发生什么？\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-25\" class=\"editorjs-toc__link\">第4级——长时程控制：当任务变成真实工作时，成功会发生变化\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-29\" class=\"editorjs-toc__link\">第5级——状态感知：环境可能在计划之下发生变化\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-33\" class=\"editorjs-toc__link\">第6级——结果验证：操作真的成功了吗？\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-37\" class=\"editorjs-toc__link\">第7级——安全目标处理：智能体必须知道何时不应继续\u003C\u002Fa>\u003C\u002Fli>\u003C\u002Fol>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-41\" class=\"editorjs-toc__link\">从演示到生产的压力测试\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-44\" class=\"editorjs-toc__link\">基准成功具有有效性边界\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-48\" class=\"editorjs-toc__link\">过程成功和结果成功必须分开评分\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-51\" class=\"editorjs-toc__link\">生产可靠性是一个分布，而不是单一通过率\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-54\" class=\"editorjs-toc__link\">可靠性需要失败预算，而不是完美\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-57\" class=\"editorjs-toc__link\">实用的计算机使用可靠性矩阵\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-59\" class=\"editorjs-toc__link\">计算机使用故障应记录什么\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-63\" class=\"editorjs-toc__link\">安全性是计算机使用代理可靠性的一部分\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-67\" class=\"editorjs-toc__link\">什么会改变这个答案？\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-70\" class=\"editorjs-toc__link\">局限性\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-73\" class=\"editorjs-toc__link\">结论\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-76\" class=\"editorjs-toc__link\">常见问题\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-78\" class=\"editorjs-toc__link\">术语表\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-80\" class=\"editorjs-toc__link\">主要来源和进一步阅读\u003C\u002Fa>\u003C\u002Fli>\u003C\u002Fol>\u003C\u002Fnav>\n\u003Cp>计算机使用代理现在可以点击、输入、浏览、编辑文件、操作桌面应用程序，并完成令人印象深刻的多步骤任务。这使得成功的演示易于理解，也容易被过度解读。一个完成的工作流表明代理在这些条件下能够成功。它并不能说明成功的频率、环境变化时的行为、是否验证结果，或者在目标变得模糊时行动的安全性。\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--info my-6 rounded-xl border p-5 border-blue-300 bg-blue-50 dark:border-blue-900 dark:bg-blue-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">直接回答\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">&lt;strong&gt;成功的计算机使用演示证明的是能力，而非可靠性。&lt;\u002Fstrong&gt;生产可靠性要求代理在环境变化中反复成功、从瞬时故障中恢复、在长周期内保持约束、检测隐藏或变化的状态、验证实际结果，并在目标变得模糊或不安全时停止或询问。正确的生产问题不是“代理能完成这个任务吗？”，而是“在哪些条件下我们可以信任它反复完成这个任务？”\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Caside class=\"editorjs-callout editorjs-callout--warning my-6 rounded-xl border p-5 border-amber-300 bg-amber-50 dark:border-amber-900 dark:bg-amber-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">快速发展的领域\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">本文反映了截至&lt;strong&gt;2026年9月25日&lt;\u002Fstrong&gt;可获得的计算机使用代理研究和平台指南。基准测试结果在不同任务集、环境、模型、步骤限制、评判器或测试框架之间不可直接比较。请将每个基准测试数字与其评估条件一起看待。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Caside class=\"editorjs-callout editorjs-callout--note my-6 rounded-xl border p-5 border-gray-300 bg-gray-50 dark:border-gray-700 dark:bg-gray-900\u002F40\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">本文使用的模型\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">下面的计算机使用可靠性阶梯和演示到生产压力测试是本文提出的实用评估模型。它们不是正式的行业标准。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-6\">为什么演示是最简单的可靠性测试\u003C\u002Fh2>\n\u003Cp>演示通常展示一条成功的轨迹。环境是已知的，任务是预先选定的，操作员可以在失败后重新开始，观众看到的是成功路径。生产系统面对的是一个分布：不同的页面、网络条件、账户状态、弹窗、延迟、UI变化、隐藏状态、权限、中断，以及不完美描述目标的用户。\u003C\u002Fp>\n\u003Cp>这种区别很重要，因为计算机使用代理通过为人类设计的界面操作，而不是确定性API。它们的行动循环依赖于感知、状态解释、规划、交互时机和环境响应。即使用户目标不变，微小的变化也可能改变轨迹。\u003C\u002Fp>\n\u003Cp>微软研究院的WAREX工作明确指出了这个问题：在受控设置中看起来有能力的基准代理，在引入现实的网络不稳定性后，任务成功率大幅下降。失败不一定是“模型变笨了”。而是环境不再是确定性的。\u003C\u002Fp>\n\u003Ch2 id=\"section-10\">能力、成功率、可靠性和安全性是不同的主张\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">主张\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">实际证明的内容\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">未证明的内容\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">代理完成了一次任务\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">在一条观察到的轨迹下的能力\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">可重复性、鲁棒性、安全性或泛化能力\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">代理在基准测试中得分很高\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">在该基准测试的任务和评估条件下的表现\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">在不同环境中的等效生产表现\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">代理通常能达到目标\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">结果成功频率\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">正确的过程、安全的行为，或结果已被验证的证据\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">代理遵循了预期的过程\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">在评估标准下的轨迹质量\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">外部环境实际接受了最终结果\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">代理在测试集中避免了不安全操作\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">在已代表的安全案例上的表现\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">在每个新颖的模糊性、注入或副作用下的安全性\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-12\">计算机使用可靠性阶梯\u003C\u002Fh2>\n\u003Cp>评估计算机使用系统的一个有用方法是从一次性能力逐步走向越来越难的可靠性属性。更高的层级假设了较低层级，但不会自动从它们推导出来。\u003C\u002Fp>\n\u003Csection class=\"editorjs-process my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">计算机使用可靠性阶梯\u003C\u002Fh3>\u003Cdiv class=\"grid grid-cols-1 md:grid-cols-2 xl:grid-cols-3 gap-4\">\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">1\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">1. 能力\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">代理能否在已知条件下至少完成一次任务？\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">2\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">2. 可重复性\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">它能否在重复试验中一致地完成同一任务？\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">3\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">3. 环境鲁棒性\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">它能否在时序变化、网络问题、弹窗、UI变化和小的环境扰动中存活？\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">4\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">4. 长周期控制\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">它能否在许多步骤、应用程序和延迟事件中保持目标、约束和进展？\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">5\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">5. 状态感知\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">它能否检测环境何时变化、隐藏状态何时重要，或假设何时不再有效？\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">6\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">6. 结果验证\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">它是否验证预期结果实际发生，而不是信任自己的行动序列？\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">7\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">7. 安全目标处理\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">当目标模糊、不可行、矛盾或高影响时，它能否停止、询问、拒绝或交回控制权？\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch3 id=\"section-15\">第1级——能力：演示问题\u003C\u002Fh3>\n\u003Cp>能力问的是代理能否执行任务。这很有价值。计算机使用系统发展迅速，现代代理可以完成旧系统无法可靠执行的工作流。\u003C\u002Fp>\n\u003Cp>但能力是一个薄弱的部署标准。一次成功运行并不能告诉你代理是95%的时间成功还是30%的时间成功，失败是无害的还是破坏性的，或者成功是否依赖于幸运的页面状态。\u003C\u002Fp>\n\u003Ch3 id=\"section-18\">第2级——可重复性：同一任务能否保持解决？\u003C\u002Fh3>\n\u003Cp>计算机使用轨迹具有随机性。模型输出会变化，页面加载速度不同，视觉状态会改变，而长工作流会创造许多分支机会。因此，生产测试应多次运行同一任务，而不是将一次通过的轨迹视为具有代表性。\u003C\u002Fp>\n\u003Cp>不仅要衡量平均成功率，还要衡量失败模式的分布：错误点击、过早终止、遗漏确认、字段错误、重复操作、导航循环、过期状态假设以及虚假成功报告。\u003C\u002Fp>\n\u003Ch3 id=\"section-21\">第3级——环境鲁棒性：当网络表现得像真实网络时会发生什么？\u003C\u002Fh3>\n\u003Cp>真实网站不是基准测试夹具。请求会失败，元素会延迟加载，会话会过期，页面会变化，同意横幅会出现，服务器会返回错误，网络条件也会波动。\u003C\u002Fp>\n\u003Cp>WAREX通过将真实的网络不可靠性注入现有基准环境来评估这一差距，并报告任务成功率显著下降。这是一个关键的生产洞察：基准测试可以衡量任务能力，却会低估从环境不稳定中恢复的能力。\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--tip my-6 rounded-xl border p-5 border-violet-300 bg-violet-50 dark:border-violet-900 dark:bg-violet-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">可靠性测试\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">注入延迟、瞬时HTTP故障、过期页面状态、模态对话框、会话过期、重复响应以及受控的UI变化。如果智能体只在干净路径上有效，那么它是具备演示能力的系统，而不是生产可靠系统。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch3 id=\"section-25\">第4级——长时程控制：当任务变成真实工作时，成功会发生变化\u003C\u002Fh3>\n\u003Cp>短任务会隐藏一类只有在数十或数百次操作后才会出现的失败：遗忘约束、重复工作、过早完成、遗漏状态变化、跨应用不一致以及累积的小错误。\u003C\u002Fp>\n\u003Cp>OSWorld 2.0 专门围绕长时程真实世界工作流设计。其任务让人类用户完成的中位时间约为1.6小时，并且需要比早期计算机使用基准多得多的工具调用。在其主要完成度指标下，即使是被评估的最强系统也远未达到完整的任务可靠性。\u003C\u002Fp>\n\u003Cp>WeaveBench从另一个角度得出了类似结论。它评估混合GUI、CLI和代码工作流，并报告最佳评估模型-运行时组合仅通过41.2%的任务。重要的结果不是某个排行榜数字，而是真实的跨界面编排会暴露出更简单的单界面任务所隐藏的失败。\u003C\u002Fp>\n\u003Ch3 id=\"section-29\">第5级——状态感知：环境可能在计划之下发生变化\u003C\u002Fh3>\n\u003Cp>长时间运行的任务通常依赖隐藏或变化的状态：收到一封电子邮件、日历发生变化、表单被提交、后台进程完成、浏览器会话过期、用户修改文件，或外部系统改变可用性。\u003C\u002Fp>\n\u003Cp>微软的SentinelBench认为，许多长时间运行的任务根本不应通过持续行动来解决。正确的行为可能是监控、等待外部事件，然后在状态变化时采取行动。这是一种不同于更快点击或规划更多步骤的能力。\u003C\u002Fp>\n\u003Cp>因此，可靠的计算机使用智能体需要区分现在可操作、等待状态、状态已变化以及假设已失效。\u003C\u002Fp>\n\u003Ch3 id=\"section-33\">第6级——结果验证：操作真的成功了吗？\u003C\u002Fh3>\n\u003Cp>智能体可以执行一个看似正确的序列，但仍然任务失败。按钮点击可能没有注册。表单可能拒绝隐藏验证。文件可能保存到错误目录。购买可能仍未确认。网站可能显示看似成功的屏幕，而底层操作却失败了。\u003C\u002Fp>\n\u003Cp>OpenAI当前的计算机使用指南明确建议对运行进行边界限定和验证，而不是仅依赖模型的最终答案。微软研究院关于计算机使用验证器的工作从评估角度得出了相同结论：过程和结果需要分别评判。\u003C\u002Fp>\n\u003Cp>Universal Verifier研究报告称，早期的验证器设置可能产生高假阳性率，而更强的评分标准设计以及对过程、结果、可控失败和不可控失败的明确区分，会显著提高与人工标注的一致性。\u003C\u002Fp>\n\u003Ch3 id=\"section-37\">第7级——安全目标处理：智能体必须知道何时不应继续\u003C\u002Fh3>\n\u003Cp>计算机使用智能体经过优化以完成目标，但目标持续性本身可能成为一种失败模式。一个模糊的请求、不可能的条件、矛盾的指令、可疑的网页或变化的环境可能需要澄清或停止，而不是采取更多行动。\u003C\u002Fp>\n\u003Cp>BLIND-ACT基准将这一问题作为盲目标导向性进行研究。在该工作中评估的各个系统中，智能体经常在存在模糊性、不可行性、冲突上下文或其他需要重新考虑的原因时仍继续追求任务。作者识别出执行优先偏差和请求首要性等模式。\u003C\u002Fp>\n\u003Cp>这一失败类别很重要，因为一个能力很强的智能体可以更快地让糟糕的情况变得更糟。因此，可靠性包括一项关于何时不应行动的策略。\u003C\u002Fp>\n\u003Ch2 id=\"section-41\">从演示到生产的压力测试\u003C\u002Fh2>\n\u003Cp>在部署计算机使用工作流之前，取成功的演示并系统地移除那些使其变得容易的假设。\u003C\u002Fp>\n\u003Csection class=\"editorjs-process my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">从演示到生产的压力测试\u003C\u002Fh3>\u003Cdiv class=\"grid grid-cols-1 md:grid-cols-2 xl:grid-cols-3 gap-4\">\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">1\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">1. 重新运行干净任务\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">在增加复杂性之前，通过多次试验建立可重复性。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">2\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">2. 扰动环境\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">添加延迟、重试、弹窗、页面变化、过期会话和临时故障。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">3\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">3. 延长时间范围\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">将短演示转变为包含中间状态、多个应用和延迟步骤的完整真实工作流。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">4\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">4. 更改隐藏状态\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">在智能体形成计划后修改账户、文件、任务或外部状态，并测试它是否检测到变化。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">5\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">5. 注入模糊性\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">移除一个重要假设，并测试智能体是询问而不是猜测。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">6\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">6. 注入受控矛盾\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">同时呈现旧状态和新状态，并验证权威的当前状态胜出。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">7\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">7. 要求结果证明\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">使任务完成取决于可验证的最终状态，而不是模型的自我报告。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">8\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">8. 测试后果性边界\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">确认不可逆或敏感操作会触发预期的批准、拒绝或移交。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">9\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">9. 在测试框架或模型更改后重复\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">将运行时升级视为需要回归测试的可靠性变更。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-44\">基准成功具有有效性边界\u003C\u002Fh2>\n\u003Cp>基准分数是一个条件陈述。它对于特定的模型、测试框架、环境、任务集、评判器、工具接口、步骤预算、重试策略、日期和评估方法有效。\u003C\u002Fp>\n\u003Cp>当这些条件从声明中消失时，这个数字就会变得具有误导性。“智能体X得分为80%”比“智能体X在环境Z下使用评判器J和步骤预算N在基准Y上得分为80%”更弱。第二个陈述保留了边界，告诉你这个数字是否可以迁移到你的应用中。\u003C\u002Fp>\n\u003Caside class=\"editorjs-referral my-6\">\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers\" class=\"flex flex-col sm:flex-row gap-4 rounded-xl border border-gray-200 dark:border-gray-700 p-4 transition hover:border-primary-500\">\u003Cdiv class=\"min-w-0 flex-1\">\u003Cstrong class=\"block text-lg text-gray-900 dark:text-gray-100\">答案有效性边界：相关性与可靠AI答案之间缺失的层\u003C\u002Fstrong>\u003Cp class=\"mt-2 text-sm text-gray-600 dark:text-gray-300\">一个框架，用于明确AI声明在何种条件下仍然有效，以及哪些变化需要限制、重新计算或放弃。\u003C\u002Fp>\u003Cspan class=\"mt-3 inline-flex text-sm font-medium text-primary-600 dark:text-primary-400\">阅读答案有效性边界 →\u003C\u002Fspan>\u003C\u002Fdiv>\u003C\u002Fa>\u003C\u002Faside>\n\u003Ch2 id=\"section-48\">过程成功和结果成功必须分开评分\u003C\u002Fh2>\n\u003Csection class=\"editorjs-comparison my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">一次计算机使用运行的四种可能结果\u003C\u002Fh3>\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left dark:border-gray-700 dark:bg-gray-900\">\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">过程\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">结果\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">解释\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">正确过程 \u002F 正确结果\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">错误过程 \u002F 正确结果\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">正确过程 \u002F 错误结果\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">错误过程 \u002F 错误结果\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Cp>WeaveBench报告称，仅根据结果评分可能会显著高估计算机使用性能，因为智能体可能通过捷径或伪造证据产生看似成功的产物。验证器必须检查轨迹和交付物，而不仅仅是最终声明。\u003C\u002Fp>\n\u003Ch2 id=\"section-51\">生产可靠性是一个分布，而不是单一通过率\u003C\u002Fh2>\n\u003Cp>有用的生产评估会抽样你的环境中实际变化的维度。对于浏览器工作流，这可能包括账户年龄、区域设置、视口、页面版本、网络质量、认证状态、现有购物车状态、cookie、弹窗、用户权限以及是否有人类中断运行。\u003C\u002Fp>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">维度\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">示例变化\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">为什么重要\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">环境\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">快与慢网络、临时故障、页面时序\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">测试恢复和等待行为\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">UI\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">不同视口、模态框、元素重排、轻微重新设计\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">测试脆弱的视觉\u002F操作假设\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">状态\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">已登录\u002F已登出、空\u002F非空购物车、现有文件、更改的权限\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">测试隐藏状态推理\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">任务时间范围\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">5步与50+步、一个应用与多个应用\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">测试累积轨迹误差\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模糊性\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">缺少偏好或不完整的用户指令\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">测试智能体是询问而不是猜测\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">后果\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">只读与购买\u002F发送\u002F删除\u002F更改\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">测试确认和授权控制\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">对抗性内容\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">提示注入或误导性页面文本\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">测试指令层级和遏制\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型\u002F测试框架版本\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">运行时升级\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">测试系统级更改带来的回归\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-54\">可靠性需要失败预算，而不是完美\u003C\u002Fh2>\n\u003Cp>没有生产系统是完美可靠的。有用的工程问题是哪些故障是可接受的、可检测的和可恢复的。对本地文件夹进行排序的失败尝试与发送错误的电子邮件、购买错误的产品或更改账户设置并不等同。\u003C\u002Fp>\n\u003Cp>根据后果和可逆性对操作进行分类。低影响、可逆的操作可以容忍更多的自主性。高影响、外部可见或难以逆转的操作需要更强的确认、状态验证、授权和操作后检查。\u003C\u002Fp>\n\u003Ch2 id=\"section-57\">实用的计算机使用可靠性矩阵\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">操作类别\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">示例\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">推荐控制\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">读取\u002F检查\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">打开页面、读取文件、收集信息\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">限定范围、记录来源、容忍可恢复的导航错误\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">可逆的本地更改\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">编辑草稿文件、重新整理临时工作区\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">更改前进行检查点或版本控制；验证结果\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">外部通信\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">发送电子邮件、发布内容、提交表单\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">用户确认或明确授权；验证已接受状态\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">财务\u002F交易\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">购买、结账、付费订阅\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">严格授权、金额\u002F商家限制、最终确认和收据验证\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">破坏性\u002F权限更改\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">删除数据、更改权限、撤销访问\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">窄授权、明确确认、尽可能可逆路径、操作后审计\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-59\">计算机使用故障应记录什么\u003C\u002Fh2>\n\u003Cul>\u003Cli>用户目标和明确约束。\u003C\u002Fli>\u003Cli>模型和框架版本。\u003C\u002Fli>\u003Cli>环境和应用程序版本。\u003C\u002Fli>\u003Cli>与故障相关的截图或结构化观察。\u003C\u002Fli>\u003Cli>带时间戳的操作。\u003C\u002Fli>\u003Cli>工具、点击、键盘和导航结果。\u003C\u002Fli>\u003Cli>状态转换和等待时间。\u003C\u002Fli>\u003Cli>批准、拒绝或移交事件。\u003C\u002Fli>\u003Cli>外部错误和网络故障。\u003C\u002Fli>\u003Cli>最终可观察的环境状态。\u003C\u002Fli>\u003Cli>代理报告的结果。\u003C\u002Fli>\u003Cli>验证器结果以及故障是否可由代理控制。\u003C\u002Fli>\u003C\u002Ful>\n\u003Cp>关键比较是报告的成功与可观察的成功之间的比较。无法区分这两者的系统最终会在生产中积累假阳性。\u003C\u002Fp>\n\u003Caside class=\"editorjs-referral my-6\">\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough\" class=\"flex flex-col sm:flex-row gap-4 rounded-xl border border-gray-200 dark:border-gray-700 p-4 transition hover:border-primary-500\">\u003Cdiv class=\"min-w-0 flex-1\">\u003Cstrong class=\"block text-lg text-gray-900 dark:text-gray-100\">AI代理可靠性：为什么最终答案不够\u003C\u002Fstrong>\u003Cp class=\"mt-2 text-sm text-gray-600 dark:text-gray-300\">一个更广泛的可靠性模型，用于评估代理轨迹、工具使用和中间决策，而不是接受最终答案作为系统正确工作的证明。\u003C\u002Fp>\u003Cspan class=\"mt-3 inline-flex text-sm font-medium text-primary-600 dark:text-primary-400\">阅读代理可靠性文章 →\u003C\u002Fspan>\u003C\u002Fdiv>\u003C\u002Fa>\u003C\u002Faside>\n\u003Ch2 id=\"section-63\">安全性是计算机使用代理可靠性的一部分\u003C\u002Fh2>\n\u003Cp>计算机使用代理不仅仅读取不受信任的内容；它们可以在读取后采取行动。这使得提示注入、恶意页面内容和网络钓鱼成为执行路径风险。\u003C\u002Fp>\n\u003Cp>OpenAI当前的计算机使用指南建议隔离环境、允许列出站点和操作、将屏幕内容视为不受信任、确认重要操作、限制运行并验证实际结果。ChatGPT代理同样使用确认、提示注入监控和监督模式来处理敏感上下文。\u003C\u002Fp>\n\u003Cp>架构原则比任何单一提供商更广泛：代理观察到的内容不得被允许重新定义用户的权限。网页可以提供数据。它不能授予将数据发送到其他地方、购买东西、更改凭据或覆盖任务边界的权限。\u003C\u002Fp>\n\u003Ch2 id=\"section-67\">什么会改变这个答案？\u003C\u002Fh2>\n\u003Cp>如果计算机使用模型在代表性生产分布中变得对长视野、动态状态、UI变化、环境故障和模糊目标具有鲁棒性，可靠性差距将会缩小。更好的原生状态API、标准化的机器可读接口和更强的验证器基础设施也可以减少所需的脆弱GUI交互量。\u003C\u002Fp>\n\u003Cp>部署阈值也随任务后果而变化。70%的成功率对于监督下的低风险研究任务可能有用，但对于自主的金融或破坏性工作流则不可接受。因此，可靠性必须根据每个故障类别的成本来评估，而不是一个通用的通过率阈值。\u003C\u002Fp>\n\u003Ch2 id=\"section-70\">局限性\u003C\u002Fh2>\n\u003Cp>引用的基准评估不同的环境，不应相互排名，就好像它们测量的是同一件事。WAREX强调网络不可靠性；WeaveBench针对混合长视野工作；OSWorld 2.0针对现实的长工作流；BLIND-ACT专注于模糊和不可行情况下的目标处理。\u003C\u002Fp>\n\u003Cp>基准结果也会很快过时。模型、框架和验证器的改进可以在几个月内显著改变分数。因此，持久的教训是评估方法：改变条件、将过程与结果分开、验证外部状态，并围绕每个性能声明保留边界。\u003C\u002Fp>\n\u003Ch2 id=\"section-73\">结论\u003C\u002Fh2>\n\u003Cp>计算机使用代理已经足够有用。这正是评估问题发生变化的原因。挑战不再仅仅是代理能否点击完成一个工作流程。而是当干净的演示条件消失时，系统是否仍然可靠。\u003C\u002Fp>\n\u003Cp>将一次成功的运行视为能力的证据。然后测试可重复性、环境鲁棒性、长时程控制、状态感知、结果验证和安全目标处理。生产级计算机使用代理不是能完成演示的那个。而是其失败边界已知、可测量和可控的那个。\u003C\u002Fp>\n\u003Ch2 id=\"section-76\">常见问题\u003C\u002Fh2>\n\u003Csection class=\"editorjs-faq my-6 rounded-xl border border-gray-200 p-5 dark:border-gray-700\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">计算机使用代理的可靠性\u003C\u002Fh3>\u003Cdiv id=\"faq1\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">成功的计算机使用代理演示能证明生产可靠性吗？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">不能。它只证明了在一条观察到的轨迹下的能力。生产可靠性需要在环境变化、长时间运行任务、状态变化、模糊性、恢复条件和有后果的操作中反复成功。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq2\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">为什么计算机使用基准测试看起来比实际表现好得多？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">基准测试可能使用更受控的环境、更短的任务、稳定的网络条件、更简单的应用组合或无法捕捉所有过程失败的结果标准。确切的有效性边界取决于每个基准测试。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq3\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">计算机使用操作后最重要的可靠性检查是什么？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">验证实际的外部结果。不要将代理的最终陈述或预期的点击序列视为目标系统已接受操作的证据。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq4\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">为什么长时程计算机任务仍然困难？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">错误在多个操作中累积，约束被遗忘，外部状态变化，工作跨越多个应用，隐藏状态很重要，代理必须决定何时等待、询问、验证或恢复，而不是简单地继续操作。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq5\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">在部署前我应该如何测试浏览器或桌面代理？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">重复干净任务，注入现实的环境故障，改变UI和状态，延长工作流程时程，引入模糊性，要求可观察的结果证明，测试高影响操作控制，并在模型或测试框架更改后重新运行测试套件。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq6\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">计算机使用代理是否应始终需要人工确认？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">并非每个低风险操作都需要。确认要求应随后果、可逆性、权限和不确定性而扩展。高影响、外部可见或难以逆转的操作需要更强的控制。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-78\">术语表\u003C\u002Fh2>\n\u003Csection class=\"editorjs-glossary my-6 rounded-xl border border-gray-200 dark:border-gray-700 p-5\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">关键可靠性术语\u003C\u002Fh3>\u003Cdl>\u003Cdiv id=\"computer-use-agent\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">计算机使用代理\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">一种AI代理，通过观察和操作（如点击、输入、滚动、文件操作或跨应用工作流程）与图形用户界面或计算机环境交互。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"repeatability\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">可重复性\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">代理在重复运行中一致完成任务的程度，而不是仅在选定的轨迹上成功。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"environmental-robustness\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">环境鲁棒性\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">在现实变化（如延迟、瞬时错误、UI变化、会话状态和意外页面条件）下保持正确行为的能力。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"outcome-verification\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">结果验证\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">在操作后检查实际外部状态以确认预期结果发生，而不是依赖代理的自我报告。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"blind-goal-directedness\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">盲目目标导向\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">一种失败模式，计算机使用代理在模糊性、不可行性、矛盾条件或需要停止并重新评估的原因下仍继续追求目标。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"reliability-boundary\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">可靠性边界\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">一组条件，在这些条件下观察到的成功率或能力声明对于特定部署决策仍具有足够的代表性。\u003C\u002Fdd>\u003C\u002Fdiv>\u003C\u002Fdl>\u003C\u002Fsection>\n\u003Ch2 id=\"section-80\">主要来源和进一步阅读\u003C\u002Fh2>\n\u003Ca href=\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Ftools-computer-use\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">OpenAI — 计算机使用\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">当前开发者指南，关于隔离环境、将屏幕内容视为不可信、确认有后果的操作、限制运行和验证结果。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fopenai.com\u002Findex\u002Frunning-codex-safely\u002F\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">OpenAI — 在OpenAI安全运行Codex\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">当前生产指南，关于技术边界、人工批准、遥测和对在真实系统上操作的代理的控制。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Fpublication\u002Fwarex-web-agent-reliability-evaluation-on-existing-benchmarks\u002F\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">微软研究院 — WAREX\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">2026年评估显示，现实的网络不可靠性导致浏览器代理在现有基准测试上的任务成功率显著下降。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Farticles\u002Fthe-art-of-building-verifiers-for-computer-use-agents\u002F\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">微软研究院 — 为计算机使用代理构建验证器的艺术\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">2026年关于过程与结果评估、可控与不可控失败以及可靠轨迹验证的工作。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Fpublication\u002Fweavebench-a-long-horizon-real-world-benchmark-for-computer-use-agents-with-hybrid-interfaces\u002F\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">微软研究院 — WeaveBench\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">2026年长时程基准测试，结合GUI、CLI和代码工作流程，显示当前代理与可靠现实世界完成之间的巨大差距。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2606.29537\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">OSWorld 2.0 — 在长时程现实世界任务上对计算机使用代理进行基准测试\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">2026年基准测试，专注于现实的长时程计算机使用工作流程、隐藏状态和跨源推理。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Fpublication\u002Fsentinelbench-a-benchmark-for-long-running-monitoring-agents\u002F\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">微软研究院 — SentinelBench\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">2026年基准测试，针对时间演化任务，代理必须监控环境并响应状态变化，而不是持续操作。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Fpublication\u002Fjust-do-it-computer-use-agents-exhibit-blind-goal-directedness\u002F\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">微软研究院 — 就去做吧！？计算机使用代理表现出盲目目标导向\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">ICLR 2026研究，关于代理继续追求模糊、矛盾或不可行的目标。\u003C\u002Fp>\u003C\u002Fa>",{"time":212,"blocks":213,"version":936},1790363951067,[214,222,228,236,243,250,255,260,265,270,275,305,310,315,343,348,353,358,363,368,373,378,383,388,395,400,405,410,415,420,425,430,435,440,445,450,455,460,465,470,475,480,485,517,522,527,532,541,546,580,585,590,595,636,641,646,651,656,685,690,710,715,723,728,733,738,743,748,753,758,763,768,773,778,783,788,793,823,828,858,863,873,882,891,900,909,918,927],{"id":215,"data":216,"type":220,"tunes":221},"IYG9UPcPY0",{"title":217,"maxLevel":218,"minLevel":219},"目录",3,2,"tableOfContents",{},{"id":223,"data":224,"type":226,"tunes":227},"intro",{"text":225},"计算机使用代理现在可以点击、输入、浏览、编辑文件、操作桌面应用程序，并完成令人印象深刻的多步骤任务。这使得成功的演示易于理解，也容易被过度解读。一个完成的工作流表明代理在这些条件下能够成功。它并不能说明成功的频率、环境变化时的行为、是否验证结果，或者在目标变得模糊时行动的安全性。","paragraph",{},{"id":229,"data":230,"type":234,"tunes":235},"direct",{"body":231,"title":232,"variant":233},"\u003Cstrong>成功的计算机使用演示证明的是能力，而非可靠性。\u003C\u002Fstrong>生产可靠性要求代理在环境变化中反复成功、从瞬时故障中恢复、在长周期内保持约束、检测隐藏或变化的状态、验证实际结果，并在目标变得模糊或不安全时停止或询问。正确的生产问题不是“代理能完成这个任务吗？”，而是“在哪些条件下我们可以信任它反复完成这个任务？”","直接回答","info","callout",{},{"id":237,"data":238,"type":234,"tunes":242},"freshness",{"body":239,"title":240,"variant":241},"本文反映了截至\u003Cstrong>2026年9月25日\u003C\u002Fstrong>可获得的计算机使用代理研究和平台指南。基准测试结果在不同任务集、环境、模型、步骤限制、评判器或测试框架之间不可直接比较。请将每个基准测试数字与其评估条件一起看待。","快速发展的领域","warning",{},{"id":244,"data":245,"type":234,"tunes":249},"model-note",{"body":246,"title":247,"variant":248},"下面的计算机使用可靠性阶梯和演示到生产压力测试是本文提出的实用评估模型。它们不是正式的行业标准。","本文使用的模型","note",{},{"id":251,"data":252,"type":42,"tunes":254},"h-demo",{"text":253,"level":219},"为什么演示是最简单的可靠性测试",{},{"id":256,"data":257,"type":226,"tunes":259},"p-demo-1",{"text":258},"演示通常展示一条成功的轨迹。环境是已知的，任务是预先选定的，操作员可以在失败后重新开始，观众看到的是成功路径。生产系统面对的是一个分布：不同的页面、网络条件、账户状态、弹窗、延迟、UI变化、隐藏状态、权限、中断，以及不完美描述目标的用户。",{},{"id":261,"data":262,"type":226,"tunes":264},"p-demo-2",{"text":263},"这种区别很重要，因为计算机使用代理通过为人类设计的界面操作，而不是确定性API。它们的行动循环依赖于感知、状态解释、规划、交互时机和环境响应。即使用户目标不变，微小的变化也可能改变轨迹。",{},{"id":266,"data":267,"type":226,"tunes":269},"p-demo-3",{"text":268},"微软研究院的WAREX工作明确指出了这个问题：在受控设置中看起来有能力的基准代理，在引入现实的网络不稳定性后，任务成功率大幅下降。失败不一定是“模型变笨了”。而是环境不再是确定性的。",{},{"id":271,"data":272,"type":42,"tunes":274},"h-claims",{"text":273,"level":219},"能力、成功率、可靠性和安全性是不同的主张",{},{"id":276,"data":277,"type":303,"tunes":304},"claims-table",{"content":278,"stretched":43,"withHeadings":14},[279,283,287,291,295,299],[280,281,282],"主张","实际证明的内容","未证明的内容",[284,285,286],"代理完成了一次任务","在一条观察到的轨迹下的能力","可重复性、鲁棒性、安全性或泛化能力",[288,289,290],"代理在基准测试中得分很高","在该基准测试的任务和评估条件下的表现","在不同环境中的等效生产表现",[292,293,294],"代理通常能达到目标","结果成功频率","正确的过程、安全的行为，或结果已被验证的证据",[296,297,298],"代理遵循了预期的过程","在评估标准下的轨迹质量","外部环境实际接受了最终结果",[300,301,302],"代理在测试集中避免了不安全操作","在已代表的安全案例上的表现","在每个新颖的模糊性、注入或副作用下的安全性","table",{},{"id":306,"data":307,"type":42,"tunes":309},"h-ladder",{"text":308,"level":219},"计算机使用可靠性阶梯",{},{"id":311,"data":312,"type":226,"tunes":314},"p-ladder-intro",{"text":313},"评估计算机使用系统的一个有用方法是从一次性能力逐步走向越来越难的可靠性属性。更高的层级假设了较低层级，但不会自动从它们推导出来。",{},{"id":316,"data":317,"type":341,"tunes":342},"ladder-flow",{"steps":318,"title":308,"orientation":340},[319,322,325,328,331,334,337],{"label":320,"description":321},"1. 能力","代理能否在已知条件下至少完成一次任务？",{"label":323,"description":324},"2. 可重复性","它能否在重复试验中一致地完成同一任务？",{"label":326,"description":327},"3. 环境鲁棒性","它能否在时序变化、网络问题、弹窗、UI变化和小的环境扰动中存活？",{"label":329,"description":330},"4. 长周期控制","它能否在许多步骤、应用程序和延迟事件中保持目标、约束和进展？",{"label":332,"description":333},"5. 状态感知","它能否检测环境何时变化、隐藏状态何时重要，或假设何时不再有效？",{"label":335,"description":336},"6. 结果验证","它是否验证预期结果实际发生，而不是信任自己的行动序列？",{"label":338,"description":339},"7. 安全目标处理","当目标模糊、不可行、矛盾或高影响时，它能否停止、询问、拒绝或交回控制权？","auto","processFlow",{},{"id":344,"data":345,"type":42,"tunes":347},"h-capability",{"text":346,"level":218},"第1级——能力：演示问题",{},{"id":349,"data":350,"type":226,"tunes":352},"p-capability-1",{"text":351},"能力问的是代理能否执行任务。这很有价值。计算机使用系统发展迅速，现代代理可以完成旧系统无法可靠执行的工作流。",{},{"id":354,"data":355,"type":226,"tunes":357},"p-capability-2",{"text":356},"但能力是一个薄弱的部署标准。一次成功运行并不能告诉你代理是95%的时间成功还是30%的时间成功，失败是无害的还是破坏性的，或者成功是否依赖于幸运的页面状态。",{},{"id":359,"data":360,"type":42,"tunes":362},"h-repeatability",{"text":361,"level":218},"第2级——可重复性：同一任务能否保持解决？",{},{"id":364,"data":365,"type":226,"tunes":367},"p-repeat-1",{"text":366},"计算机使用轨迹具有随机性。模型输出会变化，页面加载速度不同，视觉状态会改变，而长工作流会创造许多分支机会。因此，生产测试应多次运行同一任务，而不是将一次通过的轨迹视为具有代表性。",{},{"id":369,"data":370,"type":226,"tunes":372},"p-repeat-2",{"text":371},"不仅要衡量平均成功率，还要衡量失败模式的分布：错误点击、过早终止、遗漏确认、字段错误、重复操作、导航循环、过期状态假设以及虚假成功报告。",{},{"id":374,"data":375,"type":42,"tunes":377},"h-robustness",{"text":376,"level":218},"第3级——环境鲁棒性：当网络表现得像真实网络时会发生什么？",{},{"id":379,"data":380,"type":226,"tunes":382},"p-robust-1",{"text":381},"真实网站不是基准测试夹具。请求会失败，元素会延迟加载，会话会过期，页面会变化，同意横幅会出现，服务器会返回错误，网络条件也会波动。",{},{"id":384,"data":385,"type":226,"tunes":387},"p-robust-2",{"text":386},"WAREX通过将真实的网络不可靠性注入现有基准环境来评估这一差距，并报告任务成功率显著下降。这是一个关键的生产洞察：基准测试可以衡量任务能力，却会低估从环境不稳定中恢复的能力。",{},{"id":389,"data":390,"type":234,"tunes":394},"robust-tip",{"body":391,"title":392,"variant":393},"注入延迟、瞬时HTTP故障、过期页面状态、模态对话框、会话过期、重复响应以及受控的UI变化。如果智能体只在干净路径上有效，那么它是具备演示能力的系统，而不是生产可靠系统。","可靠性测试","tip",{},{"id":396,"data":397,"type":42,"tunes":399},"h-long",{"text":398,"level":218},"第4级——长时程控制：当任务变成真实工作时，成功会发生变化",{},{"id":401,"data":402,"type":226,"tunes":404},"p-long-1",{"text":403},"短任务会隐藏一类只有在数十或数百次操作后才会出现的失败：遗忘约束、重复工作、过早完成、遗漏状态变化、跨应用不一致以及累积的小错误。",{},{"id":406,"data":407,"type":226,"tunes":409},"p-long-2",{"text":408},"OSWorld 2.0 专门围绕长时程真实世界工作流设计。其任务让人类用户完成的中位时间约为1.6小时，并且需要比早期计算机使用基准多得多的工具调用。在其主要完成度指标下，即使是被评估的最强系统也远未达到完整的任务可靠性。",{},{"id":411,"data":412,"type":226,"tunes":414},"p-long-3",{"text":413},"WeaveBench从另一个角度得出了类似结论。它评估混合GUI、CLI和代码工作流，并报告最佳评估模型-运行时组合仅通过41.2%的任务。重要的结果不是某个排行榜数字，而是真实的跨界面编排会暴露出更简单的单界面任务所隐藏的失败。",{},{"id":416,"data":417,"type":42,"tunes":419},"h-state",{"text":418,"level":218},"第5级——状态感知：环境可能在计划之下发生变化",{},{"id":421,"data":422,"type":226,"tunes":424},"p-state-1",{"text":423},"长时间运行的任务通常依赖隐藏或变化的状态：收到一封电子邮件、日历发生变化、表单被提交、后台进程完成、浏览器会话过期、用户修改文件，或外部系统改变可用性。",{},{"id":426,"data":427,"type":226,"tunes":429},"p-state-2",{"text":428},"微软的SentinelBench认为，许多长时间运行的任务根本不应通过持续行动来解决。正确的行为可能是监控、等待外部事件，然后在状态变化时采取行动。这是一种不同于更快点击或规划更多步骤的能力。",{},{"id":431,"data":432,"type":226,"tunes":434},"p-state-3",{"text":433},"因此，可靠的计算机使用智能体需要区分现在可操作、等待状态、状态已变化以及假设已失效。",{},{"id":436,"data":437,"type":42,"tunes":439},"h-verify",{"text":438,"level":218},"第6级——结果验证：操作真的成功了吗？",{},{"id":441,"data":442,"type":226,"tunes":444},"p-verify-1",{"text":443},"智能体可以执行一个看似正确的序列，但仍然任务失败。按钮点击可能没有注册。表单可能拒绝隐藏验证。文件可能保存到错误目录。购买可能仍未确认。网站可能显示看似成功的屏幕，而底层操作却失败了。",{},{"id":446,"data":447,"type":226,"tunes":449},"p-verify-2",{"text":448},"OpenAI当前的计算机使用指南明确建议对运行进行边界限定和验证，而不是仅依赖模型的最终答案。微软研究院关于计算机使用验证器的工作从评估角度得出了相同结论：过程和结果需要分别评判。",{},{"id":451,"data":452,"type":226,"tunes":454},"p-verify-3",{"text":453},"Universal Verifier研究报告称，早期的验证器设置可能产生高假阳性率，而更强的评分标准设计以及对过程、结果、可控失败和不可控失败的明确区分，会显著提高与人工标注的一致性。",{},{"id":456,"data":457,"type":42,"tunes":459},"h-safe-goal",{"text":458,"level":218},"第7级——安全目标处理：智能体必须知道何时不应继续",{},{"id":461,"data":462,"type":226,"tunes":464},"p-safe-1",{"text":463},"计算机使用智能体经过优化以完成目标，但目标持续性本身可能成为一种失败模式。一个模糊的请求、不可能的条件、矛盾的指令、可疑的网页或变化的环境可能需要澄清或停止，而不是采取更多行动。",{},{"id":466,"data":467,"type":226,"tunes":469},"p-safe-2",{"text":468},"BLIND-ACT基准将这一问题作为盲目标导向性进行研究。在该工作中评估的各个系统中，智能体经常在存在模糊性、不可行性、冲突上下文或其他需要重新考虑的原因时仍继续追求任务。作者识别出执行优先偏差和请求首要性等模式。",{},{"id":471,"data":472,"type":226,"tunes":474},"p-safe-3",{"text":473},"这一失败类别很重要，因为一个能力很强的智能体可以更快地让糟糕的情况变得更糟。因此，可靠性包括一项关于何时不应行动的策略。",{},{"id":476,"data":477,"type":42,"tunes":479},"h-stress",{"text":478,"level":219},"从演示到生产的压力测试",{},{"id":481,"data":482,"type":226,"tunes":484},"p-stress-intro",{"text":483},"在部署计算机使用工作流之前，取成功的演示并系统地移除那些使其变得容易的假设。",{},{"id":486,"data":487,"type":341,"tunes":516},"stress-flow",{"steps":488,"title":478,"orientation":340},[489,492,495,498,501,504,507,510,513],{"label":490,"description":491},"1. 重新运行干净任务","在增加复杂性之前，通过多次试验建立可重复性。",{"label":493,"description":494},"2. 扰动环境","添加延迟、重试、弹窗、页面变化、过期会话和临时故障。",{"label":496,"description":497},"3. 延长时间范围","将短演示转变为包含中间状态、多个应用和延迟步骤的完整真实工作流。",{"label":499,"description":500},"4. 更改隐藏状态","在智能体形成计划后修改账户、文件、任务或外部状态，并测试它是否检测到变化。",{"label":502,"description":503},"5. 注入模糊性","移除一个重要假设，并测试智能体是询问而不是猜测。",{"label":505,"description":506},"6. 注入受控矛盾","同时呈现旧状态和新状态，并验证权威的当前状态胜出。",{"label":508,"description":509},"7. 要求结果证明","使任务完成取决于可验证的最终状态，而不是模型的自我报告。",{"label":511,"description":512},"8. 测试后果性边界","确认不可逆或敏感操作会触发预期的批准、拒绝或移交。",{"label":514,"description":515},"9. 在测试框架或模型更改后重复","将运行时升级视为需要回归测试的可靠性变更。",{},{"id":518,"data":519,"type":42,"tunes":521},"h-benchmark-boundary",{"text":520,"level":219},"基准成功具有有效性边界",{},{"id":523,"data":524,"type":226,"tunes":526},"p-boundary-1",{"text":525},"基准分数是一个条件陈述。它对于特定的模型、测试框架、环境、任务集、评判器、工具接口、步骤预算、重试策略、日期和评估方法有效。",{},{"id":528,"data":529,"type":226,"tunes":531},"p-boundary-2",{"text":530},"当这些条件从声明中消失时，这个数字就会变得具有误导性。“智能体X得分为80%”比“智能体X在环境Z下使用评判器J和步骤预算N在基准Y上得分为80%”更弱。第二个陈述保留了边界，告诉你这个数字是否可以迁移到你的应用中。",{},{"id":533,"data":534,"type":539,"tunes":540},"ref-avb",{"url":535,"title":536,"excerpt":537,"ctaLabel":538},"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers","答案有效性边界：相关性与可靠AI答案之间缺失的层","一个框架，用于明确AI声明在何种条件下仍然有效，以及哪些变化需要限制、重新计算或放弃。","阅读答案有效性边界","referralArticle",{},{"id":542,"data":543,"type":42,"tunes":545},"h-process-outcome",{"text":544,"level":219},"过程成功和结果成功必须分开评分",{},{"id":547,"data":548,"type":578,"tunes":579},"process-outcome-comparison",{"rows":549,"title":567,"layout":303,"columns":568},[550,555,559,563],{"id":551,"label":552,"values":553},"good-good","正确过程 \u002F 正确结果",[554,554,554],"",{"id":556,"label":557,"values":558},"bad-good","错误过程 \u002F 正确结果",[554,554,554],{"id":560,"label":561,"values":562},"good-bad","正确过程 \u002F 错误结果",[554,554,554],{"id":564,"label":565,"values":566},"bad-bad","错误过程 \u002F 错误结果",[554,554,554],"一次计算机使用运行的四种可能结果",[569,572,575],{"id":570,"label":571},"process","过程",{"id":573,"label":574},"outcome","结果",{"id":576,"label":577},"interpretation","解释","comparison",{},{"id":581,"data":582,"type":226,"tunes":584},"p-process-1",{"text":583},"WeaveBench报告称，仅根据结果评分可能会显著高估计算机使用性能，因为智能体可能通过捷径或伪造证据产生看似成功的产物。验证器必须检查轨迹和交付物，而不仅仅是最终声明。",{},{"id":586,"data":587,"type":42,"tunes":589},"h-distribution",{"text":588,"level":219},"生产可靠性是一个分布，而不是单一通过率",{},{"id":591,"data":592,"type":226,"tunes":594},"p-dist-1",{"text":593},"有用的生产评估会抽样你的环境中实际变化的维度。对于浏览器工作流，这可能包括账户年龄、区域设置、视口、页面版本、网络质量、认证状态、现有购物车状态、cookie、弹窗、用户权限以及是否有人类中断运行。",{},{"id":596,"data":597,"type":303,"tunes":635},"distribution-table",{"content":598,"stretched":43,"withHeadings":14},[599,603,607,611,615,619,623,627,631],[600,601,602],"维度","示例变化","为什么重要",[604,605,606],"环境","快与慢网络、临时故障、页面时序","测试恢复和等待行为",[608,609,610],"UI","不同视口、模态框、元素重排、轻微重新设计","测试脆弱的视觉\u002F操作假设",[612,613,614],"状态","已登录\u002F已登出、空\u002F非空购物车、现有文件、更改的权限","测试隐藏状态推理",[616,617,618],"任务时间范围","5步与50+步、一个应用与多个应用","测试累积轨迹误差",[620,621,622],"模糊性","缺少偏好或不完整的用户指令","测试智能体是询问而不是猜测",[624,625,626],"后果","只读与购买\u002F发送\u002F删除\u002F更改","测试确认和授权控制",[628,629,630],"对抗性内容","提示注入或误导性页面文本","测试指令层级和遏制",[632,633,634],"模型\u002F测试框架版本","运行时升级","测试系统级更改带来的回归",{},{"id":637,"data":638,"type":42,"tunes":640},"h-budget",{"text":639,"level":219},"可靠性需要失败预算，而不是完美",{},{"id":642,"data":643,"type":226,"tunes":645},"p-budget-1",{"text":644},"没有生产系统是完美可靠的。有用的工程问题是哪些故障是可接受的、可检测的和可恢复的。对本地文件夹进行排序的失败尝试与发送错误的电子邮件、购买错误的产品或更改账户设置并不等同。",{},{"id":647,"data":648,"type":226,"tunes":650},"p-budget-2",{"text":649},"根据后果和可逆性对操作进行分类。低影响、可逆的操作可以容忍更多的自主性。高影响、外部可见或难以逆转的操作需要更强的确认、状态验证、授权和操作后检查。",{},{"id":652,"data":653,"type":42,"tunes":655},"h-matrix",{"text":654,"level":219},"实用的计算机使用可靠性矩阵",{},{"id":657,"data":658,"type":303,"tunes":684},"control-matrix",{"content":659,"stretched":43,"withHeadings":14},[660,664,668,672,676,680],[661,662,663],"操作类别","示例","推荐控制",[665,666,667],"读取\u002F检查","打开页面、读取文件、收集信息","限定范围、记录来源、容忍可恢复的导航错误",[669,670,671],"可逆的本地更改","编辑草稿文件、重新整理临时工作区","更改前进行检查点或版本控制；验证结果",[673,674,675],"外部通信","发送电子邮件、发布内容、提交表单","用户确认或明确授权；验证已接受状态",[677,678,679],"财务\u002F交易","购买、结账、付费订阅","严格授权、金额\u002F商家限制、最终确认和收据验证",[681,682,683],"破坏性\u002F权限更改","删除数据、更改权限、撤销访问","窄授权、明确确认、尽可能可逆路径、操作后审计",{},{"id":686,"data":687,"type":42,"tunes":689},"h-log",{"text":688,"level":219},"计算机使用故障应记录什么",{},{"id":691,"data":692,"type":708,"tunes":709},"log-list",{"meta":693,"items":694,"style":707},{},[695,696,697,698,699,700,701,702,703,704,705,706],"用户目标和明确约束。","模型和框架版本。","环境和应用程序版本。","与故障相关的截图或结构化观察。","带时间戳的操作。","工具、点击、键盘和导航结果。","状态转换和等待时间。","批准、拒绝或移交事件。","外部错误和网络故障。","最终可观察的环境状态。","代理报告的结果。","验证器结果以及故障是否可由代理控制。","unordered","list",{},{"id":711,"data":712,"type":226,"tunes":714},"p-log-1",{"text":713},"关键比较是报告的成功与可观察的成功之间的比较。无法区分这两者的系统最终会在生产中积累假阳性。",{},{"id":716,"data":717,"type":539,"tunes":722},"ref-reliability",{"url":718,"title":719,"excerpt":720,"ctaLabel":721},"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough","AI代理可靠性：为什么最终答案不够","一个更广泛的可靠性模型，用于评估代理轨迹、工具使用和中间决策，而不是接受最终答案作为系统正确工作的证明。","阅读代理可靠性文章",{},{"id":724,"data":725,"type":42,"tunes":727},"h-security",{"text":726,"level":219},"安全性是计算机使用代理可靠性的一部分",{},{"id":729,"data":730,"type":226,"tunes":732},"p-sec-1",{"text":731},"计算机使用代理不仅仅读取不受信任的内容；它们可以在读取后采取行动。这使得提示注入、恶意页面内容和网络钓鱼成为执行路径风险。",{},{"id":734,"data":735,"type":226,"tunes":737},"p-sec-2",{"text":736},"OpenAI当前的计算机使用指南建议隔离环境、允许列出站点和操作、将屏幕内容视为不受信任、确认重要操作、限制运行并验证实际结果。ChatGPT代理同样使用确认、提示注入监控和监督模式来处理敏感上下文。",{},{"id":739,"data":740,"type":226,"tunes":742},"p-sec-3",{"text":741},"架构原则比任何单一提供商更广泛：代理观察到的内容不得被允许重新定义用户的权限。网页可以提供数据。它不能授予将数据发送到其他地方、购买东西、更改凭据或覆盖任务边界的权限。",{},{"id":744,"data":745,"type":42,"tunes":747},"h-change",{"text":746,"level":219},"什么会改变这个答案？",{},{"id":749,"data":750,"type":226,"tunes":752},"p-change-1",{"text":751},"如果计算机使用模型在代表性生产分布中变得对长视野、动态状态、UI变化、环境故障和模糊目标具有鲁棒性，可靠性差距将会缩小。更好的原生状态API、标准化的机器可读接口和更强的验证器基础设施也可以减少所需的脆弱GUI交互量。",{},{"id":754,"data":755,"type":226,"tunes":757},"p-change-2",{"text":756},"部署阈值也随任务后果而变化。70%的成功率对于监督下的低风险研究任务可能有用，但对于自主的金融或破坏性工作流则不可接受。因此，可靠性必须根据每个故障类别的成本来评估，而不是一个通用的通过率阈值。",{},{"id":759,"data":760,"type":42,"tunes":762},"h-limitations",{"text":761,"level":219},"局限性",{},{"id":764,"data":765,"type":226,"tunes":767},"p-limit-1",{"text":766},"引用的基准评估不同的环境，不应相互排名，就好像它们测量的是同一件事。WAREX强调网络不可靠性；WeaveBench针对混合长视野工作；OSWorld 2.0针对现实的长工作流；BLIND-ACT专注于模糊和不可行情况下的目标处理。",{},{"id":769,"data":770,"type":226,"tunes":772},"p-limit-2",{"text":771},"基准结果也会很快过时。模型、框架和验证器的改进可以在几个月内显著改变分数。因此，持久的教训是评估方法：改变条件、将过程与结果分开、验证外部状态，并围绕每个性能声明保留边界。",{},{"id":774,"data":775,"type":42,"tunes":777},"h-conclusion",{"text":776,"level":219},"结论",{},{"id":779,"data":780,"type":226,"tunes":782},"p-conclusion-1",{"text":781},"计算机使用代理已经足够有用。这正是评估问题发生变化的原因。挑战不再仅仅是代理能否点击完成一个工作流程。而是当干净的演示条件消失时，系统是否仍然可靠。",{},{"id":784,"data":785,"type":226,"tunes":787},"p-conclusion-2",{"text":786},"将一次成功的运行视为能力的证据。然后测试可重复性、环境鲁棒性、长时程控制、状态感知、结果验证和安全目标处理。生产级计算机使用代理不是能完成演示的那个。而是其失败边界已知、可测量和可控的那个。",{},{"id":789,"data":790,"type":42,"tunes":792},"h-faq",{"text":791,"level":219},"常见问题",{},{"id":794,"data":795,"type":794,"tunes":822},"faq",{"items":796,"title":821},[797,801,805,809,813,817],{"id":798,"answer":799,"question":800},"faq1","不能。它只证明了在一条观察到的轨迹下的能力。生产可靠性需要在环境变化、长时间运行任务、状态变化、模糊性、恢复条件和有后果的操作中反复成功。","成功的计算机使用代理演示能证明生产可靠性吗？",{"id":802,"answer":803,"question":804},"faq2","基准测试可能使用更受控的环境、更短的任务、稳定的网络条件、更简单的应用组合或无法捕捉所有过程失败的结果标准。确切的有效性边界取决于每个基准测试。","为什么计算机使用基准测试看起来比实际表现好得多？",{"id":806,"answer":807,"question":808},"faq3","验证实际的外部结果。不要将代理的最终陈述或预期的点击序列视为目标系统已接受操作的证据。","计算机使用操作后最重要的可靠性检查是什么？",{"id":810,"answer":811,"question":812},"faq4","错误在多个操作中累积，约束被遗忘，外部状态变化，工作跨越多个应用，隐藏状态很重要，代理必须决定何时等待、询问、验证或恢复，而不是简单地继续操作。","为什么长时程计算机任务仍然困难？",{"id":814,"answer":815,"question":816},"faq5","重复干净任务，注入现实的环境故障，改变UI和状态，延长工作流程时程，引入模糊性，要求可观察的结果证明，测试高影响操作控制，并在模型或测试框架更改后重新运行测试套件。","在部署前我应该如何测试浏览器或桌面代理？",{"id":818,"answer":819,"question":820},"faq6","并非每个低风险操作都需要。确认要求应随后果、可逆性、权限和不确定性而扩展。高影响、外部可见或难以逆转的操作需要更强的控制。","计算机使用代理是否应始终需要人工确认？","计算机使用代理的可靠性",{},{"id":824,"data":825,"type":42,"tunes":827},"h-glossary",{"text":826,"level":219},"术语表",{},{"id":829,"data":830,"type":829,"tunes":857},"glossary",{"title":831,"entries":832},"关键可靠性术语",[833,837,841,845,849,853],{"term":834,"anchor":835,"definition":836},"计算机使用代理","computer-use-agent","一种AI代理，通过观察和操作（如点击、输入、滚动、文件操作或跨应用工作流程）与图形用户界面或计算机环境交互。",{"term":838,"anchor":839,"definition":840},"可重复性","repeatability","代理在重复运行中一致完成任务的程度，而不是仅在选定的轨迹上成功。",{"term":842,"anchor":843,"definition":844},"环境鲁棒性","environmental-robustness","在现实变化（如延迟、瞬时错误、UI变化、会话状态和意外页面条件）下保持正确行为的能力。",{"term":846,"anchor":847,"definition":848},"结果验证","outcome-verification","在操作后检查实际外部状态以确认预期结果发生，而不是依赖代理的自我报告。",{"term":850,"anchor":851,"definition":852},"盲目目标导向","blind-goal-directedness","一种失败模式，计算机使用代理在模糊性、不可行性、矛盾条件或需要停止并重新评估的原因下仍继续追求目标。",{"term":854,"anchor":855,"definition":856},"可靠性边界","reliability-boundary","一组条件，在这些条件下观察到的成功率或能力声明对于特定部署决策仍具有足够的代表性。",{},{"id":859,"data":860,"type":42,"tunes":862},"h-sources",{"text":861,"level":219},"主要来源和进一步阅读",{},{"id":864,"data":865,"type":871,"tunes":872},"src-openai-computer",{"link":866,"meta":867},"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Ftools-computer-use",{"image":868,"title":869,"description":870},{"url":554},"OpenAI — 计算机使用","当前开发者指南，关于隔离环境、将屏幕内容视为不可信、确认有后果的操作、限制运行和验证结果。","linkTool",{},{"id":874,"data":875,"type":871,"tunes":881},"src-openai-safety",{"link":876,"meta":877},"https:\u002F\u002Fopenai.com\u002Findex\u002Frunning-codex-safely\u002F",{"image":878,"title":879,"description":880},{"url":554},"OpenAI — 在OpenAI安全运行Codex","当前生产指南，关于技术边界、人工批准、遥测和对在真实系统上操作的代理的控制。",{},{"id":883,"data":884,"type":871,"tunes":890},"src-ms-warex",{"link":885,"meta":886},"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Fpublication\u002Fwarex-web-agent-reliability-evaluation-on-existing-benchmarks\u002F",{"image":887,"title":888,"description":889},{"url":554},"微软研究院 — WAREX","2026年评估显示，现实的网络不可靠性导致浏览器代理在现有基准测试上的任务成功率显著下降。",{},{"id":892,"data":893,"type":871,"tunes":899},"src-ms-verifier",{"link":894,"meta":895},"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Farticles\u002Fthe-art-of-building-verifiers-for-computer-use-agents\u002F",{"image":896,"title":897,"description":898},{"url":554},"微软研究院 — 为计算机使用代理构建验证器的艺术","2026年关于过程与结果评估、可控与不可控失败以及可靠轨迹验证的工作。",{},{"id":901,"data":902,"type":871,"tunes":908},"src-ms-weavebench",{"link":903,"meta":904},"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Fpublication\u002Fweavebench-a-long-horizon-real-world-benchmark-for-computer-use-agents-with-hybrid-interfaces\u002F",{"image":905,"title":906,"description":907},{"url":554},"微软研究院 — WeaveBench","2026年长时程基准测试，结合GUI、CLI和代码工作流程，显示当前代理与可靠现实世界完成之间的巨大差距。",{},{"id":910,"data":911,"type":871,"tunes":917},"src-osworld2",{"link":912,"meta":913},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2606.29537",{"image":914,"title":915,"description":916},{"url":554},"OSWorld 2.0 — 在长时程现实世界任务上对计算机使用代理进行基准测试","2026年基准测试，专注于现实的长时程计算机使用工作流程、隐藏状态和跨源推理。",{},{"id":919,"data":920,"type":871,"tunes":926},"src-ms-sentinel",{"link":921,"meta":922},"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Fpublication\u002Fsentinelbench-a-benchmark-for-long-running-monitoring-agents\u002F",{"image":923,"title":924,"description":925},{"url":554},"微软研究院 — SentinelBench","2026年基准测试，针对时间演化任务，代理必须监控环境并响应状态变化，而不是持续操作。",{},{"id":928,"data":929,"type":871,"tunes":935},"src-ms-blind",{"link":930,"meta":931},"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Fpublication\u002Fjust-do-it-computer-use-agents-exhibit-blind-goal-directedness\u002F",{"image":932,"title":933,"description":934},{"url":554},"微软研究院 — 就去做吧！？计算机使用代理表现出盲目目标导向","ICLR 2026研究，关于代理继续追求模糊、矛盾或不可行的目标。",{},"2.31","计算机使用代理如今能够完成令人印象深刻的浏览器和桌面工作流程，但一次成功的运行证明的是能力——而非可靠性。本文展示了如何测试可重复性、环境鲁棒性、长时程控制、状态感知、结果验证以及安全的目标处理。","\u002Fuploads\u002F2026\u002F09\u002Fcomputer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system-1790352854690-75qnrg.webp","computer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system-1790352854690-75qnrg","PUBLISHED","2026-09-25T12:13:00.000Z","2026-09-25T16:13:28.344Z","2026-09-25T19:19:11.089Z",{"en":945,"de":946,"sr":947,"es":948,"fr":949,"it":950,"ru":951,"zh":952},"\u002Fblog\u002Fcomputer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system","\u002Fde\u002Fblog\u002Fcomputer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system","\u002Fsr\u002Fblog\u002Fcomputer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system","\u002Fes\u002Fblog\u002Fcomputer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system","\u002Ffr\u002Fblog\u002Fcomputer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system","\u002Fit\u002Fblog\u002Fcomputer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system","\u002Fru\u002Fblog\u002Fcomputer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system","\u002Fzh\u002Fblog\u002Fcomputer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system",[954,958,962],{"id":955,"name":956,"slug":957},58,"评估与质量门槛","evaluation",{"id":959,"name":960,"slug":961},97,"在测试集上验证","verification",{"id":963,"name":964,"slug":965},73,"验证与对比","verification-and-diffing",{"id":967,"login":968,"email":969,"displayName":970},"20","rooth8233","aleksandar@stajic.de","Aleksandar Stajić",[972,1571],{"lang":973,"title":974,"content":975,"contentJson":976,"excerpt":1570},"en","Computer-Use Agents: Why a Successful Demo Can Still Be an Unreliable System","{\"time\":1790352872794,\"blocks\":[{\"id\":\"IYG9UPcPY0\",\"type\":\"tableOfContents\",\"data\":{\"title\":\"Contents\",\"minLevel\":2,\"maxLevel\":3},\"tunes\":{}},{\"id\":\"intro\",\"type\":\"paragraph\",\"data\":{\"text\":\"Computer-use agents can now click, type, browse, edit files, operate desktop applications, and complete impressive multi-step tasks. That makes successful demos easy to understand and easy to overinterpret. A single completed workflow shows that the agent can succeed under those conditions. It does not show how often it succeeds, how it behaves when the environment changes, whether it verifies the result, or how safely it acts when the goal becomes ambiguous.\"},\"tunes\":{}},{\"id\":\"direct\",\"type\":\"callout\",\"data\":{\"variant\":\"info\",\"title\":\"Direct answer\",\"body\":\"\u003Cstrong>A successful computer-use demo proves capability, not reliability.\u003C\u002Fstrong> Production reliability requires the agent to succeed repeatedly across environmental variation, recover from transient failures, preserve constraints over long horizons, detect hidden or changing state, verify the actual outcome, and stop or ask when the goal becomes ambiguous or unsafe. The correct production question is not “Can the agent do this task?” but “Under which conditions can we trust it to do this task repeatedly?”\"},\"tunes\":{}},{\"id\":\"freshness\",\"type\":\"callout\",\"data\":{\"variant\":\"warning\",\"title\":\"Fast-moving field\",\"body\":\"This article reflects computer-use agent research and platform guidance available on \u003Cstrong>25 September 2026\u003C\u002Fstrong>. Benchmark results are not directly comparable across different task sets, environments, models, step limits, judges, or harnesses. Treat every benchmark number together with its evaluation conditions.\"},\"tunes\":{}},{\"id\":\"model-note\",\"type\":\"callout\",\"data\":{\"variant\":\"note\",\"title\":\"The model used in this article\",\"body\":\"The Computer-Use Reliability Ladder and Demo-to-Production Stress Test below are practical evaluation models proposed here. They are not formal industry standards.\"},\"tunes\":{}},{\"id\":\"h-demo\",\"type\":\"header\",\"data\":{\"text\":\"Why the demo is the easiest possible reliability test\",\"level\":2},\"tunes\":{}},{\"id\":\"p-demo-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A demo normally shows one trajectory that worked. The environment is known, the task is selected in advance, the operator can restart after a failure, and the audience sees the successful path. Production systems face a distribution instead: different pages, network conditions, account states, pop-ups, latency, UI changes, hidden state, permissions, interruptions, and users who describe goals imperfectly.\"},\"tunes\":{}},{\"id\":\"p-demo-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"That distinction matters because computer-use agents operate through interfaces designed for humans rather than deterministic APIs. Their action loop depends on perception, state interpretation, planning, interaction timing, and environment response. Small changes can alter the trajectory even when the user goal is unchanged.\"},\"tunes\":{}},{\"id\":\"p-demo-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"Microsoft Research's WAREX work makes the problem explicit: benchmark agents that look capable in controlled settings lose substantial task success when realistic web instability is introduced. The failure is not necessarily “the model became less intelligent.” The environment stopped being deterministic.\"},\"tunes\":{}},{\"id\":\"h-claims\",\"type\":\"header\",\"data\":{\"text\":\"Capability, success rate, reliability, and safety are different claims\",\"level\":2},\"tunes\":{}},{\"id\":\"claims-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Claim\",\"What it actually establishes\",\"What it does not establish\"],[\"The agent completed the task once\",\"Capability under one observed trajectory\",\"Repeatability, robustness, safety, or generalization\"],[\"The agent scores highly on a benchmark\",\"Performance under that benchmark's task and evaluation conditions\",\"Equivalent production performance on different environments\"],[\"The agent usually reaches the goal\",\"Outcome success frequency\",\"Correct process, safe behaviour, or evidence that the result was verified\"],[\"The agent follows the intended process\",\"Trajectory quality under the evaluated rubric\",\"That the external environment actually accepted the final outcome\"],[\"The agent avoids unsafe actions in a test set\",\"Performance on represented safety cases\",\"Safety under every novel ambiguity, injection, or side effect\"]]},\"tunes\":{}},{\"id\":\"h-ladder\",\"type\":\"header\",\"data\":{\"text\":\"The Computer-Use Reliability Ladder\",\"level\":2},\"tunes\":{}},{\"id\":\"p-ladder-intro\",\"type\":\"paragraph\",\"data\":{\"text\":\"A useful way to evaluate computer-use systems is to move from one-off capability toward progressively harder reliability properties. Higher levels assume the lower levels but do not follow automatically from them.\"},\"tunes\":{}},{\"id\":\"ladder-flow\",\"type\":\"processFlow\",\"data\":{\"title\":\"Computer-Use Reliability Ladder\",\"orientation\":\"auto\",\"steps\":[{\"label\":\"1. Capability\",\"description\":\"Can the agent complete the task at least once under known conditions?\"},{\"label\":\"2. Repeatability\",\"description\":\"Can it complete the same task consistently across repeated trials?\"},{\"label\":\"3. Environmental robustness\",\"description\":\"Does it survive timing changes, network issues, pop-ups, UI variation, and small environmental perturbations?\"},{\"label\":\"4. Long-horizon control\",\"description\":\"Can it preserve goals, constraints, and progress across many steps, applications, and delayed events?\"},{\"label\":\"5. State awareness\",\"description\":\"Can it detect when the environment changed, when hidden state matters, or when an assumption is no longer valid?\"},{\"label\":\"6. Outcome verification\",\"description\":\"Does it verify that the intended result actually happened instead of trusting its own action sequence?\"},{\"label\":\"7. Safe goal handling\",\"description\":\"Can it stop, ask, refuse, or hand control back when the goal is ambiguous, infeasible, contradictory, or high impact?\"}]},\"tunes\":{}},{\"id\":\"h-capability\",\"type\":\"header\",\"data\":{\"text\":\"Level 1 — Capability: the demo question\",\"level\":3},\"tunes\":{}},{\"id\":\"p-capability-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Capability asks whether an agent can perform the task at all. This is valuable. Computer-use systems have advanced rapidly, and modern agents can complete workflows that older systems could not execute reliably.\"},\"tunes\":{}},{\"id\":\"p-capability-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"But capability is a weak deployment criterion. One successful run does not tell you whether the agent succeeds 95% of the time or 30% of the time, whether failures are harmless or destructive, or whether success depends on a lucky page state.\"},\"tunes\":{}},{\"id\":\"h-repeatability\",\"type\":\"header\",\"data\":{\"text\":\"Level 2 — Repeatability: does the same task stay solved?\",\"level\":3},\"tunes\":{}},{\"id\":\"p-repeat-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Computer-use trajectories are stochastic. Model outputs vary, pages load at different speeds, visual states change, and long workflows create many branching opportunities. A production test should therefore run the same task multiple times rather than treating one passing trace as representative.\"},\"tunes\":{}},{\"id\":\"p-repeat-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Measure not only the average success rate but also the distribution of failure modes: wrong click, premature termination, missed confirmation, incorrect field, duplicate action, navigation loop, stale-state assumption, and false success report.\"},\"tunes\":{}},{\"id\":\"h-robustness\",\"type\":\"header\",\"data\":{\"text\":\"Level 3 — Environmental robustness: what happens when the web behaves like the web?\",\"level\":3},\"tunes\":{}},{\"id\":\"p-robust-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Real websites are not benchmark fixtures. Requests fail, elements load late, sessions expire, pages change, consent banners appear, servers return errors, and network conditions fluctuate.\"},\"tunes\":{}},{\"id\":\"p-robust-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"WAREX evaluates this gap by injecting realistic web unreliability into existing benchmark environments and reports significant drops in task success. This is a critical production insight: a benchmark can measure task competence while under-measuring recovery from environmental instability.\"},\"tunes\":{}},{\"id\":\"robust-tip\",\"type\":\"callout\",\"data\":{\"variant\":\"tip\",\"title\":\"Reliability test\",\"body\":\"Inject delays, transient HTTP failures, stale page state, modal dialogs, session expiration, duplicate responses, and controlled UI variation. If the agent only works on the clean path, it is a demo-capable system, not a production-reliable one.\"},\"tunes\":{}},{\"id\":\"h-long\",\"type\":\"header\",\"data\":{\"text\":\"Level 4 — Long-horizon control: success changes when the task becomes real work\",\"level\":3},\"tunes\":{}},{\"id\":\"p-long-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Short tasks hide a class of failures that appear only after dozens or hundreds of actions: forgotten constraints, duplicated work, premature completion, missed state changes, cross-application inconsistencies, and accumulated small errors.\"},\"tunes\":{}},{\"id\":\"p-long-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"OSWorld 2.0 was designed specifically around long-horizon real-world workflows. Its tasks take human users a median of roughly 1.6 hours and require many more tool calls than earlier computer-use benchmarks. Under its primary completion metric, even the strongest evaluated systems remain far from complete task reliability.\"},\"tunes\":{}},{\"id\":\"p-long-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"WeaveBench reaches a similar conclusion from another angle. It evaluates hybrid GUI, CLI and code workflows and reports that the best evaluated model-runtime pairing passes only 41.2% of tasks. The important result is not one leaderboard number; it is that realistic cross-interface orchestration exposes failures hidden by simpler single-interface tasks.\"},\"tunes\":{}},{\"id\":\"h-state\",\"type\":\"header\",\"data\":{\"text\":\"Level 5 — State awareness: the environment can change underneath the plan\",\"level\":3},\"tunes\":{}},{\"id\":\"p-state-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Long-running tasks often depend on hidden or changing state: an email arrives, a calendar changes, a form is submitted, a background process finishes, a browser session expires, a user modifies a file, or an external system changes availability.\"},\"tunes\":{}},{\"id\":\"p-state-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Microsoft's SentinelBench argues that many long-running tasks should not be solved through continuous action at all. The correct behaviour may be to monitor, wait for an external event, then act when the state changes. This is a different capability from clicking faster or planning more steps.\"},\"tunes\":{}},{\"id\":\"p-state-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"A reliable computer-use agent therefore needs to distinguish actionable now, waiting for state, state changed, and assumption invalidated.\"},\"tunes\":{}},{\"id\":\"h-verify\",\"type\":\"header\",\"data\":{\"text\":\"Level 6 — Outcome verification: did the action actually work?\",\"level\":3},\"tunes\":{}},{\"id\":\"p-verify-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"An agent can execute an apparently correct sequence and still fail the task. A button click may not register. A form may reject hidden validation. A file may save to the wrong directory. A purchase may remain unconfirmed. A site may display a success-looking screen while the underlying operation failed.\"},\"tunes\":{}},{\"id\":\"p-verify-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"OpenAI's current computer-use guidance explicitly recommends bounding and verifying the run instead of relying only on the model's final answer. Microsoft Research's work on computer-use verifiers reaches the same conclusion from evaluation: process and outcome need to be judged separately.\"},\"tunes\":{}},{\"id\":\"p-verify-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The Universal Verifier research reports that earlier verifier setups can produce high false-positive rates, while stronger rubric design and explicit separation of process, outcome, controllable failures, and uncontrollable failures substantially improve agreement with human labels.\"},\"tunes\":{}},{\"id\":\"h-safe-goal\",\"type\":\"header\",\"data\":{\"text\":\"Level 7 — Safe goal handling: the agent must know when not to continue\",\"level\":3},\"tunes\":{}},{\"id\":\"p-safe-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Computer-use agents are optimized to complete goals, but goal persistence can itself become a failure mode. An ambiguous request, impossible condition, contradictory instruction, suspicious webpage, or changed environment may require clarification or stopping rather than more action.\"},\"tunes\":{}},{\"id\":\"p-safe-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The BLIND-ACT benchmark studies this problem as Blind Goal-Directedness. Across the systems evaluated in that work, agents frequently continued pursuing tasks despite ambiguity, infeasibility, conflicting context, or other reasons to reconsider. The authors identify patterns such as execution-first bias and request primacy.\"},\"tunes\":{}},{\"id\":\"p-safe-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"This failure class matters because a highly capable agent can make a bad situation worse faster. Reliability therefore includes a policy for when not to act.\"},\"tunes\":{}},{\"id\":\"h-stress\",\"type\":\"header\",\"data\":{\"text\":\"The Demo-to-Production Stress Test\",\"level\":2},\"tunes\":{}},{\"id\":\"p-stress-intro\",\"type\":\"paragraph\",\"data\":{\"text\":\"Before deploying a computer-use workflow, take the successful demo and systematically remove the assumptions that made it easy.\"},\"tunes\":{}},{\"id\":\"stress-flow\",\"type\":\"processFlow\",\"data\":{\"title\":\"Demo-to-Production Stress Test\",\"orientation\":\"auto\",\"steps\":[{\"label\":\"1. Re-run the clean task\",\"description\":\"Establish repeatability over multiple trials before adding complexity.\"},{\"label\":\"2. Perturb the environment\",\"description\":\"Add latency, retries, pop-ups, page variation, stale sessions and temporary failures.\"},{\"label\":\"3. Extend the horizon\",\"description\":\"Turn the short demo into the full real workflow with intermediate state, multiple applications and delayed steps.\"},{\"label\":\"4. Change hidden state\",\"description\":\"Modify account, file, task or external state after the agent has formed a plan and test whether it detects the change.\"},{\"label\":\"5. Inject ambiguity\",\"description\":\"Remove one important assumption and test whether the agent asks instead of guessing.\"},{\"label\":\"6. Inject a controlled contradiction\",\"description\":\"Present old and new state together and verify that authoritative current state wins.\"},{\"label\":\"7. Require outcome proof\",\"description\":\"Make task completion depend on verifiable final state, not the model's self-report.\"},{\"label\":\"8. Test consequential boundaries\",\"description\":\"Confirm that irreversible or sensitive actions trigger the expected approval, refusal or handoff.\"},{\"label\":\"9. Repeat after harness or model changes\",\"description\":\"Treat runtime upgrades as reliability changes that need regression testing.\"}]},\"tunes\":{}},{\"id\":\"h-benchmark-boundary\",\"type\":\"header\",\"data\":{\"text\":\"Benchmark success has a validity boundary\",\"level\":2},\"tunes\":{}},{\"id\":\"p-boundary-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A benchmark score is a conditional statement. It is valid for a particular model, harness, environment, task set, judge, tool interface, step budget, retry policy, date and evaluation method.\"},\"tunes\":{}},{\"id\":\"p-boundary-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The number becomes misleading when those conditions disappear from the claim. “Agent X scores 80%” is weaker than “Agent X scored 80% on benchmark Y under environment Z with judge J and step budget N.” The second statement preserves the boundary that tells you whether the number transfers to your application.\"},\"tunes\":{}},{\"id\":\"ref-avb\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers\",\"title\":\"The Answer Validity Boundary: The Missing Layer Between Relevance and Reliable AI Answers\",\"excerpt\":\"A framework for making explicit the conditions under which an AI claim remains valid and what changes require restriction, recalculation, or abandonment.\",\"ctaLabel\":\"Read the Answer Validity Boundary\"},\"tunes\":{}},{\"id\":\"h-process-outcome\",\"type\":\"header\",\"data\":{\"text\":\"Process success and outcome success must be scored separately\",\"level\":2},\"tunes\":{}},{\"id\":\"process-outcome-comparison\",\"type\":\"comparison\",\"data\":{\"title\":\"Four possible outcomes of one computer-use run\",\"layout\":\"table\",\"columns\":[{\"id\":\"process\",\"label\":\"Process\"},{\"id\":\"outcome\",\"label\":\"Outcome\"},{\"id\":\"interpretation\",\"label\":\"Interpretation\"}],\"rows\":[{\"id\":\"good-good\",\"label\":\"Correct process \u002F correct outcome\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"bad-good\",\"label\":\"Wrong process \u002F correct outcome\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"good-bad\",\"label\":\"Correct process \u002F wrong outcome\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"bad-bad\",\"label\":\"Wrong process \u002F wrong outcome\",\"values\":[\"\",\"\",\"\"]}]},\"tunes\":{}},{\"id\":\"p-process-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"WeaveBench reports that outcome-only grading can materially overestimate computer-use performance because an agent may produce an apparently successful artifact through a shortcut or fabricated evidence. The verifier must inspect the trajectory and deliverables, not merely the final claim.\"},\"tunes\":{}},{\"id\":\"h-distribution\",\"type\":\"header\",\"data\":{\"text\":\"Production reliability is a distribution, not a single pass rate\",\"level\":2},\"tunes\":{}},{\"id\":\"p-dist-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A useful production evaluation samples the dimensions that actually vary in your environment. For a browser workflow, that might include account age, locale, viewport, page version, network quality, authentication state, existing cart state, cookies, pop-ups, user permissions and whether a human interrupts the run.\"},\"tunes\":{}},{\"id\":\"distribution-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Dimension\",\"Example variation\",\"Why it matters\"],[\"Environment\",\"Fast vs slow network, transient failures, page timing\",\"Tests recovery and waiting behaviour\"],[\"UI\",\"Different viewport, modal, reordered element, minor redesign\",\"Tests brittle visual\u002Faction assumptions\"],[\"State\",\"Logged in\u002Fout, empty\u002Fnon-empty cart, existing file, changed permissions\",\"Tests hidden-state reasoning\"],[\"Task horizon\",\"5 steps vs 50+ steps, one app vs several apps\",\"Tests accumulated trajectory error\"],[\"Ambiguity\",\"Missing preference or incomplete user instruction\",\"Tests whether the agent asks instead of guesses\"],[\"Consequence\",\"Read-only vs purchase\u002Fsend\u002Fdelete\u002Fchange\",\"Tests confirmation and authorization controls\"],[\"Adversarial content\",\"Prompt injection or misleading page text\",\"Tests instruction hierarchy and containment\"],[\"Model \u002F harness version\",\"Runtime upgrade\",\"Tests regression from system-level changes\"]]},\"tunes\":{}},{\"id\":\"h-budget\",\"type\":\"header\",\"data\":{\"text\":\"Reliability needs a failure budget, not perfection\",\"level\":2},\"tunes\":{}},{\"id\":\"p-budget-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"No production system is perfectly reliable. The useful engineering question is which failures are acceptable, detectable and recoverable. A failed attempt to sort a local folder is not equivalent to sending the wrong email, purchasing the wrong product or changing an account setting.\"},\"tunes\":{}},{\"id\":\"p-budget-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Classify actions by consequence and reversibility. Low-impact reversible actions can tolerate more autonomy. High-impact, externally visible or hard-to-reverse actions need stronger confirmation, state verification, authorization and post-action checks.\"},\"tunes\":{}},{\"id\":\"h-matrix\",\"type\":\"header\",\"data\":{\"text\":\"A practical computer-use reliability matrix\",\"level\":2},\"tunes\":{}},{\"id\":\"control-matrix\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Action class\",\"Example\",\"Recommended control\"],[\"Read \u002F inspect\",\"Open pages, read files, gather information\",\"Bound scope, log sources, tolerate recoverable navigation errors\"],[\"Reversible local change\",\"Edit draft file, reorganize temporary workspace\",\"Checkpoint or version before change; verify result\"],[\"External communication\",\"Send email, publish content, submit form\",\"User confirmation or explicit delegated authority; verify accepted state\"],[\"Financial \u002F transactional\",\"Purchase, checkout, paid subscription\",\"Strict mandate, amount\u002Fmerchant constraints, final confirmation and receipt verification\"],[\"Destructive \u002F privilege-changing\",\"Delete data, change permissions, revoke access\",\"Narrow authorization, explicit confirmation, reversible path where possible, post-action audit\"]]},\"tunes\":{}},{\"id\":\"h-log\",\"type\":\"header\",\"data\":{\"text\":\"What to log for a computer-use failure\",\"level\":2},\"tunes\":{}},{\"id\":\"log-list\",\"type\":\"list\",\"data\":{\"style\":\"unordered\",\"meta\":{},\"items\":[\"User goal and explicit constraints.\",\"Model and harness version.\",\"Environment and application versions.\",\"Screenshots or structured observations relevant to the failure.\",\"Actions taken with timestamps.\",\"Tool, click, keyboard and navigation results.\",\"State transitions and waiting periods.\",\"Approval, refusal or handoff events.\",\"External errors and network failures.\",\"Final observable environment state.\",\"The agent's reported outcome.\",\"Verifier result and whether the failure was controllable by the agent.\"]},\"tunes\":{}},{\"id\":\"p-log-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The crucial comparison is between reported success and observable success. A system that cannot distinguish those two will eventually accumulate false positives in production.\"},\"tunes\":{}},{\"id\":\"ref-reliability\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough\",\"title\":\"AI Agent Reliability: Why the Final Answer Is Not Enough\",\"excerpt\":\"A broader reliability model for evaluating agent trajectories, tool use and intermediate decisions instead of accepting the final answer as proof that the system worked correctly.\",\"ctaLabel\":\"Read the agent reliability article\"},\"tunes\":{}},{\"id\":\"h-security\",\"type\":\"header\",\"data\":{\"text\":\"Security is part of reliability for computer-use agents\",\"level\":2},\"tunes\":{}},{\"id\":\"p-sec-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Computer-use agents do not merely read untrusted content; they can act after reading it. That turns prompt injection, malicious page content and phishing into execution-path risks.\"},\"tunes\":{}},{\"id\":\"p-sec-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"OpenAI's current computer-use guidance recommends isolating the environment, allow-listing sites and actions, treating screen content as untrusted, confirming consequential actions, bounding the run and verifying the actual outcome. ChatGPT agent similarly uses confirmations, prompt-injection monitoring and supervised modes for sensitive contexts.\"},\"tunes\":{}},{\"id\":\"p-sec-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The architecture principle is broader than any one provider: content observed by the agent must not be allowed to redefine the user's authority. A webpage can provide data. It cannot grant permission to send data elsewhere, purchase something, change credentials or override the task boundary.\"},\"tunes\":{}},{\"id\":\"h-change\",\"type\":\"header\",\"data\":{\"text\":\"What would change this answer?\",\"level\":2},\"tunes\":{}},{\"id\":\"p-change-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The reliability gap would narrow if computer-use models became robust to long horizons, dynamic state, UI variation, environmental failures and ambiguous goals across representative production distributions. Better native state APIs, standardized machine-readable interfaces and stronger verifier infrastructure could also reduce the amount of fragile GUI interaction required.\"},\"tunes\":{}},{\"id\":\"p-change-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The deployment threshold also changes with task consequence. A 70% success rate can be useful for a supervised low-risk research task and unacceptable for an autonomous financial or destructive workflow. Reliability must therefore be evaluated against the cost of each failure class, not one universal pass-rate threshold.\"},\"tunes\":{}},{\"id\":\"h-limitations\",\"type\":\"header\",\"data\":{\"text\":\"Limitations\",\"level\":2},\"tunes\":{}},{\"id\":\"p-limit-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The cited benchmarks evaluate different environments and should not be ranked against one another as if they measured the same thing. WAREX stresses web unreliability; WeaveBench targets hybrid long-horizon work; OSWorld 2.0 targets realistic long workflows; BLIND-ACT focuses on goal handling under ambiguity and infeasibility.\"},\"tunes\":{}},{\"id\":\"p-limit-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Benchmark results also age quickly. Model, harness and verifier improvements can materially change scores within months. The durable lesson is therefore the evaluation method: vary conditions, separate process from outcome, verify external state, and preserve the boundary around each performance claim.\"},\"tunes\":{}},{\"id\":\"h-conclusion\",\"type\":\"header\",\"data\":{\"text\":\"Conclusion\",\"level\":2},\"tunes\":{}},{\"id\":\"p-conclusion-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Computer-use agents are already capable enough to be useful. That is exactly why the evaluation question has changed. The challenge is no longer only whether an agent can click through a workflow. It is whether the system remains dependable when the clean demo conditions disappear.\"},\"tunes\":{}},{\"id\":\"p-conclusion-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Treat one successful run as evidence of capability. Then test repeatability, environmental robustness, long-horizon control, state awareness, outcome verification and safe goal handling. A production computer-use agent is not the one that can complete the demo. It is the one whose failure boundaries are known, measured and controlled.\"},\"tunes\":{}},{\"id\":\"h-faq\",\"type\":\"header\",\"data\":{\"text\":\"FAQ\",\"level\":2},\"tunes\":{}},{\"id\":\"faq\",\"type\":\"faq\",\"data\":{\"title\":\"Computer-use agent reliability\",\"items\":[{\"id\":\"faq1\",\"question\":\"Does a successful computer-use agent demo prove production reliability?\",\"answer\":\"No. It proves capability under one observed trajectory. Production reliability requires repeated success across environmental variation, long-running tasks, changing state, ambiguity, recovery conditions and consequential actions.\"},{\"id\":\"faq2\",\"question\":\"Why can computer-use benchmarks look much better than real-world performance?\",\"answer\":\"Benchmarks can use more controlled environments, shorter tasks, stable network conditions, simpler application combinations or outcome criteria that do not capture all process failures. The exact validity boundary depends on each benchmark.\"},{\"id\":\"faq3\",\"question\":\"What is the most important reliability check after a computer-use action?\",\"answer\":\"Verify the actual external outcome. Do not treat the agent's final statement or intended click sequence as proof that the target system accepted the operation.\"},{\"id\":\"faq4\",\"question\":\"Why do long-horizon computer tasks remain difficult?\",\"answer\":\"Errors accumulate across many actions, constraints are forgotten, external state changes, work spans multiple applications, hidden state matters, and the agent must decide when to wait, ask, verify or recover rather than simply continue acting.\"},{\"id\":\"faq5\",\"question\":\"How should I test a browser or desktop agent before deployment?\",\"answer\":\"Repeat clean tasks, inject realistic environmental failures, vary UI and state, extend the workflow horizon, introduce ambiguity, require observable outcome proof, test high-impact action controls and rerun the suite after model or harness changes.\"},{\"id\":\"faq6\",\"question\":\"Should computer-use agents always require human confirmation?\",\"answer\":\"Not for every low-risk action. Confirmation requirements should scale with consequence, reversibility, authority and uncertainty. High-impact, externally visible or difficult-to-reverse actions need stronger controls.\"}]},\"tunes\":{}},{\"id\":\"h-glossary\",\"type\":\"header\",\"data\":{\"text\":\"Glossary\",\"level\":2},\"tunes\":{}},{\"id\":\"glossary\",\"type\":\"glossary\",\"data\":{\"title\":\"Key reliability terms\",\"entries\":[{\"term\":\"Computer-use agent\",\"definition\":\"An AI agent that interacts with graphical user interfaces or computer environments through observations and actions such as clicking, typing, scrolling, file operations or cross-application workflows.\",\"anchor\":\"computer-use-agent\"},{\"term\":\"Repeatability\",\"definition\":\"The degree to which an agent can complete the same task consistently across repeated runs rather than succeeding only on selected trajectories.\",\"anchor\":\"repeatability\"},{\"term\":\"Environmental robustness\",\"definition\":\"The ability to preserve correct behaviour despite realistic variation such as latency, transient errors, UI changes, session state and unexpected page conditions.\",\"anchor\":\"environmental-robustness\"},{\"term\":\"Outcome verification\",\"definition\":\"Checking the actual external state after an action to confirm that the intended result occurred instead of relying on the agent's self-report.\",\"anchor\":\"outcome-verification\"},{\"term\":\"Blind Goal-Directedness\",\"definition\":\"A failure pattern in which a computer-use agent continues pursuing a goal despite ambiguity, infeasibility, contradictory conditions or reasons to stop and reassess.\",\"anchor\":\"blind-goal-directedness\"},{\"term\":\"Reliability boundary\",\"definition\":\"The set of conditions under which an observed success rate or capability claim remains representative enough for a specific deployment decision.\",\"anchor\":\"reliability-boundary\"}]},\"tunes\":{}},{\"id\":\"h-sources\",\"type\":\"header\",\"data\":{\"text\":\"Primary sources and further reading\",\"level\":2},\"tunes\":{}},{\"id\":\"src-openai-computer\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Ftools-computer-use\",\"meta\":{\"title\":\"OpenAI — Computer use\",\"description\":\"Current developer guidance on isolating environments, treating screen content as untrusted, confirming consequential actions, bounding runs and verifying outcomes.\",\"image\":{\"url\":\"\"}}},\"tunes\":{}},{\"id\":\"src-openai-safety\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fopenai.com\u002Findex\u002Frunning-codex-safely\u002F\",\"meta\":{\"title\":\"OpenAI — Running Codex safely at OpenAI\",\"description\":\"Current production guidance on technical boundaries, human approval, telemetry and control for agents that act on real systems.\",\"image\":{\"url\":\"\"}}},\"tunes\":{}},{\"id\":\"src-ms-warex\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Fpublication\u002Fwarex-web-agent-reliability-evaluation-on-existing-benchmarks\u002F\",\"meta\":{\"title\":\"Microsoft Research — WAREX\",\"description\":\"2026 evaluation showing that realistic web unreliability causes significant drops in browser-agent task success on existing benchmarks.\",\"image\":{\"url\":\"\"}}},\"tunes\":{}},{\"id\":\"src-ms-verifier\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Farticles\u002Fthe-art-of-building-verifiers-for-computer-use-agents\u002F\",\"meta\":{\"title\":\"Microsoft Research — The Art of Building Verifiers for Computer Use Agents\",\"description\":\"2026 work on process versus outcome evaluation, controllable versus uncontrollable failures and reliable trajectory verification.\",\"image\":{\"url\":\"\"}}},\"tunes\":{}},{\"id\":\"src-ms-weavebench\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Fpublication\u002Fweavebench-a-long-horizon-real-world-benchmark-for-computer-use-agents-with-hybrid-interfaces\u002F\",\"meta\":{\"title\":\"Microsoft Research — WeaveBench\",\"description\":\"2026 long-horizon benchmark combining GUI, CLI and code workflows and showing a substantial gap between current agents and reliable real-world completion.\",\"image\":{\"url\":\"\"}}},\"tunes\":{}},{\"id\":\"src-osworld2\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2606.29537\",\"meta\":{\"title\":\"OSWorld 2.0 — Benchmarking Computer Use Agents on Long-Horizon Real-World Tasks\",\"description\":\"2026 benchmark focused on realistic long-horizon computer-use workflows, hidden state and cross-source reasoning.\",\"image\":{\"url\":\"\"}}},\"tunes\":{}},{\"id\":\"src-ms-sentinel\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Fpublication\u002Fsentinelbench-a-benchmark-for-long-running-monitoring-agents\u002F\",\"meta\":{\"title\":\"Microsoft Research — SentinelBench\",\"description\":\"2026 benchmark for time-evolving tasks where agents must monitor environments and respond to state changes rather than continuously act.\",\"image\":{\"url\":\"\"}}},\"tunes\":{}},{\"id\":\"src-ms-blind\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Fpublication\u002Fjust-do-it-computer-use-agents-exhibit-blind-goal-directedness\u002F\",\"meta\":{\"title\":\"Microsoft Research — Just Do It!? Computer-Use Agents Exhibit Blind Goal-Directedness\",\"description\":\"ICLR 2026 research on agents continuing to pursue ambiguous, contradictory or infeasible goals.\",\"image\":{\"url\":\"\"}}},\"tunes\":{}}],\"version\":\"2.31.6\"}",{"time":977,"blocks":978,"version":1569},1790352872794,[979,983,987,992,997,1002,1006,1010,1014,1018,1022,1050,1054,1058,1084,1088,1092,1096,1100,1104,1108,1112,1116,1120,1125,1129,1133,1137,1141,1145,1149,1153,1157,1161,1165,1169,1173,1177,1181,1185,1189,1193,1197,1229,1233,1237,1241,1248,1252,1276,1280,1284,1288,1327,1331,1335,1339,1343,1371,1375,1392,1396,1403,1407,1411,1415,1419,1423,1427,1431,1435,1439,1443,1447,1451,1455,1459,1482,1486,1509,1513,1520,1527,1534,1541,1548,1555,1562],{"id":215,"data":980,"type":220,"tunes":982},{"title":981,"maxLevel":218,"minLevel":219},"Contents",{},{"id":223,"data":984,"type":226,"tunes":986},{"text":985},"Computer-use agents can now click, type, browse, edit files, operate desktop applications, and complete impressive multi-step tasks. That makes successful demos easy to understand and easy to overinterpret. A single completed workflow shows that the agent can succeed under those conditions. It does not show how often it succeeds, how it behaves when the environment changes, whether it verifies the result, or how safely it acts when the goal becomes ambiguous.",{},{"id":229,"data":988,"type":234,"tunes":991},{"body":989,"title":990,"variant":233},"\u003Cstrong>A successful computer-use demo proves capability, not reliability.\u003C\u002Fstrong> Production reliability requires the agent to succeed repeatedly across environmental variation, recover from transient failures, preserve constraints over long horizons, detect hidden or changing state, verify the actual outcome, and stop or ask when the goal becomes ambiguous or unsafe. The correct production question is not “Can the agent do this task?” but “Under which conditions can we trust it to do this task repeatedly?”","Direct answer",{},{"id":237,"data":993,"type":234,"tunes":996},{"body":994,"title":995,"variant":241},"This article reflects computer-use agent research and platform guidance available on \u003Cstrong>25 September 2026\u003C\u002Fstrong>. Benchmark results are not directly comparable across different task sets, environments, models, step limits, judges, or harnesses. Treat every benchmark number together with its evaluation conditions.","Fast-moving field",{},{"id":244,"data":998,"type":234,"tunes":1001},{"body":999,"title":1000,"variant":248},"The Computer-Use Reliability Ladder and Demo-to-Production Stress Test below are practical evaluation models proposed here. They are not formal industry standards.","The model used in this article",{},{"id":251,"data":1003,"type":42,"tunes":1005},{"text":1004,"level":219},"Why the demo is the easiest possible reliability test",{},{"id":256,"data":1007,"type":226,"tunes":1009},{"text":1008},"A demo normally shows one trajectory that worked. The environment is known, the task is selected in advance, the operator can restart after a failure, and the audience sees the successful path. Production systems face a distribution instead: different pages, network conditions, account states, pop-ups, latency, UI changes, hidden state, permissions, interruptions, and users who describe goals imperfectly.",{},{"id":261,"data":1011,"type":226,"tunes":1013},{"text":1012},"That distinction matters because computer-use agents operate through interfaces designed for humans rather than deterministic APIs. Their action loop depends on perception, state interpretation, planning, interaction timing, and environment response. Small changes can alter the trajectory even when the user goal is unchanged.",{},{"id":266,"data":1015,"type":226,"tunes":1017},{"text":1016},"Microsoft Research's WAREX work makes the problem explicit: benchmark agents that look capable in controlled settings lose substantial task success when realistic web instability is introduced. The failure is not necessarily “the model became less intelligent.” The environment stopped being deterministic.",{},{"id":271,"data":1019,"type":42,"tunes":1021},{"text":1020,"level":219},"Capability, success rate, reliability, and safety are different claims",{},{"id":276,"data":1023,"type":303,"tunes":1049},{"content":1024,"stretched":43,"withHeadings":14},[1025,1029,1033,1037,1041,1045],[1026,1027,1028],"Claim","What it actually establishes","What it does not establish",[1030,1031,1032],"The agent completed the task once","Capability under one observed trajectory","Repeatability, robustness, safety, or generalization",[1034,1035,1036],"The agent scores highly on a benchmark","Performance under that benchmark's task and evaluation conditions","Equivalent production performance on different environments",[1038,1039,1040],"The agent usually reaches the goal","Outcome success frequency","Correct process, safe behaviour, or evidence that the result was verified",[1042,1043,1044],"The agent follows the intended process","Trajectory quality under the evaluated rubric","That the external environment actually accepted the final outcome",[1046,1047,1048],"The agent avoids unsafe actions in a test set","Performance on represented safety cases","Safety under every novel ambiguity, injection, or side effect",{},{"id":306,"data":1051,"type":42,"tunes":1053},{"text":1052,"level":219},"The Computer-Use Reliability Ladder",{},{"id":311,"data":1055,"type":226,"tunes":1057},{"text":1056},"A useful way to evaluate computer-use systems is to move from one-off capability toward progressively harder reliability properties. Higher levels assume the lower levels but do not follow automatically from them.",{},{"id":316,"data":1059,"type":341,"tunes":1083},{"steps":1060,"title":1082,"orientation":340},[1061,1064,1067,1070,1073,1076,1079],{"label":1062,"description":1063},"1. Capability","Can the agent complete the task at least once under known conditions?",{"label":1065,"description":1066},"2. Repeatability","Can it complete the same task consistently across repeated trials?",{"label":1068,"description":1069},"3. Environmental robustness","Does it survive timing changes, network issues, pop-ups, UI variation, and small environmental perturbations?",{"label":1071,"description":1072},"4. Long-horizon control","Can it preserve goals, constraints, and progress across many steps, applications, and delayed events?",{"label":1074,"description":1075},"5. State awareness","Can it detect when the environment changed, when hidden state matters, or when an assumption is no longer valid?",{"label":1077,"description":1078},"6. Outcome verification","Does it verify that the intended result actually happened instead of trusting its own action sequence?",{"label":1080,"description":1081},"7. Safe goal handling","Can it stop, ask, refuse, or hand control back when the goal is ambiguous, infeasible, contradictory, or high impact?","Computer-Use Reliability Ladder",{},{"id":344,"data":1085,"type":42,"tunes":1087},{"text":1086,"level":218},"Level 1 — Capability: the demo question",{},{"id":349,"data":1089,"type":226,"tunes":1091},{"text":1090},"Capability asks whether an agent can perform the task at all. This is valuable. Computer-use systems have advanced rapidly, and modern agents can complete workflows that older systems could not execute reliably.",{},{"id":354,"data":1093,"type":226,"tunes":1095},{"text":1094},"But capability is a weak deployment criterion. One successful run does not tell you whether the agent succeeds 95% of the time or 30% of the time, whether failures are harmless or destructive, or whether success depends on a lucky page state.",{},{"id":359,"data":1097,"type":42,"tunes":1099},{"text":1098,"level":218},"Level 2 — Repeatability: does the same task stay solved?",{},{"id":364,"data":1101,"type":226,"tunes":1103},{"text":1102},"Computer-use trajectories are stochastic. Model outputs vary, pages load at different speeds, visual states change, and long workflows create many branching opportunities. A production test should therefore run the same task multiple times rather than treating one passing trace as representative.",{},{"id":369,"data":1105,"type":226,"tunes":1107},{"text":1106},"Measure not only the average success rate but also the distribution of failure modes: wrong click, premature termination, missed confirmation, incorrect field, duplicate action, navigation loop, stale-state assumption, and false success report.",{},{"id":374,"data":1109,"type":42,"tunes":1111},{"text":1110,"level":218},"Level 3 — Environmental robustness: what happens when the web behaves like the web?",{},{"id":379,"data":1113,"type":226,"tunes":1115},{"text":1114},"Real websites are not benchmark fixtures. Requests fail, elements load late, sessions expire, pages change, consent banners appear, servers return errors, and network conditions fluctuate.",{},{"id":384,"data":1117,"type":226,"tunes":1119},{"text":1118},"WAREX evaluates this gap by injecting realistic web unreliability into existing benchmark environments and reports significant drops in task success. This is a critical production insight: a benchmark can measure task competence while under-measuring recovery from environmental instability.",{},{"id":389,"data":1121,"type":234,"tunes":1124},{"body":1122,"title":1123,"variant":393},"Inject delays, transient HTTP failures, stale page state, modal dialogs, session expiration, duplicate responses, and controlled UI variation. If the agent only works on the clean path, it is a demo-capable system, not a production-reliable one.","Reliability test",{},{"id":396,"data":1126,"type":42,"tunes":1128},{"text":1127,"level":218},"Level 4 — Long-horizon control: success changes when the task becomes real work",{},{"id":401,"data":1130,"type":226,"tunes":1132},{"text":1131},"Short tasks hide a class of failures that appear only after dozens or hundreds of actions: forgotten constraints, duplicated work, premature completion, missed state changes, cross-application inconsistencies, and accumulated small errors.",{},{"id":406,"data":1134,"type":226,"tunes":1136},{"text":1135},"OSWorld 2.0 was designed specifically around long-horizon real-world workflows. Its tasks take human users a median of roughly 1.6 hours and require many more tool calls than earlier computer-use benchmarks. Under its primary completion metric, even the strongest evaluated systems remain far from complete task reliability.",{},{"id":411,"data":1138,"type":226,"tunes":1140},{"text":1139},"WeaveBench reaches a similar conclusion from another angle. It evaluates hybrid GUI, CLI and code workflows and reports that the best evaluated model-runtime pairing passes only 41.2% of tasks. The important result is not one leaderboard number; it is that realistic cross-interface orchestration exposes failures hidden by simpler single-interface tasks.",{},{"id":416,"data":1142,"type":42,"tunes":1144},{"text":1143,"level":218},"Level 5 — State awareness: the environment can change underneath the plan",{},{"id":421,"data":1146,"type":226,"tunes":1148},{"text":1147},"Long-running tasks often depend on hidden or changing state: an email arrives, a calendar changes, a form is submitted, a background process finishes, a browser session expires, a user modifies a file, or an external system changes availability.",{},{"id":426,"data":1150,"type":226,"tunes":1152},{"text":1151},"Microsoft's SentinelBench argues that many long-running tasks should not be solved through continuous action at all. The correct behaviour may be to monitor, wait for an external event, then act when the state changes. This is a different capability from clicking faster or planning more steps.",{},{"id":431,"data":1154,"type":226,"tunes":1156},{"text":1155},"A reliable computer-use agent therefore needs to distinguish actionable now, waiting for state, state changed, and assumption invalidated.",{},{"id":436,"data":1158,"type":42,"tunes":1160},{"text":1159,"level":218},"Level 6 — Outcome verification: did the action actually work?",{},{"id":441,"data":1162,"type":226,"tunes":1164},{"text":1163},"An agent can execute an apparently correct sequence and still fail the task. A button click may not register. A form may reject hidden validation. A file may save to the wrong directory. A purchase may remain unconfirmed. A site may display a success-looking screen while the underlying operation failed.",{},{"id":446,"data":1166,"type":226,"tunes":1168},{"text":1167},"OpenAI's current computer-use guidance explicitly recommends bounding and verifying the run instead of relying only on the model's final answer. Microsoft Research's work on computer-use verifiers reaches the same conclusion from evaluation: process and outcome need to be judged separately.",{},{"id":451,"data":1170,"type":226,"tunes":1172},{"text":1171},"The Universal Verifier research reports that earlier verifier setups can produce high false-positive rates, while stronger rubric design and explicit separation of process, outcome, controllable failures, and uncontrollable failures substantially improve agreement with human labels.",{},{"id":456,"data":1174,"type":42,"tunes":1176},{"text":1175,"level":218},"Level 7 — Safe goal handling: the agent must know when not to continue",{},{"id":461,"data":1178,"type":226,"tunes":1180},{"text":1179},"Computer-use agents are optimized to complete goals, but goal persistence can itself become a failure mode. An ambiguous request, impossible condition, contradictory instruction, suspicious webpage, or changed environment may require clarification or stopping rather than more action.",{},{"id":466,"data":1182,"type":226,"tunes":1184},{"text":1183},"The BLIND-ACT benchmark studies this problem as Blind Goal-Directedness. Across the systems evaluated in that work, agents frequently continued pursuing tasks despite ambiguity, infeasibility, conflicting context, or other reasons to reconsider. The authors identify patterns such as execution-first bias and request primacy.",{},{"id":471,"data":1186,"type":226,"tunes":1188},{"text":1187},"This failure class matters because a highly capable agent can make a bad situation worse faster. Reliability therefore includes a policy for when not to act.",{},{"id":476,"data":1190,"type":42,"tunes":1192},{"text":1191,"level":219},"The Demo-to-Production Stress Test",{},{"id":481,"data":1194,"type":226,"tunes":1196},{"text":1195},"Before deploying a computer-use workflow, take the successful demo and systematically remove the assumptions that made it easy.",{},{"id":486,"data":1198,"type":341,"tunes":1228},{"steps":1199,"title":1227,"orientation":340},[1200,1203,1206,1209,1212,1215,1218,1221,1224],{"label":1201,"description":1202},"1. Re-run the clean task","Establish repeatability over multiple trials before adding complexity.",{"label":1204,"description":1205},"2. Perturb the environment","Add latency, retries, pop-ups, page variation, stale sessions and temporary failures.",{"label":1207,"description":1208},"3. Extend the horizon","Turn the short demo into the full real workflow with intermediate state, multiple applications and delayed steps.",{"label":1210,"description":1211},"4. Change hidden state","Modify account, file, task or external state after the agent has formed a plan and test whether it detects the change.",{"label":1213,"description":1214},"5. Inject ambiguity","Remove one important assumption and test whether the agent asks instead of guessing.",{"label":1216,"description":1217},"6. Inject a controlled contradiction","Present old and new state together and verify that authoritative current state wins.",{"label":1219,"description":1220},"7. Require outcome proof","Make task completion depend on verifiable final state, not the model's self-report.",{"label":1222,"description":1223},"8. Test consequential boundaries","Confirm that irreversible or sensitive actions trigger the expected approval, refusal or handoff.",{"label":1225,"description":1226},"9. Repeat after harness or model changes","Treat runtime upgrades as reliability changes that need regression testing.","Demo-to-Production Stress Test",{},{"id":518,"data":1230,"type":42,"tunes":1232},{"text":1231,"level":219},"Benchmark success has a validity boundary",{},{"id":523,"data":1234,"type":226,"tunes":1236},{"text":1235},"A benchmark score is a conditional statement. It is valid for a particular model, harness, environment, task set, judge, tool interface, step budget, retry policy, date and evaluation method.",{},{"id":528,"data":1238,"type":226,"tunes":1240},{"text":1239},"The number becomes misleading when those conditions disappear from the claim. “Agent X scores 80%” is weaker than “Agent X scored 80% on benchmark Y under environment Z with judge J and step budget N.” The second statement preserves the boundary that tells you whether the number transfers to your application.",{},{"id":533,"data":1242,"type":539,"tunes":1247},{"url":1243,"title":1244,"excerpt":1245,"ctaLabel":1246},"https:\u002F\u002Fstajic.de\u002Fblog\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers","The Answer Validity Boundary: The Missing Layer Between Relevance and Reliable AI Answers","A framework for making explicit the conditions under which an AI claim remains valid and what changes require restriction, recalculation, or abandonment.","Read the Answer Validity Boundary",{},{"id":542,"data":1249,"type":42,"tunes":1251},{"text":1250,"level":219},"Process success and outcome success must be scored separately",{},{"id":547,"data":1253,"type":578,"tunes":1275},{"rows":1254,"title":1267,"layout":303,"columns":1268},[1255,1258,1261,1264],{"id":551,"label":1256,"values":1257},"Correct process \u002F correct outcome",[554,554,554],{"id":556,"label":1259,"values":1260},"Wrong process \u002F correct outcome",[554,554,554],{"id":560,"label":1262,"values":1263},"Correct process \u002F wrong outcome",[554,554,554],{"id":564,"label":1265,"values":1266},"Wrong process \u002F wrong outcome",[554,554,554],"Four possible outcomes of one computer-use run",[1269,1271,1273],{"id":570,"label":1270},"Process",{"id":573,"label":1272},"Outcome",{"id":576,"label":1274},"Interpretation",{},{"id":581,"data":1277,"type":226,"tunes":1279},{"text":1278},"WeaveBench reports that outcome-only grading can materially overestimate computer-use performance because an agent may produce an apparently successful artifact through a shortcut or fabricated evidence. The verifier must inspect the trajectory and deliverables, not merely the final claim.",{},{"id":586,"data":1281,"type":42,"tunes":1283},{"text":1282,"level":219},"Production reliability is a distribution, not a single pass rate",{},{"id":591,"data":1285,"type":226,"tunes":1287},{"text":1286},"A useful production evaluation samples the dimensions that actually vary in your environment. For a browser workflow, that might include account age, locale, viewport, page version, network quality, authentication state, existing cart state, cookies, pop-ups, user permissions and whether a human interrupts the run.",{},{"id":596,"data":1289,"type":303,"tunes":1326},{"content":1290,"stretched":43,"withHeadings":14},[1291,1295,1299,1302,1306,1310,1314,1318,1322],[1292,1293,1294],"Dimension","Example variation","Why it matters",[1296,1297,1298],"Environment","Fast vs slow network, transient failures, page timing","Tests recovery and waiting behaviour",[608,1300,1301],"Different viewport, modal, reordered element, minor redesign","Tests brittle visual\u002Faction assumptions",[1303,1304,1305],"State","Logged in\u002Fout, empty\u002Fnon-empty cart, existing file, changed permissions","Tests hidden-state reasoning",[1307,1308,1309],"Task horizon","5 steps vs 50+ steps, one app vs several apps","Tests accumulated trajectory error",[1311,1312,1313],"Ambiguity","Missing preference or incomplete user instruction","Tests whether the agent asks instead of guesses",[1315,1316,1317],"Consequence","Read-only vs purchase\u002Fsend\u002Fdelete\u002Fchange","Tests confirmation and authorization controls",[1319,1320,1321],"Adversarial content","Prompt injection or misleading page text","Tests instruction hierarchy and containment",[1323,1324,1325],"Model \u002F harness version","Runtime upgrade","Tests regression from system-level changes",{},{"id":637,"data":1328,"type":42,"tunes":1330},{"text":1329,"level":219},"Reliability needs a failure budget, not perfection",{},{"id":642,"data":1332,"type":226,"tunes":1334},{"text":1333},"No production system is perfectly reliable. The useful engineering question is which failures are acceptable, detectable and recoverable. A failed attempt to sort a local folder is not equivalent to sending the wrong email, purchasing the wrong product or changing an account setting.",{},{"id":647,"data":1336,"type":226,"tunes":1338},{"text":1337},"Classify actions by consequence and reversibility. Low-impact reversible actions can tolerate more autonomy. High-impact, externally visible or hard-to-reverse actions need stronger confirmation, state verification, authorization and post-action checks.",{},{"id":652,"data":1340,"type":42,"tunes":1342},{"text":1341,"level":219},"A practical computer-use reliability matrix",{},{"id":657,"data":1344,"type":303,"tunes":1370},{"content":1345,"stretched":43,"withHeadings":14},[1346,1350,1354,1358,1362,1366],[1347,1348,1349],"Action class","Example","Recommended control",[1351,1352,1353],"Read \u002F inspect","Open pages, read files, gather information","Bound scope, log sources, tolerate recoverable navigation errors",[1355,1356,1357],"Reversible local change","Edit draft file, reorganize temporary workspace","Checkpoint or version before change; verify result",[1359,1360,1361],"External communication","Send email, publish content, submit form","User confirmation or explicit delegated authority; verify accepted state",[1363,1364,1365],"Financial \u002F transactional","Purchase, checkout, paid subscription","Strict mandate, amount\u002Fmerchant constraints, final confirmation and receipt verification",[1367,1368,1369],"Destructive \u002F privilege-changing","Delete data, change permissions, revoke access","Narrow authorization, explicit confirmation, reversible path where possible, post-action audit",{},{"id":686,"data":1372,"type":42,"tunes":1374},{"text":1373,"level":219},"What to log for a computer-use failure",{},{"id":691,"data":1376,"type":708,"tunes":1391},{"meta":1377,"items":1378,"style":707},{},[1379,1380,1381,1382,1383,1384,1385,1386,1387,1388,1389,1390],"User goal and explicit constraints.","Model and harness version.","Environment and application versions.","Screenshots or structured observations relevant to the failure.","Actions taken with timestamps.","Tool, click, keyboard and navigation results.","State transitions and waiting periods.","Approval, refusal or handoff events.","External errors and network failures.","Final observable environment state.","The agent's reported outcome.","Verifier result and whether the failure was controllable by the agent.",{},{"id":711,"data":1393,"type":226,"tunes":1395},{"text":1394},"The crucial comparison is between reported success and observable success. A system that cannot distinguish those two will eventually accumulate false positives in production.",{},{"id":716,"data":1397,"type":539,"tunes":1402},{"url":1398,"title":1399,"excerpt":1400,"ctaLabel":1401},"https:\u002F\u002Fstajic.de\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough","AI Agent Reliability: Why the Final Answer Is Not Enough","A broader reliability model for evaluating agent trajectories, tool use and intermediate decisions instead of accepting the final answer as proof that the system worked correctly.","Read the agent reliability article",{},{"id":724,"data":1404,"type":42,"tunes":1406},{"text":1405,"level":219},"Security is part of reliability for computer-use agents",{},{"id":729,"data":1408,"type":226,"tunes":1410},{"text":1409},"Computer-use agents do not merely read untrusted content; they can act after reading it. That turns prompt injection, malicious page content and phishing into execution-path risks.",{},{"id":734,"data":1412,"type":226,"tunes":1414},{"text":1413},"OpenAI's current computer-use guidance recommends isolating the environment, allow-listing sites and actions, treating screen content as untrusted, confirming consequential actions, bounding the run and verifying the actual outcome. ChatGPT agent similarly uses confirmations, prompt-injection monitoring and supervised modes for sensitive contexts.",{},{"id":739,"data":1416,"type":226,"tunes":1418},{"text":1417},"The architecture principle is broader than any one provider: content observed by the agent must not be allowed to redefine the user's authority. A webpage can provide data. It cannot grant permission to send data elsewhere, purchase something, change credentials or override the task boundary.",{},{"id":744,"data":1420,"type":42,"tunes":1422},{"text":1421,"level":219},"What would change this answer?",{},{"id":749,"data":1424,"type":226,"tunes":1426},{"text":1425},"The reliability gap would narrow if computer-use models became robust to long horizons, dynamic state, UI variation, environmental failures and ambiguous goals across representative production distributions. Better native state APIs, standardized machine-readable interfaces and stronger verifier infrastructure could also reduce the amount of fragile GUI interaction required.",{},{"id":754,"data":1428,"type":226,"tunes":1430},{"text":1429},"The deployment threshold also changes with task consequence. A 70% success rate can be useful for a supervised low-risk research task and unacceptable for an autonomous financial or destructive workflow. Reliability must therefore be evaluated against the cost of each failure class, not one universal pass-rate threshold.",{},{"id":759,"data":1432,"type":42,"tunes":1434},{"text":1433,"level":219},"Limitations",{},{"id":764,"data":1436,"type":226,"tunes":1438},{"text":1437},"The cited benchmarks evaluate different environments and should not be ranked against one another as if they measured the same thing. WAREX stresses web unreliability; WeaveBench targets hybrid long-horizon work; OSWorld 2.0 targets realistic long workflows; BLIND-ACT focuses on goal handling under ambiguity and infeasibility.",{},{"id":769,"data":1440,"type":226,"tunes":1442},{"text":1441},"Benchmark results also age quickly. Model, harness and verifier improvements can materially change scores within months. The durable lesson is therefore the evaluation method: vary conditions, separate process from outcome, verify external state, and preserve the boundary around each performance claim.",{},{"id":774,"data":1444,"type":42,"tunes":1446},{"text":1445,"level":219},"Conclusion",{},{"id":779,"data":1448,"type":226,"tunes":1450},{"text":1449},"Computer-use agents are already capable enough to be useful. That is exactly why the evaluation question has changed. The challenge is no longer only whether an agent can click through a workflow. It is whether the system remains dependable when the clean demo conditions disappear.",{},{"id":784,"data":1452,"type":226,"tunes":1454},{"text":1453},"Treat one successful run as evidence of capability. Then test repeatability, environmental robustness, long-horizon control, state awareness, outcome verification and safe goal handling. A production computer-use agent is not the one that can complete the demo. It is the one whose failure boundaries are known, measured and controlled.",{},{"id":789,"data":1456,"type":42,"tunes":1458},{"text":1457,"level":219},"FAQ",{},{"id":794,"data":1460,"type":794,"tunes":1481},{"items":1461,"title":1480},[1462,1465,1468,1471,1474,1477],{"id":798,"answer":1463,"question":1464},"No. It proves capability under one observed trajectory. Production reliability requires repeated success across environmental variation, long-running tasks, changing state, ambiguity, recovery conditions and consequential actions.","Does a successful computer-use agent demo prove production reliability?",{"id":802,"answer":1466,"question":1467},"Benchmarks can use more controlled environments, shorter tasks, stable network conditions, simpler application combinations or outcome criteria that do not capture all process failures. The exact validity boundary depends on each benchmark.","Why can computer-use benchmarks look much better than real-world performance?",{"id":806,"answer":1469,"question":1470},"Verify the actual external outcome. Do not treat the agent's final statement or intended click sequence as proof that the target system accepted the operation.","What is the most important reliability check after a computer-use action?",{"id":810,"answer":1472,"question":1473},"Errors accumulate across many actions, constraints are forgotten, external state changes, work spans multiple applications, hidden state matters, and the agent must decide when to wait, ask, verify or recover rather than simply continue acting.","Why do long-horizon computer tasks remain difficult?",{"id":814,"answer":1475,"question":1476},"Repeat clean tasks, inject realistic environmental failures, vary UI and state, extend the workflow horizon, introduce ambiguity, require observable outcome proof, test high-impact action controls and rerun the suite after model or harness changes.","How should I test a browser or desktop agent before deployment?",{"id":818,"answer":1478,"question":1479},"Not for every low-risk action. Confirmation requirements should scale with consequence, reversibility, authority and uncertainty. High-impact, externally visible or difficult-to-reverse actions need stronger controls.","Should computer-use agents always require human confirmation?","Computer-use agent reliability",{},{"id":824,"data":1483,"type":42,"tunes":1485},{"text":1484,"level":219},"Glossary",{},{"id":829,"data":1487,"type":829,"tunes":1508},{"title":1488,"entries":1489},"Key reliability terms",[1490,1493,1496,1499,1502,1505],{"term":1491,"anchor":835,"definition":1492},"Computer-use agent","An AI agent that interacts with graphical user interfaces or computer environments through observations and actions such as clicking, typing, scrolling, file operations or cross-application workflows.",{"term":1494,"anchor":839,"definition":1495},"Repeatability","The degree to which an agent can complete the same task consistently across repeated runs rather than succeeding only on selected trajectories.",{"term":1497,"anchor":843,"definition":1498},"Environmental robustness","The ability to preserve correct behaviour despite realistic variation such as latency, transient errors, UI changes, session state and unexpected page conditions.",{"term":1500,"anchor":847,"definition":1501},"Outcome verification","Checking the actual external state after an action to confirm that the intended result occurred instead of relying on the agent's self-report.",{"term":1503,"anchor":851,"definition":1504},"Blind Goal-Directedness","A failure pattern in which a computer-use agent continues pursuing a goal despite ambiguity, infeasibility, contradictory conditions or reasons to stop and reassess.",{"term":1506,"anchor":855,"definition":1507},"Reliability boundary","The set of conditions under which an observed success rate or capability claim remains representative enough for a specific deployment decision.",{},{"id":859,"data":1510,"type":42,"tunes":1512},{"text":1511,"level":219},"Primary sources and further reading",{},{"id":864,"data":1514,"type":871,"tunes":1519},{"link":866,"meta":1515},{"image":1516,"title":1517,"description":1518},{"url":554},"OpenAI — Computer use","Current developer guidance on isolating environments, treating screen content as untrusted, confirming consequential actions, bounding runs and verifying outcomes.",{},{"id":874,"data":1521,"type":871,"tunes":1526},{"link":876,"meta":1522},{"image":1523,"title":1524,"description":1525},{"url":554},"OpenAI — Running Codex safely at OpenAI","Current production guidance on technical boundaries, human approval, telemetry and control for agents that act on real systems.",{},{"id":883,"data":1528,"type":871,"tunes":1533},{"link":885,"meta":1529},{"image":1530,"title":1531,"description":1532},{"url":554},"Microsoft Research — WAREX","2026 evaluation showing that realistic web unreliability causes significant drops in browser-agent task success on existing benchmarks.",{},{"id":892,"data":1535,"type":871,"tunes":1540},{"link":894,"meta":1536},{"image":1537,"title":1538,"description":1539},{"url":554},"Microsoft Research — The Art of Building Verifiers for Computer Use Agents","2026 work on process versus outcome evaluation, controllable versus uncontrollable failures and reliable trajectory verification.",{},{"id":901,"data":1542,"type":871,"tunes":1547},{"link":903,"meta":1543},{"image":1544,"title":1545,"description":1546},{"url":554},"Microsoft Research — WeaveBench","2026 long-horizon benchmark combining GUI, CLI and code workflows and showing a substantial gap between current agents and reliable real-world completion.",{},{"id":910,"data":1549,"type":871,"tunes":1554},{"link":912,"meta":1550},{"image":1551,"title":1552,"description":1553},{"url":554},"OSWorld 2.0 — Benchmarking Computer Use Agents on Long-Horizon Real-World Tasks","2026 benchmark focused on realistic long-horizon computer-use workflows, hidden state and cross-source reasoning.",{},{"id":919,"data":1556,"type":871,"tunes":1561},{"link":921,"meta":1557},{"image":1558,"title":1559,"description":1560},{"url":554},"Microsoft Research — SentinelBench","2026 benchmark for time-evolving tasks where agents must monitor environments and respond to state changes rather than continuously act.",{},{"id":928,"data":1563,"type":871,"tunes":1568},{"link":930,"meta":1564},{"image":1565,"title":1566,"description":1567},{"url":554},"Microsoft Research — Just Do It!? Computer-Use Agents Exhibit Blind Goal-Directedness","ICLR 2026 research on agents continuing to pursue ambiguous, contradictory or infeasible goals.",{},"2.31.6","Computer-use agents can now complete impressive browser and desktop workflows, but one successful run proves capability—not reliability. This article shows how to test repeatability, environmental robustness, long-horizon control, state awareness, outcome verification, and safe goal handling.",{"lang":7,"title":208,"content":210,"contentJson":1572,"excerpt":937},{"time":212,"blocks":1573,"version":936},[1574,1577,1580,1583,1586,1589,1592,1595,1598,1601,1604,1614,1617,1620,1631,1634,1637,1640,1643,1646,1649,1652,1655,1658,1661,1664,1667,1670,1673,1676,1679,1682,1685,1688,1691,1694,1697,1700,1703,1706,1709,1712,1715,1728,1731,1734,1737,1740,1743,1759,1762,1765,1768,1781,1784,1787,1790,1793,1803,1806,1811,1814,1817,1820,1823,1826,1829,1832,1835,1838,1841,1844,1847,1850,1853,1856,1859,1869,1872,1882,1885,1890,1895,1900,1905,1910,1915,1920],{"id":215,"data":1575,"type":220,"tunes":1576},{"title":217,"maxLevel":218,"minLevel":219},{},{"id":223,"data":1578,"type":226,"tunes":1579},{"text":225},{},{"id":229,"data":1581,"type":234,"tunes":1582},{"body":231,"title":232,"variant":233},{},{"id":237,"data":1584,"type":234,"tunes":1585},{"body":239,"title":240,"variant":241},{},{"id":244,"data":1587,"type":234,"tunes":1588},{"body":246,"title":247,"variant":248},{},{"id":251,"data":1590,"type":42,"tunes":1591},{"text":253,"level":219},{},{"id":256,"data":1593,"type":226,"tunes":1594},{"text":258},{},{"id":261,"data":1596,"type":226,"tunes":1597},{"text":263},{},{"id":266,"data":1599,"type":226,"tunes":1600},{"text":268},{},{"id":271,"data":1602,"type":42,"tunes":1603},{"text":273,"level":219},{},{"id":276,"data":1605,"type":303,"tunes":1613},{"content":1606,"stretched":43,"withHeadings":14},[1607,1608,1609,1610,1611,1612],[280,281,282],[284,285,286],[288,289,290],[292,293,294],[296,297,298],[300,301,302],{},{"id":306,"data":1615,"type":42,"tunes":1616},{"text":308,"level":219},{},{"id":311,"data":1618,"type":226,"tunes":1619},{"text":313},{},{"id":316,"data":1621,"type":341,"tunes":1630},{"steps":1622,"title":308,"orientation":340},[1623,1624,1625,1626,1627,1628,1629],{"label":320,"description":321},{"label":323,"description":324},{"label":326,"description":327},{"label":329,"description":330},{"label":332,"description":333},{"label":335,"description":336},{"label":338,"description":339},{},{"id":344,"data":1632,"type":42,"tunes":1633},{"text":346,"level":218},{},{"id":349,"data":1635,"type":226,"tunes":1636},{"text":351},{},{"id":354,"data":1638,"type":226,"tunes":1639},{"text":356},{},{"id":359,"data":1641,"type":42,"tunes":1642},{"text":361,"level":218},{},{"id":364,"data":1644,"type":226,"tunes":1645},{"text":366},{},{"id":369,"data":1647,"type":226,"tunes":1648},{"text":371},{},{"id":374,"data":1650,"type":42,"tunes":1651},{"text":376,"level":218},{},{"id":379,"data":1653,"type":226,"tunes":1654},{"text":381},{},{"id":384,"data":1656,"type":226,"tunes":1657},{"text":386},{},{"id":389,"data":1659,"type":234,"tunes":1660},{"body":391,"title":392,"variant":393},{},{"id":396,"data":1662,"type":42,"tunes":1663},{"text":398,"level":218},{},{"id":401,"data":1665,"type":226,"tunes":1666},{"text":403},{},{"id":406,"data":1668,"type":226,"tunes":1669},{"text":408},{},{"id":411,"data":1671,"type":226,"tunes":1672},{"text":413},{},{"id":416,"data":1674,"type":42,"tunes":1675},{"text":418,"level":218},{},{"id":421,"data":1677,"type":226,"tunes":1678},{"text":423},{},{"id":426,"data":1680,"type":226,"tunes":1681},{"text":428},{},{"id":431,"data":1683,"type":226,"tunes":1684},{"text":433},{},{"id":436,"data":1686,"type":42,"tunes":1687},{"text":438,"level":218},{},{"id":441,"data":1689,"type":226,"tunes":1690},{"text":443},{},{"id":446,"data":1692,"type":226,"tunes":1693},{"text":448},{},{"id":451,"data":1695,"type":226,"tunes":1696},{"text":453},{},{"id":456,"data":1698,"type":42,"tunes":1699},{"text":458,"level":218},{},{"id":461,"data":1701,"type":226,"tunes":1702},{"text":463},{},{"id":466,"data":1704,"type":226,"tunes":1705},{"text":468},{},{"id":471,"data":1707,"type":226,"tunes":1708},{"text":473},{},{"id":476,"data":1710,"type":42,"tunes":1711},{"text":478,"level":219},{},{"id":481,"data":1713,"type":226,"tunes":1714},{"text":483},{},{"id":486,"data":1716,"type":341,"tunes":1727},{"steps":1717,"title":478,"orientation":340},[1718,1719,1720,1721,1722,1723,1724,1725,1726],{"label":490,"description":491},{"label":493,"description":494},{"label":496,"description":497},{"label":499,"description":500},{"label":502,"description":503},{"label":505,"description":506},{"label":508,"description":509},{"label":511,"description":512},{"label":514,"description":515},{},{"id":518,"data":1729,"type":42,"tunes":1730},{"text":520,"level":219},{},{"id":523,"data":1732,"type":226,"tunes":1733},{"text":525},{},{"id":528,"data":1735,"type":226,"tunes":1736},{"text":530},{},{"id":533,"data":1738,"type":539,"tunes":1739},{"url":535,"title":536,"excerpt":537,"ctaLabel":538},{},{"id":542,"data":1741,"type":42,"tunes":1742},{"text":544,"level":219},{},{"id":547,"data":1744,"type":578,"tunes":1758},{"rows":1745,"title":567,"layout":303,"columns":1754},[1746,1748,1750,1752],{"id":551,"label":552,"values":1747},[554,554,554],{"id":556,"label":557,"values":1749},[554,554,554],{"id":560,"label":561,"values":1751},[554,554,554],{"id":564,"label":565,"values":1753},[554,554,554],[1755,1756,1757],{"id":570,"label":571},{"id":573,"label":574},{"id":576,"label":577},{},{"id":581,"data":1760,"type":226,"tunes":1761},{"text":583},{},{"id":586,"data":1763,"type":42,"tunes":1764},{"text":588,"level":219},{},{"id":591,"data":1766,"type":226,"tunes":1767},{"text":593},{},{"id":596,"data":1769,"type":303,"tunes":1780},{"content":1770,"stretched":43,"withHeadings":14},[1771,1772,1773,1774,1775,1776,1777,1778,1779],[600,601,602],[604,605,606],[608,609,610],[612,613,614],[616,617,618],[620,621,622],[624,625,626],[628,629,630],[632,633,634],{},{"id":637,"data":1782,"type":42,"tunes":1783},{"text":639,"level":219},{},{"id":642,"data":1785,"type":226,"tunes":1786},{"text":644},{},{"id":647,"data":1788,"type":226,"tunes":1789},{"text":649},{},{"id":652,"data":1791,"type":42,"tunes":1792},{"text":654,"level":219},{},{"id":657,"data":1794,"type":303,"tunes":1802},{"content":1795,"stretched":43,"withHeadings":14},[1796,1797,1798,1799,1800,1801],[661,662,663],[665,666,667],[669,670,671],[673,674,675],[677,678,679],[681,682,683],{},{"id":686,"data":1804,"type":42,"tunes":1805},{"text":688,"level":219},{},{"id":691,"data":1807,"type":708,"tunes":1810},{"meta":1808,"items":1809,"style":707},{},[695,696,697,698,699,700,701,702,703,704,705,706],{},{"id":711,"data":1812,"type":226,"tunes":1813},{"text":713},{},{"id":716,"data":1815,"type":539,"tunes":1816},{"url":718,"title":719,"excerpt":720,"ctaLabel":721},{},{"id":724,"data":1818,"type":42,"tunes":1819},{"text":726,"level":219},{},{"id":729,"data":1821,"type":226,"tunes":1822},{"text":731},{},{"id":734,"data":1824,"type":226,"tunes":1825},{"text":736},{},{"id":739,"data":1827,"type":226,"tunes":1828},{"text":741},{},{"id":744,"data":1830,"type":42,"tunes":1831},{"text":746,"level":219},{},{"id":749,"data":1833,"type":226,"tunes":1834},{"text":751},{},{"id":754,"data":1836,"type":226,"tunes":1837},{"text":756},{},{"id":759,"data":1839,"type":42,"tunes":1840},{"text":761,"level":219},{},{"id":764,"data":1842,"type":226,"tunes":1843},{"text":766},{},{"id":769,"data":1845,"type":226,"tunes":1846},{"text":771},{},{"id":774,"data":1848,"type":42,"tunes":1849},{"text":776,"level":219},{},{"id":779,"data":1851,"type":226,"tunes":1852},{"text":781},{},{"id":784,"data":1854,"type":226,"tunes":1855},{"text":786},{},{"id":789,"data":1857,"type":42,"tunes":1858},{"text":791,"level":219},{},{"id":794,"data":1860,"type":794,"tunes":1868},{"items":1861,"title":821},[1862,1863,1864,1865,1866,1867],{"id":798,"answer":799,"question":800},{"id":802,"answer":803,"question":804},{"id":806,"answer":807,"question":808},{"id":810,"answer":811,"question":812},{"id":814,"answer":815,"question":816},{"id":818,"answer":819,"question":820},{},{"id":824,"data":1870,"type":42,"tunes":1871},{"text":826,"level":219},{},{"id":829,"data":1873,"type":829,"tunes":1881},{"title":831,"entries":1874},[1875,1876,1877,1878,1879,1880],{"term":834,"anchor":835,"definition":836},{"term":838,"anchor":839,"definition":840},{"term":842,"anchor":843,"definition":844},{"term":846,"anchor":847,"definition":848},{"term":850,"anchor":851,"definition":852},{"term":854,"anchor":855,"definition":856},{},{"id":859,"data":1883,"type":42,"tunes":1884},{"text":861,"level":219},{},{"id":864,"data":1886,"type":871,"tunes":1889},{"link":866,"meta":1887},{"image":1888,"title":869,"description":870},{"url":554},{},{"id":874,"data":1891,"type":871,"tunes":1894},{"link":876,"meta":1892},{"image":1893,"title":879,"description":880},{"url":554},{},{"id":883,"data":1896,"type":871,"tunes":1899},{"link":885,"meta":1897},{"image":1898,"title":888,"description":889},{"url":554},{},{"id":892,"data":1901,"type":871,"tunes":1904},{"link":894,"meta":1902},{"image":1903,"title":897,"description":898},{"url":554},{},{"id":901,"data":1906,"type":871,"tunes":1909},{"link":903,"meta":1907},{"image":1908,"title":906,"description":907},{"url":554},{},{"id":910,"data":1911,"type":871,"tunes":1914},{"link":912,"meta":1912},{"image":1913,"title":915,"description":916},{"url":554},{},{"id":919,"data":1916,"type":871,"tunes":1919},{"link":921,"meta":1917},{"image":1918,"title":924,"description":925},{"url":554},{},{"id":928,"data":1921,"type":871,"tunes":1924},{"link":930,"meta":1922},{"image":1923,"title":933,"description":934},{"url":554},{},"Post erfolgreich abgerufen",{"items":1927,"source":1991,"manualIds":1992,"manualMatchedIds":1993},[1928,1935,1942,1949,1956,1963,1970,1977,1984],{"id":1929,"slug":1930,"title":1931,"excerpt":1932,"featuredImage":1933,"publishedAt":1934},"457","should-you-buy-5g-openwrt-router-old-firmware","你应该购买带有旧固件的5G OpenWrt路由器吗？以ZBT Z8102AX为例","购买搭载旧版固件的5G OpenWrt路由器在特定条件下是合理的。ZBT Z8102AX型号清晰展现了利弊两面：硬件实用、调制解调器工作正常，测试中路由器保持稳定，但OpenWrt 21.02版本、简陋的包装以及不明确的升级路径，要求消费者在购买时需审慎决策。","\u002Fuploads\u002F2026\u002F06\u002Fopenwrt-router-review-dual-sim-05-1781620596218-5ldld4.webp","2026-06-16T10:41:00.000Z",{"id":1936,"slug":1937,"title":1938,"excerpt":1939,"featuredImage":1940,"publishedAt":1941},"478","what-is-rag-the-simplest-explanation-of-how-it-works","什么是RAG？对其工作原理的最简单解释","RAG听起来很复杂，但想法很简单：在AI回答之前，它先从知识源查找有用的信息，并将该信息提供给语言模型。本指南使用一个简单的思维模型来解释RAG、LLM、状态、记忆和工具。","\u002Fuploads\u002F2026\u002F09\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works-1790377492124-khjagt.webp","2026-09-25T19:03:00.000Z",{"id":1943,"slug":1944,"title":1945,"excerpt":1946,"featuredImage":1947,"publishedAt":1948},"364","tipps-fuer-die-verbesserung-der-seo-suchmaschinenoptimierung","Mastering the SEO Workflow: Essential Optimization Strategies for Organic Growth","A structured SEO workflow is crucial for sustainable organic growth. Learn the ten foundational strategies, from keyword research and technical optimization to content quality and performance analysis.","\u002Fuploads\u002F2026\u002F03\u002Ftipps-fuer-die-verbesserung-der-seo-suchmaschinenoptimierung-1774866098131-hwkzrg.webp","2024-01-26T06:35:00.000Z",{"id":1950,"slug":1951,"title":1952,"excerpt":1953,"featuredImage":1954,"publishedAt":1955},"459","ollama-is-not-the-product-building-production-ready-open-llm-applications","Ollama 并非产品：构建可投入生产的开源大语言模型应用","使用Ollama运行本地模型很简单。但构建一个可用于生产环境的开源大语言模型（Open-LLM）应用则更具挑战性：它需要RAG（检索增强生成）、访问控制、供应商抽象、评估、日志记录、部署规范，以及围绕模型构建受控的应用层。","\u002Fuploads\u002F2026\u002F06\u002Follama-is-not-the-product-building-production-ready-open-llm-applications-1782679361640-h0usqf.webp","2026-06-28T16:39:00.000Z",{"id":1957,"slug":1958,"title":1959,"excerpt":1960,"featuredImage":1961,"publishedAt":1962},"471","how-to-know-whether-an-ai-agent-actually-used-the-right-evidence","如何判断一个AI智能体是否真正使用了正确的证据","AI代理可以引用来源，却仍然使用错误的证据。本文介绍一种实用方法，用于核查主张支持、来源权威性、适用性、出处，以及证据是否实际影响了答案。","\u002Fuploads\u002F2026\u002F09\u002Fhow-to-know-whether-an-ai-agent-actually-used-the-right-evidence-1790351317188-o5z9ve.webp","2026-09-25T11:47:00.000Z",{"id":1964,"slug":1965,"title":1966,"excerpt":1967,"featuredImage":1968,"publishedAt":1969},"472","why-more-context-can-make-ai-answers-worse","为什么更多上下文会让AI的回答更糟","更大的上下文窗口并不保证更好的答案。本文解释了信号稀释、证据冲突、状态过时、位置敏感性和有损压缩如何降低AI可靠性——并介绍了一种实用的上下文压力测试。","\u002Fuploads\u002F2026\u002F09\u002Fwhy-more-context-can-make-ai-answers-worse-1790351615793-2ntv2v.webp","2026-09-25T11:51:00.000Z",{"id":1971,"slug":1972,"title":1973,"excerpt":1974,"featuredImage":1975,"publishedAt":1976},"474","migrating-from-openai-agents-sdk-to-the-agents-api-what-actually-changes-architecturally","从OpenAI Agents SDK迁移到Agents API：架构上究竟有哪些变化？","从 OpenAI Agents SDK 迁移到新的 Agents API 并不是简单的导入重命名。运行时边界发生了变化：代理循环、持久会话、编排、上下文压缩与恢复都向托管执行框架迁移。本指南说明哪些应当迁移、哪些应当保留在您的应用程序中，以及如何在切换前验证迁移。","\u002Fuploads\u002F2026\u002F09\u002Fmigrating-from-openai-agents-sdk-to-the-agents-api-what-actually-changes-architecturally-1790352171968-ienxr9.webp","2026-09-25T12:01:00.000Z",{"id":1978,"slug":1979,"title":1980,"excerpt":1981,"featuredImage":1982,"publishedAt":1983},"469","rag-failed-but-which-layer-actually-failed-a-diagnostic-method","RAG失败了——但究竟是哪一层真正失败了？一种诊断方法","当RAG答案出错时，将问题归咎于检索或模型过于笼统。这种诊断方法将来源覆盖、查询构建、检索、排序、上下文组装、生成、证据归因和时效性逐一隔离，从而使实际故障能够被复现并修复。","\u002Fuploads\u002F2026\u002F09\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method-1790350847177-pior4c.webp","2026-09-24T19:39:00.000Z",{"id":1985,"slug":1986,"title":1987,"excerpt":1988,"featuredImage":1989,"publishedAt":1990},"460","ai-agent-reliability-why-the-final-answer-is-not-enough","AI Agent可靠性：为什么最终答案并不足够","正确的输出并不能证明推理的正确性、执行的安全性，或系统的可信赖性。","\u002Fuploads\u002F2026\u002F09\u002Fai-agent-reliability-why-the-final-answer-is-not-enough-1788955466306-pl0qhz.webp","2026-09-09T04:01:00.000Z","fallback",[],[]]