[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"portal-settings:stajic:es":3,"public-menus:all":38,"post:ai-agent-reliability-why-the-final-answer-is-not-enough:es":205,"related:post:ai-agent-reliability-why-the-final-answer-is-not-enough:es:1":1130},{"statusCode":4,"data":5,"message":37},200,{"tenantId":6,"lang":7,"defaultLang":8,"siteUrl":9,"contactEmail":10,"brandName":11,"logoUrl":12,"siteName":11,"siteDescription":13,"ogImage":10,"robotsIndex":14,"socialLinks":10,"reservedSlugs":10,"seoPolicy":15},"stajic","es","de","https:\u002F\u002Fstajic.de",null,"Stajic Platform","\u002FLogo_Planet.svg","Stajic Portal",true,{"branding":16,"relatedContent":17,"crossDomainLinks":18},{"logoUrl":12},{"enabled":14},[19,22,25,28,31,34],{"url":20,"label":21,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Ffigure.rocks","figure.rocks",{"url":23,"label":24,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Floving.rocks","loving.rocks",{"url":26,"label":27,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.com","bazify.com",{"url":29,"label":30,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.de","bazify.de",{"url":32,"label":33,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.at","bazify.at",{"url":35,"label":36,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.ba","bazify.ba","Portal settings resolved",[39,45],{"id":40,"name":41,"location":42,"isActive":14,"isDefault":43,"items":44},1,"main-navigation","header",false,[],{"id":46,"name":47,"location":48,"isActive":14,"isDefault":14,"items":49},4,"main-menu","sidebar",[50,66,79,93,103,118,133],{"id":51,"title":52,"url":60,"target":61,"icon":62,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":64,"portfolioId":10,"children":65},"item-18",{"de":53,"en":54,"es":55,"fr":56,"it":54,"ru":57,"sr":58,"zh":59},"Startseite","Home","Inicio","Accueil","Главная","Почетна","首页","\u002Ffull-stack-web-developer-munich-performance-seo-and-maintainable-builds","_self","i-lucide-home","page",111,[],{"id":67,"title":68,"url":75,"target":61,"icon":76,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":77,"portfolioId":10,"children":78},"item-22",{"de":69,"en":69,"es":70,"fr":69,"it":71,"ru":72,"sr":73,"zh":74},"Vision","Visión","Visione","Видение","Визија","想象","\u002Fueber-uns-webdesign-muenchen-webaplikation","i-lucide-eye",113,[],{"id":80,"title":81,"url":89,"target":61,"icon":90,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":91,"portfolioId":10,"children":92},"item-19",{"de":82,"en":83,"es":84,"fr":83,"it":85,"ru":86,"sr":87,"zh":88},"Leistungen","Services","Servicios","Servizi","Услуги","Услуге","服务","\u002Fservices-dienstleistungen-muenchen","i-lucide-wrench",116,[],{"id":94,"title":95,"url":99,"target":61,"icon":100,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":101,"portfolioId":10,"children":102},"item-23",{"de":96,"en":96,"es":96,"fr":96,"it":96,"ru":97,"sr":97,"zh":98},"Blog","Блог","博客","\u002Fblog","i-lucide-book-open",112,[],{"id":104,"title":105,"url":114,"target":61,"icon":115,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":116,"portfolioId":10,"children":117},"item-32",{"de":106,"en":107,"es":108,"fr":109,"it":110,"ru":111,"sr":112,"zh":113},"Neue Technologien","New Technologies","Nuevas tecnologías","Nouvelles technologies","Nuove tecnologie","Новые технологии","Нове технологије","新技术！","\u002Fneue-webtechnologien","i-lucide-sparkles",122,[],{"id":119,"title":120,"url":129,"target":61,"icon":130,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":131,"portfolioId":10,"children":132},"item-20",{"de":121,"en":122,"es":123,"fr":124,"it":125,"ru":126,"sr":127,"zh":128},"Kontakt","Contact us!","Contacto","Contact","Contatto","Контакт","Контактирајте нас","联系我们！","\u002Fcontact","i-lucide-mail",115,[],{"id":134,"title":135,"url":144,"target":61,"icon":145,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":147},"item-21",{"de":136,"en":137,"es":138,"fr":139,"it":140,"ru":141,"sr":142,"zh":143},"Unsere Arbeit","Our Work","Nuestro trabajo","Nos réalisations","I nostri lavori","Наши работы","Наши радови","文件夹","\u002Fportfolio","i-lucide-briefcase",114,[148,161,175,181,193],{"id":149,"title":150,"url":144,"target":61,"icon":159,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":160},"item-24",{"de":151,"en":152,"es":153,"fr":154,"it":155,"ru":156,"sr":157,"zh":158},"Alle Projekte","All Projects","Todos los proyectos","Tous les projets","Tutti i progetti","Все проекты","Сви пројекти","所有项目","i-lucide-grid-3x3",[],{"id":162,"title":163,"url":171,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":174},"item-29",{"de":164,"en":165,"es":166,"fr":167,"it":168,"ru":169,"sr":170,"zh":143},"Local Roots, Global Reach","Local Roots - Global Reach","Empresa local ","Entreprise locale","Azienda locale","Местная компания","Локално предузеће глобално тржиште","\u002Fportfolio\u002Flocal-roots-global-reach-communication-media-systems-for-modern-business","i-lucide-folder","custom",[],{"id":176,"title":177,"url":179,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":180},"item-28",{"de":178,"en":178,"es":178,"fr":178,"it":178,"ru":178,"sr":178,"zh":178},"Solr Suggester","\u002Fportfolio\u002Fsolr-fuzzy-suggester-und-solr-infix-suggester-abfrage-ueber-ajax-und-filterung",[],{"id":182,"title":183,"url":191,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":192},"item-27",{"de":184,"en":185,"es":186,"fr":187,"it":188,"ru":189,"sr":190,"zh":185},"Firmenwebseite SEO","Company Website SEO","Sitio web corporativo SEO","Site web d’entreprise SEO","Sito web aziendale SEO","Корпоративный сайт SEO","Пословна веб-страница SEO","\u002Fportfolio\u002Fseo-sem-branding-mobile-webseite-muenchen",[],{"id":194,"title":195,"url":203,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":204},"item-31",{"de":196,"en":197,"es":198,"fr":199,"it":200,"ru":201,"sr":202,"zh":197},"Digitalisierungsportal","Digitalization Portal","Portal de digitalización","Portail de numérisation","Portale di digitalizzazione","Портал цифровизации","Портал за дигитализацију","\u002Fportfolio\u002Fdigitalisierungsportal-archiv-museum-bibliothek-ead-lido-mets-mods",[],{"statusCode":4,"data":206,"message":1129},{"id":207,"title":208,"slug":209,"content":210,"contentJson":211,"excerpt":561,"featuredImage":562,"featuredImageAlt":563,"featuredImageCaption":10,"featuredImageTitle":10,"featuredImageCopyright":10,"featuredImageAuthor":10,"featuredImageSourceUrl":10,"featuredImageLicense":10,"featuredImageIsAiGenerated":43,"status":564,"publishedAt":565,"createdAt":566,"updatedAt":567,"seoLocalePaths":568,"categories":577,"author":586,"translations":591},"460","Fiabilidad de los Agentes de IA: Por Qué la Respuesta Final No es Suficiente","ai-agent-reliability-why-the-final-answer-is-not-enough","\u003Cp>\u003Cb>La salida correcta no demuestra un razonamiento correcto, una ejecución segura o un sistema confiable.\u003C\u002Fb>\u003C\u002Fp>\n\u003Cp>Durante años, la evaluación de la IA ha estado dominada por una pregunta engañosamente simple: \u003Cb>¿Fue correcta la respuesta?\u003C\u002Fb> Para un chatbot, esto a veces puede ser suficiente. Para un agente capaz de buscar en sistemas, leer datos, llamar herramientas, modificar estados, ejecutar flujos de trabajo, escribir archivos, interactuar con APIs o tomar decisiones, no lo es.\u003C\u002Fp>\n\u003Cp>Un agente puede producir la respuesta final correcta mientras hace varias cosas mal en el camino. Puede usar la fuente equivocada, malinterpretar una instrucción y luego compensar el error, acceder a información innecesaria, ejecutar una acción intermedia no autorizada, recuperarse silenciosamente de un error que debería haber provocado una escalada, o dejar efectos secundarios que nadie notó.\u003C\u002Fp>\n\u003Cp>Eso crea uno de los problemas centrales de la IA agéntica: \u003Cb>un resultado correcto no demuestra una trayectoria correcta.\u003C\u002Fb>\u003C\u002Fp>\n\u003Ch2>La ilusión del resultado\u003C\u002Fh2>\n\u003Cp>El software tradicional nos da un modelo intuitivo de corrección. La entrada entra en un sistema determinista o mayormente determinista, se ejecuta la lógica, se produce la salida y las pruebas verifican el comportamiento esperado. Los sistemas basados en LLM debilitan esta suposición. Los sistemas agénticos van más allá.\u003C\u002Fp>\n\u003Cul>\u003Cli>interpretación del modelo\u003C\u002Fli>\u003Cli>contexto recuperado\u003C\u002Fli>\u003Cli>selección de herramientas\u003C\u002Fli>\u003Cli>observaciones intermedias\u003C\u002Fli>\u003Cli>estado externo\u003C\u002Fli>\u003Cli>acciones previas\u003C\u002Fli>\u003Cli>planes generados por el modelo\u003C\u002Fli>\u003Cli>límites de permisos\u003C\u002Fli>\u003Cli>reintentos y comportamiento de respaldo\u003C\u002Fli>\u003Cli>interacción humana\u003C\u002Fli>\u003C\u002Ful>\n\u003Cp>Dos ejecuciones que parten de entradas casi idénticas pueden alcanzar el mismo resultado a través de caminos muy diferentes. Si la evaluación observa solo la salida final, la mayor parte del sistema permanece invisible.\u003C\u002Fp>\n\u003Cp>Imagina que un agente de IA recibe la instrucción: \u003Ci>Actualiza la dirección de facturación del cliente.\u003C\u002Fi> La dirección finalmente se actualiza correctamente. Una evaluación convencional podría clasificar la tarea como exitosa.\u003C\u002Fp>\n\u003Col>\u003Cli>El agente busca varios registros de clientes no relacionados.\u003C\u002Fli>\u003Cli>Recupera más información personal de la necesaria.\u003C\u002Fli>\u003Cli>Inicialmente modifica la cuenta equivocada.\u003C\u002Fli>\u003Cli>Nota el error.\u003C\u002Fli>\u003Cli>Revierte el cambio.\u003C\u002Fli>\u003Cli>Actualiza la cuenta correcta.\u003C\u002Fli>\u003Cli>Reporta éxito.\u003C\u002Fli>\u003C\u002Fol>\n\u003Cp>\u003Cb>Estado final: correcto. Comportamiento del sistema: inaceptable.\u003C\u002Fb> Un benchmark basado solo en resultados le da a esta ejecución un aprobado. Un sistema de aseguramiento de producción no debería.\u003C\u002Fp>\n\u003Ch2>La trayectoria es parte del producto\u003C\u002Fh2>\n\u003Cp>Por eso la \u003Cb>trayectoria\u003C\u002Fb> de un agente de IA debe convertirse en un objeto de ingeniería de primera clase. Una trayectoria es la secuencia de estados y acciones relevantes entre la solicitud original y el resultado final.\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>Intención → Contexto → Decisión → Herramienta → Acción → Observación → Decisión → Cambio de estado → Resultado\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>Zachary J. Stevens desarrolla esta idea en \u003Ci>La trayectoria es el sistema\u003C\u002Fi>, argumentando que la evaluación agéntica debe ir más allá de la respuesta final y examinar el camino completo de acción a través de un entorno cambiante.\u003C\u002Fp>\n\u003Cblockquote class=\"border-l-4 border-gray-300 pl-4 italic\">Un resultado correcto no excusa una trayectoria inaceptable.\u003Ccite class=\"block mt-2 text-sm\">— Zachary J. Stevens, La trayectoria es el sistema\u003C\u002Fcite>\u003C\u002Fblockquote>\n\u003Ca href=\"https:\u002F\u002Fzacharyjstevens.com\u002Fdispatches\u002Fvanguard-signal\u002F009-the-trajectory-is-the-system\u002F\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">La trayectoria es el sistema\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">Zachary J. Stevens — DFEI.009 sobre evaluar sistemas agénticos por su trayectoria completa en lugar de solo el resultado final.\u003C\u002Fp>\u003C\u002Fa>\n\u003Cp>La distinción importa enormemente. Por lo tanto, la confiabilidad no es simplemente \u003Cb>salida correcta\u003C\u002Fb>. Está más cerca de \u003Cb>resultado aceptable + trayectoria aceptable + recuperabilidad + evidencia\u003C\u002Fb>.\u003C\u002Fp>\n\u003Ch2>Una respuesta correcta puede ocultar un sistema roto\u003C\u002Fh2>\n\u003Ctable class=\"w-full border-collapse\">\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Agente\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Resultado final\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Ejecución\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">A\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Correcto\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Camino correcto\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">B\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Correcto\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Camino inseguro\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">C\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Incorrecto\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Fallo seguro\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">D\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Incorrecto\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Fallo inseguro\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftable>\n\u003Cp>La mayoría de las evaluaciones basadas en benchmarks recompensan fuertemente a A y B y penalizan a C y D. Operativamente, sin embargo, \u003Cb>B puede ser más peligroso que C\u003C\u002Fb>. El agente C puede reconocer la incertidumbre, detener la ejecución y solicitar revisión humana. El agente B puede producir con confianza resultados correctos mientras viola suposiciones que nadie está monitoreando.\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>salida exitosa → mayor confianza → permisos más amplios → más automatización → mayor radio de explosión\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Ch2>Necesitamos evidencia, no confianza\u003C\u002Fh2>\n\u003Cp>Uno de los mayores errores en la adopción de IA es tratar la confianza del modelo, la satisfacción del usuario o la tasa de éxito histórica como evidencia de la fiabilidad del sistema. No son equivalentes.\u003C\u002Fp>\n\u003Cul>\u003Cli>¿Qué recibió el agente?\u003C\u002Fli>\u003Cli>¿Qué contexto recuperó?\u003C\u002Fli>\u003Cli>¿Qué herramientas llamó?\u003C\u002Fli>\u003Cli>¿Por qué se permitió la acción?\u003C\u002Fli>\u003Cli>¿Qué estado existía antes de la acción?\u003C\u002Fli>\u003Cli>¿Qué cambió?\u003C\u002Fli>\u003Cli>¿Qué fallos intermedios ocurrieron?\u003C\u002Fli>\u003Cli>¿Se reintentó algo?\u003C\u002Fli>\u003Cli>¿Se requirió aprobación humana?\u003C\u002Fli>\u003Cli>¿Se podría haber detenido la ejecución?\u003C\u002Fli>\u003Cli>¿Se puede revertir la acción?\u003C\u002Fli>\u003Cli>¿Qué modelo, prompt y versiones de herramientas estuvieron involucrados?\u003C\u002Fli>\u003C\u002Ful>\n\u003Cp>Sin estas respuestas, no hay una garantía operativa seria. Solo hay una salida. La observabilidad y la evidencia deben, por lo tanto, diseñarse dentro de la arquitectura del agente en lugar de agregarse después del despliegue.\u003C\u002Fp>\n\u003Ch2>Registrar no es lo mismo que controlar\u003C\u002Fh2>\n\u003Cp>Las organizaciones a menudo responden: \u003Ci>Todo está registrado.\u003C\u002Fi> Bien. Pero registrar solo no controla nada. Un registro te dice lo que sucedió. Un control determina si algo \u003Cb>puede suceder\u003C\u002Fb>.\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>El agente solicita DELETE \u002Fcustomer\u002F123 ↓\nAcción registrada ↓\nDELETE ejecutado\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>Eso proporciona observabilidad. Compáralo con:\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>El agente solicita DELETE \u002Fcustomer\u002F123 ↓\nEvaluación de políticas ↓\nIdentidad actual verificada ↓\nParámetros de acción actuales comprobados ↓\nUmbral de riesgo evaluado ↓\nAprobación humana si es necesaria ↓\nAcción ejecutada ↓\nResultado verificado ↓\nEvidencia almacenada\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>Ahora nos acercamos a un sistema de control. La diferencia es arquitectónica, no cosmética.\u003C\u002Fp>\n\u003Ch2>El permiso es necesario, pero no es garantía\u003C\u002Fh2>\n\u003Cp>Supongamos que un agente tiene permiso para enviar correos electrónicos. El control de acceso responde: \u003Cb>¿Puede este agente enviar correos?\u003C\u002Fb> No responde: \u003Cb>¿Debería este correo en particular enviarse a esta persona en particular con este adjunto en particular en este momento?\u003C\u002Fb>\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>CONTROL DE CAPACIDAD\n¿Qué se le permite técnicamente hacer al agente? + GARANTÍA DE ACCIÓN\n¿Es apropiada esta acción específica en el estado actual?\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>RBAC, alcances de OAuth, permisos de API e identidades de agentes definen el espacio de acciones posibles. No prueban que una acción dentro de ese espacio sea apropiada. Una arquitectura de agente sólida necesita ambas capas.\u003C\u002Fp>\n\u003Ch2>El Primer Paso Incorrecto Importa\u003C\u002Fh2>\n\u003Cp>Cuando un agente falla, la acción incorrecta final a menudo no es donde comenzó el fallo. El fallo real puede haber ocurrido mucho antes.\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>Recuperación incorrecta ↓\nSuposición incorrecta ↓\nRazonamiento plausible ↓\nLlamada a herramienta válida ↓\nAcción incorrecta\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>Si investigamos solo la acción final, corregimos el síntoma. Si inspeccionamos la trayectoria, podemos identificar el \u003Cb>primer paso incorrecto\u003C\u002Fb>. Eso convierte un fallo no atribuible en un problema de ingeniería concreto.\u003C\u002Fp>\n\u003Ch2>Las Pruebas de Agentes Deben Ir Más Allá de las Pruebas de Prompts\u003C\u002Fh2>\n\u003Cp>Los prompts importan, pero el comportamiento de un agente en producción surge de un sistema completo.\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>MODELO\n+\nPROMPT DEL SISTEMA\n+\nCONTEXTO\n+\nMEMORIA\n+\nRECUPERACIÓN\n+\nHERRAMIENTAS\n+\nPERMISOS\n+\nFLUJO DE TRABAJO\n+\nESTADO EXTERNO\n+\nLÓGICA DE CONTROL\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>Cambiar cualquiera de estos puede cambiar la trayectoria. Por lo tanto, versionar solo el prompt es insuficiente.\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>versión_del_modelo\nversión_del_prompt\nversión_de_la_herramienta\nversión_de_la_política\nversión_de_recuperación\nversión_del_flujo_de_trabajo\nestado_del_entorno\nid_de_ejecución\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Ch2>Los Criterios de Aceptación para Agentes Deben Incluir Comportamiento\u003C\u002Fh2>\n\u003Cp>Los criterios de aceptación tradicionales a menudo se ven así: \u003Ci>Dado X, el sistema produce Y.\u003C\u002Fi> Para sistemas agénticos, eso es incompleto. Los criterios de aceptación también deben definir restricciones sobre la trayectoria.\u003C\u002Fp>\n\u003Ch3>Resultado\u003C\u002Fh3>\n\u003Cp>La dirección del cliente se actualiza correctamente.\u003C\u002Fp>\n\u003Ch3>Autorización\u003C\u002Fh3>\n\u003Cp>El agente modifica solo al cliente explícitamente seleccionado.\u003C\u002Fp>\n\u003Ch3>Acceso a datos\u003C\u002Fh3>\n\u003Cp>No se accede a registros de clientes no relacionados.\u003C\u002Fp>\n\u003Ch3>Herramientas\u003C\u002Fh3>\n\u003Cp>Solo se utilizan operaciones CRM aprobadas.\u003C\u002Fp>\n\u003Ch3>Verificación\u003C\u002Fh3>\n\u003Cp>La nueva dirección se lee y se compara con el valor solicitado.\u003C\u002Fp>\n\u003Ch3>Fallo\u003C\u002Fh3>\n\u003Cp>La resolución ambigua de identidad detiene la ejecución.\u003C\u002Fp>\n\u003Ch3>Autoridad humana\u003C\u002Fh3>\n\u003Cp>Un humano puede rechazar la modificación antes de la ejecución cuando los umbrales de riesgo requieren aprobación.\u003C\u002Fp>\n\u003Ch3>Evidencia\u003C\u002Fh3>\n\u003Cp>La ejecución deja un rastro suficiente para reconstruir la decisión y la transición de estado.\u003C\u002Fp>\n\u003Ch3>Recuperación\u003C\u002Fh3>\n\u003Cp>El valor anterior sigue siendo recuperable.\u003C\u002Fp>\n\u003Ch2>El humano en el circuito no es suficiente\u003C\u002Fh2>\n\u003Cp>Agregar una casilla de aprobación humana no resuelve automáticamente el problema. Un humano solo puede controlar a un agente si la persona tiene visibilidad, autoridad, tiempo, contexto y capacidad de recuperación.\u003C\u002Fp>\n\u003Cul>\u003Cli>\u003Cb>Visibilidad:\u003C\u002Fb> suficiente información para entender lo que está sucediendo.\u003C\u002Fli>\u003Cli>\u003Cb>Autoridad:\u003C\u002Fb> capacidad real de detener o modificar la acción.\u003C\u002Fli>\u003Cli>\u003Cb>Tiempo:\u003C\u002Fb> intervención antes de que ocurra la consecuencia.\u003C\u002Fli>\u003Cli>\u003Cb>Contexto:\u003C\u002Fb> evidencia suficiente para tomar la decisión.\u003C\u002Fli>\u003Cli>\u003Cb>Capacidad de recuperación:\u003C\u002Fb> capacidad de revertir o reparar la acción.\u003C\u002Fli>\u003C\u002Ful>\n\u003Cp>Un usuario que hace clic en \u003Cb>Aprobar\u003C\u002Fb> sobre algo que no puede inspeccionar de manera significativa no es una gobernanza sólida. Es teatro de aprobación.\u003C\u002Fp>\n\u003Ch2>La reversión debe convertirse en una capacidad nativa de la IA\u003C\u002Fh2>\n\u003Cp>El despliegue de software tradicional nos ha enseñado algo valioso: \u003Cb>Nunca implementes lo que no puedes revertir.\u003C\u002Fb> Deberíamos aplicar el mismo principio a las acciones agénticas.\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>REVERSIBLE\nPuede deshacerse automáticamente. COMPENSABLE\nNo puede deshacerse directamente, pero puede ejecutar una acción compensatoria. IRREVERSIBLE\nNo puede restaurar de manera confiable el estado anterior.\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>Cuanto mayor es la irreversibilidad, más fuerte debe ser el requisito de control.\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>Leer documento público → baja consecuencia\nCrear borrador → reversible\nModificar registro CRM → reversible pero con consecuencias\nEnviar correo externo → prácticamente irreversible\nTransferir dinero → alta consecuencia\nEliminar datos de producción → potencialmente catastrófico\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Ch2>El Agente Necesita un Plano de Control\u003C\u002Fh2>\n\u003Cpre class=\"code-block\">\u003Ccode>USUARIO \u002F INTENCIÓN DEL SISTEMA │ ▼ AGENTE DE IA │ acción propuesta │ ▼ ┌───────────────────┐ │ PLANO DE CONTROL │ ├───────────────────┤ │ Identidad │ │ Autorización │ │ Política │ │ Riesgo │ │ Estado │ │ Evidencia │ │ Autoridad humana │ │ Reversión │ └───────────────────┘ │ ¿aprobado? \u002F \\ NO SÍ │ │ DETENER ▼ HERRAMIENTA │ ▼ CAMBIO DE ESTADO │ ▼ VERIFICACIÓN\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>\u003Cb>El LLM debe proponer. El plano de control debe gobernar.\u003C\u002Fb> Esa separación es crucial. El modelo no debe ser la autoridad última que determine si su propia acción propuesta de alto impacto es segura.\u003C\u002Fp>\n\u003Ch2>De los Benchmarks a la Confianza Operativa\u003C\u002Fh2>\n\u003Cp>Los benchmarks siguen siendo útiles. Nos informan sobre la capacidad, comparan modelos, detectan regresiones y ayudan a estimar el rendimiento esperado. Pero la evaluación de la capacidad y la confianza operativa responden a preguntas diferentes.\u003C\u002Fp>\n\u003Cp>Un benchmark pregunta: \u003Cb>¿Puede el sistema hacer esto?\u003C\u002Fb> La garantía operativa pregunta: \u003Cb>¿Podemos permitir que el sistema haga esto aquí, bajo estas condiciones, con estos permisos y consecuencias?\u003C\u002Fb>\u003C\u002Fp>\n\u003Ch2>La Fiabilidad Debe Medirse como una Propiedad del Sistema\u003C\u002Fh2>\n\u003Col>\u003Cli>\u003Cb>Corrección del resultado:\u003C\u002Fb> ¿Produjo el sistema el resultado esperado?\u003C\u002Fli>\u003Cli>\u003Cb>Corrección de la trayectoria:\u003C\u002Fb> ¿Siguió una ruta aceptable?\u003C\u002Fli>\u003Cli>\u003Cb>Integridad del control:\u003C\u002Fb> ¿Se respetaron los límites de autorización, política e intervención?\u003C\u002Fli>\u003Cli>\u003Cb>Recuperabilidad:\u003C\u002Fb> ¿Pueden los fallos contenerse, revertirse o repararse?\u003C\u002Fli>\u003Cli>\u003Cb>Completitud de la evidencia:\u003C\u002Fb> ¿Puede la ejecución reconstruirse y auditarse?\u003C\u002Fli>\u003C\u002Fol>\n\u003Cpre class=\"code-block\">\u003Ccode>Fiabilidad Operativa\n=\nResultado × Trayectoria × Control × Recuperabilidad × Evidencia\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>La multiplicación es intencional. Si una dimensión crítica se acerca a cero, una puntuación alta en otra parte no debería ocultarlo. Un resultado perfectamente correcto con integridad de autorización cero no es un sistema fiable al 80%. Es una ejecución inaceptable que casualmente produjo la respuesta correcta.\u003C\u002Fp>\n\u003Ch2>El Éxito es a Veces el Fracaso Más Peligroso\u003C\u002Fh2>\n\u003Cp>Los fracasos atraen la atención. El éxito a menudo no. Eso hace que las trayectorias de agentes exitosas pero no controladas sean particularmente peligrosas. Un fracaso evidente crea un incidente. Un defecto oculto en la trayectoria crea \u003Cb>confianza\u003C\u002Fb>. Y la confianza expande la autonomía.\u003C\u002Fp>\n\u003Cp>Las organizaciones no deberían por tanto investigar solo \u003Ci>¿Por qué falló el agente?\u003C\u002Fi> Deberían preguntarse periódicamente: \u003Cb>¿Por qué tuvo éxito el agente?\u003C\u002Fb> ¿Tuvo éxito porque la arquitectura restringió y verificó de manera fiable la ejecución, o porque esta vez nada salió mal?\u003C\u002Fp>\n\u003Ch2>Conclusión\u003C\u002Fh2>\n\u003Cp>La industria se mueve rápidamente desde la IA que \u003Cb>responde\u003C\u002Fb> hacia la IA que \u003Cb>actúa\u003C\u002Fb>. Esa transición cambia lo que significa fiabilidad. Para un sistema de respuestas, evaluar la respuesta puede ser a menudo suficiente. Para un sistema de acciones, debemos evaluar la trayectoria.\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>Prompt ↓\nLa respuesta se convierte en Intención ↓\nTrayectoria ↓\nAcciones ↓\nCambios de estado ↓\nEvidencia ↓\nResultado\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>La respuesta final sigue siendo importante, pero es solo el extremo visible de un sistema mucho más grande. Una vez que se permite que la IA afecte al mundo real, \u003Cb>el camino hacia la respuesta se convierte en parte de la respuesta.\u003C\u002Fb>\u003C\u002Fp>",{"time":212,"blocks":213,"version":560},1788955730342,[214,218,221,224,227,231,234,249,252,255,266,269,272,275,279,282,288,296,299,302,324,327,330,333,336,351,354,357,360,363,366,369,372,375,378,381,384,387,390,393,396,399,402,405,408,411,414,417,421,424,427,430,433,436,439,442,445,448,451,454,457,460,463,466,469,472,475,478,486,489,492,495,498,501,504,507,510,513,516,519,522,525,533,536,539,542,545,548,551,554,557],{"data":215,"type":217},{"text":216},"\u003Cb>La salida correcta no demuestra un razonamiento correcto, una ejecución segura o un sistema confiable.\u003C\u002Fb>","paragraph",{"data":219,"type":217},{"text":220},"Durante años, la evaluación de la IA ha estado dominada por una pregunta engañosamente simple: \u003Cb>¿Fue correcta la respuesta?\u003C\u002Fb> Para un chatbot, esto a veces puede ser suficiente. Para un agente capaz de buscar en sistemas, leer datos, llamar herramientas, modificar estados, ejecutar flujos de trabajo, escribir archivos, interactuar con APIs o tomar decisiones, no lo es.",{"data":222,"type":217},{"text":223},"Un agente puede producir la respuesta final correcta mientras hace varias cosas mal en el camino. Puede usar la fuente equivocada, malinterpretar una instrucción y luego compensar el error, acceder a información innecesaria, ejecutar una acción intermedia no autorizada, recuperarse silenciosamente de un error que debería haber provocado una escalada, o dejar efectos secundarios que nadie notó.",{"data":225,"type":217},{"text":226},"Eso crea uno de los problemas centrales de la IA agéntica: \u003Cb>un resultado correcto no demuestra una trayectoria correcta.\u003C\u002Fb>",{"data":228,"type":42},{"text":229,"level":230},"La ilusión del resultado",2,{"data":232,"type":217},{"text":233},"El software tradicional nos da un modelo intuitivo de corrección. La entrada entra en un sistema determinista o mayormente determinista, se ejecuta la lógica, se produce la salida y las pruebas verifican el comportamiento esperado. Los sistemas basados en LLM debilitan esta suposición. Los sistemas agénticos van más allá.",{"data":235,"type":248},{"items":236,"style":247},[237,238,239,240,241,242,243,244,245,246],"interpretación del modelo","contexto recuperado","selección de herramientas","observaciones intermedias","estado externo","acciones previas","planes generados por el modelo","límites de permisos","reintentos y comportamiento de respaldo","interacción humana","unordered","list",{"data":250,"type":217},{"text":251},"Dos ejecuciones que parten de entradas casi idénticas pueden alcanzar el mismo resultado a través de caminos muy diferentes. Si la evaluación observa solo la salida final, la mayor parte del sistema permanece invisible.",{"data":253,"type":217},{"text":254},"Imagina que un agente de IA recibe la instrucción: \u003Ci>Actualiza la dirección de facturación del cliente.\u003C\u002Fi> La dirección finalmente se actualiza correctamente. Una evaluación convencional podría clasificar la tarea como exitosa.",{"data":256,"type":248},{"items":257,"style":265},[258,259,260,261,262,263,264],"El agente busca varios registros de clientes no relacionados.","Recupera más información personal de la necesaria.","Inicialmente modifica la cuenta equivocada.","Nota el error.","Revierte el cambio.","Actualiza la cuenta correcta.","Reporta éxito.","ordered",{"data":267,"type":217},{"text":268},"\u003Cb>Estado final: correcto. Comportamiento del sistema: inaceptable.\u003C\u002Fb> Un benchmark basado solo en resultados le da a esta ejecución un aprobado. Un sistema de aseguramiento de producción no debería.",{"data":270,"type":42},{"text":271,"level":230},"La trayectoria es parte del producto",{"data":273,"type":217},{"text":274},"Por eso la \u003Cb>trayectoria\u003C\u002Fb> de un agente de IA debe convertirse en un objeto de ingeniería de primera clase. Una trayectoria es la secuencia de estados y acciones relevantes entre la solicitud original y el resultado final.",{"data":276,"type":278},{"code":277},"Intención → Contexto → Decisión → Herramienta → Acción → Observación → Decisión → Cambio de estado → Resultado","code",{"data":280,"type":217},{"text":281},"Zachary J. Stevens desarrolla esta idea en \u003Ci>La trayectoria es el sistema\u003C\u002Fi>, argumentando que la evaluación agéntica debe ir más allá de la respuesta final y examinar el camino completo de acción a través de un entorno cambiante.",{"data":283,"type":287},{"text":284,"caption":285,"alignment":286},"Un resultado correcto no excusa una trayectoria inaceptable.","Zachary J. Stevens, La trayectoria es el sistema","left","quote",{"data":289,"type":295},{"link":290,"meta":291},"https:\u002F\u002Fzacharyjstevens.com\u002Fdispatches\u002Fvanguard-signal\u002F009-the-trajectory-is-the-system\u002F",{"image":292,"title":293,"description":294},{},"La trayectoria es el sistema","Zachary J. Stevens — DFEI.009 sobre evaluar sistemas agénticos por su trayectoria completa en lugar de solo el resultado final.","linkTool",{"data":297,"type":217},{"text":298},"La distinción importa enormemente. Por lo tanto, la confiabilidad no es simplemente \u003Cb>salida correcta\u003C\u002Fb>. Está más cerca de \u003Cb>resultado aceptable + trayectoria aceptable + recuperabilidad + evidencia\u003C\u002Fb>.",{"data":300,"type":42},{"text":301,"level":230},"Una respuesta correcta puede ocultar un sistema roto",{"data":303,"type":323},{"content":304,"withHeadings":14},[305,309,313,316,320],[306,307,308],"Agente","Resultado final","Ejecución",[310,311,312],"A","Correcto","Camino correcto",[314,311,315],"B","Camino inseguro",[317,318,319],"C","Incorrecto","Fallo seguro",[321,318,322],"D","Fallo inseguro","table",{"data":325,"type":217},{"text":326},"La mayoría de las evaluaciones basadas en benchmarks recompensan fuertemente a A y B y penalizan a C y D. Operativamente, sin embargo, \u003Cb>B puede ser más peligroso que C\u003C\u002Fb>. El agente C puede reconocer la incertidumbre, detener la ejecución y solicitar revisión humana. El agente B puede producir con confianza resultados correctos mientras viola suposiciones que nadie está monitoreando.",{"data":328,"type":278},{"code":329},"salida exitosa → mayor confianza → permisos más amplios → más automatización → mayor radio de explosión",{"data":331,"type":42},{"text":332,"level":230},"Necesitamos evidencia, no confianza",{"data":334,"type":217},{"text":335},"Uno de los mayores errores en la adopción de IA es tratar la confianza del modelo, la satisfacción del usuario o la tasa de éxito histórica como evidencia de la fiabilidad del sistema. No son equivalentes.",{"data":337,"type":248},{"items":338,"style":247},[339,340,341,342,343,344,345,346,347,348,349,350],"¿Qué recibió el agente?","¿Qué contexto recuperó?","¿Qué herramientas llamó?","¿Por qué se permitió la acción?","¿Qué estado existía antes de la acción?","¿Qué cambió?","¿Qué fallos intermedios ocurrieron?","¿Se reintentó algo?","¿Se requirió aprobación humana?","¿Se podría haber detenido la ejecución?","¿Se puede revertir la acción?","¿Qué modelo, prompt y versiones de herramientas estuvieron involucrados?",{"data":352,"type":217},{"text":353},"Sin estas respuestas, no hay una garantía operativa seria. Solo hay una salida. La observabilidad y la evidencia deben, por lo tanto, diseñarse dentro de la arquitectura del agente en lugar de agregarse después del despliegue.",{"data":355,"type":42},{"text":356,"level":230},"Registrar no es lo mismo que controlar",{"data":358,"type":217},{"text":359},"Las organizaciones a menudo responden: \u003Ci>Todo está registrado.\u003C\u002Fi> Bien. Pero registrar solo no controla nada. Un registro te dice lo que sucedió. Un control determina si algo \u003Cb>puede suceder\u003C\u002Fb>.",{"data":361,"type":278},{"code":362},"El agente solicita DELETE \u002Fcustomer\u002F123 ↓\nAcción registrada ↓\nDELETE ejecutado",{"data":364,"type":217},{"text":365},"Eso proporciona observabilidad. Compáralo con:",{"data":367,"type":278},{"code":368},"El agente solicita DELETE \u002Fcustomer\u002F123 ↓\nEvaluación de políticas ↓\nIdentidad actual verificada ↓\nParámetros de acción actuales comprobados ↓\nUmbral de riesgo evaluado ↓\nAprobación humana si es necesaria ↓\nAcción ejecutada ↓\nResultado verificado ↓\nEvidencia almacenada",{"data":370,"type":217},{"text":371},"Ahora nos acercamos a un sistema de control. La diferencia es arquitectónica, no cosmética.",{"data":373,"type":42},{"text":374,"level":230},"El permiso es necesario, pero no es garantía",{"data":376,"type":217},{"text":377},"Supongamos que un agente tiene permiso para enviar correos electrónicos. El control de acceso responde: \u003Cb>¿Puede este agente enviar correos?\u003C\u002Fb> No responde: \u003Cb>¿Debería este correo en particular enviarse a esta persona en particular con este adjunto en particular en este momento?\u003C\u002Fb>",{"data":379,"type":278},{"code":380},"CONTROL DE CAPACIDAD\n¿Qué se le permite técnicamente hacer al agente? + GARANTÍA DE ACCIÓN\n¿Es apropiada esta acción específica en el estado actual?",{"data":382,"type":217},{"text":383},"RBAC, alcances de OAuth, permisos de API e identidades de agentes definen el espacio de acciones posibles. No prueban que una acción dentro de ese espacio sea apropiada. Una arquitectura de agente sólida necesita ambas capas.",{"data":385,"type":42},{"text":386,"level":230},"El Primer Paso Incorrecto Importa",{"data":388,"type":217},{"text":389},"Cuando un agente falla, la acción incorrecta final a menudo no es donde comenzó el fallo. El fallo real puede haber ocurrido mucho antes.",{"data":391,"type":278},{"code":392},"Recuperación incorrecta ↓\nSuposición incorrecta ↓\nRazonamiento plausible ↓\nLlamada a herramienta válida ↓\nAcción incorrecta",{"data":394,"type":217},{"text":395},"Si investigamos solo la acción final, corregimos el síntoma. Si inspeccionamos la trayectoria, podemos identificar el \u003Cb>primer paso incorrecto\u003C\u002Fb>. Eso convierte un fallo no atribuible en un problema de ingeniería concreto.",{"data":397,"type":42},{"text":398,"level":230},"Las Pruebas de Agentes Deben Ir Más Allá de las Pruebas de Prompts",{"data":400,"type":217},{"text":401},"Los prompts importan, pero el comportamiento de un agente en producción surge de un sistema completo.",{"data":403,"type":278},{"code":404},"MODELO\n+\nPROMPT DEL SISTEMA\n+\nCONTEXTO\n+\nMEMORIA\n+\nRECUPERACIÓN\n+\nHERRAMIENTAS\n+\nPERMISOS\n+\nFLUJO DE TRABAJO\n+\nESTADO EXTERNO\n+\nLÓGICA DE CONTROL",{"data":406,"type":217},{"text":407},"Cambiar cualquiera de estos puede cambiar la trayectoria. Por lo tanto, versionar solo el prompt es insuficiente.",{"data":409,"type":278},{"code":410},"versión_del_modelo\nversión_del_prompt\nversión_de_la_herramienta\nversión_de_la_política\nversión_de_recuperación\nversión_del_flujo_de_trabajo\nestado_del_entorno\nid_de_ejecución",{"data":412,"type":42},{"text":413,"level":230},"Los Criterios de Aceptación para Agentes Deben Incluir Comportamiento",{"data":415,"type":217},{"text":416},"Los criterios de aceptación tradicionales a menudo se ven así: \u003Ci>Dado X, el sistema produce Y.\u003C\u002Fi> Para sistemas agénticos, eso es incompleto. Los criterios de aceptación también deben definir restricciones sobre la trayectoria.",{"data":418,"type":42},{"text":419,"level":420},"Resultado",3,{"data":422,"type":217},{"text":423},"La dirección del cliente se actualiza correctamente.",{"data":425,"type":42},{"text":426,"level":420},"Autorización",{"data":428,"type":217},{"text":429},"El agente modifica solo al cliente explícitamente seleccionado.",{"data":431,"type":42},{"text":432,"level":420},"Acceso a datos",{"data":434,"type":217},{"text":435},"No se accede a registros de clientes no relacionados.",{"data":437,"type":42},{"text":438,"level":420},"Herramientas",{"data":440,"type":217},{"text":441},"Solo se utilizan operaciones CRM aprobadas.",{"data":443,"type":42},{"text":444,"level":420},"Verificación",{"data":446,"type":217},{"text":447},"La nueva dirección se lee y se compara con el valor solicitado.",{"data":449,"type":42},{"text":450,"level":420},"Fallo",{"data":452,"type":217},{"text":453},"La resolución ambigua de identidad detiene la ejecución.",{"data":455,"type":42},{"text":456,"level":420},"Autoridad humana",{"data":458,"type":217},{"text":459},"Un humano puede rechazar la modificación antes de la ejecución cuando los umbrales de riesgo requieren aprobación.",{"data":461,"type":42},{"text":462,"level":420},"Evidencia",{"data":464,"type":217},{"text":465},"La ejecución deja un rastro suficiente para reconstruir la decisión y la transición de estado.",{"data":467,"type":42},{"text":468,"level":420},"Recuperación",{"data":470,"type":217},{"text":471},"El valor anterior sigue siendo recuperable.",{"data":473,"type":42},{"text":474,"level":230},"El humano en el circuito no es suficiente",{"data":476,"type":217},{"text":477},"Agregar una casilla de aprobación humana no resuelve automáticamente el problema. Un humano solo puede controlar a un agente si la persona tiene visibilidad, autoridad, tiempo, contexto y capacidad de recuperación.",{"data":479,"type":248},{"items":480,"style":247},[481,482,483,484,485],"\u003Cb>Visibilidad:\u003C\u002Fb> suficiente información para entender lo que está sucediendo.","\u003Cb>Autoridad:\u003C\u002Fb> capacidad real de detener o modificar la acción.","\u003Cb>Tiempo:\u003C\u002Fb> intervención antes de que ocurra la consecuencia.","\u003Cb>Contexto:\u003C\u002Fb> evidencia suficiente para tomar la decisión.","\u003Cb>Capacidad de recuperación:\u003C\u002Fb> capacidad de revertir o reparar la acción.",{"data":487,"type":217},{"text":488},"Un usuario que hace clic en \u003Cb>Aprobar\u003C\u002Fb> sobre algo que no puede inspeccionar de manera significativa no es una gobernanza sólida. Es teatro de aprobación.",{"data":490,"type":42},{"text":491,"level":230},"La reversión debe convertirse en una capacidad nativa de la IA",{"data":493,"type":217},{"text":494},"El despliegue de software tradicional nos ha enseñado algo valioso: \u003Cb>Nunca implementes lo que no puedes revertir.\u003C\u002Fb> Deberíamos aplicar el mismo principio a las acciones agénticas.",{"data":496,"type":278},{"code":497},"REVERSIBLE\nPuede deshacerse automáticamente. COMPENSABLE\nNo puede deshacerse directamente, pero puede ejecutar una acción compensatoria. IRREVERSIBLE\nNo puede restaurar de manera confiable el estado anterior.",{"data":499,"type":217},{"text":500},"Cuanto mayor es la irreversibilidad, más fuerte debe ser el requisito de control.",{"data":502,"type":278},{"code":503},"Leer documento público → baja consecuencia\nCrear borrador → reversible\nModificar registro CRM → reversible pero con consecuencias\nEnviar correo externo → prácticamente irreversible\nTransferir dinero → alta consecuencia\nEliminar datos de producción → potencialmente catastrófico",{"data":505,"type":42},{"text":506,"level":230},"El Agente Necesita un Plano de Control",{"data":508,"type":278},{"code":509},"USUARIO \u002F INTENCIÓN DEL SISTEMA │ ▼ AGENTE DE IA │ acción propuesta │ ▼ ┌───────────────────┐ │ PLANO DE CONTROL │ ├───────────────────┤ │ Identidad │ │ Autorización │ │ Política │ │ Riesgo │ │ Estado │ │ Evidencia │ │ Autoridad humana │ │ Reversión │ └───────────────────┘ │ ¿aprobado? \u002F \\ NO SÍ │ │ DETENER ▼ HERRAMIENTA │ ▼ CAMBIO DE ESTADO │ ▼ VERIFICACIÓN",{"data":511,"type":217},{"text":512},"\u003Cb>El LLM debe proponer. El plano de control debe gobernar.\u003C\u002Fb> Esa separación es crucial. El modelo no debe ser la autoridad última que determine si su propia acción propuesta de alto impacto es segura.",{"data":514,"type":42},{"text":515,"level":230},"De los Benchmarks a la Confianza Operativa",{"data":517,"type":217},{"text":518},"Los benchmarks siguen siendo útiles. Nos informan sobre la capacidad, comparan modelos, detectan regresiones y ayudan a estimar el rendimiento esperado. Pero la evaluación de la capacidad y la confianza operativa responden a preguntas diferentes.",{"data":520,"type":217},{"text":521},"Un benchmark pregunta: \u003Cb>¿Puede el sistema hacer esto?\u003C\u002Fb> La garantía operativa pregunta: \u003Cb>¿Podemos permitir que el sistema haga esto aquí, bajo estas condiciones, con estos permisos y consecuencias?\u003C\u002Fb>",{"data":523,"type":42},{"text":524,"level":230},"La Fiabilidad Debe Medirse como una Propiedad del Sistema",{"data":526,"type":248},{"items":527,"style":265},[528,529,530,531,532],"\u003Cb>Corrección del resultado:\u003C\u002Fb> ¿Produjo el sistema el resultado esperado?","\u003Cb>Corrección de la trayectoria:\u003C\u002Fb> ¿Siguió una ruta aceptable?","\u003Cb>Integridad del control:\u003C\u002Fb> ¿Se respetaron los límites de autorización, política e intervención?","\u003Cb>Recuperabilidad:\u003C\u002Fb> ¿Pueden los fallos contenerse, revertirse o repararse?","\u003Cb>Completitud de la evidencia:\u003C\u002Fb> ¿Puede la ejecución reconstruirse y auditarse?",{"data":534,"type":278},{"code":535},"Fiabilidad Operativa\n=\nResultado × Trayectoria × Control × Recuperabilidad × Evidencia",{"data":537,"type":217},{"text":538},"La multiplicación es intencional. Si una dimensión crítica se acerca a cero, una puntuación alta en otra parte no debería ocultarlo. Un resultado perfectamente correcto con integridad de autorización cero no es un sistema fiable al 80%. Es una ejecución inaceptable que casualmente produjo la respuesta correcta.",{"data":540,"type":42},{"text":541,"level":230},"El Éxito es a Veces el Fracaso Más Peligroso",{"data":543,"type":217},{"text":544},"Los fracasos atraen la atención. El éxito a menudo no. Eso hace que las trayectorias de agentes exitosas pero no controladas sean particularmente peligrosas. Un fracaso evidente crea un incidente. Un defecto oculto en la trayectoria crea \u003Cb>confianza\u003C\u002Fb>. Y la confianza expande la autonomía.",{"data":546,"type":217},{"text":547},"Las organizaciones no deberían por tanto investigar solo \u003Ci>¿Por qué falló el agente?\u003C\u002Fi> Deberían preguntarse periódicamente: \u003Cb>¿Por qué tuvo éxito el agente?\u003C\u002Fb> ¿Tuvo éxito porque la arquitectura restringió y verificó de manera fiable la ejecución, o porque esta vez nada salió mal?",{"data":549,"type":42},{"text":550,"level":230},"Conclusión",{"data":552,"type":217},{"text":553},"La industria se mueve rápidamente desde la IA que \u003Cb>responde\u003C\u002Fb> hacia la IA que \u003Cb>actúa\u003C\u002Fb>. Esa transición cambia lo que significa fiabilidad. Para un sistema de respuestas, evaluar la respuesta puede ser a menudo suficiente. Para un sistema de acciones, debemos evaluar la trayectoria.",{"data":555,"type":278},{"code":556},"Prompt ↓\nLa respuesta se convierte en Intención ↓\nTrayectoria ↓\nAcciones ↓\nCambios de estado ↓\nEvidencia ↓\nResultado",{"data":558,"type":217},{"text":559},"La respuesta final sigue siendo importante, pero es solo el extremo visible de un sistema mucho más grande. Una vez que se permite que la IA afecte al mundo real, \u003Cb>el camino hacia la respuesta se convierte en parte de la respuesta.\u003C\u002Fb>","2.31","Una salida correcta no demuestra un razonamiento correcto, una ejecución segura ni un sistema confiable.","\u002Fuploads\u002F2026\u002F09\u002Fai-agent-reliability-why-the-final-answer-is-not-enough-1788955466306-pl0qhz.webp","ai-agent-reliability-why-the-final-answer-is-not-enough-1788955466306-pl0qhz","PUBLISHED","2026-09-09T04:01:00.000Z","2026-09-09T12:01:07.219Z","2026-09-09T13:08:53.481Z",{"en":569,"de":570,"sr":571,"es":572,"fr":573,"it":574,"ru":575,"zh":576},"\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough","\u002Fde\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough","\u002Fsr\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough","\u002Fes\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough","\u002Ffr\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough","\u002Fit\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough","\u002Fru\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough","\u002Fzh\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough",[578,582],{"id":579,"name":580,"slug":581},58,"Evaluación y compuertas de calidad","evaluation",{"id":583,"name":584,"slug":585},59,"Gobernanza y auditoría","governance",{"id":587,"login":588,"email":589,"displayName":590},"20","rooth8233","aleksandar@stajic.de","Aleksandar Stajić",[592,931],{"lang":593,"title":594,"content":595,"contentJson":596,"excerpt":930},"en","AI Agent Reliability: Why the Final Answer Is Not Enough","{\"time\":1788955485785,\"blocks\":[{\"data\":{\"text\":\"\u003Cb>Correct output does not prove correct reasoning, safe execution, or a trustworthy system.\u003C\u002Fb>\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"For years, AI evaluation has been dominated by a deceptively simple question: \u003Cb>Was the answer correct?\u003C\u002Fb> For a chatbot, this may sometimes be sufficient. For an agent capable of searching systems, reading data, calling tools, modifying state, executing workflows, writing files, interacting with APIs, or making decisions, it is not.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"An agent can produce the correct final answer while doing several things wrong on the way there. It can use the wrong source, misunderstand an instruction and later compensate for the mistake, access unnecessary information, execute an unauthorized intermediate action, silently recover from an error that should have triggered escalation, or leave behind side effects nobody noticed.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"That creates one of the central problems of agentic AI: \u003Cb>a correct outcome does not prove a correct trajectory.\u003C\u002Fb>\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"The Outcome Illusion\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"Traditional software gives us an intuitive model of correctness. Input enters a deterministic or mostly deterministic system, logic is executed, output is produced, and tests verify expected behavior. LLM-based systems weaken this assumption. Agentic systems go further.\"},\"type\":\"paragraph\"},{\"data\":{\"items\":[\"model interpretation\",\"retrieved context\",\"tool selection\",\"intermediate observations\",\"external state\",\"previous actions\",\"model-generated plans\",\"permission boundaries\",\"retries and fallback behavior\",\"human interaction\"],\"style\":\"unordered\"},\"type\":\"list\"},{\"data\":{\"text\":\"Two executions starting from nearly identical inputs may reach the same result through very different paths. If evaluation observes only the final output, most of the system remains invisible.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Imagine an AI agent receives the instruction: \u003Ci>Update the customer's billing address.\u003C\u002Fi> The address is ultimately updated correctly. A conventional evaluation might classify the task as successful.\"},\"type\":\"paragraph\"},{\"data\":{\"items\":[\"The agent searches several unrelated customer records.\",\"It retrieves more personal information than required.\",\"It initially modifies the wrong account.\",\"It notices the mistake.\",\"It reverses the change.\",\"It updates the correct account.\",\"It reports success.\"],\"style\":\"ordered\"},\"type\":\"list\"},{\"data\":{\"text\":\"\u003Cb>Final state: correct. System behavior: unacceptable.\u003C\u002Fb> An outcome-only benchmark gives this execution a pass. A production assurance system should not.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"The Trajectory Is Part of the Product\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"This is why the \u003Cb>trajectory\u003C\u002Fb> of an AI agent must become a first-class engineering object. A trajectory is the sequence of relevant states and actions between the original request and the final result.\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"Intent → Context → Decision → Tool → Action → Observation → Decision → State change → Result\"},\"type\":\"code\"},{\"data\":{\"text\":\"Zachary J. Stevens develops this idea in \u003Ci>The Trajectory Is the System\u003C\u002Fi>, arguing that agentic evaluation must move beyond the final answer and examine the complete path of action through a changing environment.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"A correct outcome does not excuse an unacceptable trajectory.\",\"caption\":\"Zachary J. Stevens, The Trajectory Is the System\",\"alignment\":\"left\"},\"type\":\"quote\"},{\"data\":{\"link\":\"https:\u002F\u002Fzacharyjstevens.com\u002Fdispatches\u002Fvanguard-signal\u002F009-the-trajectory-is-the-system\u002F\",\"meta\":{\"image\":{},\"title\":\"The Trajectory Is the System\",\"description\":\"Zachary J. Stevens — DFEI.009 on evaluating agentic systems by their complete trajectory rather than only the final outcome.\"}},\"type\":\"linkTool\"},{\"data\":{\"text\":\"The distinction matters enormously. Reliability is therefore not simply \u003Cb>correct output\u003C\u002Fb>. It is closer to \u003Cb>acceptable outcome + acceptable trajectory + recoverability + evidence\u003C\u002Fb>.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"A Correct Answer Can Hide a Broken System\",\"level\":2},\"type\":\"header\"},{\"data\":{\"content\":[[\"Agent\",\"Final result\",\"Execution\"],[\"A\",\"Correct\",\"Correct path\"],[\"B\",\"Correct\",\"Unsafe path\"],[\"C\",\"Incorrect\",\"Safe failure\"],[\"D\",\"Incorrect\",\"Unsafe failure\"]],\"withHeadings\":true},\"type\":\"table\"},{\"data\":{\"text\":\"Most benchmark-driven evaluation strongly rewards A and B and penalizes C and D. Operationally, however, \u003Cb>B can be more dangerous than C\u003C\u002Fb>. Agent C may recognize uncertainty, stop execution and request human review. Agent B may confidently produce correct results while violating assumptions that nobody is monitoring.\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"successful output → increased trust → broader permissions → more automation → larger blast radius\"},\"type\":\"code\"},{\"data\":{\"text\":\"We Need Evidence, Not Confidence\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"One of the biggest mistakes in AI adoption is treating model confidence, user satisfaction or historical success rate as evidence of system reliability. They are not equivalent.\"},\"type\":\"paragraph\"},{\"data\":{\"items\":[\"What did the agent receive?\",\"What context did it retrieve?\",\"Which tools did it call?\",\"Why was the action allowed?\",\"What state existed before the action?\",\"What changed?\",\"Which intermediate failures occurred?\",\"Was anything retried?\",\"Was human approval required?\",\"Could execution have been stopped?\",\"Can the action be reversed?\",\"Which model, prompt and tool versions were involved?\"],\"style\":\"unordered\"},\"type\":\"list\"},{\"data\":{\"text\":\"Without these answers, there is no serious operational assurance. There is only an output. Observability and evidence must therefore be designed into agent architecture rather than added after deployment.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Logging Is Not the Same as Control\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"Organizations often respond: \u003Ci>Everything is logged.\u003C\u002Fi> Good. But logging alone does not control anything. A log tells you what happened. A control determines whether something \u003Cb>may happen\u003C\u002Fb>.\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"Agent requests DELETE \u002Fcustomer\u002F123 ↓\\nAction logged ↓\\nDELETE executed\"},\"type\":\"code\"},{\"data\":{\"text\":\"That gives observability. Compare it with:\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"Agent requests DELETE \u002Fcustomer\u002F123 ↓\\nPolicy evaluation ↓\\nCurrent identity verified ↓\\nCurrent action parameters checked ↓\\nRisk threshold evaluated ↓\\nHuman approval if required ↓\\nAction executed ↓\\nResult verified ↓\\nEvidence stored\"},\"type\":\"code\"},{\"data\":{\"text\":\"Now we are approaching a control system. The difference is architectural, not cosmetic.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Permission Is Necessary — but It Is Not Assurance\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"Suppose an agent has permission to send email. Access control answers: \u003Cb>Can this agent send email?\u003C\u002Fb> It does not answer: \u003Cb>Should this particular email be sent to this particular person with this particular attachment right now?\u003C\u002Fb>\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"CAPABILITY CONTROL\\nWhat is the agent technically allowed to do? + ACTION ASSURANCE\\nIs this specific action appropriate in the current state?\"},\"type\":\"code\"},{\"data\":{\"text\":\"RBAC, OAuth scopes, API permissions and agent identities define the space of possible actions. They do not prove that an action inside that space is appropriate. Strong agent architecture needs both layers.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"The First Wrong Step Matters\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"When an agent fails, the final incorrect action is often not where the failure started. The real failure may have happened much earlier.\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"Wrong retrieval ↓\\nWrong assumption ↓\\nPlausible reasoning ↓\\nValid tool call ↓\\nWrong action\"},\"type\":\"code\"},{\"data\":{\"text\":\"If we investigate only the final action, we fix the symptom. If we inspect the trajectory, we can identify the \u003Cb>first wrong step\u003C\u002Fb>. That turns an unattributable failure into a concrete engineering problem.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Agent Testing Must Move Beyond Prompt Testing\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"Prompts matter, but production agent behavior emerges from an entire system.\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"MODEL\\n+\\nSYSTEM PROMPT\\n+\\nCONTEXT\\n+\\nMEMORY\\n+\\nRETRIEVAL\\n+\\nTOOLS\\n+\\nPERMISSIONS\\n+\\nWORKFLOW\\n+\\nEXTERNAL STATE\\n+\\nCONTROL LOGIC\"},\"type\":\"code\"},{\"data\":{\"text\":\"Changing any one of these can change the trajectory. Therefore versioning only the prompt is insufficient.\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"model_version\\nprompt_version\\ntool_version\\npolicy_version\\nretrieval_version\\nworkflow_version\\nenvironment_state\\nexecution_id\"},\"type\":\"code\"},{\"data\":{\"text\":\"Acceptance Criteria for Agents Must Include Behavior\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"Traditional acceptance criteria often look like this: \u003Ci>Given X, the system produces Y.\u003C\u002Fi> For agentic systems, that is incomplete. Acceptance criteria should also define constraints on the trajectory.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Outcome\",\"level\":3},\"type\":\"header\"},{\"data\":{\"text\":\"The customer's address is updated correctly.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Authorization\",\"level\":3},\"type\":\"header\"},{\"data\":{\"text\":\"The agent modifies only the explicitly selected customer.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Data access\",\"level\":3},\"type\":\"header\"},{\"data\":{\"text\":\"No unrelated customer records are accessed.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Tools\",\"level\":3},\"type\":\"header\"},{\"data\":{\"text\":\"Only approved CRM operations are used.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Verification\",\"level\":3},\"type\":\"header\"},{\"data\":{\"text\":\"The new address is read back and compared with the requested value.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Failure\",\"level\":3},\"type\":\"header\"},{\"data\":{\"text\":\"Ambiguous identity resolution stops execution.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Human authority\",\"level\":3},\"type\":\"header\"},{\"data\":{\"text\":\"A human can reject the modification before execution when risk thresholds require approval.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Evidence\",\"level\":3},\"type\":\"header\"},{\"data\":{\"text\":\"The execution leaves a trace sufficient to reconstruct the decision and state transition.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Recovery\",\"level\":3},\"type\":\"header\"},{\"data\":{\"text\":\"The previous value remains recoverable.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Human-in-the-Loop Is Not Enough\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"Adding a human approval box does not automatically solve the problem. A human can only control an agent if the person has visibility, authority, time, context and recovery capability.\"},\"type\":\"paragraph\"},{\"data\":{\"items\":[\"\u003Cb>Visibility:\u003C\u002Fb> enough information to understand what is happening.\",\"\u003Cb>Authority:\u003C\u002Fb> actual ability to stop or modify the action.\",\"\u003Cb>Time:\u003C\u002Fb> intervention before the consequence occurs.\",\"\u003Cb>Context:\u003C\u002Fb> sufficient evidence to make the decision.\",\"\u003Cb>Recovery capability:\u003C\u002Fb> ability to reverse or repair the action.\"],\"style\":\"unordered\"},\"type\":\"list\"},{\"data\":{\"text\":\"A user clicking \u003Cb>Approve\u003C\u002Fb> on something they cannot meaningfully inspect is not strong governance. It is approval theater.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Rollback Must Become a Native AI Capability\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"Traditional software deployment has taught us something valuable: \u003Cb>Never deploy what you cannot roll back.\u003C\u002Fb> We should apply the same principle to agentic actions.\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"REVERSIBLE\\nCan automatically undo. COMPENSATABLE\\nCannot undo directly but can execute a compensating action. IRREVERSIBLE\\nCannot reliably restore the previous state.\"},\"type\":\"code\"},{\"data\":{\"text\":\"The higher the irreversibility, the stronger the control requirement should become.\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"Read public document → low consequence\\nCreate draft → reversible\\nModify CRM record → reversible but consequential\\nSend external email → practically irreversible\\nTransfer money → high consequence\\nDelete production data → potentially catastrophic\"},\"type\":\"code\"},{\"data\":{\"text\":\"The Agent Needs a Control Plane\",\"level\":2},\"type\":\"header\"},{\"data\":{\"code\":\"USER \u002F SYSTEM INTENT │ ▼ AI AGENT │ proposed action │ ▼ ┌───────────────────┐ │ CONTROL PLANE │ ├───────────────────┤ │ Identity │ │ Authorization │ │ Policy │ │ Risk │ │ State │ │ Evidence │ │ Human authority │ │ Rollback │ └───────────────────┘ │ approved? \u002F \\\\ NO YES │ │ STOP ▼ TOOL │ ▼ STATE CHANGE │ ▼ VERIFICATION\"},\"type\":\"code\"},{\"data\":{\"text\":\"\u003Cb>The LLM should propose. The control plane should govern.\u003C\u002Fb> That separation is crucial. The model should not be the ultimate authority determining whether its own proposed high-impact action is safe.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"From Benchmarks to Operational Trust\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"Benchmarks remain useful. They tell us about capability, compare models, detect regressions and help estimate expected performance. But capability evaluation and operational trust answer different questions.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"A benchmark asks: \u003Cb>Can the system do this?\u003C\u002Fb> Operational assurance asks: \u003Cb>Can we allow the system to do this here, under these conditions, with these permissions and consequences?\u003C\u002Fb>\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Reliability Should Be Measured as a System Property\",\"level\":2},\"type\":\"header\"},{\"data\":{\"items\":[\"\u003Cb>Outcome correctness:\u003C\u002Fb> Did the system produce the expected result?\",\"\u003Cb>Trajectory correctness:\u003C\u002Fb> Did it follow an acceptable path?\",\"\u003Cb>Control integrity:\u003C\u002Fb> Were authorization, policy and intervention boundaries respected?\",\"\u003Cb>Recoverability:\u003C\u002Fb> Can failures be contained, reversed or repaired?\",\"\u003Cb>Evidence completeness:\u003C\u002Fb> Can the execution be reconstructed and audited?\"],\"style\":\"ordered\"},\"type\":\"list\"},{\"data\":{\"code\":\"Operational Reliability\\n=\\nOutcome × Trajectory × Control × Recoverability × Evidence\"},\"type\":\"code\"},{\"data\":{\"text\":\"The multiplication is intentional. If one critical dimension approaches zero, a high score elsewhere should not hide it. A perfectly correct output with zero authorization integrity is not an 80% reliable system. It is an unacceptable execution that happened to produce the right answer.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Success Is Sometimes the Most Dangerous Failure\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"Failures attract attention. Success often does not. That makes successful but uncontrolled agent trajectories particularly dangerous. An obvious failure creates an incident. A hidden trajectory defect creates \u003Cb>confidence\u003C\u002Fb>. And confidence expands autonomy.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Organizations should therefore not only investigate \u003Ci>Why did the agent fail?\u003C\u002Fi> They should periodically ask: \u003Cb>Why did the agent succeed?\u003C\u002Fb> Did it succeed because the architecture reliably constrained and verified the execution, or because nothing went wrong this time?\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Conclusion\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"The industry is moving rapidly from AI that \u003Cb>answers\u003C\u002Fb> toward AI that \u003Cb>acts\u003C\u002Fb>. That transition changes what reliability means. For an answer system, evaluating the answer may often be sufficient. For an action system, we must evaluate the path.\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"Prompt ↓\\nResponse becomes Intent ↓\\nTrajectory ↓\\nActions ↓\\nState changes ↓\\nEvidence ↓\\nOutcome\"},\"type\":\"code\"},{\"data\":{\"text\":\"The final answer remains important, but it is only the visible end of a much larger system. Once AI is allowed to affect the real world, \u003Cb>the path to the answer becomes part of the answer.\u003C\u002Fb>\"},\"type\":\"paragraph\"}],\"version\":\"2.31.0\"}",{"time":597,"blocks":598,"version":929},1788955485785,[599,602,605,608,611,614,617,630,633,636,646,649,652,655,658,661,665,671,674,677,694,697,700,703,706,721,724,727,730,733,736,739,742,745,748,751,754,757,760,763,766,769,772,775,778,781,784,787,790,793,796,799,802,805,808,811,814,817,820,823,826,829,832,835,838,841,844,847,855,858,861,864,867,870,873,876,879,882,885,888,891,894,902,905,908,911,914,917,920,923,926],{"data":600,"type":217},{"text":601},"\u003Cb>Correct output does not prove correct reasoning, safe execution, or a trustworthy system.\u003C\u002Fb>",{"data":603,"type":217},{"text":604},"For years, AI evaluation has been dominated by a deceptively simple question: \u003Cb>Was the answer correct?\u003C\u002Fb> For a chatbot, this may sometimes be sufficient. For an agent capable of searching systems, reading data, calling tools, modifying state, executing workflows, writing files, interacting with APIs, or making decisions, it is not.",{"data":606,"type":217},{"text":607},"An agent can produce the correct final answer while doing several things wrong on the way there. It can use the wrong source, misunderstand an instruction and later compensate for the mistake, access unnecessary information, execute an unauthorized intermediate action, silently recover from an error that should have triggered escalation, or leave behind side effects nobody noticed.",{"data":609,"type":217},{"text":610},"That creates one of the central problems of agentic AI: \u003Cb>a correct outcome does not prove a correct trajectory.\u003C\u002Fb>",{"data":612,"type":42},{"text":613,"level":230},"The Outcome Illusion",{"data":615,"type":217},{"text":616},"Traditional software gives us an intuitive model of correctness. Input enters a deterministic or mostly deterministic system, logic is executed, output is produced, and tests verify expected behavior. LLM-based systems weaken this assumption. Agentic systems go further.",{"data":618,"type":248},{"items":619,"style":247},[620,621,622,623,624,625,626,627,628,629],"model interpretation","retrieved context","tool selection","intermediate observations","external state","previous actions","model-generated plans","permission boundaries","retries and fallback behavior","human interaction",{"data":631,"type":217},{"text":632},"Two executions starting from nearly identical inputs may reach the same result through very different paths. If evaluation observes only the final output, most of the system remains invisible.",{"data":634,"type":217},{"text":635},"Imagine an AI agent receives the instruction: \u003Ci>Update the customer's billing address.\u003C\u002Fi> The address is ultimately updated correctly. A conventional evaluation might classify the task as successful.",{"data":637,"type":248},{"items":638,"style":265},[639,640,641,642,643,644,645],"The agent searches several unrelated customer records.","It retrieves more personal information than required.","It initially modifies the wrong account.","It notices the mistake.","It reverses the change.","It updates the correct account.","It reports success.",{"data":647,"type":217},{"text":648},"\u003Cb>Final state: correct. System behavior: unacceptable.\u003C\u002Fb> An outcome-only benchmark gives this execution a pass. A production assurance system should not.",{"data":650,"type":42},{"text":651,"level":230},"The Trajectory Is Part of the Product",{"data":653,"type":217},{"text":654},"This is why the \u003Cb>trajectory\u003C\u002Fb> of an AI agent must become a first-class engineering object. A trajectory is the sequence of relevant states and actions between the original request and the final result.",{"data":656,"type":278},{"code":657},"Intent → Context → Decision → Tool → Action → Observation → Decision → State change → Result",{"data":659,"type":217},{"text":660},"Zachary J. Stevens develops this idea in \u003Ci>The Trajectory Is the System\u003C\u002Fi>, arguing that agentic evaluation must move beyond the final answer and examine the complete path of action through a changing environment.",{"data":662,"type":287},{"text":663,"caption":664,"alignment":286},"A correct outcome does not excuse an unacceptable trajectory.","Zachary J. Stevens, The Trajectory Is the System",{"data":666,"type":295},{"link":290,"meta":667},{"image":668,"title":669,"description":670},{},"The Trajectory Is the System","Zachary J. Stevens — DFEI.009 on evaluating agentic systems by their complete trajectory rather than only the final outcome.",{"data":672,"type":217},{"text":673},"The distinction matters enormously. Reliability is therefore not simply \u003Cb>correct output\u003C\u002Fb>. It is closer to \u003Cb>acceptable outcome + acceptable trajectory + recoverability + evidence\u003C\u002Fb>.",{"data":675,"type":42},{"text":676,"level":230},"A Correct Answer Can Hide a Broken System",{"data":678,"type":323},{"content":679,"withHeadings":14},[680,684,687,689,692],[681,682,683],"Agent","Final result","Execution",[310,685,686],"Correct","Correct path",[314,685,688],"Unsafe path",[317,690,691],"Incorrect","Safe failure",[321,690,693],"Unsafe failure",{"data":695,"type":217},{"text":696},"Most benchmark-driven evaluation strongly rewards A and B and penalizes C and D. Operationally, however, \u003Cb>B can be more dangerous than C\u003C\u002Fb>. Agent C may recognize uncertainty, stop execution and request human review. Agent B may confidently produce correct results while violating assumptions that nobody is monitoring.",{"data":698,"type":278},{"code":699},"successful output → increased trust → broader permissions → more automation → larger blast radius",{"data":701,"type":42},{"text":702,"level":230},"We Need Evidence, Not Confidence",{"data":704,"type":217},{"text":705},"One of the biggest mistakes in AI adoption is treating model confidence, user satisfaction or historical success rate as evidence of system reliability. They are not equivalent.",{"data":707,"type":248},{"items":708,"style":247},[709,710,711,712,713,714,715,716,717,718,719,720],"What did the agent receive?","What context did it retrieve?","Which tools did it call?","Why was the action allowed?","What state existed before the action?","What changed?","Which intermediate failures occurred?","Was anything retried?","Was human approval required?","Could execution have been stopped?","Can the action be reversed?","Which model, prompt and tool versions were involved?",{"data":722,"type":217},{"text":723},"Without these answers, there is no serious operational assurance. There is only an output. Observability and evidence must therefore be designed into agent architecture rather than added after deployment.",{"data":725,"type":42},{"text":726,"level":230},"Logging Is Not the Same as Control",{"data":728,"type":217},{"text":729},"Organizations often respond: \u003Ci>Everything is logged.\u003C\u002Fi> Good. But logging alone does not control anything. A log tells you what happened. A control determines whether something \u003Cb>may happen\u003C\u002Fb>.",{"data":731,"type":278},{"code":732},"Agent requests DELETE \u002Fcustomer\u002F123 ↓\nAction logged ↓\nDELETE executed",{"data":734,"type":217},{"text":735},"That gives observability. Compare it with:",{"data":737,"type":278},{"code":738},"Agent requests DELETE \u002Fcustomer\u002F123 ↓\nPolicy evaluation ↓\nCurrent identity verified ↓\nCurrent action parameters checked ↓\nRisk threshold evaluated ↓\nHuman approval if required ↓\nAction executed ↓\nResult verified ↓\nEvidence stored",{"data":740,"type":217},{"text":741},"Now we are approaching a control system. The difference is architectural, not cosmetic.",{"data":743,"type":42},{"text":744,"level":230},"Permission Is Necessary — but It Is Not Assurance",{"data":746,"type":217},{"text":747},"Suppose an agent has permission to send email. Access control answers: \u003Cb>Can this agent send email?\u003C\u002Fb> It does not answer: \u003Cb>Should this particular email be sent to this particular person with this particular attachment right now?\u003C\u002Fb>",{"data":749,"type":278},{"code":750},"CAPABILITY CONTROL\nWhat is the agent technically allowed to do? + ACTION ASSURANCE\nIs this specific action appropriate in the current state?",{"data":752,"type":217},{"text":753},"RBAC, OAuth scopes, API permissions and agent identities define the space of possible actions. They do not prove that an action inside that space is appropriate. Strong agent architecture needs both layers.",{"data":755,"type":42},{"text":756,"level":230},"The First Wrong Step Matters",{"data":758,"type":217},{"text":759},"When an agent fails, the final incorrect action is often not where the failure started. The real failure may have happened much earlier.",{"data":761,"type":278},{"code":762},"Wrong retrieval ↓\nWrong assumption ↓\nPlausible reasoning ↓\nValid tool call ↓\nWrong action",{"data":764,"type":217},{"text":765},"If we investigate only the final action, we fix the symptom. If we inspect the trajectory, we can identify the \u003Cb>first wrong step\u003C\u002Fb>. That turns an unattributable failure into a concrete engineering problem.",{"data":767,"type":42},{"text":768,"level":230},"Agent Testing Must Move Beyond Prompt Testing",{"data":770,"type":217},{"text":771},"Prompts matter, but production agent behavior emerges from an entire system.",{"data":773,"type":278},{"code":774},"MODEL\n+\nSYSTEM PROMPT\n+\nCONTEXT\n+\nMEMORY\n+\nRETRIEVAL\n+\nTOOLS\n+\nPERMISSIONS\n+\nWORKFLOW\n+\nEXTERNAL STATE\n+\nCONTROL LOGIC",{"data":776,"type":217},{"text":777},"Changing any one of these can change the trajectory. Therefore versioning only the prompt is insufficient.",{"data":779,"type":278},{"code":780},"model_version\nprompt_version\ntool_version\npolicy_version\nretrieval_version\nworkflow_version\nenvironment_state\nexecution_id",{"data":782,"type":42},{"text":783,"level":230},"Acceptance Criteria for Agents Must Include Behavior",{"data":785,"type":217},{"text":786},"Traditional acceptance criteria often look like this: \u003Ci>Given X, the system produces Y.\u003C\u002Fi> For agentic systems, that is incomplete. Acceptance criteria should also define constraints on the trajectory.",{"data":788,"type":42},{"text":789,"level":420},"Outcome",{"data":791,"type":217},{"text":792},"The customer's address is updated correctly.",{"data":794,"type":42},{"text":795,"level":420},"Authorization",{"data":797,"type":217},{"text":798},"The agent modifies only the explicitly selected customer.",{"data":800,"type":42},{"text":801,"level":420},"Data access",{"data":803,"type":217},{"text":804},"No unrelated customer records are accessed.",{"data":806,"type":42},{"text":807,"level":420},"Tools",{"data":809,"type":217},{"text":810},"Only approved CRM operations are used.",{"data":812,"type":42},{"text":813,"level":420},"Verification",{"data":815,"type":217},{"text":816},"The new address is read back and compared with the requested value.",{"data":818,"type":42},{"text":819,"level":420},"Failure",{"data":821,"type":217},{"text":822},"Ambiguous identity resolution stops execution.",{"data":824,"type":42},{"text":825,"level":420},"Human authority",{"data":827,"type":217},{"text":828},"A human can reject the modification before execution when risk thresholds require approval.",{"data":830,"type":42},{"text":831,"level":420},"Evidence",{"data":833,"type":217},{"text":834},"The execution leaves a trace sufficient to reconstruct the decision and state transition.",{"data":836,"type":42},{"text":837,"level":420},"Recovery",{"data":839,"type":217},{"text":840},"The previous value remains recoverable.",{"data":842,"type":42},{"text":843,"level":230},"Human-in-the-Loop Is Not Enough",{"data":845,"type":217},{"text":846},"Adding a human approval box does not automatically solve the problem. A human can only control an agent if the person has visibility, authority, time, context and recovery capability.",{"data":848,"type":248},{"items":849,"style":247},[850,851,852,853,854],"\u003Cb>Visibility:\u003C\u002Fb> enough information to understand what is happening.","\u003Cb>Authority:\u003C\u002Fb> actual ability to stop or modify the action.","\u003Cb>Time:\u003C\u002Fb> intervention before the consequence occurs.","\u003Cb>Context:\u003C\u002Fb> sufficient evidence to make the decision.","\u003Cb>Recovery capability:\u003C\u002Fb> ability to reverse or repair the action.",{"data":856,"type":217},{"text":857},"A user clicking \u003Cb>Approve\u003C\u002Fb> on something they cannot meaningfully inspect is not strong governance. It is approval theater.",{"data":859,"type":42},{"text":860,"level":230},"Rollback Must Become a Native AI Capability",{"data":862,"type":217},{"text":863},"Traditional software deployment has taught us something valuable: \u003Cb>Never deploy what you cannot roll back.\u003C\u002Fb> We should apply the same principle to agentic actions.",{"data":865,"type":278},{"code":866},"REVERSIBLE\nCan automatically undo. COMPENSATABLE\nCannot undo directly but can execute a compensating action. IRREVERSIBLE\nCannot reliably restore the previous state.",{"data":868,"type":217},{"text":869},"The higher the irreversibility, the stronger the control requirement should become.",{"data":871,"type":278},{"code":872},"Read public document → low consequence\nCreate draft → reversible\nModify CRM record → reversible but consequential\nSend external email → practically irreversible\nTransfer money → high consequence\nDelete production data → potentially catastrophic",{"data":874,"type":42},{"text":875,"level":230},"The Agent Needs a Control Plane",{"data":877,"type":278},{"code":878},"USER \u002F SYSTEM INTENT │ ▼ AI AGENT │ proposed action │ ▼ ┌───────────────────┐ │ CONTROL PLANE │ ├───────────────────┤ │ Identity │ │ Authorization │ │ Policy │ │ Risk │ │ State │ │ Evidence │ │ Human authority │ │ Rollback │ └───────────────────┘ │ approved? \u002F \\ NO YES │ │ STOP ▼ TOOL │ ▼ STATE CHANGE │ ▼ VERIFICATION",{"data":880,"type":217},{"text":881},"\u003Cb>The LLM should propose. The control plane should govern.\u003C\u002Fb> That separation is crucial. The model should not be the ultimate authority determining whether its own proposed high-impact action is safe.",{"data":883,"type":42},{"text":884,"level":230},"From Benchmarks to Operational Trust",{"data":886,"type":217},{"text":887},"Benchmarks remain useful. They tell us about capability, compare models, detect regressions and help estimate expected performance. But capability evaluation and operational trust answer different questions.",{"data":889,"type":217},{"text":890},"A benchmark asks: \u003Cb>Can the system do this?\u003C\u002Fb> Operational assurance asks: \u003Cb>Can we allow the system to do this here, under these conditions, with these permissions and consequences?\u003C\u002Fb>",{"data":892,"type":42},{"text":893,"level":230},"Reliability Should Be Measured as a System Property",{"data":895,"type":248},{"items":896,"style":265},[897,898,899,900,901],"\u003Cb>Outcome correctness:\u003C\u002Fb> Did the system produce the expected result?","\u003Cb>Trajectory correctness:\u003C\u002Fb> Did it follow an acceptable path?","\u003Cb>Control integrity:\u003C\u002Fb> Were authorization, policy and intervention boundaries respected?","\u003Cb>Recoverability:\u003C\u002Fb> Can failures be contained, reversed or repaired?","\u003Cb>Evidence completeness:\u003C\u002Fb> Can the execution be reconstructed and audited?",{"data":903,"type":278},{"code":904},"Operational Reliability\n=\nOutcome × Trajectory × Control × Recoverability × Evidence",{"data":906,"type":217},{"text":907},"The multiplication is intentional. If one critical dimension approaches zero, a high score elsewhere should not hide it. A perfectly correct output with zero authorization integrity is not an 80% reliable system. It is an unacceptable execution that happened to produce the right answer.",{"data":909,"type":42},{"text":910,"level":230},"Success Is Sometimes the Most Dangerous Failure",{"data":912,"type":217},{"text":913},"Failures attract attention. Success often does not. That makes successful but uncontrolled agent trajectories particularly dangerous. An obvious failure creates an incident. A hidden trajectory defect creates \u003Cb>confidence\u003C\u002Fb>. And confidence expands autonomy.",{"data":915,"type":217},{"text":916},"Organizations should therefore not only investigate \u003Ci>Why did the agent fail?\u003C\u002Fi> They should periodically ask: \u003Cb>Why did the agent succeed?\u003C\u002Fb> Did it succeed because the architecture reliably constrained and verified the execution, or because nothing went wrong this time?",{"data":918,"type":42},{"text":919,"level":230},"Conclusion",{"data":921,"type":217},{"text":922},"The industry is moving rapidly from AI that \u003Cb>answers\u003C\u002Fb> toward AI that \u003Cb>acts\u003C\u002Fb>. That transition changes what reliability means. For an answer system, evaluating the answer may often be sufficient. For an action system, we must evaluate the path.",{"data":924,"type":278},{"code":925},"Prompt ↓\nResponse becomes Intent ↓\nTrajectory ↓\nActions ↓\nState changes ↓\nEvidence ↓\nOutcome",{"data":927,"type":217},{"text":928},"The final answer remains important, but it is only the visible end of a much larger system. Once AI is allowed to affect the real world, \u003Cb>the path to the answer becomes part of the answer.\u003C\u002Fb>","2.31.0","Correct output does not prove correct reasoning, safe execution, or a trustworthy system.",{"lang":7,"title":208,"content":210,"contentJson":932,"excerpt":561},{"time":212,"blocks":933,"version":560},[934,936,938,940,942,944,946,949,951,953,956,958,960,962,964,966,968,972,974,976,984,986,988,990,992,995,997,999,1001,1003,1005,1007,1009,1011,1013,1015,1017,1019,1021,1023,1025,1027,1029,1031,1033,1035,1037,1039,1041,1043,1045,1047,1049,1051,1053,1055,1057,1059,1061,1063,1065,1067,1069,1071,1073,1075,1077,1079,1082,1084,1086,1088,1090,1092,1094,1096,1098,1100,1102,1104,1106,1108,1111,1113,1115,1117,1119,1121,1123,1125,1127],{"data":935,"type":217},{"text":216},{"data":937,"type":217},{"text":220},{"data":939,"type":217},{"text":223},{"data":941,"type":217},{"text":226},{"data":943,"type":42},{"text":229,"level":230},{"data":945,"type":217},{"text":233},{"data":947,"type":248},{"items":948,"style":247},[237,238,239,240,241,242,243,244,245,246],{"data":950,"type":217},{"text":251},{"data":952,"type":217},{"text":254},{"data":954,"type":248},{"items":955,"style":265},[258,259,260,261,262,263,264],{"data":957,"type":217},{"text":268},{"data":959,"type":42},{"text":271,"level":230},{"data":961,"type":217},{"text":274},{"data":963,"type":278},{"code":277},{"data":965,"type":217},{"text":281},{"data":967,"type":287},{"text":284,"caption":285,"alignment":286},{"data":969,"type":295},{"link":290,"meta":970},{"image":971,"title":293,"description":294},{},{"data":973,"type":217},{"text":298},{"data":975,"type":42},{"text":301,"level":230},{"data":977,"type":323},{"content":978,"withHeadings":14},[979,980,981,982,983],[306,307,308],[310,311,312],[314,311,315],[317,318,319],[321,318,322],{"data":985,"type":217},{"text":326},{"data":987,"type":278},{"code":329},{"data":989,"type":42},{"text":332,"level":230},{"data":991,"type":217},{"text":335},{"data":993,"type":248},{"items":994,"style":247},[339,340,341,342,343,344,345,346,347,348,349,350],{"data":996,"type":217},{"text":353},{"data":998,"type":42},{"text":356,"level":230},{"data":1000,"type":217},{"text":359},{"data":1002,"type":278},{"code":362},{"data":1004,"type":217},{"text":365},{"data":1006,"type":278},{"code":368},{"data":1008,"type":217},{"text":371},{"data":1010,"type":42},{"text":374,"level":230},{"data":1012,"type":217},{"text":377},{"data":1014,"type":278},{"code":380},{"data":1016,"type":217},{"text":383},{"data":1018,"type":42},{"text":386,"level":230},{"data":1020,"type":217},{"text":389},{"data":1022,"type":278},{"code":392},{"data":1024,"type":217},{"text":395},{"data":1026,"type":42},{"text":398,"level":230},{"data":1028,"type":217},{"text":401},{"data":1030,"type":278},{"code":404},{"data":1032,"type":217},{"text":407},{"data":1034,"type":278},{"code":410},{"data":1036,"type":42},{"text":413,"level":230},{"data":1038,"type":217},{"text":416},{"data":1040,"type":42},{"text":419,"level":420},{"data":1042,"type":217},{"text":423},{"data":1044,"type":42},{"text":426,"level":420},{"data":1046,"type":217},{"text":429},{"data":1048,"type":42},{"text":432,"level":420},{"data":1050,"type":217},{"text":435},{"data":1052,"type":42},{"text":438,"level":420},{"data":1054,"type":217},{"text":441},{"data":1056,"type":42},{"text":444,"level":420},{"data":1058,"type":217},{"text":447},{"data":1060,"type":42},{"text":450,"level":420},{"data":1062,"type":217},{"text":453},{"data":1064,"type":42},{"text":456,"level":420},{"data":1066,"type":217},{"text":459},{"data":1068,"type":42},{"text":462,"level":420},{"data":1070,"type":217},{"text":465},{"data":1072,"type":42},{"text":468,"level":420},{"data":1074,"type":217},{"text":471},{"data":1076,"type":42},{"text":474,"level":230},{"data":1078,"type":217},{"text":477},{"data":1080,"type":248},{"items":1081,"style":247},[481,482,483,484,485],{"data":1083,"type":217},{"text":488},{"data":1085,"type":42},{"text":491,"level":230},{"data":1087,"type":217},{"text":494},{"data":1089,"type":278},{"code":497},{"data":1091,"type":217},{"text":500},{"data":1093,"type":278},{"code":503},{"data":1095,"type":42},{"text":506,"level":230},{"data":1097,"type":278},{"code":509},{"data":1099,"type":217},{"text":512},{"data":1101,"type":42},{"text":515,"level":230},{"data":1103,"type":217},{"text":518},{"data":1105,"type":217},{"text":521},{"data":1107,"type":42},{"text":524,"level":230},{"data":1109,"type":248},{"items":1110,"style":265},[528,529,530,531,532],{"data":1112,"type":278},{"code":535},{"data":1114,"type":217},{"text":538},{"data":1116,"type":42},{"text":541,"level":230},{"data":1118,"type":217},{"text":544},{"data":1120,"type":217},{"text":547},{"data":1122,"type":42},{"text":550,"level":230},{"data":1124,"type":217},{"text":553},{"data":1126,"type":278},{"code":556},{"data":1128,"type":217},{"text":559},"Post erfolgreich abgerufen",{"items":1131,"source":1202,"manualIds":1203,"manualMatchedIds":1204},[1132,1139,1146,1153,1160,1167,1174,1181,1188,1195],{"id":1133,"slug":1134,"title":1135,"excerpt":1136,"featuredImage":1137,"publishedAt":1138},"489","agentic-ai-explained-when-an-ai-system-can-plan-use-tools-and-act","IA agéntica explicada: cuando un sistema de IA puede planificar, usar herramientas y actuar","La IA agéntica utiliza modelos dentro de bucles de ejecución de varios pasos, donde pueden elegir herramientas, observar resultados, actualizar el estado y adaptar su siguiente acción dentro de límites explícitos de tiempo de ejecución y permisos.","\u002Fuploads\u002F2026\u002F10\u002Fagentic-ai-explained-when-an-ai-system-can-plan-use-tools-and-act-1791481499084-wnji2a.webp","2026-10-08T11:43:00.000Z",{"id":1140,"slug":1141,"title":1142,"excerpt":1143,"featuredImage":1144,"publishedAt":1145},"493","mlops-vs-llmops-what-changes-when-the-model-is-an-llm","MLOps vs LLMOps: qué cambia cuando el modelo es un LLM","MLOps opera sistemas de aprendizaje automático; LLMOps extiende esas prácticas a prompts, contexto, recuperación, proveedores, herramientas, evaluaciones y comportamiento en tiempo de ejecución en torno a modelos de lenguaje grandes.","\u002Fuploads\u002F2026\u002F10\u002Fmlops-vs-llmops-what-changes-when-the-model-is-an-llm-1791487319869-2v7hxo.webp","2026-10-08T15:20:00.000Z",{"id":1147,"slug":1148,"title":1149,"excerpt":1150,"featuredImage":1151,"publishedAt":1152},"477","computer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system","Agentes de uso de computadoras: por qué una demostración exitosa aún puede ser un sistema poco confiable","Los agentes de uso de computadoras ahora pueden completar impresionantes flujos de trabajo en el navegador y en el escritorio, pero una ejecución exitosa demuestra capacidad—no fiabilidad. Este artículo muestra cómo probar la repetibilidad, la robustez ambiental, el control de horizonte largo, la conciencia del estado, la verificación de resultados y la gestión segura de objetivos.","\u002Fuploads\u002F2026\u002F09\u002Fcomputer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system-1790352854690-75qnrg.webp","2026-09-25T12:13:00.000Z",{"id":1154,"slug":1155,"title":1156,"excerpt":1157,"featuredImage":1158,"publishedAt":1159},"479","where-does-an-llm-get-its-data-rag-data-sources-in-python","¿De dónde obtiene sus datos un LLM? Fuentes de datos RAG en Python","Un LLM no conoce mágicamente tus archivos, bases de datos o APIs. Esta continuación práctica de la serie RAG muestra, con Python sencillo, cómo los datos externos se convierten en evidencia recuperable: desde archivos de texto y SQL hasta búsqueda de texto completo, embeddings, ensamblaje de contexto y la llamada final al LLM.","\u002Fuploads\u002F2026\u002F09\u002Fwhere-does-an-llm-get-its-data-rag-data-sources-in-python-1790517200521-nfsi5i.webp","2026-09-27T05:51:00.000Z",{"id":1161,"slug":1162,"title":1163,"excerpt":1164,"featuredImage":1165,"publishedAt":1166},"459","ollama-is-not-the-product-building-production-ready-open-llm-applications","Ollama no es el producto: Construcción de aplicaciones de LLM abiertos listas para producción","Ejecutar un modelo local con Ollama es fácil. Construir una aplicación Open-LLM lista para producción es más difícil: requiere RAG, control de acceso, abstracción de proveedores, evaluación, registro, disciplina de despliegue y una capa de aplicación controlada alrededor del modelo.","\u002Fuploads\u002F2026\u002F06\u002Follama-is-not-the-product-building-production-ready-open-llm-applications-1782679361640-h0usqf.webp","2026-06-28T16:39:00.000Z",{"id":1168,"slug":1169,"title":1170,"excerpt":1171,"featuredImage":1172,"publishedAt":1173},"469","rag-failed-but-which-layer-actually-failed-a-diagnostic-method","RAG falló — ¿Pero qué capa falló realmente? Un método de diagnóstico","Cuando una respuesta RAG es incorrecta, culpar a la recuperación o al modelo es demasiado vago. Este método de diagnóstico aísla la cobertura de fuentes, la construcción de consultas, la recuperación, el ranking, el ensamblaje del contexto, la generación, la atribución de evidencia y la actualidad, de modo que el fallo real puede reproducirse y corregirse.","\u002Fuploads\u002F2026\u002F09\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method-1790350847177-pior4c.webp","2026-09-24T19:39:00.000Z",{"id":1175,"slug":1176,"title":1177,"excerpt":1178,"featuredImage":1179,"publishedAt":1180},"364","tipps-fuer-die-verbesserung-der-seo-suchmaschinenoptimierung","Dominando el flujo de trabajo SEO: Estrategias de optimización esenciales para el crecimiento orgánico","Un flujo de trabajo SEO estructurado es crucial para un crecimiento orgánico sostenible. Aprende las diez estrategias fundamentales, desde la investigación de palabras clave y la optimización técnica hasta la calidad del contenido y el análisis de rendimiento.","\u002Fuploads\u002F2026\u002F03\u002Ftipps-fuer-die-verbesserung-der-seo-suchmaschinenoptimierung-1774866098131-hwkzrg.webp","2024-01-26T06:35:00.000Z",{"id":1182,"slug":1183,"title":1184,"excerpt":1185,"featuredImage":1186,"publishedAt":1187},"478","what-is-rag-the-simplest-explanation-of-how-it-works","¿Qué es RAG? La explicación más sencilla de cómo funciona","RAG suena complicado, pero la idea es simple: antes de que una IA responda, primero busca información útil de una fuente de conocimiento y le da esa información al modelo de lenguaje. Esta guía explica RAG, los LLM, el estado, la memoria y las herramientas usando un modelo mental simple.","\u002Fuploads\u002F2026\u002F09\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works-1790377492124-khjagt.webp","2026-09-25T19:03:00.000Z",{"id":1189,"slug":1190,"title":1191,"excerpt":1192,"featuredImage":1193,"publishedAt":1194},"480","when-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger","¿Cuándo debería una IA dejar de confiar en su propio conocimiento? — El desencadenante de la recuperación","Un modelo de IA no necesita recuperación para cada pregunta. El problema importante es saber cuándo su conocimiento interno ya no es suficiente. El Disparador de Recuperación es un límite de decisión práctico que determina cuándo un sistema de IA debe dejar de depender únicamente del conocimiento del modelo y obtener evidencia externa antes de responder.","\u002Fuploads\u002F2026\u002F09\u002Fwhen-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger-1790574991244-f4rpyg.webp","2026-09-28T01:49:00.000Z",{"id":1196,"slug":1197,"title":1198,"excerpt":1199,"featuredImage":1200,"publishedAt":1201},"485","enterprise-ai-architecture-what-changes-when-ai-enters-a-company","Arquitectura de IA empresarial: qué cambia cuando la IA entra en una empresa","La arquitectura de IA empresarial explica cómo la IA cambia los sistemas de la empresa en materia de autoridad sobre los datos, identidad, permisos, proveedores, riesgo, gobernanza, evaluación, cumplimiento y operaciones.","\u002Fuploads\u002F2026\u002F10\u002Fenterprise-ai-architecture-what-changes-when-ai-enters-a-company-1791478161363-czrwaq.webp","2026-10-08T10:48:00.000Z","fallback",[],[]]