[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"portal-settings:stajic:sr":3,"public-menus:all":38,"post:ai-agent-reliability-why-the-final-answer-is-not-enough:sr":205,"related:post:ai-agent-reliability-why-the-final-answer-is-not-enough:sr:1":1128},{"statusCode":4,"data":5,"message":37},200,{"tenantId":6,"lang":7,"defaultLang":8,"siteUrl":9,"contactEmail":10,"brandName":11,"logoUrl":12,"siteName":11,"siteDescription":13,"ogImage":10,"robotsIndex":14,"socialLinks":10,"reservedSlugs":10,"seoPolicy":15},"stajic","sr","de","https:\u002F\u002Fstajic.de",null,"Stajic Platform","\u002FLogo_Planet.svg","Stajic Portal",true,{"branding":16,"relatedContent":17,"crossDomainLinks":18},{"logoUrl":12},{"enabled":14},[19,22,25,28,31,34],{"url":20,"label":21,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Ffigure.rocks","figure.rocks",{"url":23,"label":24,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Floving.rocks","loving.rocks",{"url":26,"label":27,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.com","bazify.com",{"url":29,"label":30,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.de","bazify.de",{"url":32,"label":33,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.at","bazify.at",{"url":35,"label":36,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.ba","bazify.ba","Portal settings resolved",[39,45],{"id":40,"name":41,"location":42,"isActive":14,"isDefault":43,"items":44},1,"main-navigation","header",false,[],{"id":46,"name":47,"location":48,"isActive":14,"isDefault":14,"items":49},4,"main-menu","sidebar",[50,66,79,93,103,118,133],{"id":51,"title":52,"url":60,"target":61,"icon":62,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":64,"portfolioId":10,"children":65},"item-18",{"de":53,"en":54,"es":55,"fr":56,"it":54,"ru":57,"sr":58,"zh":59},"Startseite","Home","Inicio","Accueil","Главная","Почетна","首页","\u002Ffull-stack-web-developer-munich-performance-seo-and-maintainable-builds","_self","i-lucide-home","page",111,[],{"id":67,"title":68,"url":75,"target":61,"icon":76,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":77,"portfolioId":10,"children":78},"item-22",{"de":69,"en":69,"es":70,"fr":69,"it":71,"ru":72,"sr":73,"zh":74},"Vision","Visión","Visione","Видение","Визија","想象","\u002Fueber-uns-webdesign-muenchen-webaplikation","i-lucide-eye",113,[],{"id":80,"title":81,"url":89,"target":61,"icon":90,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":91,"portfolioId":10,"children":92},"item-19",{"de":82,"en":83,"es":84,"fr":83,"it":85,"ru":86,"sr":87,"zh":88},"Leistungen","Services","Servicios","Servizi","Услуги","Услуге","服务","\u002Fservices-dienstleistungen-muenchen","i-lucide-wrench",116,[],{"id":94,"title":95,"url":99,"target":61,"icon":100,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":101,"portfolioId":10,"children":102},"item-23",{"de":96,"en":96,"es":96,"fr":96,"it":96,"ru":97,"sr":97,"zh":98},"Blog","Блог","博客","\u002Fblog","i-lucide-book-open",112,[],{"id":104,"title":105,"url":114,"target":61,"icon":115,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":116,"portfolioId":10,"children":117},"item-32",{"de":106,"en":107,"es":108,"fr":109,"it":110,"ru":111,"sr":112,"zh":113},"Neue Technologien","New Technologies","Nuevas tecnologías","Nouvelles technologies","Nuove tecnologie","Новые технологии","Нове технологије","新技术！","\u002Fneue-webtechnologien","i-lucide-sparkles",122,[],{"id":119,"title":120,"url":129,"target":61,"icon":130,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":131,"portfolioId":10,"children":132},"item-20",{"de":121,"en":122,"es":123,"fr":124,"it":125,"ru":126,"sr":127,"zh":128},"Kontakt","Contact us!","Contacto","Contact","Contatto","Контакт","Контактирајте нас","联系我们！","\u002Fcontact","i-lucide-mail",115,[],{"id":134,"title":135,"url":144,"target":61,"icon":145,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":147},"item-21",{"de":136,"en":137,"es":138,"fr":139,"it":140,"ru":141,"sr":142,"zh":143},"Unsere Arbeit","Our Work","Nuestro trabajo","Nos réalisations","I nostri lavori","Наши работы","Наши радови","文件夹","\u002Fportfolio","i-lucide-briefcase",114,[148,161,175,181,193],{"id":149,"title":150,"url":144,"target":61,"icon":159,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":160},"item-24",{"de":151,"en":152,"es":153,"fr":154,"it":155,"ru":156,"sr":157,"zh":158},"Alle Projekte","All Projects","Todos los proyectos","Tous les projets","Tutti i progetti","Все проекты","Сви пројекти","所有项目","i-lucide-grid-3x3",[],{"id":162,"title":163,"url":171,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":174},"item-29",{"de":164,"en":165,"es":166,"fr":167,"it":168,"ru":169,"sr":170,"zh":143},"Local Roots, Global Reach","Local Roots - Global Reach","Empresa local ","Entreprise locale","Azienda locale","Местная компания","Локално предузеће глобално тржиште","\u002Fportfolio\u002Flocal-roots-global-reach-communication-media-systems-for-modern-business","i-lucide-folder","custom",[],{"id":176,"title":177,"url":179,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":180},"item-28",{"de":178,"en":178,"es":178,"fr":178,"it":178,"ru":178,"sr":178,"zh":178},"Solr Suggester","\u002Fportfolio\u002Fsolr-fuzzy-suggester-und-solr-infix-suggester-abfrage-ueber-ajax-und-filterung",[],{"id":182,"title":183,"url":191,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":192},"item-27",{"de":184,"en":185,"es":186,"fr":187,"it":188,"ru":189,"sr":190,"zh":185},"Firmenwebseite SEO","Company Website SEO","Sitio web corporativo SEO","Site web d’entreprise SEO","Sito web aziendale SEO","Корпоративный сайт SEO","Пословна веб-страница SEO","\u002Fportfolio\u002Fseo-sem-branding-mobile-webseite-muenchen",[],{"id":194,"title":195,"url":203,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":204},"item-31",{"de":196,"en":197,"es":198,"fr":199,"it":200,"ru":201,"sr":202,"zh":197},"Digitalisierungsportal","Digitalization Portal","Portal de digitalización","Portail de numérisation","Portale di digitalizzazione","Портал цифровизации","Портал за дигитализацију","\u002Fportfolio\u002Fdigitalisierungsportal-archiv-museum-bibliothek-ead-lido-mets-mods",[],{"statusCode":4,"data":206,"message":1127},{"id":207,"title":208,"slug":209,"content":210,"contentJson":211,"excerpt":561,"featuredImage":562,"featuredImageAlt":563,"featuredImageCaption":10,"featuredImageTitle":10,"featuredImageCopyright":10,"featuredImageAuthor":10,"featuredImageSourceUrl":10,"featuredImageLicense":10,"featuredImageIsAiGenerated":43,"status":564,"publishedAt":565,"createdAt":566,"updatedAt":567,"seoLocalePaths":568,"categories":577,"author":586,"translations":591},"460","Pouzdanost AI agenata: Zašto konačni odgovor nije dovoljan","ai-agent-reliability-why-the-final-answer-is-not-enough","\u003Cp>\u003Cb>Tačan izlaz ne dokazuje ispravno rasuđivanje, bezbedno izvršavanje ili pouzdan sistem.\u003C\u002Fb>\u003C\u002Fp>\n\u003Cp>Godinama je evaluacija veštačke inteligencije bila dominirana naizgled jednostavnim pitanjem: \u003Cb>Da li je odgovor bio tačan?\u003C\u002Fb> Za ćaskbot to ponekad može biti dovoljno. Za agenta sposobnog da pretražuje sisteme, čita podatke, poziva alate, menja stanje, izvršava tokove posla, piše datoteke, komunicira sa API-jima ili donosi odluke, to nije.\u003C\u002Fp>\n\u003Cp>Agent može proizvesti tačan konačni odgovor dok na putu do njega čini više stvari pogrešno. Može koristiti pogrešan izvor, pogrešno razumeti instrukciju i kasnije nadoknaditi grešku, pristupiti nepotrebnim informacijama, izvršiti neovlašćenu međukorak, tiho se oporaviti od greške koja je trebala da pokrene eskalaciju, ili ostaviti iza sebe sporedne efekte koje niko nije primetio.\u003C\u002Fp>\n\u003Cp>To stvara jedan od centralnih problema agentske veštačke inteligencije: \u003Cb>tačan ishod ne dokazuje tačnu putanju.\u003C\u002Fb>\u003C\u002Fp>\n\u003Ch2>Iluzija ishoda\u003C\u002Fh2>\n\u003Cp>Tradicionalni softver nam daje intuitivni model ispravnosti. Ulaz ulazi u deterministički ili uglavnom deterministički sistem, logika se izvršava, izlaz se proizvodi, a testovi proveravaju očekivano ponašanje. Sistemi zasnovani na LLM-u slabe ovu pretpostavku. Agentski sistemi idu i dalje.\u003C\u002Fp>\n\u003Cul>\u003Cli>interpretacija modela\u003C\u002Fli>\u003Cli>preuzeti kontekst\u003C\u002Fli>\u003Cli>izbor alata\u003C\u002Fli>\u003Cli>međuopažanja\u003C\u002Fli>\u003Cli>spoljašnje stanje\u003C\u002Fli>\u003Cli>prethodne radnje\u003C\u002Fli>\u003Cli>planovi koje generiše model\u003C\u002Fli>\u003Cli>granice dozvola\u003C\u002Fli>\u003Cli>ponovni pokušaji i rezervno ponašanje\u003C\u002Fli>\u003Cli>interakcija sa ljudima\u003C\u002Fli>\u003C\u002Ful>\n\u003Cp>Dva izvršavanja koja počinju od gotovo identičnih ulaza mogu doći do istog rezultata kroz veoma različite puteve. Ako evaluacija posmatra samo konačni izlaz, veći deo sistema ostaje nevidljiv.\u003C\u002Fp>\n\u003Cp>Zamislite da agent veštačke inteligencije dobije instrukciju: \u003Ci>Ažuriraj adresu za naplatu kupca.\u003C\u002Fi> Adresa se na kraju ispravno ažurira. Konvencionalna evaluacija bi mogla klasifikovati zadatak kao uspešan.\u003C\u002Fp>\n\u003Col>\u003Cli>Agent pretražuje nekoliko nepovezanih evidencija kupaca.\u003C\u002Fli>\u003Cli>Preuzima više ličnih podataka nego što je potrebno.\u003C\u002Fli>\u003Cli>U početku menja pogrešan nalog.\u003C\u002Fli>\u003Cli>Primećuje grešku.\u003C\u002Fli>\u003Cli>Poništava promenu.\u003C\u002Fli>\u003Cli>Ažurira ispravan nalog.\u003C\u002Fli>\u003Cli>Prijavljuje uspeh.\u003C\u002Fli>\u003C\u002Fol>\n\u003Cp>\u003Cb>Konačno stanje: ispravno. Ponašanje sistema: neprihvatljivo.\u003C\u002Fb> Benchmark koji se zasniva samo na ishodu daje ovom izvršavanju prolaz. Sistem za osiguranje proizvodnje ne bi trebalo.\u003C\u002Fp>\n\u003Ch2>Putanja je deo proizvoda\u003C\u002Fh2>\n\u003Cp>Zato \u003Cb>putanja\u003C\u002Fb> agenta veštačke inteligencije mora postati prvoklasni inženjerski objekat. Putanja je niz relevantnih stanja i radnji između originalnog zahteva i konačnog rezultata.\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>Namera → Kontekst → Odluka → Alat → Radnja → Opažanje → Odluka → Promena stanja → Rezultat\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>Zachary J. Stevens razvija ovu ideju u \u003Ci>Putanja je sistem\u003C\u002Fi>, tvrdeći da agentska evaluacija mora prevazići konačni odgovor i ispitati kompletan put radnji kroz promenljivo okruženje.\u003C\u002Fp>\n\u003Cblockquote class=\"border-l-4 border-gray-300 pl-4 italic\">Tačan ishod ne opravdava neprihvatljivu putanju.\u003Ccite class=\"block mt-2 text-sm\">— Zachary J. Stevens, Putanja je sistem\u003C\u002Fcite>\u003C\u002Fblockquote>\n\u003Ca href=\"https:\u002F\u002Fzacharyjstevens.com\u002Fdispatches\u002Fvanguard-signal\u002F009-the-trajectory-is-the-system\u002F\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">Putanja je sistem\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">Zachary J. Stevens — DFEI.009 o evaluaciji agentskih sistema na osnovu njihove kompletne putanje, a ne samo konačnog ishoda.\u003C\u002Fp>\u003C\u002Fa>\n\u003Cp>Razlika je izuzetno važna. Pouzdanost stoga nije jednostavno \u003Cb>tačan izlaz\u003C\u002Fb>. Ona je bliža \u003Cb>prihvatljivom ishodu + prihvatljivoj putanji + mogućnosti oporavka + dokazima\u003C\u002Fb>.\u003C\u002Fp>\n\u003Ch2>Tačan odgovor može sakriti pokvaren sistem\u003C\u002Fh2>\n\u003Ctable class=\"w-full border-collapse\">\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Agent\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Konačan rezultat\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Izvršenje\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">A\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Tačno\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Ispravna putanja\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">B\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Tačno\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Nesigurna putanja\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">C\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Netačno\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Bezbedan otkaz\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">D\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Netačno\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Nesiguran otkaz\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftable>\n\u003Cp>Većina evaluacija zasnovanih na merilima snažno nagrađuje A i B, a kažnjava C i D. Operativno, međutim, \u003Cb>B može biti opasniji od C\u003C\u002Fb>. Agent C može prepoznati neizvesnost, zaustaviti izvršenje i zatražiti ljudski pregled. Agent B može sa samopouzdanjem proizvesti tačne rezultate dok krši pretpostavke koje niko ne nadgleda.\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>uspešan izlaz → povećano poverenje → šire dozvole → više automatizacije → veći radijus eksplozije\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Ch2>Potrebni su nam dokazi, ne samopouzdanje\u003C\u002Fh2>\n\u003Cp>Jedna od najvećih grešaka u usvajanju veštačke inteligencije je tretiranje samopouzdanja modela, zadovoljstva korisnika ili istorijske stope uspeha kao dokaza pouzdanosti sistema. Oni nisu ekvivalentni.\u003C\u002Fp>\n\u003Cul>\u003Cli>Šta je agent primio?\u003C\u002Fli>\u003Cli>Koji kontekst je preuzeo?\u003C\u002Fli>\u003Cli>Koje alate je pozvao?\u003C\u002Fli>\u003Cli>Zašto je radnja bila dozvoljena?\u003C\u002Fli>\u003Cli>Koje stanje je postojalo pre radnje?\u003C\u002Fli>\u003Cli>Šta se promenilo?\u003C\u002Fli>\u003Cli>Koji su međuotkazi nastali?\u003C\u002Fli>\u003Cli>Da li je nešto ponovljeno?\u003C\u002Fli>\u003Cli>Da li je bilo potrebno ljudsko odobrenje?\u003C\u002Fli>\u003Cli>Da li je izvršenje moglo biti zaustavljeno?\u003C\u002Fli>\u003Cli>Može li se radnja poništiti?\u003C\u002Fli>\u003Cli>Koji model, prompt i verzije alata su bili uključeni?\u003C\u002Fli>\u003C\u002Ful>\n\u003Cp>Bez ovih odgovora, ne postoji ozbiljno operativno osiguranje. Postoji samo izlaz. Opservabilnost i dokazi stoga moraju biti projektovani u arhitekturu agenta, a ne dodati nakon implementacije.\u003C\u002Fp>\n\u003Ch2>Vođenje evidencije nije isto što i kontrola\u003C\u002Fh2>\n\u003Cp>Organizacije često odgovore: \u003Ci>Sve se evidentira.\u003C\u002Fi> Dobro. Ali samo evidentiranje ne kontroliše ništa. Evidencija vam govori šta se dogodilo. Kontrola određuje da li nešto \u003Cb>može da se dogodi\u003C\u002Fb>.\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>Agent zahteva DELETE \u002Fcustomer\u002F123 ↓\nRadnja evidentirana ↓\nDELETE izvršen\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>To pruža opservabilnost. Uporedite to sa:\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>Agent zahteva DELETE \u002Fcustomer\u002F123 ↓\nProcena politike ↓\nTrenutni identitet verifikovan ↓\nTrenutni parametri radnje provereni ↓\nPrag rizika procenjen ↓\nLjudsko odobrenje ako je potrebno ↓\nRadnja izvršena ↓\nRezultat verifikovan ↓\nDokazi sačuvani\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>Sada se približavamo kontrolnom sistemu. Razlika je arhitektonska, a ne kozmetička.\u003C\u002Fp>\n\u003Ch2>Dozvola je neophodna — ali nije osiguranje\u003C\u002Fh2>\n\u003Cp>Pretpostavimo da agent ima dozvolu da šalje e-poštu. Kontrola pristupa odgovara: \u003Cb>Može li ovaj agent slati e-poštu?\u003C\u002Fb> Ne odgovara: \u003Cb>Da li ovaj konkretan e-mail treba poslati ovoj konkretnoj osobi sa ovim konkretnim prilogom upravo sada?\u003C\u002Fb>\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>KONTROLA SPOSOBNOSTI\nŠta je agentu tehnički dozvoljeno da uradi? + OSIGURANJE RADNJE\nDa li je ova konkretna radnja prikladna u trenutnom stanju?\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>RBAC, OAuth opsezi, API dozvole i identiteti agenata definišu prostor mogućih radnji. Oni ne dokazuju da je radnja unutar tog prostora prikladna. Snažna arhitektura agenta zahteva oba sloja.\u003C\u002Fp>\n\u003Ch2>Prvi pogrešan korak je važan\u003C\u002Fh2>\n\u003Cp>Kada agent ne uspe, konačna netačna radnja često nije mesto gde je neuspeh počeo. Pravi neuspeh se možda dogodio mnogo ranije.\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>Pogrešno pretraživanje ↓\nPogrešna pretpostavka ↓\nUverljivo zaključivanje ↓\nVažeći poziv alata ↓\nPogrešna radnja\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>Ako istražujemo samo konačnu radnju, popravljamo simptom. Ako pregledamo putanju, možemo identifikovati \u003Cb>prvi pogrešan korak\u003C\u002Fb>. To pretvara nepripisiv neuspeh u konkretan inženjerski problem.\u003C\u002Fp>\n\u003Ch2>Testiranje agenata mora prevazići testiranje promptova\u003C\u002Fh2>\n\u003Cp>Promptovi su važni, ali ponašanje agenta u produkciji proizilazi iz celog sistema.\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>MODEL\n+\nSISTEMSKI PROMPT\n+\nKONTEKST\n+\nMEMORIJA\n+\nPRETRAŽIVANJE\n+\nALATI\n+\nDOZVOLE\n+\nRADNI TOK\n+\nSPOLJNO STANJE\n+\nKONTROLNA LOGIKA\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>Promena bilo kojeg od ovih može promeniti putanju. Stoga, verzionisanje samo prompta nije dovoljno.\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>model_version\nprompt_version\ntool_version\npolicy_version\nretrieval_version\nworkflow_version\nenvironment_state\nexecution_id\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Ch2>Kriterijumi prihvatanja za agente moraju uključivati ponašanje\u003C\u002Fh2>\n\u003Cp>Tradicionalni kriterijumi prihvatanja često izgledaju ovako: \u003Ci>Dato X, sistem proizvodi Y.\u003C\u002Fi> Za agentske sisteme, to je nepotpuno. Kriterijumi prihvatanja treba da definišu i ograničenja na putanji.\u003C\u002Fp>\n\u003Ch3>Ishod\u003C\u002Fh3>\n\u003Cp>Adresa kupca je ispravno ažurirana.\u003C\u002Fp>\n\u003Ch3>Autorizacija\u003C\u002Fh3>\n\u003Cp>Agent menja samo eksplicitno izabranog kupca.\u003C\u002Fp>\n\u003Ch3>Pristup podacima\u003C\u002Fh3>\n\u003Cp>Nijedan nepovezan zapis o kupcu nije pristupljen.\u003C\u002Fp>\n\u003Ch3>Alati\u003C\u002Fh3>\n\u003Cp>Koriste se samo odobrene CRM operacije.\u003C\u002Fp>\n\u003Ch3>Verifikacija\u003C\u002Fh3>\n\u003Cp>Nova adresa se čita i poredi sa traženom vrednošću.\u003C\u002Fp>\n\u003Ch3>Neuspeh\u003C\u002Fh3>\n\u003Cp>Dvosmisleno razrešavanje identiteta zaustavlja izvršenje.\u003C\u002Fp>\n\u003Ch3>Ljudski autoritet\u003C\u002Fh3>\n\u003Cp>Čovek može odbiti modifikaciju pre izvršenja kada pragovi rizika zahtevaju odobrenje.\u003C\u002Fp>\n\u003Ch3>Dokaz\u003C\u002Fh3>\n\u003Cp>Izvršenje ostavlja trag dovoljan za rekonstrukciju odluke i prelaza stanja.\u003C\u002Fp>\n\u003Ch3>Oporavak\u003C\u002Fh3>\n\u003Cp>Prethodna vrednost ostaje povratna.\u003C\u002Fp>\n\u003Ch2>Čovek-u-petlji nije dovoljan\u003C\u002Fh2>\n\u003Cp>Dodavanje polja za ljudsko odobrenje ne rešava problem automatski. Čovek može kontrolisati agenta samo ako ima vidljivost, autoritet, vreme, kontekst i mogućnost oporavka.\u003C\u002Fp>\n\u003Cul>\u003Cli>\u003Cb>Vidljivost:\u003C\u002Fb> dovoljno informacija da se razume šta se dešava.\u003C\u002Fli>\u003Cli>\u003Cb>Autoritet:\u003C\u002Fb> stvarna sposobnost da se zaustavi ili izmeni akcija.\u003C\u002Fli>\u003Cli>\u003Cb>Vreme:\u003C\u002Fb> intervencija pre nego što se posledica dogodi.\u003C\u002Fli>\u003Cli>\u003Cb>Kontekst:\u003C\u002Fb> dovoljno dokaza za donošenje odluke.\u003C\u002Fli>\u003Cli>\u003Cb>Mogućnost oporavka:\u003C\u002Fb> sposobnost da se akcija poništi ili popravi.\u003C\u002Fli>\u003C\u002Ful>\n\u003Cp>Korisnik koji klikne \u003Cb>Odobri\u003C\u002Fb> na nešto što ne može smisleno da pregleda nije snažno upravljanje. To je pozorište odobravanja.\u003C\u002Fp>\n\u003Ch2>Povratak mora postati izvorna AI sposobnost\u003C\u002Fh2>\n\u003Cp>Tradicionalno uvođenje softvera nas je naučilo nečemu vrednom: \u003Cb>Nikada ne uvodi ono što ne možeš vratiti.\u003C\u002Fb> Trebalo bi da primenimo isti princip na agentske akcije.\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>REVERZIBILNO\nMože se automatski poništiti. KOMPENZABILNO\nNe može se direktno poništiti, ali se može izvršiti kompenzaciona akcija. NEOPOZIVO\nNe može se pouzdano vratiti prethodno stanje.\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>Što je veća nepovratnost, to bi zahtev za kontrolom trebalo da bude jači.\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>Čitanje javnog dokumenta → niska posledica\nKreiranje nacrta → povratno\nIzmena CRM zapisa → povratno ali sa posledicama\nSlanje spoljnog imejla → praktično nepovratno\nPrenos novca → visoka posledica\nBrisanje podataka iz produkcije → potencijalno katastrofalno\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Ch2>Agentu je potrebna kontrolna ravan\u003C\u002Fh2>\n\u003Cpre class=\"code-block\">\u003Ccode>KORISNIK \u002F SISTEMSKA NAMERA │ ▼ AI AGENT │ predložena akcija │ ▼ ┌───────────────────┐ │ KONTROLNA RAVAN │ ├───────────────────┤ │ Identitet │ │ Autorizacija │ │ Politika │ │ Rizik │ │ Stanje │ │ Dokazi │ │ Ljudski autoritet │ │ Povratak │ └───────────────────┘ │ odobreno? \u002F \\ NE DA │ │ STOP ▼ ALAT │ ▼ PROMENA STANJA │ ▼ VERIFIKACIJA\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>\u003Cb>LLM treba da predlaže. Kontrolna ravan treba da upravlja.\u003C\u002Fb> Ta podela je ključna. Model ne bi trebalo da bude krajnji autoritet koji određuje da li je njegova sopstvena predložena akcija sa visokim uticajem bezbedna.\u003C\u002Fp>\n\u003Ch2>Od benčmarka do operativnog poverenja\u003C\u002Fh2>\n\u003Cp>Benčmarkovi ostaju korisni. Oni nam govore o sposobnosti, porede modele, otkrivaju regresije i pomažu u proceni očekivanih performansi. Ali procena sposobnosti i operativno poverenje odgovaraju na različita pitanja.\u003C\u002Fp>\n\u003Cp>Benčmark pita: \u003Cb>Da li sistem može ovo da uradi?\u003C\u002Fb> Operativno osiguranje pita: \u003Cb>Da li možemo da dozvolimo sistemu da ovo uradi ovde, pod ovim uslovima, sa ovim dozvolama i posledicama?\u003C\u002Fb>\u003C\u002Fp>\n\u003Ch2>Pouzdanost treba meriti kao svojstvo sistema\u003C\u002Fh2>\n\u003Col>\u003Cli>\u003Cb>Ispravnost ishoda:\u003C\u002Fb> Da li je sistem proizveo očekivani rezultat?\u003C\u002Fli>\u003Cli>\u003Cb>Ispravnost putanje:\u003C\u002Fb> Da li je pratio prihvatljiv put?\u003C\u002Fli>\u003Cli>\u003Cb>Integritet kontrole:\u003C\u002Fb> Da li su poštovane granice autorizacije, politike i intervencije?\u003C\u002Fli>\u003Cli>\u003Cb>Oporavljivost:\u003C\u002Fb> Da li se kvarovi mogu ograničiti, preokrenuti ili popraviti?\u003C\u002Fli>\u003Cli>\u003Cb>Kompletnost dokaza:\u003C\u002Fb> Da li se izvršenje može rekonstruisati i revidirati?\u003C\u002Fli>\u003C\u002Fol>\n\u003Cpre class=\"code-block\">\u003Ccode>Operativna pouzdanost\n=\nIshod × Putanja × Kontrola × Oporavljivost × Dokazi\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>Množenje je namerno. Ako jedna kritična dimenzija teži nuli, visok rezultat na drugom mestu ne bi trebalo da to sakrije. Savršeno ispravan izlaz sa nultim integritetom autorizacije nije sistem sa 80% pouzdanosti. To je neprihvatljivo izvršenje koje je slučajno proizvelo tačan odgovor.\u003C\u002Fp>\n\u003Ch2>Uspeh je ponekad najopasniji neuspeh\u003C\u002Fh2>\n\u003Cp>Neuspesi privlače pažnju. Uspeh često ne. Zato su uspešne ali nekontrolisane putanje agenta posebno opasne. Očigledan neuspeh stvara incident. Skriveni defekt putanje stvara \u003Cb>poverenje\u003C\u002Fb>. A poverenje proširuje autonomiju.\u003C\u002Fp>\n\u003Cp>Organizacije stoga ne bi trebalo samo da istražuju \u003Ci>Zašto je agent zakazao?\u003C\u002Fi> One bi trebalo periodično da se pitaju: \u003Cb>Zašto je agent uspeo?\u003C\u002Fb> Da li je uspeo zato što je arhitektura pouzdano ograničila i verifikovala izvršenje, ili zato što ovaj put ništa nije pošlo naopako?\u003C\u002Fp>\n\u003Ch2>Zaključak\u003C\u002Fh2>\n\u003Cp>Industrija se brzo kreće od AI koji \u003Cb>odgovara\u003C\u002Fb> ka AI koji \u003Cb>deluje\u003C\u002Fb>. Ta tranzicija menja značenje pouzdanosti. Za sistem odgovora, procena odgovora često može biti dovoljna. Za sistem akcije, moramo proceniti putanju.\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>Prompt ↓\nOdgovor postaje namera ↓\nPutanja ↓\nAkcije ↓\nPromene stanja ↓\nDokazi ↓\nIshod\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>Konačan odgovor ostaje važan, ali on je samo vidljivi kraj mnogo većeg sistema. Kada se AI-u dozvoli da utiče na stvarni svet, \u003Cb>put do odgovora postaje deo odgovora.\u003C\u002Fb>\u003C\u002Fp>",{"time":212,"blocks":213,"version":560},1788955695223,[214,218,221,224,227,231,234,249,252,255,266,269,272,275,279,282,288,296,299,302,324,327,330,333,336,351,354,357,360,363,366,369,372,375,378,381,384,387,390,393,396,399,402,405,408,411,414,417,421,424,427,430,433,436,439,442,445,448,451,454,457,460,463,466,469,472,475,478,486,489,492,495,498,501,504,507,510,513,516,519,522,525,533,536,539,542,545,548,551,554,557],{"data":215,"type":217},{"text":216},"\u003Cb>Tačan izlaz ne dokazuje ispravno rasuđivanje, bezbedno izvršavanje ili pouzdan sistem.\u003C\u002Fb>","paragraph",{"data":219,"type":217},{"text":220},"Godinama je evaluacija veštačke inteligencije bila dominirana naizgled jednostavnim pitanjem: \u003Cb>Da li je odgovor bio tačan?\u003C\u002Fb> Za ćaskbot to ponekad može biti dovoljno. Za agenta sposobnog da pretražuje sisteme, čita podatke, poziva alate, menja stanje, izvršava tokove posla, piše datoteke, komunicira sa API-jima ili donosi odluke, to nije.",{"data":222,"type":217},{"text":223},"Agent može proizvesti tačan konačni odgovor dok na putu do njega čini više stvari pogrešno. Može koristiti pogrešan izvor, pogrešno razumeti instrukciju i kasnije nadoknaditi grešku, pristupiti nepotrebnim informacijama, izvršiti neovlašćenu međukorak, tiho se oporaviti od greške koja je trebala da pokrene eskalaciju, ili ostaviti iza sebe sporedne efekte koje niko nije primetio.",{"data":225,"type":217},{"text":226},"To stvara jedan od centralnih problema agentske veštačke inteligencije: \u003Cb>tačan ishod ne dokazuje tačnu putanju.\u003C\u002Fb>",{"data":228,"type":42},{"text":229,"level":230},"Iluzija ishoda",2,{"data":232,"type":217},{"text":233},"Tradicionalni softver nam daje intuitivni model ispravnosti. Ulaz ulazi u deterministički ili uglavnom deterministički sistem, logika se izvršava, izlaz se proizvodi, a testovi proveravaju očekivano ponašanje. Sistemi zasnovani na LLM-u slabe ovu pretpostavku. Agentski sistemi idu i dalje.",{"data":235,"type":248},{"items":236,"style":247},[237,238,239,240,241,242,243,244,245,246],"interpretacija modela","preuzeti kontekst","izbor alata","međuopažanja","spoljašnje stanje","prethodne radnje","planovi koje generiše model","granice dozvola","ponovni pokušaji i rezervno ponašanje","interakcija sa ljudima","unordered","list",{"data":250,"type":217},{"text":251},"Dva izvršavanja koja počinju od gotovo identičnih ulaza mogu doći do istog rezultata kroz veoma različite puteve. Ako evaluacija posmatra samo konačni izlaz, veći deo sistema ostaje nevidljiv.",{"data":253,"type":217},{"text":254},"Zamislite da agent veštačke inteligencije dobije instrukciju: \u003Ci>Ažuriraj adresu za naplatu kupca.\u003C\u002Fi> Adresa se na kraju ispravno ažurira. Konvencionalna evaluacija bi mogla klasifikovati zadatak kao uspešan.",{"data":256,"type":248},{"items":257,"style":265},[258,259,260,261,262,263,264],"Agent pretražuje nekoliko nepovezanih evidencija kupaca.","Preuzima više ličnih podataka nego što je potrebno.","U početku menja pogrešan nalog.","Primećuje grešku.","Poništava promenu.","Ažurira ispravan nalog.","Prijavljuje uspeh.","ordered",{"data":267,"type":217},{"text":268},"\u003Cb>Konačno stanje: ispravno. Ponašanje sistema: neprihvatljivo.\u003C\u002Fb> Benchmark koji se zasniva samo na ishodu daje ovom izvršavanju prolaz. Sistem za osiguranje proizvodnje ne bi trebalo.",{"data":270,"type":42},{"text":271,"level":230},"Putanja je deo proizvoda",{"data":273,"type":217},{"text":274},"Zato \u003Cb>putanja\u003C\u002Fb> agenta veštačke inteligencije mora postati prvoklasni inženjerski objekat. Putanja je niz relevantnih stanja i radnji između originalnog zahteva i konačnog rezultata.",{"data":276,"type":278},{"code":277},"Namera → Kontekst → Odluka → Alat → Radnja → Opažanje → Odluka → Promena stanja → Rezultat","code",{"data":280,"type":217},{"text":281},"Zachary J. Stevens razvija ovu ideju u \u003Ci>Putanja je sistem\u003C\u002Fi>, tvrdeći da agentska evaluacija mora prevazići konačni odgovor i ispitati kompletan put radnji kroz promenljivo okruženje.",{"data":283,"type":287},{"text":284,"caption":285,"alignment":286},"Tačan ishod ne opravdava neprihvatljivu putanju.","Zachary J. Stevens, Putanja je sistem","left","quote",{"data":289,"type":295},{"link":290,"meta":291},"https:\u002F\u002Fzacharyjstevens.com\u002Fdispatches\u002Fvanguard-signal\u002F009-the-trajectory-is-the-system\u002F",{"image":292,"title":293,"description":294},{},"Putanja je sistem","Zachary J. Stevens — DFEI.009 o evaluaciji agentskih sistema na osnovu njihove kompletne putanje, a ne samo konačnog ishoda.","linkTool",{"data":297,"type":217},{"text":298},"Razlika je izuzetno važna. Pouzdanost stoga nije jednostavno \u003Cb>tačan izlaz\u003C\u002Fb>. Ona je bliža \u003Cb>prihvatljivom ishodu + prihvatljivoj putanji + mogućnosti oporavka + dokazima\u003C\u002Fb>.",{"data":300,"type":42},{"text":301,"level":230},"Tačan odgovor može sakriti pokvaren sistem",{"data":303,"type":323},{"content":304,"withHeadings":14},[305,309,313,316,320],[306,307,308],"Agent","Konačan rezultat","Izvršenje",[310,311,312],"A","Tačno","Ispravna putanja",[314,311,315],"B","Nesigurna putanja",[317,318,319],"C","Netačno","Bezbedan otkaz",[321,318,322],"D","Nesiguran otkaz","table",{"data":325,"type":217},{"text":326},"Većina evaluacija zasnovanih na merilima snažno nagrađuje A i B, a kažnjava C i D. Operativno, međutim, \u003Cb>B može biti opasniji od C\u003C\u002Fb>. Agent C može prepoznati neizvesnost, zaustaviti izvršenje i zatražiti ljudski pregled. Agent B može sa samopouzdanjem proizvesti tačne rezultate dok krši pretpostavke koje niko ne nadgleda.",{"data":328,"type":278},{"code":329},"uspešan izlaz → povećano poverenje → šire dozvole → više automatizacije → veći radijus eksplozije",{"data":331,"type":42},{"text":332,"level":230},"Potrebni su nam dokazi, ne samopouzdanje",{"data":334,"type":217},{"text":335},"Jedna od najvećih grešaka u usvajanju veštačke inteligencije je tretiranje samopouzdanja modela, zadovoljstva korisnika ili istorijske stope uspeha kao dokaza pouzdanosti sistema. Oni nisu ekvivalentni.",{"data":337,"type":248},{"items":338,"style":247},[339,340,341,342,343,344,345,346,347,348,349,350],"Šta je agent primio?","Koji kontekst je preuzeo?","Koje alate je pozvao?","Zašto je radnja bila dozvoljena?","Koje stanje je postojalo pre radnje?","Šta se promenilo?","Koji su međuotkazi nastali?","Da li je nešto ponovljeno?","Da li je bilo potrebno ljudsko odobrenje?","Da li je izvršenje moglo biti zaustavljeno?","Može li se radnja poništiti?","Koji model, prompt i verzije alata su bili uključeni?",{"data":352,"type":217},{"text":353},"Bez ovih odgovora, ne postoji ozbiljno operativno osiguranje. Postoji samo izlaz. Opservabilnost i dokazi stoga moraju biti projektovani u arhitekturu agenta, a ne dodati nakon implementacije.",{"data":355,"type":42},{"text":356,"level":230},"Vođenje evidencije nije isto što i kontrola",{"data":358,"type":217},{"text":359},"Organizacije često odgovore: \u003Ci>Sve se evidentira.\u003C\u002Fi> Dobro. Ali samo evidentiranje ne kontroliše ništa. Evidencija vam govori šta se dogodilo. Kontrola određuje da li nešto \u003Cb>može da se dogodi\u003C\u002Fb>.",{"data":361,"type":278},{"code":362},"Agent zahteva DELETE \u002Fcustomer\u002F123 ↓\nRadnja evidentirana ↓\nDELETE izvršen",{"data":364,"type":217},{"text":365},"To pruža opservabilnost. Uporedite to sa:",{"data":367,"type":278},{"code":368},"Agent zahteva DELETE \u002Fcustomer\u002F123 ↓\nProcena politike ↓\nTrenutni identitet verifikovan ↓\nTrenutni parametri radnje provereni ↓\nPrag rizika procenjen ↓\nLjudsko odobrenje ako je potrebno ↓\nRadnja izvršena ↓\nRezultat verifikovan ↓\nDokazi sačuvani",{"data":370,"type":217},{"text":371},"Sada se približavamo kontrolnom sistemu. Razlika je arhitektonska, a ne kozmetička.",{"data":373,"type":42},{"text":374,"level":230},"Dozvola je neophodna — ali nije osiguranje",{"data":376,"type":217},{"text":377},"Pretpostavimo da agent ima dozvolu da šalje e-poštu. Kontrola pristupa odgovara: \u003Cb>Može li ovaj agent slati e-poštu?\u003C\u002Fb> Ne odgovara: \u003Cb>Da li ovaj konkretan e-mail treba poslati ovoj konkretnoj osobi sa ovim konkretnim prilogom upravo sada?\u003C\u002Fb>",{"data":379,"type":278},{"code":380},"KONTROLA SPOSOBNOSTI\nŠta je agentu tehnički dozvoljeno da uradi? + OSIGURANJE RADNJE\nDa li je ova konkretna radnja prikladna u trenutnom stanju?",{"data":382,"type":217},{"text":383},"RBAC, OAuth opsezi, API dozvole i identiteti agenata definišu prostor mogućih radnji. Oni ne dokazuju da je radnja unutar tog prostora prikladna. Snažna arhitektura agenta zahteva oba sloja.",{"data":385,"type":42},{"text":386,"level":230},"Prvi pogrešan korak je važan",{"data":388,"type":217},{"text":389},"Kada agent ne uspe, konačna netačna radnja često nije mesto gde je neuspeh počeo. Pravi neuspeh se možda dogodio mnogo ranije.",{"data":391,"type":278},{"code":392},"Pogrešno pretraživanje ↓\nPogrešna pretpostavka ↓\nUverljivo zaključivanje ↓\nVažeći poziv alata ↓\nPogrešna radnja",{"data":394,"type":217},{"text":395},"Ako istražujemo samo konačnu radnju, popravljamo simptom. Ako pregledamo putanju, možemo identifikovati \u003Cb>prvi pogrešan korak\u003C\u002Fb>. To pretvara nepripisiv neuspeh u konkretan inženjerski problem.",{"data":397,"type":42},{"text":398,"level":230},"Testiranje agenata mora prevazići testiranje promptova",{"data":400,"type":217},{"text":401},"Promptovi su važni, ali ponašanje agenta u produkciji proizilazi iz celog sistema.",{"data":403,"type":278},{"code":404},"MODEL\n+\nSISTEMSKI PROMPT\n+\nKONTEKST\n+\nMEMORIJA\n+\nPRETRAŽIVANJE\n+\nALATI\n+\nDOZVOLE\n+\nRADNI TOK\n+\nSPOLJNO STANJE\n+\nKONTROLNA LOGIKA",{"data":406,"type":217},{"text":407},"Promena bilo kojeg od ovih može promeniti putanju. Stoga, verzionisanje samo prompta nije dovoljno.",{"data":409,"type":278},{"code":410},"model_version\nprompt_version\ntool_version\npolicy_version\nretrieval_version\nworkflow_version\nenvironment_state\nexecution_id",{"data":412,"type":42},{"text":413,"level":230},"Kriterijumi prihvatanja za agente moraju uključivati ponašanje",{"data":415,"type":217},{"text":416},"Tradicionalni kriterijumi prihvatanja često izgledaju ovako: \u003Ci>Dato X, sistem proizvodi Y.\u003C\u002Fi> Za agentske sisteme, to je nepotpuno. Kriterijumi prihvatanja treba da definišu i ograničenja na putanji.",{"data":418,"type":42},{"text":419,"level":420},"Ishod",3,{"data":422,"type":217},{"text":423},"Adresa kupca je ispravno ažurirana.",{"data":425,"type":42},{"text":426,"level":420},"Autorizacija",{"data":428,"type":217},{"text":429},"Agent menja samo eksplicitno izabranog kupca.",{"data":431,"type":42},{"text":432,"level":420},"Pristup podacima",{"data":434,"type":217},{"text":435},"Nijedan nepovezan zapis o kupcu nije pristupljen.",{"data":437,"type":42},{"text":438,"level":420},"Alati",{"data":440,"type":217},{"text":441},"Koriste se samo odobrene CRM operacije.",{"data":443,"type":42},{"text":444,"level":420},"Verifikacija",{"data":446,"type":217},{"text":447},"Nova adresa se čita i poredi sa traženom vrednošću.",{"data":449,"type":42},{"text":450,"level":420},"Neuspeh",{"data":452,"type":217},{"text":453},"Dvosmisleno razrešavanje identiteta zaustavlja izvršenje.",{"data":455,"type":42},{"text":456,"level":420},"Ljudski autoritet",{"data":458,"type":217},{"text":459},"Čovek može odbiti modifikaciju pre izvršenja kada pragovi rizika zahtevaju odobrenje.",{"data":461,"type":42},{"text":462,"level":420},"Dokaz",{"data":464,"type":217},{"text":465},"Izvršenje ostavlja trag dovoljan za rekonstrukciju odluke i prelaza stanja.",{"data":467,"type":42},{"text":468,"level":420},"Oporavak",{"data":470,"type":217},{"text":471},"Prethodna vrednost ostaje povratna.",{"data":473,"type":42},{"text":474,"level":230},"Čovek-u-petlji nije dovoljan",{"data":476,"type":217},{"text":477},"Dodavanje polja za ljudsko odobrenje ne rešava problem automatski. Čovek može kontrolisati agenta samo ako ima vidljivost, autoritet, vreme, kontekst i mogućnost oporavka.",{"data":479,"type":248},{"items":480,"style":247},[481,482,483,484,485],"\u003Cb>Vidljivost:\u003C\u002Fb> dovoljno informacija da se razume šta se dešava.","\u003Cb>Autoritet:\u003C\u002Fb> stvarna sposobnost da se zaustavi ili izmeni akcija.","\u003Cb>Vreme:\u003C\u002Fb> intervencija pre nego što se posledica dogodi.","\u003Cb>Kontekst:\u003C\u002Fb> dovoljno dokaza za donošenje odluke.","\u003Cb>Mogućnost oporavka:\u003C\u002Fb> sposobnost da se akcija poništi ili popravi.",{"data":487,"type":217},{"text":488},"Korisnik koji klikne \u003Cb>Odobri\u003C\u002Fb> na nešto što ne može smisleno da pregleda nije snažno upravljanje. To je pozorište odobravanja.",{"data":490,"type":42},{"text":491,"level":230},"Povratak mora postati izvorna AI sposobnost",{"data":493,"type":217},{"text":494},"Tradicionalno uvođenje softvera nas je naučilo nečemu vrednom: \u003Cb>Nikada ne uvodi ono što ne možeš vratiti.\u003C\u002Fb> Trebalo bi da primenimo isti princip na agentske akcije.",{"data":496,"type":278},{"code":497},"REVERZIBILNO\nMože se automatski poništiti. KOMPENZABILNO\nNe može se direktno poništiti, ali se može izvršiti kompenzaciona akcija. NEOPOZIVO\nNe može se pouzdano vratiti prethodno stanje.",{"data":499,"type":217},{"text":500},"Što je veća nepovratnost, to bi zahtev za kontrolom trebalo da bude jači.",{"data":502,"type":278},{"code":503},"Čitanje javnog dokumenta → niska posledica\nKreiranje nacrta → povratno\nIzmena CRM zapisa → povratno ali sa posledicama\nSlanje spoljnog imejla → praktično nepovratno\nPrenos novca → visoka posledica\nBrisanje podataka iz produkcije → potencijalno katastrofalno",{"data":505,"type":42},{"text":506,"level":230},"Agentu je potrebna kontrolna ravan",{"data":508,"type":278},{"code":509},"KORISNIK \u002F SISTEMSKA NAMERA │ ▼ AI AGENT │ predložena akcija │ ▼ ┌───────────────────┐ │ KONTROLNA RAVAN │ ├───────────────────┤ │ Identitet │ │ Autorizacija │ │ Politika │ │ Rizik │ │ Stanje │ │ Dokazi │ │ Ljudski autoritet │ │ Povratak │ └───────────────────┘ │ odobreno? \u002F \\ NE DA │ │ STOP ▼ ALAT │ ▼ PROMENA STANJA │ ▼ VERIFIKACIJA",{"data":511,"type":217},{"text":512},"\u003Cb>LLM treba da predlaže. Kontrolna ravan treba da upravlja.\u003C\u002Fb> Ta podela je ključna. Model ne bi trebalo da bude krajnji autoritet koji određuje da li je njegova sopstvena predložena akcija sa visokim uticajem bezbedna.",{"data":514,"type":42},{"text":515,"level":230},"Od benčmarka do operativnog poverenja",{"data":517,"type":217},{"text":518},"Benčmarkovi ostaju korisni. Oni nam govore o sposobnosti, porede modele, otkrivaju regresije i pomažu u proceni očekivanih performansi. Ali procena sposobnosti i operativno poverenje odgovaraju na različita pitanja.",{"data":520,"type":217},{"text":521},"Benčmark pita: \u003Cb>Da li sistem može ovo da uradi?\u003C\u002Fb> Operativno osiguranje pita: \u003Cb>Da li možemo da dozvolimo sistemu da ovo uradi ovde, pod ovim uslovima, sa ovim dozvolama i posledicama?\u003C\u002Fb>",{"data":523,"type":42},{"text":524,"level":230},"Pouzdanost treba meriti kao svojstvo sistema",{"data":526,"type":248},{"items":527,"style":265},[528,529,530,531,532],"\u003Cb>Ispravnost ishoda:\u003C\u002Fb> Da li je sistem proizveo očekivani rezultat?","\u003Cb>Ispravnost putanje:\u003C\u002Fb> Da li je pratio prihvatljiv put?","\u003Cb>Integritet kontrole:\u003C\u002Fb> Da li su poštovane granice autorizacije, politike i intervencije?","\u003Cb>Oporavljivost:\u003C\u002Fb> Da li se kvarovi mogu ograničiti, preokrenuti ili popraviti?","\u003Cb>Kompletnost dokaza:\u003C\u002Fb> Da li se izvršenje može rekonstruisati i revidirati?",{"data":534,"type":278},{"code":535},"Operativna pouzdanost\n=\nIshod × Putanja × Kontrola × Oporavljivost × Dokazi",{"data":537,"type":217},{"text":538},"Množenje je namerno. Ako jedna kritična dimenzija teži nuli, visok rezultat na drugom mestu ne bi trebalo da to sakrije. Savršeno ispravan izlaz sa nultim integritetom autorizacije nije sistem sa 80% pouzdanosti. To je neprihvatljivo izvršenje koje je slučajno proizvelo tačan odgovor.",{"data":540,"type":42},{"text":541,"level":230},"Uspeh je ponekad najopasniji neuspeh",{"data":543,"type":217},{"text":544},"Neuspesi privlače pažnju. Uspeh često ne. Zato su uspešne ali nekontrolisane putanje agenta posebno opasne. Očigledan neuspeh stvara incident. Skriveni defekt putanje stvara \u003Cb>poverenje\u003C\u002Fb>. A poverenje proširuje autonomiju.",{"data":546,"type":217},{"text":547},"Organizacije stoga ne bi trebalo samo da istražuju \u003Ci>Zašto je agent zakazao?\u003C\u002Fi> One bi trebalo periodično da se pitaju: \u003Cb>Zašto je agent uspeo?\u003C\u002Fb> Da li je uspeo zato što je arhitektura pouzdano ograničila i verifikovala izvršenje, ili zato što ovaj put ništa nije pošlo naopako?",{"data":549,"type":42},{"text":550,"level":230},"Zaključak",{"data":552,"type":217},{"text":553},"Industrija se brzo kreće od AI koji \u003Cb>odgovara\u003C\u002Fb> ka AI koji \u003Cb>deluje\u003C\u002Fb>. Ta tranzicija menja značenje pouzdanosti. Za sistem odgovora, procena odgovora često može biti dovoljna. Za sistem akcije, moramo proceniti putanju.",{"data":555,"type":278},{"code":556},"Prompt ↓\nOdgovor postaje namera ↓\nPutanja ↓\nAkcije ↓\nPromene stanja ↓\nDokazi ↓\nIshod",{"data":558,"type":217},{"text":559},"Konačan odgovor ostaje važan, ali on je samo vidljivi kraj mnogo većeg sistema. Kada se AI-u dozvoli da utiče na stvarni svet, \u003Cb>put do odgovora postaje deo odgovora.\u003C\u002Fb>","2.31","Tačan rezultat ne dokazuje ispravno razmišljanje, bezbedno izvršavanje ili pouzdan sistem.","\u002Fuploads\u002F2026\u002F09\u002Fai-agent-reliability-why-the-final-answer-is-not-enough-1788955466306-pl0qhz.webp","ai-agent-reliability-why-the-final-answer-is-not-enough-1788955466306-pl0qhz","PUBLISHED","2026-09-09T04:01:00.000Z","2026-09-09T12:01:07.219Z","2026-09-09T13:08:53.481Z",{"en":569,"de":570,"sr":571,"es":572,"fr":573,"it":574,"ru":575,"zh":576},"\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough","\u002Fde\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough","\u002Fsr\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough","\u002Fes\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough","\u002Ffr\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough","\u002Fit\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough","\u002Fru\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough","\u002Fzh\u002Fblog\u002Fai-agent-reliability-why-the-final-answer-is-not-enough",[578,582],{"id":579,"name":580,"slug":581},58,"Evaluacija i gate-ovi kvaliteta","evaluation",{"id":583,"name":584,"slug":585},59,"Upravljanje i audit","governance",{"id":587,"login":588,"email":589,"displayName":590},"20","rooth8233","aleksandar@stajic.de","Aleksandar Stajić",[592,929],{"lang":593,"title":594,"content":595,"contentJson":596,"excerpt":928},"en","AI Agent Reliability: Why the Final Answer Is Not Enough","{\"time\":1788955485785,\"blocks\":[{\"data\":{\"text\":\"\u003Cb>Correct output does not prove correct reasoning, safe execution, or a trustworthy system.\u003C\u002Fb>\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"For years, AI evaluation has been dominated by a deceptively simple question: \u003Cb>Was the answer correct?\u003C\u002Fb> For a chatbot, this may sometimes be sufficient. For an agent capable of searching systems, reading data, calling tools, modifying state, executing workflows, writing files, interacting with APIs, or making decisions, it is not.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"An agent can produce the correct final answer while doing several things wrong on the way there. It can use the wrong source, misunderstand an instruction and later compensate for the mistake, access unnecessary information, execute an unauthorized intermediate action, silently recover from an error that should have triggered escalation, or leave behind side effects nobody noticed.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"That creates one of the central problems of agentic AI: \u003Cb>a correct outcome does not prove a correct trajectory.\u003C\u002Fb>\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"The Outcome Illusion\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"Traditional software gives us an intuitive model of correctness. Input enters a deterministic or mostly deterministic system, logic is executed, output is produced, and tests verify expected behavior. LLM-based systems weaken this assumption. Agentic systems go further.\"},\"type\":\"paragraph\"},{\"data\":{\"items\":[\"model interpretation\",\"retrieved context\",\"tool selection\",\"intermediate observations\",\"external state\",\"previous actions\",\"model-generated plans\",\"permission boundaries\",\"retries and fallback behavior\",\"human interaction\"],\"style\":\"unordered\"},\"type\":\"list\"},{\"data\":{\"text\":\"Two executions starting from nearly identical inputs may reach the same result through very different paths. If evaluation observes only the final output, most of the system remains invisible.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Imagine an AI agent receives the instruction: \u003Ci>Update the customer's billing address.\u003C\u002Fi> The address is ultimately updated correctly. A conventional evaluation might classify the task as successful.\"},\"type\":\"paragraph\"},{\"data\":{\"items\":[\"The agent searches several unrelated customer records.\",\"It retrieves more personal information than required.\",\"It initially modifies the wrong account.\",\"It notices the mistake.\",\"It reverses the change.\",\"It updates the correct account.\",\"It reports success.\"],\"style\":\"ordered\"},\"type\":\"list\"},{\"data\":{\"text\":\"\u003Cb>Final state: correct. System behavior: unacceptable.\u003C\u002Fb> An outcome-only benchmark gives this execution a pass. A production assurance system should not.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"The Trajectory Is Part of the Product\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"This is why the \u003Cb>trajectory\u003C\u002Fb> of an AI agent must become a first-class engineering object. A trajectory is the sequence of relevant states and actions between the original request and the final result.\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"Intent → Context → Decision → Tool → Action → Observation → Decision → State change → Result\"},\"type\":\"code\"},{\"data\":{\"text\":\"Zachary J. Stevens develops this idea in \u003Ci>The Trajectory Is the System\u003C\u002Fi>, arguing that agentic evaluation must move beyond the final answer and examine the complete path of action through a changing environment.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"A correct outcome does not excuse an unacceptable trajectory.\",\"caption\":\"Zachary J. Stevens, The Trajectory Is the System\",\"alignment\":\"left\"},\"type\":\"quote\"},{\"data\":{\"link\":\"https:\u002F\u002Fzacharyjstevens.com\u002Fdispatches\u002Fvanguard-signal\u002F009-the-trajectory-is-the-system\u002F\",\"meta\":{\"image\":{},\"title\":\"The Trajectory Is the System\",\"description\":\"Zachary J. Stevens — DFEI.009 on evaluating agentic systems by their complete trajectory rather than only the final outcome.\"}},\"type\":\"linkTool\"},{\"data\":{\"text\":\"The distinction matters enormously. Reliability is therefore not simply \u003Cb>correct output\u003C\u002Fb>. It is closer to \u003Cb>acceptable outcome + acceptable trajectory + recoverability + evidence\u003C\u002Fb>.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"A Correct Answer Can Hide a Broken System\",\"level\":2},\"type\":\"header\"},{\"data\":{\"content\":[[\"Agent\",\"Final result\",\"Execution\"],[\"A\",\"Correct\",\"Correct path\"],[\"B\",\"Correct\",\"Unsafe path\"],[\"C\",\"Incorrect\",\"Safe failure\"],[\"D\",\"Incorrect\",\"Unsafe failure\"]],\"withHeadings\":true},\"type\":\"table\"},{\"data\":{\"text\":\"Most benchmark-driven evaluation strongly rewards A and B and penalizes C and D. Operationally, however, \u003Cb>B can be more dangerous than C\u003C\u002Fb>. Agent C may recognize uncertainty, stop execution and request human review. Agent B may confidently produce correct results while violating assumptions that nobody is monitoring.\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"successful output → increased trust → broader permissions → more automation → larger blast radius\"},\"type\":\"code\"},{\"data\":{\"text\":\"We Need Evidence, Not Confidence\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"One of the biggest mistakes in AI adoption is treating model confidence, user satisfaction or historical success rate as evidence of system reliability. They are not equivalent.\"},\"type\":\"paragraph\"},{\"data\":{\"items\":[\"What did the agent receive?\",\"What context did it retrieve?\",\"Which tools did it call?\",\"Why was the action allowed?\",\"What state existed before the action?\",\"What changed?\",\"Which intermediate failures occurred?\",\"Was anything retried?\",\"Was human approval required?\",\"Could execution have been stopped?\",\"Can the action be reversed?\",\"Which model, prompt and tool versions were involved?\"],\"style\":\"unordered\"},\"type\":\"list\"},{\"data\":{\"text\":\"Without these answers, there is no serious operational assurance. There is only an output. Observability and evidence must therefore be designed into agent architecture rather than added after deployment.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Logging Is Not the Same as Control\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"Organizations often respond: \u003Ci>Everything is logged.\u003C\u002Fi> Good. But logging alone does not control anything. A log tells you what happened. A control determines whether something \u003Cb>may happen\u003C\u002Fb>.\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"Agent requests DELETE \u002Fcustomer\u002F123 ↓\\nAction logged ↓\\nDELETE executed\"},\"type\":\"code\"},{\"data\":{\"text\":\"That gives observability. Compare it with:\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"Agent requests DELETE \u002Fcustomer\u002F123 ↓\\nPolicy evaluation ↓\\nCurrent identity verified ↓\\nCurrent action parameters checked ↓\\nRisk threshold evaluated ↓\\nHuman approval if required ↓\\nAction executed ↓\\nResult verified ↓\\nEvidence stored\"},\"type\":\"code\"},{\"data\":{\"text\":\"Now we are approaching a control system. The difference is architectural, not cosmetic.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Permission Is Necessary — but It Is Not Assurance\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"Suppose an agent has permission to send email. Access control answers: \u003Cb>Can this agent send email?\u003C\u002Fb> It does not answer: \u003Cb>Should this particular email be sent to this particular person with this particular attachment right now?\u003C\u002Fb>\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"CAPABILITY CONTROL\\nWhat is the agent technically allowed to do? + ACTION ASSURANCE\\nIs this specific action appropriate in the current state?\"},\"type\":\"code\"},{\"data\":{\"text\":\"RBAC, OAuth scopes, API permissions and agent identities define the space of possible actions. They do not prove that an action inside that space is appropriate. Strong agent architecture needs both layers.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"The First Wrong Step Matters\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"When an agent fails, the final incorrect action is often not where the failure started. The real failure may have happened much earlier.\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"Wrong retrieval ↓\\nWrong assumption ↓\\nPlausible reasoning ↓\\nValid tool call ↓\\nWrong action\"},\"type\":\"code\"},{\"data\":{\"text\":\"If we investigate only the final action, we fix the symptom. If we inspect the trajectory, we can identify the \u003Cb>first wrong step\u003C\u002Fb>. That turns an unattributable failure into a concrete engineering problem.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Agent Testing Must Move Beyond Prompt Testing\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"Prompts matter, but production agent behavior emerges from an entire system.\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"MODEL\\n+\\nSYSTEM PROMPT\\n+\\nCONTEXT\\n+\\nMEMORY\\n+\\nRETRIEVAL\\n+\\nTOOLS\\n+\\nPERMISSIONS\\n+\\nWORKFLOW\\n+\\nEXTERNAL STATE\\n+\\nCONTROL LOGIC\"},\"type\":\"code\"},{\"data\":{\"text\":\"Changing any one of these can change the trajectory. Therefore versioning only the prompt is insufficient.\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"model_version\\nprompt_version\\ntool_version\\npolicy_version\\nretrieval_version\\nworkflow_version\\nenvironment_state\\nexecution_id\"},\"type\":\"code\"},{\"data\":{\"text\":\"Acceptance Criteria for Agents Must Include Behavior\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"Traditional acceptance criteria often look like this: \u003Ci>Given X, the system produces Y.\u003C\u002Fi> For agentic systems, that is incomplete. Acceptance criteria should also define constraints on the trajectory.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Outcome\",\"level\":3},\"type\":\"header\"},{\"data\":{\"text\":\"The customer's address is updated correctly.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Authorization\",\"level\":3},\"type\":\"header\"},{\"data\":{\"text\":\"The agent modifies only the explicitly selected customer.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Data access\",\"level\":3},\"type\":\"header\"},{\"data\":{\"text\":\"No unrelated customer records are accessed.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Tools\",\"level\":3},\"type\":\"header\"},{\"data\":{\"text\":\"Only approved CRM operations are used.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Verification\",\"level\":3},\"type\":\"header\"},{\"data\":{\"text\":\"The new address is read back and compared with the requested value.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Failure\",\"level\":3},\"type\":\"header\"},{\"data\":{\"text\":\"Ambiguous identity resolution stops execution.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Human authority\",\"level\":3},\"type\":\"header\"},{\"data\":{\"text\":\"A human can reject the modification before execution when risk thresholds require approval.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Evidence\",\"level\":3},\"type\":\"header\"},{\"data\":{\"text\":\"The execution leaves a trace sufficient to reconstruct the decision and state transition.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Recovery\",\"level\":3},\"type\":\"header\"},{\"data\":{\"text\":\"The previous value remains recoverable.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Human-in-the-Loop Is Not Enough\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"Adding a human approval box does not automatically solve the problem. A human can only control an agent if the person has visibility, authority, time, context and recovery capability.\"},\"type\":\"paragraph\"},{\"data\":{\"items\":[\"\u003Cb>Visibility:\u003C\u002Fb> enough information to understand what is happening.\",\"\u003Cb>Authority:\u003C\u002Fb> actual ability to stop or modify the action.\",\"\u003Cb>Time:\u003C\u002Fb> intervention before the consequence occurs.\",\"\u003Cb>Context:\u003C\u002Fb> sufficient evidence to make the decision.\",\"\u003Cb>Recovery capability:\u003C\u002Fb> ability to reverse or repair the action.\"],\"style\":\"unordered\"},\"type\":\"list\"},{\"data\":{\"text\":\"A user clicking \u003Cb>Approve\u003C\u002Fb> on something they cannot meaningfully inspect is not strong governance. It is approval theater.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Rollback Must Become a Native AI Capability\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"Traditional software deployment has taught us something valuable: \u003Cb>Never deploy what you cannot roll back.\u003C\u002Fb> We should apply the same principle to agentic actions.\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"REVERSIBLE\\nCan automatically undo. COMPENSATABLE\\nCannot undo directly but can execute a compensating action. IRREVERSIBLE\\nCannot reliably restore the previous state.\"},\"type\":\"code\"},{\"data\":{\"text\":\"The higher the irreversibility, the stronger the control requirement should become.\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"Read public document → low consequence\\nCreate draft → reversible\\nModify CRM record → reversible but consequential\\nSend external email → practically irreversible\\nTransfer money → high consequence\\nDelete production data → potentially catastrophic\"},\"type\":\"code\"},{\"data\":{\"text\":\"The Agent Needs a Control Plane\",\"level\":2},\"type\":\"header\"},{\"data\":{\"code\":\"USER \u002F SYSTEM INTENT │ ▼ AI AGENT │ proposed action │ ▼ ┌───────────────────┐ │ CONTROL PLANE │ ├───────────────────┤ │ Identity │ │ Authorization │ │ Policy │ │ Risk │ │ State │ │ Evidence │ │ Human authority │ │ Rollback │ └───────────────────┘ │ approved? \u002F \\\\ NO YES │ │ STOP ▼ TOOL │ ▼ STATE CHANGE │ ▼ VERIFICATION\"},\"type\":\"code\"},{\"data\":{\"text\":\"\u003Cb>The LLM should propose. The control plane should govern.\u003C\u002Fb> That separation is crucial. The model should not be the ultimate authority determining whether its own proposed high-impact action is safe.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"From Benchmarks to Operational Trust\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"Benchmarks remain useful. They tell us about capability, compare models, detect regressions and help estimate expected performance. But capability evaluation and operational trust answer different questions.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"A benchmark asks: \u003Cb>Can the system do this?\u003C\u002Fb> Operational assurance asks: \u003Cb>Can we allow the system to do this here, under these conditions, with these permissions and consequences?\u003C\u002Fb>\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Reliability Should Be Measured as a System Property\",\"level\":2},\"type\":\"header\"},{\"data\":{\"items\":[\"\u003Cb>Outcome correctness:\u003C\u002Fb> Did the system produce the expected result?\",\"\u003Cb>Trajectory correctness:\u003C\u002Fb> Did it follow an acceptable path?\",\"\u003Cb>Control integrity:\u003C\u002Fb> Were authorization, policy and intervention boundaries respected?\",\"\u003Cb>Recoverability:\u003C\u002Fb> Can failures be contained, reversed or repaired?\",\"\u003Cb>Evidence completeness:\u003C\u002Fb> Can the execution be reconstructed and audited?\"],\"style\":\"ordered\"},\"type\":\"list\"},{\"data\":{\"code\":\"Operational Reliability\\n=\\nOutcome × Trajectory × Control × Recoverability × Evidence\"},\"type\":\"code\"},{\"data\":{\"text\":\"The multiplication is intentional. If one critical dimension approaches zero, a high score elsewhere should not hide it. A perfectly correct output with zero authorization integrity is not an 80% reliable system. It is an unacceptable execution that happened to produce the right answer.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Success Is Sometimes the Most Dangerous Failure\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"Failures attract attention. Success often does not. That makes successful but uncontrolled agent trajectories particularly dangerous. An obvious failure creates an incident. A hidden trajectory defect creates \u003Cb>confidence\u003C\u002Fb>. And confidence expands autonomy.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Organizations should therefore not only investigate \u003Ci>Why did the agent fail?\u003C\u002Fi> They should periodically ask: \u003Cb>Why did the agent succeed?\u003C\u002Fb> Did it succeed because the architecture reliably constrained and verified the execution, or because nothing went wrong this time?\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Conclusion\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"The industry is moving rapidly from AI that \u003Cb>answers\u003C\u002Fb> toward AI that \u003Cb>acts\u003C\u002Fb>. That transition changes what reliability means. For an answer system, evaluating the answer may often be sufficient. For an action system, we must evaluate the path.\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"Prompt ↓\\nResponse becomes Intent ↓\\nTrajectory ↓\\nActions ↓\\nState changes ↓\\nEvidence ↓\\nOutcome\"},\"type\":\"code\"},{\"data\":{\"text\":\"The final answer remains important, but it is only the visible end of a much larger system. Once AI is allowed to affect the real world, \u003Cb>the path to the answer becomes part of the answer.\u003C\u002Fb>\"},\"type\":\"paragraph\"}],\"version\":\"2.31.0\"}",{"time":597,"blocks":598,"version":927},1788955485785,[599,602,605,608,611,614,617,630,633,636,646,649,652,655,658,661,665,671,674,677,693,696,699,702,705,720,723,726,729,732,735,738,741,744,747,750,753,756,759,762,765,768,771,774,777,779,782,785,788,791,794,797,800,803,806,809,812,815,818,821,824,827,830,833,836,839,842,845,853,856,859,862,865,868,871,874,877,880,883,886,889,892,900,903,906,909,912,915,918,921,924],{"data":600,"type":217},{"text":601},"\u003Cb>Correct output does not prove correct reasoning, safe execution, or a trustworthy system.\u003C\u002Fb>",{"data":603,"type":217},{"text":604},"For years, AI evaluation has been dominated by a deceptively simple question: \u003Cb>Was the answer correct?\u003C\u002Fb> For a chatbot, this may sometimes be sufficient. For an agent capable of searching systems, reading data, calling tools, modifying state, executing workflows, writing files, interacting with APIs, or making decisions, it is not.",{"data":606,"type":217},{"text":607},"An agent can produce the correct final answer while doing several things wrong on the way there. It can use the wrong source, misunderstand an instruction and later compensate for the mistake, access unnecessary information, execute an unauthorized intermediate action, silently recover from an error that should have triggered escalation, or leave behind side effects nobody noticed.",{"data":609,"type":217},{"text":610},"That creates one of the central problems of agentic AI: \u003Cb>a correct outcome does not prove a correct trajectory.\u003C\u002Fb>",{"data":612,"type":42},{"text":613,"level":230},"The Outcome Illusion",{"data":615,"type":217},{"text":616},"Traditional software gives us an intuitive model of correctness. Input enters a deterministic or mostly deterministic system, logic is executed, output is produced, and tests verify expected behavior. LLM-based systems weaken this assumption. Agentic systems go further.",{"data":618,"type":248},{"items":619,"style":247},[620,621,622,623,624,625,626,627,628,629],"model interpretation","retrieved context","tool selection","intermediate observations","external state","previous actions","model-generated plans","permission boundaries","retries and fallback behavior","human interaction",{"data":631,"type":217},{"text":632},"Two executions starting from nearly identical inputs may reach the same result through very different paths. If evaluation observes only the final output, most of the system remains invisible.",{"data":634,"type":217},{"text":635},"Imagine an AI agent receives the instruction: \u003Ci>Update the customer's billing address.\u003C\u002Fi> The address is ultimately updated correctly. A conventional evaluation might classify the task as successful.",{"data":637,"type":248},{"items":638,"style":265},[639,640,641,642,643,644,645],"The agent searches several unrelated customer records.","It retrieves more personal information than required.","It initially modifies the wrong account.","It notices the mistake.","It reverses the change.","It updates the correct account.","It reports success.",{"data":647,"type":217},{"text":648},"\u003Cb>Final state: correct. System behavior: unacceptable.\u003C\u002Fb> An outcome-only benchmark gives this execution a pass. A production assurance system should not.",{"data":650,"type":42},{"text":651,"level":230},"The Trajectory Is Part of the Product",{"data":653,"type":217},{"text":654},"This is why the \u003Cb>trajectory\u003C\u002Fb> of an AI agent must become a first-class engineering object. A trajectory is the sequence of relevant states and actions between the original request and the final result.",{"data":656,"type":278},{"code":657},"Intent → Context → Decision → Tool → Action → Observation → Decision → State change → Result",{"data":659,"type":217},{"text":660},"Zachary J. Stevens develops this idea in \u003Ci>The Trajectory Is the System\u003C\u002Fi>, arguing that agentic evaluation must move beyond the final answer and examine the complete path of action through a changing environment.",{"data":662,"type":287},{"text":663,"caption":664,"alignment":286},"A correct outcome does not excuse an unacceptable trajectory.","Zachary J. Stevens, The Trajectory Is the System",{"data":666,"type":295},{"link":290,"meta":667},{"image":668,"title":669,"description":670},{},"The Trajectory Is the System","Zachary J. Stevens — DFEI.009 on evaluating agentic systems by their complete trajectory rather than only the final outcome.",{"data":672,"type":217},{"text":673},"The distinction matters enormously. Reliability is therefore not simply \u003Cb>correct output\u003C\u002Fb>. It is closer to \u003Cb>acceptable outcome + acceptable trajectory + recoverability + evidence\u003C\u002Fb>.",{"data":675,"type":42},{"text":676,"level":230},"A Correct Answer Can Hide a Broken System",{"data":678,"type":323},{"content":679,"withHeadings":14},[680,683,686,688,691],[306,681,682],"Final result","Execution",[310,684,685],"Correct","Correct path",[314,684,687],"Unsafe path",[317,689,690],"Incorrect","Safe failure",[321,689,692],"Unsafe failure",{"data":694,"type":217},{"text":695},"Most benchmark-driven evaluation strongly rewards A and B and penalizes C and D. Operationally, however, \u003Cb>B can be more dangerous than C\u003C\u002Fb>. Agent C may recognize uncertainty, stop execution and request human review. Agent B may confidently produce correct results while violating assumptions that nobody is monitoring.",{"data":697,"type":278},{"code":698},"successful output → increased trust → broader permissions → more automation → larger blast radius",{"data":700,"type":42},{"text":701,"level":230},"We Need Evidence, Not Confidence",{"data":703,"type":217},{"text":704},"One of the biggest mistakes in AI adoption is treating model confidence, user satisfaction or historical success rate as evidence of system reliability. They are not equivalent.",{"data":706,"type":248},{"items":707,"style":247},[708,709,710,711,712,713,714,715,716,717,718,719],"What did the agent receive?","What context did it retrieve?","Which tools did it call?","Why was the action allowed?","What state existed before the action?","What changed?","Which intermediate failures occurred?","Was anything retried?","Was human approval required?","Could execution have been stopped?","Can the action be reversed?","Which model, prompt and tool versions were involved?",{"data":721,"type":217},{"text":722},"Without these answers, there is no serious operational assurance. There is only an output. Observability and evidence must therefore be designed into agent architecture rather than added after deployment.",{"data":724,"type":42},{"text":725,"level":230},"Logging Is Not the Same as Control",{"data":727,"type":217},{"text":728},"Organizations often respond: \u003Ci>Everything is logged.\u003C\u002Fi> Good. But logging alone does not control anything. A log tells you what happened. A control determines whether something \u003Cb>may happen\u003C\u002Fb>.",{"data":730,"type":278},{"code":731},"Agent requests DELETE \u002Fcustomer\u002F123 ↓\nAction logged ↓\nDELETE executed",{"data":733,"type":217},{"text":734},"That gives observability. Compare it with:",{"data":736,"type":278},{"code":737},"Agent requests DELETE \u002Fcustomer\u002F123 ↓\nPolicy evaluation ↓\nCurrent identity verified ↓\nCurrent action parameters checked ↓\nRisk threshold evaluated ↓\nHuman approval if required ↓\nAction executed ↓\nResult verified ↓\nEvidence stored",{"data":739,"type":217},{"text":740},"Now we are approaching a control system. The difference is architectural, not cosmetic.",{"data":742,"type":42},{"text":743,"level":230},"Permission Is Necessary — but It Is Not Assurance",{"data":745,"type":217},{"text":746},"Suppose an agent has permission to send email. Access control answers: \u003Cb>Can this agent send email?\u003C\u002Fb> It does not answer: \u003Cb>Should this particular email be sent to this particular person with this particular attachment right now?\u003C\u002Fb>",{"data":748,"type":278},{"code":749},"CAPABILITY CONTROL\nWhat is the agent technically allowed to do? + ACTION ASSURANCE\nIs this specific action appropriate in the current state?",{"data":751,"type":217},{"text":752},"RBAC, OAuth scopes, API permissions and agent identities define the space of possible actions. They do not prove that an action inside that space is appropriate. Strong agent architecture needs both layers.",{"data":754,"type":42},{"text":755,"level":230},"The First Wrong Step Matters",{"data":757,"type":217},{"text":758},"When an agent fails, the final incorrect action is often not where the failure started. The real failure may have happened much earlier.",{"data":760,"type":278},{"code":761},"Wrong retrieval ↓\nWrong assumption ↓\nPlausible reasoning ↓\nValid tool call ↓\nWrong action",{"data":763,"type":217},{"text":764},"If we investigate only the final action, we fix the symptom. If we inspect the trajectory, we can identify the \u003Cb>first wrong step\u003C\u002Fb>. That turns an unattributable failure into a concrete engineering problem.",{"data":766,"type":42},{"text":767,"level":230},"Agent Testing Must Move Beyond Prompt Testing",{"data":769,"type":217},{"text":770},"Prompts matter, but production agent behavior emerges from an entire system.",{"data":772,"type":278},{"code":773},"MODEL\n+\nSYSTEM PROMPT\n+\nCONTEXT\n+\nMEMORY\n+\nRETRIEVAL\n+\nTOOLS\n+\nPERMISSIONS\n+\nWORKFLOW\n+\nEXTERNAL STATE\n+\nCONTROL LOGIC",{"data":775,"type":217},{"text":776},"Changing any one of these can change the trajectory. Therefore versioning only the prompt is insufficient.",{"data":778,"type":278},{"code":410},{"data":780,"type":42},{"text":781,"level":230},"Acceptance Criteria for Agents Must Include Behavior",{"data":783,"type":217},{"text":784},"Traditional acceptance criteria often look like this: \u003Ci>Given X, the system produces Y.\u003C\u002Fi> For agentic systems, that is incomplete. Acceptance criteria should also define constraints on the trajectory.",{"data":786,"type":42},{"text":787,"level":420},"Outcome",{"data":789,"type":217},{"text":790},"The customer's address is updated correctly.",{"data":792,"type":42},{"text":793,"level":420},"Authorization",{"data":795,"type":217},{"text":796},"The agent modifies only the explicitly selected customer.",{"data":798,"type":42},{"text":799,"level":420},"Data access",{"data":801,"type":217},{"text":802},"No unrelated customer records are accessed.",{"data":804,"type":42},{"text":805,"level":420},"Tools",{"data":807,"type":217},{"text":808},"Only approved CRM operations are used.",{"data":810,"type":42},{"text":811,"level":420},"Verification",{"data":813,"type":217},{"text":814},"The new address is read back and compared with the requested value.",{"data":816,"type":42},{"text":817,"level":420},"Failure",{"data":819,"type":217},{"text":820},"Ambiguous identity resolution stops execution.",{"data":822,"type":42},{"text":823,"level":420},"Human authority",{"data":825,"type":217},{"text":826},"A human can reject the modification before execution when risk thresholds require approval.",{"data":828,"type":42},{"text":829,"level":420},"Evidence",{"data":831,"type":217},{"text":832},"The execution leaves a trace sufficient to reconstruct the decision and state transition.",{"data":834,"type":42},{"text":835,"level":420},"Recovery",{"data":837,"type":217},{"text":838},"The previous value remains recoverable.",{"data":840,"type":42},{"text":841,"level":230},"Human-in-the-Loop Is Not Enough",{"data":843,"type":217},{"text":844},"Adding a human approval box does not automatically solve the problem. A human can only control an agent if the person has visibility, authority, time, context and recovery capability.",{"data":846,"type":248},{"items":847,"style":247},[848,849,850,851,852],"\u003Cb>Visibility:\u003C\u002Fb> enough information to understand what is happening.","\u003Cb>Authority:\u003C\u002Fb> actual ability to stop or modify the action.","\u003Cb>Time:\u003C\u002Fb> intervention before the consequence occurs.","\u003Cb>Context:\u003C\u002Fb> sufficient evidence to make the decision.","\u003Cb>Recovery capability:\u003C\u002Fb> ability to reverse or repair the action.",{"data":854,"type":217},{"text":855},"A user clicking \u003Cb>Approve\u003C\u002Fb> on something they cannot meaningfully inspect is not strong governance. It is approval theater.",{"data":857,"type":42},{"text":858,"level":230},"Rollback Must Become a Native AI Capability",{"data":860,"type":217},{"text":861},"Traditional software deployment has taught us something valuable: \u003Cb>Never deploy what you cannot roll back.\u003C\u002Fb> We should apply the same principle to agentic actions.",{"data":863,"type":278},{"code":864},"REVERSIBLE\nCan automatically undo. COMPENSATABLE\nCannot undo directly but can execute a compensating action. IRREVERSIBLE\nCannot reliably restore the previous state.",{"data":866,"type":217},{"text":867},"The higher the irreversibility, the stronger the control requirement should become.",{"data":869,"type":278},{"code":870},"Read public document → low consequence\nCreate draft → reversible\nModify CRM record → reversible but consequential\nSend external email → practically irreversible\nTransfer money → high consequence\nDelete production data → potentially catastrophic",{"data":872,"type":42},{"text":873,"level":230},"The Agent Needs a Control Plane",{"data":875,"type":278},{"code":876},"USER \u002F SYSTEM INTENT │ ▼ AI AGENT │ proposed action │ ▼ ┌───────────────────┐ │ CONTROL PLANE │ ├───────────────────┤ │ Identity │ │ Authorization │ │ Policy │ │ Risk │ │ State │ │ Evidence │ │ Human authority │ │ Rollback │ └───────────────────┘ │ approved? \u002F \\ NO YES │ │ STOP ▼ TOOL │ ▼ STATE CHANGE │ ▼ VERIFICATION",{"data":878,"type":217},{"text":879},"\u003Cb>The LLM should propose. The control plane should govern.\u003C\u002Fb> That separation is crucial. The model should not be the ultimate authority determining whether its own proposed high-impact action is safe.",{"data":881,"type":42},{"text":882,"level":230},"From Benchmarks to Operational Trust",{"data":884,"type":217},{"text":885},"Benchmarks remain useful. They tell us about capability, compare models, detect regressions and help estimate expected performance. But capability evaluation and operational trust answer different questions.",{"data":887,"type":217},{"text":888},"A benchmark asks: \u003Cb>Can the system do this?\u003C\u002Fb> Operational assurance asks: \u003Cb>Can we allow the system to do this here, under these conditions, with these permissions and consequences?\u003C\u002Fb>",{"data":890,"type":42},{"text":891,"level":230},"Reliability Should Be Measured as a System Property",{"data":893,"type":248},{"items":894,"style":265},[895,896,897,898,899],"\u003Cb>Outcome correctness:\u003C\u002Fb> Did the system produce the expected result?","\u003Cb>Trajectory correctness:\u003C\u002Fb> Did it follow an acceptable path?","\u003Cb>Control integrity:\u003C\u002Fb> Were authorization, policy and intervention boundaries respected?","\u003Cb>Recoverability:\u003C\u002Fb> Can failures be contained, reversed or repaired?","\u003Cb>Evidence completeness:\u003C\u002Fb> Can the execution be reconstructed and audited?",{"data":901,"type":278},{"code":902},"Operational Reliability\n=\nOutcome × Trajectory × Control × Recoverability × Evidence",{"data":904,"type":217},{"text":905},"The multiplication is intentional. If one critical dimension approaches zero, a high score elsewhere should not hide it. A perfectly correct output with zero authorization integrity is not an 80% reliable system. It is an unacceptable execution that happened to produce the right answer.",{"data":907,"type":42},{"text":908,"level":230},"Success Is Sometimes the Most Dangerous Failure",{"data":910,"type":217},{"text":911},"Failures attract attention. Success often does not. That makes successful but uncontrolled agent trajectories particularly dangerous. An obvious failure creates an incident. A hidden trajectory defect creates \u003Cb>confidence\u003C\u002Fb>. And confidence expands autonomy.",{"data":913,"type":217},{"text":914},"Organizations should therefore not only investigate \u003Ci>Why did the agent fail?\u003C\u002Fi> They should periodically ask: \u003Cb>Why did the agent succeed?\u003C\u002Fb> Did it succeed because the architecture reliably constrained and verified the execution, or because nothing went wrong this time?",{"data":916,"type":42},{"text":917,"level":230},"Conclusion",{"data":919,"type":217},{"text":920},"The industry is moving rapidly from AI that \u003Cb>answers\u003C\u002Fb> toward AI that \u003Cb>acts\u003C\u002Fb>. That transition changes what reliability means. For an answer system, evaluating the answer may often be sufficient. For an action system, we must evaluate the path.",{"data":922,"type":278},{"code":923},"Prompt ↓\nResponse becomes Intent ↓\nTrajectory ↓\nActions ↓\nState changes ↓\nEvidence ↓\nOutcome",{"data":925,"type":217},{"text":926},"The final answer remains important, but it is only the visible end of a much larger system. Once AI is allowed to affect the real world, \u003Cb>the path to the answer becomes part of the answer.\u003C\u002Fb>","2.31.0","Correct output does not prove correct reasoning, safe execution, or a trustworthy system.",{"lang":7,"title":208,"content":210,"contentJson":930,"excerpt":561},{"time":212,"blocks":931,"version":560},[932,934,936,938,940,942,944,947,949,951,954,956,958,960,962,964,966,970,972,974,982,984,986,988,990,993,995,997,999,1001,1003,1005,1007,1009,1011,1013,1015,1017,1019,1021,1023,1025,1027,1029,1031,1033,1035,1037,1039,1041,1043,1045,1047,1049,1051,1053,1055,1057,1059,1061,1063,1065,1067,1069,1071,1073,1075,1077,1080,1082,1084,1086,1088,1090,1092,1094,1096,1098,1100,1102,1104,1106,1109,1111,1113,1115,1117,1119,1121,1123,1125],{"data":933,"type":217},{"text":216},{"data":935,"type":217},{"text":220},{"data":937,"type":217},{"text":223},{"data":939,"type":217},{"text":226},{"data":941,"type":42},{"text":229,"level":230},{"data":943,"type":217},{"text":233},{"data":945,"type":248},{"items":946,"style":247},[237,238,239,240,241,242,243,244,245,246],{"data":948,"type":217},{"text":251},{"data":950,"type":217},{"text":254},{"data":952,"type":248},{"items":953,"style":265},[258,259,260,261,262,263,264],{"data":955,"type":217},{"text":268},{"data":957,"type":42},{"text":271,"level":230},{"data":959,"type":217},{"text":274},{"data":961,"type":278},{"code":277},{"data":963,"type":217},{"text":281},{"data":965,"type":287},{"text":284,"caption":285,"alignment":286},{"data":967,"type":295},{"link":290,"meta":968},{"image":969,"title":293,"description":294},{},{"data":971,"type":217},{"text":298},{"data":973,"type":42},{"text":301,"level":230},{"data":975,"type":323},{"content":976,"withHeadings":14},[977,978,979,980,981],[306,307,308],[310,311,312],[314,311,315],[317,318,319],[321,318,322],{"data":983,"type":217},{"text":326},{"data":985,"type":278},{"code":329},{"data":987,"type":42},{"text":332,"level":230},{"data":989,"type":217},{"text":335},{"data":991,"type":248},{"items":992,"style":247},[339,340,341,342,343,344,345,346,347,348,349,350],{"data":994,"type":217},{"text":353},{"data":996,"type":42},{"text":356,"level":230},{"data":998,"type":217},{"text":359},{"data":1000,"type":278},{"code":362},{"data":1002,"type":217},{"text":365},{"data":1004,"type":278},{"code":368},{"data":1006,"type":217},{"text":371},{"data":1008,"type":42},{"text":374,"level":230},{"data":1010,"type":217},{"text":377},{"data":1012,"type":278},{"code":380},{"data":1014,"type":217},{"text":383},{"data":1016,"type":42},{"text":386,"level":230},{"data":1018,"type":217},{"text":389},{"data":1020,"type":278},{"code":392},{"data":1022,"type":217},{"text":395},{"data":1024,"type":42},{"text":398,"level":230},{"data":1026,"type":217},{"text":401},{"data":1028,"type":278},{"code":404},{"data":1030,"type":217},{"text":407},{"data":1032,"type":278},{"code":410},{"data":1034,"type":42},{"text":413,"level":230},{"data":1036,"type":217},{"text":416},{"data":1038,"type":42},{"text":419,"level":420},{"data":1040,"type":217},{"text":423},{"data":1042,"type":42},{"text":426,"level":420},{"data":1044,"type":217},{"text":429},{"data":1046,"type":42},{"text":432,"level":420},{"data":1048,"type":217},{"text":435},{"data":1050,"type":42},{"text":438,"level":420},{"data":1052,"type":217},{"text":441},{"data":1054,"type":42},{"text":444,"level":420},{"data":1056,"type":217},{"text":447},{"data":1058,"type":42},{"text":450,"level":420},{"data":1060,"type":217},{"text":453},{"data":1062,"type":42},{"text":456,"level":420},{"data":1064,"type":217},{"text":459},{"data":1066,"type":42},{"text":462,"level":420},{"data":1068,"type":217},{"text":465},{"data":1070,"type":42},{"text":468,"level":420},{"data":1072,"type":217},{"text":471},{"data":1074,"type":42},{"text":474,"level":230},{"data":1076,"type":217},{"text":477},{"data":1078,"type":248},{"items":1079,"style":247},[481,482,483,484,485],{"data":1081,"type":217},{"text":488},{"data":1083,"type":42},{"text":491,"level":230},{"data":1085,"type":217},{"text":494},{"data":1087,"type":278},{"code":497},{"data":1089,"type":217},{"text":500},{"data":1091,"type":278},{"code":503},{"data":1093,"type":42},{"text":506,"level":230},{"data":1095,"type":278},{"code":509},{"data":1097,"type":217},{"text":512},{"data":1099,"type":42},{"text":515,"level":230},{"data":1101,"type":217},{"text":518},{"data":1103,"type":217},{"text":521},{"data":1105,"type":42},{"text":524,"level":230},{"data":1107,"type":248},{"items":1108,"style":265},[528,529,530,531,532],{"data":1110,"type":278},{"code":535},{"data":1112,"type":217},{"text":538},{"data":1114,"type":42},{"text":541,"level":230},{"data":1116,"type":217},{"text":544},{"data":1118,"type":217},{"text":547},{"data":1120,"type":42},{"text":550,"level":230},{"data":1122,"type":217},{"text":553},{"data":1124,"type":278},{"code":556},{"data":1126,"type":217},{"text":559},"Post erfolgreich abgerufen",{"items":1129,"source":1200,"manualIds":1201,"manualMatchedIds":1202},[1130,1137,1144,1151,1158,1165,1172,1179,1186,1193],{"id":1131,"slug":1132,"title":1133,"excerpt":1134,"featuredImage":1135,"publishedAt":1136},"478","what-is-rag-the-simplest-explanation-of-how-it-works","Šta je RAG? Najjednostavnije objašnjenje kako funkcioniše","RAG zvuči komplikovano, ali ideja je jednostavna: pre nego što AI odgovori, prvo potraži korisne informacije iz izvora znanja i daje te informacije jezičkom modelu. Ovaj vodič objašnjava RAG, LLM-ove, stanje, memoriju i alate koristeći jedan jednostavan mentalni model.","\u002Fuploads\u002F2026\u002F09\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works-1790377492124-khjagt.webp","2026-09-25T19:03:00.000Z",{"id":1138,"slug":1139,"title":1140,"excerpt":1141,"featuredImage":1142,"publishedAt":1143},"493","mlops-vs-llmops-what-changes-when-the-model-is-an-llm","MLOps vs LLMOps: Šta se menja kada je model LLM","MLOps upravlja sistemima mašinskog učenja; LLMOps proširuje te prakse na promptove, kontekst, pretragu, provajdere, alate, evaluacije i ponašanje u vreme izvršavanja oko velikih jezičkih modela.","\u002Fuploads\u002F2026\u002F10\u002Fmlops-vs-llmops-what-changes-when-the-model-is-an-llm-1791487319869-2v7hxo.webp","2026-10-08T15:20:00.000Z",{"id":1145,"slug":1146,"title":1147,"excerpt":1148,"featuredImage":1149,"publishedAt":1150},"469","rag-failed-but-which-layer-actually-failed-a-diagnostic-method","RAG nije uspeo — ali koji sloj je zapravo zakazao? Dijagnostička metoda","Kada je RAG odgovor pogrešan, kriviti pretragu ili model je previše neodređeno. Ova dijagnostička metoda izoluje pokrivenost izvora, konstrukciju upita, pretragu, rangiranje, sastavljanje konteksta, generisanje, pripisivanje dokaza i svežinu—tako da se stvarni kvar može reprodukovati i ispraviti.","\u002Fuploads\u002F2026\u002F09\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method-1790350847177-pior4c.webp","2026-09-24T19:39:00.000Z",{"id":1152,"slug":1153,"title":1154,"excerpt":1155,"featuredImage":1156,"publishedAt":1157},"489","agentic-ai-explained-when-an-ai-system-can-plan-use-tools-and-act","Agentna AI objašnjena: Kada AI sistem može da planira, koristi alate i deluje","Agentna AI koristi modele unutar višekoračnih izvršnih petlji gde mogu da biraju alate, posmatraju rezultate, ažuriraju stanje i prilagode svoju sledeću akciju unutar eksplicitnih granica izvršavanja i dozvola.","\u002Fuploads\u002F2026\u002F10\u002Fagentic-ai-explained-when-an-ai-system-can-plan-use-tools-and-act-1791481499084-wnji2a.webp","2026-10-08T11:43:00.000Z",{"id":1159,"slug":1160,"title":1161,"excerpt":1162,"featuredImage":1163,"publishedAt":1164},"479","where-does-an-llm-get-its-data-rag-data-sources-in-python","Odakle LLM dobija svoje podatke? RAG izvori podataka u Python-u","LLM ne zna magično vaše fajlove, baze podataka ili API-je. Ovaj praktični nastavak RAG serije pokazuje, uz jednostavan Python, kako eksterni podaci postaju dokazi koji se mogu pronaći: od tekstualnih fajlova i SQL-a do pretrage punog teksta, embeddinga, sastavljanja konteksta i konačnog LLM poziva.","\u002Fuploads\u002F2026\u002F09\u002Fwhere-does-an-llm-get-its-data-rag-data-sources-in-python-1790517200521-nfsi5i.webp","2026-09-27T05:51:00.000Z",{"id":1166,"slug":1167,"title":1168,"excerpt":1169,"featuredImage":1170,"publishedAt":1171},"459","ollama-is-not-the-product-building-production-ready-open-llm-applications","Ollama nije proizvod: Izgradnja aplikacija spremnih za produkciju sa otvorenim LLM-ovima","Pokretanje lokalnog modela pomoću Ollama-e je jednostavno. Izgradnja Open-LLM aplikacije spremne za produkciju je teža: zahteva RAG, kontrolu pristupa, apstrakciju provajdera, evaluaciju, logovanje, disciplinu puštanja u rad i kontrolisani aplikativni sloj oko modela.","\u002Fuploads\u002F2026\u002F06\u002Follama-is-not-the-product-building-production-ready-open-llm-applications-1782679361640-h0usqf.webp","2026-06-28T16:39:00.000Z",{"id":1173,"slug":1174,"title":1175,"excerpt":1176,"featuredImage":1177,"publishedAt":1178},"480","when-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger","Kada bi AI trebalo da prestane da veruje sopstvenom znanju? — Okidač za pretragu","AI model ne zahteva pretragu za svako pitanje. Važan problem je znati kada njegovo interno znanje više nije dovoljno. Okidač za pretragu je praktična granica odlučivanja koja određuje kada AI sistem treba da prestane da se oslanja isključivo na znanje modela i pribavi spoljne dokaze pre odgovaranja.","\u002Fuploads\u002F2026\u002F09\u002Fwhen-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger-1790574991244-f4rpyg.webp","2026-09-28T01:49:00.000Z",{"id":1180,"slug":1181,"title":1182,"excerpt":1183,"featuredImage":1184,"publishedAt":1185},"485","enterprise-ai-architecture-what-changes-when-ai-enters-a-company","Enterprise AI arhitektura: Šta se menja kada AI uđe u kompaniju","Enterprise AI arhitektura objašnjava kako AI menja korporativne sisteme kroz autoritet podataka, identitet, dozvole, provajdere, rizik, upravljanje, evaluaciju, usklađenost i operacije.","\u002Fuploads\u002F2026\u002F10\u002Fenterprise-ai-architecture-what-changes-when-ai-enters-a-company-1791478161363-czrwaq.webp","2026-10-08T10:48:00.000Z",{"id":1187,"slug":1188,"title":1189,"excerpt":1190,"featuredImage":1191,"publishedAt":1192},"477","computer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system","Agenti za korišćenje računara: Zašto uspešan demo i dalje može biti nepouzdan sistem","Agenti za korišćenje računara sada mogu da završe impresivne radne tokove u pregledaču i na radnoj površini, ali jedno uspešno izvršavanje dokazuje sposobnost—ne pouzdanost. Ovaj članak pokazuje kako testirati ponovljivost, robusnost u odnosu na okruženje, kontrolu dugog horizonta, svest o stanju, verifikaciju ishoda i bezbedno upravljanje ciljevima.","\u002Fuploads\u002F2026\u002F09\u002Fcomputer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system-1790352854690-75qnrg.webp","2026-09-25T12:13:00.000Z",{"id":1194,"slug":1195,"title":1196,"excerpt":1197,"featuredImage":1198,"publishedAt":1199},"364","tipps-fuer-die-verbesserung-der-seo-suchmaschinenoptimierung","Ovladavanje SEO radnim tokom: Ključne strategije optimizacije za organski rast","Strukturiran SEO tok posla je ključan za održiv organski rast. Naučite deset osnovnih strategija, od istraživanja ključnih reči i tehničke optimizacije do kvaliteta sadržaja i analize performansi.","\u002Fuploads\u002F2026\u002F03\u002Ftipps-fuer-die-verbesserung-der-seo-suchmaschinenoptimierung-1774866098131-hwkzrg.webp","2024-01-26T06:35:00.000Z","fallback",[],[]]