[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"portal-settings:stajic:sr":3,"public-menus:all":38,"post:generative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing:sr":205,"related:post:generative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing:sr:1":2243},{"statusCode":4,"data":5,"message":37},200,{"tenantId":6,"lang":7,"defaultLang":8,"siteUrl":9,"contactEmail":10,"brandName":11,"logoUrl":12,"siteName":11,"siteDescription":13,"ogImage":10,"robotsIndex":14,"socialLinks":10,"reservedSlugs":10,"seoPolicy":15},"stajic","sr","de","https:\u002F\u002Fstajic.de",null,"Stajic Platform","\u002FLogo_Planet.svg","Stajic Portal",true,{"branding":16,"relatedContent":17,"crossDomainLinks":18},{"logoUrl":12},{"enabled":14},[19,22,25,28,31,34],{"url":20,"label":21,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Ffigure.rocks","figure.rocks",{"url":23,"label":24,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Floving.rocks","loving.rocks",{"url":26,"label":27,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.com","bazify.com",{"url":29,"label":30,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.de","bazify.de",{"url":32,"label":33,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.at","bazify.at",{"url":35,"label":36,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.ba","bazify.ba","Portal settings resolved",[39,45],{"id":40,"name":41,"location":42,"isActive":14,"isDefault":43,"items":44},1,"main-navigation","header",false,[],{"id":46,"name":47,"location":48,"isActive":14,"isDefault":14,"items":49},4,"main-menu","sidebar",[50,66,79,93,103,118,133],{"id":51,"title":52,"url":60,"target":61,"icon":62,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":64,"portfolioId":10,"children":65},"item-18",{"de":53,"en":54,"es":55,"fr":56,"it":54,"ru":57,"sr":58,"zh":59},"Startseite","Home","Inicio","Accueil","Главная","Почетна","首页","\u002Ffull-stack-web-developer-munich-performance-seo-and-maintainable-builds","_self","i-lucide-home","page",111,[],{"id":67,"title":68,"url":75,"target":61,"icon":76,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":77,"portfolioId":10,"children":78},"item-22",{"de":69,"en":69,"es":70,"fr":69,"it":71,"ru":72,"sr":73,"zh":74},"Vision","Visión","Visione","Видение","Визија","想象","\u002Fueber-uns-webdesign-muenchen-webaplikation","i-lucide-eye",113,[],{"id":80,"title":81,"url":89,"target":61,"icon":90,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":91,"portfolioId":10,"children":92},"item-19",{"de":82,"en":83,"es":84,"fr":83,"it":85,"ru":86,"sr":87,"zh":88},"Leistungen","Services","Servicios","Servizi","Услуги","Услуге","服务","\u002Fservices-dienstleistungen-muenchen","i-lucide-wrench",116,[],{"id":94,"title":95,"url":99,"target":61,"icon":100,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":101,"portfolioId":10,"children":102},"item-23",{"de":96,"en":96,"es":96,"fr":96,"it":96,"ru":97,"sr":97,"zh":98},"Blog","Блог","博客","\u002Fblog","i-lucide-book-open",112,[],{"id":104,"title":105,"url":114,"target":61,"icon":115,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":116,"portfolioId":10,"children":117},"item-32",{"de":106,"en":107,"es":108,"fr":109,"it":110,"ru":111,"sr":112,"zh":113},"Neue Technologien","New Technologies","Nuevas tecnologías","Nouvelles technologies","Nuove tecnologie","Новые технологии","Нове технологије","新技术！","\u002Fneue-webtechnologien","i-lucide-sparkles",122,[],{"id":119,"title":120,"url":129,"target":61,"icon":130,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":131,"portfolioId":10,"children":132},"item-20",{"de":121,"en":122,"es":123,"fr":124,"it":125,"ru":126,"sr":127,"zh":128},"Kontakt","Contact us!","Contacto","Contact","Contatto","Контакт","Контактирајте нас","联系我们！","\u002Fcontact","i-lucide-mail",115,[],{"id":134,"title":135,"url":144,"target":61,"icon":145,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":147},"item-21",{"de":136,"en":137,"es":138,"fr":139,"it":140,"ru":141,"sr":142,"zh":143},"Unsere Arbeit","Our Work","Nuestro trabajo","Nos réalisations","I nostri lavori","Наши работы","Наши радови","文件夹","\u002Fportfolio","i-lucide-briefcase",114,[148,161,175,181,193],{"id":149,"title":150,"url":144,"target":61,"icon":159,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":160},"item-24",{"de":151,"en":152,"es":153,"fr":154,"it":155,"ru":156,"sr":157,"zh":158},"Alle Projekte","All Projects","Todos los proyectos","Tous les projets","Tutti i progetti","Все проекты","Сви пројекти","所有项目","i-lucide-grid-3x3",[],{"id":162,"title":163,"url":171,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":174},"item-29",{"de":164,"en":165,"es":166,"fr":167,"it":168,"ru":169,"sr":170,"zh":143},"Local Roots, Global Reach","Local Roots - Global Reach","Empresa local ","Entreprise locale","Azienda locale","Местная компания","Локално предузеће глобално тржиште","\u002Fportfolio\u002Flocal-roots-global-reach-communication-media-systems-for-modern-business","i-lucide-folder","custom",[],{"id":176,"title":177,"url":179,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":180},"item-28",{"de":178,"en":178,"es":178,"fr":178,"it":178,"ru":178,"sr":178,"zh":178},"Solr Suggester","\u002Fportfolio\u002Fsolr-fuzzy-suggester-und-solr-infix-suggester-abfrage-ueber-ajax-und-filterung",[],{"id":182,"title":183,"url":191,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":192},"item-27",{"de":184,"en":185,"es":186,"fr":187,"it":188,"ru":189,"sr":190,"zh":185},"Firmenwebseite SEO","Company Website SEO","Sitio web corporativo SEO","Site web d’entreprise SEO","Sito web aziendale SEO","Корпоративный сайт SEO","Пословна веб-страница SEO","\u002Fportfolio\u002Fseo-sem-branding-mobile-webseite-muenchen",[],{"id":194,"title":195,"url":203,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":204},"item-31",{"de":196,"en":197,"es":198,"fr":199,"it":200,"ru":201,"sr":202,"zh":197},"Digitalisierungsportal","Digitalization Portal","Portal de digitalización","Portail de numérisation","Portale di digitalizzazione","Портал цифровизации","Портал за дигитализацију","\u002Fportfolio\u002Fdigitalisierungsportal-archiv-museum-bibliothek-ead-lido-mets-mods",[],{"statusCode":4,"data":206,"message":2242},{"id":207,"title":208,"slug":209,"content":210,"contentJson":211,"excerpt":1069,"featuredImage":1070,"featuredImageAlt":1071,"featuredImageCaption":10,"featuredImageTitle":10,"featuredImageCopyright":10,"featuredImageAuthor":10,"featuredImageSourceUrl":10,"featuredImageLicense":10,"featuredImageIsAiGenerated":43,"status":1072,"publishedAt":1073,"createdAt":1074,"updatedAt":1075,"seoLocalePaths":1076,"categories":1085,"author":1106,"translations":1111},"481","Generativna veštačka inteligencija objašnjena: modeli, pretraga, alati i aplikacije nisu ista stvar","generative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing","\u003Cp>Generativna veštačka inteligencija nije jedna komponenta. Produkcijski generativni AI sistem obično kombinuje generativni model sa aplikativnim kodom koji obezbeđuje instrukcije i kontekst, pronalazi eksterno znanje kada je potrebno, izlaže alate za čitanje ili menjanje eksternih sistema, upravlja stanjem izvršavanja i dozvolama i pretvara rezultat u upotrebljiv proizvod. Tretiranje modela, pronalaženja, alata, konteksta, izvršnog okruženja i aplikacije kao iste stvari skriva granice koje određuju svežinu, bezbednost, pouzdanost, cenu i kontrolu.\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--info my-6 rounded-xl border p-5 border-blue-300 bg-blue-50 dark:border-blue-900 dark:bg-blue-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">Direktan odgovor\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">\u003Cstrong>Model generiše; pronalaženje pronalazi eksterne dokaze; alati pristupaju podacima ili izvršavaju radnje; kontekst je ono što model može da vidi za trenutno zaključivanje; izvršno okruženje koordinira izvršavanje; aplikacija poseduje pravila proizvoda, stanje, dozvole, trajnost i korisničko iskustvo.\u003C\u002Fstrong> Ove slojeve dobavljač može pakovati zajedno, ali njihove odgovornosti ostaju različite.\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Caside class=\"editorjs-callout editorjs-callout--note my-6 rounded-xl border p-5 border-gray-300 bg-gray-50 dark:border-gray-700 dark:bg-gray-900\u002F40\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">Napomena o terminologiji i verziji\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">Ovaj članak definiše trajne arhitektonske odgovornosti, a ne jedan dobavljački stek. Primeri trenutnih implementacija ponovo su provereni \u003Cstrong>8. oktobra 2026.\u003C\u002Fstrong> API-ji dobavljača i nazivi proizvoda mogu se menjati; granice odgovornosti su stabilnije od bilo kog pojedinačnog SDK-a ili endpointa.\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Cnav class=\"editorjs-toc\" data-editorjs-toc=\"true\" aria-label=\"Sadržaj\">\u003Cstrong class=\"editorjs-toc__title\">Sadržaj\u003C\u002Fstrong>\u003Col class=\"editorjs-toc__list editorjs-toc__list--depth-0\">\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-5\" class=\"editorjs-toc__link\">Šta zapravo znači „generativna veštačka inteligencija“?\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-9\" class=\"editorjs-toc__link\">Najjednostavniji korisni model generativnog AI sistema\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-13\" class=\"editorjs-toc__link\">Šest granica koje su važne\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-15\" class=\"editorjs-toc__link\">1. Model: generisanje je njegova osnovna odgovornost\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-19\" class=\"editorjs-toc__link\">2. Pretraga: pronalaženje spoljnih dokaza je zasebna operacija\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-24\" class=\"editorjs-toc__link\">3. Alati: pristup i radnja nisu znanje modela\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-29\" class=\"editorjs-toc__link\">4. Kontekst: šta model može videti upravo sada\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-33\" class=\"editorjs-toc__link\">5. Okruženje za izvršavanje i orkestracija: koordinisanje petlje\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-37\" class=\"editorjs-toc__link\">6. Aplikacija: gde AI postaje proizvod\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-41\" class=\"editorjs-toc__link\">Kako delovi rade zajedno u stvarnom zahtevu\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-45\" class=\"editorjs-toc__link\">Različiti AI proizvodi koriste različite kombinacije\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-48\" class=\"editorjs-toc__link\">Dokaz implementacije: Aaasaasa AI Client\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-55\" class=\"editorjs-toc__link\">Česte greške u kategorizaciji\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-57\" class=\"editorjs-toc__link\">Načini otkaza kada se granice uruše\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-60\" class=\"editorjs-toc__link\">Šta je stabilno, a šta zavisi od verzije?\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-64\" class=\"editorjs-toc__link\">Test granica AI komponenti\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-67\" class=\"editorjs-toc__link\">Šta generativna AI nije\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-71\" class=\"editorjs-toc__link\">Gde dalje u grafu znanja\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-75\" class=\"editorjs-toc__link\">Ograničenja\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-79\" class=\"editorjs-toc__link\">Šta bi promenilo ovaj odgovor?\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-82\" class=\"editorjs-toc__link\">Zaključak\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-86\" class=\"editorjs-toc__link\">Često postavljana pitanja\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-88\" class=\"editorjs-toc__link\">Pojmovnik\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-90\" class=\"editorjs-toc__link\">Primarni izvori i dokazi o implementaciji\u003C\u002Fa>\u003C\u002Fli>\u003C\u002Fol>\u003C\u002Fnav>\n\u003Ch2 id=\"section-5\">Šta zapravo znači „generativna veštačka inteligencija“?\u003C\u002Fh2>\n\u003Cp>Na nivou modela, generativna veštačka inteligencija odnosi se na AI modele koji generišu izvedeni sintetički sadržaj kao što su tekst, slike, zvuk, video, kod ili drugi digitalni izlaz. NIST AI 600-1 koristi ovo značenje usmereno na model i odvojeno razmatra rizike na nivou modela, sistema, aplikacije i slučaja upotrebe.\u003C\u002Fp>\n\u003Cp>Ta razlika je važna jer AI model nije isto što i kompletan AI sistem. NIST-ov trenutni rečnik definiše AI model kao komponentu koja proizvodi izlaze iz ulaza koristeći računske, statističke ili tehnike mašinskog učenja, dok AI sistem može uključivati softver, hardver, aplikacije, alate ili uslužne programe koji rade koristeći AI.\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--success my-6 rounded-xl border p-5 border-emerald-300 bg-emerald-50 dark:border-emerald-900 dark:bg-emerald-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">Korisna granica\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">\u003Cstrong>Generativni model ≠ generativna AI aplikacija.\u003C\u002Fstrong>\u003Cbr>Model je jedna računska komponenta. Upotrebljiv AI proizvod je sistem izgrađen oko te komponente.\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-9\">Najjednostavniji korisni model generativnog AI sistema\u003C\u002Fh2>\n\u003Cp>Za prvi mentalni model, zamislite korporativnog asistenta koji odgovara: „Može li ovaj kupac danas dobiti povraćaj?“ Koristan odgovor može zahtevati nekoliko različitih odgovornosti. Jezički model može interpretirati pitanje i napisati objašnjenje, ali trenutno stanje porudžbine može doći iz alata baze podataka, politika povraćaja može doći iz pronalaženja dokumenata, dozvole može sprovoditi aplikacija, a konačna radnja može zahtevati kontrolisani API poziv.\u003C\u002Fp>\n\u003Csection class=\"editorjs-process my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">Jedna uobičajena putanja izvršavanja\u003C\u002Fh3>\u003Cdiv class=\"grid grid-cols-1 md:grid-cols-2 xl:grid-cols-3 gap-4\">\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">1\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">1. Korisnički zahtev\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Aplikacija prima pitanje ili zadatak na prirodnom jeziku.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">2\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">2. Politika i stanje aplikacije\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Identitet, zakupac, dozvole, trenutno stanje toka rada i pravila proizvoda definišu šta zahtev sme da uradi.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">3\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">3. Pronalaženje ili direktan pristup podacima\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Sistem pribavlja eksterne dokaze ili trenutne činjenice kada je znanje modela nedovoljno.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">4\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">4. Izgradnja konteksta\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Instrukcije, korisnički unos, izabrani dokazi, relevantno stanje i definicije alata sastavljaju se za model.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">5\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">5. Zaključivanje modela\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Generativni model interpretira dostavljeni kontekst i proizvodi tekst, strukturirani izlaz ili zahtev za alat.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">6\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">6. Izvršavanje alata kada je potrebno\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Izvršno okruženje ili aplikacija validira i izvršava odobrene pozive alata izvan modela.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">7\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">7. Opažanje i nastavak\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Rezultati alata mogu se vratiti modelu kao novi kontekst za sledeći korak zaključivanja.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">8\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">8. Validacija i izlaz proizvoda\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Aplikacija validira rezultat, beleži potrebno stanje ili podatke za reviziju i prikazuje ili izvršava konačni ishod.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Cp>Stvarni sistemi ne prate uvek tačno ovaj redosled. Pronalaženje se može desiti pre prvog poziva modela, alati se mogu birati tokom agentske petlje, deterministička aplikativna logika može potpuno zaobići model, a validacija se može desiti u nekoliko faza. Poenta je razdvojiti odgovornosti, a ne nametnuti jedan univerzalni tok rada.\u003C\u002Fp>\n\u003Ch2 id=\"section-13\">Šest granica koje su važne\u003C\u002Fh2>\n\u003Csection class=\"editorjs-comparison my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">Šest odgovornosti unutar jednog AI proizvoda\u003C\u002Fh3>\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left dark:border-gray-700 dark:bg-gray-900\">\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">Primarni zadatak\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">Tipični ulazi\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">Nije isto kao\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Model\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Pronalaženje\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Alati\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Kontekst\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Izvršno okruženje \u002F orkestrator\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Aplikacija\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-15\">1. Model: generisanje je njegova osnovna odgovornost\u003C\u002Fh2>\n\u003Cp>Generativni model preslikava dostavljene ulaze u generisane izlaze. Za jezički model to može uključivati tekst na prirodnom jeziku, strukturirani JSON, kod, klasifikacije, rezimee, planove ili argumente poziva alata. Multimodalni generativni modeli mogu raditi sa dodatnim tipovima ulaza i izlaza.\u003C\u002Fp>\n\u003Cp>Model može sadržati značajno naučeno znanje u svojim parametrima, ali parametrizovano znanje nije živa baza podataka. Model ne zna automatski za dokument kreiran pre pet minuta, trenutni nivo zaliha, privatni zapis kupca ili stanje aplikacije osim ako se ta informacija ne dostavi kroz trenutnu ulaznu putanju.\u003C\u002Fp>\n\u003Cp>Zato promena modela ne rešava automatski zastarelo znanje, nedostajuće dozvole, pokvareno pronalaženje, pogrešno vlasništvo nad stanjem ili nesigurno izvršavanje alata. Ti kvarovi često pripadaju drugim slojevima.\u003C\u002Fp>\n\u003Ch2 id=\"section-19\">2. Pretraga: pronalaženje spoljnih dokaza je zasebna operacija\u003C\u002Fh2>\n\u003Cp>Pretraga bira informacije iz spoljnog izvora pre ili tokom generisanja. Rad o generisanju uz pomoć pretrage iz 2020. godine autora Lewis i saradnika učinio je tu razdvojenost eksplicitnom kombinovanjem parametarskog generativnog modela sa preuzetom neparametarskom memorijom. Savremeni produkcioni sistemi koriste mnoge varijante pretrage, ali arhitektonska ideja ostaje: korisni dokazi mogu se dohvatiti u vreme izvođenja zaključivanja umesto da se oslanjamo samo na ono što je model naučio tokom obuke.\u003C\u002Fp>\n\u003Cp>Pretraga može koristiti leksičku pretragu, ugrađivanja, vektorsku pretragu, hibridnu pretragu, SQL, grafove znanja, filtere metapodataka, API-je ili druge mehanizme selekcije. Vektorska baza podataka je stoga jedna moguća komponenta pretrage, a ne definicija RAG-a.\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--warning my-6 rounded-xl border p-5 border-amber-300 bg-amber-50 dark:border-amber-900 dark:bg-amber-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">Relevantnost nije autoritet\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">Preuzeti odlomak može biti veoma relevantan, a ipak zastareo, neovlašćen, iz pogrešne verzije ili nedovoljan da podrži tvrdnju. Kvalitet pretrage i kvalitet dokaza moraju se procenjivati odvojeno.\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Caside class=\"editorjs-referral my-6\">\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fsr\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works\" class=\"flex flex-col sm:flex-row gap-4 rounded-xl border border-gray-200 dark:border-gray-700 p-4 transition hover:border-primary-500\">\u003Cdiv class=\"min-w-0 flex-1\">\u003Cstrong class=\"block text-lg text-gray-900 dark:text-gray-100\">Šta je RAG? Najjednostavnije objašnjenje kako funkcioniše\u003C\u002Fstrong>\u003Cp class=\"mt-2 text-sm text-gray-600 dark:text-gray-300\">Kanonsko objašnjenje generisanja uz pomoć pretrage jednostavnim jezikom, uključujući razdvajanje između LLM-a, znanja, stanja, memorije i alata.\u003C\u002Fp>\u003Cspan class=\"mt-3 inline-flex text-sm font-medium text-primary-600 dark:text-primary-400\">Pročitajte osnove RAG-a →\u003C\u002Fspan>\u003C\u002Fdiv>\u003C\u002Fa>\u003C\u002Faside>\n\u003Ch2 id=\"section-24\">3. Alati: pristup i radnja nisu znanje modela\u003C\u002Fh2>\n\u003Cp>Alat je interfejs preko kojeg AI okruženje za izvršavanje može zatražiti funkcionalnost izvan modela. Alat može upitovati bazu podataka, pretraživati veb, čitati datoteku, izračunati vrednost, pozvati interni servis, kreirati tiket, poslati poruku, izmeniti zapis ili pokrenuti drugu kontrolisanu operaciju.\u003C\u002Fp>\n\u003Cp>OpenAI-jeva trenutna dokumentacija o pozivanju funkcija eksplicitno ističe ovu granicu: pozivanje funkcija omogućava modelima da se povežu sa spoljnim sistemima i pristupe podacima ili radnjama koje pruža aplikacija. Model može predložiti ili izabrati poziv, ali spoljni sistem obavlja stvarnu operaciju.\u003C\u002Fp>\n\u003Cp>Upotreba alata stoga stvara dva odvojena pitanja: Može li model zatražiti ovu sposobnost? i Hoće li aplikacija odobriti i izvršiti je? Produkcioni sistem ne bi trebalo da meša nameru modela sa dozvolom da izazove sporedni efekat.\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--success my-6 rounded-xl border p-5 border-emerald-300 bg-emerald-50 dark:border-emerald-900 dark:bg-emerald-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">Namere modela nisu ovlašćenje za izvršenje\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">Model može uputiti važeći zahtev za alat i ipak biti odbijen. Autorizacija, validacija argumenata, ograničenja brzine, pravila transakcija, zahtevi za reviziju i vraćanje unazad pripadaju izvan modela.\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-29\">4. Kontekst: šta model može videti upravo sada\u003C\u002Fh2>\n\u003Cp>Kontekst su informacije dostupne modelu za određeni korak zaključivanja. Anthropic-ove smernice za inženjering konteksta opisuju kontekst kao skup tokena uključenih pri uzorkovanju iz LLM-a. U praksi, taj skup može sadržati sistemska uputstva, korisničke poruke, istoriju razgovora, preuzete dokaze, definicije alata, rezultate alata, sažetke memorije i izabrano stanje aplikacije.\u003C\u002Fp>\n\u003Cp>Kontekst stoga nije ni kompletna baza znanja ni dugoročna memorija. Kompanija može čuvati deset miliona dokumenata dok samo nekoliko odlomaka ulazi u jedan poziv modela. Okruženje za izvršavanje može čuvati godinu dana istorije razgovora dok izlaže samo delove potrebne za trenutni zadatak.\u003C\u002Fp>\n\u003Cp>Kontekstni prozor takođe stvara inženjersko ograničenje. Dodavanje više teksta ne garantuje bolji odgovor; irelevantne, zastarele, kontradiktorne ili informacije niskog autoriteta mogu razblažiti dokaze koji su zaista važni.\u003C\u002Fp>\n\u003Ch2 id=\"section-33\">5. Okruženje za izvršavanje i orkestracija: koordinisanje petlje\u003C\u002Fh2>\n\u003Cp>Sloj okruženja za izvršavanje ili orkestracije koordinira kako model učestvuje u zadatku. U zavisnosti od arhitekture, može upravljati sesijama, zahtevima modela, otkrivanjem alata, petljama pozivanja alata, ponovnim pokušajima, predajama, događajima strimovanja, istekima vremena, kontrolnim tačkama, sažimanjem ili okruženjima za izvršavanje.\u003C\u002Fp>\n\u003Cp>Neka okruženja za izvršavanje su tanki aplikacijski kod oko API-ja modela. Druga su potpuni agentni okviri. Upravljano okruženje dobavljača može posedovati deo petlje dok aplikacija i dalje poseduje domen istine, autorizaciju, poslovne sporedne efekte i životni ciklus proizvoda.\u003C\u002Fp>\n\u003Cp>Ova granica je važna jer su to gde se izvršava okruženje i gde se izvršava zaključivanje odvojene odluke. Lokalno pokrenut klijent ili agentni proces i dalje može pozivati udaljeni model, dok udaljena aplikacija može pozivati model hostovan na infrastrukturi pod kontrolom organizacije.\u003C\u002Fp>\n\u003Ch2 id=\"section-37\">6. Aplikacija: gde AI postaje proizvod\u003C\u002Fh2>\n\u003Cp>Aplikacija je granica proizvoda oko AI komponenti. Ona poseduje korisničko iskustvo, domenski model, trenutno stanje, identitet, opseg zakupca, dozvole, perzistenciju, integracije servisa, validaciju, observabilnost, logiku naplate ili kvota gde je relevantno, i pravila koja određuju šta AI sme da vidi ili radi.\u003C\u002Fp>\n\u003Cp>Ovo je sloj koji pretvara „model može da proizvede koristan izlaz“ u „sistem može da isporuči pouzdanu sposobnost.“ Isti model može da učestvuje u privatnom istraživačkom asistentu, korisničkom toku podrške, kodnom agentu ili aplikaciji za trgovinu jer okolna aplikacija menja podatke, alate, politike, stanje i ugovor o izvršavanju.\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--info my-6 rounded-xl border p-5 border-blue-300 bg-blue-50 dark:border-blue-900 dark:bg-blue-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">Model je zamenljiv; granica proizvoda nije\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">Zamena provajdera i modela može biti arhitektonski cilj. Autoritativno stanje aplikacije, dozvole, domenska pravila, revizorski trag i korisnički ugovor ne mogu se jednostavno delegirati modelu koji je trenutno izabran.\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-41\">Kako delovi rade zajedno u stvarnom zahtevu\u003C\u002Fh2>\n\u003Cp>Razmotrimo asistenta za podršku upitanog: „Refundiraj porudžbinu 4711 ako je još uvek podobna, i objasni zašto.“ Zahtev kombinuje znanje, trenutno stanje, autorizaciju, rezonovanje i sporedni efekat.\u003C\u002Fp>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Potreba\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Ispravan sloj\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Zašto\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Politika refundacije\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Preuzimanje\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Sistem mora da pronađe trenutno primenljivu politiku i sačuva njeno poreklo.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Status porudžbine 4711\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Direktan pristup podacima\u002Falatu\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Trenutni zapis porudžbine je promenljivo autoritativno stanje, ne nešto što se pogađa iz znanja modela.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Korisnikovo ovlašćenje za refundaciju\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Aplikacija \u002F autorizacija\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Dozvole se moraju sprovesti nezavisno od onoga što model traži.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Interpretacija politike prema činjenicama porudžbine\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Model + kontekst\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Model može da rezonuje nad dokazima politike i trenutnim stanjem porudžbine koji su mu dostavljeni.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Izvrši refundaciju\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Alat + pravila transakcije aplikacije\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Kontrolisana eksterna operacija menja stvarno stanje.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Objasni ishod\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Model\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Model može da generiše objašnjenje za korisnika iz validiranih rezultata.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Revizija onoga što se dogodilo\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Aplikacija \u002F vreme izvršavanja\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Sistem beleži dokaze, pozive, odluke, sporedne efekte i greške po potrebi.\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Cp>Ako asistent ima samo jezički model, može da diskutuje o refundacijama ali ne može bezbedno da zna da li je porudžbina 4711 trenutno podobna ili da izvrši transakciju. Ako ima samo preuzimanje, može da pronađe politiku ali mu i dalje nedostaje stanje porudžbine uživo. Ako ima alate bez autorizacije aplikacije, može postati sposoban ali nesiguran. Pouzdanost dolazi iz komponovanja slojeva sa eksplicitnim vlasništvom.\u003C\u002Fp>\n\u003Ch2 id=\"section-45\">Različiti AI proizvodi koriste različite kombinacije\u003C\u002Fh2>\n\u003Csection class=\"editorjs-comparison my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">Prisustvo modela ne definiše celu arhitekturu\u003C\u002Fh3>\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left dark:border-gray-700 dark:bg-gray-900\">\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">Preuzimanje\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">Alati\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">Autoritativno stanje\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">Tipična sposobnost\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Asistent samo sa modelom\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Asistent zasnovan na preuzimanju\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Asistent koji koristi alate\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Agentna aplikacija\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Cp>Ovo su arhitektonski obrasci, a ne rangiranje zrelosti. Funkcija samo sa modelom može biti ispravan dizajn kada zadatak ne zahteva spoljne činjenice ili radnje. Dodavanje preuzimanja, alata, memorije ili agentne petlje opravdano je samo kada zadatak zahteva te sposobnosti.\u003C\u002Fp>\n\u003Ch2 id=\"section-48\">Dokaz implementacije: Aaasaasa AI Client\u003C\u002Fh2>\n\u003Caside class=\"editorjs-callout editorjs-callout--note my-6 rounded-xl border p-5 border-gray-300 bg-gray-50 dark:border-gray-700 dark:bg-gray-900\u002F40\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">Primarni dokaz implementacije\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">Sledeći odeljak opisuje implementaciju koju sam izgradio i pregledao u odnosu na Aaasaasa AI Client kodnu bazu i arhitektonsku dokumentaciju zaključno sa \u003Cstrong>26. julom 2026.\u003C\u002Fstrong> To je dokaz korisnosti ovih granica, a ne tvrdnja da je jedna implementacija univerzalni standard ili komercijalno raspoređen enterprise proizvod.\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Cp>Aaasaasa AI Client je lokalno-prvenstveni desktop AI radni prostor izgrađen sa Nuxt 4, Electron i TypeScript. Njegov AI Hub namerno razdvaja agenta\u002Fklijenta, provajdera, model, lokaciju izvršavanja, dozvole i web klijenta umesto da ih tretira kao jedno „AI“ podešavanje.\u003C\u002Fp>\n\u003Cp>To razdvajanje stvara konkretno ponašanje. Direct Chat može da razgovara sa modelima bez alata za fajl sistem ili shell. Codex agent može da koristi izabrani radni prostor i profil dozvola. Ollama može da obezbedi direktno lokalno zaključivanje, dok LM Studio i konfigurabilni OpenAI-kompatibilni endpointi predstavljaju druge putanje provajdera. Lokalno pokrenut Codex proces i dalje može da koristi cloud model, tako da UI i arhitektura ne izjednačavaju lokalno izvršavanje sa lokalnim zaključivanjem.\u003C\u002Fp>\n\u003Cp>Implementacija takođe sadrži Qdrant\u002Fvektorsku podršku, mogućnosti ekstrakcije dokumenata i autentifikovani direktorijum MCP broker. Te komponente ilustruju još jednu granicu: infrastruktura za preuzimanje i pristup alatima mogu živeti u istom proizvodu a da ne postanu svojstva samog modela.\u003C\u002Fp>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">A01 koncept\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Dokaz implementacije Aaasaasa AI Client\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Model\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Identifikator modela specifičan za provajdera se bira odvojeno od provajdera i vremena izvršavanja.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Provajder\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Ollama, LM Studio, OpenAI-kompatibilni servisi i druge putanje provajdera su predstavljeni odvojeno.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Vreme izvršavanja\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Lokalna ili udaljena lokacija agenta\u002Fvremena izvršavanja se prati nezavisno od modela.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Alati \u002F pristup\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Direct Chat nema alate za fajl sistem ili shell; kontrolisani pristup direktorijumu se posreduje odvojeno.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Dozvole\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Profili dozvola radnog prostora su politika aplikacije\u002Fsesije, a ne sposobnost modela.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Infrastruktura za preuzimanje\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Vektorska podrška i ekstrakcija dokumenata postoje kao sposobnosti podataka\u002Fpreuzimanja, a ne kao funkcije modela.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Aplikacija\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Electron\u002FNuxt proizvod koordinira UI, akreditive, provajdere, otkrivanje vremena izvršavanja, dozvole, alate i interakciju sa modelom.\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Caside class=\"editorjs-callout editorjs-callout--success my-6 rounded-xl border p-5 border-emerald-300 bg-emerald-50 dark:border-emerald-900 dark:bg-emerald-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">Lekcija iz implementacije\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">Arhitektura je postala lakša za razumevanje kada su \u003Cstrong>model, provajder, vreme izvršavanja, dozvole, alati, podaci i klijent\u003C\u002Fstrong> prestali da budu predstavljeni kao jedan izbor konfiguracije. Razlika je operativna: određuje šta može da se pokrene lokalno, šta može da pristupi fajlovima, šta može da pozove plaćeno cloud zaključivanje i koji sloj poseduje autorizaciju.\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-55\">Česte greške u kategorizaciji\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Greška u kategorizaciji\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Šta se zapravo dešava\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">„AI zna naše dokumente.“\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Aplikacija ili sloj za pronalaženje čini sadržaj odabranih dokumenata dostupnim modelu.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">„RAG je naša vektorska baza podataka.“\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Vektorska baza podataka može biti jedan indeks ili skladište koje koristi pipeline za pronalaženje; RAG je obrazac pronalaženja i generisanja.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">„Model je pozvao naš CRM.“\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Model je proizveo zahtev za alat; runtime\u002Faplikacija je autorizovala i izvršila eksterni poziv.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">„To je lokalna AI jer desktop agent radi lokalno.“\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Lokacija izvršavanja i lokacija inferencije su odvojene. Lokalni runtime i dalje može pozvati udaljeni model.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">„Model ima dozvolu da menja fajlove.“\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Aplikacija\u002Fruntime dodeljuje sposobnost alata prema politici dozvola; dozvola nije intrinzično svojstvo modela.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">„Više konteksta znači više znanja.“\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Kontekst je konačan unos koji je dostupan za jednu inferenciju. Veći kontekst može sadržati više šuma, konflikata ili zastarelih informacija.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">„Chatbot je AI arhitektura.“\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Chat UI je jedan interfejs. Sistem takođe može uključivati identitet, stanje, pronalaženje, alate, runtime, validaciju, perzistenciju i observabilnost.\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-57\">Načini otkaza kada se granice uruše\u003C\u002Fh2>\n\u003Cp>Greške u granicama nisu samo terminološki problemi. One stvaraju različite produkcione otkaze koji zahtevaju različita rešenja.\u003C\u002Fp>\n\u003Csection class=\"editorjs-comparison my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">Dijagnostikujte sloj koji otkazuje pre nego što zamenite model\u003C\u002Fh3>\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left dark:border-gray-700 dark:bg-gray-900\">\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">Simptom\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">Verovatni problem granice\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">Prva arhitektonska provera\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Zastareo odgovor\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Nedostaje činjenica o kompaniji\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Nesigurna sporedna radnja\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Zbunjen odgovor sa puno priloženog teksta\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Neočekivano korišćenje oblaka\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Agent se zaglavljuje ili ponavlja\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-60\">Šta je stabilno, a šta zavisi od verzije?\u003C\u002Fh2>\n\u003Cp>Arhitektonske razlike u ovom članku su namerno neutralne prema dobavljaču. Trenutni primeri ispod su činjenice o implementaciji koje treba ponovo proveriti kada se API-ji razviju.\u003C\u002Fp>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Oblast\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Stabilna arhitektonska ideja\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Provereni trenutni primer na dan 8. oktobar 2026.\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">AI model naspram sistema\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Model je komponenta unutar šireg sistema\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">NIST-ov trenutni rečnik odvojeno definiše AI model i AI sistem.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">RAG\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Generisanje može biti uslovljeno pronađenim eksternim informacijama\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Formulacija Lewis et al. iz 2020. ostaje temeljna referenca; produkcione metode pronalaženja sada sežu daleko izvan jednog dizajna gustog indeksa.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Hostovano pronalaženje\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Pronalaženje može biti izloženo kao upravljani alat\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">OpenAI File Search je trenutno alat Responses API-ja koji pretražuje baze znanja otpremljenih fajlova koristeći semantičko i pretraživanje po ključnim rečima.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Pozivanje funkcija\u002Falata\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Model može zahtevati eksterne sposobnosti definisane aplikacijom\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">OpenAI trenutno dokumentuje pozivanje funkcija kao interfejs ka eksternim sistemima, podacima i radnjama.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Inženjering konteksta\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Ponašanje modela zavisi od konačne informacije koja se pruža za trenutnu inferenciju\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Anthropic-ove trenutne inženjerske smernice definišu kontekst kao skup tokena uključenih pri uzorkovanju iz LLM-a i fokusiraju se na njegovo pažljivo odabiranje.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">API-ji dobavljača\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">SDK-ovi, imena alata, oblici endpointa i podržane funkcije se menjaju\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Tretirajte dokumentaciju dobavljača kao osetljivu na verziju čak i kada granica odgovornosti ostaje stabilna.\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Cp>Članak koji je izvor istine treba stoga da očuva oba nivoa: stabilne koncepte za arhitekturu i datirane dokaze za trenutne implementacije. Mešanje ova dva čini da članak nepotrebno brzo zastari.\u003C\u002Fp>\n\u003Ch2 id=\"section-64\">Test granica AI komponenti\u003C\u002Fh2>\n\u003Cp>Kada procenjujete AI funkciju, postavite sledeća pitanja po redu. Odgovori otkrivaju koje komponente sistem zapravo ima i koje su odgovornosti još uvek implicitne.\u003C\u002Fp>\n\u003Csection class=\"editorjs-process my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">Sedam pitanja za produkcioni dizajn\u003C\u002Fh3>\u003Cdiv class=\"grid grid-cols-1 md:grid-cols-2 xl:grid-cols-3 gap-4\">\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">1\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">1. Šta generiše izlaz?\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Identifikujte tačan model i modalitete ili strukturisane izlaze koje pruža.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">2\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">2. Koje činjenice su autoritativne izvan modela?\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Identifikujte dokumente, baze podataka, API-je, trenutno stanje i druge izvore istine.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">3\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">3. Kako se biraju relevantne informacije?\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Razdvojite direktno pretraživanje, pretragu, pronalaženje, rangiranje i konstrukciju konteksta.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">4\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">4. Šta može izazvati stvarne sporedne efekte?\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Navedite alate i eksterne radnje, zatim identifikujte ko ih validira i autorizuje.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">5\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">5. Šta dospeva do modela kao kontekst?\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Učinite eksplicitnim instrukcije, dokaze, stanje, istoriju, memoriju i definicije alata.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">6\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">6. Ko poseduje petlju?\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Identifikujte runtime ili harness koji upravlja pozivima, događajima, ponovnim pokušajima, petljama alata i sesijama.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">7\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">7. Šta ostaje odgovornost aplikacije?\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Učinite eksplicitnim identitet, dozvole, stanje domena, validaciju, perzistenciju, observabilnost i UX.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-67\">Šta generativna AI nije\u003C\u002Fh2>\n\u003Cp>Generativna AI nije sinonim za LLM, iako su LLM-ovi glavna klasa generativnih modela. Takođe nije sinonim za RAG, vektorsku bazu podataka, agenta, protokol alata, chat UI ili aplikaciju.\u003C\u002Fp>\n\u003Cp>Ti koncepti mogu biti povezani, ali svaki odgovara na drugo arhitektonsko pitanje. LLM pita kako se proizvodi jezički izlaz. Pronalaženje pita odakle dolaze eksterni dokazi. Alati pitaju kako se izlažu eksterne sposobnosti. Kontekst pita šta model može da vidi. Runtime pita kako se koordinira izvršavanje. Aplikacija pita kako sposobnost postaje kontrolisan proizvod.\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--success my-6 rounded-xl border p-5 border-emerald-300 bg-emerald-50 dark:border-emerald-900 dark:bg-emerald-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">Ako zapamtite samo jedan model\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">\u003Cstrong>Model = generiše.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Pronalaženje = pronalazi dokaze.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Alati = čitaju ili deluju izvan modela.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Kontekst = šta model vidi sada.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Runtime = koordinira izvršavanje.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Aplikacija = poseduje proizvod, stanje, pravila i dozvole.\u003C\u002Fstrong>\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-71\">Gde dalje u grafu znanja\u003C\u002Fh2>\n\u003Cp>Kada su ove granice jasne, dublje teme postaje lakše smestiti. RAG pripada pronalaženju i konstrukciji konteksta. Retrieval Trigger odlučuje kada su eksterni dokazi potrebni. Memorija agenta se tiče onoga što se održava kroz vreme. Pozivanje alata i MCP pripadaju pristupu sposobnostima. Agent harnesses pripadaju runtime orkestraciji. RBAC, izolacija zakupaca i autorizacija domena pripadaju aplikaciji i bezbednosnoj granici platforme.\u003C\u002Fp>\n\u003Caside class=\"editorjs-referral my-6\">\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fsr\u002Fblog\u002Fwhere-does-an-llm-get-its-data-rag-data-sources-in-python\" class=\"flex flex-col sm:flex-row gap-4 rounded-xl border border-gray-200 dark:border-gray-700 p-4 transition hover:border-primary-500\">\u003Cdiv class=\"min-w-0 flex-1\">\u003Cstrong class=\"block text-lg text-gray-900 dark:text-gray-100\">Odakle LLM dobija svoje podatke? RAG izvori podataka u Python-u\u003C\u002Fstrong>\u003Cp class=\"mt-2 text-sm text-gray-600 dark:text-gray-300\">Praktičan nastavak koji pokazuje kako fajlovi, SQL, API-ji, pretraga punog teksta, embedding-ovi i sastavljanje konteksta povezuju eksterne podatke sa LLM-om.\u003C\u002Fp>\u003Cspan class=\"mt-3 inline-flex text-sm font-medium text-primary-600 dark:text-primary-400\">Pogledajte putanju podataka u kodu →\u003C\u002Fspan>\u003C\u002Fdiv>\u003C\u002Fa>\u003C\u002Faside>\n\u003Caside class=\"editorjs-referral my-6\">\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fsr\u002Fblog\u002Fwhen-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger\" class=\"flex flex-col sm:flex-row gap-4 rounded-xl border border-gray-200 dark:border-gray-700 p-4 transition hover:border-primary-500\">\u003Cdiv class=\"min-w-0 flex-1\">\u003Cstrong class=\"block text-lg text-gray-900 dark:text-gray-100\">Kada AI treba da prestane da veruje sopstvenom znanju? — Okidač za pretragu\u003C\u002Fstrong>\u003Cp class=\"mt-2 text-sm text-gray-600 dark:text-gray-300\">Model odlučivanja o tome kada AI sistem treba da prestane da se oslanja samo na znanje modela i pribavi eksterne dokaze.\u003C\u002Fp>\u003Cspan class=\"mt-3 inline-flex text-sm font-medium text-primary-600 dark:text-primary-400\">Pročitajte model odlučivanja o pretrazi →\u003C\u002Fspan>\u003C\u002Fdiv>\u003C\u002Fa>\u003C\u002Faside>\n\u003Ch2 id=\"section-75\">Ograničenja\u003C\u002Fh2>\n\u003Cp>Model sa šest slojeva je mapa odgovornosti, a ne zahtev da svaki proizvod implementira šest odvojenih servisa. Mala aplikacija može implementirati konstrukciju konteksta, pretragu i orkestraciju unutar jednog procesa. Upravljana platforma može objediniti nekoliko odgovornosti iza jednog API-ja. Fizičko postavljanje može biti kombinovano dok semantičko vlasništvo ostaje različito.\u003C\u002Fp>\n\u003Cp>Terminologija se takođe razlikuje među dobavljačima i u istraživanjima. „Agent“, „runtime“, „memorija“, „alat“, „konektor“ i „kontekst“ mogu biti definisani na različite načine. Definicije ovde su izabrane da učine operativno vlasništvo i dijagnostiku kvarova eksplicitnim, a ne da tvrde da svaki okvir koristi identičan rečnik.\u003C\u002Fp>\n\u003Cp>Odeljak Aaasaasa AI Client dokumentuje jedan obrazac implementacije. On pokazuje da su eksplicitne granice praktične, ali ne dokazuje da je isti raspored komponenti optimalan za svaki AI proizvod.\u003C\u002Fp>\n\u003Ch2 id=\"section-79\">Šta bi promenilo ovaj odgovor?\u003C\u002Fh2>\n\u003Cp>Mapa odgovornosti bi zahtevala reviziju ako bi same arhitekture modela počele da poseduju autoritativno eksterno stanje, dozvole, trajne transakcione sporedne efekte i verifikovan pristup izvorima kao intrinzične osobine, a ne kao sposobnosti koje pruža okolni sistem. Trenutne produkcijske arhitekture to ne čine bezbednom opštom pretpostavkom.\u003C\u002Fp>\n\u003Cp>Pojedinačni primeri implementacije će se promeniti mnogo ranije. Hostovani alati za pretragu, agent API-ji, MCP integracije, funkcije za upravljanje kontekstom i mogućnosti dobavljača se brzo razvijaju. Te detalje treba ažurirati bez urušavanja osnovnih razlika između generisanja, dokaza, pristupa sposobnostima, konteksta, izvršavanja i kontrole aplikacije.\u003C\u002Fp>\n\u003Ch2 id=\"section-82\">Zaključak\u003C\u002Fh2>\n\u003Cp>Generativnu AI je lakše dizajnirati kada „AI“ prestane da se tretira kao jedna crna kutija. Model je generativna komponenta, a ne kompletan proizvod. Pretraga pruža eksterne dokaze. Alati izlažu sposobnosti. Kontekst nosi izabrane informacije u trenutnu inferenciju. Runtime koordinira izvršavanje. Aplikacija poseduje autoritativnu granicu proizvoda.\u003C\u002Fp>\n\u003Cp>Ta razdvojenost je korisna za više od objašnjenja. Ona govori inženjerima odakle potiču zastarele činjenice, gde pripada autorizacija, zašto lokalni runtime i dalje može da koristi cloud inferenciju, zašto RAG nije isto što i vektorska baza podataka, zašto pozivi alata zahtevaju validaciju i zašto promena modela ne može da popravi svaki sistemski kvar.\u003C\u002Fp>\n\u003Cp>Trajno arhitektonsko pitanje stoga nije „Koji AI model koristimo?“ Već: Koju odgovornost poseduje svaka komponenta, koji dokazi prelaze svaku granicu i koji sloj sme da menja stvarno stanje?\u003C\u002Fp>\n\u003Ch2 id=\"section-86\">Često postavljana pitanja\u003C\u002Fh2>\n\u003Csection class=\"editorjs-faq my-6 rounded-xl border border-gray-200 p-5 dark:border-gray-700\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">Granice generativnog AI sistema\u003C\u002Fh3>\u003Cdiv id=\"faq1\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">Da li je generativna AI isto što i LLM?\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">Ne. LLM je jedna vrsta generativnog modela. Generativna AI takođe uključuje druge modalitete, a produkcijski generativni AI sistem može uključivati pretragu, alate, runtime logiku, stanje aplikacije, dozvole, perzistenciju i korisničke interfejse oko modela.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq2\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">Da li je RAG deo modela?\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">Obično ne. RAG je obrazac aplikacije\u002Fsistema koji pribavlja eksterne informacije i dostavlja izabrane dokaze modelu. Neke platforme čvrsto pakuju pretragu sa model API-jevima, ali odgovornost ostaje različita.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq3\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">Da li je vektorska baza podataka neophodna za RAG?\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">Ne. RAG može koristiti vektorsku pretragu, leksičku pretragu, hibridnu pretragu, SQL, API-je, grafove znanja ili druge metode. Definišuće svojstvo je pribavljanje eksternih informacija za generisanje, a ne jedna tehnologija skladištenja.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq4\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">Da li su alati isto što i kontekst?\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">Ne. Alat je eksterna sposobnost. Njegova definicija može biti predstavljena u kontekstu, a njegov rezultat može kasnije ući u kontekst, ali sama sposobnost se izvršava izvan modela.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq5\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">Da li pokretanje AI klijenta lokalno znači da je model lokalni?\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">Ne. Lokacija runtime-a i lokacija inferencije su odvojene. Lokalna desktop aplikacija ili agent može pozvati udaljeni model, dok udaljena aplikacija može pozvati interno hostovan model.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq6\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">Ko treba da sprovodi dozvole za AI alate?\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">Bezbednosna granica aplikacije ili runtime-a treba da sprovodi autorizaciju. Model može zatražiti operaciju, ali namera modela nikada ne treba da se smatra dovoljnim autoritetom za izvršavanje.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq7\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">Gde pripada trenutno stanje aplikacije?\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">Autoritativno volatilno stanje obično treba da ostane u aplikaciji ili domenskom sistemu koji ga poseduje. AI može primiti relevantno stanje kroz kontrolisani kontekst ili pristup alatima kada je to potrebno.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-88\">Pojmovnik\u003C\u002Fh2>\n\u003Csection class=\"editorjs-glossary my-6 rounded-xl border border-gray-200 dark:border-gray-700 p-5\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">Ključni pojmovi\u003C\u002Fh3>\u003Cdl>\u003Cdiv id=\"generative-model\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">Generativni model\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">AI model dizajniran da generiše izvedeni sintetički sadržaj kao što su tekst, slike, audio, video, kod ili strukturirani izlaz.\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"retrieval\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">Pretraga\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Proces odabira relevantnih informacija iz eksternog izvora ili skladišta za trenutni zadatak.\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"rag\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">RAG\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Retrieval-Augmented Generation: obrazac u kojem se pribavljene eksterne informacije dostavljaju generativnom modelu radi poboljšanja trenutnog izlaza.\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"tool\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">Alat\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Sposobnost izložena AI runtime-u za čitanje podataka, izračunavanje, pretragu ili izvršavanje eksterne radnje.\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"context\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">Kontekst\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Informacije dostupne modelu za određeni korak inferencije.\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"runtime-orchestrator\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">Runtime \u002F orkestrator\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Softverski sloj koji koordinira pozive modela, pozive alata, petlje zadataka, sesije, ponovne pokušaje, događaje ili okruženja za izvršavanje.\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"application\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">Aplikacija\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Sloj proizvoda i domena koji poseduje korisničku interakciju, autoritativno stanje, dozvole, validaciju, perzistenciju i poslovno ponašanje.\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"provider\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">Dobavljač\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Servis ili runtime koji izlaže pristup jednom ili više modela; identitet dobavljača i identitet modela su odvojene stvari.\u003C\u002Fdd>\u003C\u002Fdiv>\u003C\u002Fdl>\u003C\u002Fsection>\n\u003Ch2 id=\"section-90\">Primarni izvori i dokazi o implementaciji\u003C\u002Fh2>\n\u003Cp>Stabilne definicije u nastavku su zasnovane na standardima\u002Fistraživanjima; primeri implementacije koji se brzo menjaju koriste aktuelnu zvaničnu inženjersku dokumentaciju. Aaasaasa AI Client je dokaz originalne implementacije i proveren je u odnosu na stanje svoje baze koda\u002Fdokumentacije od 26. jula 2026.\u003C\u002Fp>\n\u003Ca href=\"https:\u002F\u002Fnvlpubs.nist.gov\u002Fnistpubs\u002Fai\u002FNIST.AI.600-1.pdf\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">NIST AI 600-1 — Profil generativne veštačke inteligencije\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">NIST-ov profil generativne AI, uključujući definiciju generativne AI i eksplicitnu razliku između pitanja na nivou modela, sistema, aplikacije i slučaja upotrebe.\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fcsrc.nist.gov\u002Fglossary\u002Fterm\u002Fartificial_intelligence_model\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">NIST — Model veštačke inteligencije\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">Aktuelna NIST-ova definicija iz glosara za AI model kao komponentu informacionog sistema koja proizvodi izlaze iz ulaza koristeći AI tehnike.\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fcsrc.nist.gov\u002Fglossary\u002Fterm\u002Fartificial_intelligence_system\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">NIST — Sistem veštačke inteligencije\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">Aktuelna NIST-ova definicija iz glosara koja pokazuje da AI sistem može uključivati podatkovne sisteme, softver, hardver, aplikacije, alate ili uslužne programe koji koriste AI.\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2005.11401\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">Lewis i sar. — Generisanje uz pomoć pretrage za NLP zadatke intenzivne znanjem\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">Rad iz 2020. godine koji uvodi RAG formulaciju koja kombinuje generativni model sa pronađenom neparametarskom memorijom.\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Ftools-file-search\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">OpenAI — Pretraga fajlova\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">Aktuelna zvanična dokumentacija za hostovanu pretragu fajlova u Responses API-ju korišćenjem baza znanja sa otpremljenim fajlovima, semantičke pretrage i pretrage po ključnim rečima.\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Ffunction-calling\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">OpenAI — Pozivanje funkcija\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">Aktuelna zvanična dokumentacija koja opisuje pozivanje alata\u002Ffunkcija kao interfejs između modela i eksternih sistema, podataka i radnji.\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Feffective-context-engineering-for-ai-agents\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">Anthropic — Efikasno inženjerstvo konteksta za AI agente\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">Inženjerske smernice koje definišu kontekst kao skup tokena dostupnih tokom LLM uzorkovanja i objašnjavaju zašto je izbor konteksta problem ograničenih resursa.\u003C\u002Fp>\u003C\u002Fa>",{"time":212,"blocks":213,"version":1068},1791475535658,[214,220,228,235,243,248,253,258,265,270,275,307,312,317,360,365,370,375,380,385,390,395,402,411,416,421,426,431,437,442,447,452,457,462,467,472,477,482,487,492,498,503,508,544,549,554,585,590,595,601,606,611,616,643,649,654,683,688,693,733,738,743,776,781,786,791,818,823,828,833,839,844,849,857,865,870,875,880,885,890,895,900,905,910,915,920,925,959,964,994,999,1004,1014,1023,1032,1041,1050,1059],{"id":215,"data":216,"type":218,"tunes":219},"intro",{"text":217},"Generativna veštačka inteligencija nije jedna komponenta. Produkcijski generativni AI sistem obično kombinuje generativni model sa aplikativnim kodom koji obezbeđuje instrukcije i kontekst, pronalazi eksterno znanje kada je potrebno, izlaže alate za čitanje ili menjanje eksternih sistema, upravlja stanjem izvršavanja i dozvolama i pretvara rezultat u upotrebljiv proizvod. Tretiranje modela, pronalaženja, alata, konteksta, izvršnog okruženja i aplikacije kao iste stvari skriva granice koje određuju svežinu, bezbednost, pouzdanost, cenu i kontrolu.","paragraph",{},{"id":221,"data":222,"type":226,"tunes":227},"direct",{"body":223,"title":224,"variant":225},"\u003Cstrong>Model generiše; pronalaženje pronalazi eksterne dokaze; alati pristupaju podacima ili izvršavaju radnje; kontekst je ono što model može da vidi za trenutno zaključivanje; izvršno okruženje koordinira izvršavanje; aplikacija poseduje pravila proizvoda, stanje, dozvole, trajnost i korisničko iskustvo.\u003C\u002Fstrong> Ove slojeve dobavljač može pakovati zajedno, ali njihove odgovornosti ostaju različite.","Direktan odgovor","info","callout",{},{"id":229,"data":230,"type":226,"tunes":234},"scope-note",{"body":231,"title":232,"variant":233},"Ovaj članak definiše trajne arhitektonske odgovornosti, a ne jedan dobavljački stek. Primeri trenutnih implementacija ponovo su provereni \u003Cstrong>8. oktobra 2026.\u003C\u002Fstrong> API-ji dobavljača i nazivi proizvoda mogu se menjati; granice odgovornosti su stabilnije od bilo kog pojedinačnog SDK-a ili endpointa.","Napomena o terminologiji i verziji","note",{},{"id":236,"data":237,"type":241,"tunes":242},"toc",{"title":238,"maxLevel":239,"minLevel":240},"Sadržaj",3,2,"tableOfContents",{},{"id":244,"data":245,"type":42,"tunes":247},"h-meaning",{"text":246,"level":240},"Šta zapravo znači „generativna veštačka inteligencija“?",{},{"id":249,"data":250,"type":218,"tunes":252},"p-meaning-1",{"text":251},"Na nivou modela, generativna veštačka inteligencija odnosi se na AI modele koji generišu izvedeni sintetički sadržaj kao što su tekst, slike, zvuk, video, kod ili drugi digitalni izlaz. NIST AI 600-1 koristi ovo značenje usmereno na model i odvojeno razmatra rizike na nivou modela, sistema, aplikacije i slučaja upotrebe.",{},{"id":254,"data":255,"type":218,"tunes":257},"p-meaning-2",{"text":256},"Ta razlika je važna jer AI model nije isto što i kompletan AI sistem. NIST-ov trenutni rečnik definiše AI model kao komponentu koja proizvodi izlaze iz ulaza koristeći računske, statističke ili tehnike mašinskog učenja, dok AI sistem može uključivati softver, hardver, aplikacije, alate ili uslužne programe koji rade koristeći AI.",{},{"id":259,"data":260,"type":226,"tunes":264},"model-system-rule",{"body":261,"title":262,"variant":263},"\u003Cstrong>Generativni model ≠ generativna AI aplikacija.\u003C\u002Fstrong>\u003Cbr>Model je jedna računska komponenta. Upotrebljiv AI proizvod je sistem izgrađen oko te komponente.","Korisna granica","success",{},{"id":266,"data":267,"type":42,"tunes":269},"h-simple",{"text":268,"level":240},"Najjednostavniji korisni model generativnog AI sistema",{},{"id":271,"data":272,"type":218,"tunes":274},"p-simple-1",{"text":273},"Za prvi mentalni model, zamislite korporativnog asistenta koji odgovara: „Može li ovaj kupac danas dobiti povraćaj?“ Koristan odgovor može zahtevati nekoliko različitih odgovornosti. Jezički model može interpretirati pitanje i napisati objašnjenje, ali trenutno stanje porudžbine može doći iz alata baze podataka, politika povraćaja može doći iz pronalaženja dokumenata, dozvole može sprovoditi aplikacija, a konačna radnja može zahtevati kontrolisani API poziv.",{},{"id":276,"data":277,"type":305,"tunes":306},"simple-flow",{"steps":278,"title":303,"orientation":304},[279,282,285,288,291,294,297,300],{"label":280,"description":281},"1. Korisnički zahtev","Aplikacija prima pitanje ili zadatak na prirodnom jeziku.",{"label":283,"description":284},"2. Politika i stanje aplikacije","Identitet, zakupac, dozvole, trenutno stanje toka rada i pravila proizvoda definišu šta zahtev sme da uradi.",{"label":286,"description":287},"3. Pronalaženje ili direktan pristup podacima","Sistem pribavlja eksterne dokaze ili trenutne činjenice kada je znanje modela nedovoljno.",{"label":289,"description":290},"4. Izgradnja konteksta","Instrukcije, korisnički unos, izabrani dokazi, relevantno stanje i definicije alata sastavljaju se za model.",{"label":292,"description":293},"5. Zaključivanje modela","Generativni model interpretira dostavljeni kontekst i proizvodi tekst, strukturirani izlaz ili zahtev za alat.",{"label":295,"description":296},"6. Izvršavanje alata kada je potrebno","Izvršno okruženje ili aplikacija validira i izvršava odobrene pozive alata izvan modela.",{"label":298,"description":299},"7. Opažanje i nastavak","Rezultati alata mogu se vratiti modelu kao novi kontekst za sledeći korak zaključivanja.",{"label":301,"description":302},"8. Validacija i izlaz proizvoda","Aplikacija validira rezultat, beleži potrebno stanje ili podatke za reviziju i prikazuje ili izvršava konačni ishod.","Jedna uobičajena putanja izvršavanja","auto","processFlow",{},{"id":308,"data":309,"type":218,"tunes":311},"p-simple-2",{"text":310},"Stvarni sistemi ne prate uvek tačno ovaj redosled. Pronalaženje se može desiti pre prvog poziva modela, alati se mogu birati tokom agentske petlje, deterministička aplikativna logika može potpuno zaobići model, a validacija se može desiti u nekoliko faza. Poenta je razdvojiti odgovornosti, a ne nametnuti jedan univerzalni tok rada.",{},{"id":313,"data":314,"type":42,"tunes":316},"h-boundaries",{"text":315,"level":240},"Šest granica koje su važne",{},{"id":318,"data":319,"type":358,"tunes":359},"boundary-comparison",{"rows":320,"title":346,"layout":347,"columns":348},[321,326,330,334,338,342],{"id":322,"label":323,"values":324},"model","Model",[325,325,325],"",{"id":327,"label":328,"values":329},"retrieval","Pronalaženje",[325,325,325],{"id":331,"label":332,"values":333},"tools","Alati",[325,325,325],{"id":335,"label":336,"values":337},"context","Kontekst",[325,325,325],{"id":339,"label":340,"values":341},"runtime","Izvršno okruženje \u002F orkestrator",[325,325,325],{"id":343,"label":344,"values":345},"application","Aplikacija",[325,325,325],"Šest odgovornosti unutar jednog AI proizvoda","table",[349,352,355],{"id":350,"label":351},"job","Primarni zadatak",{"id":353,"label":354},"input","Tipični ulazi",{"id":356,"label":357},"not","Nije isto kao","comparison",{},{"id":361,"data":362,"type":42,"tunes":364},"h-model",{"text":363,"level":240},"1. Model: generisanje je njegova osnovna odgovornost",{},{"id":366,"data":367,"type":218,"tunes":369},"p-model-1",{"text":368},"Generativni model preslikava dostavljene ulaze u generisane izlaze. Za jezički model to može uključivati tekst na prirodnom jeziku, strukturirani JSON, kod, klasifikacije, rezimee, planove ili argumente poziva alata. Multimodalni generativni modeli mogu raditi sa dodatnim tipovima ulaza i izlaza.",{},{"id":371,"data":372,"type":218,"tunes":374},"p-model-2",{"text":373},"Model može sadržati značajno naučeno znanje u svojim parametrima, ali parametrizovano znanje nije živa baza podataka. Model ne zna automatski za dokument kreiran pre pet minuta, trenutni nivo zaliha, privatni zapis kupca ili stanje aplikacije osim ako se ta informacija ne dostavi kroz trenutnu ulaznu putanju.",{},{"id":376,"data":377,"type":218,"tunes":379},"p-model-3",{"text":378},"Zato promena modela ne rešava automatski zastarelo znanje, nedostajuće dozvole, pokvareno pronalaženje, pogrešno vlasništvo nad stanjem ili nesigurno izvršavanje alata. Ti kvarovi često pripadaju drugim slojevima.",{},{"id":381,"data":382,"type":42,"tunes":384},"h-retrieval",{"text":383,"level":240},"2. Pretraga: pronalaženje spoljnih dokaza je zasebna operacija",{},{"id":386,"data":387,"type":218,"tunes":389},"p-retrieval-1",{"text":388},"Pretraga bira informacije iz spoljnog izvora pre ili tokom generisanja. Rad o generisanju uz pomoć pretrage iz 2020. godine autora Lewis i saradnika učinio je tu razdvojenost eksplicitnom kombinovanjem parametarskog generativnog modela sa preuzetom neparametarskom memorijom. Savremeni produkcioni sistemi koriste mnoge varijante pretrage, ali arhitektonska ideja ostaje: korisni dokazi mogu se dohvatiti u vreme izvođenja zaključivanja umesto da se oslanjamo samo na ono što je model naučio tokom obuke.",{},{"id":391,"data":392,"type":218,"tunes":394},"p-retrieval-2",{"text":393},"Pretraga može koristiti leksičku pretragu, ugrađivanja, vektorsku pretragu, hibridnu pretragu, SQL, grafove znanja, filtere metapodataka, API-je ili druge mehanizme selekcije. Vektorska baza podataka je stoga jedna moguća komponenta pretrage, a ne definicija RAG-a.",{},{"id":396,"data":397,"type":226,"tunes":401},"retrieval-rule",{"body":398,"title":399,"variant":400},"Preuzeti odlomak može biti veoma relevantan, a ipak zastareo, neovlašćen, iz pogrešne verzije ili nedovoljan da podrži tvrdnju. Kvalitet pretrage i kvalitet dokaza moraju se procenjivati odvojeno.","Relevantnost nije autoritet","warning",{},{"id":403,"data":404,"type":409,"tunes":410},"ref-rag",{"url":405,"title":406,"excerpt":407,"ctaLabel":408},"https:\u002F\u002Fstajic.de\u002Fsr\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works","Šta je RAG? Najjednostavnije objašnjenje kako funkcioniše","Kanonsko objašnjenje generisanja uz pomoć pretrage jednostavnim jezikom, uključujući razdvajanje između LLM-a, znanja, stanja, memorije i alata.","Pročitajte osnove RAG-a","referralArticle",{},{"id":412,"data":413,"type":42,"tunes":415},"h-tools",{"text":414,"level":240},"3. Alati: pristup i radnja nisu znanje modela",{},{"id":417,"data":418,"type":218,"tunes":420},"p-tools-1",{"text":419},"Alat je interfejs preko kojeg AI okruženje za izvršavanje može zatražiti funkcionalnost izvan modela. Alat može upitovati bazu podataka, pretraživati veb, čitati datoteku, izračunati vrednost, pozvati interni servis, kreirati tiket, poslati poruku, izmeniti zapis ili pokrenuti drugu kontrolisanu operaciju.",{},{"id":422,"data":423,"type":218,"tunes":425},"p-tools-2",{"text":424},"OpenAI-jeva trenutna dokumentacija o pozivanju funkcija eksplicitno ističe ovu granicu: pozivanje funkcija omogućava modelima da se povežu sa spoljnim sistemima i pristupe podacima ili radnjama koje pruža aplikacija. Model može predložiti ili izabrati poziv, ali spoljni sistem obavlja stvarnu operaciju.",{},{"id":427,"data":428,"type":218,"tunes":430},"p-tools-3",{"text":429},"Upotreba alata stoga stvara dva odvojena pitanja: Može li model zatražiti ovu sposobnost? i Hoće li aplikacija odobriti i izvršiti je? Produkcioni sistem ne bi trebalo da meša nameru modela sa dozvolom da izazove sporedni efekat.",{},{"id":432,"data":433,"type":226,"tunes":436},"tool-rule",{"body":434,"title":435,"variant":263},"Model može uputiti važeći zahtev za alat i ipak biti odbijen. Autorizacija, validacija argumenata, ograničenja brzine, pravila transakcija, zahtevi za reviziju i vraćanje unazad pripadaju izvan modela.","Namere modela nisu ovlašćenje za izvršenje",{},{"id":438,"data":439,"type":42,"tunes":441},"h-context",{"text":440,"level":240},"4. Kontekst: šta model može videti upravo sada",{},{"id":443,"data":444,"type":218,"tunes":446},"p-context-1",{"text":445},"Kontekst su informacije dostupne modelu za određeni korak zaključivanja. Anthropic-ove smernice za inženjering konteksta opisuju kontekst kao skup tokena uključenih pri uzorkovanju iz LLM-a. U praksi, taj skup može sadržati sistemska uputstva, korisničke poruke, istoriju razgovora, preuzete dokaze, definicije alata, rezultate alata, sažetke memorije i izabrano stanje aplikacije.",{},{"id":448,"data":449,"type":218,"tunes":451},"p-context-2",{"text":450},"Kontekst stoga nije ni kompletna baza znanja ni dugoročna memorija. Kompanija može čuvati deset miliona dokumenata dok samo nekoliko odlomaka ulazi u jedan poziv modela. Okruženje za izvršavanje može čuvati godinu dana istorije razgovora dok izlaže samo delove potrebne za trenutni zadatak.",{},{"id":453,"data":454,"type":218,"tunes":456},"p-context-3",{"text":455},"Kontekstni prozor takođe stvara inženjersko ograničenje. Dodavanje više teksta ne garantuje bolji odgovor; irelevantne, zastarele, kontradiktorne ili informacije niskog autoriteta mogu razblažiti dokaze koji su zaista važni.",{},{"id":458,"data":459,"type":42,"tunes":461},"h-runtime",{"text":460,"level":240},"5. Okruženje za izvršavanje i orkestracija: koordinisanje petlje",{},{"id":463,"data":464,"type":218,"tunes":466},"p-runtime-1",{"text":465},"Sloj okruženja za izvršavanje ili orkestracije koordinira kako model učestvuje u zadatku. U zavisnosti od arhitekture, može upravljati sesijama, zahtevima modela, otkrivanjem alata, petljama pozivanja alata, ponovnim pokušajima, predajama, događajima strimovanja, istekima vremena, kontrolnim tačkama, sažimanjem ili okruženjima za izvršavanje.",{},{"id":468,"data":469,"type":218,"tunes":471},"p-runtime-2",{"text":470},"Neka okruženja za izvršavanje su tanki aplikacijski kod oko API-ja modela. Druga su potpuni agentni okviri. Upravljano okruženje dobavljača može posedovati deo petlje dok aplikacija i dalje poseduje domen istine, autorizaciju, poslovne sporedne efekte i životni ciklus proizvoda.",{},{"id":473,"data":474,"type":218,"tunes":476},"p-runtime-3",{"text":475},"Ova granica je važna jer su to gde se izvršava okruženje i gde se izvršava zaključivanje odvojene odluke. Lokalno pokrenut klijent ili agentni proces i dalje može pozivati udaljeni model, dok udaljena aplikacija može pozivati model hostovan na infrastrukturi pod kontrolom organizacije.",{},{"id":478,"data":479,"type":42,"tunes":481},"h-application",{"text":480,"level":240},"6. Aplikacija: gde AI postaje proizvod",{},{"id":483,"data":484,"type":218,"tunes":486},"p-app-1",{"text":485},"Aplikacija je granica proizvoda oko AI komponenti. Ona poseduje korisničko iskustvo, domenski model, trenutno stanje, identitet, opseg zakupca, dozvole, perzistenciju, integracije servisa, validaciju, observabilnost, logiku naplate ili kvota gde je relevantno, i pravila koja određuju šta AI sme da vidi ili radi.",{},{"id":488,"data":489,"type":218,"tunes":491},"p-app-2",{"text":490},"Ovo je sloj koji pretvara „model može da proizvede koristan izlaz“ u „sistem može da isporuči pouzdanu sposobnost.“ Isti model može da učestvuje u privatnom istraživačkom asistentu, korisničkom toku podrške, kodnom agentu ili aplikaciji za trgovinu jer okolna aplikacija menja podatke, alate, politike, stanje i ugovor o izvršavanju.",{},{"id":493,"data":494,"type":226,"tunes":497},"app-rule",{"body":495,"title":496,"variant":225},"Zamena provajdera i modela može biti arhitektonski cilj. Autoritativno stanje aplikacije, dozvole, domenska pravila, revizorski trag i korisnički ugovor ne mogu se jednostavno delegirati modelu koji je trenutno izabran.","Model je zamenljiv; granica proizvoda nije",{},{"id":499,"data":500,"type":42,"tunes":502},"h-work-together",{"text":501,"level":240},"Kako delovi rade zajedno u stvarnom zahtevu",{},{"id":504,"data":505,"type":218,"tunes":507},"p-together-1",{"text":506},"Razmotrimo asistenta za podršku upitanog: „Refundiraj porudžbinu 4711 ako je još uvek podobna, i objasni zašto.“ Zahtev kombinuje znanje, trenutno stanje, autorizaciju, rezonovanje i sporedni efekat.",{},{"id":509,"data":510,"type":347,"tunes":543},"support-table",{"content":511,"stretched":43,"withHeadings":14},[512,516,520,524,528,532,536,539],[513,514,515],"Potreba","Ispravan sloj","Zašto",[517,518,519],"Politika refundacije","Preuzimanje","Sistem mora da pronađe trenutno primenljivu politiku i sačuva njeno poreklo.",[521,522,523],"Status porudžbine 4711","Direktan pristup podacima\u002Falatu","Trenutni zapis porudžbine je promenljivo autoritativno stanje, ne nešto što se pogađa iz znanja modela.",[525,526,527],"Korisnikovo ovlašćenje za refundaciju","Aplikacija \u002F autorizacija","Dozvole se moraju sprovesti nezavisno od onoga što model traži.",[529,530,531],"Interpretacija politike prema činjenicama porudžbine","Model + kontekst","Model može da rezonuje nad dokazima politike i trenutnim stanjem porudžbine koji su mu dostavljeni.",[533,534,535],"Izvrši refundaciju","Alat + pravila transakcije aplikacije","Kontrolisana eksterna operacija menja stvarno stanje.",[537,323,538],"Objasni ishod","Model može da generiše objašnjenje za korisnika iz validiranih rezultata.",[540,541,542],"Revizija onoga što se dogodilo","Aplikacija \u002F vreme izvršavanja","Sistem beleži dokaze, pozive, odluke, sporedne efekte i greške po potrebi.",{},{"id":545,"data":546,"type":218,"tunes":548},"p-together-2",{"text":547},"Ako asistent ima samo jezički model, može da diskutuje o refundacijama ali ne može bezbedno da zna da li je porudžbina 4711 trenutno podobna ili da izvrši transakciju. Ako ima samo preuzimanje, može da pronađe politiku ali mu i dalje nedostaje stanje porudžbine uživo. Ako ima alate bez autorizacije aplikacije, može postati sposoban ali nesiguran. Pouzdanost dolazi iz komponovanja slojeva sa eksplicitnim vlasništvom.",{},{"id":550,"data":551,"type":42,"tunes":553},"h-configs",{"text":552,"level":240},"Različiti AI proizvodi koriste različite kombinacije",{},{"id":555,"data":556,"type":358,"tunes":584},"config-comparison",{"rows":557,"title":574,"layout":347,"columns":575},[558,562,566,570],{"id":559,"label":560,"values":561},"bare","Asistent samo sa modelom",[325,325,325,325],{"id":563,"label":564,"values":565},"rag","Asistent zasnovan na preuzimanju",[325,325,325,325],{"id":567,"label":568,"values":569},"tool","Asistent koji koristi alate",[325,325,325,325],{"id":571,"label":572,"values":573},"agent","Agentna aplikacija",[325,325,325,325],"Prisustvo modela ne definiše celu arhitekturu",[576,577,578,581],{"id":327,"label":518},{"id":331,"label":332},{"id":579,"label":580},"state","Autoritativno stanje",{"id":582,"label":583},"result","Tipična sposobnost",{},{"id":586,"data":587,"type":218,"tunes":589},"p-configs-1",{"text":588},"Ovo su arhitektonski obrasci, a ne rangiranje zrelosti. Funkcija samo sa modelom može biti ispravan dizajn kada zadatak ne zahteva spoljne činjenice ili radnje. Dodavanje preuzimanja, alata, memorije ili agentne petlje opravdano je samo kada zadatak zahteva te sposobnosti.",{},{"id":591,"data":592,"type":42,"tunes":594},"h-implementation",{"text":593,"level":240},"Dokaz implementacije: Aaasaasa AI Client",{},{"id":596,"data":597,"type":226,"tunes":600},"implementation-scope",{"body":598,"title":599,"variant":233},"Sledeći odeljak opisuje implementaciju koju sam izgradio i pregledao u odnosu na Aaasaasa AI Client kodnu bazu i arhitektonsku dokumentaciju zaključno sa \u003Cstrong>26. julom 2026.\u003C\u002Fstrong> To je dokaz korisnosti ovih granica, a ne tvrdnja da je jedna implementacija univerzalni standard ili komercijalno raspoređen enterprise proizvod.","Primarni dokaz implementacije",{},{"id":602,"data":603,"type":218,"tunes":605},"p-impl-1",{"text":604},"Aaasaasa AI Client je lokalno-prvenstveni desktop AI radni prostor izgrađen sa Nuxt 4, Electron i TypeScript. Njegov AI Hub namerno razdvaja agenta\u002Fklijenta, provajdera, model, lokaciju izvršavanja, dozvole i web klijenta umesto da ih tretira kao jedno „AI“ podešavanje.",{},{"id":607,"data":608,"type":218,"tunes":610},"p-impl-2",{"text":609},"To razdvajanje stvara konkretno ponašanje. Direct Chat može da razgovara sa modelima bez alata za fajl sistem ili shell. Codex agent može da koristi izabrani radni prostor i profil dozvola. Ollama može da obezbedi direktno lokalno zaključivanje, dok LM Studio i konfigurabilni OpenAI-kompatibilni endpointi predstavljaju druge putanje provajdera. Lokalno pokrenut Codex proces i dalje može da koristi cloud model, tako da UI i arhitektura ne izjednačavaju lokalno izvršavanje sa lokalnim zaključivanjem.",{},{"id":612,"data":613,"type":218,"tunes":615},"p-impl-3",{"text":614},"Implementacija takođe sadrži Qdrant\u002Fvektorsku podršku, mogućnosti ekstrakcije dokumenata i autentifikovani direktorijum MCP broker. Te komponente ilustruju još jednu granicu: infrastruktura za preuzimanje i pristup alatima mogu živeti u istom proizvodu a da ne postanu svojstva samog modela.",{},{"id":617,"data":618,"type":347,"tunes":642},"impl-map",{"content":619,"stretched":43,"withHeadings":14},[620,623,625,628,631,634,637,640],[621,622],"A01 koncept","Dokaz implementacije Aaasaasa AI Client",[323,624],"Identifikator modela specifičan za provajdera se bira odvojeno od provajdera i vremena izvršavanja.",[626,627],"Provajder","Ollama, LM Studio, OpenAI-kompatibilni servisi i druge putanje provajdera su predstavljeni odvojeno.",[629,630],"Vreme izvršavanja","Lokalna ili udaljena lokacija agenta\u002Fvremena izvršavanja se prati nezavisno od modela.",[632,633],"Alati \u002F pristup","Direct Chat nema alate za fajl sistem ili shell; kontrolisani pristup direktorijumu se posreduje odvojeno.",[635,636],"Dozvole","Profili dozvola radnog prostora su politika aplikacije\u002Fsesije, a ne sposobnost modela.",[638,639],"Infrastruktura za preuzimanje","Vektorska podrška i ekstrakcija dokumenata postoje kao sposobnosti podataka\u002Fpreuzimanja, a ne kao funkcije modela.",[344,641],"Electron\u002FNuxt proizvod koordinira UI, akreditive, provajdere, otkrivanje vremena izvršavanja, dozvole, alate i interakciju sa modelom.",{},{"id":644,"data":645,"type":226,"tunes":648},"impl-lesson",{"body":646,"title":647,"variant":263},"Arhitektura je postala lakša za razumevanje kada su \u003Cstrong>model, provajder, vreme izvršavanja, dozvole, alati, podaci i klijent\u003C\u002Fstrong> prestali da budu predstavljeni kao jedan izbor konfiguracije. Razlika je operativna: određuje šta može da se pokrene lokalno, šta može da pristupi fajlovima, šta može da pozove plaćeno cloud zaključivanje i koji sloj poseduje autorizaciju.","Lekcija iz implementacije",{},{"id":650,"data":651,"type":42,"tunes":653},"h-errors",{"text":652,"level":240},"Česte greške u kategorizaciji",{},{"id":655,"data":656,"type":347,"tunes":682},"errors-table",{"content":657,"stretched":43,"withHeadings":14},[658,661,664,667,670,673,676,679],[659,660],"Greška u kategorizaciji","Šta se zapravo dešava",[662,663],"„AI zna naše dokumente.“","Aplikacija ili sloj za pronalaženje čini sadržaj odabranih dokumenata dostupnim modelu.",[665,666],"„RAG je naša vektorska baza podataka.“","Vektorska baza podataka može biti jedan indeks ili skladište koje koristi pipeline za pronalaženje; RAG je obrazac pronalaženja i generisanja.",[668,669],"„Model je pozvao naš CRM.“","Model je proizveo zahtev za alat; runtime\u002Faplikacija je autorizovala i izvršila eksterni poziv.",[671,672],"„To je lokalna AI jer desktop agent radi lokalno.“","Lokacija izvršavanja i lokacija inferencije su odvojene. Lokalni runtime i dalje može pozvati udaljeni model.",[674,675],"„Model ima dozvolu da menja fajlove.“","Aplikacija\u002Fruntime dodeljuje sposobnost alata prema politici dozvola; dozvola nije intrinzično svojstvo modela.",[677,678],"„Više konteksta znači više znanja.“","Kontekst je konačan unos koji je dostupan za jednu inferenciju. Veći kontekst može sadržati više šuma, konflikata ili zastarelih informacija.",[680,681],"„Chatbot je AI arhitektura.“","Chat UI je jedan interfejs. Sistem takođe može uključivati identitet, stanje, pronalaženje, alate, runtime, validaciju, perzistenciju i observabilnost.",{},{"id":684,"data":685,"type":42,"tunes":687},"h-failures",{"text":686,"level":240},"Načini otkaza kada se granice uruše",{},{"id":689,"data":690,"type":218,"tunes":692},"p-failure-intro",{"text":691},"Greške u granicama nisu samo terminološki problemi. One stvaraju različite produkcione otkaze koji zahtevaju različita rešenja.",{},{"id":694,"data":695,"type":358,"tunes":732},"failure-comparison",{"rows":696,"title":721,"layout":347,"columns":722},[697,701,705,709,713,717],{"id":698,"label":699,"values":700},"stale","Zastareo odgovor",[325,325,325],{"id":702,"label":703,"values":704},"missing","Nedostaje činjenica o kompaniji",[325,325,325],{"id":706,"label":707,"values":708},"unsafe","Nesigurna sporedna radnja",[325,325,325],{"id":710,"label":711,"values":712},"noise","Zbunjen odgovor sa puno priloženog teksta",[325,325,325],{"id":714,"label":715,"values":716},"route","Neočekivano korišćenje oblaka",[325,325,325],{"id":718,"label":719,"values":720},"loop","Agent se zaglavljuje ili ponavlja",[325,325,325],"Dijagnostikujte sloj koji otkazuje pre nego što zamenite model",[723,726,729],{"id":724,"label":725},"symptom","Simptom",{"id":727,"label":728},"likely","Verovatni problem granice",{"id":730,"label":731},"fix","Prva arhitektonska provera",{},{"id":734,"data":735,"type":42,"tunes":737},"h-version",{"text":736,"level":240},"Šta je stabilno, a šta zavisi od verzije?",{},{"id":739,"data":740,"type":218,"tunes":742},"p-version-1",{"text":741},"Arhitektonske razlike u ovom članku su namerno neutralne prema dobavljaču. Trenutni primeri ispod su činjenice o implementaciji koje treba ponovo proveriti kada se API-ji razviju.",{},{"id":744,"data":745,"type":347,"tunes":775},"version-table",{"content":746,"stretched":43,"withHeadings":14},[747,751,755,759,763,767,771],[748,749,750],"Oblast","Stabilna arhitektonska ideja","Provereni trenutni primer na dan 8. oktobar 2026.",[752,753,754],"AI model naspram sistema","Model je komponenta unutar šireg sistema","NIST-ov trenutni rečnik odvojeno definiše AI model i AI sistem.",[756,757,758],"RAG","Generisanje može biti uslovljeno pronađenim eksternim informacijama","Formulacija Lewis et al. iz 2020. ostaje temeljna referenca; produkcione metode pronalaženja sada sežu daleko izvan jednog dizajna gustog indeksa.",[760,761,762],"Hostovano pronalaženje","Pronalaženje može biti izloženo kao upravljani alat","OpenAI File Search je trenutno alat Responses API-ja koji pretražuje baze znanja otpremljenih fajlova koristeći semantičko i pretraživanje po ključnim rečima.",[764,765,766],"Pozivanje funkcija\u002Falata","Model može zahtevati eksterne sposobnosti definisane aplikacijom","OpenAI trenutno dokumentuje pozivanje funkcija kao interfejs ka eksternim sistemima, podacima i radnjama.",[768,769,770],"Inženjering konteksta","Ponašanje modela zavisi od konačne informacije koja se pruža za trenutnu inferenciju","Anthropic-ove trenutne inženjerske smernice definišu kontekst kao skup tokena uključenih pri uzorkovanju iz LLM-a i fokusiraju se na njegovo pažljivo odabiranje.",[772,773,774],"API-ji dobavljača","SDK-ovi, imena alata, oblici endpointa i podržane funkcije se menjaju","Tretirajte dokumentaciju dobavljača kao osetljivu na verziju čak i kada granica odgovornosti ostaje stabilna.",{},{"id":777,"data":778,"type":218,"tunes":780},"p-version-2",{"text":779},"Članak koji je izvor istine treba stoga da očuva oba nivoa: stabilne koncepte za arhitekturu i datirane dokaze za trenutne implementacije. Mešanje ova dva čini da članak nepotrebno brzo zastari.",{},{"id":782,"data":783,"type":42,"tunes":785},"h-test",{"text":784,"level":240},"Test granica AI komponenti",{},{"id":787,"data":788,"type":218,"tunes":790},"p-test-1",{"text":789},"Kada procenjujete AI funkciju, postavite sledeća pitanja po redu. Odgovori otkrivaju koje komponente sistem zapravo ima i koje su odgovornosti još uvek implicitne.",{},{"id":792,"data":793,"type":305,"tunes":817},"boundary-test",{"steps":794,"title":816,"orientation":304},[795,798,801,804,807,810,813],{"label":796,"description":797},"1. Šta generiše izlaz?","Identifikujte tačan model i modalitete ili strukturisane izlaze koje pruža.",{"label":799,"description":800},"2. Koje činjenice su autoritativne izvan modela?","Identifikujte dokumente, baze podataka, API-je, trenutno stanje i druge izvore istine.",{"label":802,"description":803},"3. Kako se biraju relevantne informacije?","Razdvojite direktno pretraživanje, pretragu, pronalaženje, rangiranje i konstrukciju konteksta.",{"label":805,"description":806},"4. Šta može izazvati stvarne sporedne efekte?","Navedite alate i eksterne radnje, zatim identifikujte ko ih validira i autorizuje.",{"label":808,"description":809},"5. Šta dospeva do modela kao kontekst?","Učinite eksplicitnim instrukcije, dokaze, stanje, istoriju, memoriju i definicije alata.",{"label":811,"description":812},"6. Ko poseduje petlju?","Identifikujte runtime ili harness koji upravlja pozivima, događajima, ponovnim pokušajima, petljama alata i sesijama.",{"label":814,"description":815},"7. Šta ostaje odgovornost aplikacije?","Učinite eksplicitnim identitet, dozvole, stanje domena, validaciju, perzistenciju, observabilnost i UX.","Sedam pitanja za produkcioni dizajn",{},{"id":819,"data":820,"type":42,"tunes":822},"h-not",{"text":821,"level":240},"Šta generativna AI nije",{},{"id":824,"data":825,"type":218,"tunes":827},"p-not-1",{"text":826},"Generativna AI nije sinonim za LLM, iako su LLM-ovi glavna klasa generativnih modela. Takođe nije sinonim za RAG, vektorsku bazu podataka, agenta, protokol alata, chat UI ili aplikaciju.",{},{"id":829,"data":830,"type":218,"tunes":832},"p-not-2",{"text":831},"Ti koncepti mogu biti povezani, ali svaki odgovara na drugo arhitektonsko pitanje. LLM pita kako se proizvodi jezički izlaz. Pronalaženje pita odakle dolaze eksterni dokazi. Alati pitaju kako se izlažu eksterne sposobnosti. Kontekst pita šta model može da vidi. Runtime pita kako se koordinira izvršavanje. Aplikacija pita kako sposobnost postaje kontrolisan proizvod.",{},{"id":834,"data":835,"type":226,"tunes":838},"remember",{"body":836,"title":837,"variant":263},"\u003Cstrong>Model = generiše.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Pronalaženje = pronalazi dokaze.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Alati = čitaju ili deluju izvan modela.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Kontekst = šta model vidi sada.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Runtime = koordinira izvršavanje.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Aplikacija = poseduje proizvod, stanje, pravila i dozvole.\u003C\u002Fstrong>","Ako zapamtite samo jedan model",{},{"id":840,"data":841,"type":42,"tunes":843},"h-next",{"text":842,"level":240},"Gde dalje u grafu znanja",{},{"id":845,"data":846,"type":218,"tunes":848},"p-next-1",{"text":847},"Kada su ove granice jasne, dublje teme postaje lakše smestiti. RAG pripada pronalaženju i konstrukciji konteksta. Retrieval Trigger odlučuje kada su eksterni dokazi potrebni. Memorija agenta se tiče onoga što se održava kroz vreme. Pozivanje alata i MCP pripadaju pristupu sposobnostima. Agent harnesses pripadaju runtime orkestraciji. RBAC, izolacija zakupaca i autorizacija domena pripadaju aplikaciji i bezbednosnoj granici platforme.",{},{"id":850,"data":851,"type":409,"tunes":856},"ref-data",{"url":852,"title":853,"excerpt":854,"ctaLabel":855},"https:\u002F\u002Fstajic.de\u002Fsr\u002Fblog\u002Fwhere-does-an-llm-get-its-data-rag-data-sources-in-python","Odakle LLM dobija svoje podatke? RAG izvori podataka u Python-u","Praktičan nastavak koji pokazuje kako fajlovi, SQL, API-ji, pretraga punog teksta, embedding-ovi i sastavljanje konteksta povezuju eksterne podatke sa LLM-om.","Pogledajte putanju podataka u kodu",{},{"id":858,"data":859,"type":409,"tunes":864},"ref-trigger",{"url":860,"title":861,"excerpt":862,"ctaLabel":863},"https:\u002F\u002Fstajic.de\u002Fsr\u002Fblog\u002Fwhen-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger","Kada AI treba da prestane da veruje sopstvenom znanju? — Okidač za pretragu","Model odlučivanja o tome kada AI sistem treba da prestane da se oslanja samo na znanje modela i pribavi eksterne dokaze.","Pročitajte model odlučivanja o pretrazi",{},{"id":866,"data":867,"type":42,"tunes":869},"h-limit",{"text":868,"level":240},"Ograničenja",{},{"id":871,"data":872,"type":218,"tunes":874},"p-limit-1",{"text":873},"Model sa šest slojeva je mapa odgovornosti, a ne zahtev da svaki proizvod implementira šest odvojenih servisa. Mala aplikacija može implementirati konstrukciju konteksta, pretragu i orkestraciju unutar jednog procesa. Upravljana platforma može objediniti nekoliko odgovornosti iza jednog API-ja. Fizičko postavljanje može biti kombinovano dok semantičko vlasništvo ostaje različito.",{},{"id":876,"data":877,"type":218,"tunes":879},"p-limit-2",{"text":878},"Terminologija se takođe razlikuje među dobavljačima i u istraživanjima. „Agent“, „runtime“, „memorija“, „alat“, „konektor“ i „kontekst“ mogu biti definisani na različite načine. Definicije ovde su izabrane da učine operativno vlasništvo i dijagnostiku kvarova eksplicitnim, a ne da tvrde da svaki okvir koristi identičan rečnik.",{},{"id":881,"data":882,"type":218,"tunes":884},"p-limit-3",{"text":883},"Odeljak Aaasaasa AI Client dokumentuje jedan obrazac implementacije. On pokazuje da su eksplicitne granice praktične, ali ne dokazuje da je isti raspored komponenti optimalan za svaki AI proizvod.",{},{"id":886,"data":887,"type":42,"tunes":889},"h-change",{"text":888,"level":240},"Šta bi promenilo ovaj odgovor?",{},{"id":891,"data":892,"type":218,"tunes":894},"p-change-1",{"text":893},"Mapa odgovornosti bi zahtevala reviziju ako bi same arhitekture modela počele da poseduju autoritativno eksterno stanje, dozvole, trajne transakcione sporedne efekte i verifikovan pristup izvorima kao intrinzične osobine, a ne kao sposobnosti koje pruža okolni sistem. Trenutne produkcijske arhitekture to ne čine bezbednom opštom pretpostavkom.",{},{"id":896,"data":897,"type":218,"tunes":899},"p-change-2",{"text":898},"Pojedinačni primeri implementacije će se promeniti mnogo ranije. Hostovani alati za pretragu, agent API-ji, MCP integracije, funkcije za upravljanje kontekstom i mogućnosti dobavljača se brzo razvijaju. Te detalje treba ažurirati bez urušavanja osnovnih razlika između generisanja, dokaza, pristupa sposobnostima, konteksta, izvršavanja i kontrole aplikacije.",{},{"id":901,"data":902,"type":42,"tunes":904},"h-conclusion",{"text":903,"level":240},"Zaključak",{},{"id":906,"data":907,"type":218,"tunes":909},"p-conclusion-1",{"text":908},"Generativnu AI je lakše dizajnirati kada „AI“ prestane da se tretira kao jedna crna kutija. Model je generativna komponenta, a ne kompletan proizvod. Pretraga pruža eksterne dokaze. Alati izlažu sposobnosti. Kontekst nosi izabrane informacije u trenutnu inferenciju. Runtime koordinira izvršavanje. Aplikacija poseduje autoritativnu granicu proizvoda.",{},{"id":911,"data":912,"type":218,"tunes":914},"p-conclusion-2",{"text":913},"Ta razdvojenost je korisna za više od objašnjenja. Ona govori inženjerima odakle potiču zastarele činjenice, gde pripada autorizacija, zašto lokalni runtime i dalje može da koristi cloud inferenciju, zašto RAG nije isto što i vektorska baza podataka, zašto pozivi alata zahtevaju validaciju i zašto promena modela ne može da popravi svaki sistemski kvar.",{},{"id":916,"data":917,"type":218,"tunes":919},"p-conclusion-3",{"text":918},"Trajno arhitektonsko pitanje stoga nije „Koji AI model koristimo?“ Već: Koju odgovornost poseduje svaka komponenta, koji dokazi prelaze svaku granicu i koji sloj sme da menja stvarno stanje?",{},{"id":921,"data":922,"type":42,"tunes":924},"h-faq",{"text":923,"level":240},"Često postavljana pitanja",{},{"id":926,"data":927,"type":926,"tunes":958},"faq",{"items":928,"title":957},[929,933,937,941,945,949,953],{"id":930,"answer":931,"question":932},"faq1","Ne. LLM je jedna vrsta generativnog modela. Generativna AI takođe uključuje druge modalitete, a produkcijski generativni AI sistem može uključivati pretragu, alate, runtime logiku, stanje aplikacije, dozvole, perzistenciju i korisničke interfejse oko modela.","Da li je generativna AI isto što i LLM?",{"id":934,"answer":935,"question":936},"faq2","Obično ne. RAG je obrazac aplikacije\u002Fsistema koji pribavlja eksterne informacije i dostavlja izabrane dokaze modelu. Neke platforme čvrsto pakuju pretragu sa model API-jevima, ali odgovornost ostaje različita.","Da li je RAG deo modela?",{"id":938,"answer":939,"question":940},"faq3","Ne. RAG može koristiti vektorsku pretragu, leksičku pretragu, hibridnu pretragu, SQL, API-je, grafove znanja ili druge metode. Definišuće svojstvo je pribavljanje eksternih informacija za generisanje, a ne jedna tehnologija skladištenja.","Da li je vektorska baza podataka neophodna za RAG?",{"id":942,"answer":943,"question":944},"faq4","Ne. Alat je eksterna sposobnost. Njegova definicija može biti predstavljena u kontekstu, a njegov rezultat može kasnije ući u kontekst, ali sama sposobnost se izvršava izvan modela.","Da li su alati isto što i kontekst?",{"id":946,"answer":947,"question":948},"faq5","Ne. Lokacija runtime-a i lokacija inferencije su odvojene. Lokalna desktop aplikacija ili agent može pozvati udaljeni model, dok udaljena aplikacija može pozvati interno hostovan model.","Da li pokretanje AI klijenta lokalno znači da je model lokalni?",{"id":950,"answer":951,"question":952},"faq6","Bezbednosna granica aplikacije ili runtime-a treba da sprovodi autorizaciju. Model može zatražiti operaciju, ali namera modela nikada ne treba da se smatra dovoljnim autoritetom za izvršavanje.","Ko treba da sprovodi dozvole za AI alate?",{"id":954,"answer":955,"question":956},"faq7","Autoritativno volatilno stanje obično treba da ostane u aplikaciji ili domenskom sistemu koji ga poseduje. AI može primiti relevantno stanje kroz kontrolisani kontekst ili pristup alatima kada je to potrebno.","Gde pripada trenutno stanje aplikacije?","Granice generativnog AI sistema",{},{"id":960,"data":961,"type":42,"tunes":963},"h-glossary",{"text":962,"level":240},"Pojmovnik",{},{"id":965,"data":966,"type":965,"tunes":993},"glossary",{"title":967,"entries":968},"Ključni pojmovi",[969,973,976,978,981,983,987,989],{"term":970,"anchor":971,"definition":972},"Generativni model","generative-model","AI model dizajniran da generiše izvedeni sintetički sadržaj kao što su tekst, slike, audio, video, kod ili strukturirani izlaz.",{"term":974,"anchor":327,"definition":975},"Pretraga","Proces odabira relevantnih informacija iz eksternog izvora ili skladišta za trenutni zadatak.",{"term":756,"anchor":563,"definition":977},"Retrieval-Augmented Generation: obrazac u kojem se pribavljene eksterne informacije dostavljaju generativnom modelu radi poboljšanja trenutnog izlaza.",{"term":979,"anchor":567,"definition":980},"Alat","Sposobnost izložena AI runtime-u za čitanje podataka, izračunavanje, pretragu ili izvršavanje eksterne radnje.",{"term":336,"anchor":335,"definition":982},"Informacije dostupne modelu za određeni korak inferencije.",{"term":984,"anchor":985,"definition":986},"Runtime \u002F orkestrator","runtime-orchestrator","Softverski sloj koji koordinira pozive modela, pozive alata, petlje zadataka, sesije, ponovne pokušaje, događaje ili okruženja za izvršavanje.",{"term":344,"anchor":343,"definition":988},"Sloj proizvoda i domena koji poseduje korisničku interakciju, autoritativno stanje, dozvole, validaciju, perzistenciju i poslovno ponašanje.",{"term":990,"anchor":991,"definition":992},"Dobavljač","provider","Servis ili runtime koji izlaže pristup jednom ili više modela; identitet dobavljača i identitet modela su odvojene stvari.",{},{"id":995,"data":996,"type":42,"tunes":998},"h-sources",{"text":997,"level":240},"Primarni izvori i dokazi o implementaciji",{},{"id":1000,"data":1001,"type":218,"tunes":1003},"p-sources-note",{"text":1002},"Stabilne definicije u nastavku su zasnovane na standardima\u002Fistraživanjima; primeri implementacije koji se brzo menjaju koriste aktuelnu zvaničnu inženjersku dokumentaciju. Aaasaasa AI Client je dokaz originalne implementacije i proveren je u odnosu na stanje svoje baze koda\u002Fdokumentacije od 26. jula 2026.",{},{"id":1005,"data":1006,"type":1012,"tunes":1013},"src-nist-profile",{"link":1007,"meta":1008},"https:\u002F\u002Fnvlpubs.nist.gov\u002Fnistpubs\u002Fai\u002FNIST.AI.600-1.pdf",{"image":1009,"title":1010,"description":1011},{"url":325},"NIST AI 600-1 — Profil generativne veštačke inteligencije","NIST-ov profil generativne AI, uključujući definiciju generativne AI i eksplicitnu razliku između pitanja na nivou modela, sistema, aplikacije i slučaja upotrebe.","linkTool",{},{"id":1015,"data":1016,"type":1012,"tunes":1022},"src-nist-model",{"link":1017,"meta":1018},"https:\u002F\u002Fcsrc.nist.gov\u002Fglossary\u002Fterm\u002Fartificial_intelligence_model",{"image":1019,"title":1020,"description":1021},{"url":325},"NIST — Model veštačke inteligencije","Aktuelna NIST-ova definicija iz glosara za AI model kao komponentu informacionog sistema koja proizvodi izlaze iz ulaza koristeći AI tehnike.",{},{"id":1024,"data":1025,"type":1012,"tunes":1031},"src-nist-system",{"link":1026,"meta":1027},"https:\u002F\u002Fcsrc.nist.gov\u002Fglossary\u002Fterm\u002Fartificial_intelligence_system",{"image":1028,"title":1029,"description":1030},{"url":325},"NIST — Sistem veštačke inteligencije","Aktuelna NIST-ova definicija iz glosara koja pokazuje da AI sistem može uključivati podatkovne sisteme, softver, hardver, aplikacije, alate ili uslužne programe koji koriste AI.",{},{"id":1033,"data":1034,"type":1012,"tunes":1040},"src-rag-paper",{"link":1035,"meta":1036},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2005.11401",{"image":1037,"title":1038,"description":1039},{"url":325},"Lewis i sar. — Generisanje uz pomoć pretrage za NLP zadatke intenzivne znanjem","Rad iz 2020. godine koji uvodi RAG formulaciju koja kombinuje generativni model sa pronađenom neparametarskom memorijom.",{},{"id":1042,"data":1043,"type":1012,"tunes":1049},"src-openai-file-search",{"link":1044,"meta":1045},"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Ftools-file-search",{"image":1046,"title":1047,"description":1048},{"url":325},"OpenAI — Pretraga fajlova","Aktuelna zvanična dokumentacija za hostovanu pretragu fajlova u Responses API-ju korišćenjem baza znanja sa otpremljenim fajlovima, semantičke pretrage i pretrage po ključnim rečima.",{},{"id":1051,"data":1052,"type":1012,"tunes":1058},"src-openai-functions",{"link":1053,"meta":1054},"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Ffunction-calling",{"image":1055,"title":1056,"description":1057},{"url":325},"OpenAI — Pozivanje funkcija","Aktuelna zvanična dokumentacija koja opisuje pozivanje alata\u002Ffunkcija kao interfejs između modela i eksternih sistema, podataka i radnji.",{},{"id":1060,"data":1061,"type":1012,"tunes":1067},"src-anthropic-context",{"link":1062,"meta":1063},"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Feffective-context-engineering-for-ai-agents",{"image":1064,"title":1065,"description":1066},{"url":325},"Anthropic — Efikasno inženjerstvo konteksta za AI agente","Inženjerske smernice koje definišu kontekst kao skup tokena dostupnih tokom LLM uzorkovanja i objašnjavaju zašto je izbor konteksta problem ograničenih resursa.",{},"2.31","Generativna AI je više od modela. Saznajte kako se modeli, pretraga, alati, kontekst, okruženja i aplikacije uklapaju u produkcione AI sisteme.","\u002Fuploads\u002F2026\u002F10\u002Fgenerative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing-1791475411822-pp0dvz.webp","generative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing-1791475411822-pp0dvz","PUBLISHED","2026-10-08T12:00:00.000Z","2026-10-08T16:00:39.350Z","2026-10-08T16:10:51.437Z",{"en":1077,"de":1078,"sr":1079,"es":1080,"fr":1081,"it":1082,"ru":1083,"zh":1084},"\u002Fblog\u002Fgenerative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing","\u002Fde\u002Fblog\u002Fgenerative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing","\u002Fsr\u002Fblog\u002Fgenerative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing","\u002Fes\u002Fblog\u002Fgenerative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing","\u002Ffr\u002Fblog\u002Fgenerative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing","\u002Fit\u002Fblog\u002Fgenerative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing","\u002Fru\u002Fblog\u002Fgenerative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing","\u002Fzh\u002Fblog\u002Fgenerative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing",[1086,1090,1094,1098,1102],{"id":1087,"name":1088,"slug":1089},84,"Politike i granice podataka","policy-and-data",{"id":1091,"name":1092,"slug":1093},57,"Granice podataka","data-boundaries",{"id":1095,"name":1096,"slug":1097},80,"Pristup i identitet","access-and-identity",{"id":1099,"name":1100,"slug":1101},68,"Rizici, kontrole i dokazi","risks-and-controls",{"id":1103,"name":1104,"slug":1105},54,"Model prijetnji","threat-model",{"id":1107,"login":1108,"email":1109,"displayName":1110},"20","rooth8233","aleksandar@stajic.de","Aleksandar Stajić",[1112,1814],{"lang":1113,"title":1114,"content":1115,"contentJson":1116,"excerpt":1813},"en","Generative AI Explained: Models, Retrieval, Tools and Applications Are Not the Same Thing","{\"time\":1791475413504,\"blocks\":[{\"id\":\"intro\",\"type\":\"paragraph\",\"data\":{\"text\":\"Generative AI is not one component. A production generative AI system usually combines a generative model with application code that supplies instructions and context, retrieves external knowledge when needed, exposes tools for reading or changing external systems, manages runtime state and permissions, and turns the result into a usable product. Treating the model, retrieval, tools, context, runtime, and application as the same thing hides the boundaries that determine freshness, security, reliability, cost, and control.\"},\"tunes\":{}},{\"id\":\"direct\",\"type\":\"callout\",\"data\":{\"variant\":\"info\",\"title\":\"Direct answer\",\"body\":\"\u003Cstrong>The model generates; retrieval finds external evidence; tools access data or perform actions; context is what the model can see for the current inference; the runtime coordinates execution; the application owns product rules, state, permissions, persistence, and user experience.\u003C\u002Fstrong> These layers can be packaged together by a vendor, but their responsibilities remain different.\"},\"tunes\":{}},{\"id\":\"scope-note\",\"type\":\"callout\",\"data\":{\"variant\":\"note\",\"title\":\"Terminology and version note\",\"body\":\"This article defines durable architectural responsibilities rather than one vendor stack. Current implementation examples were re-checked on \u003Cstrong>8 October 2026\u003C\u002Fstrong>. Vendor APIs and product names can change; the responsibility boundaries are more stable than any individual SDK or endpoint.\"},\"tunes\":{}},{\"id\":\"toc\",\"type\":\"tableOfContents\",\"data\":{\"title\":\"Contents\",\"minLevel\":2,\"maxLevel\":3},\"tunes\":{}},{\"id\":\"h-meaning\",\"type\":\"header\",\"data\":{\"text\":\"What does “generative AI” actually mean?\",\"level\":2},\"tunes\":{}},{\"id\":\"p-meaning-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"At the model level, generative AI refers to AI models that generate derived synthetic content such as text, images, audio, video, code, or other digital output. NIST AI 600-1 uses this model-oriented meaning and separately discusses risks at model, system, application, and use-case levels.\"},\"tunes\":{}},{\"id\":\"p-meaning-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"That distinction matters because an AI model is not the same thing as the complete AI system. NIST's current glossary defines an AI model as a component that produces outputs from inputs using computational, statistical, or machine-learning techniques, while an AI system can include software, hardware, applications, tools, or utilities that operate using AI.\"},\"tunes\":{}},{\"id\":\"model-system-rule\",\"type\":\"callout\",\"data\":{\"variant\":\"success\",\"title\":\"A useful boundary\",\"body\":\"\u003Cstrong>Generative model ≠ generative AI application.\u003C\u002Fstrong>\u003Cbr>A model is one computational component. A usable AI product is a system built around that component.\"},\"tunes\":{}},{\"id\":\"h-simple\",\"type\":\"header\",\"data\":{\"text\":\"The simplest useful model of a generative AI system\",\"level\":2},\"tunes\":{}},{\"id\":\"p-simple-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"For a first mental model, imagine a company assistant answering: “Can this customer receive a refund today?” A useful answer may require several different responsibilities. The language model can interpret the question and write the explanation, but the current order state may come from a database tool, the refund policy may come from document retrieval, permissions may be enforced by the application, and the final action may require a controlled API call.\"},\"tunes\":{}},{\"id\":\"simple-flow\",\"type\":\"processFlow\",\"data\":{\"title\":\"One common execution path\",\"orientation\":\"auto\",\"steps\":[{\"label\":\"1. User request\",\"description\":\"The application receives a natural-language question or task.\"},{\"label\":\"2. Application policy and state\",\"description\":\"Identity, tenant, permissions, current workflow state, and product rules define what the request is allowed to do.\"},{\"label\":\"3. Retrieval or direct data access\",\"description\":\"The system obtains external evidence or current facts when model knowledge is insufficient.\"},{\"label\":\"4. Context construction\",\"description\":\"Instructions, user input, selected evidence, relevant state, and tool definitions are assembled for the model.\"},{\"label\":\"5. Model inference\",\"description\":\"The generative model interprets the supplied context and produces text, structured output, or a tool request.\"},{\"label\":\"6. Tool execution when needed\",\"description\":\"The runtime or application validates and executes approved tool calls outside the model.\"},{\"label\":\"7. Observation and continuation\",\"description\":\"Tool results can return to the model as new context for another inference step.\"},{\"label\":\"8. Validation and product output\",\"description\":\"The application validates the result, records required state or audit data, and presents or executes the final outcome.\"}]},\"tunes\":{}},{\"id\":\"p-simple-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Real systems do not always follow this sequence exactly. Retrieval can happen before the first model call, tools can be selected during an agent loop, deterministic application logic can bypass the model entirely, and validation can occur at several stages. The point is to separate responsibilities, not to impose one universal workflow.\"},\"tunes\":{}},{\"id\":\"h-boundaries\",\"type\":\"header\",\"data\":{\"text\":\"The six boundaries that matter\",\"level\":2},\"tunes\":{}},{\"id\":\"boundary-comparison\",\"type\":\"comparison\",\"data\":{\"title\":\"Six responsibilities inside one AI product\",\"layout\":\"table\",\"columns\":[{\"id\":\"job\",\"label\":\"Primary job\"},{\"id\":\"input\",\"label\":\"Typical inputs\"},{\"id\":\"not\",\"label\":\"Not the same as\"}],\"rows\":[{\"id\":\"model\",\"label\":\"Model\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"retrieval\",\"label\":\"Retrieval\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"tools\",\"label\":\"Tools\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"context\",\"label\":\"Context\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"runtime\",\"label\":\"Runtime \u002F orchestrator\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"application\",\"label\":\"Application\",\"values\":[\"\",\"\",\"\"]}]},\"tunes\":{}},{\"id\":\"h-model\",\"type\":\"header\",\"data\":{\"text\":\"1. The model: generation is its core responsibility\",\"level\":2},\"tunes\":{}},{\"id\":\"p-model-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A generative model maps supplied inputs to generated outputs. For a language model, that can include natural-language text, structured JSON, code, classifications, summaries, plans, or tool-call arguments. Multimodal generative models can work with additional input and output types.\"},\"tunes\":{}},{\"id\":\"p-model-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The model can contain substantial learned knowledge in its parameters, but parameterized knowledge is not a live database. The model does not automatically know a document created five minutes ago, the current stock level, a private customer record, or the state of an application unless that information is supplied through the current input path.\"},\"tunes\":{}},{\"id\":\"p-model-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"This is why changing the model does not automatically solve stale knowledge, missing permissions, broken retrieval, incorrect state ownership, or unsafe tool execution. Those failures often belong to other layers.\"},\"tunes\":{}},{\"id\":\"h-retrieval\",\"type\":\"header\",\"data\":{\"text\":\"2. Retrieval: finding external evidence is a separate operation\",\"level\":2},\"tunes\":{}},{\"id\":\"p-retrieval-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Retrieval selects information from an external source before or during generation. The 2020 Retrieval-Augmented Generation work by Lewis et al. made the separation explicit by combining a parametric generative model with retrieved non-parametric memory. Modern production systems use many retrieval variants, but the architectural idea remains: useful evidence can be fetched at inference time instead of relying only on what the model learned during training.\"},\"tunes\":{}},{\"id\":\"p-retrieval-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Retrieval can use lexical search, embeddings, vector search, hybrid search, SQL, knowledge graphs, metadata filters, APIs, or other selection mechanisms. A vector database is therefore one possible retrieval component, not the definition of RAG.\"},\"tunes\":{}},{\"id\":\"retrieval-rule\",\"type\":\"callout\",\"data\":{\"variant\":\"warning\",\"title\":\"Relevance is not authority\",\"body\":\"A retrieved passage can be highly relevant and still be stale, unauthorized, from the wrong version, or insufficient to support a claim. Retrieval quality and evidence quality must be evaluated separately.\"},\"tunes\":{}},{\"id\":\"ref-rag\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works\",\"title\":\"What Is RAG? The Simplest Explanation of How It Works\",\"excerpt\":\"The canonical plain-English explanation of retrieval-augmented generation, including the separation between LLM, knowledge, state, memory and tools.\",\"ctaLabel\":\"Read the RAG foundation\"},\"tunes\":{}},{\"id\":\"h-tools\",\"type\":\"header\",\"data\":{\"text\":\"3. Tools: access and action are not model knowledge\",\"level\":2},\"tunes\":{}},{\"id\":\"p-tools-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A tool is an interface through which an AI runtime can request functionality outside the model. A tool can query a database, search the web, read a file, calculate a value, call an internal service, create a ticket, send a message, modify a record, or trigger another controlled operation.\"},\"tunes\":{}},{\"id\":\"p-tools-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"OpenAI's current function-calling documentation makes this boundary explicit: function calling lets models interface with external systems and access data or actions provided by the application. The model can propose or select a call, but the external system performs the real operation.\"},\"tunes\":{}},{\"id\":\"p-tools-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"Tool use therefore creates two separate questions: Can the model request this capability? and Will the application authorize and execute it? A production system should not confuse model intent with permission to cause a side effect.\"},\"tunes\":{}},{\"id\":\"tool-rule\",\"type\":\"callout\",\"data\":{\"variant\":\"success\",\"title\":\"Model intent is not execution authority\",\"body\":\"A model can emit a valid tool request and still be denied. Authorization, argument validation, rate limits, transaction rules, audit requirements, and rollback belong outside the model.\"},\"tunes\":{}},{\"id\":\"h-context\",\"type\":\"header\",\"data\":{\"text\":\"4. Context: what the model can see right now\",\"level\":2},\"tunes\":{}},{\"id\":\"p-context-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Context is the information available to the model for a particular inference step. Anthropic's context-engineering guidance describes context as the set of tokens included when sampling from an LLM. In practice, that set can contain system instructions, user messages, conversation history, retrieved evidence, tool definitions, tool results, memory summaries, and selected application state.\"},\"tunes\":{}},{\"id\":\"p-context-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Context is therefore neither the complete knowledge base nor long-term memory. A company may store ten million documents while only a handful of passages enter one model call. A runtime may persist a year of conversation history while exposing only the pieces needed for the current task.\"},\"tunes\":{}},{\"id\":\"p-context-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The context window also creates an engineering constraint. Adding more text does not guarantee a better answer; irrelevant, stale, contradictory, or low-authority information can dilute the evidence that actually matters.\"},\"tunes\":{}},{\"id\":\"h-runtime\",\"type\":\"header\",\"data\":{\"text\":\"5. Runtime and orchestration: coordinating the loop\",\"level\":2},\"tunes\":{}},{\"id\":\"p-runtime-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The runtime or orchestration layer coordinates how the model participates in a task. Depending on the architecture, it can manage sessions, model requests, tool discovery, tool-call loops, retries, handoffs, streaming events, timeouts, checkpoints, compaction, or execution environments.\"},\"tunes\":{}},{\"id\":\"p-runtime-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Some runtimes are thin application code around a model API. Others are full agent harnesses. A managed vendor runtime can own part of the loop while the application still owns domain truth, authorization, business side effects, and product lifecycle.\"},\"tunes\":{}},{\"id\":\"p-runtime-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"This boundary is important because where the runtime runs and where inference runs are separate decisions. A locally running client or agent process can still call a remote model, while a remote application can call a model hosted on infrastructure under the organization's control.\"},\"tunes\":{}},{\"id\":\"h-application\",\"type\":\"header\",\"data\":{\"text\":\"6. The application: where AI becomes a product\",\"level\":2},\"tunes\":{}},{\"id\":\"p-app-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The application is the product boundary around the AI components. It owns the user experience, domain model, current state, identity, tenant scope, permissions, persistence, service integrations, validation, observability, billing or quota logic where relevant, and the rules that determine what the AI is allowed to see or do.\"},\"tunes\":{}},{\"id\":\"p-app-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"This is the layer that turns “a model can produce useful output” into “a system can deliver a reliable capability.” The same model can participate in a private research assistant, a support workflow, a code agent, or a commerce application because the surrounding application changes the data, tools, policies, state, and execution contract.\"},\"tunes\":{}},{\"id\":\"app-rule\",\"type\":\"callout\",\"data\":{\"variant\":\"info\",\"title\":\"The model is replaceable; the product boundary is not\",\"body\":\"Provider and model substitution can be an architectural goal. The application's authoritative state, permissions, domain rules, audit trail, and user contract cannot simply be delegated to whichever model is currently selected.\"},\"tunes\":{}},{\"id\":\"h-work-together\",\"type\":\"header\",\"data\":{\"text\":\"How the parts work together in a real request\",\"level\":2},\"tunes\":{}},{\"id\":\"p-together-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Consider a support assistant asked: “Refund order 4711 if it is still eligible, and explain why.” The request combines knowledge, current state, authorization, reasoning, and a side effect.\"},\"tunes\":{}},{\"id\":\"support-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Need\",\"Correct layer\",\"Why\"],[\"Refund policy\",\"Retrieval\",\"The system must find the current applicable policy and preserve its provenance.\"],[\"Order 4711 status\",\"Direct data\u002Ftool access\",\"The current order record is volatile authoritative state, not something to guess from model knowledge.\"],[\"User's authority to refund\",\"Application \u002F authorization\",\"Permissions must be enforced independently of what the model asks for.\"],[\"Interpret policy against order facts\",\"Model + context\",\"The model can reason over the policy evidence and current order state supplied to it.\"],[\"Execute refund\",\"Tool + application transaction rules\",\"A controlled external operation changes real state.\"],[\"Explain outcome\",\"Model\",\"The model can generate the user-facing explanation from validated results.\"],[\"Audit what happened\",\"Application \u002F runtime\",\"The system records evidence, calls, decisions, side effects, and errors as required.\"]]},\"tunes\":{}},{\"id\":\"p-together-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"If the assistant only has the language model, it can discuss refunds but cannot safely know whether order 4711 is currently eligible or perform the transaction. If it only has retrieval, it may find the policy but still lack live order state. If it has tools without application authorization, it may become capable but unsafe. Reliability comes from composing the layers with explicit ownership.\"},\"tunes\":{}},{\"id\":\"h-configs\",\"type\":\"header\",\"data\":{\"text\":\"Different AI products use different combinations\",\"level\":2},\"tunes\":{}},{\"id\":\"config-comparison\",\"type\":\"comparison\",\"data\":{\"title\":\"The presence of a model does not define the whole architecture\",\"layout\":\"table\",\"columns\":[{\"id\":\"retrieval\",\"label\":\"Retrieval\"},{\"id\":\"tools\",\"label\":\"Tools\"},{\"id\":\"state\",\"label\":\"Authoritative state\"},{\"id\":\"result\",\"label\":\"Typical capability\"}],\"rows\":[{\"id\":\"bare\",\"label\":\"Model-only assistant\",\"values\":[\"\",\"\",\"\",\"\"]},{\"id\":\"rag\",\"label\":\"Retrieval-grounded assistant\",\"values\":[\"\",\"\",\"\",\"\"]},{\"id\":\"tool\",\"label\":\"Tool-using assistant\",\"values\":[\"\",\"\",\"\",\"\"]},{\"id\":\"agent\",\"label\":\"Agentic application\",\"values\":[\"\",\"\",\"\",\"\"]}]},\"tunes\":{}},{\"id\":\"p-configs-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"These are architecture patterns, not maturity rankings. A model-only feature can be the correct design when the task needs no external facts or actions. Adding retrieval, tools, memory, or an agent loop is justified only when the task requires those capabilities.\"},\"tunes\":{}},{\"id\":\"h-implementation\",\"type\":\"header\",\"data\":{\"text\":\"Implementation evidence: Aaasaasa AI Client\",\"level\":2},\"tunes\":{}},{\"id\":\"implementation-scope\",\"type\":\"callout\",\"data\":{\"variant\":\"note\",\"title\":\"Primary implementation evidence\",\"body\":\"The following section describes an implementation I built and reviewed against the Aaasaasa AI Client codebase and architecture documentation as of \u003Cstrong>26 July 2026\u003C\u002Fstrong>. It is evidence for the usefulness of these boundaries, not a claim that one implementation is a universal standard or a commercially deployed enterprise product.\"},\"tunes\":{}},{\"id\":\"p-impl-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Aaasaasa AI Client is a local-first desktop AI workspace built with Nuxt 4, Electron and TypeScript. Its AI Hub deliberately separates agent\u002Fclient, provider, model, runtime location, permissions, and web client instead of treating them as one “AI” setting.\"},\"tunes\":{}},{\"id\":\"p-impl-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"That separation creates concrete behavior. Direct Chat can talk to models without filesystem or shell tools. A Codex agent can use a selected workspace and permission profile. Ollama can provide direct local inference, while LM Studio and configurable OpenAI-compatible endpoints represent other provider paths. A locally running Codex process can still use a cloud model, so the UI and architecture do not equate local runtime with local inference.\"},\"tunes\":{}},{\"id\":\"p-impl-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The implementation also contains Qdrant\u002Fvector support, document-extraction capabilities and an authenticated directory MCP broker. Those components illustrate another boundary: retrieval infrastructure and tool access can live in the same product without becoming properties of the model itself.\"},\"tunes\":{}},{\"id\":\"impl-map\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"A01 concept\",\"Aaasaasa AI Client implementation evidence\"],[\"Model\",\"A provider-specific model identifier is selected separately from provider and runtime.\"],[\"Provider\",\"Ollama, LM Studio, OpenAI-compatible services and other provider paths are represented separately.\"],[\"Runtime\",\"Local or remote agent\u002Fruntime location is tracked independently of the model.\"],[\"Tools \u002F access\",\"Direct Chat has no filesystem or shell tools; controlled directory access is brokered separately.\"],[\"Permissions\",\"Workspace permission profiles are application\u002Fsession policy, not model capability.\"],[\"Retrieval infrastructure\",\"Vector support and document extraction exist as data\u002Fretrieval capabilities rather than model features.\"],[\"Application\",\"The Electron\u002FNuxt product coordinates UI, credentials, providers, runtime discovery, permissions, tools and model interaction.\"]]},\"tunes\":{}},{\"id\":\"impl-lesson\",\"type\":\"callout\",\"data\":{\"variant\":\"success\",\"title\":\"Implementation lesson\",\"body\":\"The architecture became easier to reason about once \u003Cstrong>model, provider, runtime, permissions, tools, data and client\u003C\u002Fstrong> stopped being represented as one configuration choice. The distinction is operational: it determines what can run locally, what can access files, what may call paid cloud inference, and which layer owns authorization.\"},\"tunes\":{}},{\"id\":\"h-errors\",\"type\":\"header\",\"data\":{\"text\":\"Common category errors\",\"level\":2},\"tunes\":{}},{\"id\":\"errors-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Category error\",\"What is actually happening\"],[\"“The AI knows our documents.”\",\"The application or retrieval layer makes selected document content available to the model.\"],[\"“RAG is our vector database.”\",\"The vector database can be one index or store used by a retrieval pipeline; RAG is the retrieval-plus-generation pattern.\"],[\"“The model called our CRM.”\",\"The model produced a tool request; the runtime\u002Fapplication authorized and executed the external call.\"],[\"“It is local AI because the desktop agent runs locally.”\",\"Runtime location and inference location are separate. A local runtime can still invoke a remote model.\"],[\"“The model has permission to edit files.”\",\"The application\u002Fruntime grants a tool capability under a permission policy; permission is not an intrinsic model property.\"],[\"“More context means more knowledge.”\",\"Context is the finite input made available for one inference. Larger context can contain more noise, conflict or stale information.\"],[\"“The chatbot is the AI architecture.”\",\"The chat UI is one interface. The system can also include identity, state, retrieval, tools, runtime, validation, persistence and observability.\"]]},\"tunes\":{}},{\"id\":\"h-failures\",\"type\":\"header\",\"data\":{\"text\":\"Failure modes when the boundaries collapse\",\"level\":2},\"tunes\":{}},{\"id\":\"p-failure-intro\",\"type\":\"paragraph\",\"data\":{\"text\":\"Boundary mistakes are not merely terminology problems. They create distinct production failures that require different fixes.\"},\"tunes\":{}},{\"id\":\"failure-comparison\",\"type\":\"comparison\",\"data\":{\"title\":\"Diagnose the failing layer before replacing the model\",\"layout\":\"table\",\"columns\":[{\"id\":\"symptom\",\"label\":\"Symptom\"},{\"id\":\"likely\",\"label\":\"Likely boundary problem\"},{\"id\":\"fix\",\"label\":\"First architectural check\"}],\"rows\":[{\"id\":\"stale\",\"label\":\"Stale answer\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"missing\",\"label\":\"Missing company fact\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"unsafe\",\"label\":\"Unsafe side effect\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"noise\",\"label\":\"Confused answer with lots of supplied text\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"route\",\"label\":\"Unexpected cloud use\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"loop\",\"label\":\"Agent stalls or repeats\",\"values\":[\"\",\"\",\"\"]}]},\"tunes\":{}},{\"id\":\"h-version\",\"type\":\"header\",\"data\":{\"text\":\"What is stable and what is version-sensitive?\",\"level\":2},\"tunes\":{}},{\"id\":\"p-version-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The architectural distinctions in this article are intentionally vendor-neutral. The current examples below are implementation facts that should be re-checked when APIs evolve.\"},\"tunes\":{}},{\"id\":\"version-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Area\",\"Stable architectural idea\",\"Verified current example on 8 Oct 2026\"],[\"AI model vs system\",\"A model is a component inside a broader system\",\"NIST's current glossary separately defines AI model and AI system.\"],[\"RAG\",\"Generation can be conditioned on retrieved external information\",\"The Lewis et al. 2020 formulation remains the foundational reference; production retrieval methods now extend far beyond one dense index design.\"],[\"Hosted retrieval\",\"Retrieval can be exposed as a managed tool\",\"OpenAI File Search is currently a Responses API tool that searches uploaded-file knowledge bases using semantic and keyword retrieval.\"],[\"Function\u002Ftool calling\",\"A model can request application-defined external capabilities\",\"OpenAI currently documents function calling as an interface to external systems, data and actions.\"],[\"Context engineering\",\"Model behavior depends on the finite information supplied for the current inference\",\"Anthropic's current engineering guidance defines context as the token set included when sampling from the LLM and focuses on curating that set.\"],[\"Vendor APIs\",\"SDKs, tool names, endpoint shapes and supported features change\",\"Treat vendor documentation as version-sensitive even when the responsibility boundary remains stable.\"]]},\"tunes\":{}},{\"id\":\"p-version-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"A source-of-truth article should therefore preserve both levels: stable concepts for architecture, and dated evidence for current implementations. Mixing the two makes an article age unnecessarily fast.\"},\"tunes\":{}},{\"id\":\"h-test\",\"type\":\"header\",\"data\":{\"text\":\"The AI component-boundary test\",\"level\":2},\"tunes\":{}},{\"id\":\"p-test-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"When evaluating an AI feature, ask the following questions in order. The answers reveal which components the system actually has and which responsibilities are still implicit.\"},\"tunes\":{}},{\"id\":\"boundary-test\",\"type\":\"processFlow\",\"data\":{\"title\":\"Seven questions for a production design\",\"orientation\":\"auto\",\"steps\":[{\"label\":\"1. What generates the output?\",\"description\":\"Identify the exact model and the modalities or structured outputs it provides.\"},{\"label\":\"2. What facts are authoritative outside the model?\",\"description\":\"Identify documents, databases, APIs, current state and other sources of truth.\"},{\"label\":\"3. How is relevant information selected?\",\"description\":\"Separate direct lookup, search, retrieval, ranking and context construction.\"},{\"label\":\"4. What can cause real side effects?\",\"description\":\"List tools and external actions, then identify who validates and authorizes them.\"},{\"label\":\"5. What reaches the model as context?\",\"description\":\"Make instructions, evidence, state, history, memory and tool definitions explicit.\"},{\"label\":\"6. Who owns the loop?\",\"description\":\"Identify the runtime or harness that manages calls, events, retries, tool loops and sessions.\"},{\"label\":\"7. What remains the application's responsibility?\",\"description\":\"Make identity, permissions, domain state, validation, persistence, observability and UX explicit.\"}]},\"tunes\":{}},{\"id\":\"h-not\",\"type\":\"header\",\"data\":{\"text\":\"What generative AI is not\",\"level\":2},\"tunes\":{}},{\"id\":\"p-not-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Generative AI is not synonymous with an LLM, even though LLMs are a major class of generative model. It is also not synonymous with RAG, a vector database, an agent, a tool protocol, a chatbot UI, or an application.\"},\"tunes\":{}},{\"id\":\"p-not-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Those concepts can be connected, but each answers a different architectural question. An LLM asks how language output is produced. Retrieval asks where external evidence comes from. Tools ask how external capabilities are exposed. Context asks what the model can see. Runtime asks how execution is coordinated. The application asks how the capability becomes a controlled product.\"},\"tunes\":{}},{\"id\":\"remember\",\"type\":\"callout\",\"data\":{\"variant\":\"success\",\"title\":\"If you remember only one model\",\"body\":\"\u003Cstrong>Model = generate.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Retrieval = find evidence.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Tools = read or act outside the model.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Context = what the model sees now.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Runtime = coordinate execution.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Application = own the product, state, rules and permissions.\u003C\u002Fstrong>\"},\"tunes\":{}},{\"id\":\"h-next\",\"type\":\"header\",\"data\":{\"text\":\"Where to go next in the knowledge graph\",\"level\":2},\"tunes\":{}},{\"id\":\"p-next-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Once these boundaries are clear, deeper topics become easier to place. RAG belongs in retrieval and context construction. Retrieval Trigger decides when external evidence is required. Agent memory concerns what persists across time. Tool calling and MCP belong to capability access. Agent harnesses belong to runtime orchestration. RBAC, tenant isolation and domain authorization belong to the application and platform security boundary.\"},\"tunes\":{}},{\"id\":\"ref-data\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fwhere-does-an-llm-get-its-data-rag-data-sources-in-python\",\"title\":\"Where Does an LLM Get Its Data? RAG Data Sources in Python\",\"excerpt\":\"A practical continuation showing how files, SQL, APIs, full-text search, embeddings and context assembly connect external data to an LLM.\",\"ctaLabel\":\"See the data path in code\"},\"tunes\":{}},{\"id\":\"ref-trigger\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fwhen-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger\",\"title\":\"When Should an AI Stop Trusting Its Own Knowledge? — The Retrieval Trigger\",\"excerpt\":\"A decision model for when an AI system should stop relying only on model knowledge and obtain external evidence.\",\"ctaLabel\":\"Read the retrieval decision model\"},\"tunes\":{}},{\"id\":\"h-limit\",\"type\":\"header\",\"data\":{\"text\":\"Limitations\",\"level\":2},\"tunes\":{}},{\"id\":\"p-limit-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The six-layer model is a responsibility map, not a requirement that every product deploy six separate services. A small application may implement context construction, retrieval and orchestration inside one process. A managed platform may bundle several responsibilities behind one API. Physical deployment can be combined while semantic ownership remains distinct.\"},\"tunes\":{}},{\"id\":\"p-limit-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Terminology also varies across vendors and research. “Agent,” “runtime,” “memory,” “tool,” “connector,” and “context” can be defined differently. The definitions here are chosen to make operational ownership and failure diagnosis explicit rather than to claim that every framework uses identical vocabulary.\"},\"tunes\":{}},{\"id\":\"p-limit-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The Aaasaasa AI Client section documents one implementation pattern. It demonstrates that explicit boundaries are practical, but it does not prove that the same component layout is optimal for every AI product.\"},\"tunes\":{}},{\"id\":\"h-change\",\"type\":\"header\",\"data\":{\"text\":\"What would change this answer?\",\"level\":2},\"tunes\":{}},{\"id\":\"p-change-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The responsibility map would need revision if model architectures themselves began to own authoritative external state, permissions, durable transactional side effects, and verifiable source access as intrinsic properties rather than capabilities supplied by a surrounding system. Current production architectures do not make that a safe general assumption.\"},\"tunes\":{}},{\"id\":\"p-change-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Individual implementation examples will change much sooner. Hosted retrieval tools, agent APIs, MCP integrations, context-management features and provider capabilities evolve quickly. Those details should be updated without collapsing the underlying distinctions between generation, evidence, capability access, context, execution and application control.\"},\"tunes\":{}},{\"id\":\"h-conclusion\",\"type\":\"header\",\"data\":{\"text\":\"Conclusion\",\"level\":2},\"tunes\":{}},{\"id\":\"p-conclusion-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Generative AI becomes easier to design once “the AI” stops being treated as one black box. The model is the generative component, not the complete product. Retrieval provides external evidence. Tools expose capabilities. Context carries selected information into the current inference. The runtime coordinates execution. The application owns the authoritative product boundary.\"},\"tunes\":{}},{\"id\":\"p-conclusion-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"That separation is useful for more than explanation. It tells engineers where stale facts originate, where authorization belongs, why a local runtime can still use cloud inference, why RAG does not equal a vector database, why tool calls require validation, and why changing the model cannot repair every system failure.\"},\"tunes\":{}},{\"id\":\"p-conclusion-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The durable architecture question is therefore not “Which AI model are we using?” It is: Which responsibility does each component own, what evidence crosses each boundary, and which layer is allowed to change real state?\"},\"tunes\":{}},{\"id\":\"h-faq\",\"type\":\"header\",\"data\":{\"text\":\"FAQ\",\"level\":2},\"tunes\":{}},{\"id\":\"faq\",\"type\":\"faq\",\"data\":{\"title\":\"Generative AI system boundaries\",\"items\":[{\"id\":\"faq1\",\"question\":\"Is generative AI the same as an LLM?\",\"answer\":\"No. An LLM is one type of generative model. Generative AI also includes other modalities, and a production generative AI system can include retrieval, tools, runtime logic, application state, permissions, persistence and user interfaces around the model.\"},{\"id\":\"faq2\",\"question\":\"Is RAG part of the model?\",\"answer\":\"Usually no. RAG is an application\u002Fsystem pattern that retrieves external information and supplies selected evidence to the model. Some platforms package retrieval tightly with model APIs, but the responsibility remains distinct.\"},{\"id\":\"faq3\",\"question\":\"Is a vector database required for RAG?\",\"answer\":\"No. RAG can use vector search, lexical search, hybrid retrieval, SQL, APIs, knowledge graphs or other methods. The defining property is retrieval of external information for generation, not one storage technology.\"},{\"id\":\"faq4\",\"question\":\"Are tools the same as context?\",\"answer\":\"No. A tool is an external capability. Its definition may be represented in context, and its result may later enter context, but the actual capability executes outside the model.\"},{\"id\":\"faq5\",\"question\":\"Does running an AI client locally mean the model is local?\",\"answer\":\"No. Runtime location and inference location are separate. A local desktop application or agent can call a remote model, while a remote application can call an internally hosted model.\"},{\"id\":\"faq6\",\"question\":\"Who should enforce permissions for AI tools?\",\"answer\":\"The application or runtime security boundary should enforce authorization. A model can request an operation, but model intent should never be treated as sufficient execution authority.\"},{\"id\":\"faq7\",\"question\":\"Where does current application state belong?\",\"answer\":\"Authoritative volatile state should normally remain in the application or domain system that owns it. The AI can receive the relevant state through controlled context or tool access when needed.\"}]},\"tunes\":{}},{\"id\":\"h-glossary\",\"type\":\"header\",\"data\":{\"text\":\"Glossary\",\"level\":2},\"tunes\":{}},{\"id\":\"glossary\",\"type\":\"glossary\",\"data\":{\"title\":\"Core terms\",\"entries\":[{\"term\":\"Generative model\",\"definition\":\"An AI model designed to generate derived synthetic content such as text, images, audio, video, code or structured output.\",\"anchor\":\"generative-model\"},{\"term\":\"Retrieval\",\"definition\":\"The process of selecting relevant information from an external source or store for the current task.\",\"anchor\":\"retrieval\"},{\"term\":\"RAG\",\"definition\":\"Retrieval-Augmented Generation: a pattern in which retrieved external information is supplied to a generative model to improve the current output.\",\"anchor\":\"rag\"},{\"term\":\"Tool\",\"definition\":\"A capability exposed to an AI runtime for reading data, calculating, searching, or performing an external action.\",\"anchor\":\"tool\"},{\"term\":\"Context\",\"definition\":\"The information available to the model for a particular inference step.\",\"anchor\":\"context\"},{\"term\":\"Runtime \u002F orchestrator\",\"definition\":\"The software layer that coordinates model calls, tool calls, task loops, sessions, retries, events or execution environments.\",\"anchor\":\"runtime-orchestrator\"},{\"term\":\"Application\",\"definition\":\"The product and domain layer that owns user interaction, authoritative state, permissions, validation, persistence and business behavior.\",\"anchor\":\"application\"},{\"term\":\"Provider\",\"definition\":\"The service or runtime that exposes access to one or more models; provider identity and model identity are separate concerns.\",\"anchor\":\"provider\"}]},\"tunes\":{}},{\"id\":\"h-sources\",\"type\":\"header\",\"data\":{\"text\":\"Primary sources and implementation evidence\",\"level\":2},\"tunes\":{}},{\"id\":\"p-sources-note\",\"type\":\"paragraph\",\"data\":{\"text\":\"Stable definitions below are anchored in standards\u002Fresearch; fast-moving implementation examples use current official engineering documentation. Aaasaasa AI Client is original implementation evidence and was checked against its codebase\u002Fdocumentation state dated 26 July 2026.\"},\"tunes\":{}},{\"id\":\"src-nist-profile\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fnvlpubs.nist.gov\u002Fnistpubs\u002Fai\u002FNIST.AI.600-1.pdf\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"NIST AI 600-1 — Generative Artificial Intelligence Profile\",\"description\":\"NIST's Generative AI profile, including the generative-AI definition and explicit distinction between model-, system-, application- and use-case-level concerns.\"}},\"tunes\":{}},{\"id\":\"src-nist-model\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fcsrc.nist.gov\u002Fglossary\u002Fterm\u002Fartificial_intelligence_model\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"NIST — Artificial Intelligence Model\",\"description\":\"Current NIST glossary definition of an AI model as a component of an information system that produces outputs from inputs using AI techniques.\"}},\"tunes\":{}},{\"id\":\"src-nist-system\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fcsrc.nist.gov\u002Fglossary\u002Fterm\u002Fartificial_intelligence_system\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"NIST — Artificial Intelligence System\",\"description\":\"Current NIST glossary definition showing that an AI system can include data systems, software, hardware, applications, tools or utilities using AI.\"}},\"tunes\":{}},{\"id\":\"src-rag-paper\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2005.11401\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Lewis et al. — Retrieval-Augmented Generation for Knowledge-Intensive NLP Tasks\",\"description\":\"The 2020 paper introducing the RAG formulation that combines a generative model with retrieved non-parametric memory.\"}},\"tunes\":{}},{\"id\":\"src-openai-file-search\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Ftools-file-search\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — File Search\",\"description\":\"Current official documentation for hosted file retrieval in the Responses API using uploaded-file knowledge bases, semantic search and keyword search.\"}},\"tunes\":{}},{\"id\":\"src-openai-functions\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Ffunction-calling\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — Function Calling\",\"description\":\"Current official documentation describing tool\u002Ffunction calling as the interface between models and external systems, data and actions.\"}},\"tunes\":{}},{\"id\":\"src-anthropic-context\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Feffective-context-engineering-for-ai-agents\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Anthropic — Effective Context Engineering for AI Agents\",\"description\":\"Engineering guidance defining context as the token set available during LLM sampling and explaining why context selection is a finite-resource problem.\"}},\"tunes\":{}}],\"version\":\"2.31.6\"}",{"time":1117,"blocks":1118,"version":1812},1791475413504,[1119,1123,1128,1133,1137,1141,1145,1149,1154,1158,1162,1191,1195,1199,1228,1232,1236,1240,1244,1248,1252,1256,1261,1268,1272,1276,1280,1284,1289,1293,1297,1301,1305,1309,1313,1317,1321,1325,1329,1333,1338,1342,1346,1380,1384,1388,1412,1416,1420,1425,1429,1433,1437,1463,1468,1472,1500,1504,1508,1538,1542,1546,1577,1581,1585,1589,1615,1619,1623,1627,1632,1636,1640,1647,1654,1658,1662,1666,1670,1674,1678,1682,1686,1690,1694,1698,1702,1728,1732,1755,1759,1763,1770,1777,1784,1791,1798,1805],{"id":215,"data":1120,"type":218,"tunes":1122},{"text":1121},"Generative AI is not one component. A production generative AI system usually combines a generative model with application code that supplies instructions and context, retrieves external knowledge when needed, exposes tools for reading or changing external systems, manages runtime state and permissions, and turns the result into a usable product. Treating the model, retrieval, tools, context, runtime, and application as the same thing hides the boundaries that determine freshness, security, reliability, cost, and control.",{},{"id":221,"data":1124,"type":226,"tunes":1127},{"body":1125,"title":1126,"variant":225},"\u003Cstrong>The model generates; retrieval finds external evidence; tools access data or perform actions; context is what the model can see for the current inference; the runtime coordinates execution; the application owns product rules, state, permissions, persistence, and user experience.\u003C\u002Fstrong> These layers can be packaged together by a vendor, but their responsibilities remain different.","Direct answer",{},{"id":229,"data":1129,"type":226,"tunes":1132},{"body":1130,"title":1131,"variant":233},"This article defines durable architectural responsibilities rather than one vendor stack. Current implementation examples were re-checked on \u003Cstrong>8 October 2026\u003C\u002Fstrong>. Vendor APIs and product names can change; the responsibility boundaries are more stable than any individual SDK or endpoint.","Terminology and version note",{},{"id":236,"data":1134,"type":241,"tunes":1136},{"title":1135,"maxLevel":239,"minLevel":240},"Contents",{},{"id":244,"data":1138,"type":42,"tunes":1140},{"text":1139,"level":240},"What does “generative AI” actually mean?",{},{"id":249,"data":1142,"type":218,"tunes":1144},{"text":1143},"At the model level, generative AI refers to AI models that generate derived synthetic content such as text, images, audio, video, code, or other digital output. NIST AI 600-1 uses this model-oriented meaning and separately discusses risks at model, system, application, and use-case levels.",{},{"id":254,"data":1146,"type":218,"tunes":1148},{"text":1147},"That distinction matters because an AI model is not the same thing as the complete AI system. NIST's current glossary defines an AI model as a component that produces outputs from inputs using computational, statistical, or machine-learning techniques, while an AI system can include software, hardware, applications, tools, or utilities that operate using AI.",{},{"id":259,"data":1150,"type":226,"tunes":1153},{"body":1151,"title":1152,"variant":263},"\u003Cstrong>Generative model ≠ generative AI application.\u003C\u002Fstrong>\u003Cbr>A model is one computational component. A usable AI product is a system built around that component.","A useful boundary",{},{"id":266,"data":1155,"type":42,"tunes":1157},{"text":1156,"level":240},"The simplest useful model of a generative AI system",{},{"id":271,"data":1159,"type":218,"tunes":1161},{"text":1160},"For a first mental model, imagine a company assistant answering: “Can this customer receive a refund today?” A useful answer may require several different responsibilities. The language model can interpret the question and write the explanation, but the current order state may come from a database tool, the refund policy may come from document retrieval, permissions may be enforced by the application, and the final action may require a controlled API call.",{},{"id":276,"data":1163,"type":305,"tunes":1190},{"steps":1164,"title":1189,"orientation":304},[1165,1168,1171,1174,1177,1180,1183,1186],{"label":1166,"description":1167},"1. User request","The application receives a natural-language question or task.",{"label":1169,"description":1170},"2. Application policy and state","Identity, tenant, permissions, current workflow state, and product rules define what the request is allowed to do.",{"label":1172,"description":1173},"3. Retrieval or direct data access","The system obtains external evidence or current facts when model knowledge is insufficient.",{"label":1175,"description":1176},"4. Context construction","Instructions, user input, selected evidence, relevant state, and tool definitions are assembled for the model.",{"label":1178,"description":1179},"5. Model inference","The generative model interprets the supplied context and produces text, structured output, or a tool request.",{"label":1181,"description":1182},"6. Tool execution when needed","The runtime or application validates and executes approved tool calls outside the model.",{"label":1184,"description":1185},"7. Observation and continuation","Tool results can return to the model as new context for another inference step.",{"label":1187,"description":1188},"8. Validation and product output","The application validates the result, records required state or audit data, and presents or executes the final outcome.","One common execution path",{},{"id":308,"data":1192,"type":218,"tunes":1194},{"text":1193},"Real systems do not always follow this sequence exactly. Retrieval can happen before the first model call, tools can be selected during an agent loop, deterministic application logic can bypass the model entirely, and validation can occur at several stages. The point is to separate responsibilities, not to impose one universal workflow.",{},{"id":313,"data":1196,"type":42,"tunes":1198},{"text":1197,"level":240},"The six boundaries that matter",{},{"id":318,"data":1200,"type":358,"tunes":1227},{"rows":1201,"title":1219,"layout":347,"columns":1220},[1202,1204,1207,1210,1213,1216],{"id":322,"label":323,"values":1203},[325,325,325],{"id":327,"label":1205,"values":1206},"Retrieval",[325,325,325],{"id":331,"label":1208,"values":1209},"Tools",[325,325,325],{"id":335,"label":1211,"values":1212},"Context",[325,325,325],{"id":339,"label":1214,"values":1215},"Runtime \u002F orchestrator",[325,325,325],{"id":343,"label":1217,"values":1218},"Application",[325,325,325],"Six responsibilities inside one AI product",[1221,1223,1225],{"id":350,"label":1222},"Primary job",{"id":353,"label":1224},"Typical inputs",{"id":356,"label":1226},"Not the same as",{},{"id":361,"data":1229,"type":42,"tunes":1231},{"text":1230,"level":240},"1. The model: generation is its core responsibility",{},{"id":366,"data":1233,"type":218,"tunes":1235},{"text":1234},"A generative model maps supplied inputs to generated outputs. For a language model, that can include natural-language text, structured JSON, code, classifications, summaries, plans, or tool-call arguments. Multimodal generative models can work with additional input and output types.",{},{"id":371,"data":1237,"type":218,"tunes":1239},{"text":1238},"The model can contain substantial learned knowledge in its parameters, but parameterized knowledge is not a live database. The model does not automatically know a document created five minutes ago, the current stock level, a private customer record, or the state of an application unless that information is supplied through the current input path.",{},{"id":376,"data":1241,"type":218,"tunes":1243},{"text":1242},"This is why changing the model does not automatically solve stale knowledge, missing permissions, broken retrieval, incorrect state ownership, or unsafe tool execution. Those failures often belong to other layers.",{},{"id":381,"data":1245,"type":42,"tunes":1247},{"text":1246,"level":240},"2. Retrieval: finding external evidence is a separate operation",{},{"id":386,"data":1249,"type":218,"tunes":1251},{"text":1250},"Retrieval selects information from an external source before or during generation. The 2020 Retrieval-Augmented Generation work by Lewis et al. made the separation explicit by combining a parametric generative model with retrieved non-parametric memory. Modern production systems use many retrieval variants, but the architectural idea remains: useful evidence can be fetched at inference time instead of relying only on what the model learned during training.",{},{"id":391,"data":1253,"type":218,"tunes":1255},{"text":1254},"Retrieval can use lexical search, embeddings, vector search, hybrid search, SQL, knowledge graphs, metadata filters, APIs, or other selection mechanisms. A vector database is therefore one possible retrieval component, not the definition of RAG.",{},{"id":396,"data":1257,"type":226,"tunes":1260},{"body":1258,"title":1259,"variant":400},"A retrieved passage can be highly relevant and still be stale, unauthorized, from the wrong version, or insufficient to support a claim. Retrieval quality and evidence quality must be evaluated separately.","Relevance is not authority",{},{"id":403,"data":1262,"type":409,"tunes":1267},{"url":1263,"title":1264,"excerpt":1265,"ctaLabel":1266},"https:\u002F\u002Fstajic.de\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works","What Is RAG? The Simplest Explanation of How It Works","The canonical plain-English explanation of retrieval-augmented generation, including the separation between LLM, knowledge, state, memory and tools.","Read the RAG foundation",{},{"id":412,"data":1269,"type":42,"tunes":1271},{"text":1270,"level":240},"3. Tools: access and action are not model knowledge",{},{"id":417,"data":1273,"type":218,"tunes":1275},{"text":1274},"A tool is an interface through which an AI runtime can request functionality outside the model. A tool can query a database, search the web, read a file, calculate a value, call an internal service, create a ticket, send a message, modify a record, or trigger another controlled operation.",{},{"id":422,"data":1277,"type":218,"tunes":1279},{"text":1278},"OpenAI's current function-calling documentation makes this boundary explicit: function calling lets models interface with external systems and access data or actions provided by the application. The model can propose or select a call, but the external system performs the real operation.",{},{"id":427,"data":1281,"type":218,"tunes":1283},{"text":1282},"Tool use therefore creates two separate questions: Can the model request this capability? and Will the application authorize and execute it? A production system should not confuse model intent with permission to cause a side effect.",{},{"id":432,"data":1285,"type":226,"tunes":1288},{"body":1286,"title":1287,"variant":263},"A model can emit a valid tool request and still be denied. Authorization, argument validation, rate limits, transaction rules, audit requirements, and rollback belong outside the model.","Model intent is not execution authority",{},{"id":438,"data":1290,"type":42,"tunes":1292},{"text":1291,"level":240},"4. Context: what the model can see right now",{},{"id":443,"data":1294,"type":218,"tunes":1296},{"text":1295},"Context is the information available to the model for a particular inference step. Anthropic's context-engineering guidance describes context as the set of tokens included when sampling from an LLM. In practice, that set can contain system instructions, user messages, conversation history, retrieved evidence, tool definitions, tool results, memory summaries, and selected application state.",{},{"id":448,"data":1298,"type":218,"tunes":1300},{"text":1299},"Context is therefore neither the complete knowledge base nor long-term memory. A company may store ten million documents while only a handful of passages enter one model call. A runtime may persist a year of conversation history while exposing only the pieces needed for the current task.",{},{"id":453,"data":1302,"type":218,"tunes":1304},{"text":1303},"The context window also creates an engineering constraint. Adding more text does not guarantee a better answer; irrelevant, stale, contradictory, or low-authority information can dilute the evidence that actually matters.",{},{"id":458,"data":1306,"type":42,"tunes":1308},{"text":1307,"level":240},"5. Runtime and orchestration: coordinating the loop",{},{"id":463,"data":1310,"type":218,"tunes":1312},{"text":1311},"The runtime or orchestration layer coordinates how the model participates in a task. Depending on the architecture, it can manage sessions, model requests, tool discovery, tool-call loops, retries, handoffs, streaming events, timeouts, checkpoints, compaction, or execution environments.",{},{"id":468,"data":1314,"type":218,"tunes":1316},{"text":1315},"Some runtimes are thin application code around a model API. Others are full agent harnesses. A managed vendor runtime can own part of the loop while the application still owns domain truth, authorization, business side effects, and product lifecycle.",{},{"id":473,"data":1318,"type":218,"tunes":1320},{"text":1319},"This boundary is important because where the runtime runs and where inference runs are separate decisions. A locally running client or agent process can still call a remote model, while a remote application can call a model hosted on infrastructure under the organization's control.",{},{"id":478,"data":1322,"type":42,"tunes":1324},{"text":1323,"level":240},"6. The application: where AI becomes a product",{},{"id":483,"data":1326,"type":218,"tunes":1328},{"text":1327},"The application is the product boundary around the AI components. It owns the user experience, domain model, current state, identity, tenant scope, permissions, persistence, service integrations, validation, observability, billing or quota logic where relevant, and the rules that determine what the AI is allowed to see or do.",{},{"id":488,"data":1330,"type":218,"tunes":1332},{"text":1331},"This is the layer that turns “a model can produce useful output” into “a system can deliver a reliable capability.” The same model can participate in a private research assistant, a support workflow, a code agent, or a commerce application because the surrounding application changes the data, tools, policies, state, and execution contract.",{},{"id":493,"data":1334,"type":226,"tunes":1337},{"body":1335,"title":1336,"variant":225},"Provider and model substitution can be an architectural goal. The application's authoritative state, permissions, domain rules, audit trail, and user contract cannot simply be delegated to whichever model is currently selected.","The model is replaceable; the product boundary is not",{},{"id":499,"data":1339,"type":42,"tunes":1341},{"text":1340,"level":240},"How the parts work together in a real request",{},{"id":504,"data":1343,"type":218,"tunes":1345},{"text":1344},"Consider a support assistant asked: “Refund order 4711 if it is still eligible, and explain why.” The request combines knowledge, current state, authorization, reasoning, and a side effect.",{},{"id":509,"data":1347,"type":347,"tunes":1379},{"content":1348,"stretched":43,"withHeadings":14},[1349,1353,1356,1360,1364,1368,1372,1375],[1350,1351,1352],"Need","Correct layer","Why",[1354,1205,1355],"Refund policy","The system must find the current applicable policy and preserve its provenance.",[1357,1358,1359],"Order 4711 status","Direct data\u002Ftool access","The current order record is volatile authoritative state, not something to guess from model knowledge.",[1361,1362,1363],"User's authority to refund","Application \u002F authorization","Permissions must be enforced independently of what the model asks for.",[1365,1366,1367],"Interpret policy against order facts","Model + context","The model can reason over the policy evidence and current order state supplied to it.",[1369,1370,1371],"Execute refund","Tool + application transaction rules","A controlled external operation changes real state.",[1373,323,1374],"Explain outcome","The model can generate the user-facing explanation from validated results.",[1376,1377,1378],"Audit what happened","Application \u002F runtime","The system records evidence, calls, decisions, side effects, and errors as required.",{},{"id":545,"data":1381,"type":218,"tunes":1383},{"text":1382},"If the assistant only has the language model, it can discuss refunds but cannot safely know whether order 4711 is currently eligible or perform the transaction. If it only has retrieval, it may find the policy but still lack live order state. If it has tools without application authorization, it may become capable but unsafe. Reliability comes from composing the layers with explicit ownership.",{},{"id":550,"data":1385,"type":42,"tunes":1387},{"text":1386,"level":240},"Different AI products use different combinations",{},{"id":555,"data":1389,"type":358,"tunes":1411},{"rows":1390,"title":1403,"layout":347,"columns":1404},[1391,1394,1397,1400],{"id":559,"label":1392,"values":1393},"Model-only assistant",[325,325,325,325],{"id":563,"label":1395,"values":1396},"Retrieval-grounded assistant",[325,325,325,325],{"id":567,"label":1398,"values":1399},"Tool-using assistant",[325,325,325,325],{"id":571,"label":1401,"values":1402},"Agentic application",[325,325,325,325],"The presence of a model does not define the whole architecture",[1405,1406,1407,1409],{"id":327,"label":1205},{"id":331,"label":1208},{"id":579,"label":1408},"Authoritative state",{"id":582,"label":1410},"Typical capability",{},{"id":586,"data":1413,"type":218,"tunes":1415},{"text":1414},"These are architecture patterns, not maturity rankings. A model-only feature can be the correct design when the task needs no external facts or actions. Adding retrieval, tools, memory, or an agent loop is justified only when the task requires those capabilities.",{},{"id":591,"data":1417,"type":42,"tunes":1419},{"text":1418,"level":240},"Implementation evidence: Aaasaasa AI Client",{},{"id":596,"data":1421,"type":226,"tunes":1424},{"body":1422,"title":1423,"variant":233},"The following section describes an implementation I built and reviewed against the Aaasaasa AI Client codebase and architecture documentation as of \u003Cstrong>26 July 2026\u003C\u002Fstrong>. It is evidence for the usefulness of these boundaries, not a claim that one implementation is a universal standard or a commercially deployed enterprise product.","Primary implementation evidence",{},{"id":602,"data":1426,"type":218,"tunes":1428},{"text":1427},"Aaasaasa AI Client is a local-first desktop AI workspace built with Nuxt 4, Electron and TypeScript. Its AI Hub deliberately separates agent\u002Fclient, provider, model, runtime location, permissions, and web client instead of treating them as one “AI” setting.",{},{"id":607,"data":1430,"type":218,"tunes":1432},{"text":1431},"That separation creates concrete behavior. Direct Chat can talk to models without filesystem or shell tools. A Codex agent can use a selected workspace and permission profile. Ollama can provide direct local inference, while LM Studio and configurable OpenAI-compatible endpoints represent other provider paths. A locally running Codex process can still use a cloud model, so the UI and architecture do not equate local runtime with local inference.",{},{"id":612,"data":1434,"type":218,"tunes":1436},{"text":1435},"The implementation also contains Qdrant\u002Fvector support, document-extraction capabilities and an authenticated directory MCP broker. Those components illustrate another boundary: retrieval infrastructure and tool access can live in the same product without becoming properties of the model itself.",{},{"id":617,"data":1438,"type":347,"tunes":1462},{"content":1439,"stretched":43,"withHeadings":14},[1440,1443,1445,1448,1451,1454,1457,1460],[1441,1442],"A01 concept","Aaasaasa AI Client implementation evidence",[323,1444],"A provider-specific model identifier is selected separately from provider and runtime.",[1446,1447],"Provider","Ollama, LM Studio, OpenAI-compatible services and other provider paths are represented separately.",[1449,1450],"Runtime","Local or remote agent\u002Fruntime location is tracked independently of the model.",[1452,1453],"Tools \u002F access","Direct Chat has no filesystem or shell tools; controlled directory access is brokered separately.",[1455,1456],"Permissions","Workspace permission profiles are application\u002Fsession policy, not model capability.",[1458,1459],"Retrieval infrastructure","Vector support and document extraction exist as data\u002Fretrieval capabilities rather than model features.",[1217,1461],"The Electron\u002FNuxt product coordinates UI, credentials, providers, runtime discovery, permissions, tools and model interaction.",{},{"id":644,"data":1464,"type":226,"tunes":1467},{"body":1465,"title":1466,"variant":263},"The architecture became easier to reason about once \u003Cstrong>model, provider, runtime, permissions, tools, data and client\u003C\u002Fstrong> stopped being represented as one configuration choice. The distinction is operational: it determines what can run locally, what can access files, what may call paid cloud inference, and which layer owns authorization.","Implementation lesson",{},{"id":650,"data":1469,"type":42,"tunes":1471},{"text":1470,"level":240},"Common category errors",{},{"id":655,"data":1473,"type":347,"tunes":1499},{"content":1474,"stretched":43,"withHeadings":14},[1475,1478,1481,1484,1487,1490,1493,1496],[1476,1477],"Category error","What is actually happening",[1479,1480],"“The AI knows our documents.”","The application or retrieval layer makes selected document content available to the model.",[1482,1483],"“RAG is our vector database.”","The vector database can be one index or store used by a retrieval pipeline; RAG is the retrieval-plus-generation pattern.",[1485,1486],"“The model called our CRM.”","The model produced a tool request; the runtime\u002Fapplication authorized and executed the external call.",[1488,1489],"“It is local AI because the desktop agent runs locally.”","Runtime location and inference location are separate. A local runtime can still invoke a remote model.",[1491,1492],"“The model has permission to edit files.”","The application\u002Fruntime grants a tool capability under a permission policy; permission is not an intrinsic model property.",[1494,1495],"“More context means more knowledge.”","Context is the finite input made available for one inference. Larger context can contain more noise, conflict or stale information.",[1497,1498],"“The chatbot is the AI architecture.”","The chat UI is one interface. The system can also include identity, state, retrieval, tools, runtime, validation, persistence and observability.",{},{"id":684,"data":1501,"type":42,"tunes":1503},{"text":1502,"level":240},"Failure modes when the boundaries collapse",{},{"id":689,"data":1505,"type":218,"tunes":1507},{"text":1506},"Boundary mistakes are not merely terminology problems. They create distinct production failures that require different fixes.",{},{"id":694,"data":1509,"type":358,"tunes":1537},{"rows":1510,"title":1529,"layout":347,"columns":1530},[1511,1514,1517,1520,1523,1526],{"id":698,"label":1512,"values":1513},"Stale answer",[325,325,325],{"id":702,"label":1515,"values":1516},"Missing company fact",[325,325,325],{"id":706,"label":1518,"values":1519},"Unsafe side effect",[325,325,325],{"id":710,"label":1521,"values":1522},"Confused answer with lots of supplied text",[325,325,325],{"id":714,"label":1524,"values":1525},"Unexpected cloud use",[325,325,325],{"id":718,"label":1527,"values":1528},"Agent stalls or repeats",[325,325,325],"Diagnose the failing layer before replacing the model",[1531,1533,1535],{"id":724,"label":1532},"Symptom",{"id":727,"label":1534},"Likely boundary problem",{"id":730,"label":1536},"First architectural check",{},{"id":734,"data":1539,"type":42,"tunes":1541},{"text":1540,"level":240},"What is stable and what is version-sensitive?",{},{"id":739,"data":1543,"type":218,"tunes":1545},{"text":1544},"The architectural distinctions in this article are intentionally vendor-neutral. The current examples below are implementation facts that should be re-checked when APIs evolve.",{},{"id":744,"data":1547,"type":347,"tunes":1576},{"content":1548,"stretched":43,"withHeadings":14},[1549,1553,1557,1560,1564,1568,1572],[1550,1551,1552],"Area","Stable architectural idea","Verified current example on 8 Oct 2026",[1554,1555,1556],"AI model vs system","A model is a component inside a broader system","NIST's current glossary separately defines AI model and AI system.",[756,1558,1559],"Generation can be conditioned on retrieved external information","The Lewis et al. 2020 formulation remains the foundational reference; production retrieval methods now extend far beyond one dense index design.",[1561,1562,1563],"Hosted retrieval","Retrieval can be exposed as a managed tool","OpenAI File Search is currently a Responses API tool that searches uploaded-file knowledge bases using semantic and keyword retrieval.",[1565,1566,1567],"Function\u002Ftool calling","A model can request application-defined external capabilities","OpenAI currently documents function calling as an interface to external systems, data and actions.",[1569,1570,1571],"Context engineering","Model behavior depends on the finite information supplied for the current inference","Anthropic's current engineering guidance defines context as the token set included when sampling from the LLM and focuses on curating that set.",[1573,1574,1575],"Vendor APIs","SDKs, tool names, endpoint shapes and supported features change","Treat vendor documentation as version-sensitive even when the responsibility boundary remains stable.",{},{"id":777,"data":1578,"type":218,"tunes":1580},{"text":1579},"A source-of-truth article should therefore preserve both levels: stable concepts for architecture, and dated evidence for current implementations. Mixing the two makes an article age unnecessarily fast.",{},{"id":782,"data":1582,"type":42,"tunes":1584},{"text":1583,"level":240},"The AI component-boundary test",{},{"id":787,"data":1586,"type":218,"tunes":1588},{"text":1587},"When evaluating an AI feature, ask the following questions in order. The answers reveal which components the system actually has and which responsibilities are still implicit.",{},{"id":792,"data":1590,"type":305,"tunes":1614},{"steps":1591,"title":1613,"orientation":304},[1592,1595,1598,1601,1604,1607,1610],{"label":1593,"description":1594},"1. What generates the output?","Identify the exact model and the modalities or structured outputs it provides.",{"label":1596,"description":1597},"2. What facts are authoritative outside the model?","Identify documents, databases, APIs, current state and other sources of truth.",{"label":1599,"description":1600},"3. How is relevant information selected?","Separate direct lookup, search, retrieval, ranking and context construction.",{"label":1602,"description":1603},"4. What can cause real side effects?","List tools and external actions, then identify who validates and authorizes them.",{"label":1605,"description":1606},"5. What reaches the model as context?","Make instructions, evidence, state, history, memory and tool definitions explicit.",{"label":1608,"description":1609},"6. Who owns the loop?","Identify the runtime or harness that manages calls, events, retries, tool loops and sessions.",{"label":1611,"description":1612},"7. What remains the application's responsibility?","Make identity, permissions, domain state, validation, persistence, observability and UX explicit.","Seven questions for a production design",{},{"id":819,"data":1616,"type":42,"tunes":1618},{"text":1617,"level":240},"What generative AI is not",{},{"id":824,"data":1620,"type":218,"tunes":1622},{"text":1621},"Generative AI is not synonymous with an LLM, even though LLMs are a major class of generative model. It is also not synonymous with RAG, a vector database, an agent, a tool protocol, a chatbot UI, or an application.",{},{"id":829,"data":1624,"type":218,"tunes":1626},{"text":1625},"Those concepts can be connected, but each answers a different architectural question. An LLM asks how language output is produced. Retrieval asks where external evidence comes from. Tools ask how external capabilities are exposed. Context asks what the model can see. Runtime asks how execution is coordinated. The application asks how the capability becomes a controlled product.",{},{"id":834,"data":1628,"type":226,"tunes":1631},{"body":1629,"title":1630,"variant":263},"\u003Cstrong>Model = generate.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Retrieval = find evidence.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Tools = read or act outside the model.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Context = what the model sees now.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Runtime = coordinate execution.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Application = own the product, state, rules and permissions.\u003C\u002Fstrong>","If you remember only one model",{},{"id":840,"data":1633,"type":42,"tunes":1635},{"text":1634,"level":240},"Where to go next in the knowledge graph",{},{"id":845,"data":1637,"type":218,"tunes":1639},{"text":1638},"Once these boundaries are clear, deeper topics become easier to place. RAG belongs in retrieval and context construction. Retrieval Trigger decides when external evidence is required. Agent memory concerns what persists across time. Tool calling and MCP belong to capability access. Agent harnesses belong to runtime orchestration. RBAC, tenant isolation and domain authorization belong to the application and platform security boundary.",{},{"id":850,"data":1641,"type":409,"tunes":1646},{"url":1642,"title":1643,"excerpt":1644,"ctaLabel":1645},"https:\u002F\u002Fstajic.de\u002Fblog\u002Fwhere-does-an-llm-get-its-data-rag-data-sources-in-python","Where Does an LLM Get Its Data? RAG Data Sources in Python","A practical continuation showing how files, SQL, APIs, full-text search, embeddings and context assembly connect external data to an LLM.","See the data path in code",{},{"id":858,"data":1648,"type":409,"tunes":1653},{"url":1649,"title":1650,"excerpt":1651,"ctaLabel":1652},"https:\u002F\u002Fstajic.de\u002Fblog\u002Fwhen-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger","When Should an AI Stop Trusting Its Own Knowledge? — The Retrieval Trigger","A decision model for when an AI system should stop relying only on model knowledge and obtain external evidence.","Read the retrieval decision model",{},{"id":866,"data":1655,"type":42,"tunes":1657},{"text":1656,"level":240},"Limitations",{},{"id":871,"data":1659,"type":218,"tunes":1661},{"text":1660},"The six-layer model is a responsibility map, not a requirement that every product deploy six separate services. A small application may implement context construction, retrieval and orchestration inside one process. A managed platform may bundle several responsibilities behind one API. Physical deployment can be combined while semantic ownership remains distinct.",{},{"id":876,"data":1663,"type":218,"tunes":1665},{"text":1664},"Terminology also varies across vendors and research. “Agent,” “runtime,” “memory,” “tool,” “connector,” and “context” can be defined differently. The definitions here are chosen to make operational ownership and failure diagnosis explicit rather than to claim that every framework uses identical vocabulary.",{},{"id":881,"data":1667,"type":218,"tunes":1669},{"text":1668},"The Aaasaasa AI Client section documents one implementation pattern. It demonstrates that explicit boundaries are practical, but it does not prove that the same component layout is optimal for every AI product.",{},{"id":886,"data":1671,"type":42,"tunes":1673},{"text":1672,"level":240},"What would change this answer?",{},{"id":891,"data":1675,"type":218,"tunes":1677},{"text":1676},"The responsibility map would need revision if model architectures themselves began to own authoritative external state, permissions, durable transactional side effects, and verifiable source access as intrinsic properties rather than capabilities supplied by a surrounding system. Current production architectures do not make that a safe general assumption.",{},{"id":896,"data":1679,"type":218,"tunes":1681},{"text":1680},"Individual implementation examples will change much sooner. Hosted retrieval tools, agent APIs, MCP integrations, context-management features and provider capabilities evolve quickly. Those details should be updated without collapsing the underlying distinctions between generation, evidence, capability access, context, execution and application control.",{},{"id":901,"data":1683,"type":42,"tunes":1685},{"text":1684,"level":240},"Conclusion",{},{"id":906,"data":1687,"type":218,"tunes":1689},{"text":1688},"Generative AI becomes easier to design once “the AI” stops being treated as one black box. The model is the generative component, not the complete product. Retrieval provides external evidence. Tools expose capabilities. Context carries selected information into the current inference. The runtime coordinates execution. The application owns the authoritative product boundary.",{},{"id":911,"data":1691,"type":218,"tunes":1693},{"text":1692},"That separation is useful for more than explanation. It tells engineers where stale facts originate, where authorization belongs, why a local runtime can still use cloud inference, why RAG does not equal a vector database, why tool calls require validation, and why changing the model cannot repair every system failure.",{},{"id":916,"data":1695,"type":218,"tunes":1697},{"text":1696},"The durable architecture question is therefore not “Which AI model are we using?” It is: Which responsibility does each component own, what evidence crosses each boundary, and which layer is allowed to change real state?",{},{"id":921,"data":1699,"type":42,"tunes":1701},{"text":1700,"level":240},"FAQ",{},{"id":926,"data":1703,"type":926,"tunes":1727},{"items":1704,"title":1726},[1705,1708,1711,1714,1717,1720,1723],{"id":930,"answer":1706,"question":1707},"No. An LLM is one type of generative model. Generative AI also includes other modalities, and a production generative AI system can include retrieval, tools, runtime logic, application state, permissions, persistence and user interfaces around the model.","Is generative AI the same as an LLM?",{"id":934,"answer":1709,"question":1710},"Usually no. RAG is an application\u002Fsystem pattern that retrieves external information and supplies selected evidence to the model. Some platforms package retrieval tightly with model APIs, but the responsibility remains distinct.","Is RAG part of the model?",{"id":938,"answer":1712,"question":1713},"No. RAG can use vector search, lexical search, hybrid retrieval, SQL, APIs, knowledge graphs or other methods. The defining property is retrieval of external information for generation, not one storage technology.","Is a vector database required for RAG?",{"id":942,"answer":1715,"question":1716},"No. A tool is an external capability. Its definition may be represented in context, and its result may later enter context, but the actual capability executes outside the model.","Are tools the same as context?",{"id":946,"answer":1718,"question":1719},"No. Runtime location and inference location are separate. A local desktop application or agent can call a remote model, while a remote application can call an internally hosted model.","Does running an AI client locally mean the model is local?",{"id":950,"answer":1721,"question":1722},"The application or runtime security boundary should enforce authorization. A model can request an operation, but model intent should never be treated as sufficient execution authority.","Who should enforce permissions for AI tools?",{"id":954,"answer":1724,"question":1725},"Authoritative volatile state should normally remain in the application or domain system that owns it. The AI can receive the relevant state through controlled context or tool access when needed.","Where does current application state belong?","Generative AI system boundaries",{},{"id":960,"data":1729,"type":42,"tunes":1731},{"text":1730,"level":240},"Glossary",{},{"id":965,"data":1733,"type":965,"tunes":1754},{"title":1734,"entries":1735},"Core terms",[1736,1739,1741,1743,1746,1748,1750,1752],{"term":1737,"anchor":971,"definition":1738},"Generative model","An AI model designed to generate derived synthetic content such as text, images, audio, video, code or structured output.",{"term":1205,"anchor":327,"definition":1740},"The process of selecting relevant information from an external source or store for the current task.",{"term":756,"anchor":563,"definition":1742},"Retrieval-Augmented Generation: a pattern in which retrieved external information is supplied to a generative model to improve the current output.",{"term":1744,"anchor":567,"definition":1745},"Tool","A capability exposed to an AI runtime for reading data, calculating, searching, or performing an external action.",{"term":1211,"anchor":335,"definition":1747},"The information available to the model for a particular inference step.",{"term":1214,"anchor":985,"definition":1749},"The software layer that coordinates model calls, tool calls, task loops, sessions, retries, events or execution environments.",{"term":1217,"anchor":343,"definition":1751},"The product and domain layer that owns user interaction, authoritative state, permissions, validation, persistence and business behavior.",{"term":1446,"anchor":991,"definition":1753},"The service or runtime that exposes access to one or more models; provider identity and model identity are separate concerns.",{},{"id":995,"data":1756,"type":42,"tunes":1758},{"text":1757,"level":240},"Primary sources and implementation evidence",{},{"id":1000,"data":1760,"type":218,"tunes":1762},{"text":1761},"Stable definitions below are anchored in standards\u002Fresearch; fast-moving implementation examples use current official engineering documentation. Aaasaasa AI Client is original implementation evidence and was checked against its codebase\u002Fdocumentation state dated 26 July 2026.",{},{"id":1005,"data":1764,"type":1012,"tunes":1769},{"link":1007,"meta":1765},{"image":1766,"title":1767,"description":1768},{"url":325},"NIST AI 600-1 — Generative Artificial Intelligence Profile","NIST's Generative AI profile, including the generative-AI definition and explicit distinction between model-, system-, application- and use-case-level concerns.",{},{"id":1015,"data":1771,"type":1012,"tunes":1776},{"link":1017,"meta":1772},{"image":1773,"title":1774,"description":1775},{"url":325},"NIST — Artificial Intelligence Model","Current NIST glossary definition of an AI model as a component of an information system that produces outputs from inputs using AI techniques.",{},{"id":1024,"data":1778,"type":1012,"tunes":1783},{"link":1026,"meta":1779},{"image":1780,"title":1781,"description":1782},{"url":325},"NIST — Artificial Intelligence System","Current NIST glossary definition showing that an AI system can include data systems, software, hardware, applications, tools or utilities using AI.",{},{"id":1033,"data":1785,"type":1012,"tunes":1790},{"link":1035,"meta":1786},{"image":1787,"title":1788,"description":1789},{"url":325},"Lewis et al. — Retrieval-Augmented Generation for Knowledge-Intensive NLP Tasks","The 2020 paper introducing the RAG formulation that combines a generative model with retrieved non-parametric memory.",{},{"id":1042,"data":1792,"type":1012,"tunes":1797},{"link":1044,"meta":1793},{"image":1794,"title":1795,"description":1796},{"url":325},"OpenAI — File Search","Current official documentation for hosted file retrieval in the Responses API using uploaded-file knowledge bases, semantic search and keyword search.",{},{"id":1051,"data":1799,"type":1012,"tunes":1804},{"link":1053,"meta":1800},{"image":1801,"title":1802,"description":1803},{"url":325},"OpenAI — Function Calling","Current official documentation describing tool\u002Ffunction calling as the interface between models and external systems, data and actions.",{},{"id":1060,"data":1806,"type":1012,"tunes":1811},{"link":1062,"meta":1807},{"image":1808,"title":1809,"description":1810},{"url":325},"Anthropic — Effective Context Engineering for AI Agents","Engineering guidance defining context as the token set available during LLM sampling and explaining why context selection is a finite-resource problem.",{},"2.31.6","Generative AI is more than a model. Learn how models, retrieval, tools, context, runtimes and applications fit together in production AI systems.",{"lang":7,"title":208,"content":210,"contentJson":1815,"excerpt":1069},{"time":212,"blocks":1816,"version":1068},[1817,1820,1823,1826,1829,1832,1835,1838,1841,1844,1847,1859,1862,1865,1885,1888,1891,1894,1897,1900,1903,1906,1909,1912,1915,1918,1921,1924,1927,1930,1933,1936,1939,1942,1945,1948,1951,1954,1957,1960,1963,1966,1969,1981,1984,1987,2004,2007,2010,2013,2016,2019,2022,2034,2037,2040,2052,2055,2058,2078,2081,2084,2095,2098,2101,2104,2115,2118,2121,2124,2127,2130,2133,2136,2139,2142,2145,2148,2151,2154,2157,2160,2163,2166,2169,2172,2175,2186,2189,2201,2204,2207,2212,2217,2222,2227,2232,2237],{"id":215,"data":1818,"type":218,"tunes":1819},{"text":217},{},{"id":221,"data":1821,"type":226,"tunes":1822},{"body":223,"title":224,"variant":225},{},{"id":229,"data":1824,"type":226,"tunes":1825},{"body":231,"title":232,"variant":233},{},{"id":236,"data":1827,"type":241,"tunes":1828},{"title":238,"maxLevel":239,"minLevel":240},{},{"id":244,"data":1830,"type":42,"tunes":1831},{"text":246,"level":240},{},{"id":249,"data":1833,"type":218,"tunes":1834},{"text":251},{},{"id":254,"data":1836,"type":218,"tunes":1837},{"text":256},{},{"id":259,"data":1839,"type":226,"tunes":1840},{"body":261,"title":262,"variant":263},{},{"id":266,"data":1842,"type":42,"tunes":1843},{"text":268,"level":240},{},{"id":271,"data":1845,"type":218,"tunes":1846},{"text":273},{},{"id":276,"data":1848,"type":305,"tunes":1858},{"steps":1849,"title":303,"orientation":304},[1850,1851,1852,1853,1854,1855,1856,1857],{"label":280,"description":281},{"label":283,"description":284},{"label":286,"description":287},{"label":289,"description":290},{"label":292,"description":293},{"label":295,"description":296},{"label":298,"description":299},{"label":301,"description":302},{},{"id":308,"data":1860,"type":218,"tunes":1861},{"text":310},{},{"id":313,"data":1863,"type":42,"tunes":1864},{"text":315,"level":240},{},{"id":318,"data":1866,"type":358,"tunes":1884},{"rows":1867,"title":346,"layout":347,"columns":1880},[1868,1870,1872,1874,1876,1878],{"id":322,"label":323,"values":1869},[325,325,325],{"id":327,"label":328,"values":1871},[325,325,325],{"id":331,"label":332,"values":1873},[325,325,325],{"id":335,"label":336,"values":1875},[325,325,325],{"id":339,"label":340,"values":1877},[325,325,325],{"id":343,"label":344,"values":1879},[325,325,325],[1881,1882,1883],{"id":350,"label":351},{"id":353,"label":354},{"id":356,"label":357},{},{"id":361,"data":1886,"type":42,"tunes":1887},{"text":363,"level":240},{},{"id":366,"data":1889,"type":218,"tunes":1890},{"text":368},{},{"id":371,"data":1892,"type":218,"tunes":1893},{"text":373},{},{"id":376,"data":1895,"type":218,"tunes":1896},{"text":378},{},{"id":381,"data":1898,"type":42,"tunes":1899},{"text":383,"level":240},{},{"id":386,"data":1901,"type":218,"tunes":1902},{"text":388},{},{"id":391,"data":1904,"type":218,"tunes":1905},{"text":393},{},{"id":396,"data":1907,"type":226,"tunes":1908},{"body":398,"title":399,"variant":400},{},{"id":403,"data":1910,"type":409,"tunes":1911},{"url":405,"title":406,"excerpt":407,"ctaLabel":408},{},{"id":412,"data":1913,"type":42,"tunes":1914},{"text":414,"level":240},{},{"id":417,"data":1916,"type":218,"tunes":1917},{"text":419},{},{"id":422,"data":1919,"type":218,"tunes":1920},{"text":424},{},{"id":427,"data":1922,"type":218,"tunes":1923},{"text":429},{},{"id":432,"data":1925,"type":226,"tunes":1926},{"body":434,"title":435,"variant":263},{},{"id":438,"data":1928,"type":42,"tunes":1929},{"text":440,"level":240},{},{"id":443,"data":1931,"type":218,"tunes":1932},{"text":445},{},{"id":448,"data":1934,"type":218,"tunes":1935},{"text":450},{},{"id":453,"data":1937,"type":218,"tunes":1938},{"text":455},{},{"id":458,"data":1940,"type":42,"tunes":1941},{"text":460,"level":240},{},{"id":463,"data":1943,"type":218,"tunes":1944},{"text":465},{},{"id":468,"data":1946,"type":218,"tunes":1947},{"text":470},{},{"id":473,"data":1949,"type":218,"tunes":1950},{"text":475},{},{"id":478,"data":1952,"type":42,"tunes":1953},{"text":480,"level":240},{},{"id":483,"data":1955,"type":218,"tunes":1956},{"text":485},{},{"id":488,"data":1958,"type":218,"tunes":1959},{"text":490},{},{"id":493,"data":1961,"type":226,"tunes":1962},{"body":495,"title":496,"variant":225},{},{"id":499,"data":1964,"type":42,"tunes":1965},{"text":501,"level":240},{},{"id":504,"data":1967,"type":218,"tunes":1968},{"text":506},{},{"id":509,"data":1970,"type":347,"tunes":1980},{"content":1971,"stretched":43,"withHeadings":14},[1972,1973,1974,1975,1976,1977,1978,1979],[513,514,515],[517,518,519],[521,522,523],[525,526,527],[529,530,531],[533,534,535],[537,323,538],[540,541,542],{},{"id":545,"data":1982,"type":218,"tunes":1983},{"text":547},{},{"id":550,"data":1985,"type":42,"tunes":1986},{"text":552,"level":240},{},{"id":555,"data":1988,"type":358,"tunes":2003},{"rows":1989,"title":574,"layout":347,"columns":1998},[1990,1992,1994,1996],{"id":559,"label":560,"values":1991},[325,325,325,325],{"id":563,"label":564,"values":1993},[325,325,325,325],{"id":567,"label":568,"values":1995},[325,325,325,325],{"id":571,"label":572,"values":1997},[325,325,325,325],[1999,2000,2001,2002],{"id":327,"label":518},{"id":331,"label":332},{"id":579,"label":580},{"id":582,"label":583},{},{"id":586,"data":2005,"type":218,"tunes":2006},{"text":588},{},{"id":591,"data":2008,"type":42,"tunes":2009},{"text":593,"level":240},{},{"id":596,"data":2011,"type":226,"tunes":2012},{"body":598,"title":599,"variant":233},{},{"id":602,"data":2014,"type":218,"tunes":2015},{"text":604},{},{"id":607,"data":2017,"type":218,"tunes":2018},{"text":609},{},{"id":612,"data":2020,"type":218,"tunes":2021},{"text":614},{},{"id":617,"data":2023,"type":347,"tunes":2033},{"content":2024,"stretched":43,"withHeadings":14},[2025,2026,2027,2028,2029,2030,2031,2032],[621,622],[323,624],[626,627],[629,630],[632,633],[635,636],[638,639],[344,641],{},{"id":644,"data":2035,"type":226,"tunes":2036},{"body":646,"title":647,"variant":263},{},{"id":650,"data":2038,"type":42,"tunes":2039},{"text":652,"level":240},{},{"id":655,"data":2041,"type":347,"tunes":2051},{"content":2042,"stretched":43,"withHeadings":14},[2043,2044,2045,2046,2047,2048,2049,2050],[659,660],[662,663],[665,666],[668,669],[671,672],[674,675],[677,678],[680,681],{},{"id":684,"data":2053,"type":42,"tunes":2054},{"text":686,"level":240},{},{"id":689,"data":2056,"type":218,"tunes":2057},{"text":691},{},{"id":694,"data":2059,"type":358,"tunes":2077},{"rows":2060,"title":721,"layout":347,"columns":2073},[2061,2063,2065,2067,2069,2071],{"id":698,"label":699,"values":2062},[325,325,325],{"id":702,"label":703,"values":2064},[325,325,325],{"id":706,"label":707,"values":2066},[325,325,325],{"id":710,"label":711,"values":2068},[325,325,325],{"id":714,"label":715,"values":2070},[325,325,325],{"id":718,"label":719,"values":2072},[325,325,325],[2074,2075,2076],{"id":724,"label":725},{"id":727,"label":728},{"id":730,"label":731},{},{"id":734,"data":2079,"type":42,"tunes":2080},{"text":736,"level":240},{},{"id":739,"data":2082,"type":218,"tunes":2083},{"text":741},{},{"id":744,"data":2085,"type":347,"tunes":2094},{"content":2086,"stretched":43,"withHeadings":14},[2087,2088,2089,2090,2091,2092,2093],[748,749,750],[752,753,754],[756,757,758],[760,761,762],[764,765,766],[768,769,770],[772,773,774],{},{"id":777,"data":2096,"type":218,"tunes":2097},{"text":779},{},{"id":782,"data":2099,"type":42,"tunes":2100},{"text":784,"level":240},{},{"id":787,"data":2102,"type":218,"tunes":2103},{"text":789},{},{"id":792,"data":2105,"type":305,"tunes":2114},{"steps":2106,"title":816,"orientation":304},[2107,2108,2109,2110,2111,2112,2113],{"label":796,"description":797},{"label":799,"description":800},{"label":802,"description":803},{"label":805,"description":806},{"label":808,"description":809},{"label":811,"description":812},{"label":814,"description":815},{},{"id":819,"data":2116,"type":42,"tunes":2117},{"text":821,"level":240},{},{"id":824,"data":2119,"type":218,"tunes":2120},{"text":826},{},{"id":829,"data":2122,"type":218,"tunes":2123},{"text":831},{},{"id":834,"data":2125,"type":226,"tunes":2126},{"body":836,"title":837,"variant":263},{},{"id":840,"data":2128,"type":42,"tunes":2129},{"text":842,"level":240},{},{"id":845,"data":2131,"type":218,"tunes":2132},{"text":847},{},{"id":850,"data":2134,"type":409,"tunes":2135},{"url":852,"title":853,"excerpt":854,"ctaLabel":855},{},{"id":858,"data":2137,"type":409,"tunes":2138},{"url":860,"title":861,"excerpt":862,"ctaLabel":863},{},{"id":866,"data":2140,"type":42,"tunes":2141},{"text":868,"level":240},{},{"id":871,"data":2143,"type":218,"tunes":2144},{"text":873},{},{"id":876,"data":2146,"type":218,"tunes":2147},{"text":878},{},{"id":881,"data":2149,"type":218,"tunes":2150},{"text":883},{},{"id":886,"data":2152,"type":42,"tunes":2153},{"text":888,"level":240},{},{"id":891,"data":2155,"type":218,"tunes":2156},{"text":893},{},{"id":896,"data":2158,"type":218,"tunes":2159},{"text":898},{},{"id":901,"data":2161,"type":42,"tunes":2162},{"text":903,"level":240},{},{"id":906,"data":2164,"type":218,"tunes":2165},{"text":908},{},{"id":911,"data":2167,"type":218,"tunes":2168},{"text":913},{},{"id":916,"data":2170,"type":218,"tunes":2171},{"text":918},{},{"id":921,"data":2173,"type":42,"tunes":2174},{"text":923,"level":240},{},{"id":926,"data":2176,"type":926,"tunes":2185},{"items":2177,"title":957},[2178,2179,2180,2181,2182,2183,2184],{"id":930,"answer":931,"question":932},{"id":934,"answer":935,"question":936},{"id":938,"answer":939,"question":940},{"id":942,"answer":943,"question":944},{"id":946,"answer":947,"question":948},{"id":950,"answer":951,"question":952},{"id":954,"answer":955,"question":956},{},{"id":960,"data":2187,"type":42,"tunes":2188},{"text":962,"level":240},{},{"id":965,"data":2190,"type":965,"tunes":2200},{"title":967,"entries":2191},[2192,2193,2194,2195,2196,2197,2198,2199],{"term":970,"anchor":971,"definition":972},{"term":974,"anchor":327,"definition":975},{"term":756,"anchor":563,"definition":977},{"term":979,"anchor":567,"definition":980},{"term":336,"anchor":335,"definition":982},{"term":984,"anchor":985,"definition":986},{"term":344,"anchor":343,"definition":988},{"term":990,"anchor":991,"definition":992},{},{"id":995,"data":2202,"type":42,"tunes":2203},{"text":997,"level":240},{},{"id":1000,"data":2205,"type":218,"tunes":2206},{"text":1002},{},{"id":1005,"data":2208,"type":1012,"tunes":2211},{"link":1007,"meta":2209},{"image":2210,"title":1010,"description":1011},{"url":325},{},{"id":1015,"data":2213,"type":1012,"tunes":2216},{"link":1017,"meta":2214},{"image":2215,"title":1020,"description":1021},{"url":325},{},{"id":1024,"data":2218,"type":1012,"tunes":2221},{"link":1026,"meta":2219},{"image":2220,"title":1029,"description":1030},{"url":325},{},{"id":1033,"data":2223,"type":1012,"tunes":2226},{"link":1035,"meta":2224},{"image":2225,"title":1038,"description":1039},{"url":325},{},{"id":1042,"data":2228,"type":1012,"tunes":2231},{"link":1044,"meta":2229},{"image":2230,"title":1047,"description":1048},{"url":325},{},{"id":1051,"data":2233,"type":1012,"tunes":2236},{"link":1053,"meta":2234},{"image":2235,"title":1056,"description":1057},{"url":325},{},{"id":1060,"data":2238,"type":1012,"tunes":2241},{"link":1062,"meta":2239},{"image":2240,"title":1065,"description":1066},{"url":325},{},"Post erfolgreich abgerufen",{"items":2244,"source":2328,"manualIds":2329,"manualMatchedIds":2330},[2245,2252,2259,2266,2272,2279,2286,2293,2300,2307,2314,2321],{"id":2246,"slug":2247,"title":2248,"excerpt":2249,"featuredImage":2250,"publishedAt":2251},"486","source-of-truth-in-ai-systems-where-reliable-knowledge-actually-comes-from","Izvor istine u AI sistemima: Odakle pouzdano znanje zaista dolazi","Izvor istine definiše koji je izvor merodavan za određenu činjenicu ili stanje. Saznajte kako se razlikuje od RAG-a, porekla, memorije, konteksta, vektorskih baza podataka i sistema evidencije.","\u002Fuploads\u002F2026\u002F10\u002Fsource-of-truth-in-ai-systems-where-reliable-knowledge-actually-comes-from-1791479103235-6bq9em.webp","2026-10-08T13:02:00.000Z",{"id":2253,"slug":2254,"title":2255,"excerpt":2256,"featuredImage":2257,"publishedAt":2258},"466","the-gpu-is-not-the-product-future-proof-private-ai-architecture","GPU nije proizvod: Privatna AI arhitektura spremna za budućnost","Privatna AI infrastruktura ne bi trebalo da bude projektovana oko jednog GPU-a ili jednog modela. Otporniji pristup kombinuje brze GPU-ove za inferenciju, memorijski bogate AI sisteme, čvorove za fizički AI i opcione vodeće modele u oblaku iza sloja za rutiranje koji prepoznaje mogućnosti.","\u002Fuploads\u002F2026\u002F09\u002Fthe-gpu-is-not-the-product-future-proof-private-ai-architecture-1790140878812-8hsl39.webp","2026-09-23T01:19:00.000Z",{"id":2260,"slug":2261,"title":2262,"excerpt":2263,"featuredImage":2264,"publishedAt":2265},"483","what-is-an-ai-solution-architect-system-boundaries-responsibilities-and-trade-offs","Šta je AI rešenje arhitekta? Granice sistema, odgovornosti i kompromisi","AI Solution Architect pretvara poslovne zahteve u AI sistem spreman za produkciju, obuhvatajući podatke, modele, alate, bezbednost, izvršno okruženje, evaluaciju i operacije.","\u002Fuploads\u002F2026\u002F10\u002Fwhat-is-an-ai-solution-architect-system-boundaries-responsibilities-and-trade-offs-1791476643267-1st5xz.webp","2026-10-08T12:23:00.000Z",{"id":2267,"slug":2268,"title":853,"excerpt":2269,"featuredImage":2270,"publishedAt":2271},"479","where-does-an-llm-get-its-data-rag-data-sources-in-python","LLM ne zna magično vaše fajlove, baze podataka ili API-je. Ovaj praktični nastavak RAG serije pokazuje, uz jednostavan Python, kako eksterni podaci postaju dokazi koji se mogu pronaći: od tekstualnih fajlova i SQL-a do pretrage punog teksta, embeddinga, sastavljanja konteksta i konačnog LLM poziva.","\u002Fuploads\u002F2026\u002F09\u002Fwhere-does-an-llm-get-its-data-rag-data-sources-in-python-1790517200521-nfsi5i.webp","2026-09-27T05:51:00.000Z",{"id":2273,"slug":2274,"title":2275,"excerpt":2276,"featuredImage":2277,"publishedAt":2278},"482","adr-vs-nfr-architecture-decisions-and-system-quality-are-not-the-same-thing","ADR vs NFR: Arhitektonske odluke i kvalitet sistema nisu ista stvar","ADR vs NFR objašnjeno: naučite kako zahtevi za kvalitet sistema pokreću arhitekturne odluke, kako ADR-ovi beleže kompromise i zašto validacija ostaje odvojena.","\u002Fuploads\u002F2026\u002F10\u002Fadr-vs-nfr-architecture-decisions-and-system-quality-are-not-the-same-thing-1791475921511-6zgen1.webp","2026-10-08T12:11:00.000Z",{"id":2280,"slug":2281,"title":2282,"excerpt":2283,"featuredImage":2284,"publishedAt":2285},"495","sovereign-ai-control-of-models-data-infrastructure-and-dependencies","Suverena AI: Kontrola modela, podataka, infrastrukture i zavisnosti","Suverena AI se odnosi na efektivnu kontrolu nad modelima, podacima, infrastrukturom, softverom, operacijama i strateškim zavisnostima — a ne samo na to gde je AI model hostovan.","\u002Fuploads\u002F2026\u002F10\u002Fsovereign-ai-control-of-models-data-infrastructure-and-dependencies-1791488833132-niy85x.webp","2026-10-08T15:45:00.000Z",{"id":2287,"slug":2288,"title":2289,"excerpt":2290,"featuredImage":2291,"publishedAt":2292},"487","vector-databases-embeddings-and-reranking-three-different-parts-of-retrieval","Vektorske baze podataka, ugrađivanja i ponovno rangiranje: Tri različita dela pretraživanja","Embedinzi predstavljaju značenje, vektorske baze podataka pronalaze kandidate, a rerangirači prečišćavaju rezultate. Saznajte kako se ova tri sloja pronalaženja razlikuju i kako rade zajedno u RAG-u.","\u002Fuploads\u002F2026\u002F10\u002Fvector-databases-embeddings-and-reranking-three-different-parts-of-retrieval-1791480129884-9dtasz.webp","2026-10-08T11:21:00.000Z",{"id":2294,"slug":2295,"title":2296,"excerpt":2297,"featuredImage":2298,"publishedAt":2299},"476","mcp-vs-a2a-vs-ucp-vs-ap2-vs-a2ui-the-agent-protocol-stack-explained","MCP vs A2A vs UCP vs AP2 vs A2UI: Objašnjen stek agentskih protokola","MCP, A2A, UCP, AP2 i A2UI se često predstavljaju kao konkurentski standardi za agente. Oni uglavnom rešavaju različite probleme interoperabilnosti. Ovaj vodič mapira svaki protokol na granicu koju zapravo standardizuje—i pokazuje kako oni mogu da rade zajedno u jednom produkcionom sistemu.","\u002Fuploads\u002F2026\u002F09\u002Fmcp-vs-a2a-vs-ucp-vs-ap2-vs-a2ui-the-agent-protocol-stack-explained-1790352625869-2ezle0.webp","2026-09-25T12:09:00.000Z",{"id":2301,"slug":2302,"title":2303,"excerpt":2304,"featuredImage":2305,"publishedAt":2306},"470","what-should-an-ai-agent-remember-forget-recompute-or-retrieve-again","Šta bi AI agent trebalo da zapamti, zaboravi, ponovo izračuna ili ponovo preuzme?","Dugotrajni agenti ne bi trebalo da pamte sve. Ovaj članak pruža praktičan model životnog ciklusa za odlučivanje o tome šta pripada trajnoj memoriji, šta bi trebalo ponovo preuzeti, šta je bezbednije ponovo izračunati i šta bi trebalo da istekne ili bude zamenjeno.","\u002Fuploads\u002F2026\u002F09\u002Fwhat-should-an-ai-agent-remember-forget-recompute-or-retrieve-again-1790351131087-iehz28.webp","2026-09-25T09:43:00.000Z",{"id":2308,"slug":2309,"title":2310,"excerpt":2311,"featuredImage":2312,"publishedAt":2313},"471","how-to-know-whether-an-ai-agent-actually-used-the-right-evidence","Kako znati da li je AI agent zaista koristio prave dokaze","AI agent može citirati izvore i ipak koristiti pogrešne dokaze. Ovaj članak predstavlja praktičnu metodu za proveru potkrepljenosti tvrdnji, autoriteta izvora, primenjivosti, porekla i toga da li su dokazi zaista uticali na odgovor.","\u002Fuploads\u002F2026\u002F09\u002Fhow-to-know-whether-an-ai-agent-actually-used-the-right-evidence-1790351317188-o5z9ve.webp","2026-09-25T11:47:00.000Z",{"id":2315,"slug":2316,"title":2317,"excerpt":2318,"featuredImage":2319,"publishedAt":2320},"494","air-gapped-ai-how-ai-systems-work-without-internet-or-cloud-access","Air-Gapped AI: Kako AI sistemi funkcionišu bez interneta ili pristupa oblaku","Air-gapped AI pokreće modele, RAG i AI aplikacije unutar izolovanog bezbednosnog domena bez internet ili cloud zavisnosti. Saznajte kako modeli, podaci, ažuriranja i alati funkcionišu offline.","\u002Fuploads\u002F2026\u002F10\u002Fair-gapped-ai-how-ai-systems-work-without-internet-or-cloud-access-1791487983978-e6xqf0.webp","2026-10-08T11:32:00.000Z",{"id":2322,"slug":2323,"title":2324,"excerpt":2325,"featuredImage":2326,"publishedAt":2327},"467","the-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers","Granica valjanosti odgovora: Nedostajući sloj između relevantnosti i pouzdanih AI odgovora","Izvor može biti relevantan, autoritativan i ipak pogrešan za pitanje koje se postavlja. Sloj koji nedostaje je primenljivost: uslovi pod kojima odgovor važi i promene koje ga primoravaju na preispitivanje. Ovaj članak predstavlja Granicu važenja odgovora kao obrazac za dizajn izvora za ljude, AI pretragu i RAG sisteme.","\u002Fuploads\u002F2026\u002F09\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers-1790272901306-1g5jly.webp","2026-09-24T11:59:00.000Z","fallback",[],[]]