[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"portal-settings:stajic:sr":3,"public-menus:all":38,"post:what-is-context-engineering-what-the-model-receives-before-it-answers:sr":205,"related:post:what-is-context-engineering-what-the-model-receives-before-it-answers:sr:1":3035},{"statusCode":4,"data":5,"message":37},200,{"tenantId":6,"lang":7,"defaultLang":8,"siteUrl":9,"contactEmail":10,"brandName":11,"logoUrl":12,"siteName":11,"siteDescription":13,"ogImage":10,"robotsIndex":14,"socialLinks":10,"reservedSlugs":10,"seoPolicy":15},"stajic","sr","de","https:\u002F\u002Fstajic.de",null,"Stajic Platform","\u002FLogo_Planet.svg","Stajic Portal",true,{"branding":16,"relatedContent":17,"crossDomainLinks":18},{"logoUrl":12},{"enabled":14},[19,22,25,28,31,34],{"url":20,"label":21,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Ffigure.rocks","figure.rocks",{"url":23,"label":24,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Floving.rocks","loving.rocks",{"url":26,"label":27,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.com","bazify.com",{"url":29,"label":30,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.de","bazify.de",{"url":32,"label":33,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.at","bazify.at",{"url":35,"label":36,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.ba","bazify.ba","Portal settings resolved",[39,45],{"id":40,"name":41,"location":42,"isActive":14,"isDefault":43,"items":44},1,"main-navigation","header",false,[],{"id":46,"name":47,"location":48,"isActive":14,"isDefault":14,"items":49},4,"main-menu","sidebar",[50,66,79,93,103,118,133],{"id":51,"title":52,"url":60,"target":61,"icon":62,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":64,"portfolioId":10,"children":65},"item-18",{"de":53,"en":54,"es":55,"fr":56,"it":54,"ru":57,"sr":58,"zh":59},"Startseite","Home","Inicio","Accueil","Главная","Почетна","首页","\u002Ffull-stack-web-developer-munich-performance-seo-and-maintainable-builds","_self","i-lucide-home","page",111,[],{"id":67,"title":68,"url":75,"target":61,"icon":76,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":77,"portfolioId":10,"children":78},"item-22",{"de":69,"en":69,"es":70,"fr":69,"it":71,"ru":72,"sr":73,"zh":74},"Vision","Visión","Visione","Видение","Визија","想象","\u002Fueber-uns-webdesign-muenchen-webaplikation","i-lucide-eye",113,[],{"id":80,"title":81,"url":89,"target":61,"icon":90,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":91,"portfolioId":10,"children":92},"item-19",{"de":82,"en":83,"es":84,"fr":83,"it":85,"ru":86,"sr":87,"zh":88},"Leistungen","Services","Servicios","Servizi","Услуги","Услуге","服务","\u002Fservices-dienstleistungen-muenchen","i-lucide-wrench",116,[],{"id":94,"title":95,"url":99,"target":61,"icon":100,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":101,"portfolioId":10,"children":102},"item-23",{"de":96,"en":96,"es":96,"fr":96,"it":96,"ru":97,"sr":97,"zh":98},"Blog","Блог","博客","\u002Fblog","i-lucide-book-open",112,[],{"id":104,"title":105,"url":114,"target":61,"icon":115,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":116,"portfolioId":10,"children":117},"item-32",{"de":106,"en":107,"es":108,"fr":109,"it":110,"ru":111,"sr":112,"zh":113},"Neue Technologien","New Technologies","Nuevas tecnologías","Nouvelles technologies","Nuove tecnologie","Новые технологии","Нове технологије","新技术！","\u002Fneue-webtechnologien","i-lucide-sparkles",122,[],{"id":119,"title":120,"url":129,"target":61,"icon":130,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":131,"portfolioId":10,"children":132},"item-20",{"de":121,"en":122,"es":123,"fr":124,"it":125,"ru":126,"sr":127,"zh":128},"Kontakt","Contact us!","Contacto","Contact","Contatto","Контакт","Контактирајте нас","联系我们！","\u002Fcontact","i-lucide-mail",115,[],{"id":134,"title":135,"url":144,"target":61,"icon":145,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":147},"item-21",{"de":136,"en":137,"es":138,"fr":139,"it":140,"ru":141,"sr":142,"zh":143},"Unsere Arbeit","Our Work","Nuestro trabajo","Nos réalisations","I nostri lavori","Наши работы","Наши радови","文件夹","\u002Fportfolio","i-lucide-briefcase",114,[148,161,175,181,193],{"id":149,"title":150,"url":144,"target":61,"icon":159,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":160},"item-24",{"de":151,"en":152,"es":153,"fr":154,"it":155,"ru":156,"sr":157,"zh":158},"Alle Projekte","All Projects","Todos los proyectos","Tous les projets","Tutti i progetti","Все проекты","Сви пројекти","所有项目","i-lucide-grid-3x3",[],{"id":162,"title":163,"url":171,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":174},"item-29",{"de":164,"en":165,"es":166,"fr":167,"it":168,"ru":169,"sr":170,"zh":143},"Local Roots, Global Reach","Local Roots - Global Reach","Empresa local ","Entreprise locale","Azienda locale","Местная компания","Локално предузеће глобално тржиште","\u002Fportfolio\u002Flocal-roots-global-reach-communication-media-systems-for-modern-business","i-lucide-folder","custom",[],{"id":176,"title":177,"url":179,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":180},"item-28",{"de":178,"en":178,"es":178,"fr":178,"it":178,"ru":178,"sr":178,"zh":178},"Solr Suggester","\u002Fportfolio\u002Fsolr-fuzzy-suggester-und-solr-infix-suggester-abfrage-ueber-ajax-und-filterung",[],{"id":182,"title":183,"url":191,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":192},"item-27",{"de":184,"en":185,"es":186,"fr":187,"it":188,"ru":189,"sr":190,"zh":185},"Firmenwebseite SEO","Company Website SEO","Sitio web corporativo SEO","Site web d’entreprise SEO","Sito web aziendale SEO","Корпоративный сайт SEO","Пословна веб-страница SEO","\u002Fportfolio\u002Fseo-sem-branding-mobile-webseite-muenchen",[],{"id":194,"title":195,"url":203,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":204},"item-31",{"de":196,"en":197,"es":198,"fr":199,"it":200,"ru":201,"sr":202,"zh":197},"Digitalisierungsportal","Digitalization Portal","Portal de digitalización","Portail de numérisation","Portale di digitalizzazione","Портал цифровизации","Портал за дигитализацију","\u002Fportfolio\u002Fdigitalisierungsportal-archiv-museum-bibliothek-ead-lido-mets-mods",[],{"statusCode":4,"data":206,"message":3034},{"id":207,"title":208,"slug":209,"content":210,"contentJson":211,"excerpt":1409,"featuredImage":1410,"featuredImageAlt":1411,"featuredImageCaption":10,"featuredImageTitle":10,"featuredImageCopyright":10,"featuredImageAuthor":10,"featuredImageSourceUrl":10,"featuredImageLicense":10,"featuredImageIsAiGenerated":43,"status":1412,"publishedAt":1413,"createdAt":1414,"updatedAt":1415,"seoLocalePaths":1416,"categories":1425,"author":1438,"translations":1443},"488","Šta je kontekstualno inženjerstvo? Šta model prima pre nego što odgovori","what-is-context-engineering-what-the-model-receives-before-it-answers","\u003Cp>Inženjering konteksta je dizajniranje toga koje informacije jezički model prima u trenutku zaključivanja, u kom obliku, u kom redosledu i koliko dugo. Širi je od inženjeringa upita jer kontekst modela može uključivati sistemska uputstva, korisničke poruke, preuzete dokumente, rezultate alata, memoriju, trenutno stanje aplikacije, primere, strukturirane podatke i međurezultate. Cilj nije da se maksimizira broj tokena, već da se konstruiše najmanji korisni kontekst koji čuva informacije, ograničenja i dokaze potrebne za trenutni zadatak.\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--info my-6 rounded-xl border p-5 border-blue-300 bg-blue-50 dark:border-blue-900 dark:bg-blue-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">Direktan odgovor\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">Inženjering upita pita \u003Cstrong>kako treba da uputimo model?\u003C\u002Fstrong> Inženjering konteksta pita \u003Cstrong>šta model treba da zna upravo sada i kako te informacije treba sastaviti?\u003C\u002Fstrong>\u003Cbr>\u003Cbr>Preuzimanje, memorija, upravljanje stanjem, dizajn alata, skraćivanje istorije, kompakcija i redosled su stoga mehanizmi inženjeringa konteksta kada određuju tokene dostupne modelu pre nego što proizvede sledeći izlaz.\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Caside class=\"editorjs-callout editorjs-callout--warning my-6 rounded-xl border p-5 border-amber-300 bg-amber-50 dark:border-amber-900 dark:bg-amber-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">Kontekst nije isto što i znanje ili memorija\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">Sistem može nešto da zna, a da to ne postavi u trenutni kontekst. Može da pamti nešto izvan prozora modela. Može da preuzme dokument, ali da ga kasnije isključi iz konačnog upita. Model može direktno da koristi samo kontekst koji stigne do trenutnog zaključivanja.\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Caside class=\"editorjs-callout editorjs-callout--note my-6 rounded-xl border p-5 border-gray-300 bg-gray-50 dark:border-gray-700 dark:bg-gray-900\u002F40\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">Napomena o aktuelnim izvorima — 8. oktobar 2026.\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">Inženjering konteksta je sada ustaljena praktična terminologija u vodećim smernicama za inženjering veštačke inteligencije, ali nije jedinstveni formalni standard sa jednom obaveznom arhitekturom. Anthropic ga opisuje kao odabir i održavanje optimalnog skupa tokena za zaključivanje; OpenAI-ove aktuelne smernice za agente tretiraju kontekst sesije, skraćivanje i kompresiju kao eksplicitne inženjerske brige za dugotrajne sisteme.\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Cnav class=\"editorjs-toc\" data-editorjs-toc=\"true\" aria-label=\"Sadržaj\">\u003Cstrong class=\"editorjs-toc__title\">Sadržaj\u003C\u002Fstrong>\u003Col class=\"editorjs-toc__list editorjs-toc__list--depth-0\">\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-6\" class=\"editorjs-toc__link\">Šta inženjering konteksta zaista znači\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-10\" class=\"editorjs-toc__link\">Najjednostavniji primer\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-15\" class=\"editorjs-toc__link\">Gde se jednostavan primer zaustavlja\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-18\" class=\"editorjs-toc__link\">Šta može da uđe u kontekst modela?\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-20\" class=\"editorjs-toc__link\">Inženjering konteksta naspram inženjeringa upita\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-23\" class=\"editorjs-toc__link\">Inženjering konteksta naspram pretraživanja\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-28\" class=\"editorjs-toc__link\">Inženjering konteksta naspram memorije\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-33\" class=\"editorjs-toc__link\">Inženjering konteksta naspram stanja aplikacije\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-37\" class=\"editorjs-toc__link\">Dizajn alata je deo inženjeringa konteksta\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-41\" class=\"editorjs-toc__link\">Kontekst tačno na vreme naspram unapred učitanog konteksta\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-44\" class=\"editorjs-toc__link\">Kontekst je budžet, a ne sistem za skladištenje\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-48\" class=\"editorjs-toc__link\">Zašto više konteksta može biti gore\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-52\" class=\"editorjs-toc__link\">Redosled konteksta treba da bude nameran\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-56\" class=\"editorjs-toc__link\">Konfliktni kontekst zahteva eksplicitni prioritet\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-60\" class=\"editorjs-toc__link\">Sažimanje je transformacija konteksta, a ne bezgubitničko skladištenje\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-65\" class=\"editorjs-toc__link\">Sačuvaj granice validnosti\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-69\" class=\"editorjs-toc__link\">Inženjering konteksta je takođe bezbednosna granica\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-73\" class=\"editorjs-toc__link\">Praktična arhitektura za inženjering konteksta\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-77\" class=\"editorjs-toc__link\">Praktična politika izgradnje konteksta\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-79\" class=\"editorjs-toc__link\">Kako evaluirati inženjering konteksta\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-81\" class=\"editorjs-toc__link\">Sastavljanje konteksta je poseban sloj RAG neuspeha\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-85\" class=\"editorjs-toc__link\">Dokazi iz originalne implementacije\u003C\u002Fa>\u003Col class=\"editorjs-toc__list editorjs-toc__list--depth-1\">\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-86\" class=\"editorjs-toc__link\">Source of Truth Research Engine: ograničeno istraživanje umesto neograničenog konteksta\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-90\" class=\"editorjs-toc__link\">Aaasaasa AI Client: runtime, dozvole i kontekst su odvojene brige\u003C\u002Fa>\u003C\u002Fli>\u003C\u002Fol>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-96\" class=\"editorjs-toc__link\">Uobičajeni načini neuspeha u inženjeringu konteksta\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-98\" class=\"editorjs-toc__link\">Uobičajene zablude\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-100\" class=\"editorjs-toc__link\">Praktičan sled inženjeringa konteksta\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-102\" class=\"editorjs-toc__link\">Kontrolna lista za inženjering konteksta\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-104\" class=\"editorjs-toc__link\">Rubni slučajevi i ograničenja\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-110\" class=\"editorjs-toc__link\">Šta bi promenilo ovaj odgovor?\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-114\" class=\"editorjs-toc__link\">Povezano kanonsko znanje\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-119\" class=\"editorjs-toc__link\">Često postavljana pitanja\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-121\" class=\"editorjs-toc__link\">Pojmovnik\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-123\" class=\"editorjs-toc__link\">Zaključak\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-127\" class=\"editorjs-toc__link\">Primarni izvori i aktuelne smernice\u003C\u002Fa>\u003C\u002Fli>\u003C\u002Fol>\u003C\u002Fnav>\n\u003Ch2 id=\"section-6\">Šta inženjering konteksta zaista znači\u003C\u002Fh2>\n\u003Cp>Svaki poziv modela izvršava se u privremenom radnom okruženju: trenutna uputstva, poruke, preuzeti dokazi, izlazi alata i stanje koji se uklapaju u aktivni prozor konteksta. Inženjering konteksta je disciplina namernog konstruisanja tog okruženja.\u003C\u002Fp>\n\u003Cp>Ključna reč je namerno. Naivan sistem jednostavno spaja sve što ima: punu istoriju, sve preuzete dokumente, svaki odgovor alata i velike sistemske upite. Sistem sa inženjeringom konteksta odlučuje koje su informacije potrebne za trenutnu odluku, a koje treba da ostanu izvan prozora dok ne budu potrebne.\u003C\u002Fp>\n\u003Cp>To čini inženjering konteksta delom problemom informacione arhitekture, delom problemom izvršavanja i delom problemom evaluacije. Dizajn mora da odluči šta može da uđe u kontekst, odakle dolazi, koja verzija je aktuelna, kako se rešavaju sukobi, koliko detalja se zadržava i kako se rezultat testira.\u003C\u002Fp>\n\u003Ch2 id=\"section-10\">Najjednostavniji primer\u003C\u002Fh2>\n\u003Cp>Zamislite internog asistenta za podršku. Korisnik pita: „Može li ovaj klijent da otkaže bez naknade?“\u003C\u002Fp>\n\u003Cp>Modelu može biti potrebno pet stvari: aktuelna politika otkazivanja, trenutni tip ugovora klijenta, datum stupanja ugovora na snagu, relevantna pravila o izuzecima i obim ovlašćenja korisnika.\u003C\u002Fp>\n\u003Cp>Nije mu neophodna cela baza podataka klijenata, kompletan arhiv politika, svaki prethodni razgovor ili svaki tiket podrške. Inženjering konteksta je proces koji bira i sastavlja pet korisnih delova, a isključuje nepovezane informacije.\u003C\u002Fp>\n\u003Csection class=\"editorjs-process my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">Od stanja aplikacije do konteksta modela\u003C\u002Fh3>\u003Cdiv class=\"grid grid-cols-1 md:grid-cols-2 xl:grid-cols-3 gap-4\">\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">1\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">1. Razumeti zadatak\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Klasifikovati šta trenutno pitanje zahteva i koje vrste informacija mogu uticati na odgovor.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">2\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">2. Utvrditi merodavno stanje\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Pročitati trenutno stanje aplikacije ili poslovanja koje ne treba pogađati iz memorije.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">3\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">3. Preuzeti prateće znanje\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Pronaći politiku, dokumente ili spoljne dokaze relevantne za konkretan zadatak.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">4\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">4. Primeniti podobnost i dozvole\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Isključiti podatke koje trenutni korisnik ili izvršno okruženje ne smeju da izlože modelu.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">5\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">5. Smanjiti i strukturirati\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Ukloniti duplikate, izabrati korisne izvode i sačuvati ključne metapodatke, uslove i izuzetke.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">6\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">6. Poredati kontekst\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Postaviti uputstva, trenutno stanje i odlučujuće dokaze tamo gde model može dosledno da ih koristi.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">7\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">7. Izvršiti zaključivanje\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Model prima sastavljeni kontekst i proizvodi sledeći odgovor ili predlog radnje.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-15\">Gde se jednostavan primer zaustavlja\u003C\u002Fh2>\n\u003Cp>Stvarni sistemi su teži jer informacije potrebne za jedan korak možda nisu poznate pre početka izvršavanja. Agent može da otkrije nove činjenice pomoću alata, kreira međufajlove, prima promenljivo spoljno stanje ili obuhvata zadatak duži od jednog prozora konteksta.\u003C\u002Fp>\n\u003Cp>Inženjering konteksta zato postaje dinamičan. Kontekst za korak 12 ne treba da bude prosto kontekst koraka 1 plus jedanaest slojeva nagomilanog izlaza. Treba da odražava trenutno stanje zadatka, odluke koje su još važne i dokaze potrebne za sledeću radnju.\u003C\u002Fp>\n\u003Ch2 id=\"section-18\">Šta može da uđe u kontekst modela?\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Komponenta konteksta\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Svrha\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Tipičan rizik\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Sistemske \u002F razvojne instrukcije\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Definišu ulogu, ograničenja, politike i ponašanje\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Previše nejasne, kontradiktorne ili preopterećene krhkom logikom\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Trenutni zahtev korisnika\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Definiše neposredni zadatak i nameru\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Dvosmislenost ili konflikt sa prethodnom istorijom\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Istorija razgovora\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Očuvava kontinuitet kroz različite razmene\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Zastarele pretpostavke, ponavljanje i rast broja tokena\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Preuzeti dokumenti\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Pružaju eksterno znanje\u002Fdokaze\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Nerelevantnost, zastarele verzije, slab autoritet ili dupliranje\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Trenutno stanje aplikacije\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Pruža promenljive poslovne\u002Fsistemske činjenice\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Korišćenje keširanog ili zapamćenog stanja umesto trenutnog autoriteta\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Definicije alata\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Govore modelu koje sposobnosti postoje i kako ih pozvati\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Previše preklapajućih alata ili opširne šeme\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Rezultati alata\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Unose zapažanja iz okruženja u petlju\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Veliki neuredni izlazi, nepouzdan sadržaj ili zastarela zapažanja\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Memorija\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Ponovo uvodi izabrane informacije iz prethodnih interakcija\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Zastarelost, netačna generalizacija ili preterana personalizacija\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Primeri\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Demonstriraju željeno ponašanje\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Previše graničnih slučajeva može potisnuti trenutni zadatak\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Međukoraci\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Prenose planove, rezimee, kod, proračune ili beleške\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Staro međustanje može se pogrešno smatrati konačnom istinom\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Politike \u002F zaštitne mere\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Definišu zabranjeno ili ograničeno ponašanje\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Konflikt sa poslovnom logikom ili skrivene praznine u sprovođenju\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-20\">Inženjering konteksta naspram inženjeringa upita\u003C\u002Fh2>\n\u003Csection class=\"editorjs-comparison my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">Inženjering upita i inženjering konteksta rešavaju različite slojeve\u003C\u002Fh3>\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left dark:border-gray-700 dark:bg-gray-900\">\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">Inženjering upita\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">Inženjering konteksta\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Primarni fokus\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Tipičan obim\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Kada se menja\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Tipičan neuspeh\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Odnos\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Cp>Anthropic eksplicitno opisuje inženjering konteksta kao prirodan napredak inženjeringa upita za sisteme u kojima model mora da radi sa alatima, eksternim podacima, istorijom poruka i dugotrajnim stanjem agenta. Praktična razlika je korisna jer savršeno napisan upit ne može da nadoknadi nedostatak autoritativnih podataka ili kontekst zagađen kontradiktornim stanjem.\u003C\u002Fp>\n\u003Ch2 id=\"section-23\">Inženjering konteksta naspram pretraživanja\u003C\u002Fh2>\n\u003Cp>Pretraživanje bira kandidatske informacije iz eksternog korpusa ili izvora. Inženjering konteksta odlučuje šta se dešava nakon i oko tog pretraživanja.\u003C\u002Fp>\n\u003Cp>Pretraživač može da vrati 30 pasusa. Reranker ih može svesti na 10. Sloj konteksta može izabrati četiri pasusa, ukloniti duplikate, priložiti metapodatke o izvoru\u002Fverziji, kombinovati ih sa trenutnim stanjem aplikacije i postaviti ih nakon sistemskih instrukcija.\u003C\u002Fp>\n\u003Cp>Zato RAG sistem može da pronađe tačan pasus i ipak odgovori loše: neuspeh može nastati tokom sastavljanja konteksta, a ne tokom pretraživanja.\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--success my-6 rounded-xl border p-5 border-emerald-300 bg-emerald-50 dark:border-emerald-900 dark:bg-emerald-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">Pretraživanje pronalazi kandidate; inženjering konteksta konstruiše ulaz modela\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">Tačan rezultat pretraživanja je koristan samo ako preživi filtriranje, redosled, kompresiju i odluke o budžetu tokena i zaista stigne do modela u upotrebljivom obliku.\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-28\">Inženjering konteksta naspram memorije\u003C\u002Fh2>\n\u003Cp>Memorija je informacija sačuvana izvan neposrednog poziva modela kako bi se kasnije ponovo koristila. Kontekst je informacija koja je stvarno učitana u trenutni poziv.\u003C\u002Fp>\n\u003Cp>Memorijski sistem može sadržati hiljade činjenica, beleški ili prethodnih odluka. Inženjering konteksta bira koje od njih treba ponovo uvesti za trenutni zadatak. Učitavanje cele memorije u svakom koraku obesmišljava svrhu postojanja eksternog memorijskog sloja.\u003C\u002Fp>\n\u003Cp>Razlika postaje ključna za promenljivo stanje. Zapamćen status projekta ili korisnička preferencija mogu biti korisni, ali trenutno autoritativno stanje može biti potrebno ponovo pročitati pre odluke sa posledicama.\u003C\u002Fp>\n\u003Caside class=\"editorjs-referral my-6\">\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fsr\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context\" class=\"flex flex-col sm:flex-row gap-4 rounded-xl border border-gray-200 dark:border-gray-700 p-4 transition hover:border-primary-500\">\u003Cdiv class=\"min-w-0 flex-1\">\u003Cstrong class=\"block text-lg text-gray-900 dark:text-gray-100\">Memorija AI agenta nije RAG: Kako razdvojiti memoriju, pretraživanje, stanje i kontekst\u003C\u002Fstrong>\u003Cp class=\"mt-2 text-sm text-gray-600 dark:text-gray-300\">Praktična arhitektura koja razdvaja šta se čuva, šta je trenutno autoritativno, šta se pretražuje i šta model zaista prima.\u003C\u002Fp>\u003Cspan class=\"mt-3 inline-flex text-sm font-medium text-primary-600 dark:text-primary-400\">Pročitajte članak o arhitekturi memorije →\u003C\u002Fspan>\u003C\u002Fdiv>\u003C\u002Fa>\u003C\u002Faside>\n\u003Ch2 id=\"section-33\">Inženjering konteksta naspram stanja aplikacije\u003C\u002Fh2>\n\u003Cp>Stanje aplikacije je trenutno stanje spoljnog sistema: stanje računa, status tiketa, verzija fajla, faza radnog toka, stanje implementacije ili napredak zadatka.\u003C\u002Fp>\n\u003Cp>Stanje može biti sažeto u kontekst, ali sažetak nije samo stanje. Za operacije sa posledicama, izvršno okruženje može morati ponovo pročitati autoritativni sistem neposredno pre akcije, umesto da veruje ranijem snimku vidljivom modelu.\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--warning my-6 rounded-xl border p-5 border-amber-300 bg-amber-50 dark:border-amber-900 dark:bg-amber-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">Kontekst je snimak\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">Kada se stanje jednom kopira u upit, može postati zastarelo. Inženjering konteksta mora definisati kada promenljivo stanje treba osvežiti i koje operacije zahtevaju novo autoritativno čitanje.\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-37\">Dizajn alata je deo inženjeringa konteksta\u003C\u002Fh2>\n\u003Cp>Alati ne daju agentima samo sposobnosti. Imena alata, opisi, šeme i rezultati postaju informacije vidljive modelu koje oblikuju odluke.\u003C\u002Fp>\n\u003Cp>Anthropic-ove trenutne smernice za inženjering konteksta naglašavaju alate efikasne po pitanju tokena i upozoravaju na pretrpane skupove alata sa preklapajućom funkcionalnošću. Katalog alata koji je teško razlikovati čoveku takođe je teško pouzdano usmeravati modelu.\u003C\u002Fp>\n\u003Cp>Izlazi alata takođe zahtevaju disciplinu konteksta. Vraćanje celog loga od 20.000 linija kada je agent zatražio jedan uslov greške troši pažnju i može da zatrpa odlučujući dokaz.\u003C\u002Fp>\n\u003Ch2 id=\"section-41\">Kontekst tačno na vreme naspram unapred učitanog konteksta\u003C\u002Fh2>\n\u003Csection class=\"editorjs-comparison my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">Dva načina da se obezbedi informacija\u003C\u002Fh3>\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left dark:border-gray-700 dark:bg-gray-900\">\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">Unapred učitani kontekst\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">Kontekst tačno na vreme\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Metod\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Prednost\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Rizik\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Korisno kada\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Cp>Anthropic opisuje hibridni obrazac u kojem je neki stabilan kontekst unapred učitan dok agenti preuzimaju dodatne informacije u vreme izvršavanja. Ovo je koristan arhitektonski obrazac jer nije svaka važna činjenica zaslužuje trajno prebivalište u kontekstnom prozoru.\u003C\u002Fp>\n\u003Ch2 id=\"section-44\">Kontekst je budžet, a ne sistem za skladištenje\u003C\u002Fh2>\n\u003Cp>Kontekstni prozor definiše kapacitet. On ne garantuje da će svaki token biti jednako dobro iskorišćen. Model mora da rasporedi pažnju na instrukcije, istoriju, dokaze, alate i međustanje.\u003C\u002Fp>\n\u003Cp>Praktični cilj stoga nije „napuniti prozor“. Cilj je maksimizirati korisnost ograničenog budžeta pažnje.\u003C\u002Fp>\n\u003Cp>Anthropic formuliše sličan princip kao pronalaženje najmanjeg skupa tokena sa visokim signalom koji maksimizira verovatnoću željenog ponašanja. OpenAI-ove smernice za upravljanje kontekstom takođe upozoravaju da neobrađena istorija, redundantni rezultati alata i bučno preuzimanje mogu da preplave čak i velike prozore.\u003C\u002Fp>\n\u003Ch2 id=\"section-48\">Zašto više konteksta može biti gore\u003C\u002Fh2>\n\u003Cp>Dodatni kontekst može da unese irelevantne informacije, zastarelo stanje, duplirane dokaze, kontradiktorne instrukcije ili poziciono takmičenje. Takođe može da navede sisteme sažimanja da odbace detalje koji kasnije postanu važni.\u003C\u002Fp>\n\u003Cp>Klasična studija Izgubljeni u sredini pokazala je da modeli sa dugim kontekstom mogu da koriste informacije različito u zavisnosti od toga gde se relevantni sadržaj pojavljuje, pri čemu performanse često opadaju kada se odlučujuća informacija postavi u sredinu dugih ulaza.\u003C\u002Fp>\n\u003Cp>To ne znači da je dug kontekst inherentno loš. To znači da dostupnost unutar prozora nije isto što i pouzdano korišćenje.\u003C\u002Fp>\n\u003Ch2 id=\"section-52\">Redosled konteksta treba da bude nameran\u003C\u002Fh2>\n\u003Cp>Izgradnja konteksta je takođe problem redosleda. Kritične instrukcije, trenutno stanje, odlučujući dokazi i ograničenja specifična za zadatak ne bi trebalo proizvoljno nadovezivati.\u003C\u002Fp>\n\u003Cp>Ne postoji univerzalno savršen redosled za svaki model i zadatak. Arhitektura stoga treba da testira da li promena redosleda dokaza menja tačnost i da li važne informacije ostaju robusne kroz realistične varijacije konteksta.\u003C\u002Fp>\n\u003Cp>Stabilan odgovor koji se dramatično menja kada dva jednako validna odlomka zamene pozicije ukazuje na osetljivost na kontekst koju treba meriti, a ne ignorisati.\u003C\u002Fp>\n\u003Ch2 id=\"section-56\">Konfliktni kontekst zahteva eksplicitni prioritet\u003C\u002Fh2>\n\u003Cp>Model može primiti staru politiku i novu politiku, zapamćenu preferenciju i trenutnu eksplicitnu instrukciju, ili keširani status i rezultat API-ja uživo. Sistem ne treba da očekuje da model zaključi prioritet iz stila proze.\u003C\u002Fp>\n\u003Cp>Inženjering konteksta treba da kodira prioritet kroz selekciju izvora, metapodatke, redosled ili eksplicitne instrukcije: trenutno autoritativno stanje nadjačava zastarele kopije; eksplicitna trenutna korisnička instrukcija nadjačava stariju zaključenu preferenciju; odobrena politika zamenjuje zastarele nacrte.\u003C\u002Fp>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Konflikt\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Preferentno pravilo konteksta\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Trenutno stanje vs zapamćeno stanje\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Osveži i preferiraj autoritativni trenutni izvor.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Trenutna politika vs zamenjena politika\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Uključi trenutnu verziju; zadrži staru verziju samo kada je potrebno istorijsko poređenje.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Eksplicitna korisnička instrukcija vs stara zaključena preferencija\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Preferiraj trenutnu eksplicitnu instrukciju.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Primarni izvor vs sekundarni rezime\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Koristi primarni izvor za tvrdnje koje zahtevaju autoritet; rezime može podržati objašnjenje.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Zapažanje alata vs prethodno uverenje modela\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Preferiraj trenutno zapaženo stanje kada je alat autoritativan za tu činjenicu.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Dva nerešena autoritativna izvora\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Izloži konflikt umesto fabrikovanja jednog konzistentnog odgovora.\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-60\">Sažimanje je transformacija konteksta, a ne bezgubitničko skladištenje\u003C\u002Fh2>\n\u003Cp>Dugotrajni sistemi na kraju moraju da skrate, sažmu ili kompaktuju istoriju. Kompaktovanje stvara novu reprezentaciju prethodnog konteksta kako bi agent mogao da nastavi bez reprodukovanja svakog tokena.\u003C\u002Fp>\n\u003Cp>OpenAI-ovi primeri upravljanja kontekstom koriste skraćivanje i kompresiju za dugotrajne sesije. Anthropic opisuje kompaktovanje kao primarnu tehniku za održavanje koherentnosti kada se interakcija približava granici konteksta.\u003C\u002Fp>\n\u003Cp>Težak deo je odlučivanje šta se ne može bezbedno ukloniti: nerešeni zadaci, identifikatori, korisnička ograničenja, bezbednosne granice, arhitekturne odluke, izuzeci, poreklo izvora i uslovi koji čine prethodni zaključak validnim.\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--warning my-6 rounded-xl border p-5 border-amber-300 bg-amber-50 dark:border-amber-900 dark:bg-amber-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">Rezime može sačuvati zaključak i uništiti razlog\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">Ako kompaktovanje zadrži „koristi pristup X“ ali odbaci zašto je X izabran, koja verzija je testirana ili koji uslov bi ga učinio nevažećim, kasniji odgovori mogu ostati interno konzistentni dok postaju eksterno netačni.\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-65\">Sačuvaj granice validnosti\u003C\u002Fh2>\n\u003Cp>Važni zaključci treba da nose uslove pod kojima ostaju podržani: verziju, datum, obim, pretpostavke, autoritet izvora i nerešeno neslaganje.\u003C\u002Fp>\n\u003Cp>Inženjering konteksta je stoga povezan sa granicom validnosti odgovora. Sastavljač konteksta ne treba da uklanja metapodatke koji određuju da li dokazi još uvek važe.\u003C\u002Fp>\n\u003Caside class=\"editorjs-referral my-6\">\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fsr\u002Fblog\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers\" class=\"flex flex-col sm:flex-row gap-4 rounded-xl border border-gray-200 dark:border-gray-700 p-4 transition hover:border-primary-500\">\u003Cdiv class=\"min-w-0 flex-1\">\u003Cstrong class=\"block text-lg text-gray-900 dark:text-gray-100\">Granica validnosti odgovora: sloj koji nedostaje između relevantnosti i pouzdanih AI odgovora\u003C\u002Fstrong>\u003Cp class=\"mt-2 text-sm text-gray-600 dark:text-gray-300\">Okvir za očuvanje obima, pretpostavki, verzija i uslova dokaza pod kojima AI tvrdnja ostaje podržana.\u003C\u002Fp>\u003Cspan class=\"mt-3 inline-flex text-sm font-medium text-primary-600 dark:text-primary-400\">Pročitajte granicu validnosti odgovora →\u003C\u002Fspan>\u003C\u002Fdiv>\u003C\u002Fa>\u003C\u002Faside>\n\u003Ch2 id=\"section-69\">Inženjering konteksta je takođe bezbednosna granica\u003C\u002Fh2>\n\u003Cp>Podaci koji stignu do modela prešli su važnu sistemsku granicu. Sastavljanje konteksta stoga mora poštovati autorizaciju, izolaciju zakupaca, poverljivost i pravila minimizacije podataka.\u003C\u002Fp>\n\u003Cp>Pretraživač može tehnički pronaći odlomak kojem trenutni korisnik ne može pristupiti. Ispravan dizajn je sprečiti da taj odlomak uđe u kontekst modela, umesto da se oslanja na to da će ga model ignorisati.\u003C\u002Fp>\n\u003Cp>Izlazi alata takođe mogu sadržati nepouzdane instrukcije ili neprijateljski sadržaj. Inženjering konteksta treba da očuva razliku između instrukcija aplikacije i eksternih podataka kako pronađeni tekst ne bi mogao tiho steći autoritet instrukcije.\u003C\u002Fp>\n\u003Ch2 id=\"section-73\">Praktična arhitektura za inženjering konteksta\u003C\u002Fh2>\n\u003Caside class=\"editorjs-callout editorjs-callout--note my-6 rounded-xl border p-5 border-gray-300 bg-gray-50 dark:border-gray-700 dark:bg-gray-900\u002F40\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">Predloženi arhitektonski model\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">Sledeći slojevi predstavljaju praktičnu sintezu za produkcione sisteme, a ne formalni industrijski standard. Cilj je da se vlasništvo nad informacijama odvoji od privremenog konteksta koji je okrenut modelu.\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Sloj\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Odgovornost\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Autoritativni sistemi\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Vlasnici trenutnog poslovnog\u002Fsistemskog stanja i zvaničnih zapisa.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Izvori znanja\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Vlasnici dokumenata, politika, specifikacija, istraživanja ili eksternih dokaza.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Skladište memorije\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Čuva odabrane informacije kroz različite interakcije ili sesije.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Sloj za pronalaženje\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Pronalazi kandidate relevantne za zadatak iz eksternih izvora.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Sloj alata\u002Fruntime-a\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Čita stanje, izvršava radnje i vraća zapažanja.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Sastavljač konteksta\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Bira, filtrira, uklanja duplikate, raspoređuje i formatira informacije vidljive modelu.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Model\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Razmišlja i generiše na osnovu sastavljenog konteksta.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Validacija\u002Fevaluacija\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Proverava da li odabrani kontekst i rezultujući izlaz zadovoljavaju zahteve specifične za zadatak.\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Cp>Sastavljač konteksta je konceptualno važan čak i kada nijedan modul ne nosi tačno taj naziv. U maloj aplikaciji to može biti običan aplikativni kod. U velikoj agentskoj platformi može kombinovati upravljanje sesijama, pronalaženje, memoriju, middleware za alate, kompakciju i sprovođenje politika.\u003C\u002Fp>\n\u003Ch2 id=\"section-77\">Praktična politika izgradnje konteksta\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Pravilo\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Zašto je važno\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Počnite od trenutnog zadatka\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Ne nosite informacije samo zato što su postojale ranije.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Ponovo pročitajte promenljivo stanje\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Memorija i stari kontekst mogu biti zastareli.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Pronađite samo dovoljno dokaza\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Veliki skupovi kandidata mogu razblažiti odlučujuće informacije.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Sačuvajte metapodatke izvora\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Verzija, datum i autoritet određuju da li dokaz još uvek važi.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Uklonite duplirani sadržaj\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Redundantnost troši tokene bez dodavanja informacija.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Preferirajte strukturisane sažetke za veliki izlaz alata\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Izložite odlučujuća polja umesto sirovog šuma gde to vernost dozvoljava.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Čuvajte pravila zajedno sa izuzecima\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Razdvajanje pravila od njegovog izuzetka stvara lažnu sigurnost.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Učinite prioritet eksplicitnim\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Ne tražite od modela da zaključi koji konfliktni izvor pobeđuje.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Čuvajte trajno stanje izvan konteksta\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Kontekst je privremena radna memorija, a ne baza podataka.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Kompaktirajte uz testove zadržavanja\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Proverite da li identifikatori, ograničenja, poreklo i nerešeno stanje preživljavaju.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Izmerite osetljivost na redosled\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Ispravnost ne bi trebalo slučajno da zavisi od proizvoljnog redosleda dokumenata.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Evaluacija konteksta odvojeno od kvaliteta modela\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Jači model ne može pouzdano da kompenzuje nedostajuće ili neovlašćene dokaze.\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-79\">Kako evaluirati inženjering konteksta\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Svojstvo\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Pitanje\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Primer testa\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Dovoljnost\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Da li kontekst sadrži sve što je potrebno za rešavanje zadatka?\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Uklonite jedan dokaz i posmatrajte da li odgovor postaje nepotkrepljen.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Relevantnost\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Koliko je konteksta nepotrebno za zadatak?\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Izmerite kvalitet dok se irelevantni odlomci dodaju ili uklanjaju.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Autoritet\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Da li su odlučujuće tvrdnje zasnovane na ispravnoj klasi izvora?\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Ubacite fluentniji ali neautoritativni konfliktni izvor.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Svežina\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Da li trenutno stanje nadjačava zastarele kopije?\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Promenite autoritativno stanje nakon prethodne interakcije i ponovo pokrenite.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Robusnost pozicije\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Da li kvalitet odgovora snažno zavisi od pozicije dokaza?\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Randomizujte redosled kandidata kroz ponovljene pokušaje.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Rešavanje konflikata\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Da li model prati eksplicitna pravila prioriteta?\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Prikažite staro i novo stanje zajedno.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Zadržavanje pri kompakciji\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Da li sažimanje čuva ograničenja i granice validnosti?\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Uporedite performanse zadatka pre i posle kompakcije.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Efikasnost tokena\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Da li dodatni kontekst dovoljno poboljšava kvalitet da opravda latenciju\u002Ftrošak?\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Izvršite kontrolisane ablacije veličine konteksta.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Bezbednost\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Mogu li neovlašćeni ili adversarieski sadržaj ući u kontekst modela?\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Testirajte granice zakupca, dozvola i prompt-injection-a.\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-81\">Sastavljanje konteksta je poseban sloj RAG neuspeha\u003C\u002Fh2>\n\u003Cp>RAG pipeline može uspešno da izvrši pronalaženje i ipak ne uspe dalje. Relevantni izvor se može pojaviti na rangu 2, ali sastavljač konteksta ga može izostaviti, skratiti, kombinovati sa zastarelim kontradiktornim materijalom ili prekoračiti budžet tokena.\u003C\u002Fp>\n\u003Cp>Zato tragove pronalaženja treba uporediti sa stvarnim kontekstom poslatim modelu. Bez tog poređenja, neuspesi konteksta se lako pogrešno dijagnostikuju kao neuspesi embedding-a ili modela.\u003C\u002Fp>\n\u003Caside class=\"editorjs-referral my-6\">\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fsr\u002Fblog\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method\" class=\"flex flex-col sm:flex-row gap-4 rounded-xl border border-gray-200 dark:border-gray-700 p-4 transition hover:border-primary-500\">\u003Cdiv class=\"min-w-0 flex-1\">\u003Cstrong class=\"block text-lg text-gray-900 dark:text-gray-100\">RAG nije uspeo — ali koji sloj je zapravo zakazao? Dijagnostička metoda\u003C\u002Fstrong>\u003Cp class=\"mt-2 text-sm text-gray-600 dark:text-gray-300\">Pristup sloj po sloj za razdvajanje neuspeha pokrivenosti izvora, pronalaženja, rangiranja, sastavljanja konteksta, generisanja, atribucije dokaza i svežine.\u003C\u002Fp>\u003Cspan class=\"mt-3 inline-flex text-sm font-medium text-primary-600 dark:text-primary-400\">Pročitajte RAG dijagnostičku metodu →\u003C\u002Fspan>\u003C\u002Fdiv>\u003C\u002Fa>\u003C\u002Faside>\n\u003Ch2 id=\"section-85\">Dokazi iz originalne implementacije\u003C\u002Fh2>\n\u003Ch3 id=\"section-86\">Source of Truth Research Engine: ograničeno istraživanje umesto neograničenog konteksta\u003C\u002Fh3>\n\u003Cp>Source of Truth Research Engine razdvaja otkrivanje, prikupljanje, ekstrakciju, verifikaciju, analizu kontradikcija i sintezu u ograničene istraživačke faze, umesto da jedan ogroman istraživački zadatak i sav prikupljeni materijal pošalje u jedan jedini poziv modelu.\u003C\u002Fp>\n\u003Cp>Njegov model dokaza čuva Sources, Artifacts, Claims, Relations, Contradictions i poreklo izvan konteksta modela. Model može da primi podskup potreban za trenutni istraživački korak, dok trajni dokazi ostaju u eksternom skladištu.\u003C\u002Fp>\n\u003Cp>To je konkretan obrazac inženjeringa konteksta: trajno istraživačko stanje živi izvan prozora modela; aktivni kontekst modela se rekonstruiše za trenutnu fazu.\u003C\u002Fp>\n\u003Ch3 id=\"section-90\">Aaasaasa AI Client: runtime, dozvole i kontekst su odvojene brige\u003C\u002Fh3>\n\u003Cp>Aaasaasa AI klijent razdvaja izbor provajdera\u002Fmodela, lokaciju izvršavanja, dozvole radnog prostora, lokalne resurse i pristup alatima. To sprečava da kontekst modela postane vlasnik autorizacije ili stanja aplikacije.\u003C\u002Fp>\n\u003Cp>Direktan razgovor i agentska okruženja za izvršavanje mogu imati različite mogućnosti alata. Profili dozvola radnog prostora se sprovode od strane okruženja za izvršavanje, a ne samo opisuju u kontekstu prirodnog jezika. Ova razlika je važna: kontekst može reći modelu šta bi trebalo da radi, dok okruženje za izvršavanje i dalje mora da sprovede ono što mu je zaista dozvoljeno da radi.\u003C\u002Fp>\n\u003Cp>Dokaz implementacije ovde je arhitektonsko razdvajanje, a ne tvrdnja da je svaka napredna tehnika upravljanja kontekstom opisana u ovom članku već implementirana.\u003C\u002Fp>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Obrazac implementacije\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Lekcija o inženjeringu konteksta\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Eksterno skladište dokaza\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Trajno znanje ne mora da ostane u prozoru modela.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Ograničene faze istraživanja\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Različiti koraci mogu da dobiju različit kontekst umesto da akumuliraju jednu ogromnu istoriju.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Tvrdnje + poreklo van konteksta\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Identitet dokaza nadživljava privremeno stanje zaključivanja.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Dozvole koje sprovodi okruženje za izvršavanje\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Bezbednosni autoritet ne zavisi od toga da se model seća instrukcije.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Razdvojeni koncepti lokalnog\u002Fprovajdera\u002Fmodela\u002Fokruženja za izvršavanje\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Kontekst je samo jedan sloj šire arhitekture AI aplikacije.\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Caside class=\"editorjs-callout editorjs-callout--note my-6 rounded-xl border p-5 border-gray-300 bg-gray-50 dark:border-gray-700 dark:bg-gray-900\u002F40\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">Granica dokaza\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">Ove implementacije podržavaju arhitektonsko razdvajanje između trajnog stanja, pronalaženja, kontrola okruženja za izvršavanje i konteksta okrenutog modelu. One nisu predstavljene kao dokaz putem referentnih vrednosti da je jedna strategija konteksta univerzalno optimalna.\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-96\">Uobičajeni načini neuspeha u inženjeringu konteksta\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Način neuspeha\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Šta pođe naopako\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Beskonačno ponavljanje celog razgovora\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Stare pretpostavke, ponavljanje i rast tokena nadvladaju trenutnu nameru.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Stavljanje svakog pronađenog rezultata u upit\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Šum, dupliranje i konfliktne verzije razblažuju odlučujuće dokaze.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Korišćenje memorije kao trenutnog stanja\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Zastarele informacije tiho zamenjuju autoritativno aktivno stanje.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Vraćanje sirovog izlaza alata\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Veliki dnevnici ili odgovori troše pažnju bez dodavanja vrednosti za odluku.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Skrivanje opisa alata iza nejasnih imena\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Model ne može pouzdano da odluči koju sposobnost da koristi.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Sažimanje bez testova zadržavanja\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Kritična ograničenja, identifikatori ili izuzeci nestaju.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Mešanje instrukcija i nepouzdanih podataka\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Spoljni sadržaj može biti protumačen kao instrukcija višeg autoriteta.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Korišćenje jednog statičnog šablona konteksta za svaki zadatak\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Različiti zadaci dobijaju irelevantne informacije i propuštaju dokaze specifične za zadatak.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Ignorisanje verzije\u002Fdatuma izvora\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Zastareli ali relevantni dokazi mogu da nadvladaju trenutno autoritativno stanje.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Tretiranje većeg kontekstnog prozora kao garancije kvaliteta\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Kapacitet se povećava dok problemi pažnje i konflikata ostaju.\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-98\">Uobičajene zablude\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Zabluda\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Ispravka\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">„Inženjering konteksta je samo inženjering upita pod novim imenom.“\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Upiti su jedna komponenta; inženjering konteksta takođe pokriva pronalaženje, memoriju, stanje, rezultate alata, istoriju i sažimanje.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">„Kontekst znači istoriju razgovora.“\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Istorija je samo jedan mogući izvor konteksta.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">„Više konteksta je uvek bolje.“\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Dodatne informacije mogu smanjiti signal, uneti konflikte i povećati troškove.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">„Ako je pronalaženje našlo, model je video.“\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Pronađeni kandidati mogu biti filtrirani, skraćeni ili izostavljeni pre zaključivanja.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">„Dugačak kontekst uklanja potrebu za RAG.“\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Veliki prozori povećavaju kapacitet ali ne rešavaju svežinu, autoritet, dozvole ili dinamičko pronalaženje.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">„Memorija treba uvek da bude učitana.“\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Memoriju treba birati u skladu sa trenutnim zadatkom.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">„Sažetak čuva sve važno.“\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Sažimanje je gubitničko osim ako se eksplicitno ne proceni zadržavanje.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">„Instrukcije mogu da sprovedu dozvole.“\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Autorizaciju moraju da sprovedu kontrole okruženja za izvršavanje\u002Faplikacije, a ne samo kontekst.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">„Jedan recept za kontekst radi za svaki model.“\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Osetljivost na kontekst varira u zavisnosti od modela, zadatka, korpusa i okruženja za izvršavanje.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">„Inženjering konteksta je samo za agente.“\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Agenti pojačavaju potrebu, ali obične RAG i konverzacione aplikacije takođe zahtevaju konstrukciju konteksta.\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-100\">Praktičan sled inženjeringa konteksta\u003C\u002Fh2>\n\u003Csection class=\"editorjs-process my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">Konstruisati kontekst od trenutne odluke unazad\u003C\u002Fh3>\u003Cdiv class=\"grid grid-cols-1 md:grid-cols-2 xl:grid-cols-3 gap-4\">\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">1\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">1. Definisati sledeću odluku modela\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Odrediti na šta model mora da odgovori, klasifikuje, planira ili izabere u ovom koraku.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">2\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">2. Identifikovati potrebne činjenice i ograničenja\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Navesti minimalno stanje, pravila, dokaze i instrukcije koje mogu materijalno da promene rezultat.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">3\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">3. Rešiti autoritet i dozvole\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Odrediti koji izvori su trenutni, autoritativni i dostupni trenutnom principalu.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">4\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">4. Pronaći ili čitati na zahtev\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Pribaviti neophodne dokaze i promenljivo stanje umesto oslanjanja na zastareli kontekst.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">5\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">5. Smanjiti šum\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Ukloniti duplikate, sažeti ili izabrati odlomke bez odbacivanja odlučujućih izuzetaka ili porekla.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">6\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">6. Strukturisati i poredati\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Učiniti instrukcije, trenutno stanje, dokaze i zapažanja alata razlikovnim.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">7\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">7. Uklopiti u budžet tokena\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Preferirati kontekst visokog signala i premestiti trajne informacije van prozora.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">8\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">8. Pokrenuti model\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Izvršiti zaključivanje nad sastavljenim kontekstom.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">9\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">9. Posmatrati neuspehe\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Zabeležiti da li problem potiče od nedostajućeg, zastarelog, šumnog, konfliktnog ili loše poređanog konteksta.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">10\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">10. Ponovo proceniti nakon promena modela\u002Fokruženja za izvršavanje\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Strategija konteksta važi samo za modele, alate i radna opterećenja na kojima je testirana.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-102\">Kontrolna lista za inženjering konteksta\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Pitanje\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Očekivani odgovor\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Koju tačno odluku će model doneti sledeće?\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Ograničen zadatak, a ne nejasan dugoročni cilj.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Koje informacije mogu materijalno da promene tu odluku?\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Eksplicitni minimalni skup dokaza\u002Fstanja.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Koji podaci su sada autoritativni?\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Trenutni izvor\u002Fverzija i pravilo svežine.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Koji podaci su opcionalna pozadina?\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Odvojeni od odlučujućih dokaza.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Šta ne sme da uđe u kontekst?\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Neovlašćeni, nepotrebni ili previše osetljivi podaci.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Koje stavke memorije su relevantne?\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Izabrane prema zadatku, a ne automatski ponavljane.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Koje izlaze alata treba smanjiti?\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Veliki odgovori se transformišu u formu relevantnu za odluku.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Koja ograničenja moraju da prežive sažimanje?\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Identifikatori, izuzeci, obaveze, nerešeno stanje i poreklo.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Kako je predstavljen prioritet?\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Trenutne\u002Fautoritativne informacije mogu pouzdano da nadvladaju zastarele ili slabije izvore.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Kako ćete znati da je kontekst zakazao?\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Postoje evaluacije i tragovi specifični za kontekst.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Može li odgovor biti reprodukovan?\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Ulaz modela ili rekonstruktivni trag konteksta je dostupan gde je prikladno.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Može li jači ili veći model da promeni strategiju?\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Politika konteksta je svesna verzije i empirijski se ponovo procenjuje.\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-104\">Rubni slučajevi i ograničenja\u003C\u002Fh2>\n\u003Cp>Neki zadaci su dovoljno jednostavni da se inženjering konteksta svodi na kratak sistemski upit i jednu korisničku poruku. Dodavanje pronalaženja, memorije i sažimanja samo bi uvelo nepotrebnu arhitekturu.\u003C\u002Fp>\n\u003Cp>Neki zadaci zahtevaju visok opoziv i mogu namerno da uključe više konteksta pre kasnije sinteze. Istraživanje, otkrivanje i pravni pregled mogu da preferiraju izbegavanje izostavljanja umesto minimalnog broja tokena.\u003C\u002Fp>\n\u003Cp>Neke informacije ne bi trebalo nikada sažimati pre upotrebe. Tačni ugovori, kod, kriptografski materijal, numerički zapisi i regulatorni tekst mogu zahtevati doslovno ili strukturisano pronalaženje gde kompresija može da promeni značenje.\u003C\u002Fp>\n\u003Cp>Ponašanje dugog konteksta značajno varira između modela. Strategija validirana na jednom modelu, dužini konteksta ili okviru alata ne bi trebalo automatski da se prenosi na drugi.\u003C\u002Fp>\n\u003Cp>Model i dalje može da ignoriše ili pogrešno protumači izuzetan kontekst. Kontekstualno inženjerstvo poboljšava informaciono okruženje; ono ne garantuje ispravnost zaključivanja.\u003C\u002Fp>\n\u003Ch2 id=\"section-110\">Šta bi promenilo ovaj odgovor?\u003C\u002Fh2>\n\u003Cp>Budući modeli mogu postati robusniji na dugi kontekst, pozicione efekte i konfliktne informacije. To bi moglo smanjiti količinu ručne kuratacije koja je potrebna.\u003C\u002Fp>\n\u003Cp>Arhitektonska razlika bi i dalje ostala korisna jer dozvole, svežina, trajnost memorije, autoritet izvora i stanje eksterne aplikacije postoje izvan modela bez obzira na veličinu kontekstnog prozora.\u003C\u002Fp>\n\u003Cp>Preporučeni balans između unapred učitanog i konteksta koji se učitava po potrebi takođe se menja u zavisnosti od zahteva za latencijom, pouzdanosti alata, veličine korpusa, cene modela i toga koliko su dinamične osnovne informacije.\u003C\u002Fp>\n\u003Ch2 id=\"section-114\">Povezano kanonsko znanje\u003C\u002Fh2>\n\u003Cp>Kontekstualno inženjerstvo se nalazi između pretrage i generisanja. RAG objašnjava kako se eksterno znanje pronalazi; R01 razdvaja embedding-e, vektorsku pretragu i rerangiranje; kontekstualno inženjerstvo objašnjava šta na kraju stigne do modela.\u003C\u002Fp>\n\u003Caside class=\"editorjs-referral my-6\">\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fsr\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works\" class=\"flex flex-col sm:flex-row gap-4 rounded-xl border border-gray-200 dark:border-gray-700 p-4 transition hover:border-primary-500\">\u003Cdiv class=\"min-w-0 flex-1\">\u003Cstrong class=\"block text-lg text-gray-900 dark:text-gray-100\">Šta je RAG? Najjednostavnije objašnjenje kako funkcioniše\u003C\u002Fstrong>\u003Cp class=\"mt-2 text-sm text-gray-600 dark:text-gray-300\">Temelj pretrage za razumevanje kako se eksterno znanje može dostaviti modelu pre generisanja.\u003C\u002Fp>\u003Cspan class=\"mt-3 inline-flex text-sm font-medium text-primary-600 dark:text-primary-400\">Pročitajte osnove RAG-a →\u003C\u002Fspan>\u003C\u002Fdiv>\u003C\u002Fa>\u003C\u002Faside>\n\u003Cp>Arhitektura izvora istine odgovara na drugačije pitanje: ne koje informacije su prisutne u kontekstu, već koji izvor je ovlašćen da utvrdi tvrdnju.\u003C\u002Fp>\n\u003Cp>Postojeći članak Zašto više konteksta može pogoršati AI odgovore je dijagnostički pratilac ovoj kanonskoj definiciji. Fokusira se na zagađenje konteksta, pozicione efekte, rast top-k, gubitak kompakcije i degradaciju odgovora, umesto da redefiniše samo kontekstualno inženjerstvo.\u003C\u002Fp>\n\u003Ch2 id=\"section-119\">Često postavljana pitanja\u003C\u002Fh2>\n\u003Csection class=\"editorjs-faq my-6 rounded-xl border border-gray-200 p-5 dark:border-gray-700\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">Često postavljana pitanja o kontekstualnom inženjerstvu\u003C\u002Fh3>\u003Cdiv id=\"faq1\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">Šta je kontekstualno inženjerstvo?\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">Kontekstualno inženjerstvo je dizajn i upravljanje u vreme izvršavanja onim informacijama koje jezički model prima u trenutku inferencije, uključujući instrukcije, istoriju, pronađene dokaze, memoriju, stanje, alate i rezultate alata.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq2\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">Po čemu se kontekstualno inženjerstvo razlikuje od prompt inženjerstva?\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">Prompt inženjerstvo se fokusira na to kako su instrukcije i primeri napisani. Kontekstualno inženjerstvo uključuje promptove, ali takođe odlučuje koje eksterne informacije, stanje, istorija, memorija i zapažanja alata se postavljaju oko njih.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq3\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">Da li je RAG isto što i kontekstualno inženjerstvo?\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">Ne. RAG pronalazi eksterne informacije. Kontekstualno inženjerstvo odlučuje kako se pronađene informacije filtriraju, kombinuju sa drugim stanjem i zaista dostavljaju modelu.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq4\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">Da li je memorija isto što i kontekst?\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">Ne. Memorija čuva informacije izvan trenutnog poziva modela. Kontekst je podskup informacija učitanih u trenutnu inferenciju.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq5\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">Zašto više konteksta može pogoršati odgovor?\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">Dodatni kontekst može uneti šum, zastarelo stanje, konfliktne dokaze, dupliranje i poziciono takmičenje. Veliki kapacitet konteksta ne garantuje jednako pouzdano korišćenje svakog tokena.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq6\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">Šta je kompakcija konteksta?\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">Kompakcija sažima ili transformiše nagomilanu istoriju u manju reprezentaciju kako bi dugotrajni sistem mogao da nastavi bez ponovnog reprodukovanja svakog prethodnog tokena.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq7\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">Da li trenutno stanje aplikacije treba čuvati u kontekstu?\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">Može biti predstavljeno u kontekstu radi zaključivanja, ali operacije sa posledicama često treba ponovo da pročitaju autoritativni izvor jer snimci konteksta mogu postati zastareli.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq8\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">Da li je kontekstualno inženjerstvo potrebno samo za AI agente?\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">Ne. Agenti čine upravljanje kontekstom dinamičnijim, ali RAG sistemi, asistenti, kopiloti i aplikacije sa više krugova takođe zahtevaju namerno konstruisanje konteksta.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-121\">Pojmovnik\u003C\u002Fh2>\n\u003Csection class=\"editorjs-glossary my-6 rounded-xl border border-gray-200 dark:border-gray-700 p-5\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">Ključni pojmovi kontekstualnog inženjerstva\u003C\u002Fh3>\u003Cdl>\u003Cdiv id=\"context-engineering\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">Kontekstualno inženjerstvo\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Dizajn i upravljanje u vreme izvršavanja informacijama koje se dostavljaju jezičkom modelu za određeni korak inferencije.\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"context-window\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">Kontekstni prozor\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Konačni kapacitet tokena modela za ulaz i, u zavisnosti od interfejsa modela, povezane generisane tokene ili aktivnu sekvencu.\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"prompt-engineering\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">Prompt inženjerstvo\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Dizajn instrukcija, primera i strukture prompta koji imaju za cilj da izazovu korisno ponašanje modela.\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"context-assembly\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">Sastavljanje konteksta\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Proces odabira, filtriranja, raspoređivanja i formatiranja informacija vidljivih modelu pre inferencije.\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"just-in-time-retrieval\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">Pretraga u pravo vreme\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Dinamičko učitavanje informacija kada trenutni zadatak to zahteva, umesto unapred učitavanja svih potencijalno relevantnih podataka.\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"compaction\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">Kompakcija\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Svođenje nagomilanog konteksta na manju reprezentaciju uz nastojanje da se sačuvaju informacije potrebne za buduće korake.\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"context-pollution\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">Zagađenje konteksta\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Degradacija uzrokovana irelevantnim, zastarelim, kontradiktornim ili redundantnim informacijama koje zauzimaju radni kontekst modela.\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"application-state\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">Stanje aplikacije\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Trenutno autoritativno stanje eksternog sistema, radnog toka ili domena koje postoji nezavisno od konteksta modela.\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"memory\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">Memorija\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Informacije uskladištene izvan neposrednog poziva modela za moguću upotrebu u kasnijim krugovima ili sesijama.\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"retrieved-context\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">Pronađeni kontekst\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Eksterne informacije koje je odabrao sistem za pretragu i učinio dostupnim modelu, u celosti ili delimično.\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"position-robustness\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">Robusnost pozicije\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Stepen u kojem ispravnost modela ostaje stabilna kada se lokacija ili redosled relevantnog konteksta promene.\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"validity-boundary\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">Granica važenja\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Obim, vreme, pretpostavke, verzije i uslovi dokaza u okviru kojih zaključak ostaje potkrepljen.\u003C\u002Fdd>\u003C\u002Fdiv>\u003C\u002Fdl>\u003C\u002Fsection>\n\u003Ch2 id=\"section-123\">Zaključak\u003C\u002Fh2>\n\u003Cp>Kontekstualno inženjerstvo je sloj koji odlučuje šta model može da vidi pre nego što odgovori. To ga čini širim od promptovanja i nizvodnim od pretrage, dok istovremeno ostaje različito od trajne memorije i autoritativnog stanja aplikacije.\u003C\u002Fp>\n\u003Cp>Snažna arhitektura konteksta ne tretira kontekstni prozor kao bazu podataka. Ona čuva trajno stanje i znanje izvan modela, učitava ono što je potrebno za trenutnu odluku, čuva autoritet i poreklo, uklanja nepotreban šum i osvežava volatilne informacije kada je potrebno.\u003C\u002Fp>\n\u003Cp>Praktični cilj stoga nije maksimalan kontekst. To je minimalni dovoljan, visoko signalni, ispravno autorizovan i kontekst koji čuva važenje za sledeću odluku modela.\u003C\u002Fp>\n\u003Ch2 id=\"section-127\">Primarni izvori i aktuelne smernice\u003C\u002Fh2>\n\u003Cp>Izvori navedeni u nastavku podržavaju trenutnu terminologiju kontekstualnog inženjeringa, ponašanje dugog konteksta i operativne obrasce upravljanja kontekstom. Sekcije projekta su eksplicitno dokazi implementacije, a ne univerzalne tvrdnje.\u003C\u002Fp>\n\u003Ca href=\"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Feffective-context-engineering-for-ai-agents\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">Anthropic — Efikasan kontekstualni inženjering za AI agente\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">Zvanične inženjerske smernice koje definišu kontekstualni inženjering, pravovremeno preuzimanje, kompakciju, strukturisanu memoriju i kuriranje konteksta za agente.\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fdevelopers.openai.com\u002Fcookbook\u002Fexamples\u002Fagents_sdk\u002Fsession_memory\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">OpenAI — Kontekstualni inženjering: Upravljanje kratkoročnom memorijom pomoću sesija\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">Zvanične smernice iz kuvara znanja o upravljanju kontekstom, skraćivanju i kompresiji za dugotrajne agentske sesije.\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fagents\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">OpenAI — Vodič za agente\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">Aktuelne OpenAI smernice za programere o agentskim runtime okruženjima, kontekstu kroz korake i vlasništvu nad orkestracijom.\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2307.03172\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">Izgubljeni u sredini: Kako jezički modeli koriste duge kontekste\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">Istraživanje koje pokazuje da performanse modela sa dugim kontekstom mogu snažno zavisiti od pozicije relevantnih informacija u ulazu.\u003C\u002Fp>\u003C\u002Fa>",{"time":212,"blocks":213,"version":1408},1791480770253,[214,220,228,235,242,250,255,260,265,270,275,280,285,290,319,324,329,334,339,393,398,433,438,443,448,453,458,465,470,475,480,485,494,499,504,509,515,520,525,530,535,540,569,574,579,584,589,594,599,604,609,614,619,624,629,634,639,644,649,675,680,685,690,695,701,706,711,716,724,729,734,739,744,749,755,787,792,797,841,846,891,896,901,906,914,919,924,929,934,939,944,949,954,959,982,988,993,1031,1036,1074,1079,1115,1120,1163,1168,1173,1178,1183,1188,1193,1198,1203,1208,1213,1218,1223,1231,1236,1241,1246,1284,1289,1341,1346,1351,1356,1361,1366,1371,1381,1390,1399],{"id":215,"data":216,"type":218,"tunes":219},"intro",{"text":217},"Inženjering konteksta je dizajniranje toga koje informacije jezički model prima u trenutku zaključivanja, u kom obliku, u kom redosledu i koliko dugo. Širi je od inženjeringa upita jer kontekst modela može uključivati sistemska uputstva, korisničke poruke, preuzete dokumente, rezultate alata, memoriju, trenutno stanje aplikacije, primere, strukturirane podatke i međurezultate. Cilj nije da se maksimizira broj tokena, već da se konstruiše najmanji korisni kontekst koji čuva informacije, ograničenja i dokaze potrebne za trenutni zadatak.","paragraph",{},{"id":221,"data":222,"type":226,"tunes":227},"direct",{"body":223,"title":224,"variant":225},"Inženjering upita pita \u003Cstrong>kako treba da uputimo model?\u003C\u002Fstrong> Inženjering konteksta pita \u003Cstrong>šta model treba da zna upravo sada i kako te informacije treba sastaviti?\u003C\u002Fstrong>\u003Cbr>\u003Cbr>Preuzimanje, memorija, upravljanje stanjem, dizajn alata, skraćivanje istorije, kompakcija i redosled su stoga mehanizmi inženjeringa konteksta kada određuju tokene dostupne modelu pre nego što proizvede sledeći izlaz.","Direktan odgovor","info","callout",{},{"id":229,"data":230,"type":226,"tunes":234},"boundary",{"body":231,"title":232,"variant":233},"Sistem može nešto da zna, a da to ne postavi u trenutni kontekst. Može da pamti nešto izvan prozora modela. Može da preuzme dokument, ali da ga kasnije isključi iz konačnog upita. Model može direktno da koristi samo kontekst koji stigne do trenutnog zaključivanja.","Kontekst nije isto što i znanje ili memorija","warning",{},{"id":236,"data":237,"type":226,"tunes":241},"current",{"body":238,"title":239,"variant":240},"Inženjering konteksta je sada ustaljena praktična terminologija u vodećim smernicama za inženjering veštačke inteligencije, ali nije jedinstveni formalni standard sa jednom obaveznom arhitekturom. Anthropic ga opisuje kao odabir i održavanje optimalnog skupa tokena za zaključivanje; OpenAI-ove aktuelne smernice za agente tretiraju kontekst sesije, skraćivanje i kompresiju kao eksplicitne inženjerske brige za dugotrajne sisteme.","Napomena o aktuelnim izvorima — 8. oktobar 2026.","note",{},{"id":243,"data":244,"type":248,"tunes":249},"toc",{"title":245,"maxLevel":246,"minLevel":247},"Sadržaj",3,2,"tableOfContents",{},{"id":251,"data":252,"type":42,"tunes":254},"h-meaning",{"text":253,"level":247},"Šta inženjering konteksta zaista znači",{},{"id":256,"data":257,"type":218,"tunes":259},"p-meaning-1",{"text":258},"Svaki poziv modela izvršava se u privremenom radnom okruženju: trenutna uputstva, poruke, preuzeti dokazi, izlazi alata i stanje koji se uklapaju u aktivni prozor konteksta. Inženjering konteksta je disciplina namernog konstruisanja tog okruženja.",{},{"id":261,"data":262,"type":218,"tunes":264},"p-meaning-2",{"text":263},"Ključna reč je namerno. Naivan sistem jednostavno spaja sve što ima: punu istoriju, sve preuzete dokumente, svaki odgovor alata i velike sistemske upite. Sistem sa inženjeringom konteksta odlučuje koje su informacije potrebne za trenutnu odluku, a koje treba da ostanu izvan prozora dok ne budu potrebne.",{},{"id":266,"data":267,"type":218,"tunes":269},"p-meaning-3",{"text":268},"To čini inženjering konteksta delom problemom informacione arhitekture, delom problemom izvršavanja i delom problemom evaluacije. Dizajn mora da odluči šta može da uđe u kontekst, odakle dolazi, koja verzija je aktuelna, kako se rešavaju sukobi, koliko detalja se zadržava i kako se rezultat testira.",{},{"id":271,"data":272,"type":42,"tunes":274},"h-simple",{"text":273,"level":247},"Najjednostavniji primer",{},{"id":276,"data":277,"type":218,"tunes":279},"p-simple-1",{"text":278},"Zamislite internog asistenta za podršku. Korisnik pita: „Može li ovaj klijent da otkaže bez naknade?“",{},{"id":281,"data":282,"type":218,"tunes":284},"p-simple-2",{"text":283},"Modelu može biti potrebno pet stvari: aktuelna politika otkazivanja, trenutni tip ugovora klijenta, datum stupanja ugovora na snagu, relevantna pravila o izuzecima i obim ovlašćenja korisnika.",{},{"id":286,"data":287,"type":218,"tunes":289},"p-simple-3",{"text":288},"Nije mu neophodna cela baza podataka klijenata, kompletan arhiv politika, svaki prethodni razgovor ili svaki tiket podrške. Inženjering konteksta je proces koji bira i sastavlja pet korisnih delova, a isključuje nepovezane informacije.",{},{"id":291,"data":292,"type":317,"tunes":318},"simple-flow",{"steps":293,"title":315,"orientation":316},[294,297,300,303,306,309,312],{"label":295,"description":296},"1. Razumeti zadatak","Klasifikovati šta trenutno pitanje zahteva i koje vrste informacija mogu uticati na odgovor.",{"label":298,"description":299},"2. Utvrditi merodavno stanje","Pročitati trenutno stanje aplikacije ili poslovanja koje ne treba pogađati iz memorije.",{"label":301,"description":302},"3. Preuzeti prateće znanje","Pronaći politiku, dokumente ili spoljne dokaze relevantne za konkretan zadatak.",{"label":304,"description":305},"4. Primeniti podobnost i dozvole","Isključiti podatke koje trenutni korisnik ili izvršno okruženje ne smeju da izlože modelu.",{"label":307,"description":308},"5. Smanjiti i strukturirati","Ukloniti duplikate, izabrati korisne izvode i sačuvati ključne metapodatke, uslove i izuzetke.",{"label":310,"description":311},"6. Poredati kontekst","Postaviti uputstva, trenutno stanje i odlučujuće dokaze tamo gde model može dosledno da ih koristi.",{"label":313,"description":314},"7. Izvršiti zaključivanje","Model prima sastavljeni kontekst i proizvodi sledeći odgovor ili predlog radnje.","Od stanja aplikacije do konteksta modela","auto","processFlow",{},{"id":320,"data":321,"type":42,"tunes":323},"h-stops",{"text":322,"level":247},"Gde se jednostavan primer zaustavlja",{},{"id":325,"data":326,"type":218,"tunes":328},"p-stops-1",{"text":327},"Stvarni sistemi su teži jer informacije potrebne za jedan korak možda nisu poznate pre početka izvršavanja. Agent može da otkrije nove činjenice pomoću alata, kreira međufajlove, prima promenljivo spoljno stanje ili obuhvata zadatak duži od jednog prozora konteksta.",{},{"id":330,"data":331,"type":218,"tunes":333},"p-stops-2",{"text":332},"Inženjering konteksta zato postaje dinamičan. Kontekst za korak 12 ne treba da bude prosto kontekst koraka 1 plus jedanaest slojeva nagomilanog izlaza. Treba da odražava trenutno stanje zadatka, odluke koje su još važne i dokaze potrebne za sledeću radnju.",{},{"id":335,"data":336,"type":42,"tunes":338},"h-anatomy",{"text":337,"level":247},"Šta može da uđe u kontekst modela?",{},{"id":340,"data":341,"type":391,"tunes":392},"anatomy-table",{"content":342,"stretched":43,"withHeadings":14},[343,347,351,355,359,363,367,371,375,379,383,387],[344,345,346],"Komponenta konteksta","Svrha","Tipičan rizik",[348,349,350],"Sistemske \u002F razvojne instrukcije","Definišu ulogu, ograničenja, politike i ponašanje","Previše nejasne, kontradiktorne ili preopterećene krhkom logikom",[352,353,354],"Trenutni zahtev korisnika","Definiše neposredni zadatak i nameru","Dvosmislenost ili konflikt sa prethodnom istorijom",[356,357,358],"Istorija razgovora","Očuvava kontinuitet kroz različite razmene","Zastarele pretpostavke, ponavljanje i rast broja tokena",[360,361,362],"Preuzeti dokumenti","Pružaju eksterno znanje\u002Fdokaze","Nerelevantnost, zastarele verzije, slab autoritet ili dupliranje",[364,365,366],"Trenutno stanje aplikacije","Pruža promenljive poslovne\u002Fsistemske činjenice","Korišćenje keširanog ili zapamćenog stanja umesto trenutnog autoriteta",[368,369,370],"Definicije alata","Govore modelu koje sposobnosti postoje i kako ih pozvati","Previše preklapajućih alata ili opširne šeme",[372,373,374],"Rezultati alata","Unose zapažanja iz okruženja u petlju","Veliki neuredni izlazi, nepouzdan sadržaj ili zastarela zapažanja",[376,377,378],"Memorija","Ponovo uvodi izabrane informacije iz prethodnih interakcija","Zastarelost, netačna generalizacija ili preterana personalizacija",[380,381,382],"Primeri","Demonstriraju željeno ponašanje","Previše graničnih slučajeva može potisnuti trenutni zadatak",[384,385,386],"Međukoraci","Prenose planove, rezimee, kod, proračune ili beleške","Staro međustanje može se pogrešno smatrati konačnom istinom",[388,389,390],"Politike \u002F zaštitne mere","Definišu zabranjeno ili ograničeno ponašanje","Konflikt sa poslovnom logikom ili skrivene praznine u sprovođenju","table",{},{"id":394,"data":395,"type":42,"tunes":397},"h-prompt",{"text":396,"level":247},"Inženjering konteksta naspram inženjeringa upita",{},{"id":399,"data":400,"type":431,"tunes":432},"prompt-comparison",{"rows":401,"title":423,"layout":391,"columns":424},[402,407,411,415,419],{"id":403,"label":404,"values":405},"focus","Primarni fokus",[406,406],"",{"id":408,"label":409,"values":410},"scope","Tipičan obim",[406,406],{"id":412,"label":413,"values":414},"timing","Kada se menja",[406,406],{"id":416,"label":417,"values":418},"failure","Tipičan neuspeh",[406,406],{"id":420,"label":421,"values":422},"relationship","Odnos",[406,406],"Inženjering upita i inženjering konteksta rešavaju različite slojeve",[425,428],{"id":426,"label":427},"prompt","Inženjering upita",{"id":429,"label":430},"context","Inženjering konteksta","comparison",{},{"id":434,"data":435,"type":218,"tunes":437},"p-prompt-1",{"text":436},"Anthropic eksplicitno opisuje inženjering konteksta kao prirodan napredak inženjeringa upita za sisteme u kojima model mora da radi sa alatima, eksternim podacima, istorijom poruka i dugotrajnim stanjem agenta. Praktična razlika je korisna jer savršeno napisan upit ne može da nadoknadi nedostatak autoritativnih podataka ili kontekst zagađen kontradiktornim stanjem.",{},{"id":439,"data":440,"type":42,"tunes":442},"h-retrieval",{"text":441,"level":247},"Inženjering konteksta naspram pretraživanja",{},{"id":444,"data":445,"type":218,"tunes":447},"p-ret-1",{"text":446},"Pretraživanje bira kandidatske informacije iz eksternog korpusa ili izvora. Inženjering konteksta odlučuje šta se dešava nakon i oko tog pretraživanja.",{},{"id":449,"data":450,"type":218,"tunes":452},"p-ret-2",{"text":451},"Pretraživač može da vrati 30 pasusa. Reranker ih može svesti na 10. Sloj konteksta može izabrati četiri pasusa, ukloniti duplikate, priložiti metapodatke o izvoru\u002Fverziji, kombinovati ih sa trenutnim stanjem aplikacije i postaviti ih nakon sistemskih instrukcija.",{},{"id":454,"data":455,"type":218,"tunes":457},"p-ret-3",{"text":456},"Zato RAG sistem može da pronađe tačan pasus i ipak odgovori loše: neuspeh može nastati tokom sastavljanja konteksta, a ne tokom pretraživanja.",{},{"id":459,"data":460,"type":226,"tunes":464},"retrieval-boundary",{"body":461,"title":462,"variant":463},"Tačan rezultat pretraživanja je koristan samo ako preživi filtriranje, redosled, kompresiju i odluke o budžetu tokena i zaista stigne do modela u upotrebljivom obliku.","Pretraživanje pronalazi kandidate; inženjering konteksta konstruiše ulaz modela","success",{},{"id":466,"data":467,"type":42,"tunes":469},"h-memory",{"text":468,"level":247},"Inženjering konteksta naspram memorije",{},{"id":471,"data":472,"type":218,"tunes":474},"p-memory-1",{"text":473},"Memorija je informacija sačuvana izvan neposrednog poziva modela kako bi se kasnije ponovo koristila. Kontekst je informacija koja je stvarno učitana u trenutni poziv.",{},{"id":476,"data":477,"type":218,"tunes":479},"p-memory-2",{"text":478},"Memorijski sistem može sadržati hiljade činjenica, beleški ili prethodnih odluka. Inženjering konteksta bira koje od njih treba ponovo uvesti za trenutni zadatak. Učitavanje cele memorije u svakom koraku obesmišljava svrhu postojanja eksternog memorijskog sloja.",{},{"id":481,"data":482,"type":218,"tunes":484},"p-memory-3",{"text":483},"Razlika postaje ključna za promenljivo stanje. Zapamćen status projekta ili korisnička preferencija mogu biti korisni, ali trenutno autoritativno stanje može biti potrebno ponovo pročitati pre odluke sa posledicama.",{},{"id":486,"data":487,"type":492,"tunes":493},"ref-memory",{"url":488,"title":489,"excerpt":490,"ctaLabel":491},"https:\u002F\u002Fstajic.de\u002Fsr\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context","Memorija AI agenta nije RAG: Kako razdvojiti memoriju, pretraživanje, stanje i kontekst","Praktična arhitektura koja razdvaja šta se čuva, šta je trenutno autoritativno, šta se pretražuje i šta model zaista prima.","Pročitajte članak o arhitekturi memorije","referralArticle",{},{"id":495,"data":496,"type":42,"tunes":498},"h-state",{"text":497,"level":247},"Inženjering konteksta naspram stanja aplikacije",{},{"id":500,"data":501,"type":218,"tunes":503},"p-state-1",{"text":502},"Stanje aplikacije je trenutno stanje spoljnog sistema: stanje računa, status tiketa, verzija fajla, faza radnog toka, stanje implementacije ili napredak zadatka.",{},{"id":505,"data":506,"type":218,"tunes":508},"p-state-2",{"text":507},"Stanje može biti sažeto u kontekst, ali sažetak nije samo stanje. Za operacije sa posledicama, izvršno okruženje može morati ponovo pročitati autoritativni sistem neposredno pre akcije, umesto da veruje ranijem snimku vidljivom modelu.",{},{"id":510,"data":511,"type":226,"tunes":514},"state-rule",{"body":512,"title":513,"variant":233},"Kada se stanje jednom kopira u upit, može postati zastarelo. Inženjering konteksta mora definisati kada promenljivo stanje treba osvežiti i koje operacije zahtevaju novo autoritativno čitanje.","Kontekst je snimak",{},{"id":516,"data":517,"type":42,"tunes":519},"h-tools",{"text":518,"level":247},"Dizajn alata je deo inženjeringa konteksta",{},{"id":521,"data":522,"type":218,"tunes":524},"p-tools-1",{"text":523},"Alati ne daju agentima samo sposobnosti. Imena alata, opisi, šeme i rezultati postaju informacije vidljive modelu koje oblikuju odluke.",{},{"id":526,"data":527,"type":218,"tunes":529},"p-tools-2",{"text":528},"Anthropic-ove trenutne smernice za inženjering konteksta naglašavaju alate efikasne po pitanju tokena i upozoravaju na pretrpane skupove alata sa preklapajućom funkcionalnošću. Katalog alata koji je teško razlikovati čoveku takođe je teško pouzdano usmeravati modelu.",{},{"id":531,"data":532,"type":218,"tunes":534},"p-tools-3",{"text":533},"Izlazi alata takođe zahtevaju disciplinu konteksta. Vraćanje celog loga od 20.000 linija kada je agent zatražio jedan uslov greške troši pažnju i može da zatrpa odlučujući dokaz.",{},{"id":536,"data":537,"type":42,"tunes":539},"h-jit",{"text":538,"level":247},"Kontekst tačno na vreme naspram unapred učitanog konteksta",{},{"id":541,"data":542,"type":431,"tunes":568},"jit-comparison",{"rows":543,"title":560,"layout":391,"columns":561},[544,548,552,556],{"id":545,"label":546,"values":547},"method","Metod",[406,406],{"id":549,"label":550,"values":551},"strength","Prednost",[406,406],{"id":553,"label":554,"values":555},"risk","Rizik",[406,406],{"id":557,"label":558,"values":559},"best","Korisno kada",[406,406],"Dva načina da se obezbedi informacija",[562,565],{"id":563,"label":564},"preload","Unapred učitani kontekst",{"id":566,"label":567},"jit","Kontekst tačno na vreme",{},{"id":570,"data":571,"type":218,"tunes":573},"p-jit-1",{"text":572},"Anthropic opisuje hibridni obrazac u kojem je neki stabilan kontekst unapred učitan dok agenti preuzimaju dodatne informacije u vreme izvršavanja. Ovo je koristan arhitektonski obrazac jer nije svaka važna činjenica zaslužuje trajno prebivalište u kontekstnom prozoru.",{},{"id":575,"data":576,"type":42,"tunes":578},"h-budget",{"text":577,"level":247},"Kontekst je budžet, a ne sistem za skladištenje",{},{"id":580,"data":581,"type":218,"tunes":583},"p-budget-1",{"text":582},"Kontekstni prozor definiše kapacitet. On ne garantuje da će svaki token biti jednako dobro iskorišćen. Model mora da rasporedi pažnju na instrukcije, istoriju, dokaze, alate i međustanje.",{},{"id":585,"data":586,"type":218,"tunes":588},"p-budget-2",{"text":587},"Praktični cilj stoga nije „napuniti prozor“. Cilj je maksimizirati korisnost ograničenog budžeta pažnje.",{},{"id":590,"data":591,"type":218,"tunes":593},"p-budget-3",{"text":592},"Anthropic formuliše sličan princip kao pronalaženje najmanjeg skupa tokena sa visokim signalom koji maksimizira verovatnoću željenog ponašanja. OpenAI-ove smernice za upravljanje kontekstom takođe upozoravaju da neobrađena istorija, redundantni rezultati alata i bučno preuzimanje mogu da preplave čak i velike prozore.",{},{"id":595,"data":596,"type":42,"tunes":598},"h-more",{"text":597,"level":247},"Zašto više konteksta može biti gore",{},{"id":600,"data":601,"type":218,"tunes":603},"p-more-1",{"text":602},"Dodatni kontekst može da unese irelevantne informacije, zastarelo stanje, duplirane dokaze, kontradiktorne instrukcije ili poziciono takmičenje. Takođe može da navede sisteme sažimanja da odbace detalje koji kasnije postanu važni.",{},{"id":605,"data":606,"type":218,"tunes":608},"p-more-2",{"text":607},"Klasična studija Izgubljeni u sredini pokazala je da modeli sa dugim kontekstom mogu da koriste informacije različito u zavisnosti od toga gde se relevantni sadržaj pojavljuje, pri čemu performanse često opadaju kada se odlučujuća informacija postavi u sredinu dugih ulaza.",{},{"id":610,"data":611,"type":218,"tunes":613},"p-more-3",{"text":612},"To ne znači da je dug kontekst inherentno loš. To znači da dostupnost unutar prozora nije isto što i pouzdano korišćenje.",{},{"id":615,"data":616,"type":42,"tunes":618},"h-order",{"text":617,"level":247},"Redosled konteksta treba da bude nameran",{},{"id":620,"data":621,"type":218,"tunes":623},"p-order-1",{"text":622},"Izgradnja konteksta je takođe problem redosleda. Kritične instrukcije, trenutno stanje, odlučujući dokazi i ograničenja specifična za zadatak ne bi trebalo proizvoljno nadovezivati.",{},{"id":625,"data":626,"type":218,"tunes":628},"p-order-2",{"text":627},"Ne postoji univerzalno savršen redosled za svaki model i zadatak. Arhitektura stoga treba da testira da li promena redosleda dokaza menja tačnost i da li važne informacije ostaju robusne kroz realistične varijacije konteksta.",{},{"id":630,"data":631,"type":218,"tunes":633},"p-order-3",{"text":632},"Stabilan odgovor koji se dramatično menja kada dva jednako validna odlomka zamene pozicije ukazuje na osetljivost na kontekst koju treba meriti, a ne ignorisati.",{},{"id":635,"data":636,"type":42,"tunes":638},"h-conflict",{"text":637,"level":247},"Konfliktni kontekst zahteva eksplicitni prioritet",{},{"id":640,"data":641,"type":218,"tunes":643},"p-conflict-1",{"text":642},"Model može primiti staru politiku i novu politiku, zapamćenu preferenciju i trenutnu eksplicitnu instrukciju, ili keširani status i rezultat API-ja uživo. Sistem ne treba da očekuje da model zaključi prioritet iz stila proze.",{},{"id":645,"data":646,"type":218,"tunes":648},"p-conflict-2",{"text":647},"Inženjering konteksta treba da kodira prioritet kroz selekciju izvora, metapodatke, redosled ili eksplicitne instrukcije: trenutno autoritativno stanje nadjačava zastarele kopije; eksplicitna trenutna korisnička instrukcija nadjačava stariju zaključenu preferenciju; odobrena politika zamenjuje zastarele nacrte.",{},{"id":650,"data":651,"type":391,"tunes":674},"conflict-table",{"content":652,"stretched":43,"withHeadings":14},[653,656,659,662,665,668,671],[654,655],"Konflikt","Preferentno pravilo konteksta",[657,658],"Trenutno stanje vs zapamćeno stanje","Osveži i preferiraj autoritativni trenutni izvor.",[660,661],"Trenutna politika vs zamenjena politika","Uključi trenutnu verziju; zadrži staru verziju samo kada je potrebno istorijsko poređenje.",[663,664],"Eksplicitna korisnička instrukcija vs stara zaključena preferencija","Preferiraj trenutnu eksplicitnu instrukciju.",[666,667],"Primarni izvor vs sekundarni rezime","Koristi primarni izvor za tvrdnje koje zahtevaju autoritet; rezime može podržati objašnjenje.",[669,670],"Zapažanje alata vs prethodno uverenje modela","Preferiraj trenutno zapaženo stanje kada je alat autoritativan za tu činjenicu.",[672,673],"Dva nerešena autoritativna izvora","Izloži konflikt umesto fabrikovanja jednog konzistentnog odgovora.",{},{"id":676,"data":677,"type":42,"tunes":679},"h-compaction",{"text":678,"level":247},"Sažimanje je transformacija konteksta, a ne bezgubitničko skladištenje",{},{"id":681,"data":682,"type":218,"tunes":684},"p-comp-1",{"text":683},"Dugotrajni sistemi na kraju moraju da skrate, sažmu ili kompaktuju istoriju. Kompaktovanje stvara novu reprezentaciju prethodnog konteksta kako bi agent mogao da nastavi bez reprodukovanja svakog tokena.",{},{"id":686,"data":687,"type":218,"tunes":689},"p-comp-2",{"text":688},"OpenAI-ovi primeri upravljanja kontekstom koriste skraćivanje i kompresiju za dugotrajne sesije. Anthropic opisuje kompaktovanje kao primarnu tehniku za održavanje koherentnosti kada se interakcija približava granici konteksta.",{},{"id":691,"data":692,"type":218,"tunes":694},"p-comp-3",{"text":693},"Težak deo je odlučivanje šta se ne može bezbedno ukloniti: nerešeni zadaci, identifikatori, korisnička ograničenja, bezbednosne granice, arhitekturne odluke, izuzeci, poreklo izvora i uslovi koji čine prethodni zaključak validnim.",{},{"id":696,"data":697,"type":226,"tunes":700},"compaction-rule",{"body":698,"title":699,"variant":233},"Ako kompaktovanje zadrži „koristi pristup X“ ali odbaci zašto je X izabran, koja verzija je testirana ili koji uslov bi ga učinio nevažećim, kasniji odgovori mogu ostati interno konzistentni dok postaju eksterno netačni.","Rezime može sačuvati zaključak i uništiti razlog",{},{"id":702,"data":703,"type":42,"tunes":705},"h-validity",{"text":704,"level":247},"Sačuvaj granice validnosti",{},{"id":707,"data":708,"type":218,"tunes":710},"p-validity-1",{"text":709},"Važni zaključci treba da nose uslove pod kojima ostaju podržani: verziju, datum, obim, pretpostavke, autoritet izvora i nerešeno neslaganje.",{},{"id":712,"data":713,"type":218,"tunes":715},"p-validity-2",{"text":714},"Inženjering konteksta je stoga povezan sa granicom validnosti odgovora. Sastavljač konteksta ne treba da uklanja metapodatke koji određuju da li dokazi još uvek važe.",{},{"id":717,"data":718,"type":492,"tunes":723},"ref-avb",{"url":719,"title":720,"excerpt":721,"ctaLabel":722},"https:\u002F\u002Fstajic.de\u002Fsr\u002Fblog\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers","Granica validnosti odgovora: sloj koji nedostaje između relevantnosti i pouzdanih AI odgovora","Okvir za očuvanje obima, pretpostavki, verzija i uslova dokaza pod kojima AI tvrdnja ostaje podržana.","Pročitajte granicu validnosti odgovora",{},{"id":725,"data":726,"type":42,"tunes":728},"h-security",{"text":727,"level":247},"Inženjering konteksta je takođe bezbednosna granica",{},{"id":730,"data":731,"type":218,"tunes":733},"p-sec-1",{"text":732},"Podaci koji stignu do modela prešli su važnu sistemsku granicu. Sastavljanje konteksta stoga mora poštovati autorizaciju, izolaciju zakupaca, poverljivost i pravila minimizacije podataka.",{},{"id":735,"data":736,"type":218,"tunes":738},"p-sec-2",{"text":737},"Pretraživač može tehnički pronaći odlomak kojem trenutni korisnik ne može pristupiti. Ispravan dizajn je sprečiti da taj odlomak uđe u kontekst modela, umesto da se oslanja na to da će ga model ignorisati.",{},{"id":740,"data":741,"type":218,"tunes":743},"p-sec-3",{"text":742},"Izlazi alata takođe mogu sadržati nepouzdane instrukcije ili neprijateljski sadržaj. Inženjering konteksta treba da očuva razliku između instrukcija aplikacije i eksternih podataka kako pronađeni tekst ne bi mogao tiho steći autoritet instrukcije.",{},{"id":745,"data":746,"type":42,"tunes":748},"h-architecture",{"text":747,"level":247},"Praktična arhitektura za inženjering konteksta",{},{"id":750,"data":751,"type":226,"tunes":754},"arch-note",{"body":752,"title":753,"variant":240},"Sledeći slojevi predstavljaju praktičnu sintezu za produkcione sisteme, a ne formalni industrijski standard. Cilj je da se vlasništvo nad informacijama odvoji od privremenog konteksta koji je okrenut modelu.","Predloženi arhitektonski model",{},{"id":756,"data":757,"type":391,"tunes":786},"arch-table",{"content":758,"stretched":43,"withHeadings":14},[759,762,765,768,771,774,777,780,783],[760,761],"Sloj","Odgovornost",[763,764],"Autoritativni sistemi","Vlasnici trenutnog poslovnog\u002Fsistemskog stanja i zvaničnih zapisa.",[766,767],"Izvori znanja","Vlasnici dokumenata, politika, specifikacija, istraživanja ili eksternih dokaza.",[769,770],"Skladište memorije","Čuva odabrane informacije kroz različite interakcije ili sesije.",[772,773],"Sloj za pronalaženje","Pronalazi kandidate relevantne za zadatak iz eksternih izvora.",[775,776],"Sloj alata\u002Fruntime-a","Čita stanje, izvršava radnje i vraća zapažanja.",[778,779],"Sastavljač konteksta","Bira, filtrira, uklanja duplikate, raspoređuje i formatira informacije vidljive modelu.",[781,782],"Model","Razmišlja i generiše na osnovu sastavljenog konteksta.",[784,785],"Validacija\u002Fevaluacija","Proverava da li odabrani kontekst i rezultujući izlaz zadovoljavaju zahteve specifične za zadatak.",{},{"id":788,"data":789,"type":218,"tunes":791},"p-arch-1",{"text":790},"Sastavljač konteksta je konceptualno važan čak i kada nijedan modul ne nosi tačno taj naziv. U maloj aplikaciji to može biti običan aplikativni kod. U velikoj agentskoj platformi može kombinovati upravljanje sesijama, pronalaženje, memoriju, middleware za alate, kompakciju i sprovođenje politika.",{},{"id":793,"data":794,"type":42,"tunes":796},"h-policy",{"text":795,"level":247},"Praktična politika izgradnje konteksta",{},{"id":798,"data":799,"type":391,"tunes":840},"policy-table",{"content":800,"stretched":43,"withHeadings":14},[801,804,807,810,813,816,819,822,825,828,831,834,837],[802,803],"Pravilo","Zašto je važno",[805,806],"Počnite od trenutnog zadatka","Ne nosite informacije samo zato što su postojale ranije.",[808,809],"Ponovo pročitajte promenljivo stanje","Memorija i stari kontekst mogu biti zastareli.",[811,812],"Pronađite samo dovoljno dokaza","Veliki skupovi kandidata mogu razblažiti odlučujuće informacije.",[814,815],"Sačuvajte metapodatke izvora","Verzija, datum i autoritet određuju da li dokaz još uvek važi.",[817,818],"Uklonite duplirani sadržaj","Redundantnost troši tokene bez dodavanja informacija.",[820,821],"Preferirajte strukturisane sažetke za veliki izlaz alata","Izložite odlučujuća polja umesto sirovog šuma gde to vernost dozvoljava.",[823,824],"Čuvajte pravila zajedno sa izuzecima","Razdvajanje pravila od njegovog izuzetka stvara lažnu sigurnost.",[826,827],"Učinite prioritet eksplicitnim","Ne tražite od modela da zaključi koji konfliktni izvor pobeđuje.",[829,830],"Čuvajte trajno stanje izvan konteksta","Kontekst je privremena radna memorija, a ne baza podataka.",[832,833],"Kompaktirajte uz testove zadržavanja","Proverite da li identifikatori, ograničenja, poreklo i nerešeno stanje preživljavaju.",[835,836],"Izmerite osetljivost na redosled","Ispravnost ne bi trebalo slučajno da zavisi od proizvoljnog redosleda dokumenata.",[838,839],"Evaluacija konteksta odvojeno od kvaliteta modela","Jači model ne može pouzdano da kompenzuje nedostajuće ili neovlašćene dokaze.",{},{"id":842,"data":843,"type":42,"tunes":845},"h-eval",{"text":844,"level":247},"Kako evaluirati inženjering konteksta",{},{"id":847,"data":848,"type":391,"tunes":890},"eval-table",{"content":849,"stretched":43,"withHeadings":14},[850,854,858,862,866,870,874,878,882,886],[851,852,853],"Svojstvo","Pitanje","Primer testa",[855,856,857],"Dovoljnost","Da li kontekst sadrži sve što je potrebno za rešavanje zadatka?","Uklonite jedan dokaz i posmatrajte da li odgovor postaje nepotkrepljen.",[859,860,861],"Relevantnost","Koliko je konteksta nepotrebno za zadatak?","Izmerite kvalitet dok se irelevantni odlomci dodaju ili uklanjaju.",[863,864,865],"Autoritet","Da li su odlučujuće tvrdnje zasnovane na ispravnoj klasi izvora?","Ubacite fluentniji ali neautoritativni konfliktni izvor.",[867,868,869],"Svežina","Da li trenutno stanje nadjačava zastarele kopije?","Promenite autoritativno stanje nakon prethodne interakcije i ponovo pokrenite.",[871,872,873],"Robusnost pozicije","Da li kvalitet odgovora snažno zavisi od pozicije dokaza?","Randomizujte redosled kandidata kroz ponovljene pokušaje.",[875,876,877],"Rešavanje konflikata","Da li model prati eksplicitna pravila prioriteta?","Prikažite staro i novo stanje zajedno.",[879,880,881],"Zadržavanje pri kompakciji","Da li sažimanje čuva ograničenja i granice validnosti?","Uporedite performanse zadatka pre i posle kompakcije.",[883,884,885],"Efikasnost tokena","Da li dodatni kontekst dovoljno poboljšava kvalitet da opravda latenciju\u002Ftrošak?","Izvršite kontrolisane ablacije veličine konteksta.",[887,888,889],"Bezbednost","Mogu li neovlašćeni ili adversarieski sadržaj ući u kontekst modela?","Testirajte granice zakupca, dozvola i prompt-injection-a.",{},{"id":892,"data":893,"type":42,"tunes":895},"h-rag-diagnostic",{"text":894,"level":247},"Sastavljanje konteksta je poseban sloj RAG neuspeha",{},{"id":897,"data":898,"type":218,"tunes":900},"p-ragdiag-1",{"text":899},"RAG pipeline može uspešno da izvrši pronalaženje i ipak ne uspe dalje. Relevantni izvor se može pojaviti na rangu 2, ali sastavljač konteksta ga može izostaviti, skratiti, kombinovati sa zastarelim kontradiktornim materijalom ili prekoračiti budžet tokena.",{},{"id":902,"data":903,"type":218,"tunes":905},"p-ragdiag-2",{"text":904},"Zato tragove pronalaženja treba uporediti sa stvarnim kontekstom poslatim modelu. Bez tog poređenja, neuspesi konteksta se lako pogrešno dijagnostikuju kao neuspesi embedding-a ili modela.",{},{"id":907,"data":908,"type":492,"tunes":913},"ref-ragfail",{"url":909,"title":910,"excerpt":911,"ctaLabel":912},"https:\u002F\u002Fstajic.de\u002Fsr\u002Fblog\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method","RAG nije uspeo — ali koji sloj je zapravo zakazao? Dijagnostička metoda","Pristup sloj po sloj za razdvajanje neuspeha pokrivenosti izvora, pronalaženja, rangiranja, sastavljanja konteksta, generisanja, atribucije dokaza i svežine.","Pročitajte RAG dijagnostičku metodu",{},{"id":915,"data":916,"type":42,"tunes":918},"h-implementation",{"text":917,"level":247},"Dokazi iz originalne implementacije",{},{"id":920,"data":921,"type":42,"tunes":923},"h-sot-engine",{"text":922,"level":246},"Source of Truth Research Engine: ograničeno istraživanje umesto neograničenog konteksta",{},{"id":925,"data":926,"type":218,"tunes":928},"p-sot-1",{"text":927},"Source of Truth Research Engine razdvaja otkrivanje, prikupljanje, ekstrakciju, verifikaciju, analizu kontradikcija i sintezu u ograničene istraživačke faze, umesto da jedan ogroman istraživački zadatak i sav prikupljeni materijal pošalje u jedan jedini poziv modelu.",{},{"id":930,"data":931,"type":218,"tunes":933},"p-sot-2",{"text":932},"Njegov model dokaza čuva Sources, Artifacts, Claims, Relations, Contradictions i poreklo izvan konteksta modela. Model može da primi podskup potreban za trenutni istraživački korak, dok trajni dokazi ostaju u eksternom skladištu.",{},{"id":935,"data":936,"type":218,"tunes":938},"p-sot-3",{"text":937},"To je konkretan obrazac inženjeringa konteksta: trajno istraživačko stanje živi izvan prozora modela; aktivni kontekst modela se rekonstruiše za trenutnu fazu.",{},{"id":940,"data":941,"type":42,"tunes":943},"h-ai-client",{"text":942,"level":246},"Aaasaasa AI Client: runtime, dozvole i kontekst su odvojene brige",{},{"id":945,"data":946,"type":218,"tunes":948},"p-client-1",{"text":947},"Aaasaasa AI klijent razdvaja izbor provajdera\u002Fmodela, lokaciju izvršavanja, dozvole radnog prostora, lokalne resurse i pristup alatima. To sprečava da kontekst modela postane vlasnik autorizacije ili stanja aplikacije.",{},{"id":950,"data":951,"type":218,"tunes":953},"p-client-2",{"text":952},"Direktan razgovor i agentska okruženja za izvršavanje mogu imati različite mogućnosti alata. Profili dozvola radnog prostora se sprovode od strane okruženja za izvršavanje, a ne samo opisuju u kontekstu prirodnog jezika. Ova razlika je važna: kontekst može reći modelu šta bi trebalo da radi, dok okruženje za izvršavanje i dalje mora da sprovede ono što mu je zaista dozvoljeno da radi.",{},{"id":955,"data":956,"type":218,"tunes":958},"p-client-3",{"text":957},"Dokaz implementacije ovde je arhitektonsko razdvajanje, a ne tvrdnja da je svaka napredna tehnika upravljanja kontekstom opisana u ovom članku već implementirana.",{},{"id":960,"data":961,"type":391,"tunes":981},"impl-table",{"content":962,"stretched":43,"withHeadings":14},[963,966,969,972,975,978],[964,965],"Obrazac implementacije","Lekcija o inženjeringu konteksta",[967,968],"Eksterno skladište dokaza","Trajno znanje ne mora da ostane u prozoru modela.",[970,971],"Ograničene faze istraživanja","Različiti koraci mogu da dobiju različit kontekst umesto da akumuliraju jednu ogromnu istoriju.",[973,974],"Tvrdnje + poreklo van konteksta","Identitet dokaza nadživljava privremeno stanje zaključivanja.",[976,977],"Dozvole koje sprovodi okruženje za izvršavanje","Bezbednosni autoritet ne zavisi od toga da se model seća instrukcije.",[979,980],"Razdvojeni koncepti lokalnog\u002Fprovajdera\u002Fmodela\u002Fokruženja za izvršavanje","Kontekst je samo jedan sloj šire arhitekture AI aplikacije.",{},{"id":983,"data":984,"type":226,"tunes":987},"impl-boundary",{"body":985,"title":986,"variant":240},"Ove implementacije podržavaju arhitektonsko razdvajanje između trajnog stanja, pronalaženja, kontrola okruženja za izvršavanje i konteksta okrenutog modelu. One nisu predstavljene kao dokaz putem referentnih vrednosti da je jedna strategija konteksta univerzalno optimalna.","Granica dokaza",{},{"id":989,"data":990,"type":42,"tunes":992},"h-failures",{"text":991,"level":247},"Uobičajeni načini neuspeha u inženjeringu konteksta",{},{"id":994,"data":995,"type":391,"tunes":1030},"failure-table",{"content":996,"stretched":43,"withHeadings":14},[997,1000,1003,1006,1009,1012,1015,1018,1021,1024,1027],[998,999],"Način neuspeha","Šta pođe naopako",[1001,1002],"Beskonačno ponavljanje celog razgovora","Stare pretpostavke, ponavljanje i rast tokena nadvladaju trenutnu nameru.",[1004,1005],"Stavljanje svakog pronađenog rezultata u upit","Šum, dupliranje i konfliktne verzije razblažuju odlučujuće dokaze.",[1007,1008],"Korišćenje memorije kao trenutnog stanja","Zastarele informacije tiho zamenjuju autoritativno aktivno stanje.",[1010,1011],"Vraćanje sirovog izlaza alata","Veliki dnevnici ili odgovori troše pažnju bez dodavanja vrednosti za odluku.",[1013,1014],"Skrivanje opisa alata iza nejasnih imena","Model ne može pouzdano da odluči koju sposobnost da koristi.",[1016,1017],"Sažimanje bez testova zadržavanja","Kritična ograničenja, identifikatori ili izuzeci nestaju.",[1019,1020],"Mešanje instrukcija i nepouzdanih podataka","Spoljni sadržaj može biti protumačen kao instrukcija višeg autoriteta.",[1022,1023],"Korišćenje jednog statičnog šablona konteksta za svaki zadatak","Različiti zadaci dobijaju irelevantne informacije i propuštaju dokaze specifične za zadatak.",[1025,1026],"Ignorisanje verzije\u002Fdatuma izvora","Zastareli ali relevantni dokazi mogu da nadvladaju trenutno autoritativno stanje.",[1028,1029],"Tretiranje većeg kontekstnog prozora kao garancije kvaliteta","Kapacitet se povećava dok problemi pažnje i konflikata ostaju.",{},{"id":1032,"data":1033,"type":42,"tunes":1035},"h-misconceptions",{"text":1034,"level":247},"Uobičajene zablude",{},{"id":1037,"data":1038,"type":391,"tunes":1073},"misconceptions-table",{"content":1039,"stretched":43,"withHeadings":14},[1040,1043,1046,1049,1052,1055,1058,1061,1064,1067,1070],[1041,1042],"Zabluda","Ispravka",[1044,1045],"„Inženjering konteksta je samo inženjering upita pod novim imenom.“","Upiti su jedna komponenta; inženjering konteksta takođe pokriva pronalaženje, memoriju, stanje, rezultate alata, istoriju i sažimanje.",[1047,1048],"„Kontekst znači istoriju razgovora.“","Istorija je samo jedan mogući izvor konteksta.",[1050,1051],"„Više konteksta je uvek bolje.“","Dodatne informacije mogu smanjiti signal, uneti konflikte i povećati troškove.",[1053,1054],"„Ako je pronalaženje našlo, model je video.“","Pronađeni kandidati mogu biti filtrirani, skraćeni ili izostavljeni pre zaključivanja.",[1056,1057],"„Dugačak kontekst uklanja potrebu za RAG.“","Veliki prozori povećavaju kapacitet ali ne rešavaju svežinu, autoritet, dozvole ili dinamičko pronalaženje.",[1059,1060],"„Memorija treba uvek da bude učitana.“","Memoriju treba birati u skladu sa trenutnim zadatkom.",[1062,1063],"„Sažetak čuva sve važno.“","Sažimanje je gubitničko osim ako se eksplicitno ne proceni zadržavanje.",[1065,1066],"„Instrukcije mogu da sprovedu dozvole.“","Autorizaciju moraju da sprovedu kontrole okruženja za izvršavanje\u002Faplikacije, a ne samo kontekst.",[1068,1069],"„Jedan recept za kontekst radi za svaki model.“","Osetljivost na kontekst varira u zavisnosti od modela, zadatka, korpusa i okruženja za izvršavanje.",[1071,1072],"„Inženjering konteksta je samo za agente.“","Agenti pojačavaju potrebu, ali obične RAG i konverzacione aplikacije takođe zahtevaju konstrukciju konteksta.",{},{"id":1075,"data":1076,"type":42,"tunes":1078},"h-sequence",{"text":1077,"level":247},"Praktičan sled inženjeringa konteksta",{},{"id":1080,"data":1081,"type":317,"tunes":1114},"design-sequence",{"steps":1082,"title":1113,"orientation":316},[1083,1086,1089,1092,1095,1098,1101,1104,1107,1110],{"label":1084,"description":1085},"1. Definisati sledeću odluku modela","Odrediti na šta model mora da odgovori, klasifikuje, planira ili izabere u ovom koraku.",{"label":1087,"description":1088},"2. Identifikovati potrebne činjenice i ograničenja","Navesti minimalno stanje, pravila, dokaze i instrukcije koje mogu materijalno da promene rezultat.",{"label":1090,"description":1091},"3. Rešiti autoritet i dozvole","Odrediti koji izvori su trenutni, autoritativni i dostupni trenutnom principalu.",{"label":1093,"description":1094},"4. Pronaći ili čitati na zahtev","Pribaviti neophodne dokaze i promenljivo stanje umesto oslanjanja na zastareli kontekst.",{"label":1096,"description":1097},"5. Smanjiti šum","Ukloniti duplikate, sažeti ili izabrati odlomke bez odbacivanja odlučujućih izuzetaka ili porekla.",{"label":1099,"description":1100},"6. Strukturisati i poredati","Učiniti instrukcije, trenutno stanje, dokaze i zapažanja alata razlikovnim.",{"label":1102,"description":1103},"7. Uklopiti u budžet tokena","Preferirati kontekst visokog signala i premestiti trajne informacije van prozora.",{"label":1105,"description":1106},"8. Pokrenuti model","Izvršiti zaključivanje nad sastavljenim kontekstom.",{"label":1108,"description":1109},"9. Posmatrati neuspehe","Zabeležiti da li problem potiče od nedostajućeg, zastarelog, šumnog, konfliktnog ili loše poređanog konteksta.",{"label":1111,"description":1112},"10. Ponovo proceniti nakon promena modela\u002Fokruženja za izvršavanje","Strategija konteksta važi samo za modele, alate i radna opterećenja na kojima je testirana.","Konstruisati kontekst od trenutne odluke unazad",{},{"id":1116,"data":1117,"type":42,"tunes":1119},"h-checklist",{"text":1118,"level":247},"Kontrolna lista za inženjering konteksta",{},{"id":1121,"data":1122,"type":391,"tunes":1162},"checklist-table",{"content":1123,"stretched":43,"withHeadings":14},[1124,1126,1129,1132,1135,1138,1141,1144,1147,1150,1153,1156,1159],[852,1125],"Očekivani odgovor",[1127,1128],"Koju tačno odluku će model doneti sledeće?","Ograničen zadatak, a ne nejasan dugoročni cilj.",[1130,1131],"Koje informacije mogu materijalno da promene tu odluku?","Eksplicitni minimalni skup dokaza\u002Fstanja.",[1133,1134],"Koji podaci su sada autoritativni?","Trenutni izvor\u002Fverzija i pravilo svežine.",[1136,1137],"Koji podaci su opcionalna pozadina?","Odvojeni od odlučujućih dokaza.",[1139,1140],"Šta ne sme da uđe u kontekst?","Neovlašćeni, nepotrebni ili previše osetljivi podaci.",[1142,1143],"Koje stavke memorije su relevantne?","Izabrane prema zadatku, a ne automatski ponavljane.",[1145,1146],"Koje izlaze alata treba smanjiti?","Veliki odgovori se transformišu u formu relevantnu za odluku.",[1148,1149],"Koja ograničenja moraju da prežive sažimanje?","Identifikatori, izuzeci, obaveze, nerešeno stanje i poreklo.",[1151,1152],"Kako je predstavljen prioritet?","Trenutne\u002Fautoritativne informacije mogu pouzdano da nadvladaju zastarele ili slabije izvore.",[1154,1155],"Kako ćete znati da je kontekst zakazao?","Postoje evaluacije i tragovi specifični za kontekst.",[1157,1158],"Može li odgovor biti reprodukovan?","Ulaz modela ili rekonstruktivni trag konteksta je dostupan gde je prikladno.",[1160,1161],"Može li jači ili veći model da promeni strategiju?","Politika konteksta je svesna verzije i empirijski se ponovo procenjuje.",{},{"id":1164,"data":1165,"type":42,"tunes":1167},"h-edge",{"text":1166,"level":247},"Rubni slučajevi i ograničenja",{},{"id":1169,"data":1170,"type":218,"tunes":1172},"p-edge-1",{"text":1171},"Neki zadaci su dovoljno jednostavni da se inženjering konteksta svodi na kratak sistemski upit i jednu korisničku poruku. Dodavanje pronalaženja, memorije i sažimanja samo bi uvelo nepotrebnu arhitekturu.",{},{"id":1174,"data":1175,"type":218,"tunes":1177},"p-edge-2",{"text":1176},"Neki zadaci zahtevaju visok opoziv i mogu namerno da uključe više konteksta pre kasnije sinteze. Istraživanje, otkrivanje i pravni pregled mogu da preferiraju izbegavanje izostavljanja umesto minimalnog broja tokena.",{},{"id":1179,"data":1180,"type":218,"tunes":1182},"p-edge-3",{"text":1181},"Neke informacije ne bi trebalo nikada sažimati pre upotrebe. Tačni ugovori, kod, kriptografski materijal, numerički zapisi i regulatorni tekst mogu zahtevati doslovno ili strukturisano pronalaženje gde kompresija može da promeni značenje.",{},{"id":1184,"data":1185,"type":218,"tunes":1187},"p-edge-4",{"text":1186},"Ponašanje dugog konteksta značajno varira između modela. Strategija validirana na jednom modelu, dužini konteksta ili okviru alata ne bi trebalo automatski da se prenosi na drugi.",{},{"id":1189,"data":1190,"type":218,"tunes":1192},"p-edge-5",{"text":1191},"Model i dalje može da ignoriše ili pogrešno protumači izuzetan kontekst. Kontekstualno inženjerstvo poboljšava informaciono okruženje; ono ne garantuje ispravnost zaključivanja.",{},{"id":1194,"data":1195,"type":42,"tunes":1197},"h-change",{"text":1196,"level":247},"Šta bi promenilo ovaj odgovor?",{},{"id":1199,"data":1200,"type":218,"tunes":1202},"p-change-1",{"text":1201},"Budući modeli mogu postati robusniji na dugi kontekst, pozicione efekte i konfliktne informacije. To bi moglo smanjiti količinu ručne kuratacije koja je potrebna.",{},{"id":1204,"data":1205,"type":218,"tunes":1207},"p-change-2",{"text":1206},"Arhitektonska razlika bi i dalje ostala korisna jer dozvole, svežina, trajnost memorije, autoritet izvora i stanje eksterne aplikacije postoje izvan modela bez obzira na veličinu kontekstnog prozora.",{},{"id":1209,"data":1210,"type":218,"tunes":1212},"p-change-3",{"text":1211},"Preporučeni balans između unapred učitanog i konteksta koji se učitava po potrebi takođe se menja u zavisnosti od zahteva za latencijom, pouzdanosti alata, veličine korpusa, cene modela i toga koliko su dinamične osnovne informacije.",{},{"id":1214,"data":1215,"type":42,"tunes":1217},"h-related",{"text":1216,"level":247},"Povezano kanonsko znanje",{},{"id":1219,"data":1220,"type":218,"tunes":1222},"p-related-1",{"text":1221},"Kontekstualno inženjerstvo se nalazi između pretrage i generisanja. RAG objašnjava kako se eksterno znanje pronalazi; R01 razdvaja embedding-e, vektorsku pretragu i rerangiranje; kontekstualno inženjerstvo objašnjava šta na kraju stigne do modela.",{},{"id":1224,"data":1225,"type":492,"tunes":1230},"ref-rag",{"url":1226,"title":1227,"excerpt":1228,"ctaLabel":1229},"https:\u002F\u002Fstajic.de\u002Fsr\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works","Šta je RAG? Najjednostavnije objašnjenje kako funkcioniše","Temelj pretrage za razumevanje kako se eksterno znanje može dostaviti modelu pre generisanja.","Pročitajte osnove RAG-a",{},{"id":1232,"data":1233,"type":218,"tunes":1235},"p-related-2",{"text":1234},"Arhitektura izvora istine odgovara na drugačije pitanje: ne koje informacije su prisutne u kontekstu, već koji izvor je ovlašćen da utvrdi tvrdnju.",{},{"id":1237,"data":1238,"type":218,"tunes":1240},"p-related-3",{"text":1239},"Postojeći članak Zašto više konteksta može pogoršati AI odgovore je dijagnostički pratilac ovoj kanonskoj definiciji. Fokusira se na zagađenje konteksta, pozicione efekte, rast top-k, gubitak kompakcije i degradaciju odgovora, umesto da redefiniše samo kontekstualno inženjerstvo.",{},{"id":1242,"data":1243,"type":42,"tunes":1245},"h-faq",{"text":1244,"level":247},"Često postavljana pitanja",{},{"id":1247,"data":1248,"type":1247,"tunes":1283},"faq",{"items":1249,"title":1282},[1250,1254,1258,1262,1266,1270,1274,1278],{"id":1251,"answer":1252,"question":1253},"faq1","Kontekstualno inženjerstvo je dizajn i upravljanje u vreme izvršavanja onim informacijama koje jezički model prima u trenutku inferencije, uključujući instrukcije, istoriju, pronađene dokaze, memoriju, stanje, alate i rezultate alata.","Šta je kontekstualno inženjerstvo?",{"id":1255,"answer":1256,"question":1257},"faq2","Prompt inženjerstvo se fokusira na to kako su instrukcije i primeri napisani. Kontekstualno inženjerstvo uključuje promptove, ali takođe odlučuje koje eksterne informacije, stanje, istorija, memorija i zapažanja alata se postavljaju oko njih.","Po čemu se kontekstualno inženjerstvo razlikuje od prompt inženjerstva?",{"id":1259,"answer":1260,"question":1261},"faq3","Ne. RAG pronalazi eksterne informacije. Kontekstualno inženjerstvo odlučuje kako se pronađene informacije filtriraju, kombinuju sa drugim stanjem i zaista dostavljaju modelu.","Da li je RAG isto što i kontekstualno inženjerstvo?",{"id":1263,"answer":1264,"question":1265},"faq4","Ne. Memorija čuva informacije izvan trenutnog poziva modela. Kontekst je podskup informacija učitanih u trenutnu inferenciju.","Da li je memorija isto što i kontekst?",{"id":1267,"answer":1268,"question":1269},"faq5","Dodatni kontekst može uneti šum, zastarelo stanje, konfliktne dokaze, dupliranje i poziciono takmičenje. Veliki kapacitet konteksta ne garantuje jednako pouzdano korišćenje svakog tokena.","Zašto više konteksta može pogoršati odgovor?",{"id":1271,"answer":1272,"question":1273},"faq6","Kompakcija sažima ili transformiše nagomilanu istoriju u manju reprezentaciju kako bi dugotrajni sistem mogao da nastavi bez ponovnog reprodukovanja svakog prethodnog tokena.","Šta je kompakcija konteksta?",{"id":1275,"answer":1276,"question":1277},"faq7","Može biti predstavljeno u kontekstu radi zaključivanja, ali operacije sa posledicama često treba ponovo da pročitaju autoritativni izvor jer snimci konteksta mogu postati zastareli.","Da li trenutno stanje aplikacije treba čuvati u kontekstu?",{"id":1279,"answer":1280,"question":1281},"faq8","Ne. Agenti čine upravljanje kontekstom dinamičnijim, ali RAG sistemi, asistenti, kopiloti i aplikacije sa više krugova takođe zahtevaju namerno konstruisanje konteksta.","Da li je kontekstualno inženjerstvo potrebno samo za AI agente?","Često postavljana pitanja o kontekstualnom inženjerstvu",{},{"id":1285,"data":1286,"type":42,"tunes":1288},"h-glossary",{"text":1287,"level":247},"Pojmovnik",{},{"id":1290,"data":1291,"type":1290,"tunes":1340},"glossary",{"title":1292,"entries":1293},"Ključni pojmovi kontekstualnog inženjerstva",[1294,1298,1302,1306,1310,1314,1318,1322,1326,1329,1333,1336],{"term":1295,"anchor":1296,"definition":1297},"Kontekstualno inženjerstvo","context-engineering","Dizajn i upravljanje u vreme izvršavanja informacijama koje se dostavljaju jezičkom modelu za određeni korak inferencije.",{"term":1299,"anchor":1300,"definition":1301},"Kontekstni prozor","context-window","Konačni kapacitet tokena modela za ulaz i, u zavisnosti od interfejsa modela, povezane generisane tokene ili aktivnu sekvencu.",{"term":1303,"anchor":1304,"definition":1305},"Prompt inženjerstvo","prompt-engineering","Dizajn instrukcija, primera i strukture prompta koji imaju za cilj da izazovu korisno ponašanje modela.",{"term":1307,"anchor":1308,"definition":1309},"Sastavljanje konteksta","context-assembly","Proces odabira, filtriranja, raspoređivanja i formatiranja informacija vidljivih modelu pre inferencije.",{"term":1311,"anchor":1312,"definition":1313},"Pretraga u pravo vreme","just-in-time-retrieval","Dinamičko učitavanje informacija kada trenutni zadatak to zahteva, umesto unapred učitavanja svih potencijalno relevantnih podataka.",{"term":1315,"anchor":1316,"definition":1317},"Kompakcija","compaction","Svođenje nagomilanog konteksta na manju reprezentaciju uz nastojanje da se sačuvaju informacije potrebne za buduće korake.",{"term":1319,"anchor":1320,"definition":1321},"Zagađenje konteksta","context-pollution","Degradacija uzrokovana irelevantnim, zastarelim, kontradiktornim ili redundantnim informacijama koje zauzimaju radni kontekst modela.",{"term":1323,"anchor":1324,"definition":1325},"Stanje aplikacije","application-state","Trenutno autoritativno stanje eksternog sistema, radnog toka ili domena koje postoji nezavisno od konteksta modela.",{"term":376,"anchor":1327,"definition":1328},"memory","Informacije uskladištene izvan neposrednog poziva modela za moguću upotrebu u kasnijim krugovima ili sesijama.",{"term":1330,"anchor":1331,"definition":1332},"Pronađeni kontekst","retrieved-context","Eksterne informacije koje je odabrao sistem za pretragu i učinio dostupnim modelu, u celosti ili delimično.",{"term":871,"anchor":1334,"definition":1335},"position-robustness","Stepen u kojem ispravnost modela ostaje stabilna kada se lokacija ili redosled relevantnog konteksta promene.",{"term":1337,"anchor":1338,"definition":1339},"Granica važenja","validity-boundary","Obim, vreme, pretpostavke, verzije i uslovi dokaza u okviru kojih zaključak ostaje potkrepljen.",{},{"id":1342,"data":1343,"type":42,"tunes":1345},"h-conclusion",{"text":1344,"level":247},"Zaključak",{},{"id":1347,"data":1348,"type":218,"tunes":1350},"p-conclusion-1",{"text":1349},"Kontekstualno inženjerstvo je sloj koji odlučuje šta model može da vidi pre nego što odgovori. To ga čini širim od promptovanja i nizvodnim od pretrage, dok istovremeno ostaje različito od trajne memorije i autoritativnog stanja aplikacije.",{},{"id":1352,"data":1353,"type":218,"tunes":1355},"p-conclusion-2",{"text":1354},"Snažna arhitektura konteksta ne tretira kontekstni prozor kao bazu podataka. Ona čuva trajno stanje i znanje izvan modela, učitava ono što je potrebno za trenutnu odluku, čuva autoritet i poreklo, uklanja nepotreban šum i osvežava volatilne informacije kada je potrebno.",{},{"id":1357,"data":1358,"type":218,"tunes":1360},"p-conclusion-3",{"text":1359},"Praktični cilj stoga nije maksimalan kontekst. To je minimalni dovoljan, visoko signalni, ispravno autorizovan i kontekst koji čuva važenje za sledeću odluku modela.",{},{"id":1362,"data":1363,"type":42,"tunes":1365},"h-sources",{"text":1364,"level":247},"Primarni izvori i aktuelne smernice",{},{"id":1367,"data":1368,"type":218,"tunes":1370},"p-sources-note",{"text":1369},"Izvori navedeni u nastavku podržavaju trenutnu terminologiju kontekstualnog inženjeringa, ponašanje dugog konteksta i operativne obrasce upravljanja kontekstom. Sekcije projekta su eksplicitno dokazi implementacije, a ne univerzalne tvrdnje.",{},{"id":1372,"data":1373,"type":1379,"tunes":1380},"src-anthropic",{"link":1374,"meta":1375},"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Feffective-context-engineering-for-ai-agents",{"image":1376,"title":1377,"description":1378},{"url":406},"Anthropic — Efikasan kontekstualni inženjering za AI agente","Zvanične inženjerske smernice koje definišu kontekstualni inženjering, pravovremeno preuzimanje, kompakciju, strukturisanu memoriju i kuriranje konteksta za agente.","linkTool",{},{"id":1382,"data":1383,"type":1379,"tunes":1389},"src-openai-session",{"link":1384,"meta":1385},"https:\u002F\u002Fdevelopers.openai.com\u002Fcookbook\u002Fexamples\u002Fagents_sdk\u002Fsession_memory",{"image":1386,"title":1387,"description":1388},{"url":406},"OpenAI — Kontekstualni inženjering: Upravljanje kratkoročnom memorijom pomoću sesija","Zvanične smernice iz kuvara znanja o upravljanju kontekstom, skraćivanju i kompresiji za dugotrajne agentske sesije.",{},{"id":1391,"data":1392,"type":1379,"tunes":1398},"src-openai-agents",{"link":1393,"meta":1394},"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fagents",{"image":1395,"title":1396,"description":1397},{"url":406},"OpenAI — Vodič za agente","Aktuelne OpenAI smernice za programere o agentskim runtime okruženjima, kontekstu kroz korake i vlasništvu nad orkestracijom.",{},{"id":1400,"data":1401,"type":1379,"tunes":1407},"src-lost-middle",{"link":1402,"meta":1403},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2307.03172",{"image":1404,"title":1405,"description":1406},{"url":406},"Izgubljeni u sredini: Kako jezički modeli koriste duge kontekste","Istraživanje koje pokazuje da performanse modela sa dugim kontekstom mogu snažno zavisiti od pozicije relevantnih informacija u ulazu.",{},"2.31","Inženjering konteksta osmišljava koje informacije AI model prima pre inferencije, uključujući promptove, pretragu, memoriju, stanje aplikacije, rezultate alata i istoriju konverzacije.","\u002Fuploads\u002F2026\u002F10\u002Fwhat-is-context-engineering-what-the-model-receives-before-it-answers-1791480653258-018kcv.webp","what-is-context-engineering-what-the-model-receives-before-it-answers-1791480653258-018kcv","PUBLISHED","2026-10-08T13:29:00.000Z","2026-10-08T17:29:05.600Z","2026-10-08T17:43:15.694Z",{"en":1417,"de":1418,"sr":1419,"es":1420,"fr":1421,"it":1422,"ru":1423,"zh":1424},"\u002Fblog\u002Fwhat-is-context-engineering-what-the-model-receives-before-it-answers","\u002Fde\u002Fblog\u002Fwhat-is-context-engineering-what-the-model-receives-before-it-answers","\u002Fsr\u002Fblog\u002Fwhat-is-context-engineering-what-the-model-receives-before-it-answers","\u002Fes\u002Fblog\u002Fwhat-is-context-engineering-what-the-model-receives-before-it-answers","\u002Ffr\u002Fblog\u002Fwhat-is-context-engineering-what-the-model-receives-before-it-answers","\u002Fit\u002Fblog\u002Fwhat-is-context-engineering-what-the-model-receives-before-it-answers","\u002Fru\u002Fblog\u002Fwhat-is-context-engineering-what-the-model-receives-before-it-answers","\u002Fzh\u002Fblog\u002Fwhat-is-context-engineering-what-the-model-receives-before-it-answers",[1426,1430,1434],{"id":1427,"name":1428,"slug":1429},55,"Referentni model: LLM sposobnosti","llm-capability",{"id":1431,"name":1432,"slug":1433},64,"Informaciona arhitektura","information-architecture",{"id":1435,"name":1436,"slug":1437},88,"Verzionisanje (prompt, modeli)","versioning",{"id":1439,"login":1440,"email":1441,"displayName":1442},"20","rooth8233","aleksandar@stajic.de","Aleksandar Stajić",[1444,2459],{"lang":1445,"title":1446,"content":1447,"contentJson":1448,"excerpt":2458},"en","What Is Context Engineering? What the Model Receives Before It Answers","{\"time\":1791480654232,\"blocks\":[{\"id\":\"intro\",\"type\":\"paragraph\",\"data\":{\"text\":\"Context engineering is the design of what information a language model receives at inference time, in what form, in what order and for how long. It is broader than prompt engineering because the model context can include system instructions, user messages, retrieved documents, tool results, memory, current application state, examples, structured data and intermediate artifacts. The goal is not to maximize the number of tokens, but to construct the smallest useful context that preserves the information, constraints and evidence needed for the current task.\"},\"tunes\":{}},{\"id\":\"direct\",\"type\":\"callout\",\"data\":{\"variant\":\"info\",\"title\":\"Direct answer\",\"body\":\"Prompt engineering asks \u003Cstrong>how should we instruct the model?\u003C\u002Fstrong> Context engineering asks \u003Cstrong>what should the model know right now, and how should that information be assembled?\u003C\u002Fstrong>\u003Cbr>\u003Cbr>Retrieval, memory, state management, tool design, history trimming, compaction and ordering are therefore context-engineering mechanisms when they determine the tokens available to the model before it produces the next output.\"},\"tunes\":{}},{\"id\":\"boundary\",\"type\":\"callout\",\"data\":{\"variant\":\"warning\",\"title\":\"Context is not the same as knowledge or memory\",\"body\":\"A system can know something without placing it in the current context. It can remember something outside the model window. It can retrieve a document but later exclude it from the final prompt. The model can only directly use the context that reaches the current inference.\"},\"tunes\":{}},{\"id\":\"current\",\"type\":\"callout\",\"data\":{\"variant\":\"note\",\"title\":\"Current-source note — 8 October 2026\",\"body\":\"Context engineering is now established practical terminology in major AI engineering guidance, but it is not a single formal standard with one mandatory architecture. Anthropic describes it as curating and maintaining the optimal set of tokens for inference; OpenAI's current agent guidance treats session context, trimming and compression as explicit engineering concerns for long-running systems.\"},\"tunes\":{}},{\"id\":\"toc\",\"type\":\"tableOfContents\",\"data\":{\"title\":\"Contents\",\"minLevel\":2,\"maxLevel\":3},\"tunes\":{}},{\"id\":\"h-meaning\",\"type\":\"header\",\"data\":{\"text\":\"What context engineering really means\",\"level\":2},\"tunes\":{}},{\"id\":\"p-meaning-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Every model call is made under a temporary working environment: the current instructions, messages, retrieved evidence, tool outputs and state that fit into the active context window. Context engineering is the discipline of constructing that environment deliberately.\"},\"tunes\":{}},{\"id\":\"p-meaning-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The key word is deliberately. A naive system simply concatenates everything it has: full history, all retrieved documents, every tool response and large system prompts. A context-engineered system decides which information is required for the current decision and which information should remain outside the window until needed.\"},\"tunes\":{}},{\"id\":\"p-meaning-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"This makes context engineering partly an information-architecture problem, partly a runtime problem and partly an evaluation problem. The design must decide what can enter context, where it comes from, which version is current, how conflicts are resolved, how much detail is retained and how the result is tested.\"},\"tunes\":{}},{\"id\":\"h-simple\",\"type\":\"header\",\"data\":{\"text\":\"The simplest example\",\"level\":2},\"tunes\":{}},{\"id\":\"p-simple-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Imagine an internal support assistant. A user asks: “Can this customer cancel without a fee?”\"},\"tunes\":{}},{\"id\":\"p-simple-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The model might need five things: the current cancellation policy, the customer's current contract type, the effective contract date, the relevant exception rules and the user's authorization scope.\"},\"tunes\":{}},{\"id\":\"p-simple-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"It does not necessarily need the entire customer database, the full policy archive, every previous conversation or every support ticket. Context engineering is the process that selects and assembles the five useful pieces while excluding unrelated information.\"},\"tunes\":{}},{\"id\":\"simple-flow\",\"type\":\"processFlow\",\"data\":{\"title\":\"From application state to model context\",\"orientation\":\"auto\",\"steps\":[{\"label\":\"1. Understand the task\",\"description\":\"Classify what the current question requires and which information types can affect the answer.\"},{\"label\":\"2. Resolve authoritative state\",\"description\":\"Read current application or business state that should not be guessed from memory.\"},{\"label\":\"3. Retrieve supporting knowledge\",\"description\":\"Find the policy, documents or external evidence relevant to the specific task.\"},{\"label\":\"4. Apply eligibility and permissions\",\"description\":\"Exclude data the current user or runtime is not allowed to expose to the model.\"},{\"label\":\"5. Reduce and structure\",\"description\":\"Remove duplication, select useful excerpts and preserve critical metadata, conditions and exceptions.\"},{\"label\":\"6. Order the context\",\"description\":\"Place instructions, current state and decisive evidence where the model can use them consistently.\"},{\"label\":\"7. Run inference\",\"description\":\"The model receives the assembled context and produces the next answer or action proposal.\"}]},\"tunes\":{}},{\"id\":\"h-stops\",\"type\":\"header\",\"data\":{\"text\":\"Where the simple example stops\",\"level\":2},\"tunes\":{}},{\"id\":\"p-stops-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Real systems are more difficult because the information needed for one step may not be known before execution begins. An agent can discover new facts through tools, create intermediate files, receive changing external state or span a task longer than one context window.\"},\"tunes\":{}},{\"id\":\"p-stops-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Context engineering therefore becomes dynamic. The context for step 12 should not simply be step 1 context plus eleven layers of accumulated output. It should reflect the current task state, the decisions that still matter and the evidence required for the next action.\"},\"tunes\":{}},{\"id\":\"h-anatomy\",\"type\":\"header\",\"data\":{\"text\":\"What can enter a model context?\",\"level\":2},\"tunes\":{}},{\"id\":\"anatomy-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Context component\",\"Purpose\",\"Typical risk\"],[\"System \u002F developer instructions\",\"Define role, constraints, policies and behavior\",\"Too vague, contradictory or overloaded with brittle logic\"],[\"Current user request\",\"Defines immediate task and intent\",\"Ambiguity or conflict with prior history\"],[\"Conversation history\",\"Preserves continuity across turns\",\"Stale assumptions, repetition and token growth\"],[\"Retrieved documents\",\"Provide external knowledge\u002Fevidence\",\"Irrelevance, stale versions, weak authority or duplication\"],[\"Current application state\",\"Supplies volatile business\u002Fsystem facts\",\"Using cached or remembered state instead of current authority\"],[\"Tool definitions\",\"Tell the model what capabilities exist and how to call them\",\"Too many overlapping tools or verbose schemas\"],[\"Tool results\",\"Bring observations from the environment into the loop\",\"Large noisy outputs, untrusted content or obsolete observations\"],[\"Memory\",\"Reintroduces selected information from previous interactions\",\"Staleness, incorrect generalization or over-personalization\"],[\"Examples\",\"Demonstrate desired behavior\",\"Too many edge cases can crowd out the current task\"],[\"Intermediate artifacts\",\"Carry plans, summaries, code, calculations or notes\",\"Old intermediate state may be mistaken for final truth\"],[\"Policies \u002F guardrails\",\"Define prohibited or constrained behavior\",\"Conflict with business logic or hidden enforcement gaps\"]]},\"tunes\":{}},{\"id\":\"h-prompt\",\"type\":\"header\",\"data\":{\"text\":\"Context engineering vs prompt engineering\",\"level\":2},\"tunes\":{}},{\"id\":\"prompt-comparison\",\"type\":\"comparison\",\"data\":{\"title\":\"Prompt engineering and context engineering solve different layers\",\"layout\":\"table\",\"columns\":[{\"id\":\"prompt\",\"label\":\"Prompt engineering\"},{\"id\":\"context\",\"label\":\"Context engineering\"}],\"rows\":[{\"id\":\"focus\",\"label\":\"Primary focus\",\"values\":[\"\",\"\"]},{\"id\":\"scope\",\"label\":\"Typical scope\",\"values\":[\"\",\"\"]},{\"id\":\"timing\",\"label\":\"When it changes\",\"values\":[\"\",\"\"]},{\"id\":\"failure\",\"label\":\"Typical failure\",\"values\":[\"\",\"\"]},{\"id\":\"relationship\",\"label\":\"Relationship\",\"values\":[\"\",\"\"]}]},\"tunes\":{}},{\"id\":\"p-prompt-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Anthropic explicitly describes context engineering as the natural progression of prompt engineering for systems in which the model must work with tools, external data, message history and long-running agent state. The practical distinction is useful because a perfectly written prompt cannot compensate for missing authoritative data or a context polluted by contradictory state.\"},\"tunes\":{}},{\"id\":\"h-retrieval\",\"type\":\"header\",\"data\":{\"text\":\"Context engineering vs retrieval\",\"level\":2},\"tunes\":{}},{\"id\":\"p-ret-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Retrieval selects candidate information from an external corpus or source. Context engineering decides what happens after and around that retrieval.\"},\"tunes\":{}},{\"id\":\"p-ret-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The retriever may return 30 passages. A reranker may reduce them to 10. The context layer may select four passages, remove duplicates, attach source\u002Fversion metadata, combine them with current application state and place them after the system instructions.\"},\"tunes\":{}},{\"id\":\"p-ret-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"This is why a RAG system can retrieve the correct passage and still answer badly: the failure may occur during context assembly rather than retrieval.\"},\"tunes\":{}},{\"id\":\"retrieval-boundary\",\"type\":\"callout\",\"data\":{\"variant\":\"success\",\"title\":\"Retrieval finds candidates; context engineering constructs the model input\",\"body\":\"The correct retrieval result is only useful if it survives filtering, ordering, compression and token-budget decisions and actually reaches the model in a usable form.\"},\"tunes\":{}},{\"id\":\"h-memory\",\"type\":\"header\",\"data\":{\"text\":\"Context engineering vs memory\",\"level\":2},\"tunes\":{}},{\"id\":\"p-memory-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Memory is information preserved outside the immediate model invocation so it can be used again later. Context is the information actually loaded into the current invocation.\"},\"tunes\":{}},{\"id\":\"p-memory-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"A memory system may contain thousands of facts, notes or prior decisions. Context engineering selects which of those should be reintroduced for the current task. Loading all memory on every turn defeats the purpose of having an external memory layer.\"},\"tunes\":{}},{\"id\":\"p-memory-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The distinction becomes crucial for volatile state. A remembered project status or user preference can be useful, but current authoritative state may need to be re-read before a consequential decision.\"},\"tunes\":{}},{\"id\":\"ref-memory\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context\",\"title\":\"AI Agent Memory Is Not RAG: How to Separate Memory, Retrieval, State and Context\",\"excerpt\":\"A practical architecture separating what persists, what is authoritative now, what is retrieved and what the model actually receives.\",\"ctaLabel\":\"Read the memory architecture article\"},\"tunes\":{}},{\"id\":\"h-state\",\"type\":\"header\",\"data\":{\"text\":\"Context engineering vs application state\",\"level\":2},\"tunes\":{}},{\"id\":\"p-state-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Application state is the current condition of the outside system: account balance, ticket status, file version, workflow stage, deployment state or task progress.\"},\"tunes\":{}},{\"id\":\"p-state-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"State can be summarized into context, but the summary is not the state itself. For consequential operations, the runtime may need to re-read the authoritative system immediately before the action rather than trust an earlier model-visible snapshot.\"},\"tunes\":{}},{\"id\":\"state-rule\",\"type\":\"callout\",\"data\":{\"variant\":\"warning\",\"title\":\"Context is a snapshot\",\"body\":\"Once state is copied into a prompt, it can become stale. Context engineering must define when volatile state needs refreshing and which operations require a new authoritative read.\"},\"tunes\":{}},{\"id\":\"h-tools\",\"type\":\"header\",\"data\":{\"text\":\"Tool design is part of context engineering\",\"level\":2},\"tunes\":{}},{\"id\":\"p-tools-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Tools do more than give agents capabilities. Tool names, descriptions, schemas and results become model-visible information that shapes decisions.\"},\"tunes\":{}},{\"id\":\"p-tools-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Anthropic's current context-engineering guidance emphasizes token-efficient tools and warns against bloated tool sets with overlapping functionality. A tool catalog that is difficult for a human to distinguish is also difficult for a model to route reliably.\"},\"tunes\":{}},{\"id\":\"p-tools-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"Tool outputs also need context discipline. Returning an entire 20,000-line log when the agent requested one error condition consumes attention and can bury the decisive evidence.\"},\"tunes\":{}},{\"id\":\"h-jit\",\"type\":\"header\",\"data\":{\"text\":\"Just-in-time context vs preloaded context\",\"level\":2},\"tunes\":{}},{\"id\":\"jit-comparison\",\"type\":\"comparison\",\"data\":{\"title\":\"Two ways to supply information\",\"layout\":\"table\",\"columns\":[{\"id\":\"preload\",\"label\":\"Preloaded context\"},{\"id\":\"jit\",\"label\":\"Just-in-time context\"}],\"rows\":[{\"id\":\"method\",\"label\":\"Method\",\"values\":[\"\",\"\"]},{\"id\":\"strength\",\"label\":\"Strength\",\"values\":[\"\",\"\"]},{\"id\":\"risk\",\"label\":\"Risk\",\"values\":[\"\",\"\"]},{\"id\":\"best\",\"label\":\"Useful when\",\"values\":[\"\",\"\"]}]},\"tunes\":{}},{\"id\":\"p-jit-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Anthropic describes a hybrid pattern in which some stable context is preloaded while agents retrieve additional information at runtime. This is a useful architecture pattern because not every important fact deserves permanent residency in the context window.\"},\"tunes\":{}},{\"id\":\"h-budget\",\"type\":\"header\",\"data\":{\"text\":\"Context is a budget, not a storage system\",\"level\":2},\"tunes\":{}},{\"id\":\"p-budget-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A context window defines capacity. It does not guarantee that every token will be used equally well. The model must distribute attention across instructions, history, evidence, tools and intermediate state.\"},\"tunes\":{}},{\"id\":\"p-budget-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The practical objective is therefore not “fill the window.” It is to maximize the utility of the limited attention budget.\"},\"tunes\":{}},{\"id\":\"p-budget-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"Anthropic formulates a similar principle as finding the smallest high-signal set of tokens that maximizes the probability of the desired behavior. OpenAI's context-management guidance likewise warns that uncurated history, redundant tool results and noisy retrieval can overwhelm even large windows.\"},\"tunes\":{}},{\"id\":\"h-more\",\"type\":\"header\",\"data\":{\"text\":\"Why more context can be worse\",\"level\":2},\"tunes\":{}},{\"id\":\"p-more-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Additional context can introduce irrelevant information, stale state, duplicate evidence, contradictory instructions or positional competition. It can also cause compaction systems to discard details that later become important.\"},\"tunes\":{}},{\"id\":\"p-more-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The classic Lost in the Middle study demonstrated that long-context models can use information differently depending on where relevant content appears, with performance often degrading when decisive information is placed in the middle of long inputs.\"},\"tunes\":{}},{\"id\":\"p-more-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"This does not mean long context is inherently bad. It means availability inside the window is not the same as reliable utilization.\"},\"tunes\":{}},{\"id\":\"h-order\",\"type\":\"header\",\"data\":{\"text\":\"Context ordering should be intentional\",\"level\":2},\"tunes\":{}},{\"id\":\"p-order-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Context construction is also an ordering problem. Critical instructions, current state, decisive evidence and task-specific constraints should not be concatenated arbitrarily.\"},\"tunes\":{}},{\"id\":\"p-order-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"There is no universal perfect ordering for every model and task. The architecture should therefore test whether reordering evidence changes correctness and whether important information remains robust across realistic context variations.\"},\"tunes\":{}},{\"id\":\"p-order-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"A stable answer that changes dramatically when two equally valid passages swap positions indicates context sensitivity that should be measured rather than ignored.\"},\"tunes\":{}},{\"id\":\"h-conflict\",\"type\":\"header\",\"data\":{\"text\":\"Conflicting context needs explicit precedence\",\"level\":2},\"tunes\":{}},{\"id\":\"p-conflict-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A model may receive an old policy and a new policy, a remembered preference and a current explicit instruction, or a cached status and a live API result. The system should not expect the model to infer precedence from prose style.\"},\"tunes\":{}},{\"id\":\"p-conflict-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Context engineering should encode precedence through source selection, metadata, ordering or explicit instructions: current authoritative state overrides stale copies; explicit current user instruction overrides older inferred preference; approved policy supersedes obsolete drafts.\"},\"tunes\":{}},{\"id\":\"conflict-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Conflict\",\"Preferred context rule\"],[\"Current state vs remembered state\",\"Refresh and prefer the authoritative current source.\"],[\"Current policy vs superseded policy\",\"Include current version; keep old version only when historical comparison is required.\"],[\"Explicit user instruction vs old inferred preference\",\"Prefer the current explicit instruction.\"],[\"Primary source vs secondary summary\",\"Use primary source for claims that require authority; summary may support explanation.\"],[\"Tool observation vs model prior\",\"Prefer current observed state when the tool is authoritative for that fact.\"],[\"Two unresolved authoritative sources\",\"Expose the conflict rather than fabricating one consistent answer.\"]]},\"tunes\":{}},{\"id\":\"h-compaction\",\"type\":\"header\",\"data\":{\"text\":\"Compaction is context transformation, not lossless storage\",\"level\":2},\"tunes\":{}},{\"id\":\"p-comp-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Long-running systems eventually need to trim, summarize or compact history. Compaction creates a new representation of prior context so the agent can continue without replaying every token.\"},\"tunes\":{}},{\"id\":\"p-comp-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"OpenAI's context-management examples use trimming and compression for long-running sessions. Anthropic describes compaction as a primary technique for maintaining coherence when an interaction approaches the context limit.\"},\"tunes\":{}},{\"id\":\"p-comp-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The difficult part is deciding what cannot be safely removed: unresolved tasks, identifiers, user constraints, security boundaries, architecture decisions, exceptions, source provenance and the conditions that make a previous conclusion valid.\"},\"tunes\":{}},{\"id\":\"compaction-rule\",\"type\":\"callout\",\"data\":{\"variant\":\"warning\",\"title\":\"A summary can preserve the conclusion and destroy the reason\",\"body\":\"If compaction keeps “use approach X” but discards why X was chosen, which version was tested or what condition would invalidate it, later responses can remain internally consistent while becoming externally wrong.\"},\"tunes\":{}},{\"id\":\"h-validity\",\"type\":\"header\",\"data\":{\"text\":\"Preserve validity boundaries\",\"level\":2},\"tunes\":{}},{\"id\":\"p-validity-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Important conclusions should carry the conditions under which they remain supported: version, date, scope, assumptions, source authority and unresolved disagreement.\"},\"tunes\":{}},{\"id\":\"p-validity-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Context engineering is therefore connected to the Answer Validity Boundary. The context assembler should not strip away the metadata that determines whether evidence still applies.\"},\"tunes\":{}},{\"id\":\"ref-avb\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers\",\"title\":\"The Answer Validity Boundary: The Missing Layer Between Relevance and Reliable AI Answers\",\"excerpt\":\"A framework for preserving the scope, assumptions, versions and evidence conditions under which an AI claim remains supported.\",\"ctaLabel\":\"Read the Answer Validity Boundary\"},\"tunes\":{}},{\"id\":\"h-security\",\"type\":\"header\",\"data\":{\"text\":\"Context engineering is also a security boundary\",\"level\":2},\"tunes\":{}},{\"id\":\"p-sec-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Data that reaches the model has crossed an important system boundary. Context assembly must therefore respect authorization, tenant isolation, confidentiality and data-minimization rules.\"},\"tunes\":{}},{\"id\":\"p-sec-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"A retriever may technically find a passage the current user cannot access. The correct design is to prevent that passage from entering model context rather than rely on the model to ignore it.\"},\"tunes\":{}},{\"id\":\"p-sec-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"Tool outputs can also contain untrusted instructions or adversarial content. Context engineering should preserve the distinction between application instructions and external data so retrieved text cannot silently acquire instruction authority.\"},\"tunes\":{}},{\"id\":\"h-architecture\",\"type\":\"header\",\"data\":{\"text\":\"A practical context-engineering architecture\",\"level\":2},\"tunes\":{}},{\"id\":\"arch-note\",\"type\":\"callout\",\"data\":{\"variant\":\"note\",\"title\":\"Proposed architecture model\",\"body\":\"The following layers are a practical synthesis for production systems, not a formal industry standard. The purpose is to keep information ownership separate from the temporary model-facing context.\"},\"tunes\":{}},{\"id\":\"arch-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Layer\",\"Responsibility\"],[\"Authoritative systems\",\"Own current business\u002Fsystem state and official records.\"],[\"Knowledge sources\",\"Own documents, policies, specifications, research or external evidence.\"],[\"Memory store\",\"Preserves selected information across turns or sessions.\"],[\"Retrieval layer\",\"Locates task-relevant candidates from external sources.\"],[\"Tool\u002Fruntime layer\",\"Reads state, performs actions and returns observations.\"],[\"Context assembler\",\"Selects, filters, deduplicates, orders and formats model-visible information.\"],[\"Model\",\"Reasons and generates over the assembled context.\"],[\"Validation\u002Fevaluation\",\"Checks whether selected context and resulting output satisfy task-specific requirements.\"]]},\"tunes\":{}},{\"id\":\"p-arch-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The context assembler is conceptually important even when no module has that exact name. In a small application it may be ordinary application code. In a large agent platform it may combine session management, retrieval, memory, tool middleware, compaction and policy enforcement.\"},\"tunes\":{}},{\"id\":\"h-policy\",\"type\":\"header\",\"data\":{\"text\":\"A practical context construction policy\",\"level\":2},\"tunes\":{}},{\"id\":\"policy-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Rule\",\"Why it matters\"],[\"Start from the current task\",\"Do not carry information merely because it existed earlier.\"],[\"Re-read volatile state\",\"Memory and old context can be stale.\"],[\"Retrieve just enough evidence\",\"Large candidate sets can dilute decisive information.\"],[\"Preserve source metadata\",\"Version, date and authority determine whether evidence still applies.\"],[\"Remove duplicate content\",\"Redundancy consumes tokens without adding information.\"],[\"Prefer structured summaries for large tool output\",\"Expose decisive fields instead of raw noise where fidelity permits.\"],[\"Keep rules with exceptions\",\"Separating a rule from its exception creates false certainty.\"],[\"Make precedence explicit\",\"Do not ask the model to infer which conflicting source wins.\"],[\"Keep durable state outside context\",\"Context is temporary working memory, not the database.\"],[\"Compact with retention tests\",\"Verify that identifiers, constraints, provenance and unresolved state survive.\"],[\"Measure order sensitivity\",\"Correctness should not depend accidentally on arbitrary document ordering.\"],[\"Evaluate context separately from model quality\",\"A stronger model cannot compensate reliably for missing or unauthorized evidence.\"]]},\"tunes\":{}},{\"id\":\"h-eval\",\"type\":\"header\",\"data\":{\"text\":\"How to evaluate context engineering\",\"level\":2},\"tunes\":{}},{\"id\":\"eval-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Property\",\"Question\",\"Example test\"],[\"Sufficiency\",\"Does the context contain everything required to solve the task?\",\"Remove one evidence item and observe whether the answer becomes unsupported.\"],[\"Relevance\",\"How much context is unnecessary for the task?\",\"Measure quality as irrelevant passages are added or removed.\"],[\"Authority\",\"Are decisive claims grounded in the correct source class?\",\"Inject a more fluent but non-authoritative conflicting source.\"],[\"Freshness\",\"Does current state override stale copies?\",\"Change authoritative state after a previous turn and rerun.\"],[\"Position robustness\",\"Does answer quality depend strongly on evidence position?\",\"Randomize candidate ordering across repeated trials.\"],[\"Conflict handling\",\"Does the model follow explicit precedence rules?\",\"Present old and new state together.\"],[\"Compaction retention\",\"Does summarization preserve constraints and validity boundaries?\",\"Compare pre\u002Fpost-compaction task performance.\"],[\"Token efficiency\",\"Does extra context improve quality enough to justify latency\u002Fcost?\",\"Run controlled context-size ablations.\"],[\"Security\",\"Can unauthorized or adversarial content enter model context?\",\"Test tenant, permission and prompt-injection boundaries.\"]]},\"tunes\":{}},{\"id\":\"h-rag-diagnostic\",\"type\":\"header\",\"data\":{\"text\":\"Context assembly is a distinct RAG failure layer\",\"level\":2},\"tunes\":{}},{\"id\":\"p-ragdiag-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A RAG pipeline can succeed at retrieval and still fail downstream. The relevant source may appear at rank 2, yet the context assembler can drop it, truncate it, combine it with stale contradictory material or exceed the token budget.\"},\"tunes\":{}},{\"id\":\"p-ragdiag-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"This is why retrieval traces should be compared with the actual context sent to the model. Without that comparison, context failures are easily misdiagnosed as embedding or model failures.\"},\"tunes\":{}},{\"id\":\"ref-ragfail\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method\",\"title\":\"RAG Failed — But Which Layer Actually Failed? A Diagnostic Method\",\"excerpt\":\"A layer-by-layer approach to separating source coverage, retrieval, ranking, context assembly, generation, evidence attribution and freshness failures.\",\"ctaLabel\":\"Read the RAG diagnostic method\"},\"tunes\":{}},{\"id\":\"h-implementation\",\"type\":\"header\",\"data\":{\"text\":\"Original implementation evidence\",\"level\":2},\"tunes\":{}},{\"id\":\"h-sot-engine\",\"type\":\"header\",\"data\":{\"text\":\"Source of Truth Research Engine: bounded research instead of unlimited context\",\"level\":3},\"tunes\":{}},{\"id\":\"p-sot-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The Source of Truth Research Engine separates discovery, acquisition, extraction, verification, contradiction analysis and synthesis into bounded research stages instead of sending one huge research task and all accumulated material into a single model call.\"},\"tunes\":{}},{\"id\":\"p-sot-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Its evidence model stores Sources, Artifacts, Claims, Relations, Contradictions and provenance outside the model context. The model can receive the subset needed for the current research step while durable evidence remains in the external store.\"},\"tunes\":{}},{\"id\":\"p-sot-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"That is a concrete context-engineering pattern: durable research state lives outside the model window; the active model context is reconstructed for the current stage.\"},\"tunes\":{}},{\"id\":\"h-ai-client\",\"type\":\"header\",\"data\":{\"text\":\"Aaasaasa AI Client: runtime, permissions and context are separate concerns\",\"level\":3},\"tunes\":{}},{\"id\":\"p-client-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Aaasaasa AI Client separates provider\u002Fmodel selection, runtime location, workspace permissions, local resources and tool access. This prevents the model context from becoming the owner of authorization or application state.\"},\"tunes\":{}},{\"id\":\"p-client-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Direct Chat and agentic runtimes can have different tool capabilities. Workspace permission profiles are enforced by the runtime rather than merely described in natural-language context. This distinction is important: context can tell a model what it should do, while the runtime must still enforce what it is actually allowed to do.\"},\"tunes\":{}},{\"id\":\"p-client-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The implementation evidence here is architectural separation, not a claim that every advanced context-management technique described in this article is already implemented.\"},\"tunes\":{}},{\"id\":\"impl-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Implementation pattern\",\"Context-engineering lesson\"],[\"External evidence store\",\"Durable knowledge does not need to remain in the model window.\"],[\"Bounded research stages\",\"Different steps can receive different context instead of accumulating one giant history.\"],[\"Claims + provenance outside context\",\"Evidence identity survives beyond temporary inference state.\"],[\"Runtime-enforced permissions\",\"Security authority does not depend on the model remembering an instruction.\"],[\"Separate local\u002Fprovider\u002Fmodel\u002Fruntime concepts\",\"Context is only one layer of the wider AI application architecture.\"]]},\"tunes\":{}},{\"id\":\"impl-boundary\",\"type\":\"callout\",\"data\":{\"variant\":\"note\",\"title\":\"Evidence boundary\",\"body\":\"These implementations support the architectural separation between durable state, retrieval, runtime controls and model-facing context. They are not presented as benchmark proof that one context strategy is universally optimal.\"},\"tunes\":{}},{\"id\":\"h-failures\",\"type\":\"header\",\"data\":{\"text\":\"Common context-engineering failure modes\",\"level\":2},\"tunes\":{}},{\"id\":\"failure-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Failure mode\",\"What goes wrong\"],[\"Replay the entire conversation forever\",\"Old assumptions, repetition and token growth overwhelm current intent.\"],[\"Put every retrieved result into the prompt\",\"Noise, duplication and conflicting versions dilute decisive evidence.\"],[\"Use memory as current state\",\"Stale information silently replaces authoritative live state.\"],[\"Return raw tool output\",\"Large logs or responses consume attention without adding decision value.\"],[\"Hide tool descriptions behind vague names\",\"The model cannot reliably decide which capability to use.\"],[\"Compact without retention tests\",\"Critical constraints, identifiers or exceptions disappear.\"],[\"Mix instructions and untrusted data\",\"External content can be interpreted as higher-authority instruction.\"],[\"Use one static context template for every task\",\"Different tasks receive irrelevant information and miss task-specific evidence.\"],[\"Ignore source version\u002Fdate\",\"Stale but relevant evidence can dominate current authoritative state.\"],[\"Treat a larger context window as a quality guarantee\",\"Capacity increases while attention and conflict problems remain.\"]]},\"tunes\":{}},{\"id\":\"h-misconceptions\",\"type\":\"header\",\"data\":{\"text\":\"Common misconceptions\",\"level\":2},\"tunes\":{}},{\"id\":\"misconceptions-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Misconception\",\"Correction\"],[\"“Context engineering is just prompt engineering with a new name.”\",\"Prompts are one component; context engineering also covers retrieval, memory, state, tool results, history and compaction.\"],[\"“Context means chat history.”\",\"History is only one possible context source.\"],[\"“More context is always better.”\",\"Additional information can reduce signal, introduce conflicts and increase cost.\"],[\"“If retrieval found it, the model saw it.”\",\"Retrieved candidates can be filtered, truncated or omitted before inference.\"],[\"“Long context removes the need for RAG.”\",\"Large windows increase capacity but do not solve freshness, authority, permissions or dynamic retrieval.\"],[\"“Memory should always be loaded.”\",\"Memory should be selected according to the current task.\"],[\"“A summary preserves everything important.”\",\"Compaction is lossy unless explicitly evaluated for retention.\"],[\"“Instructions can enforce permissions.”\",\"Authorization must be enforced by runtime\u002Fapplication controls, not only by context.\"],[\"“One context recipe works for every model.”\",\"Context sensitivity varies by model, task, corpus and runtime.\"],[\"“Context engineering is only for agents.”\",\"Agents amplify the need, but ordinary RAG and conversational applications also require context construction.\"]]},\"tunes\":{}},{\"id\":\"h-sequence\",\"type\":\"header\",\"data\":{\"text\":\"A practical context-engineering sequence\",\"level\":2},\"tunes\":{}},{\"id\":\"design-sequence\",\"type\":\"processFlow\",\"data\":{\"title\":\"Construct context from the current decision backward\",\"orientation\":\"auto\",\"steps\":[{\"label\":\"1. Define the next model decision\",\"description\":\"Specify what the model must answer, classify, plan or choose at this step.\"},{\"label\":\"2. Identify required facts and constraints\",\"description\":\"List the minimum state, rules, evidence and instructions that can materially change the result.\"},{\"label\":\"3. Resolve authority and permissions\",\"description\":\"Determine which sources are current, authoritative and accessible to the current principal.\"},{\"label\":\"4. Retrieve or read on demand\",\"description\":\"Acquire the necessary evidence and volatile state rather than relying on stale context.\"},{\"label\":\"5. Reduce noise\",\"description\":\"Deduplicate, summarize or select passages without discarding decisive exceptions or provenance.\"},{\"label\":\"6. Structure and order\",\"description\":\"Make instructions, current state, evidence and tool observations distinguishable.\"},{\"label\":\"7. Fit the token budget\",\"description\":\"Prefer high-signal context and move durable information outside the window.\"},{\"label\":\"8. Run the model\",\"description\":\"Execute inference over the assembled context.\"},{\"label\":\"9. Observe failures\",\"description\":\"Capture whether the problem came from missing, stale, noisy, conflicting or poorly ordered context.\"},{\"label\":\"10. Re-evaluate after model\u002Fruntime changes\",\"description\":\"A context strategy is only valid for the models, tools and workloads on which it was tested.\"}]},\"tunes\":{}},{\"id\":\"h-checklist\",\"type\":\"header\",\"data\":{\"text\":\"Context-engineering checklist\",\"level\":2},\"tunes\":{}},{\"id\":\"checklist-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Question\",\"Expected answer\"],[\"What exact decision will the model make next?\",\"A bounded task, not a vague long-term objective.\"],[\"Which information can materially change that decision?\",\"Explicit minimum evidence\u002Fstate set.\"],[\"Which data is authoritative now?\",\"Current source\u002Fversion and freshness rule.\"],[\"Which data is optional background?\",\"Separated from decisive evidence.\"],[\"What must not enter context?\",\"Unauthorized, unnecessary or overly sensitive data.\"],[\"Which memory items are relevant?\",\"Selected by task, not replayed automatically.\"],[\"Which tool outputs should be reduced?\",\"Large responses are transformed into decision-relevant form.\"],[\"Which constraints must survive compaction?\",\"Identifiers, exceptions, obligations, unresolved state and provenance.\"],[\"How is precedence represented?\",\"Current\u002Fauthoritative information can reliably override stale or weaker sources.\"],[\"How will you know context failed?\",\"Context-specific evals and traces exist.\"],[\"Can the answer be reproduced?\",\"Model input or reconstructable context trace is available where appropriate.\"],[\"Can a stronger or larger model change the strategy?\",\"Context policy is version-aware and reevaluated empirically.\"]]},\"tunes\":{}},{\"id\":\"h-edge\",\"type\":\"header\",\"data\":{\"text\":\"Edge cases and limitations\",\"level\":2},\"tunes\":{}},{\"id\":\"p-edge-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Some tasks are simple enough that context engineering reduces to a short system prompt and one user message. Adding retrieval, memory and compaction would only introduce unnecessary architecture.\"},\"tunes\":{}},{\"id\":\"p-edge-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Some tasks require high recall and may intentionally include more context before later synthesis. Research, discovery and legal review can prefer omission avoidance over minimal token count.\"},\"tunes\":{}},{\"id\":\"p-edge-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"Some information should never be summarized before use. Exact contracts, code, cryptographic material, numerical records and regulatory text may require verbatim or structured retrieval where compression could alter meaning.\"},\"tunes\":{}},{\"id\":\"p-edge-4\",\"type\":\"paragraph\",\"data\":{\"text\":\"Long-context behavior varies substantially between models. A strategy validated on one model, context length or tool harness should not automatically be transferred to another.\"},\"tunes\":{}},{\"id\":\"p-edge-5\",\"type\":\"paragraph\",\"data\":{\"text\":\"The model can still ignore or misinterpret excellent context. Context engineering improves the information environment; it does not guarantee reasoning correctness.\"},\"tunes\":{}},{\"id\":\"h-change\",\"type\":\"header\",\"data\":{\"text\":\"What would change this answer?\",\"level\":2},\"tunes\":{}},{\"id\":\"p-change-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Future models may become more robust to long context, positional effects and conflicting information. That could reduce the amount of manual curation required.\"},\"tunes\":{}},{\"id\":\"p-change-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The architectural distinction would still remain useful because permissions, freshness, memory persistence, source authority and external application state exist outside the model regardless of context-window size.\"},\"tunes\":{}},{\"id\":\"p-change-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The recommended balance between preloaded and just-in-time context also changes with latency requirements, tool reliability, corpus size, model cost and how dynamic the underlying information is.\"},\"tunes\":{}},{\"id\":\"h-related\",\"type\":\"header\",\"data\":{\"text\":\"Related canonical knowledge\",\"level\":2},\"tunes\":{}},{\"id\":\"p-related-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Context engineering sits between retrieval and generation. RAG explains how external knowledge is retrieved; R01 separates embeddings, vector search and reranking; context engineering explains what eventually reaches the model.\"},\"tunes\":{}},{\"id\":\"ref-rag\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works\",\"title\":\"What Is RAG? The Simplest Explanation of How It Works\",\"excerpt\":\"The retrieval foundation for understanding how external knowledge can be supplied to a model before generation.\",\"ctaLabel\":\"Read the RAG foundation\"},\"tunes\":{}},{\"id\":\"p-related-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Source-of-Truth architecture answers a different question: not which information is present in context, but which source is authorized to establish a claim.\"},\"tunes\":{}},{\"id\":\"p-related-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The existing article Why More Context Can Make AI Answers Worse is the diagnostic companion to this canonical definition. It focuses on context pollution, position effects, top-k growth, compaction loss and answer degradation rather than redefining context engineering itself.\"},\"tunes\":{}},{\"id\":\"h-faq\",\"type\":\"header\",\"data\":{\"text\":\"Frequently asked questions\",\"level\":2},\"tunes\":{}},{\"id\":\"faq\",\"type\":\"faq\",\"data\":{\"title\":\"Context engineering FAQ\",\"items\":[{\"id\":\"faq1\",\"question\":\"What is context engineering?\",\"answer\":\"Context engineering is the design and runtime management of what information a language model receives at inference time, including instructions, history, retrieved evidence, memory, state, tools and tool results.\"},{\"id\":\"faq2\",\"question\":\"How is context engineering different from prompt engineering?\",\"answer\":\"Prompt engineering focuses on how instructions and examples are written. Context engineering includes prompts but also decides which external information, state, history, memory and tool observations are placed around them.\"},{\"id\":\"faq3\",\"question\":\"Is RAG the same as context engineering?\",\"answer\":\"No. RAG retrieves external information. Context engineering decides how retrieved information is filtered, combined with other state and actually delivered to the model.\"},{\"id\":\"faq4\",\"question\":\"Is memory the same as context?\",\"answer\":\"No. Memory persists information outside the current model call. Context is the subset of information loaded into the current inference.\"},{\"id\":\"faq5\",\"question\":\"Why can more context make an answer worse?\",\"answer\":\"Additional context can introduce noise, stale state, conflicting evidence, duplication and positional competition. Large context capacity does not guarantee equally reliable use of every token.\"},{\"id\":\"faq6\",\"question\":\"What is context compaction?\",\"answer\":\"Compaction summarizes or transforms accumulated history into a smaller representation so a long-running system can continue without replaying every prior token.\"},{\"id\":\"faq7\",\"question\":\"Should current application state be stored in context?\",\"answer\":\"It can be represented in context for reasoning, but consequential operations should often re-read the authoritative source because context snapshots can become stale.\"},{\"id\":\"faq8\",\"question\":\"Is context engineering only needed for AI agents?\",\"answer\":\"No. Agents make context management more dynamic, but RAG systems, assistants, copilots and multi-turn applications also need deliberate context construction.\"}]},\"tunes\":{}},{\"id\":\"h-glossary\",\"type\":\"header\",\"data\":{\"text\":\"Glossary\",\"level\":2},\"tunes\":{}},{\"id\":\"glossary\",\"type\":\"glossary\",\"data\":{\"title\":\"Key context-engineering terms\",\"entries\":[{\"term\":\"Context engineering\",\"definition\":\"The design and runtime management of the information supplied to a language model for a particular inference step.\",\"anchor\":\"context-engineering\"},{\"term\":\"Context window\",\"definition\":\"The model's finite token capacity for the input and, depending on the model interface, associated generated tokens or active sequence.\",\"anchor\":\"context-window\"},{\"term\":\"Prompt engineering\",\"definition\":\"The design of instructions, examples and prompt structure intended to elicit useful model behavior.\",\"anchor\":\"prompt-engineering\"},{\"term\":\"Context assembly\",\"definition\":\"The process of selecting, filtering, ordering and formatting model-visible information before inference.\",\"anchor\":\"context-assembly\"},{\"term\":\"Just-in-time retrieval\",\"definition\":\"Loading information dynamically when the current task requires it instead of preloading all potentially relevant data.\",\"anchor\":\"just-in-time-retrieval\"},{\"term\":\"Compaction\",\"definition\":\"Reducing accumulated context into a smaller representation while attempting to preserve information needed for future steps.\",\"anchor\":\"compaction\"},{\"term\":\"Context pollution\",\"definition\":\"Degradation caused by irrelevant, stale, contradictory or redundant information occupying the model's working context.\",\"anchor\":\"context-pollution\"},{\"term\":\"Application state\",\"definition\":\"The current authoritative condition of the external system, workflow or domain that exists independently of the model context.\",\"anchor\":\"application-state\"},{\"term\":\"Memory\",\"definition\":\"Information stored outside the immediate model invocation for possible use in later turns or sessions.\",\"anchor\":\"memory\"},{\"term\":\"Retrieved context\",\"definition\":\"External information selected by a retrieval system and made available, wholly or partly, to the model.\",\"anchor\":\"retrieved-context\"},{\"term\":\"Position robustness\",\"definition\":\"The degree to which model correctness remains stable when the location or order of relevant context changes.\",\"anchor\":\"position-robustness\"},{\"term\":\"Validity boundary\",\"definition\":\"The scope, time, assumptions, versions and evidence conditions within which a conclusion remains supported.\",\"anchor\":\"validity-boundary\"}]},\"tunes\":{}},{\"id\":\"h-conclusion\",\"type\":\"header\",\"data\":{\"text\":\"Conclusion\",\"level\":2},\"tunes\":{}},{\"id\":\"p-conclusion-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Context engineering is the layer that decides what the model gets to see before it answers. That makes it broader than prompting and downstream of retrieval, while remaining distinct from durable memory and authoritative application state.\"},\"tunes\":{}},{\"id\":\"p-conclusion-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"A strong context architecture does not treat the context window as a database. It keeps durable state and knowledge outside the model, loads what is required for the current decision, preserves authority and provenance, removes unnecessary noise and refreshes volatile information when needed.\"},\"tunes\":{}},{\"id\":\"p-conclusion-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The practical objective is therefore not maximum context. It is minimum sufficient, high-signal, correctly authorized and validity-preserving context for the next model decision.\"},\"tunes\":{}},{\"id\":\"h-sources\",\"type\":\"header\",\"data\":{\"text\":\"Primary sources and current guidance\",\"level\":2},\"tunes\":{}},{\"id\":\"p-sources-note\",\"type\":\"paragraph\",\"data\":{\"text\":\"The sources below support the current context-engineering terminology, long-context behavior and operational context-management patterns. Project sections are explicitly implementation evidence rather than universal claims.\"},\"tunes\":{}},{\"id\":\"src-anthropic\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Feffective-context-engineering-for-ai-agents\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Anthropic — Effective context engineering for AI agents\",\"description\":\"Official engineering guidance defining context engineering, just-in-time retrieval, compaction, structured memory and context curation for agents.\"}},\"tunes\":{}},{\"id\":\"src-openai-session\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdevelopers.openai.com\u002Fcookbook\u002Fexamples\u002Fagents_sdk\u002Fsession_memory\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — Context Engineering: Short-Term Memory Management with Sessions\",\"description\":\"Official cookbook guidance on context management, trimming and compression for long-running agent sessions.\"}},\"tunes\":{}},{\"id\":\"src-openai-agents\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fagents\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — Agents guide\",\"description\":\"Current OpenAI developer guidance on agent runtimes, context across steps and orchestration ownership.\"}},\"tunes\":{}},{\"id\":\"src-lost-middle\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2307.03172\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Lost in the Middle: How Language Models Use Long Contexts\",\"description\":\"Research showing that long-context model performance can depend strongly on the position of relevant information in the input.\"}},\"tunes\":{}}],\"version\":\"2.31.6\"}",{"time":1449,"blocks":1450,"version":2457},1791480654232,[1451,1455,1460,1465,1470,1474,1478,1482,1486,1490,1494,1498,1502,1506,1532,1536,1540,1544,1548,1600,1604,1629,1633,1637,1641,1645,1649,1654,1658,1662,1666,1670,1677,1681,1685,1689,1694,1698,1702,1706,1710,1714,1736,1740,1744,1748,1752,1756,1760,1764,1768,1772,1776,1780,1784,1788,1792,1796,1800,1825,1829,1833,1837,1841,1846,1850,1854,1858,1865,1869,1873,1877,1881,1885,1890,1920,1924,1928,1971,1975,2019,2023,2027,2031,2038,2042,2046,2050,2054,2058,2062,2066,2070,2074,2096,2101,2105,2142,2146,2183,2187,2222,2226,2268,2272,2276,2280,2284,2288,2292,2296,2300,2304,2308,2312,2316,2323,2327,2331,2335,2364,2368,2405,2409,2413,2417,2421,2425,2429,2436,2443,2450],{"id":215,"data":1452,"type":218,"tunes":1454},{"text":1453},"Context engineering is the design of what information a language model receives at inference time, in what form, in what order and for how long. It is broader than prompt engineering because the model context can include system instructions, user messages, retrieved documents, tool results, memory, current application state, examples, structured data and intermediate artifacts. The goal is not to maximize the number of tokens, but to construct the smallest useful context that preserves the information, constraints and evidence needed for the current task.",{},{"id":221,"data":1456,"type":226,"tunes":1459},{"body":1457,"title":1458,"variant":225},"Prompt engineering asks \u003Cstrong>how should we instruct the model?\u003C\u002Fstrong> Context engineering asks \u003Cstrong>what should the model know right now, and how should that information be assembled?\u003C\u002Fstrong>\u003Cbr>\u003Cbr>Retrieval, memory, state management, tool design, history trimming, compaction and ordering are therefore context-engineering mechanisms when they determine the tokens available to the model before it produces the next output.","Direct answer",{},{"id":229,"data":1461,"type":226,"tunes":1464},{"body":1462,"title":1463,"variant":233},"A system can know something without placing it in the current context. It can remember something outside the model window. It can retrieve a document but later exclude it from the final prompt. The model can only directly use the context that reaches the current inference.","Context is not the same as knowledge or memory",{},{"id":236,"data":1466,"type":226,"tunes":1469},{"body":1467,"title":1468,"variant":240},"Context engineering is now established practical terminology in major AI engineering guidance, but it is not a single formal standard with one mandatory architecture. Anthropic describes it as curating and maintaining the optimal set of tokens for inference; OpenAI's current agent guidance treats session context, trimming and compression as explicit engineering concerns for long-running systems.","Current-source note — 8 October 2026",{},{"id":243,"data":1471,"type":248,"tunes":1473},{"title":1472,"maxLevel":246,"minLevel":247},"Contents",{},{"id":251,"data":1475,"type":42,"tunes":1477},{"text":1476,"level":247},"What context engineering really means",{},{"id":256,"data":1479,"type":218,"tunes":1481},{"text":1480},"Every model call is made under a temporary working environment: the current instructions, messages, retrieved evidence, tool outputs and state that fit into the active context window. Context engineering is the discipline of constructing that environment deliberately.",{},{"id":261,"data":1483,"type":218,"tunes":1485},{"text":1484},"The key word is deliberately. A naive system simply concatenates everything it has: full history, all retrieved documents, every tool response and large system prompts. A context-engineered system decides which information is required for the current decision and which information should remain outside the window until needed.",{},{"id":266,"data":1487,"type":218,"tunes":1489},{"text":1488},"This makes context engineering partly an information-architecture problem, partly a runtime problem and partly an evaluation problem. The design must decide what can enter context, where it comes from, which version is current, how conflicts are resolved, how much detail is retained and how the result is tested.",{},{"id":271,"data":1491,"type":42,"tunes":1493},{"text":1492,"level":247},"The simplest example",{},{"id":276,"data":1495,"type":218,"tunes":1497},{"text":1496},"Imagine an internal support assistant. A user asks: “Can this customer cancel without a fee?”",{},{"id":281,"data":1499,"type":218,"tunes":1501},{"text":1500},"The model might need five things: the current cancellation policy, the customer's current contract type, the effective contract date, the relevant exception rules and the user's authorization scope.",{},{"id":286,"data":1503,"type":218,"tunes":1505},{"text":1504},"It does not necessarily need the entire customer database, the full policy archive, every previous conversation or every support ticket. Context engineering is the process that selects and assembles the five useful pieces while excluding unrelated information.",{},{"id":291,"data":1507,"type":317,"tunes":1531},{"steps":1508,"title":1530,"orientation":316},[1509,1512,1515,1518,1521,1524,1527],{"label":1510,"description":1511},"1. Understand the task","Classify what the current question requires and which information types can affect the answer.",{"label":1513,"description":1514},"2. Resolve authoritative state","Read current application or business state that should not be guessed from memory.",{"label":1516,"description":1517},"3. Retrieve supporting knowledge","Find the policy, documents or external evidence relevant to the specific task.",{"label":1519,"description":1520},"4. Apply eligibility and permissions","Exclude data the current user or runtime is not allowed to expose to the model.",{"label":1522,"description":1523},"5. Reduce and structure","Remove duplication, select useful excerpts and preserve critical metadata, conditions and exceptions.",{"label":1525,"description":1526},"6. Order the context","Place instructions, current state and decisive evidence where the model can use them consistently.",{"label":1528,"description":1529},"7. Run inference","The model receives the assembled context and produces the next answer or action proposal.","From application state to model context",{},{"id":320,"data":1533,"type":42,"tunes":1535},{"text":1534,"level":247},"Where the simple example stops",{},{"id":325,"data":1537,"type":218,"tunes":1539},{"text":1538},"Real systems are more difficult because the information needed for one step may not be known before execution begins. An agent can discover new facts through tools, create intermediate files, receive changing external state or span a task longer than one context window.",{},{"id":330,"data":1541,"type":218,"tunes":1543},{"text":1542},"Context engineering therefore becomes dynamic. The context for step 12 should not simply be step 1 context plus eleven layers of accumulated output. It should reflect the current task state, the decisions that still matter and the evidence required for the next action.",{},{"id":335,"data":1545,"type":42,"tunes":1547},{"text":1546,"level":247},"What can enter a model context?",{},{"id":340,"data":1549,"type":391,"tunes":1599},{"content":1550,"stretched":43,"withHeadings":14},[1551,1555,1559,1563,1567,1571,1575,1579,1583,1587,1591,1595],[1552,1553,1554],"Context component","Purpose","Typical risk",[1556,1557,1558],"System \u002F developer instructions","Define role, constraints, policies and behavior","Too vague, contradictory or overloaded with brittle logic",[1560,1561,1562],"Current user request","Defines immediate task and intent","Ambiguity or conflict with prior history",[1564,1565,1566],"Conversation history","Preserves continuity across turns","Stale assumptions, repetition and token growth",[1568,1569,1570],"Retrieved documents","Provide external knowledge\u002Fevidence","Irrelevance, stale versions, weak authority or duplication",[1572,1573,1574],"Current application state","Supplies volatile business\u002Fsystem facts","Using cached or remembered state instead of current authority",[1576,1577,1578],"Tool definitions","Tell the model what capabilities exist and how to call them","Too many overlapping tools or verbose schemas",[1580,1581,1582],"Tool results","Bring observations from the environment into the loop","Large noisy outputs, untrusted content or obsolete observations",[1584,1585,1586],"Memory","Reintroduces selected information from previous interactions","Staleness, incorrect generalization or over-personalization",[1588,1589,1590],"Examples","Demonstrate desired behavior","Too many edge cases can crowd out the current task",[1592,1593,1594],"Intermediate artifacts","Carry plans, summaries, code, calculations or notes","Old intermediate state may be mistaken for final truth",[1596,1597,1598],"Policies \u002F guardrails","Define prohibited or constrained behavior","Conflict with business logic or hidden enforcement gaps",{},{"id":394,"data":1601,"type":42,"tunes":1603},{"text":1602,"level":247},"Context engineering vs prompt engineering",{},{"id":399,"data":1605,"type":431,"tunes":1628},{"rows":1606,"title":1622,"layout":391,"columns":1623},[1607,1610,1613,1616,1619],{"id":403,"label":1608,"values":1609},"Primary focus",[406,406],{"id":408,"label":1611,"values":1612},"Typical scope",[406,406],{"id":412,"label":1614,"values":1615},"When it changes",[406,406],{"id":416,"label":1617,"values":1618},"Typical failure",[406,406],{"id":420,"label":1620,"values":1621},"Relationship",[406,406],"Prompt engineering and context engineering solve different layers",[1624,1626],{"id":426,"label":1625},"Prompt engineering",{"id":429,"label":1627},"Context engineering",{},{"id":434,"data":1630,"type":218,"tunes":1632},{"text":1631},"Anthropic explicitly describes context engineering as the natural progression of prompt engineering for systems in which the model must work with tools, external data, message history and long-running agent state. The practical distinction is useful because a perfectly written prompt cannot compensate for missing authoritative data or a context polluted by contradictory state.",{},{"id":439,"data":1634,"type":42,"tunes":1636},{"text":1635,"level":247},"Context engineering vs retrieval",{},{"id":444,"data":1638,"type":218,"tunes":1640},{"text":1639},"Retrieval selects candidate information from an external corpus or source. Context engineering decides what happens after and around that retrieval.",{},{"id":449,"data":1642,"type":218,"tunes":1644},{"text":1643},"The retriever may return 30 passages. A reranker may reduce them to 10. The context layer may select four passages, remove duplicates, attach source\u002Fversion metadata, combine them with current application state and place them after the system instructions.",{},{"id":454,"data":1646,"type":218,"tunes":1648},{"text":1647},"This is why a RAG system can retrieve the correct passage and still answer badly: the failure may occur during context assembly rather than retrieval.",{},{"id":459,"data":1650,"type":226,"tunes":1653},{"body":1651,"title":1652,"variant":463},"The correct retrieval result is only useful if it survives filtering, ordering, compression and token-budget decisions and actually reaches the model in a usable form.","Retrieval finds candidates; context engineering constructs the model input",{},{"id":466,"data":1655,"type":42,"tunes":1657},{"text":1656,"level":247},"Context engineering vs memory",{},{"id":471,"data":1659,"type":218,"tunes":1661},{"text":1660},"Memory is information preserved outside the immediate model invocation so it can be used again later. Context is the information actually loaded into the current invocation.",{},{"id":476,"data":1663,"type":218,"tunes":1665},{"text":1664},"A memory system may contain thousands of facts, notes or prior decisions. Context engineering selects which of those should be reintroduced for the current task. Loading all memory on every turn defeats the purpose of having an external memory layer.",{},{"id":481,"data":1667,"type":218,"tunes":1669},{"text":1668},"The distinction becomes crucial for volatile state. A remembered project status or user preference can be useful, but current authoritative state may need to be re-read before a consequential decision.",{},{"id":486,"data":1671,"type":492,"tunes":1676},{"url":1672,"title":1673,"excerpt":1674,"ctaLabel":1675},"https:\u002F\u002Fstajic.de\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context","AI Agent Memory Is Not RAG: How to Separate Memory, Retrieval, State and Context","A practical architecture separating what persists, what is authoritative now, what is retrieved and what the model actually receives.","Read the memory architecture article",{},{"id":495,"data":1678,"type":42,"tunes":1680},{"text":1679,"level":247},"Context engineering vs application state",{},{"id":500,"data":1682,"type":218,"tunes":1684},{"text":1683},"Application state is the current condition of the outside system: account balance, ticket status, file version, workflow stage, deployment state or task progress.",{},{"id":505,"data":1686,"type":218,"tunes":1688},{"text":1687},"State can be summarized into context, but the summary is not the state itself. For consequential operations, the runtime may need to re-read the authoritative system immediately before the action rather than trust an earlier model-visible snapshot.",{},{"id":510,"data":1690,"type":226,"tunes":1693},{"body":1691,"title":1692,"variant":233},"Once state is copied into a prompt, it can become stale. Context engineering must define when volatile state needs refreshing and which operations require a new authoritative read.","Context is a snapshot",{},{"id":516,"data":1695,"type":42,"tunes":1697},{"text":1696,"level":247},"Tool design is part of context engineering",{},{"id":521,"data":1699,"type":218,"tunes":1701},{"text":1700},"Tools do more than give agents capabilities. Tool names, descriptions, schemas and results become model-visible information that shapes decisions.",{},{"id":526,"data":1703,"type":218,"tunes":1705},{"text":1704},"Anthropic's current context-engineering guidance emphasizes token-efficient tools and warns against bloated tool sets with overlapping functionality. A tool catalog that is difficult for a human to distinguish is also difficult for a model to route reliably.",{},{"id":531,"data":1707,"type":218,"tunes":1709},{"text":1708},"Tool outputs also need context discipline. Returning an entire 20,000-line log when the agent requested one error condition consumes attention and can bury the decisive evidence.",{},{"id":536,"data":1711,"type":42,"tunes":1713},{"text":1712,"level":247},"Just-in-time context vs preloaded context",{},{"id":541,"data":1715,"type":431,"tunes":1735},{"rows":1716,"title":1729,"layout":391,"columns":1730},[1717,1720,1723,1726],{"id":545,"label":1718,"values":1719},"Method",[406,406],{"id":549,"label":1721,"values":1722},"Strength",[406,406],{"id":553,"label":1724,"values":1725},"Risk",[406,406],{"id":557,"label":1727,"values":1728},"Useful when",[406,406],"Two ways to supply information",[1731,1733],{"id":563,"label":1732},"Preloaded context",{"id":566,"label":1734},"Just-in-time context",{},{"id":570,"data":1737,"type":218,"tunes":1739},{"text":1738},"Anthropic describes a hybrid pattern in which some stable context is preloaded while agents retrieve additional information at runtime. This is a useful architecture pattern because not every important fact deserves permanent residency in the context window.",{},{"id":575,"data":1741,"type":42,"tunes":1743},{"text":1742,"level":247},"Context is a budget, not a storage system",{},{"id":580,"data":1745,"type":218,"tunes":1747},{"text":1746},"A context window defines capacity. It does not guarantee that every token will be used equally well. The model must distribute attention across instructions, history, evidence, tools and intermediate state.",{},{"id":585,"data":1749,"type":218,"tunes":1751},{"text":1750},"The practical objective is therefore not “fill the window.” It is to maximize the utility of the limited attention budget.",{},{"id":590,"data":1753,"type":218,"tunes":1755},{"text":1754},"Anthropic formulates a similar principle as finding the smallest high-signal set of tokens that maximizes the probability of the desired behavior. OpenAI's context-management guidance likewise warns that uncurated history, redundant tool results and noisy retrieval can overwhelm even large windows.",{},{"id":595,"data":1757,"type":42,"tunes":1759},{"text":1758,"level":247},"Why more context can be worse",{},{"id":600,"data":1761,"type":218,"tunes":1763},{"text":1762},"Additional context can introduce irrelevant information, stale state, duplicate evidence, contradictory instructions or positional competition. It can also cause compaction systems to discard details that later become important.",{},{"id":605,"data":1765,"type":218,"tunes":1767},{"text":1766},"The classic Lost in the Middle study demonstrated that long-context models can use information differently depending on where relevant content appears, with performance often degrading when decisive information is placed in the middle of long inputs.",{},{"id":610,"data":1769,"type":218,"tunes":1771},{"text":1770},"This does not mean long context is inherently bad. It means availability inside the window is not the same as reliable utilization.",{},{"id":615,"data":1773,"type":42,"tunes":1775},{"text":1774,"level":247},"Context ordering should be intentional",{},{"id":620,"data":1777,"type":218,"tunes":1779},{"text":1778},"Context construction is also an ordering problem. Critical instructions, current state, decisive evidence and task-specific constraints should not be concatenated arbitrarily.",{},{"id":625,"data":1781,"type":218,"tunes":1783},{"text":1782},"There is no universal perfect ordering for every model and task. The architecture should therefore test whether reordering evidence changes correctness and whether important information remains robust across realistic context variations.",{},{"id":630,"data":1785,"type":218,"tunes":1787},{"text":1786},"A stable answer that changes dramatically when two equally valid passages swap positions indicates context sensitivity that should be measured rather than ignored.",{},{"id":635,"data":1789,"type":42,"tunes":1791},{"text":1790,"level":247},"Conflicting context needs explicit precedence",{},{"id":640,"data":1793,"type":218,"tunes":1795},{"text":1794},"A model may receive an old policy and a new policy, a remembered preference and a current explicit instruction, or a cached status and a live API result. The system should not expect the model to infer precedence from prose style.",{},{"id":645,"data":1797,"type":218,"tunes":1799},{"text":1798},"Context engineering should encode precedence through source selection, metadata, ordering or explicit instructions: current authoritative state overrides stale copies; explicit current user instruction overrides older inferred preference; approved policy supersedes obsolete drafts.",{},{"id":650,"data":1801,"type":391,"tunes":1824},{"content":1802,"stretched":43,"withHeadings":14},[1803,1806,1809,1812,1815,1818,1821],[1804,1805],"Conflict","Preferred context rule",[1807,1808],"Current state vs remembered state","Refresh and prefer the authoritative current source.",[1810,1811],"Current policy vs superseded policy","Include current version; keep old version only when historical comparison is required.",[1813,1814],"Explicit user instruction vs old inferred preference","Prefer the current explicit instruction.",[1816,1817],"Primary source vs secondary summary","Use primary source for claims that require authority; summary may support explanation.",[1819,1820],"Tool observation vs model prior","Prefer current observed state when the tool is authoritative for that fact.",[1822,1823],"Two unresolved authoritative sources","Expose the conflict rather than fabricating one consistent answer.",{},{"id":676,"data":1826,"type":42,"tunes":1828},{"text":1827,"level":247},"Compaction is context transformation, not lossless storage",{},{"id":681,"data":1830,"type":218,"tunes":1832},{"text":1831},"Long-running systems eventually need to trim, summarize or compact history. Compaction creates a new representation of prior context so the agent can continue without replaying every token.",{},{"id":686,"data":1834,"type":218,"tunes":1836},{"text":1835},"OpenAI's context-management examples use trimming and compression for long-running sessions. Anthropic describes compaction as a primary technique for maintaining coherence when an interaction approaches the context limit.",{},{"id":691,"data":1838,"type":218,"tunes":1840},{"text":1839},"The difficult part is deciding what cannot be safely removed: unresolved tasks, identifiers, user constraints, security boundaries, architecture decisions, exceptions, source provenance and the conditions that make a previous conclusion valid.",{},{"id":696,"data":1842,"type":226,"tunes":1845},{"body":1843,"title":1844,"variant":233},"If compaction keeps “use approach X” but discards why X was chosen, which version was tested or what condition would invalidate it, later responses can remain internally consistent while becoming externally wrong.","A summary can preserve the conclusion and destroy the reason",{},{"id":702,"data":1847,"type":42,"tunes":1849},{"text":1848,"level":247},"Preserve validity boundaries",{},{"id":707,"data":1851,"type":218,"tunes":1853},{"text":1852},"Important conclusions should carry the conditions under which they remain supported: version, date, scope, assumptions, source authority and unresolved disagreement.",{},{"id":712,"data":1855,"type":218,"tunes":1857},{"text":1856},"Context engineering is therefore connected to the Answer Validity Boundary. The context assembler should not strip away the metadata that determines whether evidence still applies.",{},{"id":717,"data":1859,"type":492,"tunes":1864},{"url":1860,"title":1861,"excerpt":1862,"ctaLabel":1863},"https:\u002F\u002Fstajic.de\u002Fblog\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers","The Answer Validity Boundary: The Missing Layer Between Relevance and Reliable AI Answers","A framework for preserving the scope, assumptions, versions and evidence conditions under which an AI claim remains supported.","Read the Answer Validity Boundary",{},{"id":725,"data":1866,"type":42,"tunes":1868},{"text":1867,"level":247},"Context engineering is also a security boundary",{},{"id":730,"data":1870,"type":218,"tunes":1872},{"text":1871},"Data that reaches the model has crossed an important system boundary. Context assembly must therefore respect authorization, tenant isolation, confidentiality and data-minimization rules.",{},{"id":735,"data":1874,"type":218,"tunes":1876},{"text":1875},"A retriever may technically find a passage the current user cannot access. The correct design is to prevent that passage from entering model context rather than rely on the model to ignore it.",{},{"id":740,"data":1878,"type":218,"tunes":1880},{"text":1879},"Tool outputs can also contain untrusted instructions or adversarial content. Context engineering should preserve the distinction between application instructions and external data so retrieved text cannot silently acquire instruction authority.",{},{"id":745,"data":1882,"type":42,"tunes":1884},{"text":1883,"level":247},"A practical context-engineering architecture",{},{"id":750,"data":1886,"type":226,"tunes":1889},{"body":1887,"title":1888,"variant":240},"The following layers are a practical synthesis for production systems, not a formal industry standard. The purpose is to keep information ownership separate from the temporary model-facing context.","Proposed architecture model",{},{"id":756,"data":1891,"type":391,"tunes":1919},{"content":1892,"stretched":43,"withHeadings":14},[1893,1896,1899,1902,1905,1908,1911,1914,1916],[1894,1895],"Layer","Responsibility",[1897,1898],"Authoritative systems","Own current business\u002Fsystem state and official records.",[1900,1901],"Knowledge sources","Own documents, policies, specifications, research or external evidence.",[1903,1904],"Memory store","Preserves selected information across turns or sessions.",[1906,1907],"Retrieval layer","Locates task-relevant candidates from external sources.",[1909,1910],"Tool\u002Fruntime layer","Reads state, performs actions and returns observations.",[1912,1913],"Context assembler","Selects, filters, deduplicates, orders and formats model-visible information.",[781,1915],"Reasons and generates over the assembled context.",[1917,1918],"Validation\u002Fevaluation","Checks whether selected context and resulting output satisfy task-specific requirements.",{},{"id":788,"data":1921,"type":218,"tunes":1923},{"text":1922},"The context assembler is conceptually important even when no module has that exact name. In a small application it may be ordinary application code. In a large agent platform it may combine session management, retrieval, memory, tool middleware, compaction and policy enforcement.",{},{"id":793,"data":1925,"type":42,"tunes":1927},{"text":1926,"level":247},"A practical context construction policy",{},{"id":798,"data":1929,"type":391,"tunes":1970},{"content":1930,"stretched":43,"withHeadings":14},[1931,1934,1937,1940,1943,1946,1949,1952,1955,1958,1961,1964,1967],[1932,1933],"Rule","Why it matters",[1935,1936],"Start from the current task","Do not carry information merely because it existed earlier.",[1938,1939],"Re-read volatile state","Memory and old context can be stale.",[1941,1942],"Retrieve just enough evidence","Large candidate sets can dilute decisive information.",[1944,1945],"Preserve source metadata","Version, date and authority determine whether evidence still applies.",[1947,1948],"Remove duplicate content","Redundancy consumes tokens without adding information.",[1950,1951],"Prefer structured summaries for large tool output","Expose decisive fields instead of raw noise where fidelity permits.",[1953,1954],"Keep rules with exceptions","Separating a rule from its exception creates false certainty.",[1956,1957],"Make precedence explicit","Do not ask the model to infer which conflicting source wins.",[1959,1960],"Keep durable state outside context","Context is temporary working memory, not the database.",[1962,1963],"Compact with retention tests","Verify that identifiers, constraints, provenance and unresolved state survive.",[1965,1966],"Measure order sensitivity","Correctness should not depend accidentally on arbitrary document ordering.",[1968,1969],"Evaluate context separately from model quality","A stronger model cannot compensate reliably for missing or unauthorized evidence.",{},{"id":842,"data":1972,"type":42,"tunes":1974},{"text":1973,"level":247},"How to evaluate context engineering",{},{"id":847,"data":1976,"type":391,"tunes":2018},{"content":1977,"stretched":43,"withHeadings":14},[1978,1982,1986,1990,1994,1998,2002,2006,2010,2014],[1979,1980,1981],"Property","Question","Example test",[1983,1984,1985],"Sufficiency","Does the context contain everything required to solve the task?","Remove one evidence item and observe whether the answer becomes unsupported.",[1987,1988,1989],"Relevance","How much context is unnecessary for the task?","Measure quality as irrelevant passages are added or removed.",[1991,1992,1993],"Authority","Are decisive claims grounded in the correct source class?","Inject a more fluent but non-authoritative conflicting source.",[1995,1996,1997],"Freshness","Does current state override stale copies?","Change authoritative state after a previous turn and rerun.",[1999,2000,2001],"Position robustness","Does answer quality depend strongly on evidence position?","Randomize candidate ordering across repeated trials.",[2003,2004,2005],"Conflict handling","Does the model follow explicit precedence rules?","Present old and new state together.",[2007,2008,2009],"Compaction retention","Does summarization preserve constraints and validity boundaries?","Compare pre\u002Fpost-compaction task performance.",[2011,2012,2013],"Token efficiency","Does extra context improve quality enough to justify latency\u002Fcost?","Run controlled context-size ablations.",[2015,2016,2017],"Security","Can unauthorized or adversarial content enter model context?","Test tenant, permission and prompt-injection boundaries.",{},{"id":892,"data":2020,"type":42,"tunes":2022},{"text":2021,"level":247},"Context assembly is a distinct RAG failure layer",{},{"id":897,"data":2024,"type":218,"tunes":2026},{"text":2025},"A RAG pipeline can succeed at retrieval and still fail downstream. The relevant source may appear at rank 2, yet the context assembler can drop it, truncate it, combine it with stale contradictory material or exceed the token budget.",{},{"id":902,"data":2028,"type":218,"tunes":2030},{"text":2029},"This is why retrieval traces should be compared with the actual context sent to the model. Without that comparison, context failures are easily misdiagnosed as embedding or model failures.",{},{"id":907,"data":2032,"type":492,"tunes":2037},{"url":2033,"title":2034,"excerpt":2035,"ctaLabel":2036},"https:\u002F\u002Fstajic.de\u002Fblog\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method","RAG Failed — But Which Layer Actually Failed? A Diagnostic Method","A layer-by-layer approach to separating source coverage, retrieval, ranking, context assembly, generation, evidence attribution and freshness failures.","Read the RAG diagnostic method",{},{"id":915,"data":2039,"type":42,"tunes":2041},{"text":2040,"level":247},"Original implementation evidence",{},{"id":920,"data":2043,"type":42,"tunes":2045},{"text":2044,"level":246},"Source of Truth Research Engine: bounded research instead of unlimited context",{},{"id":925,"data":2047,"type":218,"tunes":2049},{"text":2048},"The Source of Truth Research Engine separates discovery, acquisition, extraction, verification, contradiction analysis and synthesis into bounded research stages instead of sending one huge research task and all accumulated material into a single model call.",{},{"id":930,"data":2051,"type":218,"tunes":2053},{"text":2052},"Its evidence model stores Sources, Artifacts, Claims, Relations, Contradictions and provenance outside the model context. The model can receive the subset needed for the current research step while durable evidence remains in the external store.",{},{"id":935,"data":2055,"type":218,"tunes":2057},{"text":2056},"That is a concrete context-engineering pattern: durable research state lives outside the model window; the active model context is reconstructed for the current stage.",{},{"id":940,"data":2059,"type":42,"tunes":2061},{"text":2060,"level":246},"Aaasaasa AI Client: runtime, permissions and context are separate concerns",{},{"id":945,"data":2063,"type":218,"tunes":2065},{"text":2064},"Aaasaasa AI Client separates provider\u002Fmodel selection, runtime location, workspace permissions, local resources and tool access. This prevents the model context from becoming the owner of authorization or application state.",{},{"id":950,"data":2067,"type":218,"tunes":2069},{"text":2068},"Direct Chat and agentic runtimes can have different tool capabilities. Workspace permission profiles are enforced by the runtime rather than merely described in natural-language context. This distinction is important: context can tell a model what it should do, while the runtime must still enforce what it is actually allowed to do.",{},{"id":955,"data":2071,"type":218,"tunes":2073},{"text":2072},"The implementation evidence here is architectural separation, not a claim that every advanced context-management technique described in this article is already implemented.",{},{"id":960,"data":2075,"type":391,"tunes":2095},{"content":2076,"stretched":43,"withHeadings":14},[2077,2080,2083,2086,2089,2092],[2078,2079],"Implementation pattern","Context-engineering lesson",[2081,2082],"External evidence store","Durable knowledge does not need to remain in the model window.",[2084,2085],"Bounded research stages","Different steps can receive different context instead of accumulating one giant history.",[2087,2088],"Claims + provenance outside context","Evidence identity survives beyond temporary inference state.",[2090,2091],"Runtime-enforced permissions","Security authority does not depend on the model remembering an instruction.",[2093,2094],"Separate local\u002Fprovider\u002Fmodel\u002Fruntime concepts","Context is only one layer of the wider AI application architecture.",{},{"id":983,"data":2097,"type":226,"tunes":2100},{"body":2098,"title":2099,"variant":240},"These implementations support the architectural separation between durable state, retrieval, runtime controls and model-facing context. They are not presented as benchmark proof that one context strategy is universally optimal.","Evidence boundary",{},{"id":989,"data":2102,"type":42,"tunes":2104},{"text":2103,"level":247},"Common context-engineering failure modes",{},{"id":994,"data":2106,"type":391,"tunes":2141},{"content":2107,"stretched":43,"withHeadings":14},[2108,2111,2114,2117,2120,2123,2126,2129,2132,2135,2138],[2109,2110],"Failure mode","What goes wrong",[2112,2113],"Replay the entire conversation forever","Old assumptions, repetition and token growth overwhelm current intent.",[2115,2116],"Put every retrieved result into the prompt","Noise, duplication and conflicting versions dilute decisive evidence.",[2118,2119],"Use memory as current state","Stale information silently replaces authoritative live state.",[2121,2122],"Return raw tool output","Large logs or responses consume attention without adding decision value.",[2124,2125],"Hide tool descriptions behind vague names","The model cannot reliably decide which capability to use.",[2127,2128],"Compact without retention tests","Critical constraints, identifiers or exceptions disappear.",[2130,2131],"Mix instructions and untrusted data","External content can be interpreted as higher-authority instruction.",[2133,2134],"Use one static context template for every task","Different tasks receive irrelevant information and miss task-specific evidence.",[2136,2137],"Ignore source version\u002Fdate","Stale but relevant evidence can dominate current authoritative state.",[2139,2140],"Treat a larger context window as a quality guarantee","Capacity increases while attention and conflict problems remain.",{},{"id":1032,"data":2143,"type":42,"tunes":2145},{"text":2144,"level":247},"Common misconceptions",{},{"id":1037,"data":2147,"type":391,"tunes":2182},{"content":2148,"stretched":43,"withHeadings":14},[2149,2152,2155,2158,2161,2164,2167,2170,2173,2176,2179],[2150,2151],"Misconception","Correction",[2153,2154],"“Context engineering is just prompt engineering with a new name.”","Prompts are one component; context engineering also covers retrieval, memory, state, tool results, history and compaction.",[2156,2157],"“Context means chat history.”","History is only one possible context source.",[2159,2160],"“More context is always better.”","Additional information can reduce signal, introduce conflicts and increase cost.",[2162,2163],"“If retrieval found it, the model saw it.”","Retrieved candidates can be filtered, truncated or omitted before inference.",[2165,2166],"“Long context removes the need for RAG.”","Large windows increase capacity but do not solve freshness, authority, permissions or dynamic retrieval.",[2168,2169],"“Memory should always be loaded.”","Memory should be selected according to the current task.",[2171,2172],"“A summary preserves everything important.”","Compaction is lossy unless explicitly evaluated for retention.",[2174,2175],"“Instructions can enforce permissions.”","Authorization must be enforced by runtime\u002Fapplication controls, not only by context.",[2177,2178],"“One context recipe works for every model.”","Context sensitivity varies by model, task, corpus and runtime.",[2180,2181],"“Context engineering is only for agents.”","Agents amplify the need, but ordinary RAG and conversational applications also require context construction.",{},{"id":1075,"data":2184,"type":42,"tunes":2186},{"text":2185,"level":247},"A practical context-engineering sequence",{},{"id":1080,"data":2188,"type":317,"tunes":2221},{"steps":2189,"title":2220,"orientation":316},[2190,2193,2196,2199,2202,2205,2208,2211,2214,2217],{"label":2191,"description":2192},"1. Define the next model decision","Specify what the model must answer, classify, plan or choose at this step.",{"label":2194,"description":2195},"2. Identify required facts and constraints","List the minimum state, rules, evidence and instructions that can materially change the result.",{"label":2197,"description":2198},"3. Resolve authority and permissions","Determine which sources are current, authoritative and accessible to the current principal.",{"label":2200,"description":2201},"4. Retrieve or read on demand","Acquire the necessary evidence and volatile state rather than relying on stale context.",{"label":2203,"description":2204},"5. Reduce noise","Deduplicate, summarize or select passages without discarding decisive exceptions or provenance.",{"label":2206,"description":2207},"6. Structure and order","Make instructions, current state, evidence and tool observations distinguishable.",{"label":2209,"description":2210},"7. Fit the token budget","Prefer high-signal context and move durable information outside the window.",{"label":2212,"description":2213},"8. Run the model","Execute inference over the assembled context.",{"label":2215,"description":2216},"9. Observe failures","Capture whether the problem came from missing, stale, noisy, conflicting or poorly ordered context.",{"label":2218,"description":2219},"10. Re-evaluate after model\u002Fruntime changes","A context strategy is only valid for the models, tools and workloads on which it was tested.","Construct context from the current decision backward",{},{"id":1116,"data":2223,"type":42,"tunes":2225},{"text":2224,"level":247},"Context-engineering checklist",{},{"id":1121,"data":2227,"type":391,"tunes":2267},{"content":2228,"stretched":43,"withHeadings":14},[2229,2231,2234,2237,2240,2243,2246,2249,2252,2255,2258,2261,2264],[1980,2230],"Expected answer",[2232,2233],"What exact decision will the model make next?","A bounded task, not a vague long-term objective.",[2235,2236],"Which information can materially change that decision?","Explicit minimum evidence\u002Fstate set.",[2238,2239],"Which data is authoritative now?","Current source\u002Fversion and freshness rule.",[2241,2242],"Which data is optional background?","Separated from decisive evidence.",[2244,2245],"What must not enter context?","Unauthorized, unnecessary or overly sensitive data.",[2247,2248],"Which memory items are relevant?","Selected by task, not replayed automatically.",[2250,2251],"Which tool outputs should be reduced?","Large responses are transformed into decision-relevant form.",[2253,2254],"Which constraints must survive compaction?","Identifiers, exceptions, obligations, unresolved state and provenance.",[2256,2257],"How is precedence represented?","Current\u002Fauthoritative information can reliably override stale or weaker sources.",[2259,2260],"How will you know context failed?","Context-specific evals and traces exist.",[2262,2263],"Can the answer be reproduced?","Model input or reconstructable context trace is available where appropriate.",[2265,2266],"Can a stronger or larger model change the strategy?","Context policy is version-aware and reevaluated empirically.",{},{"id":1164,"data":2269,"type":42,"tunes":2271},{"text":2270,"level":247},"Edge cases and limitations",{},{"id":1169,"data":2273,"type":218,"tunes":2275},{"text":2274},"Some tasks are simple enough that context engineering reduces to a short system prompt and one user message. Adding retrieval, memory and compaction would only introduce unnecessary architecture.",{},{"id":1174,"data":2277,"type":218,"tunes":2279},{"text":2278},"Some tasks require high recall and may intentionally include more context before later synthesis. Research, discovery and legal review can prefer omission avoidance over minimal token count.",{},{"id":1179,"data":2281,"type":218,"tunes":2283},{"text":2282},"Some information should never be summarized before use. Exact contracts, code, cryptographic material, numerical records and regulatory text may require verbatim or structured retrieval where compression could alter meaning.",{},{"id":1184,"data":2285,"type":218,"tunes":2287},{"text":2286},"Long-context behavior varies substantially between models. A strategy validated on one model, context length or tool harness should not automatically be transferred to another.",{},{"id":1189,"data":2289,"type":218,"tunes":2291},{"text":2290},"The model can still ignore or misinterpret excellent context. Context engineering improves the information environment; it does not guarantee reasoning correctness.",{},{"id":1194,"data":2293,"type":42,"tunes":2295},{"text":2294,"level":247},"What would change this answer?",{},{"id":1199,"data":2297,"type":218,"tunes":2299},{"text":2298},"Future models may become more robust to long context, positional effects and conflicting information. That could reduce the amount of manual curation required.",{},{"id":1204,"data":2301,"type":218,"tunes":2303},{"text":2302},"The architectural distinction would still remain useful because permissions, freshness, memory persistence, source authority and external application state exist outside the model regardless of context-window size.",{},{"id":1209,"data":2305,"type":218,"tunes":2307},{"text":2306},"The recommended balance between preloaded and just-in-time context also changes with latency requirements, tool reliability, corpus size, model cost and how dynamic the underlying information is.",{},{"id":1214,"data":2309,"type":42,"tunes":2311},{"text":2310,"level":247},"Related canonical knowledge",{},{"id":1219,"data":2313,"type":218,"tunes":2315},{"text":2314},"Context engineering sits between retrieval and generation. RAG explains how external knowledge is retrieved; R01 separates embeddings, vector search and reranking; context engineering explains what eventually reaches the model.",{},{"id":1224,"data":2317,"type":492,"tunes":2322},{"url":2318,"title":2319,"excerpt":2320,"ctaLabel":2321},"https:\u002F\u002Fstajic.de\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works","What Is RAG? The Simplest Explanation of How It Works","The retrieval foundation for understanding how external knowledge can be supplied to a model before generation.","Read the RAG foundation",{},{"id":1232,"data":2324,"type":218,"tunes":2326},{"text":2325},"Source-of-Truth architecture answers a different question: not which information is present in context, but which source is authorized to establish a claim.",{},{"id":1237,"data":2328,"type":218,"tunes":2330},{"text":2329},"The existing article Why More Context Can Make AI Answers Worse is the diagnostic companion to this canonical definition. It focuses on context pollution, position effects, top-k growth, compaction loss and answer degradation rather than redefining context engineering itself.",{},{"id":1242,"data":2332,"type":42,"tunes":2334},{"text":2333,"level":247},"Frequently asked questions",{},{"id":1247,"data":2336,"type":1247,"tunes":2363},{"items":2337,"title":2362},[2338,2341,2344,2347,2350,2353,2356,2359],{"id":1251,"answer":2339,"question":2340},"Context engineering is the design and runtime management of what information a language model receives at inference time, including instructions, history, retrieved evidence, memory, state, tools and tool results.","What is context engineering?",{"id":1255,"answer":2342,"question":2343},"Prompt engineering focuses on how instructions and examples are written. Context engineering includes prompts but also decides which external information, state, history, memory and tool observations are placed around them.","How is context engineering different from prompt engineering?",{"id":1259,"answer":2345,"question":2346},"No. RAG retrieves external information. Context engineering decides how retrieved information is filtered, combined with other state and actually delivered to the model.","Is RAG the same as context engineering?",{"id":1263,"answer":2348,"question":2349},"No. Memory persists information outside the current model call. Context is the subset of information loaded into the current inference.","Is memory the same as context?",{"id":1267,"answer":2351,"question":2352},"Additional context can introduce noise, stale state, conflicting evidence, duplication and positional competition. Large context capacity does not guarantee equally reliable use of every token.","Why can more context make an answer worse?",{"id":1271,"answer":2354,"question":2355},"Compaction summarizes or transforms accumulated history into a smaller representation so a long-running system can continue without replaying every prior token.","What is context compaction?",{"id":1275,"answer":2357,"question":2358},"It can be represented in context for reasoning, but consequential operations should often re-read the authoritative source because context snapshots can become stale.","Should current application state be stored in context?",{"id":1279,"answer":2360,"question":2361},"No. Agents make context management more dynamic, but RAG systems, assistants, copilots and multi-turn applications also need deliberate context construction.","Is context engineering only needed for AI agents?","Context engineering FAQ",{},{"id":1285,"data":2365,"type":42,"tunes":2367},{"text":2366,"level":247},"Glossary",{},{"id":1290,"data":2369,"type":1290,"tunes":2404},{"title":2370,"entries":2371},"Key context-engineering terms",[2372,2374,2377,2379,2382,2385,2388,2391,2394,2396,2399,2401],{"term":1627,"anchor":1296,"definition":2373},"The design and runtime management of the information supplied to a language model for a particular inference step.",{"term":2375,"anchor":1300,"definition":2376},"Context window","The model's finite token capacity for the input and, depending on the model interface, associated generated tokens or active sequence.",{"term":1625,"anchor":1304,"definition":2378},"The design of instructions, examples and prompt structure intended to elicit useful model behavior.",{"term":2380,"anchor":1308,"definition":2381},"Context assembly","The process of selecting, filtering, ordering and formatting model-visible information before inference.",{"term":2383,"anchor":1312,"definition":2384},"Just-in-time retrieval","Loading information dynamically when the current task requires it instead of preloading all potentially relevant data.",{"term":2386,"anchor":1316,"definition":2387},"Compaction","Reducing accumulated context into a smaller representation while attempting to preserve information needed for future steps.",{"term":2389,"anchor":1320,"definition":2390},"Context pollution","Degradation caused by irrelevant, stale, contradictory or redundant information occupying the model's working context.",{"term":2392,"anchor":1324,"definition":2393},"Application state","The current authoritative condition of the external system, workflow or domain that exists independently of the model context.",{"term":1584,"anchor":1327,"definition":2395},"Information stored outside the immediate model invocation for possible use in later turns or sessions.",{"term":2397,"anchor":1331,"definition":2398},"Retrieved context","External information selected by a retrieval system and made available, wholly or partly, to the model.",{"term":1999,"anchor":1334,"definition":2400},"The degree to which model correctness remains stable when the location or order of relevant context changes.",{"term":2402,"anchor":1338,"definition":2403},"Validity boundary","The scope, time, assumptions, versions and evidence conditions within which a conclusion remains supported.",{},{"id":1342,"data":2406,"type":42,"tunes":2408},{"text":2407,"level":247},"Conclusion",{},{"id":1347,"data":2410,"type":218,"tunes":2412},{"text":2411},"Context engineering is the layer that decides what the model gets to see before it answers. That makes it broader than prompting and downstream of retrieval, while remaining distinct from durable memory and authoritative application state.",{},{"id":1352,"data":2414,"type":218,"tunes":2416},{"text":2415},"A strong context architecture does not treat the context window as a database. It keeps durable state and knowledge outside the model, loads what is required for the current decision, preserves authority and provenance, removes unnecessary noise and refreshes volatile information when needed.",{},{"id":1357,"data":2418,"type":218,"tunes":2420},{"text":2419},"The practical objective is therefore not maximum context. It is minimum sufficient, high-signal, correctly authorized and validity-preserving context for the next model decision.",{},{"id":1362,"data":2422,"type":42,"tunes":2424},{"text":2423,"level":247},"Primary sources and current guidance",{},{"id":1367,"data":2426,"type":218,"tunes":2428},{"text":2427},"The sources below support the current context-engineering terminology, long-context behavior and operational context-management patterns. Project sections are explicitly implementation evidence rather than universal claims.",{},{"id":1372,"data":2430,"type":1379,"tunes":2435},{"link":1374,"meta":2431},{"image":2432,"title":2433,"description":2434},{"url":406},"Anthropic — Effective context engineering for AI agents","Official engineering guidance defining context engineering, just-in-time retrieval, compaction, structured memory and context curation for agents.",{},{"id":1382,"data":2437,"type":1379,"tunes":2442},{"link":1384,"meta":2438},{"image":2439,"title":2440,"description":2441},{"url":406},"OpenAI — Context Engineering: Short-Term Memory Management with Sessions","Official cookbook guidance on context management, trimming and compression for long-running agent sessions.",{},{"id":1391,"data":2444,"type":1379,"tunes":2449},{"link":1393,"meta":2445},{"image":2446,"title":2447,"description":2448},{"url":406},"OpenAI — Agents guide","Current OpenAI developer guidance on agent runtimes, context across steps and orchestration ownership.",{},{"id":1400,"data":2451,"type":1379,"tunes":2456},{"link":1402,"meta":2452},{"image":2453,"title":2454,"description":2455},{"url":406},"Lost in the Middle: How Language Models Use Long Contexts","Research showing that long-context model performance can depend strongly on the position of relevant information in the input.",{},"2.31.6","Context engineering designs what information an AI model receives before inference, including prompts, retrieval, memory, application state, tool results and conversation history.",{"lang":7,"title":208,"content":210,"contentJson":2460,"excerpt":1409},{"time":212,"blocks":2461,"version":1408},[2462,2465,2468,2471,2474,2477,2480,2483,2486,2489,2492,2495,2498,2501,2512,2515,2518,2521,2524,2540,2543,2560,2563,2566,2569,2572,2575,2578,2581,2584,2587,2590,2593,2596,2599,2602,2605,2608,2611,2614,2617,2620,2635,2638,2641,2644,2647,2650,2653,2656,2659,2662,2665,2668,2671,2674,2677,2680,2683,2694,2697,2700,2703,2706,2709,2712,2715,2718,2721,2724,2727,2730,2733,2736,2739,2752,2755,2758,2775,2778,2792,2795,2798,2801,2804,2807,2810,2813,2816,2819,2822,2825,2828,2831,2841,2844,2847,2862,2865,2880,2883,2897,2900,2917,2920,2923,2926,2929,2932,2935,2938,2941,2944,2947,2950,2953,2956,2959,2962,2965,2977,2980,2996,2999,3002,3005,3008,3011,3014,3019,3024,3029],{"id":215,"data":2463,"type":218,"tunes":2464},{"text":217},{},{"id":221,"data":2466,"type":226,"tunes":2467},{"body":223,"title":224,"variant":225},{},{"id":229,"data":2469,"type":226,"tunes":2470},{"body":231,"title":232,"variant":233},{},{"id":236,"data":2472,"type":226,"tunes":2473},{"body":238,"title":239,"variant":240},{},{"id":243,"data":2475,"type":248,"tunes":2476},{"title":245,"maxLevel":246,"minLevel":247},{},{"id":251,"data":2478,"type":42,"tunes":2479},{"text":253,"level":247},{},{"id":256,"data":2481,"type":218,"tunes":2482},{"text":258},{},{"id":261,"data":2484,"type":218,"tunes":2485},{"text":263},{},{"id":266,"data":2487,"type":218,"tunes":2488},{"text":268},{},{"id":271,"data":2490,"type":42,"tunes":2491},{"text":273,"level":247},{},{"id":276,"data":2493,"type":218,"tunes":2494},{"text":278},{},{"id":281,"data":2496,"type":218,"tunes":2497},{"text":283},{},{"id":286,"data":2499,"type":218,"tunes":2500},{"text":288},{},{"id":291,"data":2502,"type":317,"tunes":2511},{"steps":2503,"title":315,"orientation":316},[2504,2505,2506,2507,2508,2509,2510],{"label":295,"description":296},{"label":298,"description":299},{"label":301,"description":302},{"label":304,"description":305},{"label":307,"description":308},{"label":310,"description":311},{"label":313,"description":314},{},{"id":320,"data":2513,"type":42,"tunes":2514},{"text":322,"level":247},{},{"id":325,"data":2516,"type":218,"tunes":2517},{"text":327},{},{"id":330,"data":2519,"type":218,"tunes":2520},{"text":332},{},{"id":335,"data":2522,"type":42,"tunes":2523},{"text":337,"level":247},{},{"id":340,"data":2525,"type":391,"tunes":2539},{"content":2526,"stretched":43,"withHeadings":14},[2527,2528,2529,2530,2531,2532,2533,2534,2535,2536,2537,2538],[344,345,346],[348,349,350],[352,353,354],[356,357,358],[360,361,362],[364,365,366],[368,369,370],[372,373,374],[376,377,378],[380,381,382],[384,385,386],[388,389,390],{},{"id":394,"data":2541,"type":42,"tunes":2542},{"text":396,"level":247},{},{"id":399,"data":2544,"type":431,"tunes":2559},{"rows":2545,"title":423,"layout":391,"columns":2556},[2546,2548,2550,2552,2554],{"id":403,"label":404,"values":2547},[406,406],{"id":408,"label":409,"values":2549},[406,406],{"id":412,"label":413,"values":2551},[406,406],{"id":416,"label":417,"values":2553},[406,406],{"id":420,"label":421,"values":2555},[406,406],[2557,2558],{"id":426,"label":427},{"id":429,"label":430},{},{"id":434,"data":2561,"type":218,"tunes":2562},{"text":436},{},{"id":439,"data":2564,"type":42,"tunes":2565},{"text":441,"level":247},{},{"id":444,"data":2567,"type":218,"tunes":2568},{"text":446},{},{"id":449,"data":2570,"type":218,"tunes":2571},{"text":451},{},{"id":454,"data":2573,"type":218,"tunes":2574},{"text":456},{},{"id":459,"data":2576,"type":226,"tunes":2577},{"body":461,"title":462,"variant":463},{},{"id":466,"data":2579,"type":42,"tunes":2580},{"text":468,"level":247},{},{"id":471,"data":2582,"type":218,"tunes":2583},{"text":473},{},{"id":476,"data":2585,"type":218,"tunes":2586},{"text":478},{},{"id":481,"data":2588,"type":218,"tunes":2589},{"text":483},{},{"id":486,"data":2591,"type":492,"tunes":2592},{"url":488,"title":489,"excerpt":490,"ctaLabel":491},{},{"id":495,"data":2594,"type":42,"tunes":2595},{"text":497,"level":247},{},{"id":500,"data":2597,"type":218,"tunes":2598},{"text":502},{},{"id":505,"data":2600,"type":218,"tunes":2601},{"text":507},{},{"id":510,"data":2603,"type":226,"tunes":2604},{"body":512,"title":513,"variant":233},{},{"id":516,"data":2606,"type":42,"tunes":2607},{"text":518,"level":247},{},{"id":521,"data":2609,"type":218,"tunes":2610},{"text":523},{},{"id":526,"data":2612,"type":218,"tunes":2613},{"text":528},{},{"id":531,"data":2615,"type":218,"tunes":2616},{"text":533},{},{"id":536,"data":2618,"type":42,"tunes":2619},{"text":538,"level":247},{},{"id":541,"data":2621,"type":431,"tunes":2634},{"rows":2622,"title":560,"layout":391,"columns":2631},[2623,2625,2627,2629],{"id":545,"label":546,"values":2624},[406,406],{"id":549,"label":550,"values":2626},[406,406],{"id":553,"label":554,"values":2628},[406,406],{"id":557,"label":558,"values":2630},[406,406],[2632,2633],{"id":563,"label":564},{"id":566,"label":567},{},{"id":570,"data":2636,"type":218,"tunes":2637},{"text":572},{},{"id":575,"data":2639,"type":42,"tunes":2640},{"text":577,"level":247},{},{"id":580,"data":2642,"type":218,"tunes":2643},{"text":582},{},{"id":585,"data":2645,"type":218,"tunes":2646},{"text":587},{},{"id":590,"data":2648,"type":218,"tunes":2649},{"text":592},{},{"id":595,"data":2651,"type":42,"tunes":2652},{"text":597,"level":247},{},{"id":600,"data":2654,"type":218,"tunes":2655},{"text":602},{},{"id":605,"data":2657,"type":218,"tunes":2658},{"text":607},{},{"id":610,"data":2660,"type":218,"tunes":2661},{"text":612},{},{"id":615,"data":2663,"type":42,"tunes":2664},{"text":617,"level":247},{},{"id":620,"data":2666,"type":218,"tunes":2667},{"text":622},{},{"id":625,"data":2669,"type":218,"tunes":2670},{"text":627},{},{"id":630,"data":2672,"type":218,"tunes":2673},{"text":632},{},{"id":635,"data":2675,"type":42,"tunes":2676},{"text":637,"level":247},{},{"id":640,"data":2678,"type":218,"tunes":2679},{"text":642},{},{"id":645,"data":2681,"type":218,"tunes":2682},{"text":647},{},{"id":650,"data":2684,"type":391,"tunes":2693},{"content":2685,"stretched":43,"withHeadings":14},[2686,2687,2688,2689,2690,2691,2692],[654,655],[657,658],[660,661],[663,664],[666,667],[669,670],[672,673],{},{"id":676,"data":2695,"type":42,"tunes":2696},{"text":678,"level":247},{},{"id":681,"data":2698,"type":218,"tunes":2699},{"text":683},{},{"id":686,"data":2701,"type":218,"tunes":2702},{"text":688},{},{"id":691,"data":2704,"type":218,"tunes":2705},{"text":693},{},{"id":696,"data":2707,"type":226,"tunes":2708},{"body":698,"title":699,"variant":233},{},{"id":702,"data":2710,"type":42,"tunes":2711},{"text":704,"level":247},{},{"id":707,"data":2713,"type":218,"tunes":2714},{"text":709},{},{"id":712,"data":2716,"type":218,"tunes":2717},{"text":714},{},{"id":717,"data":2719,"type":492,"tunes":2720},{"url":719,"title":720,"excerpt":721,"ctaLabel":722},{},{"id":725,"data":2722,"type":42,"tunes":2723},{"text":727,"level":247},{},{"id":730,"data":2725,"type":218,"tunes":2726},{"text":732},{},{"id":735,"data":2728,"type":218,"tunes":2729},{"text":737},{},{"id":740,"data":2731,"type":218,"tunes":2732},{"text":742},{},{"id":745,"data":2734,"type":42,"tunes":2735},{"text":747,"level":247},{},{"id":750,"data":2737,"type":226,"tunes":2738},{"body":752,"title":753,"variant":240},{},{"id":756,"data":2740,"type":391,"tunes":2751},{"content":2741,"stretched":43,"withHeadings":14},[2742,2743,2744,2745,2746,2747,2748,2749,2750],[760,761],[763,764],[766,767],[769,770],[772,773],[775,776],[778,779],[781,782],[784,785],{},{"id":788,"data":2753,"type":218,"tunes":2754},{"text":790},{},{"id":793,"data":2756,"type":42,"tunes":2757},{"text":795,"level":247},{},{"id":798,"data":2759,"type":391,"tunes":2774},{"content":2760,"stretched":43,"withHeadings":14},[2761,2762,2763,2764,2765,2766,2767,2768,2769,2770,2771,2772,2773],[802,803],[805,806],[808,809],[811,812],[814,815],[817,818],[820,821],[823,824],[826,827],[829,830],[832,833],[835,836],[838,839],{},{"id":842,"data":2776,"type":42,"tunes":2777},{"text":844,"level":247},{},{"id":847,"data":2779,"type":391,"tunes":2791},{"content":2780,"stretched":43,"withHeadings":14},[2781,2782,2783,2784,2785,2786,2787,2788,2789,2790],[851,852,853],[855,856,857],[859,860,861],[863,864,865],[867,868,869],[871,872,873],[875,876,877],[879,880,881],[883,884,885],[887,888,889],{},{"id":892,"data":2793,"type":42,"tunes":2794},{"text":894,"level":247},{},{"id":897,"data":2796,"type":218,"tunes":2797},{"text":899},{},{"id":902,"data":2799,"type":218,"tunes":2800},{"text":904},{},{"id":907,"data":2802,"type":492,"tunes":2803},{"url":909,"title":910,"excerpt":911,"ctaLabel":912},{},{"id":915,"data":2805,"type":42,"tunes":2806},{"text":917,"level":247},{},{"id":920,"data":2808,"type":42,"tunes":2809},{"text":922,"level":246},{},{"id":925,"data":2811,"type":218,"tunes":2812},{"text":927},{},{"id":930,"data":2814,"type":218,"tunes":2815},{"text":932},{},{"id":935,"data":2817,"type":218,"tunes":2818},{"text":937},{},{"id":940,"data":2820,"type":42,"tunes":2821},{"text":942,"level":246},{},{"id":945,"data":2823,"type":218,"tunes":2824},{"text":947},{},{"id":950,"data":2826,"type":218,"tunes":2827},{"text":952},{},{"id":955,"data":2829,"type":218,"tunes":2830},{"text":957},{},{"id":960,"data":2832,"type":391,"tunes":2840},{"content":2833,"stretched":43,"withHeadings":14},[2834,2835,2836,2837,2838,2839],[964,965],[967,968],[970,971],[973,974],[976,977],[979,980],{},{"id":983,"data":2842,"type":226,"tunes":2843},{"body":985,"title":986,"variant":240},{},{"id":989,"data":2845,"type":42,"tunes":2846},{"text":991,"level":247},{},{"id":994,"data":2848,"type":391,"tunes":2861},{"content":2849,"stretched":43,"withHeadings":14},[2850,2851,2852,2853,2854,2855,2856,2857,2858,2859,2860],[998,999],[1001,1002],[1004,1005],[1007,1008],[1010,1011],[1013,1014],[1016,1017],[1019,1020],[1022,1023],[1025,1026],[1028,1029],{},{"id":1032,"data":2863,"type":42,"tunes":2864},{"text":1034,"level":247},{},{"id":1037,"data":2866,"type":391,"tunes":2879},{"content":2867,"stretched":43,"withHeadings":14},[2868,2869,2870,2871,2872,2873,2874,2875,2876,2877,2878],[1041,1042],[1044,1045],[1047,1048],[1050,1051],[1053,1054],[1056,1057],[1059,1060],[1062,1063],[1065,1066],[1068,1069],[1071,1072],{},{"id":1075,"data":2881,"type":42,"tunes":2882},{"text":1077,"level":247},{},{"id":1080,"data":2884,"type":317,"tunes":2896},{"steps":2885,"title":1113,"orientation":316},[2886,2887,2888,2889,2890,2891,2892,2893,2894,2895],{"label":1084,"description":1085},{"label":1087,"description":1088},{"label":1090,"description":1091},{"label":1093,"description":1094},{"label":1096,"description":1097},{"label":1099,"description":1100},{"label":1102,"description":1103},{"label":1105,"description":1106},{"label":1108,"description":1109},{"label":1111,"description":1112},{},{"id":1116,"data":2898,"type":42,"tunes":2899},{"text":1118,"level":247},{},{"id":1121,"data":2901,"type":391,"tunes":2916},{"content":2902,"stretched":43,"withHeadings":14},[2903,2904,2905,2906,2907,2908,2909,2910,2911,2912,2913,2914,2915],[852,1125],[1127,1128],[1130,1131],[1133,1134],[1136,1137],[1139,1140],[1142,1143],[1145,1146],[1148,1149],[1151,1152],[1154,1155],[1157,1158],[1160,1161],{},{"id":1164,"data":2918,"type":42,"tunes":2919},{"text":1166,"level":247},{},{"id":1169,"data":2921,"type":218,"tunes":2922},{"text":1171},{},{"id":1174,"data":2924,"type":218,"tunes":2925},{"text":1176},{},{"id":1179,"data":2927,"type":218,"tunes":2928},{"text":1181},{},{"id":1184,"data":2930,"type":218,"tunes":2931},{"text":1186},{},{"id":1189,"data":2933,"type":218,"tunes":2934},{"text":1191},{},{"id":1194,"data":2936,"type":42,"tunes":2937},{"text":1196,"level":247},{},{"id":1199,"data":2939,"type":218,"tunes":2940},{"text":1201},{},{"id":1204,"data":2942,"type":218,"tunes":2943},{"text":1206},{},{"id":1209,"data":2945,"type":218,"tunes":2946},{"text":1211},{},{"id":1214,"data":2948,"type":42,"tunes":2949},{"text":1216,"level":247},{},{"id":1219,"data":2951,"type":218,"tunes":2952},{"text":1221},{},{"id":1224,"data":2954,"type":492,"tunes":2955},{"url":1226,"title":1227,"excerpt":1228,"ctaLabel":1229},{},{"id":1232,"data":2957,"type":218,"tunes":2958},{"text":1234},{},{"id":1237,"data":2960,"type":218,"tunes":2961},{"text":1239},{},{"id":1242,"data":2963,"type":42,"tunes":2964},{"text":1244,"level":247},{},{"id":1247,"data":2966,"type":1247,"tunes":2976},{"items":2967,"title":1282},[2968,2969,2970,2971,2972,2973,2974,2975],{"id":1251,"answer":1252,"question":1253},{"id":1255,"answer":1256,"question":1257},{"id":1259,"answer":1260,"question":1261},{"id":1263,"answer":1264,"question":1265},{"id":1267,"answer":1268,"question":1269},{"id":1271,"answer":1272,"question":1273},{"id":1275,"answer":1276,"question":1277},{"id":1279,"answer":1280,"question":1281},{},{"id":1285,"data":2978,"type":42,"tunes":2979},{"text":1287,"level":247},{},{"id":1290,"data":2981,"type":1290,"tunes":2995},{"title":1292,"entries":2982},[2983,2984,2985,2986,2987,2988,2989,2990,2991,2992,2993,2994],{"term":1295,"anchor":1296,"definition":1297},{"term":1299,"anchor":1300,"definition":1301},{"term":1303,"anchor":1304,"definition":1305},{"term":1307,"anchor":1308,"definition":1309},{"term":1311,"anchor":1312,"definition":1313},{"term":1315,"anchor":1316,"definition":1317},{"term":1319,"anchor":1320,"definition":1321},{"term":1323,"anchor":1324,"definition":1325},{"term":376,"anchor":1327,"definition":1328},{"term":1330,"anchor":1331,"definition":1332},{"term":871,"anchor":1334,"definition":1335},{"term":1337,"anchor":1338,"definition":1339},{},{"id":1342,"data":2997,"type":42,"tunes":2998},{"text":1344,"level":247},{},{"id":1347,"data":3000,"type":218,"tunes":3001},{"text":1349},{},{"id":1352,"data":3003,"type":218,"tunes":3004},{"text":1354},{},{"id":1357,"data":3006,"type":218,"tunes":3007},{"text":1359},{},{"id":1362,"data":3009,"type":42,"tunes":3010},{"text":1364,"level":247},{},{"id":1367,"data":3012,"type":218,"tunes":3013},{"text":1369},{},{"id":1372,"data":3015,"type":1379,"tunes":3018},{"link":1374,"meta":3016},{"image":3017,"title":1377,"description":1378},{"url":406},{},{"id":1382,"data":3020,"type":1379,"tunes":3023},{"link":1384,"meta":3021},{"image":3022,"title":1387,"description":1388},{"url":406},{},{"id":1391,"data":3025,"type":1379,"tunes":3028},{"link":1393,"meta":3026},{"image":3027,"title":1396,"description":1397},{"url":406},{},{"id":1400,"data":3030,"type":1379,"tunes":3033},{"link":1402,"meta":3031},{"image":3032,"title":1405,"description":1406},{"url":406},{},"Post erfolgreich abgerufen",{"items":3036,"source":3118,"manualIds":3119,"manualMatchedIds":3120},[3037,3044,3051,3058,3063,3070,3077,3084,3091,3098,3104,3111],{"id":3038,"slug":3039,"title":3040,"excerpt":3041,"featuredImage":3042,"publishedAt":3043},"363","front-und-backend-entwicklung","Frontend i Backend Razvoj","Front-end i back-end razvoj je suštinski deo veb razvoja i obuhvata kreiranje veb aplikacija i veb-sajtova. Front-end razvoj se fokusira na korisnički interfejs, dok je back-end razvoj odgovoran za programiranje i upravljanje serverskom stranom.","\u002Fuploads\u002F2026\u002F03\u002Ffront-und-backend-entwicklung-1774872219531-wyu4i1.webp","2023-04-12T11:11:00.000Z",{"id":3045,"slug":3046,"title":3047,"excerpt":3048,"featuredImage":3049,"publishedAt":3050},"466","the-gpu-is-not-the-product-future-proof-private-ai-architecture","GPU nije proizvod: Privatna AI arhitektura spremna za budućnost","Privatna AI infrastruktura ne bi trebalo da bude projektovana oko jednog GPU-a ili jednog modela. Otporniji pristup kombinuje brze GPU-ove za inferenciju, memorijski bogate AI sisteme, čvorove za fizički AI i opcione vodeće modele u oblaku iza sloja za rutiranje koji prepoznaje mogućnosti.","\u002Fuploads\u002F2026\u002F09\u002Fthe-gpu-is-not-the-product-future-proof-private-ai-architecture-1790140878812-8hsl39.webp","2026-09-23T01:19:00.000Z",{"id":3052,"slug":3053,"title":3054,"excerpt":3055,"featuredImage":3056,"publishedAt":3057},"364","tipps-fuer-die-verbesserung-der-seo-suchmaschinenoptimierung","Ovladavanje SEO radnim tokom: Ključne strategije optimizacije za organski rast","Strukturiran SEO tok posla je ključan za održiv organski rast. Naučite deset osnovnih strategija, od istraživanja ključnih reči i tehničke optimizacije do kvaliteta sadržaja i analize performansi.","\u002Fuploads\u002F2026\u002F03\u002Ftipps-fuer-die-verbesserung-der-seo-suchmaschinenoptimierung-1774866098131-hwkzrg.webp","2024-01-26T06:35:00.000Z",{"id":3059,"slug":3060,"title":3060,"excerpt":10,"featuredImage":3061,"publishedAt":3062},"369","git-with-automatic-upload-and-synchronization-to-a-production-server","\u002Fuploads\u002F2024\u002F05\u002Fstep-by-step-guide-illustration-showing-the-process-of-setting-up-Git-with-auto-upload-and-synchronization-to-a-production-server-large.webp","2024-05-28T22:48:00.000Z",{"id":3064,"slug":3065,"title":3066,"excerpt":3067,"featuredImage":3068,"publishedAt":3069},"487","vector-databases-embeddings-and-reranking-three-different-parts-of-retrieval","Vektorske baze podataka, ugrađivanja i ponovno rangiranje: Tri različita dela pretraživanja","Embedinzi predstavljaju značenje, vektorske baze podataka pronalaze kandidate, a rerangirači prečišćavaju rezultate. Saznajte kako se ova tri sloja pronalaženja razlikuju i kako rade zajedno u RAG-u.","\u002Fuploads\u002F2026\u002F10\u002Fvector-databases-embeddings-and-reranking-three-different-parts-of-retrieval-1791480129884-9dtasz.webp","2026-10-08T11:21:00.000Z",{"id":3071,"slug":3072,"title":3073,"excerpt":3074,"featuredImage":3075,"publishedAt":3076},"493","mlops-vs-llmops-what-changes-when-the-model-is-an-llm","MLOps vs LLMOps: Šta se menja kada je model LLM","MLOps upravlja sistemima mašinskog učenja; LLMOps proširuje te prakse na promptove, kontekst, pretragu, provajdere, alate, evaluacije i ponašanje u vreme izvršavanja oko velikih jezičkih modela.","\u002Fuploads\u002F2026\u002F10\u002Fmlops-vs-llmops-what-changes-when-the-model-is-an-llm-1791487319869-2v7hxo.webp","2026-10-08T15:20:00.000Z",{"id":3078,"slug":3079,"title":3080,"excerpt":3081,"featuredImage":3082,"publishedAt":3083},"384","new-qwen-3-5-plus","Novi Qwen 3.5-Plus: AI otvorenog koda je upravo postao ozbiljan.","Otkrijte revolucionarne funkcije i prednosti Alibabinog Qwen 3.5-Plus modela, AI otvorenog koda koji menja pravila igre za programere.","\u002Fuploads\u002F2026\u002F02\u002Fnew-qwen-3-5-plus-1771515512741-dcbi9p.webp","2026-02-19T10:23:00.000Z",{"id":3085,"slug":3086,"title":3087,"excerpt":3088,"featuredImage":3089,"publishedAt":3090},"467","the-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers","Granica valjanosti odgovora: Nedostajući sloj između relevantnosti i pouzdanih AI odgovora","Izvor može biti relevantan, autoritativan i ipak pogrešan za pitanje koje se postavlja. Sloj koji nedostaje je primenljivost: uslovi pod kojima odgovor važi i promene koje ga primoravaju na preispitivanje. Ovaj članak predstavlja Granicu važenja odgovora kao obrazac za dizajn izvora za ljude, AI pretragu i RAG sisteme.","\u002Fuploads\u002F2026\u002F09\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers-1790272901306-1g5jly.webp","2026-09-24T11:59:00.000Z",{"id":3092,"slug":3093,"title":3094,"excerpt":3095,"featuredImage":3096,"publishedAt":3097},"472","why-more-context-can-make-ai-answers-worse","Zašto više konteksta može pogoršati AI odgovore","Veći kontekstni prozor ne garantuje bolji odgovor. Ovaj članak objašnjava kako razblaživanje signala, protivrečni dokazi, zastarelo stanje, osetljivost na poziciju i kompresija sa gubicima mogu smanjiti pouzdanost veštačke inteligencije—i uvodi praktičan test pritiska konteksta.","\u002Fuploads\u002F2026\u002F09\u002Fwhy-more-context-can-make-ai-answers-worse-1790351615793-2ntv2v.webp","2026-09-25T11:51:00.000Z",{"id":3099,"slug":3100,"title":3101,"excerpt":3102,"featuredImage":3082,"publishedAt":3103},"445","qwen-3-6-in-production-release-runbook-ai-rollback-and-llmops-versioning","Qwen 3.6 u produkciji: Runbook za izdavanje, AI rollback i LLMOps verziranje","Qwen 3.6 nije samo još jedna nadogradnja modela. To je istovremeno događaj objavljivanja, scenario povratka na prethodnu verziju i problem verziranja. Ovaj članak objašnjava kako Qwen 3.6 treba tretirati u produkciji kroz LLMOps disciplinu, sledljivost promptova i modela, kontrolisano uvođenje i spremnost za povratak na prethodnu verziju zasnovanu na dokazima.","2026-05-04T02:49:00.000Z",{"id":3105,"slug":3106,"title":3107,"excerpt":3108,"featuredImage":3109,"publishedAt":3110},"468","ai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context","Memorija AI agenta nije RAG: Kako razdvojiti memoriju, pronalaženje, stanje i kontekst","Memorija agenta, RAG, stanje i kontekst često se koriste kao da su međusobno zamenjivi. Oni to nisu. Ovaj praktični arhitektonski model razdvaja ova četiri sloja, pokazuje gde svaki pripada i objašnjava šta se kvari kada ih sistemi stope u jedno.","\u002Fuploads\u002F2026\u002F09\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context-1790350560308-np0xy6.webp","2026-09-25T11:34:00.000Z",{"id":3112,"slug":3113,"title":3114,"excerpt":3115,"featuredImage":3116,"publishedAt":3117},"490","rbac-vs-tenant-isolation-two-different-security-boundaries","RBAC naspram izolacije zakupaca: dve različite bezbednosne granice","RBAC kontroliše šta korisnik sme da radi; izolacija zakupaca kontroliše kojim resursima tog zakupca ta radnja može da pristupi. Saznajte zašto bezbednost višekorisničkog SaaS-a zahteva obe granice.","\u002Fuploads\u002F2026\u002F10\u002Frbac-vs-tenant-isolation-two-different-security-boundaries-1791485111528-qqtzby.webp","2026-10-08T14:43:00.000Z","fallback",[],[]]