[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"portal-settings:stajic:sr":3,"public-menus:all":38,"post:why-more-context-can-make-ai-answers-worse:sr":205,"related:post:why-more-context-can-make-ai-answers-worse:sr:1":1595},{"statusCode":4,"data":5,"message":37},200,{"tenantId":6,"lang":7,"defaultLang":8,"siteUrl":9,"contactEmail":10,"brandName":11,"logoUrl":12,"siteName":11,"siteDescription":13,"ogImage":10,"robotsIndex":14,"socialLinks":10,"reservedSlugs":10,"seoPolicy":15},"stajic","sr","de","https:\u002F\u002Fstajic.de",null,"Stajic Platform","\u002FLogo_Planet.svg","Stajic Portal",true,{"branding":16,"relatedContent":17,"crossDomainLinks":18},{"logoUrl":12},{"enabled":14},[19,22,25,28,31,34],{"url":20,"label":21,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Ffigure.rocks","figure.rocks",{"url":23,"label":24,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Floving.rocks","loving.rocks",{"url":26,"label":27,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.com","bazify.com",{"url":29,"label":30,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.de","bazify.de",{"url":32,"label":33,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.at","bazify.at",{"url":35,"label":36,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.ba","bazify.ba","Portal settings resolved",[39,45],{"id":40,"name":41,"location":42,"isActive":14,"isDefault":43,"items":44},1,"main-navigation","header",false,[],{"id":46,"name":47,"location":48,"isActive":14,"isDefault":14,"items":49},4,"main-menu","sidebar",[50,66,79,93,103,118,133],{"id":51,"title":52,"url":60,"target":61,"icon":62,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":64,"portfolioId":10,"children":65},"item-18",{"de":53,"en":54,"es":55,"fr":56,"it":54,"ru":57,"sr":58,"zh":59},"Startseite","Home","Inicio","Accueil","Главная","Почетна","首页","\u002Ffull-stack-web-developer-munich-performance-seo-and-maintainable-builds","_self","i-lucide-home","page",111,[],{"id":67,"title":68,"url":75,"target":61,"icon":76,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":77,"portfolioId":10,"children":78},"item-22",{"de":69,"en":69,"es":70,"fr":69,"it":71,"ru":72,"sr":73,"zh":74},"Vision","Visión","Visione","Видение","Визија","想象","\u002Fueber-uns-webdesign-muenchen-webaplikation","i-lucide-eye",113,[],{"id":80,"title":81,"url":89,"target":61,"icon":90,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":91,"portfolioId":10,"children":92},"item-19",{"de":82,"en":83,"es":84,"fr":83,"it":85,"ru":86,"sr":87,"zh":88},"Leistungen","Services","Servicios","Servizi","Услуги","Услуге","服务","\u002Fservices-dienstleistungen-muenchen","i-lucide-wrench",116,[],{"id":94,"title":95,"url":99,"target":61,"icon":100,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":101,"portfolioId":10,"children":102},"item-23",{"de":96,"en":96,"es":96,"fr":96,"it":96,"ru":97,"sr":97,"zh":98},"Blog","Блог","博客","\u002Fblog","i-lucide-book-open",112,[],{"id":104,"title":105,"url":114,"target":61,"icon":115,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":116,"portfolioId":10,"children":117},"item-32",{"de":106,"en":107,"es":108,"fr":109,"it":110,"ru":111,"sr":112,"zh":113},"Neue Technologien","New Technologies","Nuevas tecnologías","Nouvelles technologies","Nuove tecnologie","Новые технологии","Нове технологије","新技术！","\u002Fneue-webtechnologien","i-lucide-sparkles",122,[],{"id":119,"title":120,"url":129,"target":61,"icon":130,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":131,"portfolioId":10,"children":132},"item-20",{"de":121,"en":122,"es":123,"fr":124,"it":125,"ru":126,"sr":127,"zh":128},"Kontakt","Contact us!","Contacto","Contact","Contatto","Контакт","Контактирајте нас","联系我们！","\u002Fcontact","i-lucide-mail",115,[],{"id":134,"title":135,"url":144,"target":61,"icon":145,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":147},"item-21",{"de":136,"en":137,"es":138,"fr":139,"it":140,"ru":141,"sr":142,"zh":143},"Unsere Arbeit","Our Work","Nuestro trabajo","Nos réalisations","I nostri lavori","Наши работы","Наши радови","文件夹","\u002Fportfolio","i-lucide-briefcase",114,[148,161,175,181,193],{"id":149,"title":150,"url":144,"target":61,"icon":159,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":160},"item-24",{"de":151,"en":152,"es":153,"fr":154,"it":155,"ru":156,"sr":157,"zh":158},"Alle Projekte","All Projects","Todos los proyectos","Tous les projets","Tutti i progetti","Все проекты","Сви пројекти","所有项目","i-lucide-grid-3x3",[],{"id":162,"title":163,"url":171,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":174},"item-29",{"de":164,"en":165,"es":166,"fr":167,"it":168,"ru":169,"sr":170,"zh":143},"Local Roots, Global Reach","Local Roots - Global Reach","Empresa local ","Entreprise locale","Azienda locale","Местная компания","Локално предузеће глобално тржиште","\u002Fportfolio\u002Flocal-roots-global-reach-communication-media-systems-for-modern-business","i-lucide-folder","custom",[],{"id":176,"title":177,"url":179,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":180},"item-28",{"de":178,"en":178,"es":178,"fr":178,"it":178,"ru":178,"sr":178,"zh":178},"Solr Suggester","\u002Fportfolio\u002Fsolr-fuzzy-suggester-und-solr-infix-suggester-abfrage-ueber-ajax-und-filterung",[],{"id":182,"title":183,"url":191,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":192},"item-27",{"de":184,"en":185,"es":186,"fr":187,"it":188,"ru":189,"sr":190,"zh":185},"Firmenwebseite SEO","Company Website SEO","Sitio web corporativo SEO","Site web d’entreprise SEO","Sito web aziendale SEO","Корпоративный сайт SEO","Пословна веб-страница SEO","\u002Fportfolio\u002Fseo-sem-branding-mobile-webseite-muenchen",[],{"id":194,"title":195,"url":203,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":204},"item-31",{"de":196,"en":197,"es":198,"fr":199,"it":200,"ru":201,"sr":202,"zh":197},"Digitalisierungsportal","Digitalization Portal","Portal de digitalización","Portail de numérisation","Portale di digitalizzazione","Портал цифровизации","Портал за дигитализацију","\u002Fportfolio\u002Fdigitalisierungsportal-archiv-museum-bibliothek-ead-lido-mets-mods",[],{"statusCode":4,"data":206,"message":1594},{"id":207,"title":208,"slug":209,"content":210,"contentJson":211,"excerpt":798,"featuredImage":799,"featuredImageAlt":800,"featuredImageCaption":10,"featuredImageTitle":10,"featuredImageCopyright":10,"featuredImageAuthor":10,"featuredImageSourceUrl":10,"featuredImageLicense":10,"featuredImageIsAiGenerated":43,"status":801,"publishedAt":802,"createdAt":803,"updatedAt":804,"seoLocalePaths":805,"categories":814,"author":827,"translations":832},"472","Zašto više konteksta može pogoršati AI odgovore","why-more-context-can-make-ai-answers-worse","\u003Cnav class=\"editorjs-toc\" data-editorjs-toc=\"true\" aria-label=\"Sadržaj\">\u003Cstrong class=\"editorjs-toc__title\">Sadržaj\u003C\u002Fstrong>\u003Col class=\"editorjs-toc__list editorjs-toc__list--depth-0\">\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-5\" class=\"editorjs-toc__link\">Kapacitet konteksta nije isto što i upotrebljivost konteksta\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-9\" class=\"editorjs-toc__link\">Pet načina na koje dodatni kontekst može umanjiti kvalitet odgovora\u003C\u002Fa>\u003Col class=\"editorjs-toc__list editorjs-toc__list--depth-1\">\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-11\" class=\"editorjs-toc__link\">1. Razblaživanje signala: relevantni dokazi se takmiče sa svim ostalim\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-15\" class=\"editorjs-toc__link\">2. Konflikt dokaza: više izvora može značiti više verzija realnosti\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-19\" class=\"editorjs-toc__link\">3. Osetljivost na poziciju: mesto gde se dokaz pojavljuje može promeniti rezultat\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-22\" class=\"editorjs-toc__link\">4. Postojanost zastarelog konteksta: model vidi istinu i istorijat zajedno\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-25\" class=\"editorjs-toc__link\">5. Gubitak pri kompresiji: manji kontekst takođe može postati lošiji kontekst\u003C\u002Fa>\u003C\u002Fli>\u003C\u002Fol>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-28\" class=\"editorjs-toc__link\">Model kvaliteta konteksta\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-32\" class=\"editorjs-toc__link\">Test pritiska na kontekst\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-35\" class=\"editorjs-toc__link\">Šta meriti umesto broja tokena\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-37\" class=\"editorjs-toc__link\">RAG: zašto povećanje parametra top-k može da škodi\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-41\" class=\"editorjs-toc__link\">Dugotrajni agenti: kontinuitet nije akumulacija\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-45\" class=\"editorjs-toc__link\">Redosled u kontekstu treba da bude nameran\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-48\" class=\"editorjs-toc__link\">Očuvajte granice odlučivanja tokom sabijanja\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-52\" class=\"editorjs-toc__link\">Praktična politika konstrukcije konteksta\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-54\" class=\"editorjs-toc__link\">Šta bi promenilo ovaj odgovor?\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-58\" class=\"editorjs-toc__link\">Ograničenja\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-61\" class=\"editorjs-toc__link\">Zaključak\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-64\" class=\"editorjs-toc__link\">Često postavljana pitanja\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-66\" class=\"editorjs-toc__link\">Rečnik pojmova\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-68\" class=\"editorjs-toc__link\">Primarni izvori i dodatna literatura\u003C\u002Fa>\u003C\u002Fli>\u003C\u002Fol>\u003C\u002Fnav>\n\u003Cp>Veći kontekstni prozor pruža AI sistemu veći kapacitet. To ne garantuje da će model taj kapacitet dobro iskoristiti. U dugim razgovorima, RAG tokovima, istraživačkim agentima i radnim procesima sa intenzivnim korišćenjem alata, dodavanje više istorije, više dokumenata, više izlaznih podataka alata ili više memorije može učiniti odgovor manje pouzdanim, umesto bolje informisanim.\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--info my-6 rounded-xl border p-5 border-blue-300 bg-blue-50 dark:border-blue-900 dark:bg-blue-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">Direktan odgovor\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">&lt;strong&gt;Više konteksta može pogoršati odgovor veštačke inteligencije kada dodatne informacije smanjuju odnos signala i šuma, unose konflikte, skrivaju ključne dokaze, zadržavaju zastarelo stanje ili kompresijom uklanjaju važne uslove.&lt;\u002Fstrong&gt; Relevantan inženjerski cilj stoga nije maksimalni kontekst. To je &lt;strong&gt;minimalni dovoljni kontekst sa očuvanim dokazima i granicama odlučivanja&lt;\u002Fstrong&gt;.\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Caside class=\"editorjs-callout editorjs-callout--note my-6 rounded-xl border p-5 border-gray-300 bg-gray-50 dark:border-gray-700 dark:bg-gray-900\u002F40\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">O modelu korišćenom u ovom članku\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">Model kvaliteta konteksta (Context Quality) i test pritiska na kontekst (Context Pressure Test) u nastavku predstavljaju praktične arhitektonske metode predložene u ovom članku, a ne formalne industrijske standarde. Oni sintetišu utvrđena saznanja o efektima pozicije u dugom kontekstu, zagađenju konteksta, sažimanju, preuzimanju informacija i inženjeringu konteksta.\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-5\">Kapacitet konteksta nije isto što i upotrebljivost konteksta\u003C\u002Fh2>\n\u003Cp>Deklarisani kontekstni prozor modela opisuje koliko ulaznih podataka on može da prihvati. To ne podrazumeva da svaki token unutar tog prozora dobija jednaku pažnju ili podjednako doprinosi konačnom odgovoru. Ova razlika je važna jer produkcioni sistemi sve više popunjavaju kontekst istorijom razgovora, preuzetim dokumentima, rezultatima alata, memorijom, struktuiranim stanjem, instrukcijama i međuproduktima.\u003C\u002Fp>\n\u003Cp>Klasična studija „Lost in the Middle“ pokazala je da modeli sa dugim kontekstom mogu imati lošije rezultate kada se relevantni dokazi nalaze u sredini dugog unosa nego kada se nalaze blizu početka ili kraja. Šira inženjerska pouka nije da je dugi kontekst loš. Ona glasi da dostupnost unutar konteksta nije isto što i pouzdana upotreba.\u003C\u002Fp>\n\u003Cp>Smernice kompanije OpenAI za upravljanje kontekstom dolaze do istog operativnog zaključka iz drugog pravca: čak i veoma veliki kontekstni prozori mogu biti preopterećeni neprobranom istorijom, suvišnim izlaznim podacima alata i bučnim preuzimanjem podataka. Anthropic na sličan način tretira kontekst kao ograničen resurs koji zahteva aktivni inženjering, a ne pasivno nagomilavanje.\u003C\u002Fp>\n\u003Ch2 id=\"section-9\">Pet načina na koje dodatni kontekst može umanjiti kvalitet odgovora\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Oblik greške\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Šta se menja kada se doda više konteksta\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Tipičan simptom\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Razblaživanje signala\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Relevantni dokazi postaju manji udeo ukupnog unosa\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Model daje generički odgovor ili propušta ključni odlomak\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Konflikt dokaza\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Različiti dokumenti, verzije ili memorije se ne slažu\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Odgovor spaja nekompatibilne tvrdnje ili bira pogrešnu verziju\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Osetljivost na poziciju\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Ključne informacije prelaze u deo konteksta koji se manje pouzdano koristi\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Isti dokazi funkcionišu u jednom redosledu, ali ne uspevaju u drugom\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Zadržavanje zastarelog konteksta\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Staro stanje ili prethodni zaključci ostaju prisutni nakon što se realnost promeni\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Model nastavlja da ponavlja ranije tačan odgovor\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Gubitak usled kompresije\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Sažimanje ili rezimiranje uklanja ograde, izuzetke, poreklo ili nerazrešenu neizvesnost\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Rezime je koherentan, ali rezultujući odgovor postaje preterano samouveren ili preopšten\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch3 id=\"section-11\">1. Razblaživanje signala: relevantni dokazi se takmiče sa svim ostalim\u003C\u002Fh3>\n\u003Cp>Pretpostavimo da se na pitanje može odgovoriti na osnovu dva kratka odlomka. RAG sistem preuzima ta dva odlomka plus osamnaest labavo povezanih „za svaki slučaj“. Odziv preuzimanja se može poboljšati, ali generator sada mora da razlikuje ključne dokaze od pozadinskog materijala. Ako se slične fraze pojavljuju u više dokumenata, dodatni kontekst može učiniti odgovor manje preciznim.\u003C\u002Fp>\n\u003Cp>Ovo stvara važnu razliku između odziva preuzimanja i korisnosti konteksta. Više preuzetog materijala može povećati verovatnoću da odgovor postoji negde u kontekstu, dok istovremeno smanjuje verovatnoću da model pravim dokazima dodeli dovoljnu težinu.\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--tip my-6 rounded-xl border p-5 border-violet-300 bg-violet-50 dark:border-violet-900 dark:bg-violet-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">Inženjersko pravilo\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">Nemojte optimizovati top-k izolovano. Merite da li dodavanje dokumenata poboljšava konačnu tvrdnju, čuva atribuciju dokaza i ostaje stabilno tokom ponovljenih testiranja.\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch3 id=\"section-15\">2. Konflikt dokaza: više izvora može značiti više verzija realnosti\u003C\u002Fh3>\n\u003Cp>Dugi konteksti često sadrže međusobno neusklađene informacije: staru i novu API dokumentaciju, dve verzije pravilnika, prethodne i trenutne korisničke preferencije, konkurentne veb izvore, keširano stanje ili rezime koji je generisao model, a koji više ne odgovara izvoru.\u003C\u002Fp>\n\u003Cp>Ovaj propust nije nužno halucinacija. Model možda verno kombinuje kontradiktorne dokaze. Arhitekturi su stoga potrebna pravila prioriteta: autoritet izvora, verzija, vremenska oznaka, jurisdikcija, korisnik sistema (tenant), revizija proizvoda, stanje korisnika ili eksplicitni metapodaci o zameni starih podataka novim.\u003C\u002Fp>\n\u003Cp>Bez tih pravila, povećanje konteksta može povećati kontradiktornosti brže nego što povećava znanje.\u003C\u002Fp>\n\u003Ch3 id=\"section-19\">3. Osetljivost na poziciju: mesto gde se dokaz pojavljuje može promeniti rezultat\u003C\u002Fh3>\n\u003Cp>Rezultati istraživanja „Izgubljeni u sredini” (Lost in the Middle) pokazali su da samo promena pozicije relevantnih informacija može materijalno promeniti performanse modela. To saznanje je posebno važno za sisteme koji spajaju mnoge pronađene odlomke ili dugačke istorijate u fiksnom redosledu.\u003C\u002Fp>\n\u003Cp>Produkcioni test bi stoga trebalo da varira redosled dokumenata, a ne samo da testira jedan kanonski upit. Ako sistem tačno odgovara samo kada je odlučujući dokaz na početku ili na kraju, aplikacija je krhkija nego što to sugeriše jedan rezultat na benčmarku.\u003C\u002Fp>\n\u003Ch3 id=\"section-22\">4. Postojanost zastarelog konteksta: model vidi istinu i istorijat zajedno\u003C\u002Fh3>\n\u003Cp>Dugotrajni agenti često prenose ranije zaključke unapred. Taj kontinuitet je koristan sve dok se neka činjenica ne promeni. Ako jučerašnji rezultat alata kaže da je implementacija u dobrom stanju, a trenutni rezultat alata kaže da je degradirana, oba mogu ostati u kontekstu osim ako sistem eksplicitno ne zameni ili ograniči opseg starog stanja.\u003C\u002Fp>\n\u003Cp>Zato bi trenutno operativno stanje obično trebalo da potiče iz autoritativnog izvora, dok memorija čuva trajni kontekst kao što su odluke, preferencije ili procedure. Veća istorija razgovora nije zamena za ponovno sagledavanje sadašnjosti.\u003C\u002Fp>\n\u003Ch3 id=\"section-25\">5. Gubitak pri kompresiji: manji kontekst takođe može postati lošiji kontekst\u003C\u002Fh3>\n\u003Cp>Suprotna intervencija — kompresovanje konteksta — takođe ima svoje načine otkazivanja. Sažeci mogu izostaviti izuzetke, nerešena pitanja, poreklo, precizne identifikatore, negativne dokaze ili uslove pod kojima je zaključak bio validan.\u003C\u002Fp>\n\u003Cp>Rad Microsoft Research-a pod nazivom Agentic Context Engineering opisuje srodan problem kao pristrasnost ka sažetosti i kolaps konteksta: iterativno prepisivanje može ukloniti korisne detalje domena. Cilj stoga nije „kompresovati što je više moguće”. Cilj je smanjiti kontekst uz očuvanje informacija koje menjaju odluke.\u003C\u002Fp>\n\u003Ch2 id=\"section-28\">Model kvaliteta konteksta\u003C\u002Fh2>\n\u003Cp>Koristan kontekst se može proceniti kroz šest dimenzija. Nijedna od njih nije prosto broj tokena.\u003C\u002Fp>\n\u003Csection class=\"editorjs-comparison my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">Šest dimenzija kvaliteta konteksta\u003C\u002Fh3>\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left dark:border-gray-700 dark:bg-gray-900\">\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">Dimenzija\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">Pitanje\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">Ako je slabo\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Relevantnost\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Autoritativnost\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Svežina\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Doslednost\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Kompletnost za donošenje odluka\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Sledljivost\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Caside class=\"editorjs-callout editorjs-callout--success my-6 rounded-xl border p-5 border-emerald-300 bg-emerald-50 dark:border-emerald-900 dark:bg-emerald-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">Ciljno stanje\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">Najbolji kontekst nije najveći kontekst. To je &lt;strong&gt;najmanji kontekst koji i dalje čuva dokaze, ograničenja, stanje, izuzetke i poreklo neophodne za pouzdan odgovor ili akciju&lt;\u002Fstrong&gt;.\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-32\">Test pritiska na kontekst\u003C\u002Fh2>\n\u003Cp>Da biste utvrdili da li aplikacija ima koristi od više konteksta, testirajte veličinu konteksta kao eksperimentalnu varijablu umesto da pretpostavljate da je veće uvek bolje.\u003C\u002Fp>\n\u003Csection class=\"editorjs-process my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">Test pritiska na kontekst\u003C\u002Fh3>\u003Cdiv class=\"grid grid-cols-1 md:grid-cols-2 xl:grid-cols-3 gap-4\">\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">1\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">1. Definišite referentni slučaj\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Izaberite zadatak sa poznatim odgovorom i poznatim minimalnim skupom dokaza.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">2\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">2. Pokrenite minimalno dovoljan kontekst\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Obezbedite samo uputstva, trenutno stanje i dokaze neophodne za odgovor.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">3\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">3. Dodajte relevantnu pozadinu\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Dodajte koristan, ali ne i presudan kontekst i izmerite da li se kvalitet poboljšava, ostaje stabilan ili opada.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">4\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">4. Dodajte realističan šum\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Dodajte labavo povezan istorijat, izlaz alata ili pronađene odlomke koje bi produkcioni sistem mogao da uključi.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">5\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">5. Dodajte kontrolisane konflikte\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Uvedite zastarele ili kontradiktorne dokaze sa jasnim metapodacima o verziji i proverite da li tačan izvor i dalje pobeđuje.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">6\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">6. Promenite redosled odlučujućih dokaza\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Postavite ključne informacije blizu početka, sredine i kraja da biste testirali osetljivost na poziciju.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">7\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">7. Testirajte sažimanje\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Zamenite stariji kontekst sažetkom i proverite da li kvalifikatori, poreklo, nerešena pitanja i granice odlučivanja opstaju.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">8\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">8. Uporedite krivu kvaliteta\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Izmerite tačnost, korišćenje dokaza, doslednost, kašnjenje, trošak i varijansu kako se kontekst menja.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-35\">Šta meriti umesto broja tokena\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Metrika\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Šta otkriva\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Tačnost odgovora\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Da li je konačni rezultat tačan\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Potkrepljenost tvrdnji dokazima\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Da li materijalne tvrdnje ostaju utemeljene kako se kontekst menja\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Iskorišćenost dokaza\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Da li odgovor prati odlučujući dokaz umesto prethodnog znanja modela\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Tačnost rešavanja konflikata\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Da li aktuelni \u002F autoritativni dokazi pobeđuju zastarele ili slabije izvore\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Robusnost na poziciju\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Da li promena redosleda dokaza menja tačnost\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Zadržavanje pri sažimanju\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Da li sažeci čuvaju ograničenja, izuzetke, identifikatore, poreklo i nerešena stanja\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Varijansa izlaza kroz ponavljanja\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Da li dodatni kontekst čini sistem manje stabilnim\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Latencija i trošak tokena\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Da li dodate informacije donose dovoljno kvaliteta da opravdaju operativne troškove\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-37\">RAG: zašto povećanje parametra top-k može da škodi\u003C\u002Fh2>\n\u003Cp>Uobičajeni obrazac za fino podešavanje RAG-a jeste povećanje parametra top-k kada sistem propusti odgovor. To može poboljšati odziv kandidata (recall), ali takođe može povećati količinu irelevantnog konteksta, dupliranih dokaza, zastarelih odlomaka i protivrečnih dokumenata.\u003C\u002Fp>\n\u003Cp>Bolje pitanje je da li ključni dokaz nedostaje u pretrazi ili samo gubi uticaj nakon sklapanja konteksta. Ako se tačan odlomak već nalazi u skupu kandidata, povećanje parametra top-k može predstavljati rešavanje pogrešnog problema.\u003C\u002Fp>\n\u003Caside class=\"editorjs-referral my-6\">\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fsr\u002Fblog\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method\" class=\"flex flex-col sm:flex-row gap-4 rounded-xl border border-gray-200 dark:border-gray-700 p-4 transition hover:border-primary-500\">\u003Cdiv class=\"min-w-0 flex-1\">\u003Cstrong class=\"block text-lg text-gray-900 dark:text-gray-100\">RAG nije uspeo — ali koji sloj je zapravo zakazao? Dijagnostički metod\u003C\u002Fstrong>\u003Cp class=\"mt-2 text-sm text-gray-600 dark:text-gray-300\">Metod sloj-po-sloj za izolovanje grešaka u pokrivenosti izvora, pretrazi, rangiranju, sklapanju konteksta, generisanju, atribuciji dokaza i ažurnosti.\u003C\u002Fp>\u003Cspan class=\"mt-3 inline-flex text-sm font-medium text-primary-600 dark:text-primary-400\">Pročitajte dijagnostički metod za RAG →\u003C\u002Fspan>\u003C\u002Fdiv>\u003C\u002Fa>\u003C\u002Faside>\n\u003Ch2 id=\"section-41\">Dugotrajni agenti: kontinuitet nije akumulacija\u003C\u002Fh2>\n\u003Cp>Agentu je potreban kontinuitet kroz korake, ali kontinuitet ne zahteva ponavljanje svakog prethodnog tokena. OpenAI demonstrira skraćivanje i kompresiju za kontekst dugotrajnih sesija. Anthropic preporučuje sabijanje (kompakciju), strukturisano vođenje beleški i druge tehnike kako bi se sačuvale korisne informacije uz kontrolu zagađenja konteksta.\u003C\u002Fp>\n\u003Cp>Snažna arhitektura za dugotrajne procese obično razdvaja trajnu memoriju, trenutno stanje, spoljne artefakte, pretragu i kontekst namenjen modelu. To omogućava sistemu da sačuva ono što je važno, a da pritom ne ubacuje svaki istorijski detalj u svako zaključivanje.\u003C\u002Fp>\n\u003Caside class=\"editorjs-referral my-6\">\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fsr\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context\" class=\"flex flex-col sm:flex-row gap-4 rounded-xl border border-gray-200 dark:border-gray-700 p-4 transition hover:border-primary-500\">\u003Cdiv class=\"min-w-0 flex-1\">\u003Cstrong class=\"block text-lg text-gray-900 dark:text-gray-100\">Memorija AI agenata nije RAG: Kako razdvojiti memoriju, pretragu, stanje i kontekst\u003C\u002Fstrong>\u003Cp class=\"mt-2 text-sm text-gray-600 dark:text-gray-300\">Praktična četvoroslojna arhitektura za razdvajanje onoga što opstaje, onoga što je trenutno merodavno, onoga što se pretražuje i onoga što model zaista dobija.\u003C\u002Fp>\u003Cspan class=\"mt-3 inline-flex text-sm font-medium text-primary-600 dark:text-primary-400\">Pročitajte članak o arhitekturi memorije →\u003C\u002Fspan>\u003C\u002Fdiv>\u003C\u002Fa>\u003C\u002Faside>\n\u003Ch2 id=\"section-45\">Redosled u kontekstu treba da bude nameran\u003C\u002Fh2>\n\u003Cp>Konstrukcija konteksta predstavlja problem informacione arhitekture. Ključna uputstva, trenutno stanje, odlučujući dokazi i ograničenja specifična za zadatak ne bi trebalo da budu nasumično raspoređeni. Kada sistemi mehanički spajaju izvore, oni implicitno prepuštaju određivanje prioriteta pozicionim efektima i pažnji modela.\u003C\u002Fp>\n\u003Cp>Ne postoji univerzalno najbolji redosled za svaki model i zadatak, pa bi redosled trebalo empirijski procenjivati. Koristan skup testova nasumično raspoređuje ili sistematski menja poziciju dokumenta i meri da li ista tvrdnja ostaje stabilna.\u003C\u002Fp>\n\u003Ch2 id=\"section-48\">Očuvajte granice odlučivanja tokom sabijanja\u003C\u002Fh2>\n\u003Cp>Sažetak koji glasi „koristite pristup X“ slabiji je od sažetka koji čuva razlog zašto je X izabran i šta bi poništilo tu odluku. Sabijanje konteksta treba da zadrži varijable koje mogu promeniti odgovor: verziju, datum, pretpostavke, stanje, autoritet, nerešena neslaganja i poreklo dokaza.\u003C\u002Fp>\n\u003Cp>Ovo direktno povezuje inženjering konteksta sa validnošću odgovora. Ako se sabijanjem sačuva zaključak, ali se ukloni granica njegove validnosti, budući odgovori mogu ostati interno dosledni, a da pritom postanu eksterno netačni.\u003C\u002Fp>\n\u003Caside class=\"editorjs-referral my-6\">\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fsr\u002Fblog\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers\" class=\"flex flex-col sm:flex-row gap-4 rounded-xl border border-gray-200 dark:border-gray-700 p-4 transition hover:border-primary-500\">\u003Cdiv class=\"min-w-0 flex-1\">\u003Cstrong class=\"block text-lg text-gray-900 dark:text-gray-100\">Granica validnosti odgovora: Karika koja nedostaje između relevantnosti i pouzdanih AI odgovora\u003C\u002Fstrong>\u003Cp class=\"mt-2 text-sm text-gray-600 dark:text-gray-300\">Okvir za eksplicitno definisanje uslova pod kojima se tvrdnja veštačke inteligencije primenjuje i koje promene zahtevaju ograničenje, ponovno izračunavanje ili odbacivanje.\u003C\u002Fp>\u003Cspan class=\"mt-3 inline-flex text-sm font-medium text-primary-600 dark:text-primary-400\">Pročitajte o granici validnosti odgovora →\u003C\u002Fspan>\u003C\u002Fdiv>\u003C\u002Fa>\u003C\u002Faside>\n\u003Ch2 id=\"section-52\">Praktična politika konstrukcije konteksta\u003C\u002Fh2>\n\u003Cul>\u003Cli>Počnite od trenutnog zadatka, a ne od svega što sistem zna.\u003C\u002Fli>\u003Cli>Ponovo pročitajte promenljivo stanje iz merodavnih sistema pre donošenja važnih odluka.\u003C\u002Fli>\u003Cli>Pretražujte dokaze za trenutno pitanje umesto da sa sobom prenosite velike statičke korpuse.\u003C\u002Fli>\u003Cli>Uklonite duplirane ili bezvredne izlaze iz alata.\u003C\u002Fli>\u003Cli>Zadržite verziju izvora, vremensku oznaku, autoritet i poreklo uz važne dokaze.\u003C\u002Fli>\u003Cli>Jasno odredite prioritet kada su trenutne i istorijske informacije u sukobu.\u003C\u002Fli>\u003Cli>Sačuvajte pravila zajedno sa njihovim izuzecima i preduslovima.\u003C\u002Fli>\u003Cli>Čuvajte trajne odluke i višekratne procedure van neposrednog konteksta kada nije potrebno njihovo doslovno ponavljanje.\u003C\u002Fli>\u003Cli>Sabijajte istoriju samo uz testove za zadržavanje ograničenja, identifikatora, izuzetaka i porekla.\u003C\u002Fli>\u003Cli>Procenjujte veličinu konteksta, redosled i šum kroz ponovljene probe, a ne na osnovu jednog upita.\u003C\u002Fli>\u003C\u002Ful>\n\u003Ch2 id=\"section-54\">Šta bi promenilo ovaj odgovor?\u003C\u002Fh2>\n\u003Cp>Kompromis se menja u zavisnosti od arhitekture modela, obuke, tipa zadatka i dužine konteksta. Budući modeli mogu postati znatno otporniji na poziciju, šum i kontradiktorne informacije. Zadatak sa malim, čistim korpusom takođe može imati koristi od jednostavnog pružanja kompletnog izvora umesto izgradnje složenog mehanizma za preuzimanje (retrieval pipeline).\u003C\u002Fp>\n\u003Cp>Preporuka se takođe menja kada je izostavljanje opasnije od šuma. U istraživačkim zadacima ili zadacima otkrivanja sa visokim odzivom (high-recall), veći kandidatski kontekst može biti opravdan pre kasnije faze filtriranja ili sinteze. U produkcionim sistemima osetljivim na latenciju, striktniji odabir konteksta može biti poželjniji.\u003C\u002Fp>\n\u003Cp>Osnovni princip bi se promenio samo ako bi modeli postali pouzdano invarijantni na irelevantne informacije, poziciju, protivrečnosti i zastarele dokaze. Do tada, kontekst treba tretirati kao pažljivo odabran resurs za izvršavanje, a ne kao pasivno skladište.\u003C\u002Fp>\n\u003Ch2 id=\"section-58\">Ograničenja\u003C\u002Fh2>\n\u003Cp>Ponašanje u dugom kontekstu znatno varira među modelima i radnim opterećenjima. Prvobitni eksperimenti „Lost in the Middle” koristili su ranije generacije modela, pa se ne sme pretpostaviti da njihove tačne veličine efekta predstavljaju današnje sisteme. Nalaz ostaje koristan kao obrazac otkaza koji treba testirati, a ne kao univerzalna fiksna kriva performansi.\u003C\u002Fp>\n\u003Cp>Isto tako, smanjenje konteksta može ukloniti neophodne dokaze. Zbijanje (kompaktovanje) uvodi rizik sažimanja, a agresivno filtriranje preuzimanja može smanjiti odziv. Cilj nije minimalan broj tokena po svaku cenu; cilj je dovoljan, aktuelan i sledljiv kontekst za odluku koja se donosi.\u003C\u002Fp>\n\u003Ch2 id=\"section-61\">Zaključak\u003C\u002Fh2>\n\u003Cp>Pitanje „Koliko konteksta model može da primi?” manje je korisno od pitanja „Koliko ovog konteksta poboljšava odluku?” Više tokena može dodati dokaze, ali takođe može uneti ometanja, protivrečnosti, zastarelo stanje, pozicionu ranjivost i dug kompresije.\u003C\u002Fp>\n\u003Cp>Tretirajte kontekst kao projektovani radni skup. Počnite sa minimalnim dovoljnim dokazima. Dodajte informacije samo kada poboljšavaju izmerene performanse. Eksplicitno testirajte šum, konflikt, redosled i sažimanje. Veliki kontekstualni prozor je kapacitet; kvalitet konteksta je arhitektura.\u003C\u002Fp>\n\u003Ch2 id=\"section-64\">Često postavljana pitanja\u003C\u002Fh2>\n\u003Csection class=\"editorjs-faq my-6 rounded-xl border border-gray-200 p-5 dark:border-gray-700\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">Dug kontekst i kvalitet odgovora veštačke inteligencije\u003C\u002Fh3>\u003Cdiv id=\"faq1\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">Može li pružanje više konteksta AI modelu pogoršati njegov odgovor?\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">Da. Dodatni kontekst može razvodniti relevantne dokaze, uneti kontradiktorne ili zastarele informacije, premestiti odlučujuće dokaze na manje robusne pozicije i povećati šansu da model koristi slabe umesto odlučujućih signala.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq2\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">Da li veći kontekstualni prozor eliminiše potrebu za RAG-om?\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">Uglavnom ne. Veći kontekstualni prozor povećava kapacitet, ali preuzimanje i dalje pomaže u odabiru aktuelnih i relevantnih informacija, kontroli troškova, očuvanju granica izvora i izbegavanju slanja velikih količina nepovezanih podataka u svaki zahtev.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq3\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">Šta je problem „Lost in the Middle”?\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">On opisuje uočene slučajeve u kojima jezički modeli manje pouzdano koriste relevantne informacije kada se one nalaze u sredini dugog konteksta nego kada se pojavljuju blizu početka ili kraja. Tačan efekat varira u zavisnosti od modela i zadatka i treba ga testirati na savremenim sistemima.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq4\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">Treba li uvek smanjivati top-k kod RAG-a?\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">Ne. Ako relevantni dokazi nedostaju u skupu kandidata, veći top-k može poboljšati odziv. Ako su dokazi već prisutni, ali se razvodnjavaju dodatnim materijalom, povećanje top-k može pogoršati kontekst. Dijagnostikujte preuzimanje i sklapanje konteksta odvojeno.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq5\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">Šta rezime konteksta treba da sačuva?\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">Treba da sačuva trajne odluke, trenutne ciljeve, nerešena pitanja, identifikatore, ograničenja, izuzetke, poreklo dokaza i uslove koji bi promenili raniji zaključak.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-66\">Rečnik pojmova\u003C\u002Fh2>\n\u003Csection class=\"editorjs-glossary my-6 rounded-xl border border-gray-200 dark:border-gray-700 p-5\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">Ključni pojmovi inženjeringa konteksta\u003C\u002Fh3>\u003Cdl>\u003Cdiv id=\"context-window\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">Kontekstualni prozor (Context window)\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Količina ulaznih i izlaznih informacija u tokenima na koju model može obratiti pažnju unutar jedne sekvence inferencije.\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"context-pollution\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">Zagađenje konteksta (Context pollution)\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Degradacija uzrokovana time što irelevantne, zastarele, suvišne, kontradiktorne ili na drugi način bezvredne informacije zauzimaju kontekst modela.\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"signal-dilution\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">Razvodnjavanje signala (Signal dilution)\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Smanjenje relativne istaknutosti odlučujućih dokaza usled dodavanja dodatnih informacija niske vrednosti ili konkurentskih informacija.\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"context-compaction\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">Kompaktovanje konteksta (Context compaction)\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Smanjivanje nagomilanog konteksta sažimanjem, restrukturiranjem, eksternalizacijom ili na drugi način očuvanjem suštinskih informacija u manjoj radnoj reprezentaciji.\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"position-robustness\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">Poziciona robusnost (Position robustness)\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Stepen u kojem performanse modela ostaju stabilne kada se relevantne informacije pojavljuju na različitim pozicijama unutar konteksta.\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"minimum-sufficient-context\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">Minimalni dovoljni kontekst (Minimum sufficient context)\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Najmanji praktični radni kontekst koji i dalje čuva dokaze, stanje, ograničenja, izuzetke i poreklo potrebne za pouzdano izvršavanje.\u003C\u002Fdd>\u003C\u002Fdiv>\u003C\u002Fdl>\u003C\u002Fsection>\n\u003Ch2 id=\"section-68\">Primarni izvori i dodatna literatura\u003C\u002Fh2>\n\u003Ca href=\"https:\u002F\u002Fdevelopers.openai.com\u002Fcookbook\u002Fexamples\u002Fagents_sdk\u002Fsession_memory\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">OpenAI — Context Engineering: Short-Term Memory Management with Sessions\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">Smernice o skraćivanju i kompresiji, sa diskusijom o ometanju, neefikasnosti, zastarelom kontekstu, bučnom preuzimanju i dugotrajnim sesijama.\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Feffective-context-engineering-for-ai-agents\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">Anthropic — Effective Context Engineering for AI Agents\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">Inženjerske smernice o zagađenju konteksta, sažimanju, strukturiranom vođenju beleški i upravljanju kontekstom agenata na dugim horizontima.\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Faclanthology.org\u002F2024.tacl-1.9\u002F\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">Liu et al. — Lost in the Middle: How Language Models Use Long Contexts\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">TACL rad koji prikazuje poziciono osetljivu upotrebu relevantnih informacija u dugim kontekstima i motiviše eksplicitne testove robusnosti na dugi kontekst.\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Fpublication\u002Fagentic-context-engineering-evolving-contexts-for-self-improving-language-models\u002F\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">Microsoft Research — Agentic Context Engineering (ACE)\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">Istraživanje o razvoju strukturiranih konteksta uz istovremeno rešavanje pristrasnosti ka sažetosti i kolapsa konteksta.\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fevaluation-best-practices\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">OpenAI — Najbolje prakse za evaluaciju\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">Smernice za testiranje graničnih slučajeva, uključujući dugačak kontekst i dugotrajne razgovore, korišćenjem eksplicitnih, ponovljivih evaluacija.\u003C\u002Fp>\u003C\u002Fa>",{"time":212,"blocks":213,"version":797},1790366689232,[214,222,228,236,243,248,253,258,263,268,298,303,308,313,320,325,330,335,340,345,350,355,360,365,370,375,380,385,390,395,437,444,449,454,485,490,522,527,532,537,546,551,556,561,569,574,579,584,589,594,599,607,612,630,635,640,645,650,655,660,665,670,675,680,685,711,716,746,751,761,770,779,788],{"id":215,"data":216,"type":220,"tunes":221},"Qfxj3iD3g1",{"title":217,"maxLevel":218,"minLevel":219},"Sadržaj",3,2,"tableOfContents",{},{"id":223,"data":224,"type":226,"tunes":227},"intro",{"text":225},"Veći kontekstni prozor pruža AI sistemu veći kapacitet. To ne garantuje da će model taj kapacitet dobro iskoristiti. U dugim razgovorima, RAG tokovima, istraživačkim agentima i radnim procesima sa intenzivnim korišćenjem alata, dodavanje više istorije, više dokumenata, više izlaznih podataka alata ili više memorije može učiniti odgovor manje pouzdanim, umesto bolje informisanim.","paragraph",{},{"id":229,"data":230,"type":234,"tunes":235},"direct",{"body":231,"title":232,"variant":233},"\u003Cstrong>Više konteksta može pogoršati odgovor veštačke inteligencije kada dodatne informacije smanjuju odnos signala i šuma, unose konflikte, skrivaju ključne dokaze, zadržavaju zastarelo stanje ili kompresijom uklanjaju važne uslove.\u003C\u002Fstrong> Relevantan inženjerski cilj stoga nije maksimalni kontekst. To je \u003Cstrong>minimalni dovoljni kontekst sa očuvanim dokazima i granicama odlučivanja\u003C\u002Fstrong>.","Direktan odgovor","info","callout",{},{"id":237,"data":238,"type":234,"tunes":242},"model-note",{"body":239,"title":240,"variant":241},"Model kvaliteta konteksta (Context Quality) i test pritiska na kontekst (Context Pressure Test) u nastavku predstavljaju praktične arhitektonske metode predložene u ovom članku, a ne formalne industrijske standarde. Oni sintetišu utvrđena saznanja o efektima pozicije u dugom kontekstu, zagađenju konteksta, sažimanju, preuzimanju informacija i inženjeringu konteksta.","O modelu korišćenom u ovom članku","note",{},{"id":244,"data":245,"type":42,"tunes":247},"h-capacity",{"text":246,"level":219},"Kapacitet konteksta nije isto što i upotrebljivost konteksta",{},{"id":249,"data":250,"type":226,"tunes":252},"p-capacity-1",{"text":251},"Deklarisani kontekstni prozor modela opisuje koliko ulaznih podataka on može da prihvati. To ne podrazumeva da svaki token unutar tog prozora dobija jednaku pažnju ili podjednako doprinosi konačnom odgovoru. Ova razlika je važna jer produkcioni sistemi sve više popunjavaju kontekst istorijom razgovora, preuzetim dokumentima, rezultatima alata, memorijom, struktuiranim stanjem, instrukcijama i međuproduktima.",{},{"id":254,"data":255,"type":226,"tunes":257},"p-capacity-2",{"text":256},"Klasična studija „Lost in the Middle“ pokazala je da modeli sa dugim kontekstom mogu imati lošije rezultate kada se relevantni dokazi nalaze u sredini dugog unosa nego kada se nalaze blizu početka ili kraja. Šira inženjerska pouka nije da je dugi kontekst loš. Ona glasi da dostupnost unutar konteksta nije isto što i pouzdana upotreba.",{},{"id":259,"data":260,"type":226,"tunes":262},"p-capacity-3",{"text":261},"Smernice kompanije OpenAI za upravljanje kontekstom dolaze do istog operativnog zaključka iz drugog pravca: čak i veoma veliki kontekstni prozori mogu biti preopterećeni neprobranom istorijom, suvišnim izlaznim podacima alata i bučnim preuzimanjem podataka. Anthropic na sličan način tretira kontekst kao ograničen resurs koji zahteva aktivni inženjering, a ne pasivno nagomilavanje.",{},{"id":264,"data":265,"type":42,"tunes":267},"h-five",{"text":266,"level":219},"Pet načina na koje dodatni kontekst može umanjiti kvalitet odgovora",{},{"id":269,"data":270,"type":296,"tunes":297},"five-table",{"content":271,"stretched":43,"withHeadings":14},[272,276,280,284,288,292],[273,274,275],"Oblik greške","Šta se menja kada se doda više konteksta","Tipičan simptom",[277,278,279],"Razblaživanje signala","Relevantni dokazi postaju manji udeo ukupnog unosa","Model daje generički odgovor ili propušta ključni odlomak",[281,282,283],"Konflikt dokaza","Različiti dokumenti, verzije ili memorije se ne slažu","Odgovor spaja nekompatibilne tvrdnje ili bira pogrešnu verziju",[285,286,287],"Osetljivost na poziciju","Ključne informacije prelaze u deo konteksta koji se manje pouzdano koristi","Isti dokazi funkcionišu u jednom redosledu, ali ne uspevaju u drugom",[289,290,291],"Zadržavanje zastarelog konteksta","Staro stanje ili prethodni zaključci ostaju prisutni nakon što se realnost promeni","Model nastavlja da ponavlja ranije tačan odgovor",[293,294,295],"Gubitak usled kompresije","Sažimanje ili rezimiranje uklanja ograde, izuzetke, poreklo ili nerazrešenu neizvesnost","Rezime je koherentan, ali rezultujući odgovor postaje preterano samouveren ili preopšten","table",{},{"id":299,"data":300,"type":42,"tunes":302},"h-dilution",{"text":301,"level":218},"1. Razblaživanje signala: relevantni dokazi se takmiče sa svim ostalim",{},{"id":304,"data":305,"type":226,"tunes":307},"p-dilution-1",{"text":306},"Pretpostavimo da se na pitanje može odgovoriti na osnovu dva kratka odlomka. RAG sistem preuzima ta dva odlomka plus osamnaest labavo povezanih „za svaki slučaj“. Odziv preuzimanja se može poboljšati, ali generator sada mora da razlikuje ključne dokaze od pozadinskog materijala. Ako se slične fraze pojavljuju u više dokumenata, dodatni kontekst može učiniti odgovor manje preciznim.",{},{"id":309,"data":310,"type":226,"tunes":312},"p-dilution-2",{"text":311},"Ovo stvara važnu razliku između odziva preuzimanja i korisnosti konteksta. Više preuzetog materijala može povećati verovatnoću da odgovor postoji negde u kontekstu, dok istovremeno smanjuje verovatnoću da model pravim dokazima dodeli dovoljnu težinu.",{},{"id":314,"data":315,"type":234,"tunes":319},"dilution-tip",{"body":316,"title":317,"variant":318},"Nemojte optimizovati top-k izolovano. Merite da li dodavanje dokumenata poboljšava konačnu tvrdnju, čuva atribuciju dokaza i ostaje stabilno tokom ponovljenih testiranja.","Inženjersko pravilo","tip",{},{"id":321,"data":322,"type":42,"tunes":324},"h-conflict",{"text":323,"level":218},"2. Konflikt dokaza: više izvora može značiti više verzija realnosti",{},{"id":326,"data":327,"type":226,"tunes":329},"p-conflict-1",{"text":328},"Dugi konteksti često sadrže međusobno neusklađene informacije: staru i novu API dokumentaciju, dve verzije pravilnika, prethodne i trenutne korisničke preferencije, konkurentne veb izvore, keširano stanje ili rezime koji je generisao model, a koji više ne odgovara izvoru.",{},{"id":331,"data":332,"type":226,"tunes":334},"p-conflict-2",{"text":333},"Ovaj propust nije nužno halucinacija. Model možda verno kombinuje kontradiktorne dokaze. Arhitekturi su stoga potrebna pravila prioriteta: autoritet izvora, verzija, vremenska oznaka, jurisdikcija, korisnik sistema (tenant), revizija proizvoda, stanje korisnika ili eksplicitni metapodaci o zameni starih podataka novim.",{},{"id":336,"data":337,"type":226,"tunes":339},"p-conflict-3",{"text":338},"Bez tih pravila, povećanje konteksta može povećati kontradiktornosti brže nego što povećava znanje.",{},{"id":341,"data":342,"type":42,"tunes":344},"h-position",{"text":343,"level":218},"3. Osetljivost na poziciju: mesto gde se dokaz pojavljuje može promeniti rezultat",{},{"id":346,"data":347,"type":226,"tunes":349},"p-position-1",{"text":348},"Rezultati istraživanja „Izgubljeni u sredini” (Lost in the Middle) pokazali su da samo promena pozicije relevantnih informacija može materijalno promeniti performanse modela. To saznanje je posebno važno za sisteme koji spajaju mnoge pronađene odlomke ili dugačke istorijate u fiksnom redosledu.",{},{"id":351,"data":352,"type":226,"tunes":354},"p-position-2",{"text":353},"Produkcioni test bi stoga trebalo da varira redosled dokumenata, a ne samo da testira jedan kanonski upit. Ako sistem tačno odgovara samo kada je odlučujući dokaz na početku ili na kraju, aplikacija je krhkija nego što to sugeriše jedan rezultat na benčmarku.",{},{"id":356,"data":357,"type":42,"tunes":359},"h-stale",{"text":358,"level":218},"4. Postojanost zastarelog konteksta: model vidi istinu i istorijat zajedno",{},{"id":361,"data":362,"type":226,"tunes":364},"p-stale-1",{"text":363},"Dugotrajni agenti često prenose ranije zaključke unapred. Taj kontinuitet je koristan sve dok se neka činjenica ne promeni. Ako jučerašnji rezultat alata kaže da je implementacija u dobrom stanju, a trenutni rezultat alata kaže da je degradirana, oba mogu ostati u kontekstu osim ako sistem eksplicitno ne zameni ili ograniči opseg starog stanja.",{},{"id":366,"data":367,"type":226,"tunes":369},"p-stale-2",{"text":368},"Zato bi trenutno operativno stanje obično trebalo da potiče iz autoritativnog izvora, dok memorija čuva trajni kontekst kao što su odluke, preferencije ili procedure. Veća istorija razgovora nije zamena za ponovno sagledavanje sadašnjosti.",{},{"id":371,"data":372,"type":42,"tunes":374},"h-compression",{"text":373,"level":218},"5. Gubitak pri kompresiji: manji kontekst takođe može postati lošiji kontekst",{},{"id":376,"data":377,"type":226,"tunes":379},"p-compression-1",{"text":378},"Suprotna intervencija — kompresovanje konteksta — takođe ima svoje načine otkazivanja. Sažeci mogu izostaviti izuzetke, nerešena pitanja, poreklo, precizne identifikatore, negativne dokaze ili uslove pod kojima je zaključak bio validan.",{},{"id":381,"data":382,"type":226,"tunes":384},"p-compression-2",{"text":383},"Rad Microsoft Research-a pod nazivom Agentic Context Engineering opisuje srodan problem kao pristrasnost ka sažetosti i kolaps konteksta: iterativno prepisivanje može ukloniti korisne detalje domena. Cilj stoga nije „kompresovati što je više moguće”. Cilj je smanjiti kontekst uz očuvanje informacija koje menjaju odluke.",{},{"id":386,"data":387,"type":42,"tunes":389},"h-quality",{"text":388,"level":219},"Model kvaliteta konteksta",{},{"id":391,"data":392,"type":226,"tunes":394},"p-quality-intro",{"text":393},"Koristan kontekst se može proceniti kroz šest dimenzija. Nijedna od njih nije prosto broj tokena.",{},{"id":396,"data":397,"type":435,"tunes":436},"quality-comparison",{"rows":398,"title":424,"layout":296,"columns":425},[399,404,408,412,416,420],{"id":400,"label":401,"values":402},"relevance","Relevantnost",[403,403,403],"",{"id":405,"label":406,"values":407},"authority","Autoritativnost",[403,403,403],{"id":409,"label":410,"values":411},"freshness","Svežina",[403,403,403],{"id":413,"label":414,"values":415},"consistency","Doslednost",[403,403,403],{"id":417,"label":418,"values":419},"completeness","Kompletnost za donošenje odluka",[403,403,403],{"id":421,"label":422,"values":423},"traceability","Sledljivost",[403,403,403],"Šest dimenzija kvaliteta konteksta",[426,429,432],{"id":427,"label":428},"dimension","Dimenzija",{"id":430,"label":431},"question","Pitanje",{"id":433,"label":434},"failure","Ako je slabo","comparison",{},{"id":438,"data":439,"type":234,"tunes":443},"target-state",{"body":440,"title":441,"variant":442},"Najbolji kontekst nije najveći kontekst. To je \u003Cstrong>najmanji kontekst koji i dalje čuva dokaze, ograničenja, stanje, izuzetke i poreklo neophodne za pouzdan odgovor ili akciju\u003C\u002Fstrong>.","Ciljno stanje","success",{},{"id":445,"data":446,"type":42,"tunes":448},"h-pressure",{"text":447,"level":219},"Test pritiska na kontekst",{},{"id":450,"data":451,"type":226,"tunes":453},"p-pressure-intro",{"text":452},"Da biste utvrdili da li aplikacija ima koristi od više konteksta, testirajte veličinu konteksta kao eksperimentalnu varijablu umesto da pretpostavljate da je veće uvek bolje.",{},{"id":455,"data":456,"type":483,"tunes":484},"pressure-flow",{"steps":457,"title":447,"orientation":482},[458,461,464,467,470,473,476,479],{"label":459,"description":460},"1. Definišite referentni slučaj","Izaberite zadatak sa poznatim odgovorom i poznatim minimalnim skupom dokaza.",{"label":462,"description":463},"2. Pokrenite minimalno dovoljan kontekst","Obezbedite samo uputstva, trenutno stanje i dokaze neophodne za odgovor.",{"label":465,"description":466},"3. Dodajte relevantnu pozadinu","Dodajte koristan, ali ne i presudan kontekst i izmerite da li se kvalitet poboljšava, ostaje stabilan ili opada.",{"label":468,"description":469},"4. Dodajte realističan šum","Dodajte labavo povezan istorijat, izlaz alata ili pronađene odlomke koje bi produkcioni sistem mogao da uključi.",{"label":471,"description":472},"5. Dodajte kontrolisane konflikte","Uvedite zastarele ili kontradiktorne dokaze sa jasnim metapodacima o verziji i proverite da li tačan izvor i dalje pobeđuje.",{"label":474,"description":475},"6. Promenite redosled odlučujućih dokaza","Postavite ključne informacije blizu početka, sredine i kraja da biste testirali osetljivost na poziciju.",{"label":477,"description":478},"7. Testirajte sažimanje","Zamenite stariji kontekst sažetkom i proverite da li kvalifikatori, poreklo, nerešena pitanja i granice odlučivanja opstaju.",{"label":480,"description":481},"8. Uporedite krivu kvaliteta","Izmerite tačnost, korišćenje dokaza, doslednost, kašnjenje, trošak i varijansu kako se kontekst menja.","auto","processFlow",{},{"id":486,"data":487,"type":42,"tunes":489},"h-measure",{"text":488,"level":219},"Šta meriti umesto broja tokena",{},{"id":491,"data":492,"type":296,"tunes":521},"measure-table",{"content":493,"stretched":43,"withHeadings":14},[494,497,500,503,506,509,512,515,518],[495,496],"Metrika","Šta otkriva",[498,499],"Tačnost odgovora","Da li je konačni rezultat tačan",[501,502],"Potkrepljenost tvrdnji dokazima","Da li materijalne tvrdnje ostaju utemeljene kako se kontekst menja",[504,505],"Iskorišćenost dokaza","Da li odgovor prati odlučujući dokaz umesto prethodnog znanja modela",[507,508],"Tačnost rešavanja konflikata","Da li aktuelni \u002F autoritativni dokazi pobeđuju zastarele ili slabije izvore",[510,511],"Robusnost na poziciju","Da li promena redosleda dokaza menja tačnost",[513,514],"Zadržavanje pri sažimanju","Da li sažeci čuvaju ograničenja, izuzetke, identifikatore, poreklo i nerešena stanja",[516,517],"Varijansa izlaza kroz ponavljanja","Da li dodatni kontekst čini sistem manje stabilnim",[519,520],"Latencija i trošak tokena","Da li dodate informacije donose dovoljno kvaliteta da opravdaju operativne troškove",{},{"id":523,"data":524,"type":42,"tunes":526},"h-topk",{"text":525,"level":219},"RAG: zašto povećanje parametra top-k može da škodi",{},{"id":528,"data":529,"type":226,"tunes":531},"p-topk-1",{"text":530},"Uobičajeni obrazac za fino podešavanje RAG-a jeste povećanje parametra top-k kada sistem propusti odgovor. To može poboljšati odziv kandidata (recall), ali takođe može povećati količinu irelevantnog konteksta, dupliranih dokaza, zastarelih odlomaka i protivrečnih dokumenata.",{},{"id":533,"data":534,"type":226,"tunes":536},"p-topk-2",{"text":535},"Bolje pitanje je da li ključni dokaz nedostaje u pretrazi ili samo gubi uticaj nakon sklapanja konteksta. Ako se tačan odlomak već nalazi u skupu kandidata, povećanje parametra top-k može predstavljati rešavanje pogrešnog problema.",{},{"id":538,"data":539,"type":544,"tunes":545},"internal-rag",{"url":540,"title":541,"excerpt":542,"ctaLabel":543},"https:\u002F\u002Fstajic.de\u002Fsr\u002Fblog\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method","RAG nije uspeo — ali koji sloj je zapravo zakazao? Dijagnostički metod","Metod sloj-po-sloj za izolovanje grešaka u pokrivenosti izvora, pretrazi, rangiranju, sklapanju konteksta, generisanju, atribuciji dokaza i ažurnosti.","Pročitajte dijagnostički metod za RAG","referralArticle",{},{"id":547,"data":548,"type":42,"tunes":550},"h-agents",{"text":549,"level":219},"Dugotrajni agenti: kontinuitet nije akumulacija",{},{"id":552,"data":553,"type":226,"tunes":555},"p-agents-1",{"text":554},"Agentu je potreban kontinuitet kroz korake, ali kontinuitet ne zahteva ponavljanje svakog prethodnog tokena. OpenAI demonstrira skraćivanje i kompresiju za kontekst dugotrajnih sesija. Anthropic preporučuje sabijanje (kompakciju), strukturisano vođenje beleški i druge tehnike kako bi se sačuvale korisne informacije uz kontrolu zagađenja konteksta.",{},{"id":557,"data":558,"type":226,"tunes":560},"p-agents-2",{"text":559},"Snažna arhitektura za dugotrajne procese obično razdvaja trajnu memoriju, trenutno stanje, spoljne artefakte, pretragu i kontekst namenjen modelu. To omogućava sistemu da sačuva ono što je važno, a da pritom ne ubacuje svaki istorijski detalj u svako zaključivanje.",{},{"id":562,"data":563,"type":544,"tunes":568},"internal-memory",{"url":564,"title":565,"excerpt":566,"ctaLabel":567},"https:\u002F\u002Fstajic.de\u002Fsr\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context","Memorija AI agenata nije RAG: Kako razdvojiti memoriju, pretragu, stanje i kontekst","Praktična četvoroslojna arhitektura za razdvajanje onoga što opstaje, onoga što je trenutno merodavno, onoga što se pretražuje i onoga što model zaista dobija.","Pročitajte članak o arhitekturi memorije",{},{"id":570,"data":571,"type":42,"tunes":573},"h-order",{"text":572,"level":219},"Redosled u kontekstu treba da bude nameran",{},{"id":575,"data":576,"type":226,"tunes":578},"p-order-1",{"text":577},"Konstrukcija konteksta predstavlja problem informacione arhitekture. Ključna uputstva, trenutno stanje, odlučujući dokazi i ograničenja specifična za zadatak ne bi trebalo da budu nasumično raspoređeni. Kada sistemi mehanički spajaju izvore, oni implicitno prepuštaju određivanje prioriteta pozicionim efektima i pažnji modela.",{},{"id":580,"data":581,"type":226,"tunes":583},"p-order-2",{"text":582},"Ne postoji univerzalno najbolji redosled za svaki model i zadatak, pa bi redosled trebalo empirijski procenjivati. Koristan skup testova nasumično raspoređuje ili sistematski menja poziciju dokumenta i meri da li ista tvrdnja ostaje stabilna.",{},{"id":585,"data":586,"type":42,"tunes":588},"h-boundaries",{"text":587,"level":219},"Očuvajte granice odlučivanja tokom sabijanja",{},{"id":590,"data":591,"type":226,"tunes":593},"p-bound-1",{"text":592},"Sažetak koji glasi „koristite pristup X“ slabiji je od sažetka koji čuva razlog zašto je X izabran i šta bi poništilo tu odluku. Sabijanje konteksta treba da zadrži varijable koje mogu promeniti odgovor: verziju, datum, pretpostavke, stanje, autoritet, nerešena neslaganja i poreklo dokaza.",{},{"id":595,"data":596,"type":226,"tunes":598},"p-bound-2",{"text":597},"Ovo direktno povezuje inženjering konteksta sa validnošću odgovora. Ako se sabijanjem sačuva zaključak, ali se ukloni granica njegove validnosti, budući odgovori mogu ostati interno dosledni, a da pritom postanu eksterno netačni.",{},{"id":600,"data":601,"type":544,"tunes":606},"internal-avb",{"url":602,"title":603,"excerpt":604,"ctaLabel":605},"https:\u002F\u002Fstajic.de\u002Fsr\u002Fblog\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers","Granica validnosti odgovora: Karika koja nedostaje između relevantnosti i pouzdanih AI odgovora","Okvir za eksplicitno definisanje uslova pod kojima se tvrdnja veštačke inteligencije primenjuje i koje promene zahtevaju ograničenje, ponovno izračunavanje ili odbacivanje.","Pročitajte o granici validnosti odgovora",{},{"id":608,"data":609,"type":42,"tunes":611},"h-policy",{"text":610,"level":219},"Praktična politika konstrukcije konteksta",{},{"id":613,"data":614,"type":628,"tunes":629},"policy-list",{"meta":615,"items":616,"style":627},{},[617,618,619,620,621,622,623,624,625,626],"Počnite od trenutnog zadatka, a ne od svega što sistem zna.","Ponovo pročitajte promenljivo stanje iz merodavnih sistema pre donošenja važnih odluka.","Pretražujte dokaze za trenutno pitanje umesto da sa sobom prenosite velike statičke korpuse.","Uklonite duplirane ili bezvredne izlaze iz alata.","Zadržite verziju izvora, vremensku oznaku, autoritet i poreklo uz važne dokaze.","Jasno odredite prioritet kada su trenutne i istorijske informacije u sukobu.","Sačuvajte pravila zajedno sa njihovim izuzecima i preduslovima.","Čuvajte trajne odluke i višekratne procedure van neposrednog konteksta kada nije potrebno njihovo doslovno ponavljanje.","Sabijajte istoriju samo uz testove za zadržavanje ograničenja, identifikatora, izuzetaka i porekla.","Procenjujte veličinu konteksta, redosled i šum kroz ponovljene probe, a ne na osnovu jednog upita.","unordered","list",{},{"id":631,"data":632,"type":42,"tunes":634},"h-change",{"text":633,"level":219},"Šta bi promenilo ovaj odgovor?",{},{"id":636,"data":637,"type":226,"tunes":639},"p-change-1",{"text":638},"Kompromis se menja u zavisnosti od arhitekture modela, obuke, tipa zadatka i dužine konteksta. Budući modeli mogu postati znatno otporniji na poziciju, šum i kontradiktorne informacije. Zadatak sa malim, čistim korpusom takođe može imati koristi od jednostavnog pružanja kompletnog izvora umesto izgradnje složenog mehanizma za preuzimanje (retrieval pipeline).",{},{"id":641,"data":642,"type":226,"tunes":644},"p-change-2",{"text":643},"Preporuka se takođe menja kada je izostavljanje opasnije od šuma. U istraživačkim zadacima ili zadacima otkrivanja sa visokim odzivom (high-recall), veći kandidatski kontekst može biti opravdan pre kasnije faze filtriranja ili sinteze. U produkcionim sistemima osetljivim na latenciju, striktniji odabir konteksta može biti poželjniji.",{},{"id":646,"data":647,"type":226,"tunes":649},"p-change-3",{"text":648},"Osnovni princip bi se promenio samo ako bi modeli postali pouzdano invarijantni na irelevantne informacije, poziciju, protivrečnosti i zastarele dokaze. Do tada, kontekst treba tretirati kao pažljivo odabran resurs za izvršavanje, a ne kao pasivno skladište.",{},{"id":651,"data":652,"type":42,"tunes":654},"h-limitations",{"text":653,"level":219},"Ograničenja",{},{"id":656,"data":657,"type":226,"tunes":659},"p-limit-1",{"text":658},"Ponašanje u dugom kontekstu znatno varira među modelima i radnim opterećenjima. Prvobitni eksperimenti „Lost in the Middle” koristili su ranije generacije modela, pa se ne sme pretpostaviti da njihove tačne veličine efekta predstavljaju današnje sisteme. Nalaz ostaje koristan kao obrazac otkaza koji treba testirati, a ne kao univerzalna fiksna kriva performansi.",{},{"id":661,"data":662,"type":226,"tunes":664},"p-limit-2",{"text":663},"Isto tako, smanjenje konteksta može ukloniti neophodne dokaze. Zbijanje (kompaktovanje) uvodi rizik sažimanja, a agresivno filtriranje preuzimanja može smanjiti odziv. Cilj nije minimalan broj tokena po svaku cenu; cilj je dovoljan, aktuelan i sledljiv kontekst za odluku koja se donosi.",{},{"id":666,"data":667,"type":42,"tunes":669},"h-conclusion",{"text":668,"level":219},"Zaključak",{},{"id":671,"data":672,"type":226,"tunes":674},"p-conclusion-1",{"text":673},"Pitanje „Koliko konteksta model može da primi?” manje je korisno od pitanja „Koliko ovog konteksta poboljšava odluku?” Više tokena može dodati dokaze, ali takođe može uneti ometanja, protivrečnosti, zastarelo stanje, pozicionu ranjivost i dug kompresije.",{},{"id":676,"data":677,"type":226,"tunes":679},"p-conclusion-2",{"text":678},"Tretirajte kontekst kao projektovani radni skup. Počnite sa minimalnim dovoljnim dokazima. Dodajte informacije samo kada poboljšavaju izmerene performanse. Eksplicitno testirajte šum, konflikt, redosled i sažimanje. Veliki kontekstualni prozor je kapacitet; kvalitet konteksta je arhitektura.",{},{"id":681,"data":682,"type":42,"tunes":684},"h-faq",{"text":683,"level":219},"Često postavljana pitanja",{},{"id":686,"data":687,"type":686,"tunes":710},"faq",{"items":688,"title":709},[689,693,697,701,705],{"id":690,"answer":691,"question":692},"faq1","Da. Dodatni kontekst može razvodniti relevantne dokaze, uneti kontradiktorne ili zastarele informacije, premestiti odlučujuće dokaze na manje robusne pozicije i povećati šansu da model koristi slabe umesto odlučujućih signala.","Može li pružanje više konteksta AI modelu pogoršati njegov odgovor?",{"id":694,"answer":695,"question":696},"faq2","Uglavnom ne. Veći kontekstualni prozor povećava kapacitet, ali preuzimanje i dalje pomaže u odabiru aktuelnih i relevantnih informacija, kontroli troškova, očuvanju granica izvora i izbegavanju slanja velikih količina nepovezanih podataka u svaki zahtev.","Da li veći kontekstualni prozor eliminiše potrebu za RAG-om?",{"id":698,"answer":699,"question":700},"faq3","On opisuje uočene slučajeve u kojima jezički modeli manje pouzdano koriste relevantne informacije kada se one nalaze u sredini dugog konteksta nego kada se pojavljuju blizu početka ili kraja. Tačan efekat varira u zavisnosti od modela i zadatka i treba ga testirati na savremenim sistemima.","Šta je problem „Lost in the Middle”?",{"id":702,"answer":703,"question":704},"faq4","Ne. Ako relevantni dokazi nedostaju u skupu kandidata, veći top-k može poboljšati odziv. Ako su dokazi već prisutni, ali se razvodnjavaju dodatnim materijalom, povećanje top-k može pogoršati kontekst. Dijagnostikujte preuzimanje i sklapanje konteksta odvojeno.","Treba li uvek smanjivati top-k kod RAG-a?",{"id":706,"answer":707,"question":708},"faq5","Treba da sačuva trajne odluke, trenutne ciljeve, nerešena pitanja, identifikatore, ograničenja, izuzetke, poreklo dokaza i uslove koji bi promenili raniji zaključak.","Šta rezime konteksta treba da sačuva?","Dug kontekst i kvalitet odgovora veštačke inteligencije",{},{"id":712,"data":713,"type":42,"tunes":715},"h-glossary",{"text":714,"level":219},"Rečnik pojmova",{},{"id":717,"data":718,"type":717,"tunes":745},"glossary",{"title":719,"entries":720},"Ključni pojmovi inženjeringa konteksta",[721,725,729,733,737,741],{"term":722,"anchor":723,"definition":724},"Kontekstualni prozor (Context window)","context-window","Količina ulaznih i izlaznih informacija u tokenima na koju model može obratiti pažnju unutar jedne sekvence inferencije.",{"term":726,"anchor":727,"definition":728},"Zagađenje konteksta (Context pollution)","context-pollution","Degradacija uzrokovana time što irelevantne, zastarele, suvišne, kontradiktorne ili na drugi način bezvredne informacije zauzimaju kontekst modela.",{"term":730,"anchor":731,"definition":732},"Razvodnjavanje signala (Signal dilution)","signal-dilution","Smanjenje relativne istaknutosti odlučujućih dokaza usled dodavanja dodatnih informacija niske vrednosti ili konkurentskih informacija.",{"term":734,"anchor":735,"definition":736},"Kompaktovanje konteksta (Context compaction)","context-compaction","Smanjivanje nagomilanog konteksta sažimanjem, restrukturiranjem, eksternalizacijom ili na drugi način očuvanjem suštinskih informacija u manjoj radnoj reprezentaciji.",{"term":738,"anchor":739,"definition":740},"Poziciona robusnost (Position robustness)","position-robustness","Stepen u kojem performanse modela ostaju stabilne kada se relevantne informacije pojavljuju na različitim pozicijama unutar konteksta.",{"term":742,"anchor":743,"definition":744},"Minimalni dovoljni kontekst (Minimum sufficient context)","minimum-sufficient-context","Najmanji praktični radni kontekst koji i dalje čuva dokaze, stanje, ograničenja, izuzetke i poreklo potrebne za pouzdano izvršavanje.",{},{"id":747,"data":748,"type":42,"tunes":750},"h-sources",{"text":749,"level":219},"Primarni izvori i dodatna literatura",{},{"id":752,"data":753,"type":759,"tunes":760},"src-openai-session",{"link":754,"meta":755},"https:\u002F\u002Fdevelopers.openai.com\u002Fcookbook\u002Fexamples\u002Fagents_sdk\u002Fsession_memory",{"image":756,"title":757,"description":758},{"url":403},"OpenAI — Context Engineering: Short-Term Memory Management with Sessions","Smernice o skraćivanju i kompresiji, sa diskusijom o ometanju, neefikasnosti, zastarelom kontekstu, bučnom preuzimanju i dugotrajnim sesijama.","linkTool",{},{"id":762,"data":763,"type":759,"tunes":769},"src-anthropic-context",{"link":764,"meta":765},"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Feffective-context-engineering-for-ai-agents",{"image":766,"title":767,"description":768},{"url":403},"Anthropic — Effective Context Engineering for AI Agents","Inženjerske smernice o zagađenju konteksta, sažimanju, strukturiranom vođenju beleški i upravljanju kontekstom agenata na dugim horizontima.",{},{"id":771,"data":772,"type":759,"tunes":778},"src-lost-middle",{"link":773,"meta":774},"https:\u002F\u002Faclanthology.org\u002F2024.tacl-1.9\u002F",{"image":775,"title":776,"description":777},{"url":403},"Liu et al. — Lost in the Middle: How Language Models Use Long Contexts","TACL rad koji prikazuje poziciono osetljivu upotrebu relevantnih informacija u dugim kontekstima i motiviše eksplicitne testove robusnosti na dugi kontekst.",{},{"id":780,"data":781,"type":759,"tunes":787},"src-ms-ace",{"link":782,"meta":783},"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Fpublication\u002Fagentic-context-engineering-evolving-contexts-for-self-improving-language-models\u002F",{"image":784,"title":785,"description":786},{"url":403},"Microsoft Research — Agentic Context Engineering (ACE)","Istraživanje o razvoju strukturiranih konteksta uz istovremeno rešavanje pristrasnosti ka sažetosti i kolapsa konteksta.",{},{"id":789,"data":790,"type":759,"tunes":796},"src-openai-evals",{"link":791,"meta":792},"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fevaluation-best-practices",{"image":793,"title":794,"description":795},{"url":403},"OpenAI — Najbolje prakse za evaluaciju","Smernice za testiranje graničnih slučajeva, uključujući dugačak kontekst i dugotrajne razgovore, korišćenjem eksplicitnih, ponovljivih evaluacija.",{},"2.31","Veći kontekstni prozor ne garantuje bolji odgovor. Ovaj članak objašnjava kako razblaživanje signala, protivrečni dokazi, zastarelo stanje, osetljivost na poziciju i kompresija sa gubicima mogu smanjiti pouzdanost veštačke inteligencije—i uvodi praktičan test pritiska konteksta.","\u002Fuploads\u002F2026\u002F09\u002Fwhy-more-context-can-make-ai-answers-worse-1790351615793-2ntv2v.webp","why-more-context-can-make-ai-answers-worse-1790351615793-2ntv2v","PUBLISHED","2026-09-25T11:51:00.000Z","2026-09-25T15:51:57.195Z","2026-09-25T20:09:28.015Z",{"en":806,"de":807,"sr":808,"es":809,"fr":810,"it":811,"ru":812,"zh":813},"\u002Fblog\u002Fwhy-more-context-can-make-ai-answers-worse","\u002Fde\u002Fblog\u002Fwhy-more-context-can-make-ai-answers-worse","\u002Fsr\u002Fblog\u002Fwhy-more-context-can-make-ai-answers-worse","\u002Fes\u002Fblog\u002Fwhy-more-context-can-make-ai-answers-worse","\u002Ffr\u002Fblog\u002Fwhy-more-context-can-make-ai-answers-worse","\u002Fit\u002Fblog\u002Fwhy-more-context-can-make-ai-answers-worse","\u002Fru\u002Fblog\u002Fwhy-more-context-can-make-ai-answers-worse","\u002Fzh\u002Fblog\u002Fwhy-more-context-can-make-ai-answers-worse",[815,819,823],{"id":816,"name":817,"slug":818},64,"Informaciona arhitektura","information-architecture",{"id":820,"name":821,"slug":822},60,"Kontrole troška i latencije","cost-and-latency",{"id":824,"name":825,"slug":826},97,"Verifikacija na test setu","verification",{"id":828,"login":829,"email":830,"displayName":831},"20","rooth8233","aleksandar@stajic.de","Aleksandar Stajić",[833,1304],{"lang":834,"title":835,"content":836,"contentJson":837,"excerpt":1303},"en","Why More Context Can Make AI Answers Worse","{\"time\":1790351629251,\"blocks\":[{\"id\":\"Qfxj3iD3g1\",\"type\":\"tableOfContents\",\"data\":{\"title\":\"Contents\",\"minLevel\":2,\"maxLevel\":3},\"tunes\":{}},{\"id\":\"intro\",\"type\":\"paragraph\",\"data\":{\"text\":\"A larger context window gives an AI system more capacity. It does not guarantee that the model will use that capacity well. In long conversations, RAG pipelines, research agents, and tool-heavy workflows, adding more history, more documents, more tool output, or more memory can make a response less reliable rather than more informed.\"},\"tunes\":{}},{\"id\":\"direct\",\"type\":\"callout\",\"data\":{\"variant\":\"info\",\"title\":\"Direct answer\",\"body\":\"\u003Cstrong>More context can make an AI answer worse when the additional information lowers the signal-to-noise ratio, introduces conflicts, hides decisive evidence, preserves stale state, or compresses away important conditions.\u003C\u002Fstrong> The relevant engineering objective is therefore not maximum context. It is \u003Cstrong>minimum sufficient context with preserved evidence and decision boundaries\u003C\u002Fstrong>.\"},\"tunes\":{}},{\"id\":\"model-note\",\"type\":\"callout\",\"data\":{\"variant\":\"note\",\"title\":\"About the model used in this article\",\"body\":\"The Context Quality model and Context Pressure Test below are practical architecture methods proposed in this article, not formal industry standards. They synthesize established findings on long-context position effects, context pollution, compaction, retrieval, and context engineering.\"},\"tunes\":{}},{\"id\":\"h-capacity\",\"type\":\"header\",\"data\":{\"text\":\"Context capacity is not context usability\",\"level\":2},\"tunes\":{}},{\"id\":\"p-capacity-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A model's advertised context window describes how much input it can accept. It does not imply that every token inside that window receives equal attention or contributes equally to the final answer. The distinction matters because production systems increasingly fill context with conversation history, retrieved documents, tool results, memory, structured state, instructions, and intermediate artifacts.\"},\"tunes\":{}},{\"id\":\"p-capacity-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The classic “Lost in the Middle” study showed that long-context models can perform worse when relevant evidence appears in the middle of a long input than when it appears near the beginning or end. The broader engineering lesson is not that long context is bad. It is that availability inside the context is not equivalent to reliable use.\"},\"tunes\":{}},{\"id\":\"p-capacity-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"OpenAI's context-management guidance reaches the same operational conclusion from another direction: even very large context windows can be overwhelmed by uncurated history, redundant tool output, and noisy retrieval. Anthropic likewise treats context as a finite resource that requires active engineering rather than passive accumulation.\"},\"tunes\":{}},{\"id\":\"h-five\",\"type\":\"header\",\"data\":{\"text\":\"Five ways additional context can reduce answer quality\",\"level\":2},\"tunes\":{}},{\"id\":\"five-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Failure mode\",\"What changes when more context is added\",\"Typical symptom\"],[\"Signal dilution\",\"Relevant evidence becomes a smaller fraction of the total input\",\"The model gives a generic answer or misses the decisive passage\"],[\"Evidence conflict\",\"Different documents, versions, or memories disagree\",\"The answer blends incompatible claims or chooses the wrong version\"],[\"Position sensitivity\",\"Decisive information moves into a less reliably used part of the context\",\"The same evidence works in one ordering but fails in another\"],[\"Stale-context persistence\",\"Old state or prior conclusions remain present after reality changes\",\"The model keeps repeating a formerly correct answer\"],[\"Compression loss\",\"Compaction or summarization removes qualifiers, exceptions, provenance, or unresolved uncertainty\",\"The summary is coherent but the resulting answer becomes overconfident or overgeneralized\"]]},\"tunes\":{}},{\"id\":\"h-dilution\",\"type\":\"header\",\"data\":{\"text\":\"1. Signal dilution: relevant evidence competes with everything else\",\"level\":3},\"tunes\":{}},{\"id\":\"p-dilution-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Suppose a question can be answered from two short passages. A RAG system retrieves those passages plus eighteen loosely related ones “for safety.” Retrieval recall may improve, but the generator must now distinguish decisive evidence from background material. If similar phrases appear across several documents, the additional context can make the answer less precise.\"},\"tunes\":{}},{\"id\":\"p-dilution-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"This creates an important distinction between retrieval recall and context utility. More retrieved material can increase the probability that the answer exists somewhere in the context while simultaneously reducing the probability that the model gives the right evidence enough weight.\"},\"tunes\":{}},{\"id\":\"dilution-tip\",\"type\":\"callout\",\"data\":{\"variant\":\"tip\",\"title\":\"Engineering rule\",\"body\":\"Do not optimize top-k in isolation. Measure whether adding documents improves the final claim, preserves evidence attribution, and survives repeated trials.\"},\"tunes\":{}},{\"id\":\"h-conflict\",\"type\":\"header\",\"data\":{\"text\":\"2. Evidence conflict: more sources can mean more versions of reality\",\"level\":3},\"tunes\":{}},{\"id\":\"p-conflict-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Long contexts often contain mutually inconsistent information: old and new API documentation, two policy versions, previous and current user preferences, competing web sources, cached state, or a model-generated summary that no longer matches the source.\"},\"tunes\":{}},{\"id\":\"p-conflict-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The failure is not necessarily hallucination. The model may be faithfully combining contradictory evidence. The architecture therefore needs precedence rules: source authority, version, timestamp, jurisdiction, tenant, product revision, user state, or explicit supersession metadata.\"},\"tunes\":{}},{\"id\":\"p-conflict-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"Without those rules, increasing context can increase contradiction faster than it increases knowledge.\"},\"tunes\":{}},{\"id\":\"h-position\",\"type\":\"header\",\"data\":{\"text\":\"3. Position sensitivity: where evidence appears can change the result\",\"level\":3},\"tunes\":{}},{\"id\":\"p-position-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The “Lost in the Middle” results demonstrated that changing only the position of relevant information can materially change model performance. That finding is especially important for systems that concatenate many retrieved passages or long histories in a fixed order.\"},\"tunes\":{}},{\"id\":\"p-position-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"A production test should therefore vary document order, not merely test one canonical prompt. If the system answers correctly only when the decisive evidence is first or last, the application is more fragile than a single benchmark score suggests.\"},\"tunes\":{}},{\"id\":\"h-stale\",\"type\":\"header\",\"data\":{\"text\":\"4. Stale-context persistence: the model sees truth and history together\",\"level\":3},\"tunes\":{}},{\"id\":\"p-stale-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Long-running agents frequently carry earlier conclusions forward. That continuity is useful until a fact changes. If a tool result from yesterday says a deployment is healthy and a current tool result says it is degraded, both may remain in context unless the system explicitly replaces or scopes old state.\"},\"tunes\":{}},{\"id\":\"p-stale-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"This is why current operational state should normally come from an authoritative source, while memory preserves durable context such as decisions, preferences, or procedures. More conversation history is not a substitute for re-reading the present.\"},\"tunes\":{}},{\"id\":\"h-compression\",\"type\":\"header\",\"data\":{\"text\":\"5. Compression loss: smaller context can also become worse context\",\"level\":3},\"tunes\":{}},{\"id\":\"p-compression-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The opposite intervention — compressing context — also has failure modes. Summaries can drop exceptions, unresolved questions, provenance, precise identifiers, negative evidence, or the conditions under which a conclusion was valid.\"},\"tunes\":{}},{\"id\":\"p-compression-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Microsoft Research's Agentic Context Engineering work describes a related problem as brevity bias and context collapse: iterative rewriting can remove useful domain detail. The objective is therefore not “compress as much as possible.” It is to reduce context while preserving the information that changes decisions.\"},\"tunes\":{}},{\"id\":\"h-quality\",\"type\":\"header\",\"data\":{\"text\":\"The Context Quality model\",\"level\":2},\"tunes\":{}},{\"id\":\"p-quality-intro\",\"type\":\"paragraph\",\"data\":{\"text\":\"A useful context can be evaluated across six dimensions. None of them is simply token count.\"},\"tunes\":{}},{\"id\":\"quality-comparison\",\"type\":\"comparison\",\"data\":{\"title\":\"Six dimensions of context quality\",\"layout\":\"table\",\"columns\":[{\"id\":\"dimension\",\"label\":\"Dimension\"},{\"id\":\"question\",\"label\":\"Question\"},{\"id\":\"failure\",\"label\":\"If weak\"}],\"rows\":[{\"id\":\"relevance\",\"label\":\"Relevance\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"authority\",\"label\":\"Authority\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"freshness\",\"label\":\"Freshness\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"consistency\",\"label\":\"Consistency\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"completeness\",\"label\":\"Decision completeness\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"traceability\",\"label\":\"Traceability\",\"values\":[\"\",\"\",\"\"]}]},\"tunes\":{}},{\"id\":\"target-state\",\"type\":\"callout\",\"data\":{\"variant\":\"success\",\"title\":\"Target state\",\"body\":\"The best context is not the largest context. It is the \u003Cstrong>smallest context that still preserves the evidence, constraints, state, exceptions, and provenance required for a reliable answer or action\u003C\u002Fstrong>.\"},\"tunes\":{}},{\"id\":\"h-pressure\",\"type\":\"header\",\"data\":{\"text\":\"The Context Pressure Test\",\"level\":2},\"tunes\":{}},{\"id\":\"p-pressure-intro\",\"type\":\"paragraph\",\"data\":{\"text\":\"To determine whether an application benefits from more context, test context size as an experimental variable instead of assuming that larger is better.\"},\"tunes\":{}},{\"id\":\"pressure-flow\",\"type\":\"processFlow\",\"data\":{\"title\":\"Context Pressure Test\",\"orientation\":\"auto\",\"steps\":[{\"label\":\"1. Define a gold case\",\"description\":\"Choose a task with a known answer and a known minimal evidence set.\"},{\"label\":\"2. Run minimal sufficient context\",\"description\":\"Provide only the instructions, current state, and evidence necessary for the answer.\"},{\"label\":\"3. Add relevant background\",\"description\":\"Add useful but non-decisive context and measure whether quality improves, stays stable, or falls.\"},{\"label\":\"4. Add realistic noise\",\"description\":\"Add loosely related history, tool output, or retrieved passages that a production system might include.\"},{\"label\":\"5. Add controlled conflicts\",\"description\":\"Introduce stale or contradictory evidence with clear version metadata and verify that the correct source still wins.\"},{\"label\":\"6. Reorder decisive evidence\",\"description\":\"Place the key information near the beginning, middle, and end to test position sensitivity.\"},{\"label\":\"7. Test compaction\",\"description\":\"Replace older context with a summary and verify that qualifiers, provenance, unresolved issues, and decision boundaries survive.\"},{\"label\":\"8. Compare the quality curve\",\"description\":\"Measure correctness, evidence use, consistency, latency, cost, and variance as context changes.\"}]},\"tunes\":{}},{\"id\":\"h-measure\",\"type\":\"header\",\"data\":{\"text\":\"What to measure instead of token count\",\"level\":2},\"tunes\":{}},{\"id\":\"measure-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Metric\",\"What it reveals\"],[\"Answer correctness\",\"Whether the final result is right\"],[\"Claim-level evidence support\",\"Whether material claims remain grounded as context changes\"],[\"Evidence utilization\",\"Whether the answer follows the decisive evidence instead of prior model knowledge\"],[\"Conflict resolution accuracy\",\"Whether current \u002F authoritative evidence wins over stale or weaker sources\"],[\"Position robustness\",\"Whether reordering evidence changes correctness\"],[\"Compaction retention\",\"Whether summaries preserve constraints, exceptions, identifiers, provenance, and unresolved state\"],[\"Output variance across trials\",\"Whether additional context makes the system less stable\"],[\"Latency and token cost\",\"Whether the added information produces enough quality to justify its operational cost\"]]},\"tunes\":{}},{\"id\":\"h-topk\",\"type\":\"header\",\"data\":{\"text\":\"RAG: why increasing top-k can hurt\",\"level\":2},\"tunes\":{}},{\"id\":\"p-topk-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A common RAG tuning pattern is to increase top-k when the system misses an answer. This can improve candidate recall but also increase irrelevant context, duplicate evidence, outdated passages, and conflicting documents.\"},\"tunes\":{}},{\"id\":\"p-topk-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The better question is whether the decisive evidence is missing from retrieval or merely losing influence after context assembly. If the correct passage already appears in the candidate set, increasing top-k may solve the wrong problem.\"},\"tunes\":{}},{\"id\":\"internal-rag\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method\",\"title\":\"RAG Failed — But Which Layer Actually Failed? A Diagnostic Method\",\"excerpt\":\"A layer-by-layer method for isolating source coverage, retrieval, ranking, context assembly, generation, evidence attribution, and freshness failures.\",\"ctaLabel\":\"Read the RAG diagnostic method\"},\"tunes\":{}},{\"id\":\"h-agents\",\"type\":\"header\",\"data\":{\"text\":\"Long-running agents: continuity is not accumulation\",\"level\":2},\"tunes\":{}},{\"id\":\"p-agents-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"An agent needs continuity across steps, but continuity does not require replaying every prior token. OpenAI demonstrates trimming and compression for long-running session context. Anthropic recommends compaction, structured note-taking, and other techniques to preserve useful information while controlling context pollution.\"},\"tunes\":{}},{\"id\":\"p-agents-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"A strong long-running architecture usually separates durable memory, current state, external artifacts, retrieval, and model-facing context. That allows the system to preserve what matters without forcing every historical detail into every inference.\"},\"tunes\":{}},{\"id\":\"internal-memory\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context\",\"title\":\"AI Agent Memory Is Not RAG: How to Separate Memory, Retrieval, State and Context\",\"excerpt\":\"A practical four-layer architecture for separating what persists, what is authoritative now, what is retrieved, and what the model actually receives.\",\"ctaLabel\":\"Read the memory architecture article\"},\"tunes\":{}},{\"id\":\"h-order\",\"type\":\"header\",\"data\":{\"text\":\"Context order should be intentional\",\"level\":2},\"tunes\":{}},{\"id\":\"p-order-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Context construction is an information architecture problem. Critical instructions, current state, decisive evidence, and task-specific constraints should not be placed arbitrarily. When systems concatenate sources mechanically, they implicitly delegate prioritization to positional effects and model attention.\"},\"tunes\":{}},{\"id\":\"p-order-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"There is no universal best ordering for every model and task, so ordering should be evaluated empirically. A useful test suite randomizes or systematically varies document position and measures whether the same claim remains stable.\"},\"tunes\":{}},{\"id\":\"h-boundaries\",\"type\":\"header\",\"data\":{\"text\":\"Preserve decision boundaries during compaction\",\"level\":2},\"tunes\":{}},{\"id\":\"p-bound-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A summary that says “use approach X” is weaker than a summary that preserves why X was chosen and what would invalidate the decision. Context compaction should retain the variables that can change the answer: version, date, assumptions, state, authority, unresolved disagreement, and evidence provenance.\"},\"tunes\":{}},{\"id\":\"p-bound-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"This connects context engineering directly to answer validity. If compaction preserves a conclusion but removes its validity boundary, future responses can remain internally consistent while becoming externally wrong.\"},\"tunes\":{}},{\"id\":\"internal-avb\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers\",\"title\":\"The Answer Validity Boundary: The Missing Layer Between Relevance and Reliable AI Answers\",\"excerpt\":\"A framework for making explicit the conditions under which an AI claim applies and what changes require restriction, recalculation, or abandonment.\",\"ctaLabel\":\"Read the Answer Validity Boundary\"},\"tunes\":{}},{\"id\":\"h-policy\",\"type\":\"header\",\"data\":{\"text\":\"A practical context construction policy\",\"level\":2},\"tunes\":{}},{\"id\":\"policy-list\",\"type\":\"list\",\"data\":{\"style\":\"unordered\",\"meta\":{},\"items\":[\"Start from the current task, not from everything the system knows.\",\"Re-read volatile state from authoritative systems before consequential decisions.\",\"Retrieve evidence for the current question instead of carrying large static corpora forward.\",\"Remove duplicate or low-value tool output.\",\"Keep source version, timestamp, authority, and provenance with important evidence.\",\"Make precedence explicit when current and historical information conflict.\",\"Preserve rules together with their exceptions and prerequisites.\",\"Store durable decisions and reusable procedures outside the immediate context when they do not need verbatim replay.\",\"Compact history only with tests for constraint, identifier, exception, and provenance retention.\",\"Evaluate context size, ordering, and noise with repeated trials rather than a single prompt.\"]},\"tunes\":{}},{\"id\":\"h-change\",\"type\":\"header\",\"data\":{\"text\":\"What would change this answer?\",\"level\":2},\"tunes\":{}},{\"id\":\"p-change-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The trade-off changes with model architecture, training, task type, and context length. Future models may become substantially more robust to position, noise, and conflicting information. A task with a small clean corpus can also benefit from simply providing the complete source rather than building an elaborate retrieval pipeline.\"},\"tunes\":{}},{\"id\":\"p-change-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The recommendation also changes when omission is more dangerous than noise. In high-recall research or discovery tasks, a larger candidate context may be justified before a later filtering or synthesis stage. In latency-sensitive production systems, stricter context selection may be preferable.\"},\"tunes\":{}},{\"id\":\"p-change-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The core principle would change only if models became reliably invariant to irrelevant information, position, contradiction, and stale evidence. Until then, context should be treated as a curated execution resource rather than passive storage.\"},\"tunes\":{}},{\"id\":\"h-limitations\",\"type\":\"header\",\"data\":{\"text\":\"Limitations\",\"level\":2},\"tunes\":{}},{\"id\":\"p-limit-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Long-context behaviour varies considerably across models and workloads. The original “Lost in the Middle” experiments used earlier generations of models, so their exact effect sizes should not be assumed to represent current systems. The finding remains useful as a failure pattern to test, not as a universal fixed performance curve.\"},\"tunes\":{}},{\"id\":\"p-limit-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Likewise, reducing context can remove necessary evidence. Compaction introduces summarization risk, and aggressive retrieval filtering can reduce recall. The objective is not minimal tokens at any cost; it is sufficient, current, traceable context for the decision being made.\"},\"tunes\":{}},{\"id\":\"h-conclusion\",\"type\":\"header\",\"data\":{\"text\":\"Conclusion\",\"level\":2},\"tunes\":{}},{\"id\":\"p-conclusion-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The question “How much context can the model accept?” is less useful than “How much of this context improves the decision?” More tokens can add evidence, but they can also add distraction, contradiction, stale state, positional fragility, and compression debt.\"},\"tunes\":{}},{\"id\":\"p-conclusion-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Treat context as an engineered working set. Start with minimum sufficient evidence. Add information only when it improves measured performance. Test noise, conflict, ordering, and compaction explicitly. A large context window is capacity; context quality is architecture.\"},\"tunes\":{}},{\"id\":\"h-faq\",\"type\":\"header\",\"data\":{\"text\":\"FAQ\",\"level\":2},\"tunes\":{}},{\"id\":\"faq\",\"type\":\"faq\",\"data\":{\"title\":\"Long context and AI answer quality\",\"items\":[{\"id\":\"faq1\",\"question\":\"Can giving an AI model more context make its answer worse?\",\"answer\":\"Yes. Additional context can dilute relevant evidence, introduce contradictory or stale information, move decisive evidence into less robust positions, and increase the chance that the model uses weak rather than decisive signals.\"},{\"id\":\"faq2\",\"question\":\"Does a larger context window eliminate the need for RAG?\",\"answer\":\"Not generally. A larger context window increases capacity, but retrieval still helps select current and relevant information, control cost, preserve source boundaries, and avoid sending large amounts of unrelated data into every request.\"},{\"id\":\"faq3\",\"question\":\"What is the Lost in the Middle problem?\",\"answer\":\"It describes observed cases where language models use relevant information less reliably when that information is located in the middle of a long context than when it appears near the beginning or end. The exact effect varies by model and task and should be tested on current systems.\"},{\"id\":\"faq4\",\"question\":\"Should I always reduce RAG top-k?\",\"answer\":\"No. If relevant evidence is missing from the candidate set, a larger top-k may improve recall. If the evidence is already present but gets diluted by additional material, increasing top-k can make the context worse. Diagnose retrieval and context assembly separately.\"},{\"id\":\"faq5\",\"question\":\"What should a context summary preserve?\",\"answer\":\"Preserve durable decisions, current goals, unresolved issues, identifiers, constraints, exceptions, evidence provenance, and the conditions that would change an earlier conclusion.\"}]},\"tunes\":{}},{\"id\":\"h-glossary\",\"type\":\"header\",\"data\":{\"text\":\"Glossary\",\"level\":2},\"tunes\":{}},{\"id\":\"glossary\",\"type\":\"glossary\",\"data\":{\"title\":\"Key context-engineering terms\",\"entries\":[{\"term\":\"Context window\",\"definition\":\"The amount of input and output token information a model can attend to within one inference sequence.\",\"anchor\":\"context-window\"},{\"term\":\"Context pollution\",\"definition\":\"Degradation caused by irrelevant, stale, redundant, conflicting, or otherwise low-value information occupying model context.\",\"anchor\":\"context-pollution\"},{\"term\":\"Signal dilution\",\"definition\":\"A reduction in the relative prominence of decisive evidence as additional low-value or competing information is added.\",\"anchor\":\"signal-dilution\"},{\"term\":\"Context compaction\",\"definition\":\"Reducing an accumulated context by summarizing, restructuring, externalizing, or otherwise preserving essential information in a smaller working representation.\",\"anchor\":\"context-compaction\"},{\"term\":\"Position robustness\",\"definition\":\"The degree to which model performance remains stable when relevant information appears in different positions inside the context.\",\"anchor\":\"position-robustness\"},{\"term\":\"Minimum sufficient context\",\"definition\":\"The smallest practical working context that still preserves the evidence, state, constraints, exceptions, and provenance required for reliable execution.\",\"anchor\":\"minimum-sufficient-context\"}]},\"tunes\":{}},{\"id\":\"h-sources\",\"type\":\"header\",\"data\":{\"text\":\"Primary sources and further reading\",\"level\":2},\"tunes\":{}},{\"id\":\"src-openai-session\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdevelopers.openai.com\u002Fcookbook\u002Fexamples\u002Fagents_sdk\u002Fsession_memory\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — Context Engineering: Short-Term Memory Management with Sessions\",\"description\":\"Guidance on trimming and compression, with discussion of distraction, inefficiency, stale context, noisy retrieval, and long-running sessions.\"}},\"tunes\":{}},{\"id\":\"src-anthropic-context\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Feffective-context-engineering-for-ai-agents\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Anthropic — Effective Context Engineering for AI Agents\",\"description\":\"Engineering guidance on context pollution, compaction, structured note-taking, and long-horizon agent context management.\"}},\"tunes\":{}},{\"id\":\"src-lost-middle\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Faclanthology.org\u002F2024.tacl-1.9\u002F\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Liu et al. — Lost in the Middle: How Language Models Use Long Contexts\",\"description\":\"TACL paper showing position-sensitive use of relevant information in long contexts and motivating explicit long-context robustness tests.\"}},\"tunes\":{}},{\"id\":\"src-ms-ace\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Fpublication\u002Fagentic-context-engineering-evolving-contexts-for-self-improving-language-models\u002F\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Microsoft Research — Agentic Context Engineering (ACE)\",\"description\":\"Research on evolving structured contexts while addressing brevity bias and context collapse.\"}},\"tunes\":{}},{\"id\":\"src-openai-evals\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fevaluation-best-practices\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — Evaluation Best Practices\",\"description\":\"Guidance on testing edge cases including long context and long-running conversations using explicit, repeatable evals.\"}},\"tunes\":{}}],\"version\":\"2.31.6\"}",{"time":838,"blocks":839,"version":1302},1790351629251,[840,844,848,853,858,862,866,870,874,878,906,910,914,918,923,927,931,935,939,943,947,951,955,959,963,967,971,975,979,983,1013,1018,1022,1026,1055,1059,1090,1094,1098,1102,1109,1113,1117,1121,1128,1132,1136,1140,1144,1148,1152,1159,1163,1178,1182,1186,1190,1194,1198,1202,1206,1210,1214,1218,1222,1242,1246,1267,1271,1277,1283,1289,1295],{"id":215,"data":841,"type":220,"tunes":843},{"title":842,"maxLevel":218,"minLevel":219},"Contents",{},{"id":223,"data":845,"type":226,"tunes":847},{"text":846},"A larger context window gives an AI system more capacity. It does not guarantee that the model will use that capacity well. In long conversations, RAG pipelines, research agents, and tool-heavy workflows, adding more history, more documents, more tool output, or more memory can make a response less reliable rather than more informed.",{},{"id":229,"data":849,"type":234,"tunes":852},{"body":850,"title":851,"variant":233},"\u003Cstrong>More context can make an AI answer worse when the additional information lowers the signal-to-noise ratio, introduces conflicts, hides decisive evidence, preserves stale state, or compresses away important conditions.\u003C\u002Fstrong> The relevant engineering objective is therefore not maximum context. It is \u003Cstrong>minimum sufficient context with preserved evidence and decision boundaries\u003C\u002Fstrong>.","Direct answer",{},{"id":237,"data":854,"type":234,"tunes":857},{"body":855,"title":856,"variant":241},"The Context Quality model and Context Pressure Test below are practical architecture methods proposed in this article, not formal industry standards. They synthesize established findings on long-context position effects, context pollution, compaction, retrieval, and context engineering.","About the model used in this article",{},{"id":244,"data":859,"type":42,"tunes":861},{"text":860,"level":219},"Context capacity is not context usability",{},{"id":249,"data":863,"type":226,"tunes":865},{"text":864},"A model's advertised context window describes how much input it can accept. It does not imply that every token inside that window receives equal attention or contributes equally to the final answer. The distinction matters because production systems increasingly fill context with conversation history, retrieved documents, tool results, memory, structured state, instructions, and intermediate artifacts.",{},{"id":254,"data":867,"type":226,"tunes":869},{"text":868},"The classic “Lost in the Middle” study showed that long-context models can perform worse when relevant evidence appears in the middle of a long input than when it appears near the beginning or end. The broader engineering lesson is not that long context is bad. It is that availability inside the context is not equivalent to reliable use.",{},{"id":259,"data":871,"type":226,"tunes":873},{"text":872},"OpenAI's context-management guidance reaches the same operational conclusion from another direction: even very large context windows can be overwhelmed by uncurated history, redundant tool output, and noisy retrieval. Anthropic likewise treats context as a finite resource that requires active engineering rather than passive accumulation.",{},{"id":264,"data":875,"type":42,"tunes":877},{"text":876,"level":219},"Five ways additional context can reduce answer quality",{},{"id":269,"data":879,"type":296,"tunes":905},{"content":880,"stretched":43,"withHeadings":14},[881,885,889,893,897,901],[882,883,884],"Failure mode","What changes when more context is added","Typical symptom",[886,887,888],"Signal dilution","Relevant evidence becomes a smaller fraction of the total input","The model gives a generic answer or misses the decisive passage",[890,891,892],"Evidence conflict","Different documents, versions, or memories disagree","The answer blends incompatible claims or chooses the wrong version",[894,895,896],"Position sensitivity","Decisive information moves into a less reliably used part of the context","The same evidence works in one ordering but fails in another",[898,899,900],"Stale-context persistence","Old state or prior conclusions remain present after reality changes","The model keeps repeating a formerly correct answer",[902,903,904],"Compression loss","Compaction or summarization removes qualifiers, exceptions, provenance, or unresolved uncertainty","The summary is coherent but the resulting answer becomes overconfident or overgeneralized",{},{"id":299,"data":907,"type":42,"tunes":909},{"text":908,"level":218},"1. Signal dilution: relevant evidence competes with everything else",{},{"id":304,"data":911,"type":226,"tunes":913},{"text":912},"Suppose a question can be answered from two short passages. A RAG system retrieves those passages plus eighteen loosely related ones “for safety.” Retrieval recall may improve, but the generator must now distinguish decisive evidence from background material. If similar phrases appear across several documents, the additional context can make the answer less precise.",{},{"id":309,"data":915,"type":226,"tunes":917},{"text":916},"This creates an important distinction between retrieval recall and context utility. More retrieved material can increase the probability that the answer exists somewhere in the context while simultaneously reducing the probability that the model gives the right evidence enough weight.",{},{"id":314,"data":919,"type":234,"tunes":922},{"body":920,"title":921,"variant":318},"Do not optimize top-k in isolation. Measure whether adding documents improves the final claim, preserves evidence attribution, and survives repeated trials.","Engineering rule",{},{"id":321,"data":924,"type":42,"tunes":926},{"text":925,"level":218},"2. Evidence conflict: more sources can mean more versions of reality",{},{"id":326,"data":928,"type":226,"tunes":930},{"text":929},"Long contexts often contain mutually inconsistent information: old and new API documentation, two policy versions, previous and current user preferences, competing web sources, cached state, or a model-generated summary that no longer matches the source.",{},{"id":331,"data":932,"type":226,"tunes":934},{"text":933},"The failure is not necessarily hallucination. The model may be faithfully combining contradictory evidence. The architecture therefore needs precedence rules: source authority, version, timestamp, jurisdiction, tenant, product revision, user state, or explicit supersession metadata.",{},{"id":336,"data":936,"type":226,"tunes":938},{"text":937},"Without those rules, increasing context can increase contradiction faster than it increases knowledge.",{},{"id":341,"data":940,"type":42,"tunes":942},{"text":941,"level":218},"3. Position sensitivity: where evidence appears can change the result",{},{"id":346,"data":944,"type":226,"tunes":946},{"text":945},"The “Lost in the Middle” results demonstrated that changing only the position of relevant information can materially change model performance. That finding is especially important for systems that concatenate many retrieved passages or long histories in a fixed order.",{},{"id":351,"data":948,"type":226,"tunes":950},{"text":949},"A production test should therefore vary document order, not merely test one canonical prompt. If the system answers correctly only when the decisive evidence is first or last, the application is more fragile than a single benchmark score suggests.",{},{"id":356,"data":952,"type":42,"tunes":954},{"text":953,"level":218},"4. Stale-context persistence: the model sees truth and history together",{},{"id":361,"data":956,"type":226,"tunes":958},{"text":957},"Long-running agents frequently carry earlier conclusions forward. That continuity is useful until a fact changes. If a tool result from yesterday says a deployment is healthy and a current tool result says it is degraded, both may remain in context unless the system explicitly replaces or scopes old state.",{},{"id":366,"data":960,"type":226,"tunes":962},{"text":961},"This is why current operational state should normally come from an authoritative source, while memory preserves durable context such as decisions, preferences, or procedures. More conversation history is not a substitute for re-reading the present.",{},{"id":371,"data":964,"type":42,"tunes":966},{"text":965,"level":218},"5. Compression loss: smaller context can also become worse context",{},{"id":376,"data":968,"type":226,"tunes":970},{"text":969},"The opposite intervention — compressing context — also has failure modes. Summaries can drop exceptions, unresolved questions, provenance, precise identifiers, negative evidence, or the conditions under which a conclusion was valid.",{},{"id":381,"data":972,"type":226,"tunes":974},{"text":973},"Microsoft Research's Agentic Context Engineering work describes a related problem as brevity bias and context collapse: iterative rewriting can remove useful domain detail. The objective is therefore not “compress as much as possible.” It is to reduce context while preserving the information that changes decisions.",{},{"id":386,"data":976,"type":42,"tunes":978},{"text":977,"level":219},"The Context Quality model",{},{"id":391,"data":980,"type":226,"tunes":982},{"text":981},"A useful context can be evaluated across six dimensions. None of them is simply token count.",{},{"id":396,"data":984,"type":435,"tunes":1012},{"rows":985,"title":1004,"layout":296,"columns":1005},[986,989,992,995,998,1001],{"id":400,"label":987,"values":988},"Relevance",[403,403,403],{"id":405,"label":990,"values":991},"Authority",[403,403,403],{"id":409,"label":993,"values":994},"Freshness",[403,403,403],{"id":413,"label":996,"values":997},"Consistency",[403,403,403],{"id":417,"label":999,"values":1000},"Decision completeness",[403,403,403],{"id":421,"label":1002,"values":1003},"Traceability",[403,403,403],"Six dimensions of context quality",[1006,1008,1010],{"id":427,"label":1007},"Dimension",{"id":430,"label":1009},"Question",{"id":433,"label":1011},"If weak",{},{"id":438,"data":1014,"type":234,"tunes":1017},{"body":1015,"title":1016,"variant":442},"The best context is not the largest context. It is the \u003Cstrong>smallest context that still preserves the evidence, constraints, state, exceptions, and provenance required for a reliable answer or action\u003C\u002Fstrong>.","Target state",{},{"id":445,"data":1019,"type":42,"tunes":1021},{"text":1020,"level":219},"The Context Pressure Test",{},{"id":450,"data":1023,"type":226,"tunes":1025},{"text":1024},"To determine whether an application benefits from more context, test context size as an experimental variable instead of assuming that larger is better.",{},{"id":455,"data":1027,"type":483,"tunes":1054},{"steps":1028,"title":1053,"orientation":482},[1029,1032,1035,1038,1041,1044,1047,1050],{"label":1030,"description":1031},"1. Define a gold case","Choose a task with a known answer and a known minimal evidence set.",{"label":1033,"description":1034},"2. Run minimal sufficient context","Provide only the instructions, current state, and evidence necessary for the answer.",{"label":1036,"description":1037},"3. Add relevant background","Add useful but non-decisive context and measure whether quality improves, stays stable, or falls.",{"label":1039,"description":1040},"4. Add realistic noise","Add loosely related history, tool output, or retrieved passages that a production system might include.",{"label":1042,"description":1043},"5. Add controlled conflicts","Introduce stale or contradictory evidence with clear version metadata and verify that the correct source still wins.",{"label":1045,"description":1046},"6. Reorder decisive evidence","Place the key information near the beginning, middle, and end to test position sensitivity.",{"label":1048,"description":1049},"7. Test compaction","Replace older context with a summary and verify that qualifiers, provenance, unresolved issues, and decision boundaries survive.",{"label":1051,"description":1052},"8. Compare the quality curve","Measure correctness, evidence use, consistency, latency, cost, and variance as context changes.","Context Pressure Test",{},{"id":486,"data":1056,"type":42,"tunes":1058},{"text":1057,"level":219},"What to measure instead of token count",{},{"id":491,"data":1060,"type":296,"tunes":1089},{"content":1061,"stretched":43,"withHeadings":14},[1062,1065,1068,1071,1074,1077,1080,1083,1086],[1063,1064],"Metric","What it reveals",[1066,1067],"Answer correctness","Whether the final result is right",[1069,1070],"Claim-level evidence support","Whether material claims remain grounded as context changes",[1072,1073],"Evidence utilization","Whether the answer follows the decisive evidence instead of prior model knowledge",[1075,1076],"Conflict resolution accuracy","Whether current \u002F authoritative evidence wins over stale or weaker sources",[1078,1079],"Position robustness","Whether reordering evidence changes correctness",[1081,1082],"Compaction retention","Whether summaries preserve constraints, exceptions, identifiers, provenance, and unresolved state",[1084,1085],"Output variance across trials","Whether additional context makes the system less stable",[1087,1088],"Latency and token cost","Whether the added information produces enough quality to justify its operational cost",{},{"id":523,"data":1091,"type":42,"tunes":1093},{"text":1092,"level":219},"RAG: why increasing top-k can hurt",{},{"id":528,"data":1095,"type":226,"tunes":1097},{"text":1096},"A common RAG tuning pattern is to increase top-k when the system misses an answer. This can improve candidate recall but also increase irrelevant context, duplicate evidence, outdated passages, and conflicting documents.",{},{"id":533,"data":1099,"type":226,"tunes":1101},{"text":1100},"The better question is whether the decisive evidence is missing from retrieval or merely losing influence after context assembly. If the correct passage already appears in the candidate set, increasing top-k may solve the wrong problem.",{},{"id":538,"data":1103,"type":544,"tunes":1108},{"url":1104,"title":1105,"excerpt":1106,"ctaLabel":1107},"https:\u002F\u002Fstajic.de\u002Fblog\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method","RAG Failed — But Which Layer Actually Failed? A Diagnostic Method","A layer-by-layer method for isolating source coverage, retrieval, ranking, context assembly, generation, evidence attribution, and freshness failures.","Read the RAG diagnostic method",{},{"id":547,"data":1110,"type":42,"tunes":1112},{"text":1111,"level":219},"Long-running agents: continuity is not accumulation",{},{"id":552,"data":1114,"type":226,"tunes":1116},{"text":1115},"An agent needs continuity across steps, but continuity does not require replaying every prior token. OpenAI demonstrates trimming and compression for long-running session context. Anthropic recommends compaction, structured note-taking, and other techniques to preserve useful information while controlling context pollution.",{},{"id":557,"data":1118,"type":226,"tunes":1120},{"text":1119},"A strong long-running architecture usually separates durable memory, current state, external artifacts, retrieval, and model-facing context. That allows the system to preserve what matters without forcing every historical detail into every inference.",{},{"id":562,"data":1122,"type":544,"tunes":1127},{"url":1123,"title":1124,"excerpt":1125,"ctaLabel":1126},"https:\u002F\u002Fstajic.de\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context","AI Agent Memory Is Not RAG: How to Separate Memory, Retrieval, State and Context","A practical four-layer architecture for separating what persists, what is authoritative now, what is retrieved, and what the model actually receives.","Read the memory architecture article",{},{"id":570,"data":1129,"type":42,"tunes":1131},{"text":1130,"level":219},"Context order should be intentional",{},{"id":575,"data":1133,"type":226,"tunes":1135},{"text":1134},"Context construction is an information architecture problem. Critical instructions, current state, decisive evidence, and task-specific constraints should not be placed arbitrarily. When systems concatenate sources mechanically, they implicitly delegate prioritization to positional effects and model attention.",{},{"id":580,"data":1137,"type":226,"tunes":1139},{"text":1138},"There is no universal best ordering for every model and task, so ordering should be evaluated empirically. A useful test suite randomizes or systematically varies document position and measures whether the same claim remains stable.",{},{"id":585,"data":1141,"type":42,"tunes":1143},{"text":1142,"level":219},"Preserve decision boundaries during compaction",{},{"id":590,"data":1145,"type":226,"tunes":1147},{"text":1146},"A summary that says “use approach X” is weaker than a summary that preserves why X was chosen and what would invalidate the decision. Context compaction should retain the variables that can change the answer: version, date, assumptions, state, authority, unresolved disagreement, and evidence provenance.",{},{"id":595,"data":1149,"type":226,"tunes":1151},{"text":1150},"This connects context engineering directly to answer validity. If compaction preserves a conclusion but removes its validity boundary, future responses can remain internally consistent while becoming externally wrong.",{},{"id":600,"data":1153,"type":544,"tunes":1158},{"url":1154,"title":1155,"excerpt":1156,"ctaLabel":1157},"https:\u002F\u002Fstajic.de\u002Fblog\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers","The Answer Validity Boundary: The Missing Layer Between Relevance and Reliable AI Answers","A framework for making explicit the conditions under which an AI claim applies and what changes require restriction, recalculation, or abandonment.","Read the Answer Validity Boundary",{},{"id":608,"data":1160,"type":42,"tunes":1162},{"text":1161,"level":219},"A practical context construction policy",{},{"id":613,"data":1164,"type":628,"tunes":1177},{"meta":1165,"items":1166,"style":627},{},[1167,1168,1169,1170,1171,1172,1173,1174,1175,1176],"Start from the current task, not from everything the system knows.","Re-read volatile state from authoritative systems before consequential decisions.","Retrieve evidence for the current question instead of carrying large static corpora forward.","Remove duplicate or low-value tool output.","Keep source version, timestamp, authority, and provenance with important evidence.","Make precedence explicit when current and historical information conflict.","Preserve rules together with their exceptions and prerequisites.","Store durable decisions and reusable procedures outside the immediate context when they do not need verbatim replay.","Compact history only with tests for constraint, identifier, exception, and provenance retention.","Evaluate context size, ordering, and noise with repeated trials rather than a single prompt.",{},{"id":631,"data":1179,"type":42,"tunes":1181},{"text":1180,"level":219},"What would change this answer?",{},{"id":636,"data":1183,"type":226,"tunes":1185},{"text":1184},"The trade-off changes with model architecture, training, task type, and context length. Future models may become substantially more robust to position, noise, and conflicting information. A task with a small clean corpus can also benefit from simply providing the complete source rather than building an elaborate retrieval pipeline.",{},{"id":641,"data":1187,"type":226,"tunes":1189},{"text":1188},"The recommendation also changes when omission is more dangerous than noise. In high-recall research or discovery tasks, a larger candidate context may be justified before a later filtering or synthesis stage. In latency-sensitive production systems, stricter context selection may be preferable.",{},{"id":646,"data":1191,"type":226,"tunes":1193},{"text":1192},"The core principle would change only if models became reliably invariant to irrelevant information, position, contradiction, and stale evidence. Until then, context should be treated as a curated execution resource rather than passive storage.",{},{"id":651,"data":1195,"type":42,"tunes":1197},{"text":1196,"level":219},"Limitations",{},{"id":656,"data":1199,"type":226,"tunes":1201},{"text":1200},"Long-context behaviour varies considerably across models and workloads. The original “Lost in the Middle” experiments used earlier generations of models, so their exact effect sizes should not be assumed to represent current systems. The finding remains useful as a failure pattern to test, not as a universal fixed performance curve.",{},{"id":661,"data":1203,"type":226,"tunes":1205},{"text":1204},"Likewise, reducing context can remove necessary evidence. Compaction introduces summarization risk, and aggressive retrieval filtering can reduce recall. The objective is not minimal tokens at any cost; it is sufficient, current, traceable context for the decision being made.",{},{"id":666,"data":1207,"type":42,"tunes":1209},{"text":1208,"level":219},"Conclusion",{},{"id":671,"data":1211,"type":226,"tunes":1213},{"text":1212},"The question “How much context can the model accept?” is less useful than “How much of this context improves the decision?” More tokens can add evidence, but they can also add distraction, contradiction, stale state, positional fragility, and compression debt.",{},{"id":676,"data":1215,"type":226,"tunes":1217},{"text":1216},"Treat context as an engineered working set. Start with minimum sufficient evidence. Add information only when it improves measured performance. Test noise, conflict, ordering, and compaction explicitly. A large context window is capacity; context quality is architecture.",{},{"id":681,"data":1219,"type":42,"tunes":1221},{"text":1220,"level":219},"FAQ",{},{"id":686,"data":1223,"type":686,"tunes":1241},{"items":1224,"title":1240},[1225,1228,1231,1234,1237],{"id":690,"answer":1226,"question":1227},"Yes. Additional context can dilute relevant evidence, introduce contradictory or stale information, move decisive evidence into less robust positions, and increase the chance that the model uses weak rather than decisive signals.","Can giving an AI model more context make its answer worse?",{"id":694,"answer":1229,"question":1230},"Not generally. A larger context window increases capacity, but retrieval still helps select current and relevant information, control cost, preserve source boundaries, and avoid sending large amounts of unrelated data into every request.","Does a larger context window eliminate the need for RAG?",{"id":698,"answer":1232,"question":1233},"It describes observed cases where language models use relevant information less reliably when that information is located in the middle of a long context than when it appears near the beginning or end. The exact effect varies by model and task and should be tested on current systems.","What is the Lost in the Middle problem?",{"id":702,"answer":1235,"question":1236},"No. If relevant evidence is missing from the candidate set, a larger top-k may improve recall. If the evidence is already present but gets diluted by additional material, increasing top-k can make the context worse. Diagnose retrieval and context assembly separately.","Should I always reduce RAG top-k?",{"id":706,"answer":1238,"question":1239},"Preserve durable decisions, current goals, unresolved issues, identifiers, constraints, exceptions, evidence provenance, and the conditions that would change an earlier conclusion.","What should a context summary preserve?","Long context and AI answer quality",{},{"id":712,"data":1243,"type":42,"tunes":1245},{"text":1244,"level":219},"Glossary",{},{"id":717,"data":1247,"type":717,"tunes":1266},{"title":1248,"entries":1249},"Key context-engineering terms",[1250,1253,1256,1258,1261,1263],{"term":1251,"anchor":723,"definition":1252},"Context window","The amount of input and output token information a model can attend to within one inference sequence.",{"term":1254,"anchor":727,"definition":1255},"Context pollution","Degradation caused by irrelevant, stale, redundant, conflicting, or otherwise low-value information occupying model context.",{"term":886,"anchor":731,"definition":1257},"A reduction in the relative prominence of decisive evidence as additional low-value or competing information is added.",{"term":1259,"anchor":735,"definition":1260},"Context compaction","Reducing an accumulated context by summarizing, restructuring, externalizing, or otherwise preserving essential information in a smaller working representation.",{"term":1078,"anchor":739,"definition":1262},"The degree to which model performance remains stable when relevant information appears in different positions inside the context.",{"term":1264,"anchor":743,"definition":1265},"Minimum sufficient context","The smallest practical working context that still preserves the evidence, state, constraints, exceptions, and provenance required for reliable execution.",{},{"id":747,"data":1268,"type":42,"tunes":1270},{"text":1269,"level":219},"Primary sources and further reading",{},{"id":752,"data":1272,"type":759,"tunes":1276},{"link":754,"meta":1273},{"image":1274,"title":757,"description":1275},{"url":403},"Guidance on trimming and compression, with discussion of distraction, inefficiency, stale context, noisy retrieval, and long-running sessions.",{},{"id":762,"data":1278,"type":759,"tunes":1282},{"link":764,"meta":1279},{"image":1280,"title":767,"description":1281},{"url":403},"Engineering guidance on context pollution, compaction, structured note-taking, and long-horizon agent context management.",{},{"id":771,"data":1284,"type":759,"tunes":1288},{"link":773,"meta":1285},{"image":1286,"title":776,"description":1287},{"url":403},"TACL paper showing position-sensitive use of relevant information in long contexts and motivating explicit long-context robustness tests.",{},{"id":780,"data":1290,"type":759,"tunes":1294},{"link":782,"meta":1291},{"image":1292,"title":785,"description":1293},{"url":403},"Research on evolving structured contexts while addressing brevity bias and context collapse.",{},{"id":789,"data":1296,"type":759,"tunes":1301},{"link":791,"meta":1297},{"image":1298,"title":1299,"description":1300},{"url":403},"OpenAI — Evaluation Best Practices","Guidance on testing edge cases including long context and long-running conversations using explicit, repeatable evals.",{},"2.31.6","A larger context window does not guarantee a better answer. This article explains how signal dilution, conflicting evidence, stale state, position sensitivity, and lossy compression can reduce AI reliability—and introduces a practical Context Pressure Test.",{"lang":7,"title":208,"content":210,"contentJson":1305,"excerpt":798},{"time":212,"blocks":1306,"version":797},[1307,1310,1313,1316,1319,1322,1325,1328,1331,1334,1344,1347,1350,1353,1356,1359,1362,1365,1368,1371,1374,1377,1380,1383,1386,1389,1392,1395,1398,1401,1421,1424,1427,1430,1442,1445,1458,1461,1464,1467,1470,1473,1476,1479,1482,1485,1488,1491,1494,1497,1500,1503,1506,1511,1514,1517,1520,1523,1526,1529,1532,1535,1538,1541,1544,1553,1556,1566,1569,1574,1579,1584,1589],{"id":215,"data":1308,"type":220,"tunes":1309},{"title":217,"maxLevel":218,"minLevel":219},{},{"id":223,"data":1311,"type":226,"tunes":1312},{"text":225},{},{"id":229,"data":1314,"type":234,"tunes":1315},{"body":231,"title":232,"variant":233},{},{"id":237,"data":1317,"type":234,"tunes":1318},{"body":239,"title":240,"variant":241},{},{"id":244,"data":1320,"type":42,"tunes":1321},{"text":246,"level":219},{},{"id":249,"data":1323,"type":226,"tunes":1324},{"text":251},{},{"id":254,"data":1326,"type":226,"tunes":1327},{"text":256},{},{"id":259,"data":1329,"type":226,"tunes":1330},{"text":261},{},{"id":264,"data":1332,"type":42,"tunes":1333},{"text":266,"level":219},{},{"id":269,"data":1335,"type":296,"tunes":1343},{"content":1336,"stretched":43,"withHeadings":14},[1337,1338,1339,1340,1341,1342],[273,274,275],[277,278,279],[281,282,283],[285,286,287],[289,290,291],[293,294,295],{},{"id":299,"data":1345,"type":42,"tunes":1346},{"text":301,"level":218},{},{"id":304,"data":1348,"type":226,"tunes":1349},{"text":306},{},{"id":309,"data":1351,"type":226,"tunes":1352},{"text":311},{},{"id":314,"data":1354,"type":234,"tunes":1355},{"body":316,"title":317,"variant":318},{},{"id":321,"data":1357,"type":42,"tunes":1358},{"text":323,"level":218},{},{"id":326,"data":1360,"type":226,"tunes":1361},{"text":328},{},{"id":331,"data":1363,"type":226,"tunes":1364},{"text":333},{},{"id":336,"data":1366,"type":226,"tunes":1367},{"text":338},{},{"id":341,"data":1369,"type":42,"tunes":1370},{"text":343,"level":218},{},{"id":346,"data":1372,"type":226,"tunes":1373},{"text":348},{},{"id":351,"data":1375,"type":226,"tunes":1376},{"text":353},{},{"id":356,"data":1378,"type":42,"tunes":1379},{"text":358,"level":218},{},{"id":361,"data":1381,"type":226,"tunes":1382},{"text":363},{},{"id":366,"data":1384,"type":226,"tunes":1385},{"text":368},{},{"id":371,"data":1387,"type":42,"tunes":1388},{"text":373,"level":218},{},{"id":376,"data":1390,"type":226,"tunes":1391},{"text":378},{},{"id":381,"data":1393,"type":226,"tunes":1394},{"text":383},{},{"id":386,"data":1396,"type":42,"tunes":1397},{"text":388,"level":219},{},{"id":391,"data":1399,"type":226,"tunes":1400},{"text":393},{},{"id":396,"data":1402,"type":435,"tunes":1420},{"rows":1403,"title":424,"layout":296,"columns":1416},[1404,1406,1408,1410,1412,1414],{"id":400,"label":401,"values":1405},[403,403,403],{"id":405,"label":406,"values":1407},[403,403,403],{"id":409,"label":410,"values":1409},[403,403,403],{"id":413,"label":414,"values":1411},[403,403,403],{"id":417,"label":418,"values":1413},[403,403,403],{"id":421,"label":422,"values":1415},[403,403,403],[1417,1418,1419],{"id":427,"label":428},{"id":430,"label":431},{"id":433,"label":434},{},{"id":438,"data":1422,"type":234,"tunes":1423},{"body":440,"title":441,"variant":442},{},{"id":445,"data":1425,"type":42,"tunes":1426},{"text":447,"level":219},{},{"id":450,"data":1428,"type":226,"tunes":1429},{"text":452},{},{"id":455,"data":1431,"type":483,"tunes":1441},{"steps":1432,"title":447,"orientation":482},[1433,1434,1435,1436,1437,1438,1439,1440],{"label":459,"description":460},{"label":462,"description":463},{"label":465,"description":466},{"label":468,"description":469},{"label":471,"description":472},{"label":474,"description":475},{"label":477,"description":478},{"label":480,"description":481},{},{"id":486,"data":1443,"type":42,"tunes":1444},{"text":488,"level":219},{},{"id":491,"data":1446,"type":296,"tunes":1457},{"content":1447,"stretched":43,"withHeadings":14},[1448,1449,1450,1451,1452,1453,1454,1455,1456],[495,496],[498,499],[501,502],[504,505],[507,508],[510,511],[513,514],[516,517],[519,520],{},{"id":523,"data":1459,"type":42,"tunes":1460},{"text":525,"level":219},{},{"id":528,"data":1462,"type":226,"tunes":1463},{"text":530},{},{"id":533,"data":1465,"type":226,"tunes":1466},{"text":535},{},{"id":538,"data":1468,"type":544,"tunes":1469},{"url":540,"title":541,"excerpt":542,"ctaLabel":543},{},{"id":547,"data":1471,"type":42,"tunes":1472},{"text":549,"level":219},{},{"id":552,"data":1474,"type":226,"tunes":1475},{"text":554},{},{"id":557,"data":1477,"type":226,"tunes":1478},{"text":559},{},{"id":562,"data":1480,"type":544,"tunes":1481},{"url":564,"title":565,"excerpt":566,"ctaLabel":567},{},{"id":570,"data":1483,"type":42,"tunes":1484},{"text":572,"level":219},{},{"id":575,"data":1486,"type":226,"tunes":1487},{"text":577},{},{"id":580,"data":1489,"type":226,"tunes":1490},{"text":582},{},{"id":585,"data":1492,"type":42,"tunes":1493},{"text":587,"level":219},{},{"id":590,"data":1495,"type":226,"tunes":1496},{"text":592},{},{"id":595,"data":1498,"type":226,"tunes":1499},{"text":597},{},{"id":600,"data":1501,"type":544,"tunes":1502},{"url":602,"title":603,"excerpt":604,"ctaLabel":605},{},{"id":608,"data":1504,"type":42,"tunes":1505},{"text":610,"level":219},{},{"id":613,"data":1507,"type":628,"tunes":1510},{"meta":1508,"items":1509,"style":627},{},[617,618,619,620,621,622,623,624,625,626],{},{"id":631,"data":1512,"type":42,"tunes":1513},{"text":633,"level":219},{},{"id":636,"data":1515,"type":226,"tunes":1516},{"text":638},{},{"id":641,"data":1518,"type":226,"tunes":1519},{"text":643},{},{"id":646,"data":1521,"type":226,"tunes":1522},{"text":648},{},{"id":651,"data":1524,"type":42,"tunes":1525},{"text":653,"level":219},{},{"id":656,"data":1527,"type":226,"tunes":1528},{"text":658},{},{"id":661,"data":1530,"type":226,"tunes":1531},{"text":663},{},{"id":666,"data":1533,"type":42,"tunes":1534},{"text":668,"level":219},{},{"id":671,"data":1536,"type":226,"tunes":1537},{"text":673},{},{"id":676,"data":1539,"type":226,"tunes":1540},{"text":678},{},{"id":681,"data":1542,"type":42,"tunes":1543},{"text":683,"level":219},{},{"id":686,"data":1545,"type":686,"tunes":1552},{"items":1546,"title":709},[1547,1548,1549,1550,1551],{"id":690,"answer":691,"question":692},{"id":694,"answer":695,"question":696},{"id":698,"answer":699,"question":700},{"id":702,"answer":703,"question":704},{"id":706,"answer":707,"question":708},{},{"id":712,"data":1554,"type":42,"tunes":1555},{"text":714,"level":219},{},{"id":717,"data":1557,"type":717,"tunes":1565},{"title":719,"entries":1558},[1559,1560,1561,1562,1563,1564],{"term":722,"anchor":723,"definition":724},{"term":726,"anchor":727,"definition":728},{"term":730,"anchor":731,"definition":732},{"term":734,"anchor":735,"definition":736},{"term":738,"anchor":739,"definition":740},{"term":742,"anchor":743,"definition":744},{},{"id":747,"data":1567,"type":42,"tunes":1568},{"text":749,"level":219},{},{"id":752,"data":1570,"type":759,"tunes":1573},{"link":754,"meta":1571},{"image":1572,"title":757,"description":758},{"url":403},{},{"id":762,"data":1575,"type":759,"tunes":1578},{"link":764,"meta":1576},{"image":1577,"title":767,"description":768},{"url":403},{},{"id":771,"data":1580,"type":759,"tunes":1583},{"link":773,"meta":1581},{"image":1582,"title":776,"description":777},{"url":403},{},{"id":780,"data":1585,"type":759,"tunes":1588},{"link":782,"meta":1586},{"image":1587,"title":785,"description":786},{"url":403},{},{"id":789,"data":1590,"type":759,"tunes":1593},{"link":791,"meta":1591},{"image":1592,"title":794,"description":795},{"url":403},{},"Post erfolgreich abgerufen",{"items":1596,"source":1660,"manualIds":1661,"manualMatchedIds":1662},[1597,1604,1611,1618,1625,1632,1639,1646,1653],{"id":1598,"slug":1599,"title":1600,"excerpt":1601,"featuredImage":1602,"publishedAt":1603},"364","tipps-fuer-die-verbesserung-der-seo-suchmaschinenoptimierung","Ovladavanje SEO radnim tokom: Ključne strategije optimizacije za organski rast","Strukturiran SEO tok posla je ključan za održiv organski rast. Naučite deset osnovnih strategija, od istraživanja ključnih reči i tehničke optimizacije do kvaliteta sadržaja i analize performansi.","\u002Fuploads\u002F2026\u002F03\u002Ftipps-fuer-die-verbesserung-der-seo-suchmaschinenoptimierung-1774866098131-hwkzrg.webp","2024-01-26T06:35:00.000Z",{"id":1605,"slug":1606,"title":1607,"excerpt":1608,"featuredImage":1609,"publishedAt":1610},"467","the-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers","Granica valjanosti odgovora: Nedostajući sloj između relevantnosti i pouzdanih AI odgovora","Izvor može biti relevantan, autoritativan i ipak pogrešan za pitanje koje se postavlja. Sloj koji nedostaje je primenljivost: uslovi pod kojima odgovor važi i promene koje ga primoravaju na preispitivanje. Ovaj članak predstavlja Granicu važenja odgovora kao obrazac za dizajn izvora za ljude, AI pretragu i RAG sisteme.","\u002Fuploads\u002F2026\u002F09\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers-1790272901306-1g5jly.webp","2026-09-24T11:59:00.000Z",{"id":1612,"slug":1613,"title":1614,"excerpt":1615,"featuredImage":1616,"publishedAt":1617},"471","how-to-know-whether-an-ai-agent-actually-used-the-right-evidence","Kako znati da li je AI agent zaista koristio prave dokaze","AI agent može citirati izvore i ipak koristiti pogrešne dokaze. Ovaj članak predstavlja praktičnu metodu za proveru potkrepljenosti tvrdnji, autoriteta izvora, primenjivosti, porekla i toga da li su dokazi zaista uticali na odgovor.","\u002Fuploads\u002F2026\u002F09\u002Fhow-to-know-whether-an-ai-agent-actually-used-the-right-evidence-1790351317188-o5z9ve.webp","2026-09-25T11:47:00.000Z",{"id":1619,"slug":1620,"title":1621,"excerpt":1622,"featuredImage":1623,"publishedAt":1624},"477","computer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system","Agenti za korišćenje računara: Zašto uspešan demo i dalje može biti nepouzdan sistem","Agenti za korišćenje računara sada mogu da završe impresivne radne tokove u pregledaču i na radnoj površini, ali jedno uspešno izvršavanje dokazuje sposobnost—ne pouzdanost. Ovaj članak pokazuje kako testirati ponovljivost, robusnost u odnosu na okruženje, kontrolu dugog horizonta, svest o stanju, verifikaciju ishoda i bezbedno upravljanje ciljevima.","\u002Fuploads\u002F2026\u002F09\u002Fcomputer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system-1790352854690-75qnrg.webp","2026-09-25T12:13:00.000Z",{"id":1626,"slug":1627,"title":1628,"excerpt":1629,"featuredImage":1630,"publishedAt":1631},"466","the-gpu-is-not-the-product-future-proof-private-ai-architecture","GPU nije proizvod: Privatna AI arhitektura spremna za budućnost","Privatna AI infrastruktura ne bi trebalo da bude projektovana oko jednog GPU-a ili jednog modela. Otporniji pristup kombinuje brze GPU-ove za inferenciju, memorijski bogate AI sisteme, čvorove za fizički AI i opcione vodeće modele u oblaku iza sloja za rutiranje koji prepoznaje mogućnosti.","\u002Fuploads\u002F2026\u002F09\u002Fthe-gpu-is-not-the-product-future-proof-private-ai-architecture-1790140878812-8hsl39.webp","2026-09-23T01:19:00.000Z",{"id":1633,"slug":1634,"title":1635,"excerpt":1636,"featuredImage":1637,"publishedAt":1638},"457","should-you-buy-5g-openwrt-router-old-firmware","Treba li kupiti 5G OpenWrt ruter sa starim firmverom? ZBT Z8102AX kao praktičan primer","Kupovina 5G OpenWrt rutera sa starijim firmverom može imati smisla, ali samo pod pravim uslovima. ZBT Z8102AX jasno pokazuje obe strane: hardver je koristan, modem radi, a ruter je ostao stabilan u testiranju, ali OpenWrt 21.02, slabo pakovanje i nejasni putevi nadogradnje zahtevaju pažljivu odluku o kupovini.","\u002Fuploads\u002F2026\u002F06\u002Fopenwrt-router-review-dual-sim-05-1781620596218-5ldld4.webp","2026-06-16T10:41:00.000Z",{"id":1640,"slug":1641,"title":1642,"excerpt":1643,"featuredImage":1644,"publishedAt":1645},"363","front-und-backend-entwicklung","Frontend i Backend Razvoj","Front-end i back-end razvoj je suštinski deo veb razvoja i obuhvata kreiranje veb aplikacija i veb-sajtova. Front-end razvoj se fokusira na korisnički interfejs, dok je back-end razvoj odgovoran za programiranje i upravljanje serverskom stranom.","\u002Fuploads\u002F2026\u002F03\u002Ffront-und-backend-entwicklung-1774872219531-wyu4i1.webp","2023-04-12T11:11:00.000Z",{"id":1647,"slug":1648,"title":1649,"excerpt":1650,"featuredImage":1651,"publishedAt":1652},"468","ai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context","Memorija AI agenta nije RAG: Kako razdvojiti memoriju, pronalaženje, stanje i kontekst","Memorija agenta, RAG, stanje i kontekst često se koriste kao da su međusobno zamenjivi. Oni to nisu. Ovaj praktični arhitektonski model razdvaja ova četiri sloja, pokazuje gde svaki pripada i objašnjava šta se kvari kada ih sistemi stope u jedno.","\u002Fuploads\u002F2026\u002F09\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context-1790350560308-np0xy6.webp","2026-09-25T11:34:00.000Z",{"id":1654,"slug":1655,"title":1656,"excerpt":1657,"featuredImage":1658,"publishedAt":1659},"478","what-is-rag-the-simplest-explanation-of-how-it-works","Šta je RAG? Najjednostavnije objašnjenje kako funkcioniše","RAG zvuči komplikovano, ali ideja je jednostavna: pre nego što AI odgovori, prvo potraži korisne informacije iz izvora znanja i daje te informacije jezičkom modelu. Ovaj vodič objašnjava RAG, LLM-ove, stanje, memoriju i alate koristeći jedan jednostavan mentalni model.","\u002Fuploads\u002F2026\u002F09\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works-1790377492124-khjagt.webp","2026-09-25T19:03:00.000Z","fallback",[],[]]