[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"portal-settings:stajic:sr":3,"public-menus:all":38,"post:where-does-an-llm-get-its-data-rag-data-sources-in-python:sr":205,"related:post:where-does-an-llm-get-its-data-rag-data-sources-in-python:sr:1":1406},{"statusCode":4,"data":5,"message":37},200,{"tenantId":6,"lang":7,"defaultLang":8,"siteUrl":9,"contactEmail":10,"brandName":11,"logoUrl":12,"siteName":11,"siteDescription":13,"ogImage":10,"robotsIndex":14,"socialLinks":10,"reservedSlugs":10,"seoPolicy":15},"stajic","sr","de","https:\u002F\u002Fstajic.de",null,"Stajic Platform","\u002FLogo_Planet.svg","Stajic Portal",true,{"branding":16,"relatedContent":17,"crossDomainLinks":18},{"logoUrl":12},{"enabled":14},[19,22,25,28,31,34],{"url":20,"label":21,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Ffigure.rocks","figure.rocks",{"url":23,"label":24,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Floving.rocks","loving.rocks",{"url":26,"label":27,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.com","bazify.com",{"url":29,"label":30,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.de","bazify.de",{"url":32,"label":33,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.at","bazify.at",{"url":35,"label":36,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.ba","bazify.ba","Portal settings resolved",[39,45],{"id":40,"name":41,"location":42,"isActive":14,"isDefault":43,"items":44},1,"main-navigation","header",false,[],{"id":46,"name":47,"location":48,"isActive":14,"isDefault":14,"items":49},4,"main-menu","sidebar",[50,66,79,93,103,118,133],{"id":51,"title":52,"url":60,"target":61,"icon":62,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":64,"portfolioId":10,"children":65},"item-18",{"de":53,"en":54,"es":55,"fr":56,"it":54,"ru":57,"sr":58,"zh":59},"Startseite","Home","Inicio","Accueil","Главная","Почетна","首页","\u002Ffull-stack-web-developer-munich-performance-seo-and-maintainable-builds","_self","i-lucide-home","page",111,[],{"id":67,"title":68,"url":75,"target":61,"icon":76,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":77,"portfolioId":10,"children":78},"item-22",{"de":69,"en":69,"es":70,"fr":69,"it":71,"ru":72,"sr":73,"zh":74},"Vision","Visión","Visione","Видение","Визија","想象","\u002Fueber-uns-webdesign-muenchen-webaplikation","i-lucide-eye",113,[],{"id":80,"title":81,"url":89,"target":61,"icon":90,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":91,"portfolioId":10,"children":92},"item-19",{"de":82,"en":83,"es":84,"fr":83,"it":85,"ru":86,"sr":87,"zh":88},"Leistungen","Services","Servicios","Servizi","Услуги","Услуге","服务","\u002Fservices-dienstleistungen-muenchen","i-lucide-wrench",116,[],{"id":94,"title":95,"url":99,"target":61,"icon":100,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":101,"portfolioId":10,"children":102},"item-23",{"de":96,"en":96,"es":96,"fr":96,"it":96,"ru":97,"sr":97,"zh":98},"Blog","Блог","博客","\u002Fblog","i-lucide-book-open",112,[],{"id":104,"title":105,"url":114,"target":61,"icon":115,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":116,"portfolioId":10,"children":117},"item-32",{"de":106,"en":107,"es":108,"fr":109,"it":110,"ru":111,"sr":112,"zh":113},"Neue Technologien","New Technologies","Nuevas tecnologías","Nouvelles technologies","Nuove tecnologie","Новые технологии","Нове технологије","新技术！","\u002Fneue-webtechnologien","i-lucide-sparkles",122,[],{"id":119,"title":120,"url":129,"target":61,"icon":130,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":131,"portfolioId":10,"children":132},"item-20",{"de":121,"en":122,"es":123,"fr":124,"it":125,"ru":126,"sr":127,"zh":128},"Kontakt","Contact us!","Contacto","Contact","Contatto","Контакт","Контактирајте нас","联系我们！","\u002Fcontact","i-lucide-mail",115,[],{"id":134,"title":135,"url":144,"target":61,"icon":145,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":147},"item-21",{"de":136,"en":137,"es":138,"fr":139,"it":140,"ru":141,"sr":142,"zh":143},"Unsere Arbeit","Our Work","Nuestro trabajo","Nos réalisations","I nostri lavori","Наши работы","Наши радови","文件夹","\u002Fportfolio","i-lucide-briefcase",114,[148,161,175,181,193],{"id":149,"title":150,"url":144,"target":61,"icon":159,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":160},"item-24",{"de":151,"en":152,"es":153,"fr":154,"it":155,"ru":156,"sr":157,"zh":158},"Alle Projekte","All Projects","Todos los proyectos","Tous les projets","Tutti i progetti","Все проекты","Сви пројекти","所有项目","i-lucide-grid-3x3",[],{"id":162,"title":163,"url":171,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":174},"item-29",{"de":164,"en":165,"es":166,"fr":167,"it":168,"ru":169,"sr":170,"zh":143},"Local Roots, Global Reach","Local Roots - Global Reach","Empresa local ","Entreprise locale","Azienda locale","Местная компания","Локално предузеће глобално тржиште","\u002Fportfolio\u002Flocal-roots-global-reach-communication-media-systems-for-modern-business","i-lucide-folder","custom",[],{"id":176,"title":177,"url":179,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":180},"item-28",{"de":178,"en":178,"es":178,"fr":178,"it":178,"ru":178,"sr":178,"zh":178},"Solr Suggester","\u002Fportfolio\u002Fsolr-fuzzy-suggester-und-solr-infix-suggester-abfrage-ueber-ajax-und-filterung",[],{"id":182,"title":183,"url":191,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":192},"item-27",{"de":184,"en":185,"es":186,"fr":187,"it":188,"ru":189,"sr":190,"zh":185},"Firmenwebseite SEO","Company Website SEO","Sitio web corporativo SEO","Site web d’entreprise SEO","Sito web aziendale SEO","Корпоративный сайт SEO","Пословна веб-страница SEO","\u002Fportfolio\u002Fseo-sem-branding-mobile-webseite-muenchen",[],{"id":194,"title":195,"url":203,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":204},"item-31",{"de":196,"en":197,"es":198,"fr":199,"it":200,"ru":201,"sr":202,"zh":197},"Digitalisierungsportal","Digitalization Portal","Portal de digitalización","Portail de numérisation","Portale di digitalizzazione","Портал цифровизации","Портал за дигитализацију","\u002Fportfolio\u002Fdigitalisierungsportal-archiv-museum-bibliothek-ead-lido-mets-mods",[],{"statusCode":4,"data":206,"message":1405},{"id":207,"title":208,"slug":209,"content":210,"contentJson":211,"excerpt":677,"featuredImage":678,"featuredImageAlt":679,"featuredImageCaption":10,"featuredImageTitle":10,"featuredImageCopyright":10,"featuredImageAuthor":10,"featuredImageSourceUrl":10,"featuredImageLicense":10,"featuredImageIsAiGenerated":43,"status":680,"publishedAt":681,"createdAt":682,"updatedAt":683,"seoLocalePaths":684,"categories":693,"author":694,"translations":699},"479","Odakle LLM dobija svoje podatke? RAG izvori podataka u Python-u","where-does-an-llm-get-its-data-rag-data-sources-in-python","\u003Cp>Prethodni članak, \u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fsr\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works\">Šta je RAG? Najjednostavnije objašnjenje kako funkcioniše\u003C\u002Fa>, uspostavio je mentalni model: LLM piše, RAG pronalazi korisno znanje, aplikacija poseduje trenutno stanje, a alati izvršavaju radnje. Ovaj članak čini sledeći korak: \u003Cb>odakle podaci zapravo dolaze i kako pretraga izgleda u Python-u?\u003C\u002Fb>\u003C\u002Fp>\n\u003Cp>Važno iznenađenje je da „izvor podataka za LLM“ obično nije ništa egzotično. To može biti tekstualni fajl, fascikla sa Markdown dokumentima, SQL baza podataka, odgovor API-ja, katalog proizvoda, sistem za podršku ili vektorski indeks izveden iz tih izvora. AI ne poznaje te sisteme magično. Vaša aplikacija mora da učita, upita, pretraži ili pronađe relevantne podatke i smesti rezultat u kontekst modela.\u003C\u002Fp>\n\u003Cblockquote class=\"border-l-4 border-gray-300 pl-4 italic\">Izvor podataka = gde informacije žive. Pretraga = kako aplikacija pronalazi korisne informacije. Kontekst = izabrane informacije date modelu. LLM = komponenta koja tumači taj kontekst i generiše odgovor.\u003Ccite class=\"block mt-2 text-sm\">— Model od četiri dela koji se koristi u ovom članku\u003C\u002Fcite>\u003C\u002Fblockquote>\n\u003Ch2 id=\"section-4\">Pitanje\u003C\u002Fh2>\n\u003Cp>Kako LLM koristi eksterne podatke kao što su fajlovi, baze podataka ili API-ji, i kako mali Python program može da implementira osnovne RAG korake bez skrivanja iza okvira?\u003C\u002Fp>\n\u003Ch2 id=\"section-6\">Šta to zapravo znači\u003C\u002Fh2>\n\u003Cp>Kada programeri kažu da je LLM „povezan sa podacima kompanije“, nekoliko različitih operacija može biti skriveno iza te rečenice. Jedna aplikacija može izvršavati SQL. Druga može pozivati API. Treća može pokretati pretragu punog teksta. Četvrta može izračunavati sličnost ugrađivanja nad delovima dokumenata. Sve one mogu pružiti eksterne informacije LLM-u, ali nisu ista metoda pretrage i ne treba ih tretirati kao međusobno zamenljive.\u003C\u002Fp>\n\u003Cp>Ova razlika je važna jer najbolja metoda pretrage zavisi od oblika pitanja. „Koja je naša politika povraćaja?“ je problem pretrage dokumenata. „Koji je trenutni status porudžbine 4711?“ je obično strukturirano pretraživanje baze podataka. „Koji pasus govori o oporavku naloga?“ može biti pretraga po ključnim rečima ili semantička pretraga. RAG je najkorisniji kada sistem mora da \u003Cb>otkrije relevantno znanje pre generisanja\u003C\u002Fb>.\u003C\u002Fp>\n\u003Ch2 id=\"section-9\">Najjednostavniji primer\u003C\u002Fh2>\n\u003Cp>Počnite sa tri stringa u običnom Python-u. Nema vektorske baze podataka, nema okvira, a još nema ni LLM-a. Želimo samo da učinimo korak pretrage vidljivim.\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>documents = [\n    &quot;The AKM uses 7.62 mm ammunition.&quot;,\n    &quot;A Med Kit restores health.&quot;,\n    &quot;A 4x scope can be attached to several compatible weapons.&quot;\n]\n\nquestion = &quot;Which ammunition does the AKM use?&quot;\n\nfor document in documents:\n    if &quot;AKM&quot; in document:\n        print(document)\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>Program štampa prvu rečenicu jer sadrži termin koji smo tražili. Ovo je primitivna pretraga, ali arhitektura je već vidljiva: \u003Cb>pitanje → pretraga → relevantan tekst\u003C\u002Fb>. RAG dodaje još jedan važan korak: prosledite pronađeni tekst jezičkom modelu zajedno sa pitanjem.\u003C\u002Fp>\n\u003Cp>Malo opštija verzija rangira dokumente prema preklapanju termina upita:\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>import re\n\ndocuments = [\n    {&quot;id&quot;: &quot;weapon-akm&quot;, &quot;text&quot;: &quot;The AKM uses 7.62 mm ammunition.&quot;},\n    {&quot;id&quot;: &quot;healing-medkit&quot;, &quot;text&quot;: &quot;A Med Kit restores health.&quot;},\n    {&quot;id&quot;: &quot;scope-4x&quot;, &quot;text&quot;: &quot;A 4x scope can be attached to several compatible weapons.&quot;},\n]\n\ndef words(text):\n    return set(re.findall(r&quot;[a-zA-Z0-9.]+&quot;, text.lower()))\n\ndef retrieve(question, documents, top_k=2):\n    query_terms = words(question)\n    ranked = []\n\n    for document in documents:\n        score = len(query_terms &amp; words(document[&quot;text&quot;]))\n        if score &gt; 0:\n            ranked.append((score, document))\n\n    ranked.sort(key=lambda item: item[0], reverse=True)\n    return [document for _, document in ranked[:top_k]]\n\nquestion = &quot;Which ammunition does the AKM use?&quot;\nhits = retrieve(question, documents)\n\nfor hit in hits:\n    print(hit[&quot;id&quot;], &quot;-&gt;&quot;, hit[&quot;text&quot;])\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>Ovo nije produkcioni pretraživač. Zanemaruje morfologiju, sinonime, pravopisne varijante, dužinu dokumenta i mnoge signale rangiranja. Njegova vrednost je edukativna: \u003Cb>RAG ne počinje vektorskom bazom podataka. Počinje pretragom.\u003C\u002Fb>\u003C\u002Fp>\n\u003Ch2 id=\"section-16\">Gde primer prestaje da funkcioniše\u003C\u002Fh2>\n\u003Cp>Tačno ili leksičko podudaranje postaje slabo kada pitanje i izvor koriste različite reči. Dokument može da kaže „održavanje vozila“, dok korisnik pita „kako da popravim svoj auto?“ Leksički pretraživač može da propusti vezu iako je čovek odmah vidi. Semantička pretraga se bavi ovim tako što predstavlja tekst kao vektore i poredi značenje, a ne samo tačne tokene.\u003C\u002Fp>\n\u003Cp>Dugački fajlovi stvaraju još jedan problem. Pretraga celog priručnika od 80 stranica kao jedne celine je previše gruba, ali deljenje svake rečenice može uništiti koristan kontekst. Pravi RAG sistemi stoga zahtevaju odluke o parsiranju, deljenju na delove, metapodacima, rangiranju, svežini, dozvolama i poreklu.\u003C\u002Fp>\n\u003Cp>Primer takođe ne govori ništa o strukturisanim aktuelnim činjenicama. Ako korisnik pita za trenutni status porudžbine 4711 i aplikacija već ima ključ baze podataka, semantička pretraga je obično pogrešan prvi alat. Deterministički upit baze podataka je bolji.\u003C\u002Fp>\n\u003Ch2 id=\"section-20\">Direktan odgovor\u003C\u002Fh2>\n\u003Cp>LLM izvor podataka je svaki eksterni sistem iz kojeg aplikacija može da dobije informacije za model: fajlovi, baze podataka, API-ji, indeksi pretrage, vektorski skladišta ili stanje aplikacije uživo. RAG je obrazac \u003Cb>preuzimanja relevantnog znanja iz takvih izvora pre generisanja\u003C\u002Fb>.\u003C\u002Fp>\n\u003Cp>U Python-u, suštinski pipeline može biti veoma mali: \u003Cb>učitaj podatke → kreiraj jedinice za preuzimanje → pronađi relevantne dokaze → sastavi kontekst → pozovi LLM\u003C\u002Fb>. Metoda preuzimanja treba da odgovara izvoru i pitanju. Koristite SQL za tačne strukturisane činjenice, punotekstualnu pretragu za leksičko poklapanje, embedding-e za semantičku sličnost i hibridno preuzimanje kada je nekoliko signala vredno.\u003C\u002Fp>\n\u003Ch2 id=\"section-23\">Zašto je to tako\u003C\u002Fh2>\n\u003Cp>Jezički model ne prima automatski sadržaj vašeg fajl sistema, PostgreSQL baze podataka, CRM-a, privatnog API-ja ili novoizmenjenog dokumenta. Aplikacija odlučuje koje eksterne informacije su dostupne i šta se stavlja u trenutni kontekst modela.\u003C\u002Fp>\n\u003Cp>Originalni rad Retrieval-Augmented Generation od Lewis et al. kombinovao je generativni model sa eksternom neparametarskom memorijom preuzetom iz gustog vektorskog indeksa. Šira arhitektonska ideja nadživljava tu specifičnu implementaciju: eksterni dokazi mogu se preuzeti u vreme inferencije umesto da se očekuje da sve korisno znanje bude kodirano u parametrima modela.\u003C\u002Fp>\n\u003Cp>Ovo stvara korisnu podelu odgovornosti: izvor čuva informacije, pretraživač bira dokaze, kontekst nosi te dokaze u zahtev, a model ih interpretira. Održavanje tih granica vidljivim čini kvarove mnogo lakšim za dijagnostikovanje.\u003C\u002Fp>\n\u003Ch2 id=\"section-27\">Kontekst: Glavni tipovi izvora podataka\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Izvor\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Tipična metoda preuzimanja\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Dobro za\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">TXT \u002F Markdown \u002F HTML\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Parsiranje + leksička ili semantička pretraga\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Dokumentacija, priručnici, članci, beleške\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">PDF \u002F DOCX\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Ekstrakcija svesna strukture + pretraga\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Politike, izveštaji, ugovori, priručnici\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">SQL baza podataka\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">SQL upit ili filtrirano preuzimanje\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Porudžbine, korisnici, proizvodi, strukturisani zapisi\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">REST \u002F GraphQL API\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">HTTP zahtev sa parametrima\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Udaljeni sistemi i podaci servisa uživo\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Indeks pretrage\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">BM25 \u002F punotekstualna \u002F hibridna pretraga\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Velike tekstualne kolekcije\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Vektorski indeks\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Sličnost embedding-a\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Semantičko preuzimanje dokumenata\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Stanje aplikacije\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Direktno čitanje stanja ili poziv alata\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Šta je istinito upravo sada\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Cp>Vektorski indeks zaslužuje posebnu pažnju. U mnogim arhitekturama on \u003Cb>nije kanonski izvor istine\u003C\u002Fb>. To je indeks za preuzimanje izveden iz dokumenata ili zapisa. Autoritativni dokument može živeti u objektnom skladištu, CMS-u, Git-u, PostgreSQL-u ili drugom sistemu, dok se embedding-i i metapodaci čuvaju odvojeno za brzu semantičku pretragu. Neki sistemi koriste vektorsko skladište kao primarno skladište, ali to je arhitektonski izbor, a ne zahtev RAG-a.\u003C\u002Fp>\n\u003Cp>Ako je granica između preuzimanja, trajne memorije, trenutnog stanja i konteksta modela još uvek nejasna, pogledajte \u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fsr\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context\">AI Agent Memory Is Not RAG\u003C\u002Fa>. Ti slojevi mogu koristiti neke od istih tehnologija skladištenja, a ipak imati različita pravila ispravnosti.\u003C\u002Fp>\n\u003Ch2 id=\"section-31\">Pretpostavke\u003C\u002Fh2>\n\u003Cul>\u003Cli>Aplikaciji je dozvoljen pristup eksternom izvoru.\u003C\u002Fli>\u003Cli>Relevantni izvor sadrži dovoljno informacija da odgovori na pitanje.\u003C\u002Fli>\u003Cli>Podaci se mogu parsirati ili upitati u obliku koji sloj za preuzimanje može koristiti.\u003C\u002Fli>\u003Cli>Preuzete informacije su dovoljno sveže za traženu odluku.\u003C\u002Fli>\u003Cli>Model prima izabrane dokaze u svom kontekstu.\u003C\u002Fli>\u003Cli>Autorizacija se sprovodi pre nego što zaštićeni dokazi stignu do modela.\u003C\u002Fli>\u003Cli>Generativni model i dalje može biti pogrešan čak i kada je preuzimanje ispravno.\u003C\u002Fli>\u003C\u002Ful>\n\u003Cp>Ove pretpostavke su važne jer preuzimanje ne može nadoknaditi nedostajuće dokaze, zastarele verzije izvora, pokvarene parsere ili neovlašćen pristup. RAG pipeline može biti samo onoliko pouzdan koliko je put dokaza koji ga hrani.\u003C\u002Fp>\n\u003Ch2 id=\"section-34\">Promenljive\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Promenljiva\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Zašto menja dizajn\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Struktura izvora\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">SQL tabela, pravni PDF i repozitorijum izvornog koda zahtevaju različite strategije preuzimanja\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Tip pitanja\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Tačno pretraživanje, konceptualna pretraga i višekoračno istraživanje su različiti zadaci\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Zahtev za svežinom\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Stanje uživo može zahtevati direktne upite umesto periodično obnavljanih indeksa\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Veličina korpusa\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Pretraga u memoriji može raditi za stotine delova, ali ne za veoma velike kolekcije\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Jezik\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Višejezično preuzimanje zahteva modele i tokenizaciju prikladne za stvarne jezike\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Dozvole\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Preuzimanje mora filtrirati prema pravima pristupa trenutnog korisnika\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Latencija i trošak\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Više faza preuzimanja može poboljšati kvalitet, ali dodaje vreme izvršavanja i troškove infrastrukture\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Potreba za poreklom\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Sistemi visokog poverenja zahtevaju ID-ove izvora, verzije i dokaze koji se mogu pratiti\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-36\">Dijagnostička \u002F Metoda odlučivanja\u003C\u002Fh2>\n\u003Cp>Prva odluka nije „Koju vektorsku bazu podataka treba da instaliram?“ Već: \u003Cb>Koju vrstu činjenice pokušavam da pronađem?\u003C\u002Fb>\u003C\u002Fp>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Tip pitanja\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Preferirani prvi pristup\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Razlog\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Tačan ID ili trenutni zapis\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">SQL \u002F pretraga po ključu \u002F API\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Deterministički strukturirani pristup\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Tačan tekst, kodovi, nazivi\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Pretraga punog teksta ili ključnih reči\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Leksička preciznost\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Konceptualno pitanje o dokumentima\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Semantička vektorska pretraga\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Značenje se može razlikovati od teksta\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Mešovito poslovno znanje\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Hibridna pretraga + metapodaci filteri\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Kombinuje leksičke i semantičke signale\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Trenutno stanje aplikacije\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Direktan pristup stanju\u002Falatu\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Svežina je važnija od sličnosti dokumenata\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Cp>Koristan test je: \u003Cb>Da li već znam koji zapis mi je potreban, ili sistem mora da otkrije koji je odlomak relevantan?\u003C\u002Fb> Ako je zapis poznat, upitajte ga direktno. Ako relevantnost mora da se otkrije, pretraga postaje važnija.\u003C\u002Fp>\n\u003Cp>Kada je odgovor pogrešan, dijagnostikujte pipeline po redu umesto da odmah menjate LLM:\u003C\u002Fp>\n\u003Col>\u003Cli>\u003Cb>1. Pokrivenost izvora:\u003C\u002Fb> Da li tačna informacija postoji u dostupnom skupu izvora?\u003C\u002Fli>\u003Cli>\u003Cb>2. Svežina:\u003C\u002Fb> Da li je ta verzija dovoljno aktuelna za pitanje?\u003C\u002Fli>\u003Cli>\u003Cb>3. Parsiranje:\u003C\u002Fb> Da li je relevantan sadržaj ispravno izvučen?\u003C\u002Fli>\u003Cli>\u003Cb>4. Deljenje na delove:\u003C\u002Fb> Da li su dokazi ostali zajedno sa uslovima koji im daju značenje?\u003C\u002Fli>\u003Cli>\u003Cb>5. Pretraga:\u003C\u002Fb> Da li se ispravan deo pojavljuje među kandidatima?\u003C\u002Fli>\u003Cli>\u003Cb>6. Rangiranje:\u003C\u002Fb> Da li su jači izvori rangirani iznad slabijih ili konfliktnih?\u003C\u002Fli>\u003Cli>\u003Cb>7. Sastavljanje konteksta:\u003C\u002Fb> Da li je aplikacija zaista poslala izabrane dokaze modelu?\u003C\u002Fli>\u003Cli>\u003Cb>8. Generisanje:\u003C\u002Fb> Da li je LLM verno koristio priložene dokaze?\u003C\u002Fli>\u003Cli>\u003Cb>9. Pripisivanje:\u003C\u002Fb> Može li svaka važna tvrdnja da se prati do izvora?\u003C\u002Fli>\u003C\u002Fol>\n\u003Cp>Za dublju metodu otklanjanja grešaka u produkciji, pogledajte \u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fsr\u002Fblog\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method\">RAG nije uspeo — ali koji sloj je zapravo zakazao? Dijagnostička metoda\u003C\u002Fa>, koja proširuje ovaj lanac na nezavisno testabilne slojeve otkaza.\u003C\u002Fp>\n\u003Ch2 id=\"section-43\">Dokazi\u003C\u002Fh2>\n\u003Cp>RAG rad Lewisa i saradnika formalizovao je generisanje koje se uslovljava na preuzetu eksternu memoriju umesto da se oslanja samo na parametre modela. To pruža konceptualni temelj za razdvajanje generatora od izvora znanja koji se može preuzeti.\u003C\u002Fp>\n\u003Cp>Sentence Transformers dokumentuje semantičku pretragu kao ugrađivanje korpusa i upita u vektorski prostor i preuzimanje stavki sa visokom semantičkom sličnošću. Njegov trenutni API takođe razlikuje kodiranje upita od kodiranja dokumenata za zadatke preuzimanja.\u003C\u002Fp>\n\u003Cp>SQLite FTS5 demonstrira drugu stranu spektra: zrelo preuzimanje punog teksta može rangirati dokumente bez ugrađivanja. To je važno jer leksička pretraga ostaje vredna za identifikatore, tačnu terminologiju i mnoge hibridne dizajne preuzimanja.\u003C\u002Fp>\n\u003Cp>OpenAI dokumentacija o ugrađivanjima opisuje ugrađivanja kao numeričke vektorske reprezentacije koje se koriste za povezanost i pretragu. To je jedan put implementacije za semantičko preuzimanje, a ne definicija samog RAG-a.\u003C\u002Fp>\n\u003Ch2 id=\"section-48\">Stvarni primer 1: Fascikla sa tekstualnim datotekama\u003C\u002Fh2>\n\u003Cp>Pretpostavimo da direktorijum pod nazivom \u003Ccode>knowledge\u002F\u003C\u002Fcode> sadrži obične tekstualne datoteke. Python ih može učitati bez ikakve AI biblioteke.\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>from pathlib import Path\n\ndef load_text_files(folder=&quot;knowledge&quot;):\n    documents = []\n\n    for path in Path(folder).glob(&quot;*.txt&quot;):\n        documents.append({\n            &quot;source&quot;: path.name,\n            &quot;text&quot;: path.read_text(encoding=&quot;utf-8&quot;)\n        })\n\n    return documents\n\ndocuments = load_text_files()\n\nfor document in documents:\n    print(document[&quot;source&quot;], len(document[&quot;text&quot;]))\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>Fajl sistem je izvor podataka. Sledeće pitanje je koliko teksta treba da postane jedna jedinica koja se može preuzeti. Za duge dokumente, pretraga jednog kompletnog fajla je često previše gruba. Zato RAG pipeline-ovi obično kreiraju delove.\u003C\u002Fp>\n\u003Ch3 id=\"section-52\">Vrlo jednostavan alat za deljenje na delove\u003C\u002Fh3>\n\u003Cpre class=\"code-block\">\u003Ccode>def chunk_text(text, max_chars=800):\n    paragraphs = [p.strip() for p in text.split(&quot;\\n\\n&quot;) if p.strip()]\n\n    chunks = []\n    current = &quot;&quot;\n\n    for paragraph in paragraphs:\n        candidate = f&quot;{current}\\n\\n{paragraph}&quot;.strip()\n\n        if current and len(candidate) &gt; max_chars:\n            chunks.append(current)\n            current = paragraph\n        else:\n            current = candidate\n\n    if current:\n        chunks.append(current)\n\n    return chunks\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>Ovaj primer grupiše pasuse dok se ne dostigne grubo ograničenje broja znakova. Namerno je razumljiv, a ne optimalan. Produkcijski sistemi često dele na delove po tokenima, naslovima, sekcijama, granicama rečenica ili strukturi dokumenta. Tabele, izvorni kod, ugovori i API dokumentacija mogu zahtevati različite strategije.\u003C\u002Fp>\n\u003Ch3 id=\"section-55\">Očuvajte poreklo prilikom deljenja na delove\u003C\u002Fh3>\n\u003Cpre class=\"code-block\">\u003Ccode>def build_chunks(documents):\n    chunks = []\n\n    for document in documents:\n        for index, text in enumerate(chunk_text(document[&quot;text&quot;])):\n            chunks.append({\n                &quot;id&quot;: f&#39;{document[&quot;source&quot;]}:{index}&#39;,\n                &quot;source&quot;: document[&quot;source&quot;],\n                &quot;chunk&quot;: index,\n                &quot;text&quot;: text,\n            })\n\n    return chunks\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>Korisni deo nosi više od teksta. Naziv izvora, ID dokumenta, URL, vremenska oznaka, verzija ili odeljak mogu kasnije podržati citiranje, otklanjanje grešaka i provere svežine. Ako se poreklo izgubi tokom unosa, postaje mnogo teže objasniti zašto je određeni odgovor proizveden.\u003C\u002Fp>\n\u003Ch2 id=\"section-58\">Stvarni primer 2: Strukturirani podaci — Koristite SQL kada je SQL pravi alat\u003C\u002Fh2>\n\u003Cp>Ne treba svaka eksterna činjenica da prođe kroz semantičku pretragu. Ako pitanje zahteva tačan trenutni zapis, direktan upit baze podataka je obično jasniji i determinističkiji.\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>import sqlite3\n\ndef get_order_status(order_id):\n    connection = sqlite3.connect(&quot;shop.db&quot;)\n    cursor = connection.cursor()\n\n    cursor.execute(\n        &quot;SELECT status, total, currency FROM orders WHERE id = ?&quot;,\n        (order_id,)\n    )\n\n    row = cursor.fetchone()\n    connection.close()\n\n    if row is None:\n        return None\n\n    return {\n        &quot;order_id&quot;: order_id,\n        &quot;status&quot;: row[0],\n        &quot;total&quot;: row[1],\n        &quot;currency&quot;: row[2],\n    }\n\nprint(get_order_status(4711))\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>Ako aplikacija već zna da korisnik pita o porudžbini 4711, ugrađivanje cele tabele porudžbina i traženje od semantičke pretrage da ponovo otkrije taj red obično dodaje složenost bez koristi. Snažno pravilo dizajna je: \u003Cb>dohvatite strukturirane činjenice strukturiranim upitima; dohvatite nestrukturirano znanje pretragom.\u003C\u002Fb>\u003C\u002Fp>\n\u003Cp>Vraćeni red baze podataka i dalje može biti smešten u kontekst modela kako bi LLM mogao da ga objasni prirodnim jezikom. Ali direktan pristup stanju ili zapisu je konceptualno drugačiji od pretrage korpusa znanja.\u003C\u002Fp>\n\u003Ch2 id=\"section-63\">Stvarni primer 3: Pretraga punog teksta pre ugrađivanja\u003C\u002Fh2>\n\u003Cp>Između naivne Python petlje i vektorske pretrage leži zrela klasa sistema leksičkog dohvata. SQLite uključuje FTS5 za pretragu punog teksta, uključujući BM25 rangiranje.\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>import sqlite3\n\nconnection = sqlite3.connect(&quot;knowledge.db&quot;)\ncursor = connection.cursor()\n\ncursor.execute(\n    &quot;CREATE VIRTUAL TABLE IF NOT EXISTS docs USING fts5(title, body)&quot;\n)\n\ncursor.execute(\n    &quot;INSERT INTO docs(title, body) VALUES (?, ?)&quot;,\n    (&quot;AKM&quot;, &quot;The AKM uses 7.62 mm ammunition.&quot;)\n)\n\nconnection.commit()\n\nquery = &quot;AKM ammunition&quot;\n\nrows = cursor.execute(\n    &quot;SELECT title, body, bm25(docs) AS score &quot;\n    &quot;FROM docs WHERE docs MATCH ? &quot;\n    &quot;ORDER BY score LIMIT 5&quot;,\n    (query,)\n).fetchall()\n\nfor row in rows:\n    print(row)\n\nconnection.close()\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>Leksička pretraga je posebno korisna kada su tačna terminologija, kodovi proizvoda, imena, identifikatori ili reči specifične za domen važni. Semantička pretraga nije automatski bolja. Produkcioni sistemi često kombinuju oba signala.\u003C\u002Fp>\n\u003Ch2 id=\"section-67\">Stvarni primer 4: Semantički dohvat sa ugrađivanjima\u003C\u002Fh2>\n\u003Cp>Ugrađivanja pretvaraju tekst u numeričke vektore tako da semantički povezani odlomci mogu biti upoređeni čak i kada ne koriste identične reči. Sentence Transformers pruža jednostavnu lokalnu implementaciju.\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode># pip install sentence-transformers\n\nfrom sentence_transformers import SentenceTransformer, util\n\ndocuments = [\n    &quot;The AKM uses 7.62 mm ammunition.&quot;,\n    &quot;A Med Kit restores health.&quot;,\n    &quot;Vehicle maintenance includes checking oil, brakes and tires.&quot;,\n    &quot;Account recovery requires access to the registered email address.&quot;\n]\n\nmodel = SentenceTransformer(\n    &quot;sentence-transformers\u002Fmulti-qa-mpnet-base-cos-v1&quot;\n)\n\ndocument_embeddings = model.encode_document(\n    documents,\n    convert_to_tensor=True\n)\n\nquestion = &quot;How do I repair my car?&quot;\n\nquery_embedding = model.encode_query(\n    question,\n    convert_to_tensor=True\n)\n\nhits = util.semantic_search(\n    query_embedding,\n    document_embeddings,\n    top_k=2\n)[0]\n\nfor hit in hits:\n    print(round(float(hit[&quot;score&quot;]), 3), documents[hit[&quot;corpus_id&quot;]])\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>Upit ne sadrži frazu „održavanje vozila“, ali semantički model i dalje može visoko rangirati taj odlomak jer su koncepti povezani. To je praktični razlog zašto su ugrađivanja uobičajena u RAG sistemima.\u003C\u002Fp>\n\u003Cp>Za male kolekcije, ugrađivanja mogu ostati u memoriji. Veći sistemi ih obično čuvaju u indeksu ili bazi podataka sa podrškom za vektore i tamo obavljaju pretragu najbližih suseda. Skladištenje se menja, ali logika ostaje: kodirajte pitanje, pronađite relevantne reprezentacije dokumenata, vratite najbolje dokaze.\u003C\u002Fp>\n\u003Ch2 id=\"section-72\">Stvarni primer 5: Izgradite kontekst za LLM\u003C\u002Fh2>\n\u003Cp>Retriver treba da vrati dokaze. LLM zatim treba da primi pitanje plus te dokaze. Održavanje pretrage i generisanja odvojenim čini oba lakšim za pregled i testiranje.\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>def build_prompt(question, retrieved_documents):\n    context = &quot;\\n\\n&quot;.join(\n        f&#39;[{doc[&quot;id&quot;]}] {doc[&quot;text&quot;]}&#39;\n        for doc in retrieved_documents\n    )\n\n    return f&quot;&quot;&quot;\nAnswer the question using the supplied context.\n\nRules:\n- Do not invent facts that are not supported by the context.\n- If the context is insufficient, say so.\n- Cite the source IDs you used.\n\nQuestion:\n{question}\n\nContext:\n{context}\n&quot;&quot;&quot;.strip()\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>Instrukcija ne čini model nepogrešivim. Ona jednostavno stvara eksplicitnu granicu dokaza. Model i dalje može pogrešno razumeti dobre dokaze, ignorisati uslov ili preterano generalizovati. Zato se kvalitet pretrage i kvalitet generisanja moraju evaluirati odvojeno.\u003C\u002Fp>\n\u003Ch2 id=\"section-76\">Stvarni primer 6: Kompletan minimalni pipeline\u003C\u002Fh2>\n\u003Cpre class=\"code-block\">\u003Ccode>def answer_question(question, all_documents, call_llm):\n    # 1. Retrieve evidence\n    retrieved = retrieve(question, all_documents, top_k=3)\n\n    # 2. Build model context\n    prompt = build_prompt(question, retrieved)\n\n    # 3. Generate the answer\n    answer = call_llm(prompt)\n\n    return {\n        &quot;answer&quot;: answer,\n        &quot;sources&quot;: [doc[&quot;id&quot;] for doc in retrieved]\n    }\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>Funkcija namerno prima \u003Ccode>call_llm\u003C\u002Fcode> kao zavisnost. Pretragu ne treba da zanima da li generisanje obavlja cloud model, lokalni model ili drugi provajder. Putanja podataka pripada aplikaciji.\u003C\u002Fp>\n\u003Ch3 id=\"section-79\">Opcioni generator: OpenAI Responses API\u003C\u002Fh3>\n\u003Cp>Jedan mogući generator je OpenAI Responses API. Održavanje imena modela u promenljivoj okruženja izbegava hardkodiranje određenog modela u RAG arhitekturu.\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode># pip install openai\n\nimport os\nfrom openai import OpenAI\n\nclient = OpenAI()\n\ndef call_llm(prompt):\n    response = client.responses.create(\n        model=os.environ[&quot;OPENAI_MODEL&quot;],\n        input=prompt,\n    )\n    return response.output_text\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>Isti pipeline za pretragu može se povezati sa lokalnim serverom za inferenciju. Ovo je važna arhitektonska tačka: \u003Cb>RAG nije vlasništvo LLM provajdera.\u003C\u002Fb> Aplikacija poseduje izvor, pretragu i sastavljanje konteksta.\u003C\u002Fp>\n\u003Ch3 id=\"section-83\">Cela arhitektura u jednom pregledu\u003C\u002Fh3>\n\u003Cpre class=\"code-block\">\u003Ccode>USER QUESTION\n     |\n     v\n+-------------+\n|  Retriever  |\n+-------------+\n   |       |\n   |       +----&gt; SQL \u002F API \u002F state query\n   |\n   +------------&gt; keyword \u002F full-text search\n   |\n   +------------&gt; embedding \u002F vector search\n                     |\n                     v\n              relevant evidence\n                     |\n                     v\n+-----------------------------------+\n| question + evidence + instructions |\n+-----------------------------------+\n                     |\n                     v\n                   LLM\n                     |\n                     v\n                  answer\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>Ovaj model toka podataka je trajniji od memorisanja jednog framework-a. Biblioteke, baze podataka i prodavci modela će se menjati; granice odgovornosti ostaju.\u003C\u002Fp>\n\u003Ch2 id=\"section-86\">Uobičajene zablude i načini neuspeha\u003C\u002Fh2>\n\u003Ch3 id=\"section-87\">„RAG znači vektorska baza podataka.“\u003C\u002Fh3>\n\u003Cp>Ne. Vektorska pretraga je jedna metoda pretrage. RAG može koristiti full-text pretragu, SQL, API-je, grafove znanja, vektorsku pretragu ili njihove kombinacije. Definišući obrazac je pretraga eksternih informacija za generisanje.\u003C\u002Fp>\n\u003Ch3 id=\"section-89\">„Ako su podaci u PostgreSQL-u, moram ugraditi celu bazu podataka.“\u003C\u002Fh3>\n\u003Cp>Ne. Strukturirani zapisi obično treba da ostanu upitljivi kao strukturirani zapisi. Embeddings su korisni za semantičku relevantnost, ne kao zamena za determinističke upite.\u003C\u002Fp>\n\u003Ch3 id=\"section-91\">„Više delova znači bolji odgovor.“\u003C\u002Fh3>\n\u003Cp>Ne nužno. Dodatni kontekst može uneti šum, konfliktne verzije i irelevantan materijal. Pretraga treba da optimizuje za korisne dokaze, a ne za maksimalni obim.\u003C\u002Fp>\n\u003Ch3 id=\"section-93\">„Visok skor sličnosti dokazuje odgovor.“\u003C\u002Fh3>\n\u003Cp>Ne. Sličnost meri relevantnost, a ne istinitost ili primenljivost. Veoma sličan odlomak može biti zastareo, iz pogrešne verzije proizvoda ili važeći samo pod uslovima koji se ne poklapaju sa pitanjem.\u003C\u002Fp>\n\u003Ch3 id=\"section-95\">„Kada se preuzme ispravan deo, halucinacija je rešena.“\u003C\u002Fh3>\n\u003Cp>Ne. Pretraga poboljšava utemeljenost, ali ne garantuje verodostojno zaključivanje. Generisanje i dalje zahteva evaluaciju, a radni tokovi visokog rizika mogu zahtevati determinističku validaciju ili ljudski pregled.\u003C\u002Fp>\n\u003Ch3 id=\"section-97\">„Model je zakazao, pa promenite model.“\u003C\u002Fh3>\n\u003Cp>Ne nužno. Ispravan izvor je možda nedostajao, bio pogrešno parsiran, loše podeljen, filtriran, rangiran prenisko ili izostavljen iz sastavljenog konteksta. Zamena modela ne bi trebalo da bude prvi dijagnostički korak.\u003C\u002Fp>\n\u003Ch2 id=\"section-99\">Rubni slučajevi\u003C\u002Fh2>\n\u003Cul>\u003Cli>\u003Cb>Konfliktni dokumenti:\u003C\u002Fb> dva izvora se mogu ne slagati jer se verzije, jurisdikcije ili proizvodi razlikuju.\u003C\u002Fli>\u003Cli>\u003Cb>Vremenski osetljive činjenice:\u003C\u002Fb> semantički relevantan izvor može već biti zastareo.\u003C\u002Fli>\u003Cli>\u003Cb>Dozvole:\u003C\u002Fb> pretraživač ne sme da vrati dokumente kojima trenutni korisnik nije ovlašćen da pristupi.\u003C\u002Fli>\u003Cli>\u003Cb>Višejezične kolekcije:\u003C\u002Fb> model za ugrađivanje i strategija pretrage moraju podržavati jezike koji se stvarno koriste.\u003C\u002Fli>\u003Cli>\u003Cb>Tabele i izvorni kod:\u003C\u002Fb> obično deljenje na pasuse može uništiti strukturu koja je ključna za odgovor.\u003C\u002Fli>\u003Cli>\u003Cb>Vrlo kratki identifikatori:\u003C\u002Fb> semantička pretraga može biti slabija od tačnog poklapanja za SKU-ove, ID-ove, kodove grešaka ili akronime.\u003C\u002Fli>\u003Cli>\u003Cb>Duga pitanja koja zahtevaju nekoliko činjenica:\u003C\u002Fb> pretraga može zahtevati dekompoziciju, nekoliko pretraga ili ponovno rangiranje umesto jednog top-k upita.\u003C\u002Fli>\u003Cli>\u003Cb>Hijerarhija izvora:\u003C\u002Fb> zvanična aktuelna politika može trebati da bude rangirana iznad starijeg, ali semantički bližeg diskusionog dokumenta.\u003C\u002Fli>\u003C\u002Ful>\n\u003Ch2 id=\"section-101\">Ograničenja\u003C\u002Fh2>\n\u003Cp>Python primeri namerno optimizuju za transparentnost, a ne za skalabilnost. Pretraživač po ključnim rečima je naivan, delilac koristi dužinu znakova, SQLite primeri ne uključuju upravljanje vezama za produkciju, a semantički primer drži sve ugrađene vektore u memoriji.\u003C\u002Fp>\n\u003Cp>Produkcijski sistem može zahtevati vektorske indekse, ponovne rangere, hibridnu pretragu, parsere dokumenata, keširanje, inkrementalno indeksiranje, verzionisanje izvora, filtere kontrole pristupa, observabilnost, skupove podataka za evaluaciju i rukovanje greškama. Nijedan od tih dodataka ne menja osnovnu arhitekturu; oni čine svaku granicu pouzdanijom.\u003C\u002Fp>\n\u003Cp>RAG takođe ne može da stvori dokaze koji ne postoje u skupu izvora. Ako je izvor pogrešan, nepotpun ili zastareo, bolji model za ugrađivanje ne može ga pretvoriti u autoritativno znanje.\u003C\u002Fp>\n\u003Ch2 id=\"section-105\">Šta bi promenilo ovaj odgovor?\u003C\u002Fh2>\n\u003Cp>Arhitektura se menja kada zadatak zahteva više od pronalaženja znanja. Status porudžbine uživo zahteva trenutno stanje. Finansijski izračun može zahtevati deterministički kod. Zadatak veb-istraživanja može zahtevati aktivnu pretragu. Radni tok može zahtevati alate koji mogu da upišu podatke nazad u drugi sistem. Autonomni agent može zahtevati planiranje, dozvole i kontrolu izvršavanja pored pretrage.\u003C\u002Fp>\n\u003Cp>RAG je stoga najbolje razumeti kao \u003Cb>jedan sloj za prikupljanje dokaza unutar većeg AI sistema\u003C\u002Fb>. Moćan je upravo zato što ima uzak zadatak: pronaći korisne spoljne informacije i smestiti ih u radni kontekst modela.\u003C\u002Fp>\n\u003Ch2 id=\"section-108\">Zaključak\u003C\u002Fh2>\n\u003Cp>RAG postaje mnogo lakši za razumevanje kada se uklone nazivi tehnologija. Datoteka je izvor. Baza podataka je izvor. API je izvor. Funkcija pretrage pronalazi dokaze. Upit nosi te dokaze do modela. LLM ih zatim tumači i proizvodi jezik.\u003C\u002Fp>\n\u003Cp>Teži deo produkcijskog RAG-a nije pozivanje modela za ugrađivanje. To je izgradnja pouzdane putanje dokaza od originalnog izvora do konačne tvrdnje: očuvanje porekla, odabir prave metode pretrage, održavanje informacija ažurnim, kontrola pristupa, evaluacija pretrage odvojeno od generisanja i znanje kada je direktan poziv baze podataka ili alata bolji od semantičke pretrage.\u003C\u002Fp>\n\u003Cp>To je praktični nastavak osnovnog RAG modela: \u003Cb>prvo razumeti uloge, zatim učiniti putanju podataka eksplicitnom.\u003C\u002Fb>\u003C\u002Fp>\n\u003Ch2 id=\"section-112\">Primarni izvori\u003C\u002Fh2>\n\u003Cul>\u003Cli>\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2005.11401\">Lewis et al. — Retrieval-Augmented Generation for Knowledge-Intensive NLP Tasks\u003C\u002Fa> — rad iz 2020. koji uvodi RAG formulaciju koja kombinuje generisanje sa pronađenom neparametarskom memorijom.\u003C\u002Fli>\u003Cli>\u003Ca href=\"https:\u002F\u002Fwww.sbert.net\u002Fexamples\u002Fsentence_transformer\u002Fapplications\u002Fsemantic-search\u002FREADME.html\">Sentence Transformers — Semantic Search\u003C\u002Fa> — zvanična dokumentacija za semantičku pretragu, ugrađivanje upita i ugrađivanje dokumenata.\u003C\u002Fli>\u003Cli>\u003Ca href=\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fembeddings\">OpenAI — Vector Embeddings\u003C\u002Fa> — zvanična dokumentacija koja opisuje ugrađivanja kao numeričke reprezentacije koje se koriste za povezanost i pretragu.\u003C\u002Fli>\u003Cli>\u003Ca href=\"https:\u002F\u002Fwww.sqlite.org\u002Ffts5.html\">SQLite — FTS5 Extension\u003C\u002Fa> — zvanična dokumentacija za pretragu punog teksta i BM25 rangiranje u SQLite-u.\u003C\u002Fli>\u003Cli>\u003Ca href=\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Flibraries\">OpenAI — SDKs and CLI\u003C\u002Fa> — zvanični primer Python SDK-a za Responses API koji se koristi u opcionom primeru generatora.\u003C\u002Fli>\u003Cli>\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fsr\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works\">What Is RAG? The Simplest Explanation of How It Works\u003C\u002Fa> — konceptualni prvi deo ove serije.\u003C\u002Fli>\u003C\u002Ful>",{"time":212,"blocks":213,"version":676},1790517301991,[214,218,221,227,231,234,237,240,243,246,249,253,256,259,262,265,268,271,274,277,280,283,286,289,292,295,298,301,337,340,343,346,358,361,364,394,397,400,426,429,432,445,448,451,454,457,460,463,466,469,472,475,479,482,485,488,491,494,497,500,503,506,509,512,515,518,521,524,527,530,533,536,539,542,545,548,551,554,557,560,563,566,569,572,575,578,581,584,587,590,593,596,599,602,605,608,611,614,617,620,631,634,637,640,643,646,649,652,655,658,661,664,667],{"data":215,"type":217},{"text":216},"Prethodni članak, \u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fsr\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works\">Šta je RAG? Najjednostavnije objašnjenje kako funkcioniše\u003C\u002Fa>, uspostavio je mentalni model: LLM piše, RAG pronalazi korisno znanje, aplikacija poseduje trenutno stanje, a alati izvršavaju radnje. Ovaj članak čini sledeći korak: \u003Cb>odakle podaci zapravo dolaze i kako pretraga izgleda u Python-u?\u003C\u002Fb>","paragraph",{"data":219,"type":217},{"text":220},"Važno iznenađenje je da „izvor podataka za LLM“ obično nije ništa egzotično. To može biti tekstualni fajl, fascikla sa Markdown dokumentima, SQL baza podataka, odgovor API-ja, katalog proizvoda, sistem za podršku ili vektorski indeks izveden iz tih izvora. AI ne poznaje te sisteme magično. Vaša aplikacija mora da učita, upita, pretraži ili pronađe relevantne podatke i smesti rezultat u kontekst modela.",{"data":222,"type":226},{"text":223,"caption":224,"alignment":225},"Izvor podataka = gde informacije žive. Pretraga = kako aplikacija pronalazi korisne informacije. Kontekst = izabrane informacije date modelu. LLM = komponenta koja tumači taj kontekst i generiše odgovor.","Model od četiri dela koji se koristi u ovom članku","left","quote",{"data":228,"type":42},{"text":229,"level":230},"Pitanje",2,{"data":232,"type":217},{"text":233},"Kako LLM koristi eksterne podatke kao što su fajlovi, baze podataka ili API-ji, i kako mali Python program može da implementira osnovne RAG korake bez skrivanja iza okvira?",{"data":235,"type":42},{"text":236,"level":230},"Šta to zapravo znači",{"data":238,"type":217},{"text":239},"Kada programeri kažu da je LLM „povezan sa podacima kompanije“, nekoliko različitih operacija može biti skriveno iza te rečenice. Jedna aplikacija može izvršavati SQL. Druga može pozivati API. Treća može pokretati pretragu punog teksta. Četvrta može izračunavati sličnost ugrađivanja nad delovima dokumenata. Sve one mogu pružiti eksterne informacije LLM-u, ali nisu ista metoda pretrage i ne treba ih tretirati kao međusobno zamenljive.",{"data":241,"type":217},{"text":242},"Ova razlika je važna jer najbolja metoda pretrage zavisi od oblika pitanja. „Koja je naša politika povraćaja?“ je problem pretrage dokumenata. „Koji je trenutni status porudžbine 4711?“ je obično strukturirano pretraživanje baze podataka. „Koji pasus govori o oporavku naloga?“ može biti pretraga po ključnim rečima ili semantička pretraga. RAG je najkorisniji kada sistem mora da \u003Cb>otkrije relevantno znanje pre generisanja\u003C\u002Fb>.",{"data":244,"type":42},{"text":245,"level":230},"Najjednostavniji primer",{"data":247,"type":217},{"text":248},"Počnite sa tri stringa u običnom Python-u. Nema vektorske baze podataka, nema okvira, a još nema ni LLM-a. Želimo samo da učinimo korak pretrage vidljivim.",{"data":250,"type":252},{"code":251},"documents = [\n    \"The AKM uses 7.62 mm ammunition.\",\n    \"A Med Kit restores health.\",\n    \"A 4x scope can be attached to several compatible weapons.\"\n]\n\nquestion = \"Which ammunition does the AKM use?\"\n\nfor document in documents:\n    if \"AKM\" in document:\n        print(document)","code",{"data":254,"type":217},{"text":255},"Program štampa prvu rečenicu jer sadrži termin koji smo tražili. Ovo je primitivna pretraga, ali arhitektura je već vidljiva: \u003Cb>pitanje → pretraga → relevantan tekst\u003C\u002Fb>. RAG dodaje još jedan važan korak: prosledite pronađeni tekst jezičkom modelu zajedno sa pitanjem.",{"data":257,"type":217},{"text":258},"Malo opštija verzija rangira dokumente prema preklapanju termina upita:",{"data":260,"type":252},{"code":261},"import re\n\ndocuments = [\n    {\"id\": \"weapon-akm\", \"text\": \"The AKM uses 7.62 mm ammunition.\"},\n    {\"id\": \"healing-medkit\", \"text\": \"A Med Kit restores health.\"},\n    {\"id\": \"scope-4x\", \"text\": \"A 4x scope can be attached to several compatible weapons.\"},\n]\n\ndef words(text):\n    return set(re.findall(r\"[a-zA-Z0-9.]+\", text.lower()))\n\ndef retrieve(question, documents, top_k=2):\n    query_terms = words(question)\n    ranked = []\n\n    for document in documents:\n        score = len(query_terms & words(document[\"text\"]))\n        if score > 0:\n            ranked.append((score, document))\n\n    ranked.sort(key=lambda item: item[0], reverse=True)\n    return [document for _, document in ranked[:top_k]]\n\nquestion = \"Which ammunition does the AKM use?\"\nhits = retrieve(question, documents)\n\nfor hit in hits:\n    print(hit[\"id\"], \"->\", hit[\"text\"])",{"data":263,"type":217},{"text":264},"Ovo nije produkcioni pretraživač. Zanemaruje morfologiju, sinonime, pravopisne varijante, dužinu dokumenta i mnoge signale rangiranja. Njegova vrednost je edukativna: \u003Cb>RAG ne počinje vektorskom bazom podataka. Počinje pretragom.\u003C\u002Fb>",{"data":266,"type":42},{"text":267,"level":230},"Gde primer prestaje da funkcioniše",{"data":269,"type":217},{"text":270},"Tačno ili leksičko podudaranje postaje slabo kada pitanje i izvor koriste različite reči. Dokument može da kaže „održavanje vozila“, dok korisnik pita „kako da popravim svoj auto?“ Leksički pretraživač može da propusti vezu iako je čovek odmah vidi. Semantička pretraga se bavi ovim tako što predstavlja tekst kao vektore i poredi značenje, a ne samo tačne tokene.",{"data":272,"type":217},{"text":273},"Dugački fajlovi stvaraju još jedan problem. Pretraga celog priručnika od 80 stranica kao jedne celine je previše gruba, ali deljenje svake rečenice može uništiti koristan kontekst. Pravi RAG sistemi stoga zahtevaju odluke o parsiranju, deljenju na delove, metapodacima, rangiranju, svežini, dozvolama i poreklu.",{"data":275,"type":217},{"text":276},"Primer takođe ne govori ništa o strukturisanim aktuelnim činjenicama. Ako korisnik pita za trenutni status porudžbine 4711 i aplikacija već ima ključ baze podataka, semantička pretraga je obično pogrešan prvi alat. Deterministički upit baze podataka je bolji.",{"data":278,"type":42},{"text":279,"level":230},"Direktan odgovor",{"data":281,"type":217},{"text":282},"LLM izvor podataka je svaki eksterni sistem iz kojeg aplikacija može da dobije informacije za model: fajlovi, baze podataka, API-ji, indeksi pretrage, vektorski skladišta ili stanje aplikacije uživo. RAG je obrazac \u003Cb>preuzimanja relevantnog znanja iz takvih izvora pre generisanja\u003C\u002Fb>.",{"data":284,"type":217},{"text":285},"U Python-u, suštinski pipeline može biti veoma mali: \u003Cb>učitaj podatke → kreiraj jedinice za preuzimanje → pronađi relevantne dokaze → sastavi kontekst → pozovi LLM\u003C\u002Fb>. Metoda preuzimanja treba da odgovara izvoru i pitanju. Koristite SQL za tačne strukturisane činjenice, punotekstualnu pretragu za leksičko poklapanje, embedding-e za semantičku sličnost i hibridno preuzimanje kada je nekoliko signala vredno.",{"data":287,"type":42},{"text":288,"level":230},"Zašto je to tako",{"data":290,"type":217},{"text":291},"Jezički model ne prima automatski sadržaj vašeg fajl sistema, PostgreSQL baze podataka, CRM-a, privatnog API-ja ili novoizmenjenog dokumenta. Aplikacija odlučuje koje eksterne informacije su dostupne i šta se stavlja u trenutni kontekst modela.",{"data":293,"type":217},{"text":294},"Originalni rad Retrieval-Augmented Generation od Lewis et al. kombinovao je generativni model sa eksternom neparametarskom memorijom preuzetom iz gustog vektorskog indeksa. Šira arhitektonska ideja nadživljava tu specifičnu implementaciju: eksterni dokazi mogu se preuzeti u vreme inferencije umesto da se očekuje da sve korisno znanje bude kodirano u parametrima modela.",{"data":296,"type":217},{"text":297},"Ovo stvara korisnu podelu odgovornosti: izvor čuva informacije, pretraživač bira dokaze, kontekst nosi te dokaze u zahtev, a model ih interpretira. Održavanje tih granica vidljivim čini kvarove mnogo lakšim za dijagnostikovanje.",{"data":299,"type":42},{"text":300,"level":230},"Kontekst: Glavni tipovi izvora podataka",{"data":302,"type":336},{"content":303,"withHeadings":14},[304,308,312,316,320,324,328,332],[305,306,307],"Izvor","Tipična metoda preuzimanja","Dobro za",[309,310,311],"TXT \u002F Markdown \u002F HTML","Parsiranje + leksička ili semantička pretraga","Dokumentacija, priručnici, članci, beleške",[313,314,315],"PDF \u002F DOCX","Ekstrakcija svesna strukture + pretraga","Politike, izveštaji, ugovori, priručnici",[317,318,319],"SQL baza podataka","SQL upit ili filtrirano preuzimanje","Porudžbine, korisnici, proizvodi, strukturisani zapisi",[321,322,323],"REST \u002F GraphQL API","HTTP zahtev sa parametrima","Udaljeni sistemi i podaci servisa uživo",[325,326,327],"Indeks pretrage","BM25 \u002F punotekstualna \u002F hibridna pretraga","Velike tekstualne kolekcije",[329,330,331],"Vektorski indeks","Sličnost embedding-a","Semantičko preuzimanje dokumenata",[333,334,335],"Stanje aplikacije","Direktno čitanje stanja ili poziv alata","Šta je istinito upravo sada","table",{"data":338,"type":217},{"text":339},"Vektorski indeks zaslužuje posebnu pažnju. U mnogim arhitekturama on \u003Cb>nije kanonski izvor istine\u003C\u002Fb>. To je indeks za preuzimanje izveden iz dokumenata ili zapisa. Autoritativni dokument može živeti u objektnom skladištu, CMS-u, Git-u, PostgreSQL-u ili drugom sistemu, dok se embedding-i i metapodaci čuvaju odvojeno za brzu semantičku pretragu. Neki sistemi koriste vektorsko skladište kao primarno skladište, ali to je arhitektonski izbor, a ne zahtev RAG-a.",{"data":341,"type":217},{"text":342},"Ako je granica između preuzimanja, trajne memorije, trenutnog stanja i konteksta modela još uvek nejasna, pogledajte \u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fsr\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context\">AI Agent Memory Is Not RAG\u003C\u002Fa>. Ti slojevi mogu koristiti neke od istih tehnologija skladištenja, a ipak imati različita pravila ispravnosti.",{"data":344,"type":42},{"text":345,"level":230},"Pretpostavke",{"data":347,"type":357},{"items":348,"style":356},[349,350,351,352,353,354,355],"Aplikaciji je dozvoljen pristup eksternom izvoru.","Relevantni izvor sadrži dovoljno informacija da odgovori na pitanje.","Podaci se mogu parsirati ili upitati u obliku koji sloj za preuzimanje može koristiti.","Preuzete informacije su dovoljno sveže za traženu odluku.","Model prima izabrane dokaze u svom kontekstu.","Autorizacija se sprovodi pre nego što zaštićeni dokazi stignu do modela.","Generativni model i dalje može biti pogrešan čak i kada je preuzimanje ispravno.","unordered","list",{"data":359,"type":217},{"text":360},"Ove pretpostavke su važne jer preuzimanje ne može nadoknaditi nedostajuće dokaze, zastarele verzije izvora, pokvarene parsere ili neovlašćen pristup. RAG pipeline može biti samo onoliko pouzdan koliko je put dokaza koji ga hrani.",{"data":362,"type":42},{"text":363,"level":230},"Promenljive",{"data":365,"type":336},{"content":366,"withHeadings":14},[367,370,373,376,379,382,385,388,391],[368,369],"Promenljiva","Zašto menja dizajn",[371,372],"Struktura izvora","SQL tabela, pravni PDF i repozitorijum izvornog koda zahtevaju različite strategije preuzimanja",[374,375],"Tip pitanja","Tačno pretraživanje, konceptualna pretraga i višekoračno istraživanje su različiti zadaci",[377,378],"Zahtev za svežinom","Stanje uživo može zahtevati direktne upite umesto periodično obnavljanih indeksa",[380,381],"Veličina korpusa","Pretraga u memoriji može raditi za stotine delova, ali ne za veoma velike kolekcije",[383,384],"Jezik","Višejezično preuzimanje zahteva modele i tokenizaciju prikladne za stvarne jezike",[386,387],"Dozvole","Preuzimanje mora filtrirati prema pravima pristupa trenutnog korisnika",[389,390],"Latencija i trošak","Više faza preuzimanja može poboljšati kvalitet, ali dodaje vreme izvršavanja i troškove infrastrukture",[392,393],"Potreba za poreklom","Sistemi visokog poverenja zahtevaju ID-ove izvora, verzije i dokaze koji se mogu pratiti",{"data":395,"type":42},{"text":396,"level":230},"Dijagnostička \u002F Metoda odlučivanja",{"data":398,"type":217},{"text":399},"Prva odluka nije „Koju vektorsku bazu podataka treba da instaliram?“ Već: \u003Cb>Koju vrstu činjenice pokušavam da pronađem?\u003C\u002Fb>",{"data":401,"type":336},{"content":402,"withHeadings":14},[403,406,410,414,418,422],[374,404,405],"Preferirani prvi pristup","Razlog",[407,408,409],"Tačan ID ili trenutni zapis","SQL \u002F pretraga po ključu \u002F API","Deterministički strukturirani pristup",[411,412,413],"Tačan tekst, kodovi, nazivi","Pretraga punog teksta ili ključnih reči","Leksička preciznost",[415,416,417],"Konceptualno pitanje o dokumentima","Semantička vektorska pretraga","Značenje se može razlikovati od teksta",[419,420,421],"Mešovito poslovno znanje","Hibridna pretraga + metapodaci filteri","Kombinuje leksičke i semantičke signale",[423,424,425],"Trenutno stanje aplikacije","Direktan pristup stanju\u002Falatu","Svežina je važnija od sličnosti dokumenata",{"data":427,"type":217},{"text":428},"Koristan test je: \u003Cb>Da li već znam koji zapis mi je potreban, ili sistem mora da otkrije koji je odlomak relevantan?\u003C\u002Fb> Ako je zapis poznat, upitajte ga direktno. Ako relevantnost mora da se otkrije, pretraga postaje važnija.",{"data":430,"type":217},{"text":431},"Kada je odgovor pogrešan, dijagnostikujte pipeline po redu umesto da odmah menjate LLM:",{"data":433,"type":357},{"items":434,"style":444},[435,436,437,438,439,440,441,442,443],"\u003Cb>1. Pokrivenost izvora:\u003C\u002Fb> Da li tačna informacija postoji u dostupnom skupu izvora?","\u003Cb>2. Svežina:\u003C\u002Fb> Da li je ta verzija dovoljno aktuelna za pitanje?","\u003Cb>3. Parsiranje:\u003C\u002Fb> Da li je relevantan sadržaj ispravno izvučen?","\u003Cb>4. Deljenje na delove:\u003C\u002Fb> Da li su dokazi ostali zajedno sa uslovima koji im daju značenje?","\u003Cb>5. Pretraga:\u003C\u002Fb> Da li se ispravan deo pojavljuje među kandidatima?","\u003Cb>6. Rangiranje:\u003C\u002Fb> Da li su jači izvori rangirani iznad slabijih ili konfliktnih?","\u003Cb>7. Sastavljanje konteksta:\u003C\u002Fb> Da li je aplikacija zaista poslala izabrane dokaze modelu?","\u003Cb>8. Generisanje:\u003C\u002Fb> Da li je LLM verno koristio priložene dokaze?","\u003Cb>9. Pripisivanje:\u003C\u002Fb> Može li svaka važna tvrdnja da se prati do izvora?","ordered",{"data":446,"type":217},{"text":447},"Za dublju metodu otklanjanja grešaka u produkciji, pogledajte \u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fsr\u002Fblog\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method\">RAG nije uspeo — ali koji sloj je zapravo zakazao? Dijagnostička metoda\u003C\u002Fa>, koja proširuje ovaj lanac na nezavisno testabilne slojeve otkaza.",{"data":449,"type":42},{"text":450,"level":230},"Dokazi",{"data":452,"type":217},{"text":453},"RAG rad Lewisa i saradnika formalizovao je generisanje koje se uslovljava na preuzetu eksternu memoriju umesto da se oslanja samo na parametre modela. To pruža konceptualni temelj za razdvajanje generatora od izvora znanja koji se može preuzeti.",{"data":455,"type":217},{"text":456},"Sentence Transformers dokumentuje semantičku pretragu kao ugrađivanje korpusa i upita u vektorski prostor i preuzimanje stavki sa visokom semantičkom sličnošću. Njegov trenutni API takođe razlikuje kodiranje upita od kodiranja dokumenata za zadatke preuzimanja.",{"data":458,"type":217},{"text":459},"SQLite FTS5 demonstrira drugu stranu spektra: zrelo preuzimanje punog teksta može rangirati dokumente bez ugrađivanja. To je važno jer leksička pretraga ostaje vredna za identifikatore, tačnu terminologiju i mnoge hibridne dizajne preuzimanja.",{"data":461,"type":217},{"text":462},"OpenAI dokumentacija o ugrađivanjima opisuje ugrađivanja kao numeričke vektorske reprezentacije koje se koriste za povezanost i pretragu. To je jedan put implementacije za semantičko preuzimanje, a ne definicija samog RAG-a.",{"data":464,"type":42},{"text":465,"level":230},"Stvarni primer 1: Fascikla sa tekstualnim datotekama",{"data":467,"type":217},{"text":468},"Pretpostavimo da direktorijum pod nazivom \u003Ccode>knowledge\u002F\u003C\u002Fcode> sadrži obične tekstualne datoteke. Python ih može učitati bez ikakve AI biblioteke.",{"data":470,"type":252},{"code":471},"from pathlib import Path\n\ndef load_text_files(folder=\"knowledge\"):\n    documents = []\n\n    for path in Path(folder).glob(\"*.txt\"):\n        documents.append({\n            \"source\": path.name,\n            \"text\": path.read_text(encoding=\"utf-8\")\n        })\n\n    return documents\n\ndocuments = load_text_files()\n\nfor document in documents:\n    print(document[\"source\"], len(document[\"text\"]))",{"data":473,"type":217},{"text":474},"Fajl sistem je izvor podataka. Sledeće pitanje je koliko teksta treba da postane jedna jedinica koja se može preuzeti. Za duge dokumente, pretraga jednog kompletnog fajla je često previše gruba. Zato RAG pipeline-ovi obično kreiraju delove.",{"data":476,"type":42},{"text":477,"level":478},"Vrlo jednostavan alat za deljenje na delove",3,{"data":480,"type":252},{"code":481},"def chunk_text(text, max_chars=800):\n    paragraphs = [p.strip() for p in text.split(\"\\n\\n\") if p.strip()]\n\n    chunks = []\n    current = \"\"\n\n    for paragraph in paragraphs:\n        candidate = f\"{current}\\n\\n{paragraph}\".strip()\n\n        if current and len(candidate) > max_chars:\n            chunks.append(current)\n            current = paragraph\n        else:\n            current = candidate\n\n    if current:\n        chunks.append(current)\n\n    return chunks",{"data":483,"type":217},{"text":484},"Ovaj primer grupiše pasuse dok se ne dostigne grubo ograničenje broja znakova. Namerno je razumljiv, a ne optimalan. Produkcijski sistemi često dele na delove po tokenima, naslovima, sekcijama, granicama rečenica ili strukturi dokumenta. Tabele, izvorni kod, ugovori i API dokumentacija mogu zahtevati različite strategije.",{"data":486,"type":42},{"text":487,"level":478},"Očuvajte poreklo prilikom deljenja na delove",{"data":489,"type":252},{"code":490},"def build_chunks(documents):\n    chunks = []\n\n    for document in documents:\n        for index, text in enumerate(chunk_text(document[\"text\"])):\n            chunks.append({\n                \"id\": f'{document[\"source\"]}:{index}',\n                \"source\": document[\"source\"],\n                \"chunk\": index,\n                \"text\": text,\n            })\n\n    return chunks",{"data":492,"type":217},{"text":493},"Korisni deo nosi više od teksta. Naziv izvora, ID dokumenta, URL, vremenska oznaka, verzija ili odeljak mogu kasnije podržati citiranje, otklanjanje grešaka i provere svežine. Ako se poreklo izgubi tokom unosa, postaje mnogo teže objasniti zašto je određeni odgovor proizveden.",{"data":495,"type":42},{"text":496,"level":230},"Stvarni primer 2: Strukturirani podaci — Koristite SQL kada je SQL pravi alat",{"data":498,"type":217},{"text":499},"Ne treba svaka eksterna činjenica da prođe kroz semantičku pretragu. Ako pitanje zahteva tačan trenutni zapis, direktan upit baze podataka je obično jasniji i determinističkiji.",{"data":501,"type":252},{"code":502},"import sqlite3\n\ndef get_order_status(order_id):\n    connection = sqlite3.connect(\"shop.db\")\n    cursor = connection.cursor()\n\n    cursor.execute(\n        \"SELECT status, total, currency FROM orders WHERE id = ?\",\n        (order_id,)\n    )\n\n    row = cursor.fetchone()\n    connection.close()\n\n    if row is None:\n        return None\n\n    return {\n        \"order_id\": order_id,\n        \"status\": row[0],\n        \"total\": row[1],\n        \"currency\": row[2],\n    }\n\nprint(get_order_status(4711))",{"data":504,"type":217},{"text":505},"Ako aplikacija već zna da korisnik pita o porudžbini 4711, ugrađivanje cele tabele porudžbina i traženje od semantičke pretrage da ponovo otkrije taj red obično dodaje složenost bez koristi. Snažno pravilo dizajna je: \u003Cb>dohvatite strukturirane činjenice strukturiranim upitima; dohvatite nestrukturirano znanje pretragom.\u003C\u002Fb>",{"data":507,"type":217},{"text":508},"Vraćeni red baze podataka i dalje može biti smešten u kontekst modela kako bi LLM mogao da ga objasni prirodnim jezikom. Ali direktan pristup stanju ili zapisu je konceptualno drugačiji od pretrage korpusa znanja.",{"data":510,"type":42},{"text":511,"level":230},"Stvarni primer 3: Pretraga punog teksta pre ugrađivanja",{"data":513,"type":217},{"text":514},"Između naivne Python petlje i vektorske pretrage leži zrela klasa sistema leksičkog dohvata. SQLite uključuje FTS5 za pretragu punog teksta, uključujući BM25 rangiranje.",{"data":516,"type":252},{"code":517},"import sqlite3\n\nconnection = sqlite3.connect(\"knowledge.db\")\ncursor = connection.cursor()\n\ncursor.execute(\n    \"CREATE VIRTUAL TABLE IF NOT EXISTS docs USING fts5(title, body)\"\n)\n\ncursor.execute(\n    \"INSERT INTO docs(title, body) VALUES (?, ?)\",\n    (\"AKM\", \"The AKM uses 7.62 mm ammunition.\")\n)\n\nconnection.commit()\n\nquery = \"AKM ammunition\"\n\nrows = cursor.execute(\n    \"SELECT title, body, bm25(docs) AS score \"\n    \"FROM docs WHERE docs MATCH ? \"\n    \"ORDER BY score LIMIT 5\",\n    (query,)\n).fetchall()\n\nfor row in rows:\n    print(row)\n\nconnection.close()",{"data":519,"type":217},{"text":520},"Leksička pretraga je posebno korisna kada su tačna terminologija, kodovi proizvoda, imena, identifikatori ili reči specifične za domen važni. Semantička pretraga nije automatski bolja. Produkcioni sistemi često kombinuju oba signala.",{"data":522,"type":42},{"text":523,"level":230},"Stvarni primer 4: Semantički dohvat sa ugrađivanjima",{"data":525,"type":217},{"text":526},"Ugrađivanja pretvaraju tekst u numeričke vektore tako da semantički povezani odlomci mogu biti upoređeni čak i kada ne koriste identične reči. Sentence Transformers pruža jednostavnu lokalnu implementaciju.",{"data":528,"type":252},{"code":529},"# pip install sentence-transformers\n\nfrom sentence_transformers import SentenceTransformer, util\n\ndocuments = [\n    \"The AKM uses 7.62 mm ammunition.\",\n    \"A Med Kit restores health.\",\n    \"Vehicle maintenance includes checking oil, brakes and tires.\",\n    \"Account recovery requires access to the registered email address.\"\n]\n\nmodel = SentenceTransformer(\n    \"sentence-transformers\u002Fmulti-qa-mpnet-base-cos-v1\"\n)\n\ndocument_embeddings = model.encode_document(\n    documents,\n    convert_to_tensor=True\n)\n\nquestion = \"How do I repair my car?\"\n\nquery_embedding = model.encode_query(\n    question,\n    convert_to_tensor=True\n)\n\nhits = util.semantic_search(\n    query_embedding,\n    document_embeddings,\n    top_k=2\n)[0]\n\nfor hit in hits:\n    print(round(float(hit[\"score\"]), 3), documents[hit[\"corpus_id\"]])",{"data":531,"type":217},{"text":532},"Upit ne sadrži frazu „održavanje vozila“, ali semantički model i dalje može visoko rangirati taj odlomak jer su koncepti povezani. To je praktični razlog zašto su ugrađivanja uobičajena u RAG sistemima.",{"data":534,"type":217},{"text":535},"Za male kolekcije, ugrađivanja mogu ostati u memoriji. Veći sistemi ih obično čuvaju u indeksu ili bazi podataka sa podrškom za vektore i tamo obavljaju pretragu najbližih suseda. Skladištenje se menja, ali logika ostaje: kodirajte pitanje, pronađite relevantne reprezentacije dokumenata, vratite najbolje dokaze.",{"data":537,"type":42},{"text":538,"level":230},"Stvarni primer 5: Izgradite kontekst za LLM",{"data":540,"type":217},{"text":541},"Retriver treba da vrati dokaze. LLM zatim treba da primi pitanje plus te dokaze. Održavanje pretrage i generisanja odvojenim čini oba lakšim za pregled i testiranje.",{"data":543,"type":252},{"code":544},"def build_prompt(question, retrieved_documents):\n    context = \"\\n\\n\".join(\n        f'[{doc[\"id\"]}] {doc[\"text\"]}'\n        for doc in retrieved_documents\n    )\n\n    return f\"\"\"\nAnswer the question using the supplied context.\n\nRules:\n- Do not invent facts that are not supported by the context.\n- If the context is insufficient, say so.\n- Cite the source IDs you used.\n\nQuestion:\n{question}\n\nContext:\n{context}\n\"\"\".strip()",{"data":546,"type":217},{"text":547},"Instrukcija ne čini model nepogrešivim. Ona jednostavno stvara eksplicitnu granicu dokaza. Model i dalje može pogrešno razumeti dobre dokaze, ignorisati uslov ili preterano generalizovati. Zato se kvalitet pretrage i kvalitet generisanja moraju evaluirati odvojeno.",{"data":549,"type":42},{"text":550,"level":230},"Stvarni primer 6: Kompletan minimalni pipeline",{"data":552,"type":252},{"code":553},"def answer_question(question, all_documents, call_llm):\n    # 1. Retrieve evidence\n    retrieved = retrieve(question, all_documents, top_k=3)\n\n    # 2. Build model context\n    prompt = build_prompt(question, retrieved)\n\n    # 3. Generate the answer\n    answer = call_llm(prompt)\n\n    return {\n        \"answer\": answer,\n        \"sources\": [doc[\"id\"] for doc in retrieved]\n    }",{"data":555,"type":217},{"text":556},"Funkcija namerno prima \u003Ccode>call_llm\u003C\u002Fcode> kao zavisnost. Pretragu ne treba da zanima da li generisanje obavlja cloud model, lokalni model ili drugi provajder. Putanja podataka pripada aplikaciji.",{"data":558,"type":42},{"text":559,"level":478},"Opcioni generator: OpenAI Responses API",{"data":561,"type":217},{"text":562},"Jedan mogući generator je OpenAI Responses API. Održavanje imena modela u promenljivoj okruženja izbegava hardkodiranje određenog modela u RAG arhitekturu.",{"data":564,"type":252},{"code":565},"# pip install openai\n\nimport os\nfrom openai import OpenAI\n\nclient = OpenAI()\n\ndef call_llm(prompt):\n    response = client.responses.create(\n        model=os.environ[\"OPENAI_MODEL\"],\n        input=prompt,\n    )\n    return response.output_text",{"data":567,"type":217},{"text":568},"Isti pipeline za pretragu može se povezati sa lokalnim serverom za inferenciju. Ovo je važna arhitektonska tačka: \u003Cb>RAG nije vlasništvo LLM provajdera.\u003C\u002Fb> Aplikacija poseduje izvor, pretragu i sastavljanje konteksta.",{"data":570,"type":42},{"text":571,"level":478},"Cela arhitektura u jednom pregledu",{"data":573,"type":252},{"code":574},"USER QUESTION\n     |\n     v\n+-------------+\n|  Retriever  |\n+-------------+\n   |       |\n   |       +----> SQL \u002F API \u002F state query\n   |\n   +------------> keyword \u002F full-text search\n   |\n   +------------> embedding \u002F vector search\n                     |\n                     v\n              relevant evidence\n                     |\n                     v\n+-----------------------------------+\n| question + evidence + instructions |\n+-----------------------------------+\n                     |\n                     v\n                   LLM\n                     |\n                     v\n                  answer",{"data":576,"type":217},{"text":577},"Ovaj model toka podataka je trajniji od memorisanja jednog framework-a. Biblioteke, baze podataka i prodavci modela će se menjati; granice odgovornosti ostaju.",{"data":579,"type":42},{"text":580,"level":230},"Uobičajene zablude i načini neuspeha",{"data":582,"type":42},{"text":583,"level":478},"„RAG znači vektorska baza podataka.“",{"data":585,"type":217},{"text":586},"Ne. Vektorska pretraga je jedna metoda pretrage. RAG može koristiti full-text pretragu, SQL, API-je, grafove znanja, vektorsku pretragu ili njihove kombinacije. Definišući obrazac je pretraga eksternih informacija za generisanje.",{"data":588,"type":42},{"text":589,"level":478},"„Ako su podaci u PostgreSQL-u, moram ugraditi celu bazu podataka.“",{"data":591,"type":217},{"text":592},"Ne. Strukturirani zapisi obično treba da ostanu upitljivi kao strukturirani zapisi. Embeddings su korisni za semantičku relevantnost, ne kao zamena za determinističke upite.",{"data":594,"type":42},{"text":595,"level":478},"„Više delova znači bolji odgovor.“",{"data":597,"type":217},{"text":598},"Ne nužno. Dodatni kontekst može uneti šum, konfliktne verzije i irelevantan materijal. Pretraga treba da optimizuje za korisne dokaze, a ne za maksimalni obim.",{"data":600,"type":42},{"text":601,"level":478},"„Visok skor sličnosti dokazuje odgovor.“",{"data":603,"type":217},{"text":604},"Ne. Sličnost meri relevantnost, a ne istinitost ili primenljivost. Veoma sličan odlomak može biti zastareo, iz pogrešne verzije proizvoda ili važeći samo pod uslovima koji se ne poklapaju sa pitanjem.",{"data":606,"type":42},{"text":607,"level":478},"„Kada se preuzme ispravan deo, halucinacija je rešena.“",{"data":609,"type":217},{"text":610},"Ne. Pretraga poboljšava utemeljenost, ali ne garantuje verodostojno zaključivanje. Generisanje i dalje zahteva evaluaciju, a radni tokovi visokog rizika mogu zahtevati determinističku validaciju ili ljudski pregled.",{"data":612,"type":42},{"text":613,"level":478},"„Model je zakazao, pa promenite model.“",{"data":615,"type":217},{"text":616},"Ne nužno. Ispravan izvor je možda nedostajao, bio pogrešno parsiran, loše podeljen, filtriran, rangiran prenisko ili izostavljen iz sastavljenog konteksta. Zamena modela ne bi trebalo da bude prvi dijagnostički korak.",{"data":618,"type":42},{"text":619,"level":230},"Rubni slučajevi",{"data":621,"type":357},{"items":622,"style":356},[623,624,625,626,627,628,629,630],"\u003Cb>Konfliktni dokumenti:\u003C\u002Fb> dva izvora se mogu ne slagati jer se verzije, jurisdikcije ili proizvodi razlikuju.","\u003Cb>Vremenski osetljive činjenice:\u003C\u002Fb> semantički relevantan izvor može već biti zastareo.","\u003Cb>Dozvole:\u003C\u002Fb> pretraživač ne sme da vrati dokumente kojima trenutni korisnik nije ovlašćen da pristupi.","\u003Cb>Višejezične kolekcije:\u003C\u002Fb> model za ugrađivanje i strategija pretrage moraju podržavati jezike koji se stvarno koriste.","\u003Cb>Tabele i izvorni kod:\u003C\u002Fb> obično deljenje na pasuse može uništiti strukturu koja je ključna za odgovor.","\u003Cb>Vrlo kratki identifikatori:\u003C\u002Fb> semantička pretraga može biti slabija od tačnog poklapanja za SKU-ove, ID-ove, kodove grešaka ili akronime.","\u003Cb>Duga pitanja koja zahtevaju nekoliko činjenica:\u003C\u002Fb> pretraga može zahtevati dekompoziciju, nekoliko pretraga ili ponovno rangiranje umesto jednog top-k upita.","\u003Cb>Hijerarhija izvora:\u003C\u002Fb> zvanična aktuelna politika može trebati da bude rangirana iznad starijeg, ali semantički bližeg diskusionog dokumenta.",{"data":632,"type":42},{"text":633,"level":230},"Ograničenja",{"data":635,"type":217},{"text":636},"Python primeri namerno optimizuju za transparentnost, a ne za skalabilnost. Pretraživač po ključnim rečima je naivan, delilac koristi dužinu znakova, SQLite primeri ne uključuju upravljanje vezama za produkciju, a semantički primer drži sve ugrađene vektore u memoriji.",{"data":638,"type":217},{"text":639},"Produkcijski sistem može zahtevati vektorske indekse, ponovne rangere, hibridnu pretragu, parsere dokumenata, keširanje, inkrementalno indeksiranje, verzionisanje izvora, filtere kontrole pristupa, observabilnost, skupove podataka za evaluaciju i rukovanje greškama. Nijedan od tih dodataka ne menja osnovnu arhitekturu; oni čine svaku granicu pouzdanijom.",{"data":641,"type":217},{"text":642},"RAG takođe ne može da stvori dokaze koji ne postoje u skupu izvora. Ako je izvor pogrešan, nepotpun ili zastareo, bolji model za ugrađivanje ne može ga pretvoriti u autoritativno znanje.",{"data":644,"type":42},{"text":645,"level":230},"Šta bi promenilo ovaj odgovor?",{"data":647,"type":217},{"text":648},"Arhitektura se menja kada zadatak zahteva više od pronalaženja znanja. Status porudžbine uživo zahteva trenutno stanje. Finansijski izračun može zahtevati deterministički kod. Zadatak veb-istraživanja može zahtevati aktivnu pretragu. Radni tok može zahtevati alate koji mogu da upišu podatke nazad u drugi sistem. Autonomni agent može zahtevati planiranje, dozvole i kontrolu izvršavanja pored pretrage.",{"data":650,"type":217},{"text":651},"RAG je stoga najbolje razumeti kao \u003Cb>jedan sloj za prikupljanje dokaza unutar većeg AI sistema\u003C\u002Fb>. Moćan je upravo zato što ima uzak zadatak: pronaći korisne spoljne informacije i smestiti ih u radni kontekst modela.",{"data":653,"type":42},{"text":654,"level":230},"Zaključak",{"data":656,"type":217},{"text":657},"RAG postaje mnogo lakši za razumevanje kada se uklone nazivi tehnologija. Datoteka je izvor. Baza podataka je izvor. API je izvor. Funkcija pretrage pronalazi dokaze. Upit nosi te dokaze do modela. LLM ih zatim tumači i proizvodi jezik.",{"data":659,"type":217},{"text":660},"Teži deo produkcijskog RAG-a nije pozivanje modela za ugrađivanje. To je izgradnja pouzdane putanje dokaza od originalnog izvora do konačne tvrdnje: očuvanje porekla, odabir prave metode pretrage, održavanje informacija ažurnim, kontrola pristupa, evaluacija pretrage odvojeno od generisanja i znanje kada je direktan poziv baze podataka ili alata bolji od semantičke pretrage.",{"data":662,"type":217},{"text":663},"To je praktični nastavak osnovnog RAG modela: \u003Cb>prvo razumeti uloge, zatim učiniti putanju podataka eksplicitnom.\u003C\u002Fb>",{"data":665,"type":42},{"text":666,"level":230},"Primarni izvori",{"data":668,"type":357},{"items":669,"style":356},[670,671,672,673,674,675],"\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2005.11401\">Lewis et al. — Retrieval-Augmented Generation for Knowledge-Intensive NLP Tasks\u003C\u002Fa> — rad iz 2020. koji uvodi RAG formulaciju koja kombinuje generisanje sa pronađenom neparametarskom memorijom.","\u003Ca href=\"https:\u002F\u002Fwww.sbert.net\u002Fexamples\u002Fsentence_transformer\u002Fapplications\u002Fsemantic-search\u002FREADME.html\">Sentence Transformers — Semantic Search\u003C\u002Fa> — zvanična dokumentacija za semantičku pretragu, ugrađivanje upita i ugrađivanje dokumenata.","\u003Ca href=\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fembeddings\">OpenAI — Vector Embeddings\u003C\u002Fa> — zvanična dokumentacija koja opisuje ugrađivanja kao numeričke reprezentacije koje se koriste za povezanost i pretragu.","\u003Ca href=\"https:\u002F\u002Fwww.sqlite.org\u002Ffts5.html\">SQLite — FTS5 Extension\u003C\u002Fa> — zvanična dokumentacija za pretragu punog teksta i BM25 rangiranje u SQLite-u.","\u003Ca href=\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Flibraries\">OpenAI — SDKs and CLI\u003C\u002Fa> — zvanični primer Python SDK-a za Responses API koji se koristi u opcionom primeru generatora.","\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fsr\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works\">What Is RAG? The Simplest Explanation of How It Works\u003C\u002Fa> — konceptualni prvi deo ove serije.","2.31","LLM ne zna magično vaše fajlove, baze podataka ili API-je. Ovaj praktični nastavak RAG serije pokazuje, uz jednostavan Python, kako eksterni podaci postaju dokazi koji se mogu pronaći: od tekstualnih fajlova i SQL-a do pretrage punog teksta, embeddinga, sastavljanja konteksta i konačnog LLM poziva.","\u002Fuploads\u002F2026\u002F09\u002Fwhere-does-an-llm-get-its-data-rag-data-sources-in-python-1790517200521-nfsi5i.webp","where-does-an-llm-get-its-data-rag-data-sources-in-python-1790517200521-nfsi5i","PUBLISHED","2026-09-27T09:51:00.000Z","2026-09-27T13:51:43.843Z","2026-09-27T13:58:03.762Z",{"en":685,"de":686,"sr":687,"es":688,"fr":689,"it":690,"ru":691,"zh":692},"\u002Fblog\u002Fwhere-does-an-llm-get-its-data-rag-data-sources-in-python","\u002Fde\u002Fblog\u002Fwhere-does-an-llm-get-its-data-rag-data-sources-in-python","\u002Fsr\u002Fblog\u002Fwhere-does-an-llm-get-its-data-rag-data-sources-in-python","\u002Fes\u002Fblog\u002Fwhere-does-an-llm-get-its-data-rag-data-sources-in-python","\u002Ffr\u002Fblog\u002Fwhere-does-an-llm-get-its-data-rag-data-sources-in-python","\u002Fit\u002Fblog\u002Fwhere-does-an-llm-get-its-data-rag-data-sources-in-python","\u002Fru\u002Fblog\u002Fwhere-does-an-llm-get-its-data-rag-data-sources-in-python","\u002Fzh\u002Fblog\u002Fwhere-does-an-llm-get-its-data-rag-data-sources-in-python",[],{"id":695,"login":696,"email":697,"displayName":698},"20","rooth8233","aleksandar@stajic.de","Aleksandar Stajić",[700,1146],{"lang":701,"title":702,"content":703,"contentJson":704,"excerpt":1145},"en","Where Does an LLM Get Its Data? RAG Data Sources in Python","{\"time\":1790516400000,\"blocks\":[{\"data\":{\"text\":\"The previous article, \u003Ca href=\\\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works\\\">What Is RAG? The Simplest Explanation of How It Works\u003C\u002Fa>, established the mental model: the LLM writes, RAG retrieves useful knowledge, the application owns current state, and tools perform actions. This article takes the next step: \u003Cb>where does the data actually come from, and what does retrieval look like in Python?\u003C\u002Fb>\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"The important surprise is that an “LLM data source” is usually nothing exotic. It can be a text file, a folder of Markdown documents, a SQL database, an API response, a product catalog, a support system, or a vector index derived from those sources. The AI does not magically know these systems. Your application has to load, query, search, or retrieve the relevant data and place the result into the model’s context.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Data source = where information lives. Retrieval = how the application finds useful information. Context = the selected information given to the model. LLM = the component that interprets that context and generates an answer.\",\"caption\":\"The four-part model used throughout this article\",\"alignment\":\"left\"},\"type\":\"quote\"},{\"data\":{\"text\":\"Question\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"How does an LLM use external data such as files, databases or APIs, and how can a small Python program implement the essential RAG steps without hiding them behind a framework?\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"What This Really Means\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"When developers say that an LLM is “connected to company data,” several different operations may be hidden behind that sentence. One application may execute SQL. Another may call an API. Another may run full-text search. Another may calculate embedding similarity over document chunks. All of them can provide external information to an LLM, but they are not the same retrieval method and they should not be treated as interchangeable.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"This distinction matters because the best retrieval method depends on the shape of the question. “What is our refund policy?” is a document-retrieval problem. “What is order 4711’s current status?” is usually a structured database lookup. “Which paragraph discusses account recovery?” can be keyword or semantic search. RAG is most useful when the system must \u003Cb>discover relevant knowledge before generation\u003C\u002Fb>.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Simplest Example\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"Start with three strings in ordinary Python. There is no vector database, no framework, and no LLM yet. We only want to make the retrieval step visible.\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"documents = [\\n    \\\"The AKM uses 7.62 mm ammunition.\\\",\\n    \\\"A Med Kit restores health.\\\",\\n    \\\"A 4x scope can be attached to several compatible weapons.\\\"\\n]\\n\\nquestion = \\\"Which ammunition does the AKM use?\\\"\\n\\nfor document in documents:\\n    if \\\"AKM\\\" in document:\\n        print(document)\"},\"type\":\"code\"},{\"data\":{\"text\":\"The program prints the first sentence because it contains the term we searched for. This is primitive retrieval, but the architecture is already visible: \u003Cb>question → search → relevant text\u003C\u002Fb>. RAG adds one more major step: pass the retrieved text to a language model together with the question.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"A slightly more general version ranks documents by overlapping query terms:\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"import re\\n\\ndocuments = [\\n    {\\\"id\\\": \\\"weapon-akm\\\", \\\"text\\\": \\\"The AKM uses 7.62 mm ammunition.\\\"},\\n    {\\\"id\\\": \\\"healing-medkit\\\", \\\"text\\\": \\\"A Med Kit restores health.\\\"},\\n    {\\\"id\\\": \\\"scope-4x\\\", \\\"text\\\": \\\"A 4x scope can be attached to several compatible weapons.\\\"},\\n]\\n\\ndef words(text):\\n    return set(re.findall(r\\\"[a-zA-Z0-9.]+\\\", text.lower()))\\n\\ndef retrieve(question, documents, top_k=2):\\n    query_terms = words(question)\\n    ranked = []\\n\\n    for document in documents:\\n        score = len(query_terms & words(document[\\\"text\\\"]))\\n        if score > 0:\\n            ranked.append((score, document))\\n\\n    ranked.sort(key=lambda item: item[0], reverse=True)\\n    return [document for _, document in ranked[:top_k]]\\n\\nquestion = \\\"Which ammunition does the AKM use?\\\"\\nhits = retrieve(question, documents)\\n\\nfor hit in hits:\\n    print(hit[\\\"id\\\"], \\\"->\\\", hit[\\\"text\\\"])\"},\"type\":\"code\"},{\"data\":{\"text\":\"This is not a production search engine. It ignores morphology, synonyms, spelling variants, document length and many ranking signals. Its value is educational: \u003Cb>RAG does not begin with a vector database. It begins with retrieval.\u003C\u002Fb>\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Where the Example Stops Working\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"Exact or lexical matching becomes weak when the question and the source use different words. A document may say “vehicle maintenance,” while the user asks “how do I repair my car?” A lexical retriever can miss the relationship even though a human sees it immediately. Semantic retrieval addresses this by representing text as vectors and comparing meaning rather than only exact tokens.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Long files create another problem. Searching an entire 80-page manual as one unit is too coarse, but splitting every sentence can destroy useful context. Real RAG systems therefore need decisions about parsing, chunking, metadata, ranking, freshness, permissions and provenance.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"The example also says nothing about structured live facts. If the user asks for the current status of order 4711 and the application already has a database key, semantic search is usually the wrong first tool. A deterministic database query is better.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Direct Answer\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"An LLM data source is any external system from which an application can obtain information for the model: files, databases, APIs, search indexes, vector stores or live application state. RAG is the pattern of \u003Cb>retrieving relevant knowledge from such sources before generation\u003C\u002Fb>.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"In Python, the essential pipeline can be very small: \u003Cb>load data → create retrievable units → find relevant evidence → assemble context → call the LLM\u003C\u002Fb>. The retrieval method should match the source and the question. Use SQL for exact structured facts, full-text search for lexical matching, embeddings for semantic similarity, and hybrid retrieval when several signals are valuable.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Why This Is So\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"A language model does not automatically receive the contents of your filesystem, PostgreSQL database, CRM, private API or newly edited document. The application decides what external information is accessible and what is placed into the model’s current context.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"The original Retrieval-Augmented Generation work by Lewis et al. combined a generative model with external non-parametric memory retrieved from a dense vector index. The broader architectural idea survives beyond that specific implementation: external evidence can be retrieved at inference time instead of expecting all useful knowledge to be encoded in model parameters.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"This creates a useful separation of responsibilities: the source stores information, the retriever selects evidence, the context carries that evidence into the request, and the model interprets it. Keeping those boundaries visible makes failures much easier to diagnose.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Context: The Main Types of Data Sources\",\"level\":2},\"type\":\"header\"},{\"data\":{\"content\":[[\"Source\",\"Typical retrieval method\",\"Good for\"],[\"TXT \u002F Markdown \u002F HTML\",\"Parsing + lexical or semantic search\",\"Documentation, manuals, articles, notes\"],[\"PDF \u002F DOCX\",\"Structure-aware extraction + search\",\"Policies, reports, contracts, manuals\"],[\"SQL database\",\"SQL query or filtered retrieval\",\"Orders, users, products, structured records\"],[\"REST \u002F GraphQL API\",\"HTTP request with parameters\",\"Remote systems and live service data\"],[\"Search index\",\"BM25 \u002F full-text \u002F hybrid search\",\"Large text collections\"],[\"Vector index\",\"Embedding similarity\",\"Semantic document retrieval\"],[\"Application state\",\"Direct state read or tool call\",\"What is true right now\"]],\"withHeadings\":true},\"type\":\"table\"},{\"data\":{\"text\":\"A vector index deserves special attention. In many architectures it is \u003Cb>not the canonical source of truth\u003C\u002Fb>. It is a retrieval index derived from documents or records. The authoritative document may live in object storage, a CMS, Git, PostgreSQL or another system, while embeddings and metadata are stored separately for fast semantic lookup. Some systems do use a vector store as primary storage, but that is an architectural choice rather than a requirement of RAG.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"If the boundary between retrieval, persistent memory, current state and model context is still unclear, see \u003Ca href=\\\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context\\\">AI Agent Memory Is Not RAG\u003C\u002Fa>. Those layers can use some of the same storage technologies while still having different correctness rules.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Assumptions\",\"level\":2},\"type\":\"header\"},{\"data\":{\"items\":[\"The application is allowed to access the external source.\",\"The relevant source contains enough information to answer the question.\",\"The data can be parsed or queried in a form the retrieval layer can use.\",\"The retrieved information is fresh enough for the requested decision.\",\"The model receives the selected evidence in its context.\",\"Authorization is enforced before protected evidence reaches the model.\",\"The generation model can still be wrong even when retrieval is correct.\"],\"style\":\"unordered\"},\"type\":\"list\"},{\"data\":{\"text\":\"These assumptions matter because retrieval cannot compensate for missing evidence, stale source versions, broken parsers or unauthorized access. A RAG pipeline can only be as trustworthy as the evidence path that feeds it.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Variables\",\"level\":2},\"type\":\"header\"},{\"data\":{\"content\":[[\"Variable\",\"Why it changes the design\"],[\"Source structure\",\"A SQL table, legal PDF and source-code repository need different retrieval strategies\"],[\"Question type\",\"Exact lookup, conceptual search and multi-hop research are different tasks\"],[\"Freshness requirement\",\"Live state may need direct queries instead of periodically rebuilt indexes\"],[\"Corpus size\",\"In-memory search may work for hundreds of chunks but not for very large collections\"],[\"Language\",\"Multilingual retrieval requires models and tokenization suitable for the actual languages\"],[\"Permissions\",\"Retrieval must filter by the current user’s access rights\"],[\"Latency and cost\",\"More retrieval stages can improve quality but add runtime and infrastructure cost\"],[\"Need for provenance\",\"High-trust systems need source IDs, versions and traceable evidence\"]],\"withHeadings\":true},\"type\":\"table\"},{\"data\":{\"text\":\"Diagnostic \u002F Decision Method\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"The first decision is not “Which vector database should I install?” It is: \u003Cb>What kind of fact am I trying to retrieve?\u003C\u002Fb>\"},\"type\":\"paragraph\"},{\"data\":{\"content\":[[\"Question type\",\"Preferred first approach\",\"Reason\"],[\"Exact ID or current record\",\"SQL \u002F key lookup \u002F API\",\"Deterministic structured access\"],[\"Exact wording, codes, names\",\"Full-text or keyword search\",\"Lexical precision\"],[\"Conceptual question over documents\",\"Semantic vector search\",\"Meaning can differ from wording\"],[\"Mixed enterprise knowledge\",\"Hybrid retrieval + metadata filters\",\"Combines lexical and semantic signals\"],[\"Current application state\",\"Direct state\u002Ftool access\",\"Freshness matters more than document similarity\"]],\"withHeadings\":true},\"type\":\"table\"},{\"data\":{\"text\":\"A useful test is: \u003Cb>Do I already know which record I need, or must the system discover which passage is relevant?\u003C\u002Fb> If the record is known, query it directly. If relevance must be discovered, search becomes more important.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"When an answer is wrong, diagnose the pipeline in order instead of immediately changing the LLM:\"},\"type\":\"paragraph\"},{\"data\":{\"items\":[\"\u003Cb>1. Source coverage:\u003C\u002Fb> Does the correct information exist in the accessible source set?\",\"\u003Cb>2. Freshness:\u003C\u002Fb> Is that version current enough for the question?\",\"\u003Cb>3. Parsing:\u003C\u002Fb> Was the relevant content extracted correctly?\",\"\u003Cb>4. Chunking:\u003C\u002Fb> Did the evidence stay together with the conditions that give it meaning?\",\"\u003Cb>5. Retrieval:\u003C\u002Fb> Does the correct chunk appear among the candidates?\",\"\u003Cb>6. Ranking:\u003C\u002Fb> Are stronger sources ranked above weaker or conflicting ones?\",\"\u003Cb>7. Context assembly:\u003C\u002Fb> Did the application actually send the selected evidence to the model?\",\"\u003Cb>8. Generation:\u003C\u002Fb> Did the LLM faithfully use the supplied evidence?\",\"\u003Cb>9. Attribution:\u003C\u002Fb> Can each important claim be traced to a source?\"],\"style\":\"ordered\"},\"type\":\"list\"},{\"data\":{\"text\":\"For a deeper production-debugging method, see \u003Ca href=\\\"https:\u002F\u002Fstajic.de\u002Fblog\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method\\\">RAG Failed — But Which Layer Actually Failed? A Diagnostic Method\u003C\u002Fa>, which expands this chain into independently testable failure layers.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Evidence\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"The RAG paper by Lewis et al. formalized generation that conditions on retrieved external memory rather than relying only on model parameters. That provides the conceptual foundation for separating the generator from a retrievable knowledge source.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Sentence Transformers documents semantic search as embedding the corpus and the query into a vector space and retrieving items with high semantic similarity. Its current API also distinguishes query encoding from document encoding for retrieval tasks.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"SQLite FTS5 demonstrates the other side of the spectrum: mature full-text retrieval can rank documents without embeddings. This matters because lexical search remains valuable for identifiers, exact terminology and many hybrid retrieval designs.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"OpenAI’s embeddings documentation describes embeddings as numerical vector representations used for relatedness and search. This is one implementation path for semantic retrieval, not the definition of RAG itself.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Real Example 1: A Folder of Text Files\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"Suppose a directory named \u003Ccode>knowledge\u002F\u003C\u002Fcode> contains ordinary text files. Python can load them with no AI library at all.\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"from pathlib import Path\\n\\ndef load_text_files(folder=\\\"knowledge\\\"):\\n    documents = []\\n\\n    for path in Path(folder).glob(\\\"*.txt\\\"):\\n        documents.append({\\n            \\\"source\\\": path.name,\\n            \\\"text\\\": path.read_text(encoding=\\\"utf-8\\\")\\n        })\\n\\n    return documents\\n\\ndocuments = load_text_files()\\n\\nfor document in documents:\\n    print(document[\\\"source\\\"], len(document[\\\"text\\\"]))\"},\"type\":\"code\"},{\"data\":{\"text\":\"The filesystem is the data source. The next question is how much text should become one retrievable unit. For long documents, searching one complete file is often too coarse. This is why RAG pipelines commonly create chunks.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"A Very Simple Chunker\",\"level\":3},\"type\":\"header\"},{\"data\":{\"code\":\"def chunk_text(text, max_chars=800):\\n    paragraphs = [p.strip() for p in text.split(\\\"\\\\n\\\\n\\\") if p.strip()]\\n\\n    chunks = []\\n    current = \\\"\\\"\\n\\n    for paragraph in paragraphs:\\n        candidate = f\\\"{current}\\\\n\\\\n{paragraph}\\\".strip()\\n\\n        if current and len(candidate) > max_chars:\\n            chunks.append(current)\\n            current = paragraph\\n        else:\\n            current = candidate\\n\\n    if current:\\n        chunks.append(current)\\n\\n    return chunks\"},\"type\":\"code\"},{\"data\":{\"text\":\"This example groups paragraphs until a rough character limit is reached. It is intentionally understandable rather than optimal. Production systems often chunk by tokens, headings, sections, sentence boundaries or document structure. Tables, source code, contracts and API documentation may need different strategies.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Preserve Provenance While Chunking\",\"level\":3},\"type\":\"header\"},{\"data\":{\"code\":\"def build_chunks(documents):\\n    chunks = []\\n\\n    for document in documents:\\n        for index, text in enumerate(chunk_text(document[\\\"text\\\"])):\\n            chunks.append({\\n                \\\"id\\\": f'{document[\\\"source\\\"]}:{index}',\\n                \\\"source\\\": document[\\\"source\\\"],\\n                \\\"chunk\\\": index,\\n                \\\"text\\\": text,\\n            })\\n\\n    return chunks\"},\"type\":\"code\"},{\"data\":{\"text\":\"A useful chunk carries more than text. Source name, document ID, URL, timestamp, version or section can later support citation, debugging and freshness checks. If provenance is lost during ingestion, it becomes much harder to explain why a particular answer was produced.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Real Example 2: Structured Data — Use SQL When SQL Is the Right Tool\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"Not every external fact should go through semantic search. If the question asks for an exact current record, a direct database query is usually clearer and more deterministic.\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"import sqlite3\\n\\ndef get_order_status(order_id):\\n    connection = sqlite3.connect(\\\"shop.db\\\")\\n    cursor = connection.cursor()\\n\\n    cursor.execute(\\n        \\\"SELECT status, total, currency FROM orders WHERE id = ?\\\",\\n        (order_id,)\\n    )\\n\\n    row = cursor.fetchone()\\n    connection.close()\\n\\n    if row is None:\\n        return None\\n\\n    return {\\n        \\\"order_id\\\": order_id,\\n        \\\"status\\\": row[0],\\n        \\\"total\\\": row[1],\\n        \\\"currency\\\": row[2],\\n    }\\n\\nprint(get_order_status(4711))\"},\"type\":\"code\"},{\"data\":{\"text\":\"If the application already knows that the user is asking about order 4711, embedding the entire orders table and asking semantic search to rediscover that row usually adds complexity without benefit. A strong design rule is: \u003Cb>retrieve structured facts with structured queries; retrieve unstructured knowledge with search.\u003C\u002Fb>\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"The returned database row can still be placed into the model context so the LLM can explain it in natural language. But direct state or record access is conceptually different from searching a knowledge corpus.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Real Example 3: Full-Text Search Before Embeddings\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"Between a naive Python loop and vector search lies a mature class of lexical retrieval systems. SQLite includes FTS5 for full-text search, including BM25 ranking.\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"import sqlite3\\n\\nconnection = sqlite3.connect(\\\"knowledge.db\\\")\\ncursor = connection.cursor()\\n\\ncursor.execute(\\n    \\\"CREATE VIRTUAL TABLE IF NOT EXISTS docs USING fts5(title, body)\\\"\\n)\\n\\ncursor.execute(\\n    \\\"INSERT INTO docs(title, body) VALUES (?, ?)\\\",\\n    (\\\"AKM\\\", \\\"The AKM uses 7.62 mm ammunition.\\\")\\n)\\n\\nconnection.commit()\\n\\nquery = \\\"AKM ammunition\\\"\\n\\nrows = cursor.execute(\\n    \\\"SELECT title, body, bm25(docs) AS score \\\"\\n    \\\"FROM docs WHERE docs MATCH ? \\\"\\n    \\\"ORDER BY score LIMIT 5\\\",\\n    (query,)\\n).fetchall()\\n\\nfor row in rows:\\n    print(row)\\n\\nconnection.close()\"},\"type\":\"code\"},{\"data\":{\"text\":\"Lexical search is especially useful when exact terminology, product codes, names, identifiers or domain-specific words matter. Semantic search is not automatically better. Production systems often combine both signals.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Real Example 4: Semantic Retrieval With Embeddings\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"Embeddings turn text into numerical vectors so semantically related passages can be compared even when they do not use identical wording. Sentence Transformers provides a straightforward local implementation.\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"# pip install sentence-transformers\\n\\nfrom sentence_transformers import SentenceTransformer, util\\n\\ndocuments = [\\n    \\\"The AKM uses 7.62 mm ammunition.\\\",\\n    \\\"A Med Kit restores health.\\\",\\n    \\\"Vehicle maintenance includes checking oil, brakes and tires.\\\",\\n    \\\"Account recovery requires access to the registered email address.\\\"\\n]\\n\\nmodel = SentenceTransformer(\\n    \\\"sentence-transformers\u002Fmulti-qa-mpnet-base-cos-v1\\\"\\n)\\n\\ndocument_embeddings = model.encode_document(\\n    documents,\\n    convert_to_tensor=True\\n)\\n\\nquestion = \\\"How do I repair my car?\\\"\\n\\nquery_embedding = model.encode_query(\\n    question,\\n    convert_to_tensor=True\\n)\\n\\nhits = util.semantic_search(\\n    query_embedding,\\n    document_embeddings,\\n    top_k=2\\n)[0]\\n\\nfor hit in hits:\\n    print(round(float(hit[\\\"score\\\"]), 3), documents[hit[\\\"corpus_id\\\"]])\"},\"type\":\"code\"},{\"data\":{\"text\":\"The query does not contain the phrase “vehicle maintenance,” but a semantic model can still rank that passage highly because the concepts are related. This is the practical reason embeddings are common in RAG systems.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"For small collections, embeddings can stay in memory. Larger systems usually persist them in a vector-capable index or database and perform nearest-neighbor search there. The storage changes, but the logic remains: encode the question, find relevant document representations, return the best evidence.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Real Example 5: Build the Context for the LLM\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"A retriever should return evidence. The LLM should then receive the question plus that evidence. Keeping retrieval and generation separate makes both easier to inspect and test.\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"def build_prompt(question, retrieved_documents):\\n    context = \\\"\\\\n\\\\n\\\".join(\\n        f'[{doc[\\\"id\\\"]}] {doc[\\\"text\\\"]}'\\n        for doc in retrieved_documents\\n    )\\n\\n    return f\\\"\\\"\\\"\\nAnswer the question using the supplied context.\\n\\nRules:\\n- Do not invent facts that are not supported by the context.\\n- If the context is insufficient, say so.\\n- Cite the source IDs you used.\\n\\nQuestion:\\n{question}\\n\\nContext:\\n{context}\\n\\\"\\\"\\\".strip()\"},\"type\":\"code\"},{\"data\":{\"text\":\"The instruction does not make the model infallible. It simply creates an explicit evidence boundary. The model can still misunderstand good evidence, ignore a condition or overgeneralize. That is why retrieval quality and generation quality must be evaluated separately.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Real Example 6: A Complete Minimal Pipeline\",\"level\":2},\"type\":\"header\"},{\"data\":{\"code\":\"def answer_question(question, all_documents, call_llm):\\n    # 1. Retrieve evidence\\n    retrieved = retrieve(question, all_documents, top_k=3)\\n\\n    # 2. Build model context\\n    prompt = build_prompt(question, retrieved)\\n\\n    # 3. Generate the answer\\n    answer = call_llm(prompt)\\n\\n    return {\\n        \\\"answer\\\": answer,\\n        \\\"sources\\\": [doc[\\\"id\\\"] for doc in retrieved]\\n    }\"},\"type\":\"code\"},{\"data\":{\"text\":\"The function receives \u003Ccode>call_llm\u003C\u002Fcode> as a dependency on purpose. Retrieval should not care whether generation is performed by a cloud model, a local model or another provider. The data path belongs to the application.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Optional Generator: OpenAI Responses API\",\"level\":3},\"type\":\"header\"},{\"data\":{\"text\":\"One possible generator is the OpenAI Responses API. Keeping the model name in an environment variable avoids hard-coding a particular model into the RAG architecture.\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"# pip install openai\\n\\nimport os\\nfrom openai import OpenAI\\n\\nclient = OpenAI()\\n\\ndef call_llm(prompt):\\n    response = client.responses.create(\\n        model=os.environ[\\\"OPENAI_MODEL\\\"],\\n        input=prompt,\\n    )\\n    return response.output_text\"},\"type\":\"code\"},{\"data\":{\"text\":\"The same retrieval pipeline can be connected to a local inference server. This is an important architectural point: \u003Cb>RAG is not owned by the LLM provider.\u003C\u002Fb> The application owns the source, retrieval and context assembly.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"The Whole Architecture in One View\",\"level\":3},\"type\":\"header\"},{\"data\":{\"code\":\"USER QUESTION\\n     |\\n     v\\n+-------------+\\n|  Retriever  |\\n+-------------+\\n   |       |\\n   |       +----> SQL \u002F API \u002F state query\\n   |\\n   +------------> keyword \u002F full-text search\\n   |\\n   +------------> embedding \u002F vector search\\n                     |\\n                     v\\n              relevant evidence\\n                     |\\n                     v\\n+-----------------------------------+\\n| question + evidence + instructions |\\n+-----------------------------------+\\n                     |\\n                     v\\n                   LLM\\n                     |\\n                     v\\n                  answer\"},\"type\":\"code\"},{\"data\":{\"text\":\"This data-flow model is more durable than memorizing one framework. Libraries, databases and model vendors will change; the responsibility boundaries remain.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Common Misconceptions and Failure Modes\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"“RAG means vector database.”\",\"level\":3},\"type\":\"header\"},{\"data\":{\"text\":\"No. Vector search is one retrieval method. RAG can use full-text search, SQL, APIs, knowledge graphs, vector search or combinations of them. The defining pattern is retrieval of external information for generation.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"“If the data is in PostgreSQL, I must embed the whole database.”\",\"level\":3},\"type\":\"header\"},{\"data\":{\"text\":\"No. Structured records should usually remain queryable as structured records. Embeddings are useful for semantic relevance, not as a replacement for deterministic queries.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"“More chunks means a better answer.”\",\"level\":3},\"type\":\"header\"},{\"data\":{\"text\":\"Not necessarily. Extra context can introduce noise, conflicting versions and irrelevant material. Retrieval should optimize for useful evidence, not maximum volume.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"“A high similarity score proves the answer.”\",\"level\":3},\"type\":\"header\"},{\"data\":{\"text\":\"No. Similarity measures relevance, not truth or applicability. A highly similar passage can be outdated, from the wrong product version or valid only under conditions that do not match the question.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"“Once the correct chunk is retrieved, hallucination is solved.”\",\"level\":3},\"type\":\"header\"},{\"data\":{\"text\":\"No. Retrieval improves grounding but does not guarantee faithful reasoning. Generation still needs evaluation, and high-risk workflows may require deterministic validation or human review.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"“The model failed, so change the model.”\",\"level\":3},\"type\":\"header\"},{\"data\":{\"text\":\"Not necessarily. The correct source may have been missing, parsed incorrectly, split badly, filtered out, ranked too low or omitted from the assembled context. Model replacement should not be the first diagnostic step.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Edge Cases\",\"level\":2},\"type\":\"header\"},{\"data\":{\"items\":[\"\u003Cb>Conflicting documents:\u003C\u002Fb> two sources may disagree because versions, jurisdictions or products differ.\",\"\u003Cb>Time-sensitive facts:\u003C\u002Fb> a semantically relevant source may already be stale.\",\"\u003Cb>Permissions:\u003C\u002Fb> a retriever must not return documents the current user is not authorized to access.\",\"\u003Cb>Multi-language collections:\u003C\u002Fb> the embedding model and retrieval strategy must support the languages actually used.\",\"\u003Cb>Tables and source code:\u003C\u002Fb> plain paragraph chunking can destroy structure that is essential to the answer.\",\"\u003Cb>Very short identifiers:\u003C\u002Fb> semantic retrieval can be weaker than exact matching for SKUs, IDs, error codes or acronyms.\",\"\u003Cb>Long questions requiring several facts:\u003C\u002Fb> retrieval may need decomposition, several searches or reranking rather than one top-k query.\",\"\u003Cb>Source hierarchy:\u003C\u002Fb> an official current policy may need to outrank an older but semantically closer discussion document.\"],\"style\":\"unordered\"},\"type\":\"list\"},{\"data\":{\"text\":\"Limitations\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"The Python examples intentionally optimize for transparency, not scale. The keyword retriever is naive, the chunker uses character length, the SQLite examples do not include production connection management, and the semantic example keeps all embeddings in memory.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"A production system may require vector indexes, rerankers, hybrid retrieval, document parsers, caching, incremental indexing, source versioning, access-control filters, observability, evaluation datasets and failure handling. None of those additions change the core architecture; they make each boundary more reliable.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"RAG also cannot create evidence that is absent from the source set. If the source is wrong, incomplete or stale, a better embedding model cannot turn it into authoritative knowledge.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"What Would Change This Answer?\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"The architecture changes when the task requires more than knowledge lookup. A live order status needs current state. A financial calculation may need deterministic code. A web-research task may need active search. A workflow may need tools that can write data back to another system. An autonomous agent may need planning, permissions and execution control in addition to retrieval.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"RAG is therefore best understood as \u003Cb>one evidence-acquisition layer inside a larger AI system\u003C\u002Fb>. It is powerful precisely because it has a narrow job: find useful external information and place it in the model’s working context.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Conclusion\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"RAG becomes much easier to understand when the technology names are removed. A file is a source. A database is a source. An API is a source. A search function retrieves evidence. A prompt carries that evidence to the model. The LLM then interprets it and produces language.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"The hard part of production RAG is not calling an embedding model. It is building a trustworthy evidence path from the original source to the final claim: preserving provenance, selecting the right retrieval method, keeping information current, controlling access, evaluating retrieval separately from generation, and knowing when a direct database or tool call is better than semantic search.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"That is the practical continuation of the basic RAG model: \u003Cb>first understand the roles, then make the data path explicit.\u003C\u002Fb>\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Primary Sources\",\"level\":2},\"type\":\"header\"},{\"data\":{\"items\":[\"\u003Ca href=\\\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2005.11401\\\">Lewis et al. — Retrieval-Augmented Generation for Knowledge-Intensive NLP Tasks\u003C\u002Fa> — the 2020 paper introducing the RAG formulation that combines generation with retrieved non-parametric memory.\",\"\u003Ca href=\\\"https:\u002F\u002Fwww.sbert.net\u002Fexamples\u002Fsentence_transformer\u002Fapplications\u002Fsemantic-search\u002FREADME.html\\\">Sentence Transformers — Semantic Search\u003C\u002Fa> — official documentation for semantic retrieval, query embeddings and document embeddings.\",\"\u003Ca href=\\\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fembeddings\\\">OpenAI — Vector Embeddings\u003C\u002Fa> — official documentation describing embeddings as numerical representations used for relatedness and search.\",\"\u003Ca href=\\\"https:\u002F\u002Fwww.sqlite.org\u002Ffts5.html\\\">SQLite — FTS5 Extension\u003C\u002Fa> — official documentation for full-text search and BM25 ranking in SQLite.\",\"\u003Ca href=\\\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Flibraries\\\">OpenAI — SDKs and CLI\u003C\u002Fa> — official Python SDK example for the Responses API used in the optional generator example.\",\"\u003Ca href=\\\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works\\\">What Is RAG? The Simplest Explanation of How It Works\u003C\u002Fa> — the conceptual first part of this series.\"],\"style\":\"unordered\"},\"type\":\"list\"}],\"version\":\"2.31.0\"}",{"time":705,"blocks":706,"version":1144},1790516400000,[707,710,713,717,720,723,726,729,732,735,738,740,743,746,748,751,754,757,760,763,766,769,772,775,778,781,784,787,819,822,825,828,838,841,844,874,877,880,906,909,912,924,927,930,933,936,939,942,945,948,950,953,956,958,961,964,966,969,972,975,977,980,983,986,989,991,994,997,1000,1002,1005,1008,1011,1014,1016,1019,1022,1024,1027,1030,1033,1035,1038,1041,1043,1046,1049,1052,1055,1058,1061,1064,1067,1070,1073,1076,1079,1082,1085,1088,1099,1102,1105,1108,1111,1114,1117,1120,1123,1126,1129,1132,1135],{"data":708,"type":217},{"text":709},"The previous article, \u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works\">What Is RAG? The Simplest Explanation of How It Works\u003C\u002Fa>, established the mental model: the LLM writes, RAG retrieves useful knowledge, the application owns current state, and tools perform actions. This article takes the next step: \u003Cb>where does the data actually come from, and what does retrieval look like in Python?\u003C\u002Fb>",{"data":711,"type":217},{"text":712},"The important surprise is that an “LLM data source” is usually nothing exotic. It can be a text file, a folder of Markdown documents, a SQL database, an API response, a product catalog, a support system, or a vector index derived from those sources. The AI does not magically know these systems. Your application has to load, query, search, or retrieve the relevant data and place the result into the model’s context.",{"data":714,"type":226},{"text":715,"caption":716,"alignment":225},"Data source = where information lives. Retrieval = how the application finds useful information. Context = the selected information given to the model. LLM = the component that interprets that context and generates an answer.","The four-part model used throughout this article",{"data":718,"type":42},{"text":719,"level":230},"Question",{"data":721,"type":217},{"text":722},"How does an LLM use external data such as files, databases or APIs, and how can a small Python program implement the essential RAG steps without hiding them behind a framework?",{"data":724,"type":42},{"text":725,"level":230},"What This Really Means",{"data":727,"type":217},{"text":728},"When developers say that an LLM is “connected to company data,” several different operations may be hidden behind that sentence. One application may execute SQL. Another may call an API. Another may run full-text search. Another may calculate embedding similarity over document chunks. All of them can provide external information to an LLM, but they are not the same retrieval method and they should not be treated as interchangeable.",{"data":730,"type":217},{"text":731},"This distinction matters because the best retrieval method depends on the shape of the question. “What is our refund policy?” is a document-retrieval problem. “What is order 4711’s current status?” is usually a structured database lookup. “Which paragraph discusses account recovery?” can be keyword or semantic search. RAG is most useful when the system must \u003Cb>discover relevant knowledge before generation\u003C\u002Fb>.",{"data":733,"type":42},{"text":734,"level":230},"Simplest Example",{"data":736,"type":217},{"text":737},"Start with three strings in ordinary Python. There is no vector database, no framework, and no LLM yet. We only want to make the retrieval step visible.",{"data":739,"type":252},{"code":251},{"data":741,"type":217},{"text":742},"The program prints the first sentence because it contains the term we searched for. This is primitive retrieval, but the architecture is already visible: \u003Cb>question → search → relevant text\u003C\u002Fb>. RAG adds one more major step: pass the retrieved text to a language model together with the question.",{"data":744,"type":217},{"text":745},"A slightly more general version ranks documents by overlapping query terms:",{"data":747,"type":252},{"code":261},{"data":749,"type":217},{"text":750},"This is not a production search engine. It ignores morphology, synonyms, spelling variants, document length and many ranking signals. Its value is educational: \u003Cb>RAG does not begin with a vector database. It begins with retrieval.\u003C\u002Fb>",{"data":752,"type":42},{"text":753,"level":230},"Where the Example Stops Working",{"data":755,"type":217},{"text":756},"Exact or lexical matching becomes weak when the question and the source use different words. A document may say “vehicle maintenance,” while the user asks “how do I repair my car?” A lexical retriever can miss the relationship even though a human sees it immediately. Semantic retrieval addresses this by representing text as vectors and comparing meaning rather than only exact tokens.",{"data":758,"type":217},{"text":759},"Long files create another problem. Searching an entire 80-page manual as one unit is too coarse, but splitting every sentence can destroy useful context. Real RAG systems therefore need decisions about parsing, chunking, metadata, ranking, freshness, permissions and provenance.",{"data":761,"type":217},{"text":762},"The example also says nothing about structured live facts. If the user asks for the current status of order 4711 and the application already has a database key, semantic search is usually the wrong first tool. A deterministic database query is better.",{"data":764,"type":42},{"text":765,"level":230},"Direct Answer",{"data":767,"type":217},{"text":768},"An LLM data source is any external system from which an application can obtain information for the model: files, databases, APIs, search indexes, vector stores or live application state. RAG is the pattern of \u003Cb>retrieving relevant knowledge from such sources before generation\u003C\u002Fb>.",{"data":770,"type":217},{"text":771},"In Python, the essential pipeline can be very small: \u003Cb>load data → create retrievable units → find relevant evidence → assemble context → call the LLM\u003C\u002Fb>. The retrieval method should match the source and the question. Use SQL for exact structured facts, full-text search for lexical matching, embeddings for semantic similarity, and hybrid retrieval when several signals are valuable.",{"data":773,"type":42},{"text":774,"level":230},"Why This Is So",{"data":776,"type":217},{"text":777},"A language model does not automatically receive the contents of your filesystem, PostgreSQL database, CRM, private API or newly edited document. The application decides what external information is accessible and what is placed into the model’s current context.",{"data":779,"type":217},{"text":780},"The original Retrieval-Augmented Generation work by Lewis et al. combined a generative model with external non-parametric memory retrieved from a dense vector index. The broader architectural idea survives beyond that specific implementation: external evidence can be retrieved at inference time instead of expecting all useful knowledge to be encoded in model parameters.",{"data":782,"type":217},{"text":783},"This creates a useful separation of responsibilities: the source stores information, the retriever selects evidence, the context carries that evidence into the request, and the model interprets it. Keeping those boundaries visible makes failures much easier to diagnose.",{"data":785,"type":42},{"text":786,"level":230},"Context: The Main Types of Data Sources",{"data":788,"type":336},{"content":789,"withHeadings":14},[790,794,797,800,804,807,811,815],[791,792,793],"Source","Typical retrieval method","Good for",[309,795,796],"Parsing + lexical or semantic search","Documentation, manuals, articles, notes",[313,798,799],"Structure-aware extraction + search","Policies, reports, contracts, manuals",[801,802,803],"SQL database","SQL query or filtered retrieval","Orders, users, products, structured records",[321,805,806],"HTTP request with parameters","Remote systems and live service data",[808,809,810],"Search index","BM25 \u002F full-text \u002F hybrid search","Large text collections",[812,813,814],"Vector index","Embedding similarity","Semantic document retrieval",[816,817,818],"Application state","Direct state read or tool call","What is true right now",{"data":820,"type":217},{"text":821},"A vector index deserves special attention. In many architectures it is \u003Cb>not the canonical source of truth\u003C\u002Fb>. It is a retrieval index derived from documents or records. The authoritative document may live in object storage, a CMS, Git, PostgreSQL or another system, while embeddings and metadata are stored separately for fast semantic lookup. Some systems do use a vector store as primary storage, but that is an architectural choice rather than a requirement of RAG.",{"data":823,"type":217},{"text":824},"If the boundary between retrieval, persistent memory, current state and model context is still unclear, see \u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context\">AI Agent Memory Is Not RAG\u003C\u002Fa>. Those layers can use some of the same storage technologies while still having different correctness rules.",{"data":826,"type":42},{"text":827,"level":230},"Assumptions",{"data":829,"type":357},{"items":830,"style":356},[831,832,833,834,835,836,837],"The application is allowed to access the external source.","The relevant source contains enough information to answer the question.","The data can be parsed or queried in a form the retrieval layer can use.","The retrieved information is fresh enough for the requested decision.","The model receives the selected evidence in its context.","Authorization is enforced before protected evidence reaches the model.","The generation model can still be wrong even when retrieval is correct.",{"data":839,"type":217},{"text":840},"These assumptions matter because retrieval cannot compensate for missing evidence, stale source versions, broken parsers or unauthorized access. A RAG pipeline can only be as trustworthy as the evidence path that feeds it.",{"data":842,"type":42},{"text":843,"level":230},"Variables",{"data":845,"type":336},{"content":846,"withHeadings":14},[847,850,853,856,859,862,865,868,871],[848,849],"Variable","Why it changes the design",[851,852],"Source structure","A SQL table, legal PDF and source-code repository need different retrieval strategies",[854,855],"Question type","Exact lookup, conceptual search and multi-hop research are different tasks",[857,858],"Freshness requirement","Live state may need direct queries instead of periodically rebuilt indexes",[860,861],"Corpus size","In-memory search may work for hundreds of chunks but not for very large collections",[863,864],"Language","Multilingual retrieval requires models and tokenization suitable for the actual languages",[866,867],"Permissions","Retrieval must filter by the current user’s access rights",[869,870],"Latency and cost","More retrieval stages can improve quality but add runtime and infrastructure cost",[872,873],"Need for provenance","High-trust systems need source IDs, versions and traceable evidence",{"data":875,"type":42},{"text":876,"level":230},"Diagnostic \u002F Decision Method",{"data":878,"type":217},{"text":879},"The first decision is not “Which vector database should I install?” It is: \u003Cb>What kind of fact am I trying to retrieve?\u003C\u002Fb>",{"data":881,"type":336},{"content":882,"withHeadings":14},[883,886,890,894,898,902],[854,884,885],"Preferred first approach","Reason",[887,888,889],"Exact ID or current record","SQL \u002F key lookup \u002F API","Deterministic structured access",[891,892,893],"Exact wording, codes, names","Full-text or keyword search","Lexical precision",[895,896,897],"Conceptual question over documents","Semantic vector search","Meaning can differ from wording",[899,900,901],"Mixed enterprise knowledge","Hybrid retrieval + metadata filters","Combines lexical and semantic signals",[903,904,905],"Current application state","Direct state\u002Ftool access","Freshness matters more than document similarity",{"data":907,"type":217},{"text":908},"A useful test is: \u003Cb>Do I already know which record I need, or must the system discover which passage is relevant?\u003C\u002Fb> If the record is known, query it directly. If relevance must be discovered, search becomes more important.",{"data":910,"type":217},{"text":911},"When an answer is wrong, diagnose the pipeline in order instead of immediately changing the LLM:",{"data":913,"type":357},{"items":914,"style":444},[915,916,917,918,919,920,921,922,923],"\u003Cb>1. Source coverage:\u003C\u002Fb> Does the correct information exist in the accessible source set?","\u003Cb>2. Freshness:\u003C\u002Fb> Is that version current enough for the question?","\u003Cb>3. Parsing:\u003C\u002Fb> Was the relevant content extracted correctly?","\u003Cb>4. Chunking:\u003C\u002Fb> Did the evidence stay together with the conditions that give it meaning?","\u003Cb>5. Retrieval:\u003C\u002Fb> Does the correct chunk appear among the candidates?","\u003Cb>6. Ranking:\u003C\u002Fb> Are stronger sources ranked above weaker or conflicting ones?","\u003Cb>7. Context assembly:\u003C\u002Fb> Did the application actually send the selected evidence to the model?","\u003Cb>8. Generation:\u003C\u002Fb> Did the LLM faithfully use the supplied evidence?","\u003Cb>9. Attribution:\u003C\u002Fb> Can each important claim be traced to a source?",{"data":925,"type":217},{"text":926},"For a deeper production-debugging method, see \u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fblog\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method\">RAG Failed — But Which Layer Actually Failed? A Diagnostic Method\u003C\u002Fa>, which expands this chain into independently testable failure layers.",{"data":928,"type":42},{"text":929,"level":230},"Evidence",{"data":931,"type":217},{"text":932},"The RAG paper by Lewis et al. formalized generation that conditions on retrieved external memory rather than relying only on model parameters. That provides the conceptual foundation for separating the generator from a retrievable knowledge source.",{"data":934,"type":217},{"text":935},"Sentence Transformers documents semantic search as embedding the corpus and the query into a vector space and retrieving items with high semantic similarity. Its current API also distinguishes query encoding from document encoding for retrieval tasks.",{"data":937,"type":217},{"text":938},"SQLite FTS5 demonstrates the other side of the spectrum: mature full-text retrieval can rank documents without embeddings. This matters because lexical search remains valuable for identifiers, exact terminology and many hybrid retrieval designs.",{"data":940,"type":217},{"text":941},"OpenAI’s embeddings documentation describes embeddings as numerical vector representations used for relatedness and search. This is one implementation path for semantic retrieval, not the definition of RAG itself.",{"data":943,"type":42},{"text":944,"level":230},"Real Example 1: A Folder of Text Files",{"data":946,"type":217},{"text":947},"Suppose a directory named \u003Ccode>knowledge\u002F\u003C\u002Fcode> contains ordinary text files. Python can load them with no AI library at all.",{"data":949,"type":252},{"code":471},{"data":951,"type":217},{"text":952},"The filesystem is the data source. The next question is how much text should become one retrievable unit. For long documents, searching one complete file is often too coarse. This is why RAG pipelines commonly create chunks.",{"data":954,"type":42},{"text":955,"level":478},"A Very Simple Chunker",{"data":957,"type":252},{"code":481},{"data":959,"type":217},{"text":960},"This example groups paragraphs until a rough character limit is reached. It is intentionally understandable rather than optimal. Production systems often chunk by tokens, headings, sections, sentence boundaries or document structure. Tables, source code, contracts and API documentation may need different strategies.",{"data":962,"type":42},{"text":963,"level":478},"Preserve Provenance While Chunking",{"data":965,"type":252},{"code":490},{"data":967,"type":217},{"text":968},"A useful chunk carries more than text. Source name, document ID, URL, timestamp, version or section can later support citation, debugging and freshness checks. If provenance is lost during ingestion, it becomes much harder to explain why a particular answer was produced.",{"data":970,"type":42},{"text":971,"level":230},"Real Example 2: Structured Data — Use SQL When SQL Is the Right Tool",{"data":973,"type":217},{"text":974},"Not every external fact should go through semantic search. If the question asks for an exact current record, a direct database query is usually clearer and more deterministic.",{"data":976,"type":252},{"code":502},{"data":978,"type":217},{"text":979},"If the application already knows that the user is asking about order 4711, embedding the entire orders table and asking semantic search to rediscover that row usually adds complexity without benefit. A strong design rule is: \u003Cb>retrieve structured facts with structured queries; retrieve unstructured knowledge with search.\u003C\u002Fb>",{"data":981,"type":217},{"text":982},"The returned database row can still be placed into the model context so the LLM can explain it in natural language. But direct state or record access is conceptually different from searching a knowledge corpus.",{"data":984,"type":42},{"text":985,"level":230},"Real Example 3: Full-Text Search Before Embeddings",{"data":987,"type":217},{"text":988},"Between a naive Python loop and vector search lies a mature class of lexical retrieval systems. SQLite includes FTS5 for full-text search, including BM25 ranking.",{"data":990,"type":252},{"code":517},{"data":992,"type":217},{"text":993},"Lexical search is especially useful when exact terminology, product codes, names, identifiers or domain-specific words matter. Semantic search is not automatically better. Production systems often combine both signals.",{"data":995,"type":42},{"text":996,"level":230},"Real Example 4: Semantic Retrieval With Embeddings",{"data":998,"type":217},{"text":999},"Embeddings turn text into numerical vectors so semantically related passages can be compared even when they do not use identical wording. Sentence Transformers provides a straightforward local implementation.",{"data":1001,"type":252},{"code":529},{"data":1003,"type":217},{"text":1004},"The query does not contain the phrase “vehicle maintenance,” but a semantic model can still rank that passage highly because the concepts are related. This is the practical reason embeddings are common in RAG systems.",{"data":1006,"type":217},{"text":1007},"For small collections, embeddings can stay in memory. Larger systems usually persist them in a vector-capable index or database and perform nearest-neighbor search there. The storage changes, but the logic remains: encode the question, find relevant document representations, return the best evidence.",{"data":1009,"type":42},{"text":1010,"level":230},"Real Example 5: Build the Context for the LLM",{"data":1012,"type":217},{"text":1013},"A retriever should return evidence. The LLM should then receive the question plus that evidence. Keeping retrieval and generation separate makes both easier to inspect and test.",{"data":1015,"type":252},{"code":544},{"data":1017,"type":217},{"text":1018},"The instruction does not make the model infallible. It simply creates an explicit evidence boundary. The model can still misunderstand good evidence, ignore a condition or overgeneralize. That is why retrieval quality and generation quality must be evaluated separately.",{"data":1020,"type":42},{"text":1021,"level":230},"Real Example 6: A Complete Minimal Pipeline",{"data":1023,"type":252},{"code":553},{"data":1025,"type":217},{"text":1026},"The function receives \u003Ccode>call_llm\u003C\u002Fcode> as a dependency on purpose. Retrieval should not care whether generation is performed by a cloud model, a local model or another provider. The data path belongs to the application.",{"data":1028,"type":42},{"text":1029,"level":478},"Optional Generator: OpenAI Responses API",{"data":1031,"type":217},{"text":1032},"One possible generator is the OpenAI Responses API. Keeping the model name in an environment variable avoids hard-coding a particular model into the RAG architecture.",{"data":1034,"type":252},{"code":565},{"data":1036,"type":217},{"text":1037},"The same retrieval pipeline can be connected to a local inference server. This is an important architectural point: \u003Cb>RAG is not owned by the LLM provider.\u003C\u002Fb> The application owns the source, retrieval and context assembly.",{"data":1039,"type":42},{"text":1040,"level":478},"The Whole Architecture in One View",{"data":1042,"type":252},{"code":574},{"data":1044,"type":217},{"text":1045},"This data-flow model is more durable than memorizing one framework. Libraries, databases and model vendors will change; the responsibility boundaries remain.",{"data":1047,"type":42},{"text":1048,"level":230},"Common Misconceptions and Failure Modes",{"data":1050,"type":42},{"text":1051,"level":478},"“RAG means vector database.”",{"data":1053,"type":217},{"text":1054},"No. Vector search is one retrieval method. RAG can use full-text search, SQL, APIs, knowledge graphs, vector search or combinations of them. The defining pattern is retrieval of external information for generation.",{"data":1056,"type":42},{"text":1057,"level":478},"“If the data is in PostgreSQL, I must embed the whole database.”",{"data":1059,"type":217},{"text":1060},"No. Structured records should usually remain queryable as structured records. Embeddings are useful for semantic relevance, not as a replacement for deterministic queries.",{"data":1062,"type":42},{"text":1063,"level":478},"“More chunks means a better answer.”",{"data":1065,"type":217},{"text":1066},"Not necessarily. Extra context can introduce noise, conflicting versions and irrelevant material. Retrieval should optimize for useful evidence, not maximum volume.",{"data":1068,"type":42},{"text":1069,"level":478},"“A high similarity score proves the answer.”",{"data":1071,"type":217},{"text":1072},"No. Similarity measures relevance, not truth or applicability. A highly similar passage can be outdated, from the wrong product version or valid only under conditions that do not match the question.",{"data":1074,"type":42},{"text":1075,"level":478},"“Once the correct chunk is retrieved, hallucination is solved.”",{"data":1077,"type":217},{"text":1078},"No. Retrieval improves grounding but does not guarantee faithful reasoning. Generation still needs evaluation, and high-risk workflows may require deterministic validation or human review.",{"data":1080,"type":42},{"text":1081,"level":478},"“The model failed, so change the model.”",{"data":1083,"type":217},{"text":1084},"Not necessarily. The correct source may have been missing, parsed incorrectly, split badly, filtered out, ranked too low or omitted from the assembled context. Model replacement should not be the first diagnostic step.",{"data":1086,"type":42},{"text":1087,"level":230},"Edge Cases",{"data":1089,"type":357},{"items":1090,"style":356},[1091,1092,1093,1094,1095,1096,1097,1098],"\u003Cb>Conflicting documents:\u003C\u002Fb> two sources may disagree because versions, jurisdictions or products differ.","\u003Cb>Time-sensitive facts:\u003C\u002Fb> a semantically relevant source may already be stale.","\u003Cb>Permissions:\u003C\u002Fb> a retriever must not return documents the current user is not authorized to access.","\u003Cb>Multi-language collections:\u003C\u002Fb> the embedding model and retrieval strategy must support the languages actually used.","\u003Cb>Tables and source code:\u003C\u002Fb> plain paragraph chunking can destroy structure that is essential to the answer.","\u003Cb>Very short identifiers:\u003C\u002Fb> semantic retrieval can be weaker than exact matching for SKUs, IDs, error codes or acronyms.","\u003Cb>Long questions requiring several facts:\u003C\u002Fb> retrieval may need decomposition, several searches or reranking rather than one top-k query.","\u003Cb>Source hierarchy:\u003C\u002Fb> an official current policy may need to outrank an older but semantically closer discussion document.",{"data":1100,"type":42},{"text":1101,"level":230},"Limitations",{"data":1103,"type":217},{"text":1104},"The Python examples intentionally optimize for transparency, not scale. The keyword retriever is naive, the chunker uses character length, the SQLite examples do not include production connection management, and the semantic example keeps all embeddings in memory.",{"data":1106,"type":217},{"text":1107},"A production system may require vector indexes, rerankers, hybrid retrieval, document parsers, caching, incremental indexing, source versioning, access-control filters, observability, evaluation datasets and failure handling. None of those additions change the core architecture; they make each boundary more reliable.",{"data":1109,"type":217},{"text":1110},"RAG also cannot create evidence that is absent from the source set. If the source is wrong, incomplete or stale, a better embedding model cannot turn it into authoritative knowledge.",{"data":1112,"type":42},{"text":1113,"level":230},"What Would Change This Answer?",{"data":1115,"type":217},{"text":1116},"The architecture changes when the task requires more than knowledge lookup. A live order status needs current state. A financial calculation may need deterministic code. A web-research task may need active search. A workflow may need tools that can write data back to another system. An autonomous agent may need planning, permissions and execution control in addition to retrieval.",{"data":1118,"type":217},{"text":1119},"RAG is therefore best understood as \u003Cb>one evidence-acquisition layer inside a larger AI system\u003C\u002Fb>. It is powerful precisely because it has a narrow job: find useful external information and place it in the model’s working context.",{"data":1121,"type":42},{"text":1122,"level":230},"Conclusion",{"data":1124,"type":217},{"text":1125},"RAG becomes much easier to understand when the technology names are removed. A file is a source. A database is a source. An API is a source. A search function retrieves evidence. A prompt carries that evidence to the model. The LLM then interprets it and produces language.",{"data":1127,"type":217},{"text":1128},"The hard part of production RAG is not calling an embedding model. It is building a trustworthy evidence path from the original source to the final claim: preserving provenance, selecting the right retrieval method, keeping information current, controlling access, evaluating retrieval separately from generation, and knowing when a direct database or tool call is better than semantic search.",{"data":1130,"type":217},{"text":1131},"That is the practical continuation of the basic RAG model: \u003Cb>first understand the roles, then make the data path explicit.\u003C\u002Fb>",{"data":1133,"type":42},{"text":1134,"level":230},"Primary Sources",{"data":1136,"type":357},{"items":1137,"style":356},[1138,1139,1140,1141,1142,1143],"\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2005.11401\">Lewis et al. — Retrieval-Augmented Generation for Knowledge-Intensive NLP Tasks\u003C\u002Fa> — the 2020 paper introducing the RAG formulation that combines generation with retrieved non-parametric memory.","\u003Ca href=\"https:\u002F\u002Fwww.sbert.net\u002Fexamples\u002Fsentence_transformer\u002Fapplications\u002Fsemantic-search\u002FREADME.html\">Sentence Transformers — Semantic Search\u003C\u002Fa> — official documentation for semantic retrieval, query embeddings and document embeddings.","\u003Ca href=\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fembeddings\">OpenAI — Vector Embeddings\u003C\u002Fa> — official documentation describing embeddings as numerical representations used for relatedness and search.","\u003Ca href=\"https:\u002F\u002Fwww.sqlite.org\u002Ffts5.html\">SQLite — FTS5 Extension\u003C\u002Fa> — official documentation for full-text search and BM25 ranking in SQLite.","\u003Ca href=\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Flibraries\">OpenAI — SDKs and CLI\u003C\u002Fa> — official Python SDK example for the Responses API used in the optional generator example.","\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works\">What Is RAG? The Simplest Explanation of How It Works\u003C\u002Fa> — the conceptual first part of this series.","2.31.0","An LLM does not magically know your files, databases or APIs. This practical continuation of the RAG series shows, with simple Python, how external data becomes retrievable evidence: from text files and SQL to full-text search, embeddings, context assembly and the final LLM call.",{"lang":7,"title":208,"content":210,"contentJson":1147,"excerpt":677},{"time":212,"blocks":1148,"version":676},[1149,1151,1153,1155,1157,1159,1161,1163,1165,1167,1169,1171,1173,1175,1177,1179,1181,1183,1185,1187,1189,1191,1193,1195,1197,1199,1201,1203,1214,1216,1218,1220,1223,1225,1227,1239,1241,1243,1252,1254,1256,1259,1261,1263,1265,1267,1269,1271,1273,1275,1277,1279,1281,1283,1285,1287,1289,1291,1293,1295,1297,1299,1301,1303,1305,1307,1309,1311,1313,1315,1317,1319,1321,1323,1325,1327,1329,1331,1333,1335,1337,1339,1341,1343,1345,1347,1349,1351,1353,1355,1357,1359,1361,1363,1365,1367,1369,1371,1373,1375,1378,1380,1382,1384,1386,1388,1390,1392,1394,1396,1398,1400,1402],{"data":1150,"type":217},{"text":216},{"data":1152,"type":217},{"text":220},{"data":1154,"type":226},{"text":223,"caption":224,"alignment":225},{"data":1156,"type":42},{"text":229,"level":230},{"data":1158,"type":217},{"text":233},{"data":1160,"type":42},{"text":236,"level":230},{"data":1162,"type":217},{"text":239},{"data":1164,"type":217},{"text":242},{"data":1166,"type":42},{"text":245,"level":230},{"data":1168,"type":217},{"text":248},{"data":1170,"type":252},{"code":251},{"data":1172,"type":217},{"text":255},{"data":1174,"type":217},{"text":258},{"data":1176,"type":252},{"code":261},{"data":1178,"type":217},{"text":264},{"data":1180,"type":42},{"text":267,"level":230},{"data":1182,"type":217},{"text":270},{"data":1184,"type":217},{"text":273},{"data":1186,"type":217},{"text":276},{"data":1188,"type":42},{"text":279,"level":230},{"data":1190,"type":217},{"text":282},{"data":1192,"type":217},{"text":285},{"data":1194,"type":42},{"text":288,"level":230},{"data":1196,"type":217},{"text":291},{"data":1198,"type":217},{"text":294},{"data":1200,"type":217},{"text":297},{"data":1202,"type":42},{"text":300,"level":230},{"data":1204,"type":336},{"content":1205,"withHeadings":14},[1206,1207,1208,1209,1210,1211,1212,1213],[305,306,307],[309,310,311],[313,314,315],[317,318,319],[321,322,323],[325,326,327],[329,330,331],[333,334,335],{"data":1215,"type":217},{"text":339},{"data":1217,"type":217},{"text":342},{"data":1219,"type":42},{"text":345,"level":230},{"data":1221,"type":357},{"items":1222,"style":356},[349,350,351,352,353,354,355],{"data":1224,"type":217},{"text":360},{"data":1226,"type":42},{"text":363,"level":230},{"data":1228,"type":336},{"content":1229,"withHeadings":14},[1230,1231,1232,1233,1234,1235,1236,1237,1238],[368,369],[371,372],[374,375],[377,378],[380,381],[383,384],[386,387],[389,390],[392,393],{"data":1240,"type":42},{"text":396,"level":230},{"data":1242,"type":217},{"text":399},{"data":1244,"type":336},{"content":1245,"withHeadings":14},[1246,1247,1248,1249,1250,1251],[374,404,405],[407,408,409],[411,412,413],[415,416,417],[419,420,421],[423,424,425],{"data":1253,"type":217},{"text":428},{"data":1255,"type":217},{"text":431},{"data":1257,"type":357},{"items":1258,"style":444},[435,436,437,438,439,440,441,442,443],{"data":1260,"type":217},{"text":447},{"data":1262,"type":42},{"text":450,"level":230},{"data":1264,"type":217},{"text":453},{"data":1266,"type":217},{"text":456},{"data":1268,"type":217},{"text":459},{"data":1270,"type":217},{"text":462},{"data":1272,"type":42},{"text":465,"level":230},{"data":1274,"type":217},{"text":468},{"data":1276,"type":252},{"code":471},{"data":1278,"type":217},{"text":474},{"data":1280,"type":42},{"text":477,"level":478},{"data":1282,"type":252},{"code":481},{"data":1284,"type":217},{"text":484},{"data":1286,"type":42},{"text":487,"level":478},{"data":1288,"type":252},{"code":490},{"data":1290,"type":217},{"text":493},{"data":1292,"type":42},{"text":496,"level":230},{"data":1294,"type":217},{"text":499},{"data":1296,"type":252},{"code":502},{"data":1298,"type":217},{"text":505},{"data":1300,"type":217},{"text":508},{"data":1302,"type":42},{"text":511,"level":230},{"data":1304,"type":217},{"text":514},{"data":1306,"type":252},{"code":517},{"data":1308,"type":217},{"text":520},{"data":1310,"type":42},{"text":523,"level":230},{"data":1312,"type":217},{"text":526},{"data":1314,"type":252},{"code":529},{"data":1316,"type":217},{"text":532},{"data":1318,"type":217},{"text":535},{"data":1320,"type":42},{"text":538,"level":230},{"data":1322,"type":217},{"text":541},{"data":1324,"type":252},{"code":544},{"data":1326,"type":217},{"text":547},{"data":1328,"type":42},{"text":550,"level":230},{"data":1330,"type":252},{"code":553},{"data":1332,"type":217},{"text":556},{"data":1334,"type":42},{"text":559,"level":478},{"data":1336,"type":217},{"text":562},{"data":1338,"type":252},{"code":565},{"data":1340,"type":217},{"text":568},{"data":1342,"type":42},{"text":571,"level":478},{"data":1344,"type":252},{"code":574},{"data":1346,"type":217},{"text":577},{"data":1348,"type":42},{"text":580,"level":230},{"data":1350,"type":42},{"text":583,"level":478},{"data":1352,"type":217},{"text":586},{"data":1354,"type":42},{"text":589,"level":478},{"data":1356,"type":217},{"text":592},{"data":1358,"type":42},{"text":595,"level":478},{"data":1360,"type":217},{"text":598},{"data":1362,"type":42},{"text":601,"level":478},{"data":1364,"type":217},{"text":604},{"data":1366,"type":42},{"text":607,"level":478},{"data":1368,"type":217},{"text":610},{"data":1370,"type":42},{"text":613,"level":478},{"data":1372,"type":217},{"text":616},{"data":1374,"type":42},{"text":619,"level":230},{"data":1376,"type":357},{"items":1377,"style":356},[623,624,625,626,627,628,629,630],{"data":1379,"type":42},{"text":633,"level":230},{"data":1381,"type":217},{"text":636},{"data":1383,"type":217},{"text":639},{"data":1385,"type":217},{"text":642},{"data":1387,"type":42},{"text":645,"level":230},{"data":1389,"type":217},{"text":648},{"data":1391,"type":217},{"text":651},{"data":1393,"type":42},{"text":654,"level":230},{"data":1395,"type":217},{"text":657},{"data":1397,"type":217},{"text":660},{"data":1399,"type":217},{"text":663},{"data":1401,"type":42},{"text":666,"level":230},{"data":1403,"type":357},{"items":1404,"style":356},[670,671,672,673,674,675],"Post erfolgreich abgerufen",{"items":1407,"source":1492,"manualIds":1493,"manualMatchedIds":1494},[1408,1415,1422,1429,1436,1443,1450,1457,1464,1471,1478,1485],{"id":1409,"slug":1410,"title":1411,"excerpt":1412,"featuredImage":1413,"publishedAt":1414},"463","prompt-invariance-does-the-conclusion-survive-the-prompt","Invarijantnost prompta: Da li zaključak preživljava prompt?","Praktična metodologija za testiranje da li zaključak veštačke inteligencije zavisi od načina na koji je problem uokviren. Prompt Invariance upoređuje originalne, slepe, invertovane i suparničke formulacije, dok strukturu dokaza drži kontrolisanom.","\u002Fuploads\u002F2026\u002F09\u002Fprompt-invariance-does-the-conclusion-survive-the-prompt-1789809799910-s0vbcb.webp","2026-09-19T01:09:00.000Z",{"id":1416,"slug":1417,"title":1418,"excerpt":1419,"featuredImage":1420,"publishedAt":1421},"454","zbt-z8102ax-rm500u-ea-5g-modem-test","Quectel RM500U-EA u ZBT Z8102AX: 5G opsezi, o2 Nemačka i ponašanje signala u stvarnom svetu","ZBT Z8102AX koristi Quectel RM500U-EA modem za 4G i 5G povezivost. U prvom praktičnom testu, ruter se uspešno povezao na o2 Germany sa LTE Band 3 i NR n28. Modem radi, ali dublja dijagnostika poput RSRP, RSRQ, SINR, zaključavanja opsega i ponašanja ćelije još uvek zahteva odgovarajuće testiranje.","\u002Fuploads\u002F2026\u002F06\u002Fopenwrt-router-review-dual-sim-06-1781620597879-qay2sx.webp","2026-06-16T08:39:00.000Z",{"id":1423,"slug":1424,"title":1425,"excerpt":1426,"featuredImage":1427,"publishedAt":1428},"436","triggers","Sveobuhvatni vodič za okidače povlačenja u poslovnim AI runbooks-ovima","Ovaj vodič istražuje Rollback Triggers, ključne mehanizme u enterprise AI runbooks koji automatski otkrivaju anomalije i pokreću rollback-ove radi održavanja stabilnosti sistema. Naučite kako da konfigurišete, nadgledate i optimizujete ove triggere za robusne AI deploymente.","\u002Fuploads\u002F2026\u002F06\u002Ftriggers-1781855500994-cibuou.webp","2026-03-01T16:51:00.000Z",{"id":1430,"slug":1431,"title":1432,"excerpt":1433,"featuredImage":1434,"publishedAt":1435},"476","mcp-vs-a2a-vs-ucp-vs-ap2-vs-a2ui-the-agent-protocol-stack-explained","MCP vs A2A vs UCP vs AP2 vs A2UI: Objašnjen stek agentskih protokola","MCP, A2A, UCP, AP2 i A2UI se često predstavljaju kao konkurentski standardi za agente. Oni uglavnom rešavaju različite probleme interoperabilnosti. Ovaj vodič mapira svaki protokol na granicu koju zapravo standardizuje—i pokazuje kako oni mogu da rade zajedno u jednom produkcionom sistemu.","\u002Fuploads\u002F2026\u002F09\u002Fmcp-vs-a2a-vs-ucp-vs-ap2-vs-a2ui-the-agent-protocol-stack-explained-1790352625869-2ezle0.webp","2026-09-25T12:09:00.000Z",{"id":1437,"slug":1438,"title":1439,"excerpt":1440,"featuredImage":1441,"publishedAt":1442},"376","laravel-12-custom-cms-with-filament3","Laravel 12 Prilagođeni CMS sa Filament 3: Ekspertski radni tok","Detaljan pregled sinergija između Laravel 12 i Filament 3 za kreiranje prilagođenih sistema za upravljanje sadržajem. Stručnjaci analiziraju inovativni tok posla, prednosti, mane i izazov Jetstream toka posla.","\u002Fuploads\u002F2025\u002F01\u002FLaravel-12-Custom-CMS-with-a-Filament3-large.webp","2025-01-12T02:18:00.000Z",{"id":1444,"slug":1445,"title":1446,"excerpt":1447,"featuredImage":1448,"publishedAt":1449},"370","boosting-productivity-with-erp-systems-a-case-study-on-relational-databases","Povećanje produktivnosti sa ERP sistemima: Studija slučaja o relacionim bazama podataka","Integracija relacionih baza podataka sa ERP sistemima značajno povećava produktivnost. Kom","\u002Fuploads\u002F2024\u002F07\u002F2024-07-25-A-visual-representation-of-an-ERP-Enterprise-Resource-Planning-model-showing-relational-databases-improving-productivity-large.webp","2024-07-25T11:29:00.000Z",{"id":1451,"slug":1452,"title":1453,"excerpt":1454,"featuredImage":1455,"publishedAt":1456},"478","what-is-rag-the-simplest-explanation-of-how-it-works","Šta je RAG? Najjednostavnije objašnjenje kako funkcioniše","RAG zvuči komplikovano, ali ideja je jednostavna: pre nego što AI odgovori, prvo potraži korisne informacije iz izvora znanja i daje te informacije jezičkom modelu. Ovaj vodič objašnjava RAG, LLM-ove, stanje, memoriju i alate koristeći jedan jednostavan mentalni model.","\u002Fuploads\u002F2026\u002F09\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works-1790377492124-khjagt.webp","2026-09-25T19:03:00.000Z",{"id":1458,"slug":1459,"title":1460,"excerpt":1461,"featuredImage":1462,"publishedAt":1463},"375","database-marketing","Marketing baze podataka: Moderan pristup odnosima sa klijentima","Marketing baza podataka je neophodan za moderno upravljanje odnosima sa klijentima. Saznajte kako strateško korišćenje podataka, tehnička ekspertiza i inovacije pokreću personalizovane interakcije sa klijentima i održivi rast.","\u002Fuploads\u002F2025\u002F01\u002FDatabasemarketing.png-medium.webp","2025-01-06T00:20:00.000Z",{"id":1465,"slug":1466,"title":1467,"excerpt":1468,"featuredImage":1469,"publishedAt":1470},"451","test-dev-enterprise","Sveobuhvatni vodič za Test DEv Enterprise Stajic.de: Arhitektura i najbolje prakse","Istražite arhitektonske principe, prednosti i tehničke detalje upravljanja okruženjem za razvoj i testiranje nivoa preduzeća pomoću Test DEv Enterprise Stajic.de.","\u002Fuploads\u002F2026\u002F05\u002Ftest-dev-enterprise-1779534260081-r4dvxn.webp","2026-05-22T23:01:00.000Z",{"id":1472,"slug":1473,"title":1474,"excerpt":1475,"featuredImage":1476,"publishedAt":1477},"459","ollama-is-not-the-product-building-production-ready-open-llm-applications","Ollama nije proizvod: Izgradnja aplikacija spremnih za produkciju sa otvorenim LLM-ovima","Pokretanje lokalnog modela pomoću Ollama-e je jednostavno. Izgradnja Open-LLM aplikacije spremne za produkciju je teža: zahteva RAG, kontrolu pristupa, apstrakciju provajdera, evaluaciju, logovanje, disciplinu puštanja u rad i kontrolisani aplikativni sloj oko modela.","\u002Fuploads\u002F2026\u002F06\u002Follama-is-not-the-product-building-production-ready-open-llm-applications-1782679361640-h0usqf.webp","2026-06-28T16:39:00.000Z",{"id":1479,"slug":1480,"title":1481,"excerpt":1482,"featuredImage":1483,"publishedAt":1484},"477","computer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system","Agenti za korišćenje računara: Zašto uspešan demo i dalje može biti nepouzdan sistem","Agenti za korišćenje računara sada mogu da završe impresivne radne tokove u pregledaču i na radnoj površini, ali jedno uspešno izvršavanje dokazuje sposobnost—ne pouzdanost. Ovaj članak pokazuje kako testirati ponovljivost, robusnost u odnosu na okruženje, kontrolu dugog horizonta, svest o stanju, verifikaciju ishoda i bezbedno upravljanje ciljevima.","\u002Fuploads\u002F2026\u002F09\u002Fcomputer-use-agents-why-a-successful-demo-can-still-be-an-unreliable-system-1790352854690-75qnrg.webp","2026-09-25T12:13:00.000Z",{"id":1486,"slug":1487,"title":1488,"excerpt":1489,"featuredImage":1490,"publishedAt":1491},"473","openai-agents-api-vs-agents-sdk-vs-responses-api-what-should-you-build-on-in-2026","OpenAI Agents API vs Agents SDK vs Responses API: Na čemu bi trebalo da gradite u 2026?","OpenAI-jev stek agenata promenio se u septembru 2026. Ovaj arhitekturni vodič razdvaja Agents API, Agents SDK, Responses API i Codex SDK prema vlasništvu nad izvršnim okruženjem—kako bi timovi mogli da izaberu pravu granicu kontrole umesto da porede nazive proizvoda.","\u002Fuploads\u002F2026\u002F09\u002Fopenai-agents-api-vs-agents-sdk-vs-responses-api-what-should-you-build-on-in-2026-1790351846714-zi7lus.webp","2026-09-25T11:56:00.000Z","fallback",[],[]]