[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"portal-settings:stajic:zh":3,"public-menus:all":38,"post:where-does-an-llm-get-its-data-rag-data-sources-in-python:zh":205,"related:post:where-does-an-llm-get-its-data-rag-data-sources-in-python:zh:1":1405},{"statusCode":4,"data":5,"message":37},200,{"tenantId":6,"lang":7,"defaultLang":8,"siteUrl":9,"contactEmail":10,"brandName":11,"logoUrl":12,"siteName":11,"siteDescription":13,"ogImage":10,"robotsIndex":14,"socialLinks":10,"reservedSlugs":10,"seoPolicy":15},"stajic","zh","de","https:\u002F\u002Fstajic.de",null,"Stajic Platform","\u002FLogo_Planet.svg","Stajic Portal",true,{"branding":16,"relatedContent":17,"crossDomainLinks":18},{"logoUrl":12},{"enabled":14},[19,22,25,28,31,34],{"url":20,"label":21,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Ffigure.rocks","figure.rocks",{"url":23,"label":24,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Floving.rocks","loving.rocks",{"url":26,"label":27,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.com","bazify.com",{"url":29,"label":30,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.de","bazify.de",{"url":32,"label":33,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.at","bazify.at",{"url":35,"label":36,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.ba","bazify.ba","Portal settings resolved",[39,45],{"id":40,"name":41,"location":42,"isActive":14,"isDefault":43,"items":44},1,"main-navigation","header",false,[],{"id":46,"name":47,"location":48,"isActive":14,"isDefault":14,"items":49},4,"main-menu","sidebar",[50,66,79,93,103,118,133],{"id":51,"title":52,"url":60,"target":61,"icon":62,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":64,"portfolioId":10,"children":65},"item-18",{"de":53,"en":54,"es":55,"fr":56,"it":54,"ru":57,"sr":58,"zh":59},"Startseite","Home","Inicio","Accueil","Главная","Почетна","首页","\u002Ffull-stack-web-developer-munich-performance-seo-and-maintainable-builds","_self","i-lucide-home","page",111,[],{"id":67,"title":68,"url":75,"target":61,"icon":76,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":77,"portfolioId":10,"children":78},"item-22",{"de":69,"en":69,"es":70,"fr":69,"it":71,"ru":72,"sr":73,"zh":74},"Vision","Visión","Visione","Видение","Визија","想象","\u002Fueber-uns-webdesign-muenchen-webaplikation","i-lucide-eye",113,[],{"id":80,"title":81,"url":89,"target":61,"icon":90,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":91,"portfolioId":10,"children":92},"item-19",{"de":82,"en":83,"es":84,"fr":83,"it":85,"ru":86,"sr":87,"zh":88},"Leistungen","Services","Servicios","Servizi","Услуги","Услуге","服务","\u002Fservices-dienstleistungen-muenchen","i-lucide-wrench",116,[],{"id":94,"title":95,"url":99,"target":61,"icon":100,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":101,"portfolioId":10,"children":102},"item-23",{"de":96,"en":96,"es":96,"fr":96,"it":96,"ru":97,"sr":97,"zh":98},"Blog","Блог","博客","\u002Fblog","i-lucide-book-open",112,[],{"id":104,"title":105,"url":114,"target":61,"icon":115,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":116,"portfolioId":10,"children":117},"item-32",{"de":106,"en":107,"es":108,"fr":109,"it":110,"ru":111,"sr":112,"zh":113},"Neue Technologien","New Technologies","Nuevas tecnologías","Nouvelles technologies","Nuove tecnologie","Новые технологии","Нове технологије","新技术！","\u002Fneue-webtechnologien","i-lucide-sparkles",122,[],{"id":119,"title":120,"url":129,"target":61,"icon":130,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":131,"portfolioId":10,"children":132},"item-20",{"de":121,"en":122,"es":123,"fr":124,"it":125,"ru":126,"sr":127,"zh":128},"Kontakt","Contact us!","Contacto","Contact","Contatto","Контакт","Контактирајте нас","联系我们！","\u002Fcontact","i-lucide-mail",115,[],{"id":134,"title":135,"url":144,"target":61,"icon":145,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":147},"item-21",{"de":136,"en":137,"es":138,"fr":139,"it":140,"ru":141,"sr":142,"zh":143},"Unsere Arbeit","Our Work","Nuestro trabajo","Nos réalisations","I nostri lavori","Наши работы","Наши радови","文件夹","\u002Fportfolio","i-lucide-briefcase",114,[148,161,175,181,193],{"id":149,"title":150,"url":144,"target":61,"icon":159,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":160},"item-24",{"de":151,"en":152,"es":153,"fr":154,"it":155,"ru":156,"sr":157,"zh":158},"Alle Projekte","All Projects","Todos los proyectos","Tous les projets","Tutti i progetti","Все проекты","Сви пројекти","所有项目","i-lucide-grid-3x3",[],{"id":162,"title":163,"url":171,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":174},"item-29",{"de":164,"en":165,"es":166,"fr":167,"it":168,"ru":169,"sr":170,"zh":143},"Local Roots, Global Reach","Local Roots - Global Reach","Empresa local ","Entreprise locale","Azienda locale","Местная компания","Локално предузеће глобално тржиште","\u002Fportfolio\u002Flocal-roots-global-reach-communication-media-systems-for-modern-business","i-lucide-folder","custom",[],{"id":176,"title":177,"url":179,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":180},"item-28",{"de":178,"en":178,"es":178,"fr":178,"it":178,"ru":178,"sr":178,"zh":178},"Solr Suggester","\u002Fportfolio\u002Fsolr-fuzzy-suggester-und-solr-infix-suggester-abfrage-ueber-ajax-und-filterung",[],{"id":182,"title":183,"url":191,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":192},"item-27",{"de":184,"en":185,"es":186,"fr":187,"it":188,"ru":189,"sr":190,"zh":185},"Firmenwebseite SEO","Company Website SEO","Sitio web corporativo SEO","Site web d’entreprise SEO","Sito web aziendale SEO","Корпоративный сайт SEO","Пословна веб-страница SEO","\u002Fportfolio\u002Fseo-sem-branding-mobile-webseite-muenchen",[],{"id":194,"title":195,"url":203,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":204},"item-31",{"de":196,"en":197,"es":198,"fr":199,"it":200,"ru":201,"sr":202,"zh":197},"Digitalisierungsportal","Digitalization Portal","Portal de digitalización","Portail de numérisation","Portale di digitalizzazione","Портал цифровизации","Портал за дигитализацију","\u002Fportfolio\u002Fdigitalisierungsportal-archiv-museum-bibliothek-ead-lido-mets-mods",[],{"statusCode":4,"data":206,"message":1404},{"id":207,"title":208,"slug":209,"content":210,"contentJson":211,"excerpt":676,"featuredImage":677,"featuredImageAlt":678,"featuredImageCaption":10,"featuredImageTitle":10,"featuredImageCopyright":10,"featuredImageAuthor":10,"featuredImageSourceUrl":10,"featuredImageLicense":10,"featuredImageIsAiGenerated":43,"status":679,"publishedAt":680,"createdAt":681,"updatedAt":682,"seoLocalePaths":683,"categories":692,"author":693,"translations":698},"479","LLM从哪里获取数据？Python中的RAG数据源","where-does-an-llm-get-its-data-rag-data-sources-in-python","\u003Cp>上一篇文章，\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works\">什么是 RAG？对其工作原理的最简单解释\u003C\u002Fa>，建立了思维模型：LLM 负责写作，RAG 检索有用的知识，应用程序拥有当前状态，工具执行操作。本文迈出下一步：\u003Cb>数据实际上来自哪里，以及在 Python 中检索是什么样子的？\u003C\u002Fb>\u003C\u002Fp>\n\u003Cp>重要的惊喜在于，“LLM 数据源”通常并没有什么特别之处。它可以是一个文本文件、一个包含 Markdown 文档的文件夹、一个 SQL 数据库、一个 API 响应、一个产品目录、一个支持系统，或从这些来源派生的向量索引。AI 并不会神奇地知道这些系统。你的应用程序必须加载、查询、搜索或检索相关数据，并将结果放入模型的上下文中。\u003C\u002Fp>\n\u003Cblockquote class=\"border-l-4 border-gray-300 pl-4 italic\">数据源 = 信息所在之处。检索 = 应用程序如何找到有用的信息。上下文 = 提供给模型的选定信息。LLM = 解释该上下文并生成答案的组件。\u003Ccite class=\"block mt-2 text-sm\">— 本文通篇使用的四部分模型\u003C\u002Fcite>\u003C\u002Fblockquote>\n\u003Ch2 id=\"section-4\">问题\u003C\u002Fh2>\n\u003Cp>LLM 如何使用文件、数据库或 API 等外部数据，以及一个小型 Python 程序如何在不借助框架隐藏这些步骤的情况下实现基本的 RAG 步骤？\u003C\u002Fp>\n\u003Ch2 id=\"section-6\">这真正意味着什么\u003C\u002Fh2>\n\u003Cp>当开发者说 LLM“连接到公司数据”时，这句话背后可能隐藏着几种不同的操作。一个应用程序可能执行 SQL。另一个可能调用 API。另一个可能运行全文搜索。另一个可能计算文档块之间的嵌入相似度。所有这些都可以向 LLM 提供外部信息，但它们不是相同的检索方法，不应被视为可以互换。\u003C\u002Fp>\n\u003Cp>这种区分很重要，因为最佳检索方法取决于问题的形态。“我们的退款政策是什么？”是一个文档检索问题。“订单 4711 的当前状态是什么？”通常是结构化数据库查询。“哪一段讨论了账户恢复？”可以是关键词或语义搜索。当系统必须在生成之前\u003Cb>发现相关知识\u003C\u002Fb>时，RAG 最为有用。\u003C\u002Fp>\n\u003Ch2 id=\"section-9\">最简单的例子\u003C\u002Fh2>\n\u003Cp>从普通 Python 中的三个字符串开始。没有向量数据库，没有框架，也还没有 LLM。我们只想让检索步骤变得可见。\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>documents = [\n    &quot;The AKM uses 7.62 mm ammunition.&quot;,\n    &quot;A Med Kit restores health.&quot;,\n    &quot;A 4x scope can be attached to several compatible weapons.&quot;\n]\n\nquestion = &quot;Which ammunition does the AKM use?&quot;\n\nfor document in documents:\n    if &quot;AKM&quot; in document:\n        print(document)\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>程序打印第一句话，因为它包含我们搜索的词。这是原始的检索，但架构已经可见：\u003Cb>问题 → 搜索 → 相关文本\u003C\u002Fb>。RAG 增加了一个重要步骤：将检索到的文本与问题一起传递给语言模型。\u003C\u002Fp>\n\u003Cp>一个稍微更通用的版本根据查询词的重叠对文档进行排名：\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>import re\n\ndocuments = [\n    {&quot;id&quot;: &quot;weapon-akm&quot;, &quot;text&quot;: &quot;The AKM uses 7.62 mm ammunition.&quot;},\n    {&quot;id&quot;: &quot;healing-medkit&quot;, &quot;text&quot;: &quot;A Med Kit restores health.&quot;},\n    {&quot;id&quot;: &quot;scope-4x&quot;, &quot;text&quot;: &quot;A 4x scope can be attached to several compatible weapons.&quot;},\n]\n\ndef words(text):\n    return set(re.findall(r&quot;[a-zA-Z0-9.]+&quot;, text.lower()))\n\ndef retrieve(question, documents, top_k=2):\n    query_terms = words(question)\n    ranked = []\n\n    for document in documents:\n        score = len(query_terms &amp; words(document[&quot;text&quot;]))\n        if score &gt; 0:\n            ranked.append((score, document))\n\n    ranked.sort(key=lambda item: item[0], reverse=True)\n    return [document for _, document in ranked[:top_k]]\n\nquestion = &quot;Which ammunition does the AKM use?&quot;\nhits = retrieve(question, documents)\n\nfor hit in hits:\n    print(hit[&quot;id&quot;], &quot;-&gt;&quot;, hit[&quot;text&quot;])\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>这不是生产级搜索引擎。它忽略了词形变化、同义词、拼写变体、文档长度和许多排名信号。它的价值在于教育意义：\u003Cb>RAG 不是从向量数据库开始的。它是从检索开始的。\u003C\u002Fb>\u003C\u002Fp>\n\u003Ch2 id=\"section-16\">示例在哪里失效\u003C\u002Fh2>\n\u003Cp>当问题和来源使用不同的词时，精确匹配或词汇匹配会变得薄弱。文档可能说“车辆维护”，而用户问“我如何修理我的车？”词汇检索器可能会错过这种关系，即使人类立即就能看出。语义检索通过将文本表示为向量并比较含义而不仅仅是精确的词元来解决这个问题。\u003C\u002Fp>\n\u003Cp>长文件会带来另一个问题。将整本 80 页的手册作为一个单元搜索太粗糙，但拆分每个句子可能会破坏有用的上下文。因此，真正的 RAG 系统需要对解析、分块、元数据、排名、新鲜度、权限和来源做出决策。\u003C\u002Fp>\n\u003Cp>该示例也没有提及结构化的实时事实。如果用户询问订单 4711 的当前状态，而应用程序已经拥有数据库键，那么语义搜索通常是错误的首选工具。确定性的数据库查询更好。\u003C\u002Fp>\n\u003Ch2 id=\"section-20\">直接回答\u003C\u002Fh2>\n\u003Cp>LLM 数据源是应用程序可以从中为模型获取信息的任何外部系统：文件、数据库、API、搜索索引、向量存储或实时应用程序状态。RAG 是\u003Cb>在生成之前从此类来源检索相关知识\u003C\u002Fb>的模式。\u003C\u002Fp>\n\u003Cp>在 Python 中，基本流程可以非常小：\u003Cb>加载数据 → 创建可检索单元 → 查找相关证据 → 组装上下文 → 调用 LLM\u003C\u002Fb>。检索方法应与来源和问题相匹配。使用 SQL 获取精确的结构化事实，使用全文搜索进行词汇匹配，使用嵌入进行语义相似性匹配，并在多种信号有价值时使用混合检索。\u003C\u002Fp>\n\u003Ch2 id=\"section-23\">为什么如此\u003C\u002Fh2>\n\u003Cp>语言模型不会自动接收你的文件系统、PostgreSQL 数据库、CRM、私有 API 或新编辑文档的内容。应用程序决定哪些外部信息可访问，以及哪些信息被放入模型的当前上下文中。\u003C\u002Fp>\n\u003Cp>Lewis 等人的原始检索增强生成工作将生成模型与从密集向量索引中检索的外部非参数记忆相结合。更广泛的架构思想在该特定实现之外仍然存在：可以在推理时检索外部证据，而不是期望所有有用的知识都编码在模型参数中。\u003C\u002Fp>\n\u003Cp>这创建了有用的职责分离：来源存储信息，检索器选择证据，上下文将证据带入请求，模型解释它。保持这些边界可见使得故障更容易诊断。\u003C\u002Fp>\n\u003Ch2 id=\"section-27\">上下文：数据源的主要类型\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">来源\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">典型检索方法\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">适用于\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">TXT \u002F Markdown \u002F HTML\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">解析 + 词汇或语义搜索\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">文档、手册、文章、笔记\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">PDF \u002F DOCX\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">结构感知提取 + 搜索\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">政策、报告、合同、手册\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">SQL 数据库\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">SQL 查询或过滤检索\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">订单、用户、产品、结构化记录\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">REST \u002F GraphQL API\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">带参数的 HTTP 请求\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">远程系统和实时服务数据\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">搜索索引\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">BM25 \u002F 全文 \u002F 混合搜索\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">大型文本集合\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">向量索引\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">嵌入相似性\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">语义文档检索\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">应用程序状态\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">直接状态读取或工具调用\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">当前真实情况\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Cp>向量索引值得特别关注。在许多架构中，它\u003Cb>不是规范的真相来源\u003C\u002Fb>。它是从文档或记录派生的检索索引。权威文档可能存在于对象存储、CMS、Git、PostgreSQL 或其他系统中，而嵌入和元数据则单独存储以便快速语义查找。有些系统确实使用向量存储作为主存储，但这是架构选择，而不是 RAG 的要求。\u003C\u002Fp>\n\u003Cp>如果检索、持久记忆、当前状态和模型上下文之间的边界仍然不清楚，请参阅\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context\">AI Agent Memory Is Not RAG\u003C\u002Fa>。这些层可以使用一些相同的存储技术，但仍然有不同的正确性规则。\u003C\u002Fp>\n\u003Ch2 id=\"section-31\">假设\u003C\u002Fh2>\n\u003Cul>\u003Cli>允许应用程序访问外部来源。\u003C\u002Fli>\u003Cli>相关来源包含足够的信息来回答问题。\u003C\u002Fli>\u003Cli>数据可以以检索层可以使用的形式解析或查询。\u003C\u002Fli>\u003Cli>检索到的信息对于所请求的决策足够新鲜。\u003C\u002Fli>\u003Cli>模型在其上下文中接收选定的证据。\u003C\u002Fli>\u003Cli>在受保护的证据到达模型之前强制执行授权。\u003C\u002Fli>\u003Cli>即使检索正确，生成模型仍然可能出错。\u003C\u002Fli>\u003C\u002Ful>\n\u003Cp>这些假设很重要，因为检索无法弥补缺失的证据、过时的来源版本、损坏的解析器或未经授权的访问。RAG 管道的可信度只能与其提供的证据路径一样高。\u003C\u002Fp>\n\u003Ch2 id=\"section-34\">变量\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">变量\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">为什么它会改变设计\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">来源结构\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">SQL 表、法律 PDF 和源代码仓库需要不同的检索策略\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">问题类型\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">精确查找、概念搜索和多跳研究是不同的任务\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">新鲜度要求\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">实时状态可能需要直接查询，而不是定期重建索引\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">语料库大小\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">内存搜索可能适用于数百个块，但不适用于非常大的集合\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">语言\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">多语言检索需要适合实际语言的模型和分词\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">权限\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">检索必须根据当前用户的访问权限进行过滤\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">延迟和成本\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">更多的检索阶段可以提高质量，但会增加运行时和基础设施成本\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">来源需求\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">高信任系统需要源 ID、版本和可追溯的证据\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-36\">诊断 \u002F 决策方法\u003C\u002Fh2>\n\u003Cp>第一个决策不是“我应该安装哪个向量数据库？”而是：\u003Cb>我试图检索的是哪种事实？\u003C\u002Fb>\u003C\u002Fp>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">问题类型\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">首选方法\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">原因\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">精确 ID 或当前记录\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">SQL \u002F 键查找 \u002F API\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">确定性的结构化访问\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">精确措辞、代码、名称\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">全文或关键词搜索\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">词汇精确性\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">针对文档的概念性问题\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">语义向量搜索\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">含义可能与措辞不同\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">混合企业知识\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">混合检索 + 元数据过滤\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">结合词汇和语义信号\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">当前应用状态\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">直接状态\u002F工具访问\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">新鲜度比文档相似性更重要\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Cp>一个有用的测试是：\u003Cb>我已经知道我需要哪条记录，还是系统必须发现哪个段落是相关的？\u003C\u002Fb>如果记录已知，直接查询它。如果相关性必须被发现，搜索就变得更加重要。\u003C\u002Fp>\n\u003Cp>当答案错误时，按顺序诊断流水线，而不是立即更换 LLM：\u003C\u002Fp>\n\u003Col>\u003Cli>\u003Cb>1. 源覆盖：\u003C\u002Fb>正确信息是否存在于可访问的源集合中？\u003C\u002Fli>\u003Cli>\u003Cb>2. 新鲜度：\u003C\u002Fb>该版本对于问题来说是否足够新？\u003C\u002Fli>\u003Cli>\u003Cb>3. 解析：\u003C\u002Fb>相关内容是否被正确提取？\u003C\u002Fli>\u003Cli>\u003Cb>4. 分块：\u003C\u002Fb>证据是否与其赋予含义的条件保持在一起？\u003C\u002Fli>\u003Cli>\u003Cb>5. 检索：\u003C\u002Fb>正确的块是否出现在候选中？\u003C\u002Fli>\u003Cli>\u003Cb>6. 排序：\u003C\u002Fb>更强的来源是否排在较弱或冲突的来源之上？\u003C\u002Fli>\u003Cli>\u003Cb>7. 上下文组装：\u003C\u002Fb>应用程序是否实际将选定的证据发送给模型？\u003C\u002Fli>\u003Cli>\u003Cb>8. 生成：\u003C\u002Fb>LLM 是否忠实地使用了提供的证据？\u003C\u002Fli>\u003Cli>\u003Cb>9. 归因：\u003C\u002Fb>每个重要声明是否都能追溯到来源？\u003C\u002Fli>\u003C\u002Fol>\n\u003Cp>有关更深入的生产调试方法，请参阅\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method\">RAG 失败了——但究竟是哪一层失败了？一种诊断方法\u003C\u002Fa>，它将此链条扩展为可独立测试的故障层。\u003C\u002Fp>\n\u003Ch2 id=\"section-43\">证据\u003C\u002Fh2>\n\u003Cp>Lewis 等人的 RAG 论文形式化了基于检索到的外部记忆进行条件生成，而不是仅依赖模型参数。这为将生成器与可检索的知识源分离提供了概念基础。\u003C\u002Fp>\n\u003Cp>Sentence Transformers 将语义搜索记录为将语料库和查询嵌入到向量空间中，并检索具有高语义相似性的项目。其当前 API 还区分了检索任务中的查询编码和文档编码。\u003C\u002Fp>\n\u003Cp>SQLite FTS5 展示了光谱的另一端：成熟的全文检索可以在没有嵌入的情况下对文档进行排序。这很重要，因为词汇搜索对于标识符、精确术语和许多混合检索设计仍然有价值。\u003C\u002Fp>\n\u003Cp>OpenAI 的嵌入文档将嵌入描述为用于相关性和搜索的数值向量表示。这是语义检索的一种实现路径，而不是 RAG 本身的定义。\u003C\u002Fp>\n\u003Ch2 id=\"section-48\">真实示例 1：一个文本文件文件夹\u003C\u002Fh2>\n\u003Cp>假设一个名为 \u003Ccode>knowledge\u002F\u003C\u002Fcode> 的目录包含普通文本文件。Python 可以在完全不使用 AI 库的情况下加载它们。\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>from pathlib import Path\n\ndef load_text_files(folder=&quot;knowledge&quot;):\n    documents = []\n\n    for path in Path(folder).glob(&quot;*.txt&quot;):\n        documents.append({\n            &quot;source&quot;: path.name,\n            &quot;text&quot;: path.read_text(encoding=&quot;utf-8&quot;)\n        })\n\n    return documents\n\ndocuments = load_text_files()\n\nfor document in documents:\n    print(document[&quot;source&quot;], len(document[&quot;text&quot;]))\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>文件系统就是数据源。下一个问题是多少文本应该成为一个可检索单元。对于长文档，搜索一个完整文件通常太粗糙。这就是 RAG 流水线通常创建块的原因。\u003C\u002Fp>\n\u003Ch3 id=\"section-52\">一个非常简单的分块器\u003C\u002Fh3>\n\u003Cpre class=\"code-block\">\u003Ccode>def chunk_text(text, max_chars=800):\n    paragraphs = [p.strip() for p in text.split(&quot;\\n\\n&quot;) if p.strip()]\n\n    chunks = []\n    current = &quot;&quot;\n\n    for paragraph in paragraphs:\n        candidate = f&quot;{current}\\n\\n{paragraph}&quot;.strip()\n\n        if current and len(candidate) &gt; max_chars:\n            chunks.append(current)\n            current = paragraph\n        else:\n            current = candidate\n\n    if current:\n        chunks.append(current)\n\n    return chunks\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>此示例将段落分组，直到达到粗略的字符限制。它有意设计为易于理解而非最优。生产系统通常按标记、标题、章节、句子边界或文档结构进行分块。表格、源代码、合同和 API 文档可能需要不同的策略。\u003C\u002Fp>\n\u003Ch3 id=\"section-55\">分块时保留来源信息\u003C\u002Fh3>\n\u003Cpre class=\"code-block\">\u003Ccode>def build_chunks(documents):\n    chunks = []\n\n    for document in documents:\n        for index, text in enumerate(chunk_text(document[&quot;text&quot;])):\n            chunks.append({\n                &quot;id&quot;: f&#39;{document[&quot;source&quot;]}:{index}&#39;,\n                &quot;source&quot;: document[&quot;source&quot;],\n                &quot;chunk&quot;: index,\n                &quot;text&quot;: text,\n            })\n\n    return chunks\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>有用的分块承载的不仅仅是文本。来源名称、文档 ID、URL、时间戳、版本或章节，之后都可以支持引用、调试和新鲜度检查。如果在摄取过程中丢失了来源信息，解释某个特定答案为何产生就会变得困难得多。\u003C\u002Fp>\n\u003Ch2 id=\"section-58\">真实示例 2：结构化数据——当 SQL 是正确工具时就使用 SQL\u003C\u002Fh2>\n\u003Cp>并非每个外部事实都应该经过语义搜索。如果问题要求的是某个精确的当前记录，直接查询数据库通常更清晰、更具确定性。\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>import sqlite3\n\ndef get_order_status(order_id):\n    connection = sqlite3.connect(&quot;shop.db&quot;)\n    cursor = connection.cursor()\n\n    cursor.execute(\n        &quot;SELECT status, total, currency FROM orders WHERE id = ?&quot;,\n        (order_id,)\n    )\n\n    row = cursor.fetchone()\n    connection.close()\n\n    if row is None:\n        return None\n\n    return {\n        &quot;order_id&quot;: order_id,\n        &quot;status&quot;: row[0],\n        &quot;total&quot;: row[1],\n        &quot;currency&quot;: row[2],\n    }\n\nprint(get_order_status(4711))\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>如果应用已经知道用户询问的是订单 4711，那么把整个订单表嵌入，并让语义搜索重新发现那一行，通常只会增加复杂性而没有收益。一个强有力的设计规则是：\u003Cb>用结构化查询检索结构化事实；用搜索检索非结构化知识。\u003C\u002Fb>\u003C\u002Fp>\n\u003Cp>返回的数据库行仍然可以放入模型上下文中，以便 LLM 用自然语言解释它。但直接访问状态或记录，在概念上不同于搜索知识语料库。\u003C\u002Fp>\n\u003Ch2 id=\"section-63\">真实示例 3：在嵌入之前先做全文搜索\u003C\u002Fh2>\n\u003Cp>在朴素的 Python 循环和向量搜索之间，存在一类成熟的词法检索系统。SQLite 包含用于全文搜索的 FTS5，其中包括 BM25 排序。\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>import sqlite3\n\nconnection = sqlite3.connect(&quot;knowledge.db&quot;)\ncursor = connection.cursor()\n\ncursor.execute(\n    &quot;CREATE VIRTUAL TABLE IF NOT EXISTS docs USING fts5(title, body)&quot;\n)\n\ncursor.execute(\n    &quot;INSERT INTO docs(title, body) VALUES (?, ?)&quot;,\n    (&quot;AKM&quot;, &quot;The AKM uses 7.62 mm ammunition.&quot;)\n)\n\nconnection.commit()\n\nquery = &quot;AKM ammunition&quot;\n\nrows = cursor.execute(\n    &quot;SELECT title, body, bm25(docs) AS score &quot;\n    &quot;FROM docs WHERE docs MATCH ? &quot;\n    &quot;ORDER BY score LIMIT 5&quot;,\n    (query,)\n).fetchall()\n\nfor row in rows:\n    print(row)\n\nconnection.close()\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>当精确术语、产品代码、名称、标识符或领域特定词汇很重要时，词法搜索尤其有用。语义搜索并不自动更好。生产系统通常会结合这两种信号。\u003C\u002Fp>\n\u003Ch2 id=\"section-67\">真实示例 4：使用嵌入进行语义检索\u003C\u002Fh2>\n\u003Cp>嵌入将文本转换为数值向量，因此即使语义相关的段落没有使用完全相同的措辞，也可以进行比较。Sentence Transformers 提供了一种直接的本地实现。\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode># pip install sentence-transformers\n\nfrom sentence_transformers import SentenceTransformer, util\n\ndocuments = [\n    &quot;The AKM uses 7.62 mm ammunition.&quot;,\n    &quot;A Med Kit restores health.&quot;,\n    &quot;Vehicle maintenance includes checking oil, brakes and tires.&quot;,\n    &quot;Account recovery requires access to the registered email address.&quot;\n]\n\nmodel = SentenceTransformer(\n    &quot;sentence-transformers\u002Fmulti-qa-mpnet-base-cos-v1&quot;\n)\n\ndocument_embeddings = model.encode_document(\n    documents,\n    convert_to_tensor=True\n)\n\nquestion = &quot;How do I repair my car?&quot;\n\nquery_embedding = model.encode_query(\n    question,\n    convert_to_tensor=True\n)\n\nhits = util.semantic_search(\n    query_embedding,\n    document_embeddings,\n    top_k=2\n)[0]\n\nfor hit in hits:\n    print(round(float(hit[&quot;score&quot;]), 3), documents[hit[&quot;corpus_id&quot;]])\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>查询中不包含“车辆维护”这个短语，但语义模型仍然可以给该段落很高的排名，因为这些概念是相关的。这就是嵌入在 RAG 系统中常见的实际原因。\u003C\u002Fp>\n\u003Cp>对于小型集合，嵌入可以保存在内存中。更大的系统通常会将它们持久化到支持向量的索引或数据库中，并在那里执行最近邻搜索。存储方式变了，但逻辑仍然不变：编码问题，找到相关的文档表示，返回最佳证据。\u003C\u002Fp>\n\u003Ch2 id=\"section-72\">真实示例 5：为 LLM 构建上下文\u003C\u002Fh2>\n\u003Cp>检索器应返回证据。然后，LLM 应接收问题以及该证据。将检索和生成分开，使两者都更易于检查和测试。\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode>def build_prompt(question, retrieved_documents):\n    context = &quot;\\n\\n&quot;.join(\n        f&#39;[{doc[&quot;id&quot;]}] {doc[&quot;text&quot;]}&#39;\n        for doc in retrieved_documents\n    )\n\n    return f&quot;&quot;&quot;\nAnswer the question using the supplied context.\n\nRules:\n- Do not invent facts that are not supported by the context.\n- If the context is insufficient, say so.\n- Cite the source IDs you used.\n\nQuestion:\n{question}\n\nContext:\n{context}\n&quot;&quot;&quot;.strip()\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>该指令并不会使模型变得绝对可靠。它只是创建了一个明确的证据边界。模型仍然可能误解好的证据、忽略某个条件或过度泛化。这就是为什么检索质量和生成质量必须分别评估。\u003C\u002Fp>\n\u003Ch2 id=\"section-76\">真实示例 6：一个完整的最小流水线\u003C\u002Fh2>\n\u003Cpre class=\"code-block\">\u003Ccode>def answer_question(question, all_documents, call_llm):\n    # 1. Retrieve evidence\n    retrieved = retrieve(question, all_documents, top_k=3)\n\n    # 2. Build model context\n    prompt = build_prompt(question, retrieved)\n\n    # 3. Generate the answer\n    answer = call_llm(prompt)\n\n    return {\n        &quot;answer&quot;: answer,\n        &quot;sources&quot;: [doc[&quot;id&quot;] for doc in retrieved]\n    }\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>该函数有意将 \u003Ccode>call_llm\u003C\u002Fcode> 作为依赖项接收。检索不应关心生成是由云模型、本地模型还是其他提供方执行的。数据路径属于应用程序。\u003C\u002Fp>\n\u003Ch3 id=\"section-79\">可选生成器：OpenAI Responses API\u003C\u002Fh3>\n\u003Cp>一种可能的生成器是 OpenAI Responses API。将模型名称保留在环境变量中，可以避免将特定模型硬编码到 RAG 架构中。\u003C\u002Fp>\n\u003Cpre class=\"code-block\">\u003Ccode># pip install openai\n\nimport os\nfrom openai import OpenAI\n\nclient = OpenAI()\n\ndef call_llm(prompt):\n    response = client.responses.create(\n        model=os.environ[&quot;OPENAI_MODEL&quot;],\n        input=prompt,\n    )\n    return response.output_text\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>同一个检索流水线可以连接到本地推理服务器。这是一个重要的架构要点：\u003Cb>RAG 并不属于 LLM 提供方。\u003C\u002Fb>应用程序拥有来源、检索和上下文组装。\u003C\u002Fp>\n\u003Ch3 id=\"section-83\">整体架构一览\u003C\u002Fh3>\n\u003Cpre class=\"code-block\">\u003Ccode>USER QUESTION\n     |\n     v\n+-------------+\n|  Retriever  |\n+-------------+\n   |       |\n   |       +----&gt; SQL \u002F API \u002F state query\n   |\n   +------------&gt; keyword \u002F full-text search\n   |\n   +------------&gt; embedding \u002F vector search\n                     |\n                     v\n              relevant evidence\n                     |\n                     v\n+-----------------------------------+\n| question + evidence + instructions |\n+-----------------------------------+\n                     |\n                     v\n                   LLM\n                     |\n                     v\n                  answer\u003C\u002Fcode>\u003C\u002Fpre>\n\u003Cp>这种数据流模型比记住某一个框架更持久。库、数据库和模型供应商都会变化；但职责边界保持不变。\u003C\u002Fp>\n\u003Ch2 id=\"section-86\">常见误解与失败模式\u003C\u002Fh2>\n\u003Ch3 id=\"section-87\">“RAG 就意味着向量数据库。”\u003C\u002Fh3>\n\u003Cp>不。向量搜索只是一种检索方法。RAG 可以使用全文搜索、SQL、API、知识图谱、向量搜索或它们的组合。其定义性模式是为生成而检索外部信息。\u003C\u002Fp>\n\u003Ch3 id=\"section-89\">“如果数据在 PostgreSQL 中，我就必须嵌入整个数据库。”\u003C\u002Fh3>\n\u003Cp>不。结构化记录通常应保持作为结构化记录可查询。嵌入对于语义相关性很有用，但不能替代确定性查询。\u003C\u002Fp>\n\u003Ch3 id=\"section-91\">“更多分块意味着更好的答案。”\u003C\u002Fh3>\n\u003Cp>未必如此。额外的上下文可能引入噪声、冲突版本和无关材料。检索应优化有用的证据，而非最大数量。\u003C\u002Fp>\n\u003Ch3 id=\"section-93\">“高相似度分数证明了答案。”\u003C\u002Fh3>\n\u003Cp>不。相似度衡量的是相关性，而非真实性或适用性。高度相似的段落可能已过时、来自错误的产品版本，或仅在不符合问题的条件下有效。\u003C\u002Fp>\n\u003Ch3 id=\"section-95\">“一旦检索到正确的分块，幻觉就解决了。”\u003C\u002Fh3>\n\u003Cp>不。检索改善了基础，但不保证忠实的推理。生成仍需评估，高风险工作流可能需要确定性验证或人工审查。\u003C\u002Fp>\n\u003Ch3 id=\"section-97\">“模型失败了，所以更换模型。”\u003C\u002Fh3>\n\u003Cp>未必如此。正确的来源可能缺失、解析错误、分割不当、被过滤掉、排名过低或未包含在组装的上下文中。更换模型不应是第一个诊断步骤。\u003C\u002Fp>\n\u003Ch2 id=\"section-99\">边缘情况\u003C\u002Fh2>\n\u003Cul>\u003Cli>\u003Cb>冲突文档：\u003C\u002Fb>两个来源可能因版本、司法管辖区或产品不同而不一致。\u003C\u002Fli>\u003Cli>\u003Cb>时间敏感事实：\u003C\u002Fb>语义相关的来源可能已经过时。\u003C\u002Fli>\u003Cli>\u003Cb>权限：\u003C\u002Fb>检索器不得返回当前用户无权访问的文档。\u003C\u002Fli>\u003Cli>\u003Cb>多语言集合：\u003C\u002Fb>嵌入模型和检索策略必须支持实际使用的语言。\u003C\u002Fli>\u003Cli>\u003Cb>表格和源代码：\u003C\u002Fb>普通段落分块可能破坏对答案至关重要的结构。\u003C\u002Fli>\u003Cli>\u003Cb>极短标识符：\u003C\u002Fb>对于SKU、ID、错误代码或缩写，语义检索可能弱于精确匹配。\u003C\u002Fli>\u003Cli>\u003Cb>需要多个事实的长问题：\u003C\u002Fb>检索可能需要分解、多次搜索或重排序，而非一次top-k查询。\u003C\u002Fli>\u003Cli>\u003Cb>来源层级：\u003C\u002Fb>官方当前政策可能需要优先于较旧但语义更接近的讨论文档。\u003C\u002Fli>\u003C\u002Ful>\n\u003Ch2 id=\"section-101\">局限性\u003C\u002Fh2>\n\u003Cp>Python示例有意优化透明度，而非规模。关键词检索器是朴素的，分块器使用字符长度，SQLite示例不包含生产连接管理，语义示例将所有嵌入保存在内存中。\u003C\u002Fp>\n\u003Cp>生产系统可能需要向量索引、重排序器、混合检索、文档解析器、缓存、增量索引、来源版本控制、访问控制过滤器、可观测性、评估数据集和故障处理。这些添加都不会改变核心架构；它们使每个边界更可靠。\u003C\u002Fp>\n\u003Cp>RAG也无法创造来源集中缺失的证据。如果来源错误、不完整或过时，更好的嵌入模型也无法将其转化为权威知识。\u003C\u002Fp>\n\u003Ch2 id=\"section-105\">什么会改变这个答案？\u003C\u002Fh2>\n\u003Cp>当任务需要的不只是知识查找时，架构就会改变。实时订单状态需要当前状态。财务计算可能需要确定性代码。网络研究任务可能需要主动搜索。工作流可能需要能将数据写回另一个系统的工具。自主代理可能需要在检索之外进行规划、权限和执行控制。\u003C\u002Fp>\n\u003Cp>因此，RAG最好被理解为\u003Cb>更大AI系统中的一个证据获取层\u003C\u002Fb>。它之所以强大，正是因为它有一个狭窄的职责：找到有用的外部信息并将其放入模型的工作上下文中。\u003C\u002Fp>\n\u003Ch2 id=\"section-108\">结论\u003C\u002Fh2>\n\u003Cp>当去掉技术名称后，RAG 就变得容易理解多了。一个文件是一个来源。一个数据库是一个来源。一个 API 是一个来源。搜索函数检索证据。提示词将证据传递给模型。然后 LLM 对其进行解释并生成语言。\u003C\u002Fp>\n\u003Cp>生产级 RAG 的难点不在于调用嵌入模型。而在于从原始来源到最终主张构建一条可信的证据路径：保留来源信息、选择正确的检索方法、保持信息最新、控制访问权限、将检索评估与生成评估分开，以及知道何时直接调用数据库或工具比语义搜索更好。\u003C\u002Fp>\n\u003Cp>这就是基础 RAG 模型的实用延续：\u003Cb>先理解角色，然后让数据路径变得明确。\u003C\u002Fb>\u003C\u002Fp>\n\u003Ch2 id=\"section-112\">主要来源\u003C\u002Fh2>\n\u003Cul>\u003Cli>\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2005.11401\">Lewis 等人 — 面向知识密集型 NLP 任务的检索增强生成\u003C\u002Fa> — 2020 年的论文，介绍了将生成与检索到的非参数记忆相结合的 RAG 公式。\u003C\u002Fli>\u003Cli>\u003Ca href=\"https:\u002F\u002Fwww.sbert.net\u002Fexamples\u002Fsentence_transformer\u002Fapplications\u002Fsemantic-search\u002FREADME.html\">Sentence Transformers — 语义搜索\u003C\u002Fa> — 关于语义检索、查询嵌入和文档嵌入的官方文档。\u003C\u002Fli>\u003Cli>\u003Ca href=\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fembeddings\">OpenAI — 向量嵌入\u003C\u002Fa> — 官方文档，将嵌入描述为用于相关性和搜索的数值表示。\u003C\u002Fli>\u003Cli>\u003Ca href=\"https:\u002F\u002Fwww.sqlite.org\u002Ffts5.html\">SQLite — FTS5 扩展\u003C\u002Fa> — 关于 SQLite 中全文搜索和 BM25 排名的官方文档。\u003C\u002Fli>\u003Cli>\u003Ca href=\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Flibraries\">OpenAI — SDK 和 CLI\u003C\u002Fa> — 用于可选生成器示例中 Responses API 的官方 Python SDK 示例。\u003C\u002Fli>\u003Cli>\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works\">什么是 RAG？对其工作原理的最简单解释\u003C\u002Fa> — 本系列的概念性第一部分。\u003C\u002Fli>\u003C\u002Ful>",{"time":212,"blocks":213,"version":675},1790517477362,[214,218,221,227,231,234,237,240,243,246,249,253,256,259,262,265,268,271,274,277,280,283,286,289,292,295,298,301,337,340,343,346,358,361,364,393,396,399,425,428,431,444,447,450,453,456,459,462,465,468,471,474,478,481,484,487,490,493,496,499,502,505,508,511,514,517,520,523,526,529,532,535,538,541,544,547,550,553,556,559,562,565,568,571,574,577,580,583,586,589,592,595,598,601,604,607,610,613,616,619,630,633,636,639,642,645,648,651,654,657,660,663,666],{"data":215,"type":217},{"text":216},"上一篇文章，\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works\">什么是 RAG？对其工作原理的最简单解释\u003C\u002Fa>，建立了思维模型：LLM 负责写作，RAG 检索有用的知识，应用程序拥有当前状态，工具执行操作。本文迈出下一步：\u003Cb>数据实际上来自哪里，以及在 Python 中检索是什么样子的？\u003C\u002Fb>","paragraph",{"data":219,"type":217},{"text":220},"重要的惊喜在于，“LLM 数据源”通常并没有什么特别之处。它可以是一个文本文件、一个包含 Markdown 文档的文件夹、一个 SQL 数据库、一个 API 响应、一个产品目录、一个支持系统，或从这些来源派生的向量索引。AI 并不会神奇地知道这些系统。你的应用程序必须加载、查询、搜索或检索相关数据，并将结果放入模型的上下文中。",{"data":222,"type":226},{"text":223,"caption":224,"alignment":225},"数据源 = 信息所在之处。检索 = 应用程序如何找到有用的信息。上下文 = 提供给模型的选定信息。LLM = 解释该上下文并生成答案的组件。","本文通篇使用的四部分模型","left","quote",{"data":228,"type":42},{"text":229,"level":230},"问题",2,{"data":232,"type":217},{"text":233},"LLM 如何使用文件、数据库或 API 等外部数据，以及一个小型 Python 程序如何在不借助框架隐藏这些步骤的情况下实现基本的 RAG 步骤？",{"data":235,"type":42},{"text":236,"level":230},"这真正意味着什么",{"data":238,"type":217},{"text":239},"当开发者说 LLM“连接到公司数据”时，这句话背后可能隐藏着几种不同的操作。一个应用程序可能执行 SQL。另一个可能调用 API。另一个可能运行全文搜索。另一个可能计算文档块之间的嵌入相似度。所有这些都可以向 LLM 提供外部信息，但它们不是相同的检索方法，不应被视为可以互换。",{"data":241,"type":217},{"text":242},"这种区分很重要，因为最佳检索方法取决于问题的形态。“我们的退款政策是什么？”是一个文档检索问题。“订单 4711 的当前状态是什么？”通常是结构化数据库查询。“哪一段讨论了账户恢复？”可以是关键词或语义搜索。当系统必须在生成之前\u003Cb>发现相关知识\u003C\u002Fb>时，RAG 最为有用。",{"data":244,"type":42},{"text":245,"level":230},"最简单的例子",{"data":247,"type":217},{"text":248},"从普通 Python 中的三个字符串开始。没有向量数据库，没有框架，也还没有 LLM。我们只想让检索步骤变得可见。",{"data":250,"type":252},{"code":251},"documents = [\n    \"The AKM uses 7.62 mm ammunition.\",\n    \"A Med Kit restores health.\",\n    \"A 4x scope can be attached to several compatible weapons.\"\n]\n\nquestion = \"Which ammunition does the AKM use?\"\n\nfor document in documents:\n    if \"AKM\" in document:\n        print(document)","code",{"data":254,"type":217},{"text":255},"程序打印第一句话，因为它包含我们搜索的词。这是原始的检索，但架构已经可见：\u003Cb>问题 → 搜索 → 相关文本\u003C\u002Fb>。RAG 增加了一个重要步骤：将检索到的文本与问题一起传递给语言模型。",{"data":257,"type":217},{"text":258},"一个稍微更通用的版本根据查询词的重叠对文档进行排名：",{"data":260,"type":252},{"code":261},"import re\n\ndocuments = [\n    {\"id\": \"weapon-akm\", \"text\": \"The AKM uses 7.62 mm ammunition.\"},\n    {\"id\": \"healing-medkit\", \"text\": \"A Med Kit restores health.\"},\n    {\"id\": \"scope-4x\", \"text\": \"A 4x scope can be attached to several compatible weapons.\"},\n]\n\ndef words(text):\n    return set(re.findall(r\"[a-zA-Z0-9.]+\", text.lower()))\n\ndef retrieve(question, documents, top_k=2):\n    query_terms = words(question)\n    ranked = []\n\n    for document in documents:\n        score = len(query_terms & words(document[\"text\"]))\n        if score > 0:\n            ranked.append((score, document))\n\n    ranked.sort(key=lambda item: item[0], reverse=True)\n    return [document for _, document in ranked[:top_k]]\n\nquestion = \"Which ammunition does the AKM use?\"\nhits = retrieve(question, documents)\n\nfor hit in hits:\n    print(hit[\"id\"], \"->\", hit[\"text\"])",{"data":263,"type":217},{"text":264},"这不是生产级搜索引擎。它忽略了词形变化、同义词、拼写变体、文档长度和许多排名信号。它的价值在于教育意义：\u003Cb>RAG 不是从向量数据库开始的。它是从检索开始的。\u003C\u002Fb>",{"data":266,"type":42},{"text":267,"level":230},"示例在哪里失效",{"data":269,"type":217},{"text":270},"当问题和来源使用不同的词时，精确匹配或词汇匹配会变得薄弱。文档可能说“车辆维护”，而用户问“我如何修理我的车？”词汇检索器可能会错过这种关系，即使人类立即就能看出。语义检索通过将文本表示为向量并比较含义而不仅仅是精确的词元来解决这个问题。",{"data":272,"type":217},{"text":273},"长文件会带来另一个问题。将整本 80 页的手册作为一个单元搜索太粗糙，但拆分每个句子可能会破坏有用的上下文。因此，真正的 RAG 系统需要对解析、分块、元数据、排名、新鲜度、权限和来源做出决策。",{"data":275,"type":217},{"text":276},"该示例也没有提及结构化的实时事实。如果用户询问订单 4711 的当前状态，而应用程序已经拥有数据库键，那么语义搜索通常是错误的首选工具。确定性的数据库查询更好。",{"data":278,"type":42},{"text":279,"level":230},"直接回答",{"data":281,"type":217},{"text":282},"LLM 数据源是应用程序可以从中为模型获取信息的任何外部系统：文件、数据库、API、搜索索引、向量存储或实时应用程序状态。RAG 是\u003Cb>在生成之前从此类来源检索相关知识\u003C\u002Fb>的模式。",{"data":284,"type":217},{"text":285},"在 Python 中，基本流程可以非常小：\u003Cb>加载数据 → 创建可检索单元 → 查找相关证据 → 组装上下文 → 调用 LLM\u003C\u002Fb>。检索方法应与来源和问题相匹配。使用 SQL 获取精确的结构化事实，使用全文搜索进行词汇匹配，使用嵌入进行语义相似性匹配，并在多种信号有价值时使用混合检索。",{"data":287,"type":42},{"text":288,"level":230},"为什么如此",{"data":290,"type":217},{"text":291},"语言模型不会自动接收你的文件系统、PostgreSQL 数据库、CRM、私有 API 或新编辑文档的内容。应用程序决定哪些外部信息可访问，以及哪些信息被放入模型的当前上下文中。",{"data":293,"type":217},{"text":294},"Lewis 等人的原始检索增强生成工作将生成模型与从密集向量索引中检索的外部非参数记忆相结合。更广泛的架构思想在该特定实现之外仍然存在：可以在推理时检索外部证据，而不是期望所有有用的知识都编码在模型参数中。",{"data":296,"type":217},{"text":297},"这创建了有用的职责分离：来源存储信息，检索器选择证据，上下文将证据带入请求，模型解释它。保持这些边界可见使得故障更容易诊断。",{"data":299,"type":42},{"text":300,"level":230},"上下文：数据源的主要类型",{"data":302,"type":336},{"content":303,"withHeadings":14},[304,308,312,316,320,324,328,332],[305,306,307],"来源","典型检索方法","适用于",[309,310,311],"TXT \u002F Markdown \u002F HTML","解析 + 词汇或语义搜索","文档、手册、文章、笔记",[313,314,315],"PDF \u002F DOCX","结构感知提取 + 搜索","政策、报告、合同、手册",[317,318,319],"SQL 数据库","SQL 查询或过滤检索","订单、用户、产品、结构化记录",[321,322,323],"REST \u002F GraphQL API","带参数的 HTTP 请求","远程系统和实时服务数据",[325,326,327],"搜索索引","BM25 \u002F 全文 \u002F 混合搜索","大型文本集合",[329,330,331],"向量索引","嵌入相似性","语义文档检索",[333,334,335],"应用程序状态","直接状态读取或工具调用","当前真实情况","table",{"data":338,"type":217},{"text":339},"向量索引值得特别关注。在许多架构中，它\u003Cb>不是规范的真相来源\u003C\u002Fb>。它是从文档或记录派生的检索索引。权威文档可能存在于对象存储、CMS、Git、PostgreSQL 或其他系统中，而嵌入和元数据则单独存储以便快速语义查找。有些系统确实使用向量存储作为主存储，但这是架构选择，而不是 RAG 的要求。",{"data":341,"type":217},{"text":342},"如果检索、持久记忆、当前状态和模型上下文之间的边界仍然不清楚，请参阅\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context\">AI Agent Memory Is Not RAG\u003C\u002Fa>。这些层可以使用一些相同的存储技术，但仍然有不同的正确性规则。",{"data":344,"type":42},{"text":345,"level":230},"假设",{"data":347,"type":357},{"items":348,"style":356},[349,350,351,352,353,354,355],"允许应用程序访问外部来源。","相关来源包含足够的信息来回答问题。","数据可以以检索层可以使用的形式解析或查询。","检索到的信息对于所请求的决策足够新鲜。","模型在其上下文中接收选定的证据。","在受保护的证据到达模型之前强制执行授权。","即使检索正确，生成模型仍然可能出错。","unordered","list",{"data":359,"type":217},{"text":360},"这些假设很重要，因为检索无法弥补缺失的证据、过时的来源版本、损坏的解析器或未经授权的访问。RAG 管道的可信度只能与其提供的证据路径一样高。",{"data":362,"type":42},{"text":363,"level":230},"变量",{"data":365,"type":336},{"content":366,"withHeadings":14},[367,369,372,375,378,381,384,387,390],[363,368],"为什么它会改变设计",[370,371],"来源结构","SQL 表、法律 PDF 和源代码仓库需要不同的检索策略",[373,374],"问题类型","精确查找、概念搜索和多跳研究是不同的任务",[376,377],"新鲜度要求","实时状态可能需要直接查询，而不是定期重建索引",[379,380],"语料库大小","内存搜索可能适用于数百个块，但不适用于非常大的集合",[382,383],"语言","多语言检索需要适合实际语言的模型和分词",[385,386],"权限","检索必须根据当前用户的访问权限进行过滤",[388,389],"延迟和成本","更多的检索阶段可以提高质量，但会增加运行时和基础设施成本",[391,392],"来源需求","高信任系统需要源 ID、版本和可追溯的证据",{"data":394,"type":42},{"text":395,"level":230},"诊断 \u002F 决策方法",{"data":397,"type":217},{"text":398},"第一个决策不是“我应该安装哪个向量数据库？”而是：\u003Cb>我试图检索的是哪种事实？\u003C\u002Fb>",{"data":400,"type":336},{"content":401,"withHeadings":14},[402,405,409,413,417,421],[373,403,404],"首选方法","原因",[406,407,408],"精确 ID 或当前记录","SQL \u002F 键查找 \u002F API","确定性的结构化访问",[410,411,412],"精确措辞、代码、名称","全文或关键词搜索","词汇精确性",[414,415,416],"针对文档的概念性问题","语义向量搜索","含义可能与措辞不同",[418,419,420],"混合企业知识","混合检索 + 元数据过滤","结合词汇和语义信号",[422,423,424],"当前应用状态","直接状态\u002F工具访问","新鲜度比文档相似性更重要",{"data":426,"type":217},{"text":427},"一个有用的测试是：\u003Cb>我已经知道我需要哪条记录，还是系统必须发现哪个段落是相关的？\u003C\u002Fb>如果记录已知，直接查询它。如果相关性必须被发现，搜索就变得更加重要。",{"data":429,"type":217},{"text":430},"当答案错误时，按顺序诊断流水线，而不是立即更换 LLM：",{"data":432,"type":357},{"items":433,"style":443},[434,435,436,437,438,439,440,441,442],"\u003Cb>1. 源覆盖：\u003C\u002Fb>正确信息是否存在于可访问的源集合中？","\u003Cb>2. 新鲜度：\u003C\u002Fb>该版本对于问题来说是否足够新？","\u003Cb>3. 解析：\u003C\u002Fb>相关内容是否被正确提取？","\u003Cb>4. 分块：\u003C\u002Fb>证据是否与其赋予含义的条件保持在一起？","\u003Cb>5. 检索：\u003C\u002Fb>正确的块是否出现在候选中？","\u003Cb>6. 排序：\u003C\u002Fb>更强的来源是否排在较弱或冲突的来源之上？","\u003Cb>7. 上下文组装：\u003C\u002Fb>应用程序是否实际将选定的证据发送给模型？","\u003Cb>8. 生成：\u003C\u002Fb>LLM 是否忠实地使用了提供的证据？","\u003Cb>9. 归因：\u003C\u002Fb>每个重要声明是否都能追溯到来源？","ordered",{"data":445,"type":217},{"text":446},"有关更深入的生产调试方法，请参阅\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method\">RAG 失败了——但究竟是哪一层失败了？一种诊断方法\u003C\u002Fa>，它将此链条扩展为可独立测试的故障层。",{"data":448,"type":42},{"text":449,"level":230},"证据",{"data":451,"type":217},{"text":452},"Lewis 等人的 RAG 论文形式化了基于检索到的外部记忆进行条件生成，而不是仅依赖模型参数。这为将生成器与可检索的知识源分离提供了概念基础。",{"data":454,"type":217},{"text":455},"Sentence Transformers 将语义搜索记录为将语料库和查询嵌入到向量空间中，并检索具有高语义相似性的项目。其当前 API 还区分了检索任务中的查询编码和文档编码。",{"data":457,"type":217},{"text":458},"SQLite FTS5 展示了光谱的另一端：成熟的全文检索可以在没有嵌入的情况下对文档进行排序。这很重要，因为词汇搜索对于标识符、精确术语和许多混合检索设计仍然有价值。",{"data":460,"type":217},{"text":461},"OpenAI 的嵌入文档将嵌入描述为用于相关性和搜索的数值向量表示。这是语义检索的一种实现路径，而不是 RAG 本身的定义。",{"data":463,"type":42},{"text":464,"level":230},"真实示例 1：一个文本文件文件夹",{"data":466,"type":217},{"text":467},"假设一个名为 \u003Ccode>knowledge\u002F\u003C\u002Fcode> 的目录包含普通文本文件。Python 可以在完全不使用 AI 库的情况下加载它们。",{"data":469,"type":252},{"code":470},"from pathlib import Path\n\ndef load_text_files(folder=\"knowledge\"):\n    documents = []\n\n    for path in Path(folder).glob(\"*.txt\"):\n        documents.append({\n            \"source\": path.name,\n            \"text\": path.read_text(encoding=\"utf-8\")\n        })\n\n    return documents\n\ndocuments = load_text_files()\n\nfor document in documents:\n    print(document[\"source\"], len(document[\"text\"]))",{"data":472,"type":217},{"text":473},"文件系统就是数据源。下一个问题是多少文本应该成为一个可检索单元。对于长文档，搜索一个完整文件通常太粗糙。这就是 RAG 流水线通常创建块的原因。",{"data":475,"type":42},{"text":476,"level":477},"一个非常简单的分块器",3,{"data":479,"type":252},{"code":480},"def chunk_text(text, max_chars=800):\n    paragraphs = [p.strip() for p in text.split(\"\\n\\n\") if p.strip()]\n\n    chunks = []\n    current = \"\"\n\n    for paragraph in paragraphs:\n        candidate = f\"{current}\\n\\n{paragraph}\".strip()\n\n        if current and len(candidate) > max_chars:\n            chunks.append(current)\n            current = paragraph\n        else:\n            current = candidate\n\n    if current:\n        chunks.append(current)\n\n    return chunks",{"data":482,"type":217},{"text":483},"此示例将段落分组，直到达到粗略的字符限制。它有意设计为易于理解而非最优。生产系统通常按标记、标题、章节、句子边界或文档结构进行分块。表格、源代码、合同和 API 文档可能需要不同的策略。",{"data":485,"type":42},{"text":486,"level":477},"分块时保留来源信息",{"data":488,"type":252},{"code":489},"def build_chunks(documents):\n    chunks = []\n\n    for document in documents:\n        for index, text in enumerate(chunk_text(document[\"text\"])):\n            chunks.append({\n                \"id\": f'{document[\"source\"]}:{index}',\n                \"source\": document[\"source\"],\n                \"chunk\": index,\n                \"text\": text,\n            })\n\n    return chunks",{"data":491,"type":217},{"text":492},"有用的分块承载的不仅仅是文本。来源名称、文档 ID、URL、时间戳、版本或章节，之后都可以支持引用、调试和新鲜度检查。如果在摄取过程中丢失了来源信息，解释某个特定答案为何产生就会变得困难得多。",{"data":494,"type":42},{"text":495,"level":230},"真实示例 2：结构化数据——当 SQL 是正确工具时就使用 SQL",{"data":497,"type":217},{"text":498},"并非每个外部事实都应该经过语义搜索。如果问题要求的是某个精确的当前记录，直接查询数据库通常更清晰、更具确定性。",{"data":500,"type":252},{"code":501},"import sqlite3\n\ndef get_order_status(order_id):\n    connection = sqlite3.connect(\"shop.db\")\n    cursor = connection.cursor()\n\n    cursor.execute(\n        \"SELECT status, total, currency FROM orders WHERE id = ?\",\n        (order_id,)\n    )\n\n    row = cursor.fetchone()\n    connection.close()\n\n    if row is None:\n        return None\n\n    return {\n        \"order_id\": order_id,\n        \"status\": row[0],\n        \"total\": row[1],\n        \"currency\": row[2],\n    }\n\nprint(get_order_status(4711))",{"data":503,"type":217},{"text":504},"如果应用已经知道用户询问的是订单 4711，那么把整个订单表嵌入，并让语义搜索重新发现那一行，通常只会增加复杂性而没有收益。一个强有力的设计规则是：\u003Cb>用结构化查询检索结构化事实；用搜索检索非结构化知识。\u003C\u002Fb>",{"data":506,"type":217},{"text":507},"返回的数据库行仍然可以放入模型上下文中，以便 LLM 用自然语言解释它。但直接访问状态或记录，在概念上不同于搜索知识语料库。",{"data":509,"type":42},{"text":510,"level":230},"真实示例 3：在嵌入之前先做全文搜索",{"data":512,"type":217},{"text":513},"在朴素的 Python 循环和向量搜索之间，存在一类成熟的词法检索系统。SQLite 包含用于全文搜索的 FTS5，其中包括 BM25 排序。",{"data":515,"type":252},{"code":516},"import sqlite3\n\nconnection = sqlite3.connect(\"knowledge.db\")\ncursor = connection.cursor()\n\ncursor.execute(\n    \"CREATE VIRTUAL TABLE IF NOT EXISTS docs USING fts5(title, body)\"\n)\n\ncursor.execute(\n    \"INSERT INTO docs(title, body) VALUES (?, ?)\",\n    (\"AKM\", \"The AKM uses 7.62 mm ammunition.\")\n)\n\nconnection.commit()\n\nquery = \"AKM ammunition\"\n\nrows = cursor.execute(\n    \"SELECT title, body, bm25(docs) AS score \"\n    \"FROM docs WHERE docs MATCH ? \"\n    \"ORDER BY score LIMIT 5\",\n    (query,)\n).fetchall()\n\nfor row in rows:\n    print(row)\n\nconnection.close()",{"data":518,"type":217},{"text":519},"当精确术语、产品代码、名称、标识符或领域特定词汇很重要时，词法搜索尤其有用。语义搜索并不自动更好。生产系统通常会结合这两种信号。",{"data":521,"type":42},{"text":522,"level":230},"真实示例 4：使用嵌入进行语义检索",{"data":524,"type":217},{"text":525},"嵌入将文本转换为数值向量，因此即使语义相关的段落没有使用完全相同的措辞，也可以进行比较。Sentence Transformers 提供了一种直接的本地实现。",{"data":527,"type":252},{"code":528},"# pip install sentence-transformers\n\nfrom sentence_transformers import SentenceTransformer, util\n\ndocuments = [\n    \"The AKM uses 7.62 mm ammunition.\",\n    \"A Med Kit restores health.\",\n    \"Vehicle maintenance includes checking oil, brakes and tires.\",\n    \"Account recovery requires access to the registered email address.\"\n]\n\nmodel = SentenceTransformer(\n    \"sentence-transformers\u002Fmulti-qa-mpnet-base-cos-v1\"\n)\n\ndocument_embeddings = model.encode_document(\n    documents,\n    convert_to_tensor=True\n)\n\nquestion = \"How do I repair my car?\"\n\nquery_embedding = model.encode_query(\n    question,\n    convert_to_tensor=True\n)\n\nhits = util.semantic_search(\n    query_embedding,\n    document_embeddings,\n    top_k=2\n)[0]\n\nfor hit in hits:\n    print(round(float(hit[\"score\"]), 3), documents[hit[\"corpus_id\"]])",{"data":530,"type":217},{"text":531},"查询中不包含“车辆维护”这个短语，但语义模型仍然可以给该段落很高的排名，因为这些概念是相关的。这就是嵌入在 RAG 系统中常见的实际原因。",{"data":533,"type":217},{"text":534},"对于小型集合，嵌入可以保存在内存中。更大的系统通常会将它们持久化到支持向量的索引或数据库中，并在那里执行最近邻搜索。存储方式变了，但逻辑仍然不变：编码问题，找到相关的文档表示，返回最佳证据。",{"data":536,"type":42},{"text":537,"level":230},"真实示例 5：为 LLM 构建上下文",{"data":539,"type":217},{"text":540},"检索器应返回证据。然后，LLM 应接收问题以及该证据。将检索和生成分开，使两者都更易于检查和测试。",{"data":542,"type":252},{"code":543},"def build_prompt(question, retrieved_documents):\n    context = \"\\n\\n\".join(\n        f'[{doc[\"id\"]}] {doc[\"text\"]}'\n        for doc in retrieved_documents\n    )\n\n    return f\"\"\"\nAnswer the question using the supplied context.\n\nRules:\n- Do not invent facts that are not supported by the context.\n- If the context is insufficient, say so.\n- Cite the source IDs you used.\n\nQuestion:\n{question}\n\nContext:\n{context}\n\"\"\".strip()",{"data":545,"type":217},{"text":546},"该指令并不会使模型变得绝对可靠。它只是创建了一个明确的证据边界。模型仍然可能误解好的证据、忽略某个条件或过度泛化。这就是为什么检索质量和生成质量必须分别评估。",{"data":548,"type":42},{"text":549,"level":230},"真实示例 6：一个完整的最小流水线",{"data":551,"type":252},{"code":552},"def answer_question(question, all_documents, call_llm):\n    # 1. Retrieve evidence\n    retrieved = retrieve(question, all_documents, top_k=3)\n\n    # 2. Build model context\n    prompt = build_prompt(question, retrieved)\n\n    # 3. Generate the answer\n    answer = call_llm(prompt)\n\n    return {\n        \"answer\": answer,\n        \"sources\": [doc[\"id\"] for doc in retrieved]\n    }",{"data":554,"type":217},{"text":555},"该函数有意将 \u003Ccode>call_llm\u003C\u002Fcode> 作为依赖项接收。检索不应关心生成是由云模型、本地模型还是其他提供方执行的。数据路径属于应用程序。",{"data":557,"type":42},{"text":558,"level":477},"可选生成器：OpenAI Responses API",{"data":560,"type":217},{"text":561},"一种可能的生成器是 OpenAI Responses API。将模型名称保留在环境变量中，可以避免将特定模型硬编码到 RAG 架构中。",{"data":563,"type":252},{"code":564},"# pip install openai\n\nimport os\nfrom openai import OpenAI\n\nclient = OpenAI()\n\ndef call_llm(prompt):\n    response = client.responses.create(\n        model=os.environ[\"OPENAI_MODEL\"],\n        input=prompt,\n    )\n    return response.output_text",{"data":566,"type":217},{"text":567},"同一个检索流水线可以连接到本地推理服务器。这是一个重要的架构要点：\u003Cb>RAG 并不属于 LLM 提供方。\u003C\u002Fb>应用程序拥有来源、检索和上下文组装。",{"data":569,"type":42},{"text":570,"level":477},"整体架构一览",{"data":572,"type":252},{"code":573},"USER QUESTION\n     |\n     v\n+-------------+\n|  Retriever  |\n+-------------+\n   |       |\n   |       +----> SQL \u002F API \u002F state query\n   |\n   +------------> keyword \u002F full-text search\n   |\n   +------------> embedding \u002F vector search\n                     |\n                     v\n              relevant evidence\n                     |\n                     v\n+-----------------------------------+\n| question + evidence + instructions |\n+-----------------------------------+\n                     |\n                     v\n                   LLM\n                     |\n                     v\n                  answer",{"data":575,"type":217},{"text":576},"这种数据流模型比记住某一个框架更持久。库、数据库和模型供应商都会变化；但职责边界保持不变。",{"data":578,"type":42},{"text":579,"level":230},"常见误解与失败模式",{"data":581,"type":42},{"text":582,"level":477},"“RAG 就意味着向量数据库。”",{"data":584,"type":217},{"text":585},"不。向量搜索只是一种检索方法。RAG 可以使用全文搜索、SQL、API、知识图谱、向量搜索或它们的组合。其定义性模式是为生成而检索外部信息。",{"data":587,"type":42},{"text":588,"level":477},"“如果数据在 PostgreSQL 中，我就必须嵌入整个数据库。”",{"data":590,"type":217},{"text":591},"不。结构化记录通常应保持作为结构化记录可查询。嵌入对于语义相关性很有用，但不能替代确定性查询。",{"data":593,"type":42},{"text":594,"level":477},"“更多分块意味着更好的答案。”",{"data":596,"type":217},{"text":597},"未必如此。额外的上下文可能引入噪声、冲突版本和无关材料。检索应优化有用的证据，而非最大数量。",{"data":599,"type":42},{"text":600,"level":477},"“高相似度分数证明了答案。”",{"data":602,"type":217},{"text":603},"不。相似度衡量的是相关性，而非真实性或适用性。高度相似的段落可能已过时、来自错误的产品版本，或仅在不符合问题的条件下有效。",{"data":605,"type":42},{"text":606,"level":477},"“一旦检索到正确的分块，幻觉就解决了。”",{"data":608,"type":217},{"text":609},"不。检索改善了基础，但不保证忠实的推理。生成仍需评估，高风险工作流可能需要确定性验证或人工审查。",{"data":611,"type":42},{"text":612,"level":477},"“模型失败了，所以更换模型。”",{"data":614,"type":217},{"text":615},"未必如此。正确的来源可能缺失、解析错误、分割不当、被过滤掉、排名过低或未包含在组装的上下文中。更换模型不应是第一个诊断步骤。",{"data":617,"type":42},{"text":618,"level":230},"边缘情况",{"data":620,"type":357},{"items":621,"style":356},[622,623,624,625,626,627,628,629],"\u003Cb>冲突文档：\u003C\u002Fb>两个来源可能因版本、司法管辖区或产品不同而不一致。","\u003Cb>时间敏感事实：\u003C\u002Fb>语义相关的来源可能已经过时。","\u003Cb>权限：\u003C\u002Fb>检索器不得返回当前用户无权访问的文档。","\u003Cb>多语言集合：\u003C\u002Fb>嵌入模型和检索策略必须支持实际使用的语言。","\u003Cb>表格和源代码：\u003C\u002Fb>普通段落分块可能破坏对答案至关重要的结构。","\u003Cb>极短标识符：\u003C\u002Fb>对于SKU、ID、错误代码或缩写，语义检索可能弱于精确匹配。","\u003Cb>需要多个事实的长问题：\u003C\u002Fb>检索可能需要分解、多次搜索或重排序，而非一次top-k查询。","\u003Cb>来源层级：\u003C\u002Fb>官方当前政策可能需要优先于较旧但语义更接近的讨论文档。",{"data":631,"type":42},{"text":632,"level":230},"局限性",{"data":634,"type":217},{"text":635},"Python示例有意优化透明度，而非规模。关键词检索器是朴素的，分块器使用字符长度，SQLite示例不包含生产连接管理，语义示例将所有嵌入保存在内存中。",{"data":637,"type":217},{"text":638},"生产系统可能需要向量索引、重排序器、混合检索、文档解析器、缓存、增量索引、来源版本控制、访问控制过滤器、可观测性、评估数据集和故障处理。这些添加都不会改变核心架构；它们使每个边界更可靠。",{"data":640,"type":217},{"text":641},"RAG也无法创造来源集中缺失的证据。如果来源错误、不完整或过时，更好的嵌入模型也无法将其转化为权威知识。",{"data":643,"type":42},{"text":644,"level":230},"什么会改变这个答案？",{"data":646,"type":217},{"text":647},"当任务需要的不只是知识查找时，架构就会改变。实时订单状态需要当前状态。财务计算可能需要确定性代码。网络研究任务可能需要主动搜索。工作流可能需要能将数据写回另一个系统的工具。自主代理可能需要在检索之外进行规划、权限和执行控制。",{"data":649,"type":217},{"text":650},"因此，RAG最好被理解为\u003Cb>更大AI系统中的一个证据获取层\u003C\u002Fb>。它之所以强大，正是因为它有一个狭窄的职责：找到有用的外部信息并将其放入模型的工作上下文中。",{"data":652,"type":42},{"text":653,"level":230},"结论",{"data":655,"type":217},{"text":656},"当去掉技术名称后，RAG 就变得容易理解多了。一个文件是一个来源。一个数据库是一个来源。一个 API 是一个来源。搜索函数检索证据。提示词将证据传递给模型。然后 LLM 对其进行解释并生成语言。",{"data":658,"type":217},{"text":659},"生产级 RAG 的难点不在于调用嵌入模型。而在于从原始来源到最终主张构建一条可信的证据路径：保留来源信息、选择正确的检索方法、保持信息最新、控制访问权限、将检索评估与生成评估分开，以及知道何时直接调用数据库或工具比语义搜索更好。",{"data":661,"type":217},{"text":662},"这就是基础 RAG 模型的实用延续：\u003Cb>先理解角色，然后让数据路径变得明确。\u003C\u002Fb>",{"data":664,"type":42},{"text":665,"level":230},"主要来源",{"data":667,"type":357},{"items":668,"style":356},[669,670,671,672,673,674],"\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2005.11401\">Lewis 等人 — 面向知识密集型 NLP 任务的检索增强生成\u003C\u002Fa> — 2020 年的论文，介绍了将生成与检索到的非参数记忆相结合的 RAG 公式。","\u003Ca href=\"https:\u002F\u002Fwww.sbert.net\u002Fexamples\u002Fsentence_transformer\u002Fapplications\u002Fsemantic-search\u002FREADME.html\">Sentence Transformers — 语义搜索\u003C\u002Fa> — 关于语义检索、查询嵌入和文档嵌入的官方文档。","\u003Ca href=\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fembeddings\">OpenAI — 向量嵌入\u003C\u002Fa> — 官方文档，将嵌入描述为用于相关性和搜索的数值表示。","\u003Ca href=\"https:\u002F\u002Fwww.sqlite.org\u002Ffts5.html\">SQLite — FTS5 扩展\u003C\u002Fa> — 关于 SQLite 中全文搜索和 BM25 排名的官方文档。","\u003Ca href=\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Flibraries\">OpenAI — SDK 和 CLI\u003C\u002Fa> — 用于可选生成器示例中 Responses API 的官方 Python SDK 示例。","\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works\">什么是 RAG？对其工作原理的最简单解释\u003C\u002Fa> — 本系列的概念性第一部分。","2.31","LLM 并不会神奇地知道你的文件、数据库或 API。这个 RAG 系列的实用续篇用简单的 Python 展示了外部数据如何变成可检索的证据：从文本文件和 SQL 到全文搜索、嵌入、上下文组装以及最终的 LLM 调用。","\u002Fuploads\u002F2026\u002F09\u002Fwhere-does-an-llm-get-its-data-rag-data-sources-in-python-1790517200521-nfsi5i.webp","where-does-an-llm-get-its-data-rag-data-sources-in-python-1790517200521-nfsi5i","PUBLISHED","2026-09-27T09:51:00.000Z","2026-09-27T13:51:43.843Z","2026-09-27T13:58:03.762Z",{"en":684,"de":685,"sr":686,"es":687,"fr":688,"it":689,"ru":690,"zh":691},"\u002Fblog\u002Fwhere-does-an-llm-get-its-data-rag-data-sources-in-python","\u002Fde\u002Fblog\u002Fwhere-does-an-llm-get-its-data-rag-data-sources-in-python","\u002Fsr\u002Fblog\u002Fwhere-does-an-llm-get-its-data-rag-data-sources-in-python","\u002Fes\u002Fblog\u002Fwhere-does-an-llm-get-its-data-rag-data-sources-in-python","\u002Ffr\u002Fblog\u002Fwhere-does-an-llm-get-its-data-rag-data-sources-in-python","\u002Fit\u002Fblog\u002Fwhere-does-an-llm-get-its-data-rag-data-sources-in-python","\u002Fru\u002Fblog\u002Fwhere-does-an-llm-get-its-data-rag-data-sources-in-python","\u002Fzh\u002Fblog\u002Fwhere-does-an-llm-get-its-data-rag-data-sources-in-python",[],{"id":694,"login":695,"email":696,"displayName":697},"20","rooth8233","aleksandar@stajic.de","Aleksandar Stajić",[699,1145],{"lang":700,"title":701,"content":702,"contentJson":703,"excerpt":1144},"en","Where Does an LLM Get Its Data? RAG Data Sources in Python","{\"time\":1790516400000,\"blocks\":[{\"data\":{\"text\":\"The previous article, \u003Ca href=\\\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works\\\">What Is RAG? The Simplest Explanation of How It Works\u003C\u002Fa>, established the mental model: the LLM writes, RAG retrieves useful knowledge, the application owns current state, and tools perform actions. This article takes the next step: \u003Cb>where does the data actually come from, and what does retrieval look like in Python?\u003C\u002Fb>\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"The important surprise is that an “LLM data source” is usually nothing exotic. It can be a text file, a folder of Markdown documents, a SQL database, an API response, a product catalog, a support system, or a vector index derived from those sources. The AI does not magically know these systems. Your application has to load, query, search, or retrieve the relevant data and place the result into the model’s context.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Data source = where information lives. Retrieval = how the application finds useful information. Context = the selected information given to the model. LLM = the component that interprets that context and generates an answer.\",\"caption\":\"The four-part model used throughout this article\",\"alignment\":\"left\"},\"type\":\"quote\"},{\"data\":{\"text\":\"Question\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"How does an LLM use external data such as files, databases or APIs, and how can a small Python program implement the essential RAG steps without hiding them behind a framework?\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"What This Really Means\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"When developers say that an LLM is “connected to company data,” several different operations may be hidden behind that sentence. One application may execute SQL. Another may call an API. Another may run full-text search. Another may calculate embedding similarity over document chunks. All of them can provide external information to an LLM, but they are not the same retrieval method and they should not be treated as interchangeable.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"This distinction matters because the best retrieval method depends on the shape of the question. “What is our refund policy?” is a document-retrieval problem. “What is order 4711’s current status?” is usually a structured database lookup. “Which paragraph discusses account recovery?” can be keyword or semantic search. RAG is most useful when the system must \u003Cb>discover relevant knowledge before generation\u003C\u002Fb>.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Simplest Example\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"Start with three strings in ordinary Python. There is no vector database, no framework, and no LLM yet. We only want to make the retrieval step visible.\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"documents = [\\n    \\\"The AKM uses 7.62 mm ammunition.\\\",\\n    \\\"A Med Kit restores health.\\\",\\n    \\\"A 4x scope can be attached to several compatible weapons.\\\"\\n]\\n\\nquestion = \\\"Which ammunition does the AKM use?\\\"\\n\\nfor document in documents:\\n    if \\\"AKM\\\" in document:\\n        print(document)\"},\"type\":\"code\"},{\"data\":{\"text\":\"The program prints the first sentence because it contains the term we searched for. This is primitive retrieval, but the architecture is already visible: \u003Cb>question → search → relevant text\u003C\u002Fb>. RAG adds one more major step: pass the retrieved text to a language model together with the question.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"A slightly more general version ranks documents by overlapping query terms:\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"import re\\n\\ndocuments = [\\n    {\\\"id\\\": \\\"weapon-akm\\\", \\\"text\\\": \\\"The AKM uses 7.62 mm ammunition.\\\"},\\n    {\\\"id\\\": \\\"healing-medkit\\\", \\\"text\\\": \\\"A Med Kit restores health.\\\"},\\n    {\\\"id\\\": \\\"scope-4x\\\", \\\"text\\\": \\\"A 4x scope can be attached to several compatible weapons.\\\"},\\n]\\n\\ndef words(text):\\n    return set(re.findall(r\\\"[a-zA-Z0-9.]+\\\", text.lower()))\\n\\ndef retrieve(question, documents, top_k=2):\\n    query_terms = words(question)\\n    ranked = []\\n\\n    for document in documents:\\n        score = len(query_terms & words(document[\\\"text\\\"]))\\n        if score > 0:\\n            ranked.append((score, document))\\n\\n    ranked.sort(key=lambda item: item[0], reverse=True)\\n    return [document for _, document in ranked[:top_k]]\\n\\nquestion = \\\"Which ammunition does the AKM use?\\\"\\nhits = retrieve(question, documents)\\n\\nfor hit in hits:\\n    print(hit[\\\"id\\\"], \\\"->\\\", hit[\\\"text\\\"])\"},\"type\":\"code\"},{\"data\":{\"text\":\"This is not a production search engine. It ignores morphology, synonyms, spelling variants, document length and many ranking signals. Its value is educational: \u003Cb>RAG does not begin with a vector database. It begins with retrieval.\u003C\u002Fb>\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Where the Example Stops Working\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"Exact or lexical matching becomes weak when the question and the source use different words. A document may say “vehicle maintenance,” while the user asks “how do I repair my car?” A lexical retriever can miss the relationship even though a human sees it immediately. Semantic retrieval addresses this by representing text as vectors and comparing meaning rather than only exact tokens.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Long files create another problem. Searching an entire 80-page manual as one unit is too coarse, but splitting every sentence can destroy useful context. Real RAG systems therefore need decisions about parsing, chunking, metadata, ranking, freshness, permissions and provenance.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"The example also says nothing about structured live facts. If the user asks for the current status of order 4711 and the application already has a database key, semantic search is usually the wrong first tool. A deterministic database query is better.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Direct Answer\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"An LLM data source is any external system from which an application can obtain information for the model: files, databases, APIs, search indexes, vector stores or live application state. RAG is the pattern of \u003Cb>retrieving relevant knowledge from such sources before generation\u003C\u002Fb>.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"In Python, the essential pipeline can be very small: \u003Cb>load data → create retrievable units → find relevant evidence → assemble context → call the LLM\u003C\u002Fb>. The retrieval method should match the source and the question. Use SQL for exact structured facts, full-text search for lexical matching, embeddings for semantic similarity, and hybrid retrieval when several signals are valuable.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Why This Is So\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"A language model does not automatically receive the contents of your filesystem, PostgreSQL database, CRM, private API or newly edited document. The application decides what external information is accessible and what is placed into the model’s current context.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"The original Retrieval-Augmented Generation work by Lewis et al. combined a generative model with external non-parametric memory retrieved from a dense vector index. The broader architectural idea survives beyond that specific implementation: external evidence can be retrieved at inference time instead of expecting all useful knowledge to be encoded in model parameters.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"This creates a useful separation of responsibilities: the source stores information, the retriever selects evidence, the context carries that evidence into the request, and the model interprets it. Keeping those boundaries visible makes failures much easier to diagnose.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Context: The Main Types of Data Sources\",\"level\":2},\"type\":\"header\"},{\"data\":{\"content\":[[\"Source\",\"Typical retrieval method\",\"Good for\"],[\"TXT \u002F Markdown \u002F HTML\",\"Parsing + lexical or semantic search\",\"Documentation, manuals, articles, notes\"],[\"PDF \u002F DOCX\",\"Structure-aware extraction + search\",\"Policies, reports, contracts, manuals\"],[\"SQL database\",\"SQL query or filtered retrieval\",\"Orders, users, products, structured records\"],[\"REST \u002F GraphQL API\",\"HTTP request with parameters\",\"Remote systems and live service data\"],[\"Search index\",\"BM25 \u002F full-text \u002F hybrid search\",\"Large text collections\"],[\"Vector index\",\"Embedding similarity\",\"Semantic document retrieval\"],[\"Application state\",\"Direct state read or tool call\",\"What is true right now\"]],\"withHeadings\":true},\"type\":\"table\"},{\"data\":{\"text\":\"A vector index deserves special attention. In many architectures it is \u003Cb>not the canonical source of truth\u003C\u002Fb>. It is a retrieval index derived from documents or records. The authoritative document may live in object storage, a CMS, Git, PostgreSQL or another system, while embeddings and metadata are stored separately for fast semantic lookup. Some systems do use a vector store as primary storage, but that is an architectural choice rather than a requirement of RAG.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"If the boundary between retrieval, persistent memory, current state and model context is still unclear, see \u003Ca href=\\\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context\\\">AI Agent Memory Is Not RAG\u003C\u002Fa>. Those layers can use some of the same storage technologies while still having different correctness rules.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Assumptions\",\"level\":2},\"type\":\"header\"},{\"data\":{\"items\":[\"The application is allowed to access the external source.\",\"The relevant source contains enough information to answer the question.\",\"The data can be parsed or queried in a form the retrieval layer can use.\",\"The retrieved information is fresh enough for the requested decision.\",\"The model receives the selected evidence in its context.\",\"Authorization is enforced before protected evidence reaches the model.\",\"The generation model can still be wrong even when retrieval is correct.\"],\"style\":\"unordered\"},\"type\":\"list\"},{\"data\":{\"text\":\"These assumptions matter because retrieval cannot compensate for missing evidence, stale source versions, broken parsers or unauthorized access. A RAG pipeline can only be as trustworthy as the evidence path that feeds it.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Variables\",\"level\":2},\"type\":\"header\"},{\"data\":{\"content\":[[\"Variable\",\"Why it changes the design\"],[\"Source structure\",\"A SQL table, legal PDF and source-code repository need different retrieval strategies\"],[\"Question type\",\"Exact lookup, conceptual search and multi-hop research are different tasks\"],[\"Freshness requirement\",\"Live state may need direct queries instead of periodically rebuilt indexes\"],[\"Corpus size\",\"In-memory search may work for hundreds of chunks but not for very large collections\"],[\"Language\",\"Multilingual retrieval requires models and tokenization suitable for the actual languages\"],[\"Permissions\",\"Retrieval must filter by the current user’s access rights\"],[\"Latency and cost\",\"More retrieval stages can improve quality but add runtime and infrastructure cost\"],[\"Need for provenance\",\"High-trust systems need source IDs, versions and traceable evidence\"]],\"withHeadings\":true},\"type\":\"table\"},{\"data\":{\"text\":\"Diagnostic \u002F Decision Method\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"The first decision is not “Which vector database should I install?” It is: \u003Cb>What kind of fact am I trying to retrieve?\u003C\u002Fb>\"},\"type\":\"paragraph\"},{\"data\":{\"content\":[[\"Question type\",\"Preferred first approach\",\"Reason\"],[\"Exact ID or current record\",\"SQL \u002F key lookup \u002F API\",\"Deterministic structured access\"],[\"Exact wording, codes, names\",\"Full-text or keyword search\",\"Lexical precision\"],[\"Conceptual question over documents\",\"Semantic vector search\",\"Meaning can differ from wording\"],[\"Mixed enterprise knowledge\",\"Hybrid retrieval + metadata filters\",\"Combines lexical and semantic signals\"],[\"Current application state\",\"Direct state\u002Ftool access\",\"Freshness matters more than document similarity\"]],\"withHeadings\":true},\"type\":\"table\"},{\"data\":{\"text\":\"A useful test is: \u003Cb>Do I already know which record I need, or must the system discover which passage is relevant?\u003C\u002Fb> If the record is known, query it directly. If relevance must be discovered, search becomes more important.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"When an answer is wrong, diagnose the pipeline in order instead of immediately changing the LLM:\"},\"type\":\"paragraph\"},{\"data\":{\"items\":[\"\u003Cb>1. Source coverage:\u003C\u002Fb> Does the correct information exist in the accessible source set?\",\"\u003Cb>2. Freshness:\u003C\u002Fb> Is that version current enough for the question?\",\"\u003Cb>3. Parsing:\u003C\u002Fb> Was the relevant content extracted correctly?\",\"\u003Cb>4. Chunking:\u003C\u002Fb> Did the evidence stay together with the conditions that give it meaning?\",\"\u003Cb>5. Retrieval:\u003C\u002Fb> Does the correct chunk appear among the candidates?\",\"\u003Cb>6. Ranking:\u003C\u002Fb> Are stronger sources ranked above weaker or conflicting ones?\",\"\u003Cb>7. Context assembly:\u003C\u002Fb> Did the application actually send the selected evidence to the model?\",\"\u003Cb>8. Generation:\u003C\u002Fb> Did the LLM faithfully use the supplied evidence?\",\"\u003Cb>9. Attribution:\u003C\u002Fb> Can each important claim be traced to a source?\"],\"style\":\"ordered\"},\"type\":\"list\"},{\"data\":{\"text\":\"For a deeper production-debugging method, see \u003Ca href=\\\"https:\u002F\u002Fstajic.de\u002Fblog\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method\\\">RAG Failed — But Which Layer Actually Failed? A Diagnostic Method\u003C\u002Fa>, which expands this chain into independently testable failure layers.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Evidence\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"The RAG paper by Lewis et al. formalized generation that conditions on retrieved external memory rather than relying only on model parameters. That provides the conceptual foundation for separating the generator from a retrievable knowledge source.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Sentence Transformers documents semantic search as embedding the corpus and the query into a vector space and retrieving items with high semantic similarity. Its current API also distinguishes query encoding from document encoding for retrieval tasks.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"SQLite FTS5 demonstrates the other side of the spectrum: mature full-text retrieval can rank documents without embeddings. This matters because lexical search remains valuable for identifiers, exact terminology and many hybrid retrieval designs.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"OpenAI’s embeddings documentation describes embeddings as numerical vector representations used for relatedness and search. This is one implementation path for semantic retrieval, not the definition of RAG itself.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Real Example 1: A Folder of Text Files\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"Suppose a directory named \u003Ccode>knowledge\u002F\u003C\u002Fcode> contains ordinary text files. Python can load them with no AI library at all.\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"from pathlib import Path\\n\\ndef load_text_files(folder=\\\"knowledge\\\"):\\n    documents = []\\n\\n    for path in Path(folder).glob(\\\"*.txt\\\"):\\n        documents.append({\\n            \\\"source\\\": path.name,\\n            \\\"text\\\": path.read_text(encoding=\\\"utf-8\\\")\\n        })\\n\\n    return documents\\n\\ndocuments = load_text_files()\\n\\nfor document in documents:\\n    print(document[\\\"source\\\"], len(document[\\\"text\\\"]))\"},\"type\":\"code\"},{\"data\":{\"text\":\"The filesystem is the data source. The next question is how much text should become one retrievable unit. For long documents, searching one complete file is often too coarse. This is why RAG pipelines commonly create chunks.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"A Very Simple Chunker\",\"level\":3},\"type\":\"header\"},{\"data\":{\"code\":\"def chunk_text(text, max_chars=800):\\n    paragraphs = [p.strip() for p in text.split(\\\"\\\\n\\\\n\\\") if p.strip()]\\n\\n    chunks = []\\n    current = \\\"\\\"\\n\\n    for paragraph in paragraphs:\\n        candidate = f\\\"{current}\\\\n\\\\n{paragraph}\\\".strip()\\n\\n        if current and len(candidate) > max_chars:\\n            chunks.append(current)\\n            current = paragraph\\n        else:\\n            current = candidate\\n\\n    if current:\\n        chunks.append(current)\\n\\n    return chunks\"},\"type\":\"code\"},{\"data\":{\"text\":\"This example groups paragraphs until a rough character limit is reached. It is intentionally understandable rather than optimal. Production systems often chunk by tokens, headings, sections, sentence boundaries or document structure. Tables, source code, contracts and API documentation may need different strategies.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Preserve Provenance While Chunking\",\"level\":3},\"type\":\"header\"},{\"data\":{\"code\":\"def build_chunks(documents):\\n    chunks = []\\n\\n    for document in documents:\\n        for index, text in enumerate(chunk_text(document[\\\"text\\\"])):\\n            chunks.append({\\n                \\\"id\\\": f'{document[\\\"source\\\"]}:{index}',\\n                \\\"source\\\": document[\\\"source\\\"],\\n                \\\"chunk\\\": index,\\n                \\\"text\\\": text,\\n            })\\n\\n    return chunks\"},\"type\":\"code\"},{\"data\":{\"text\":\"A useful chunk carries more than text. Source name, document ID, URL, timestamp, version or section can later support citation, debugging and freshness checks. If provenance is lost during ingestion, it becomes much harder to explain why a particular answer was produced.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Real Example 2: Structured Data — Use SQL When SQL Is the Right Tool\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"Not every external fact should go through semantic search. If the question asks for an exact current record, a direct database query is usually clearer and more deterministic.\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"import sqlite3\\n\\ndef get_order_status(order_id):\\n    connection = sqlite3.connect(\\\"shop.db\\\")\\n    cursor = connection.cursor()\\n\\n    cursor.execute(\\n        \\\"SELECT status, total, currency FROM orders WHERE id = ?\\\",\\n        (order_id,)\\n    )\\n\\n    row = cursor.fetchone()\\n    connection.close()\\n\\n    if row is None:\\n        return None\\n\\n    return {\\n        \\\"order_id\\\": order_id,\\n        \\\"status\\\": row[0],\\n        \\\"total\\\": row[1],\\n        \\\"currency\\\": row[2],\\n    }\\n\\nprint(get_order_status(4711))\"},\"type\":\"code\"},{\"data\":{\"text\":\"If the application already knows that the user is asking about order 4711, embedding the entire orders table and asking semantic search to rediscover that row usually adds complexity without benefit. A strong design rule is: \u003Cb>retrieve structured facts with structured queries; retrieve unstructured knowledge with search.\u003C\u002Fb>\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"The returned database row can still be placed into the model context so the LLM can explain it in natural language. But direct state or record access is conceptually different from searching a knowledge corpus.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Real Example 3: Full-Text Search Before Embeddings\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"Between a naive Python loop and vector search lies a mature class of lexical retrieval systems. SQLite includes FTS5 for full-text search, including BM25 ranking.\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"import sqlite3\\n\\nconnection = sqlite3.connect(\\\"knowledge.db\\\")\\ncursor = connection.cursor()\\n\\ncursor.execute(\\n    \\\"CREATE VIRTUAL TABLE IF NOT EXISTS docs USING fts5(title, body)\\\"\\n)\\n\\ncursor.execute(\\n    \\\"INSERT INTO docs(title, body) VALUES (?, ?)\\\",\\n    (\\\"AKM\\\", \\\"The AKM uses 7.62 mm ammunition.\\\")\\n)\\n\\nconnection.commit()\\n\\nquery = \\\"AKM ammunition\\\"\\n\\nrows = cursor.execute(\\n    \\\"SELECT title, body, bm25(docs) AS score \\\"\\n    \\\"FROM docs WHERE docs MATCH ? \\\"\\n    \\\"ORDER BY score LIMIT 5\\\",\\n    (query,)\\n).fetchall()\\n\\nfor row in rows:\\n    print(row)\\n\\nconnection.close()\"},\"type\":\"code\"},{\"data\":{\"text\":\"Lexical search is especially useful when exact terminology, product codes, names, identifiers or domain-specific words matter. Semantic search is not automatically better. Production systems often combine both signals.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Real Example 4: Semantic Retrieval With Embeddings\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"Embeddings turn text into numerical vectors so semantically related passages can be compared even when they do not use identical wording. Sentence Transformers provides a straightforward local implementation.\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"# pip install sentence-transformers\\n\\nfrom sentence_transformers import SentenceTransformer, util\\n\\ndocuments = [\\n    \\\"The AKM uses 7.62 mm ammunition.\\\",\\n    \\\"A Med Kit restores health.\\\",\\n    \\\"Vehicle maintenance includes checking oil, brakes and tires.\\\",\\n    \\\"Account recovery requires access to the registered email address.\\\"\\n]\\n\\nmodel = SentenceTransformer(\\n    \\\"sentence-transformers\u002Fmulti-qa-mpnet-base-cos-v1\\\"\\n)\\n\\ndocument_embeddings = model.encode_document(\\n    documents,\\n    convert_to_tensor=True\\n)\\n\\nquestion = \\\"How do I repair my car?\\\"\\n\\nquery_embedding = model.encode_query(\\n    question,\\n    convert_to_tensor=True\\n)\\n\\nhits = util.semantic_search(\\n    query_embedding,\\n    document_embeddings,\\n    top_k=2\\n)[0]\\n\\nfor hit in hits:\\n    print(round(float(hit[\\\"score\\\"]), 3), documents[hit[\\\"corpus_id\\\"]])\"},\"type\":\"code\"},{\"data\":{\"text\":\"The query does not contain the phrase “vehicle maintenance,” but a semantic model can still rank that passage highly because the concepts are related. This is the practical reason embeddings are common in RAG systems.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"For small collections, embeddings can stay in memory. Larger systems usually persist them in a vector-capable index or database and perform nearest-neighbor search there. The storage changes, but the logic remains: encode the question, find relevant document representations, return the best evidence.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Real Example 5: Build the Context for the LLM\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"A retriever should return evidence. The LLM should then receive the question plus that evidence. Keeping retrieval and generation separate makes both easier to inspect and test.\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"def build_prompt(question, retrieved_documents):\\n    context = \\\"\\\\n\\\\n\\\".join(\\n        f'[{doc[\\\"id\\\"]}] {doc[\\\"text\\\"]}'\\n        for doc in retrieved_documents\\n    )\\n\\n    return f\\\"\\\"\\\"\\nAnswer the question using the supplied context.\\n\\nRules:\\n- Do not invent facts that are not supported by the context.\\n- If the context is insufficient, say so.\\n- Cite the source IDs you used.\\n\\nQuestion:\\n{question}\\n\\nContext:\\n{context}\\n\\\"\\\"\\\".strip()\"},\"type\":\"code\"},{\"data\":{\"text\":\"The instruction does not make the model infallible. It simply creates an explicit evidence boundary. The model can still misunderstand good evidence, ignore a condition or overgeneralize. That is why retrieval quality and generation quality must be evaluated separately.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Real Example 6: A Complete Minimal Pipeline\",\"level\":2},\"type\":\"header\"},{\"data\":{\"code\":\"def answer_question(question, all_documents, call_llm):\\n    # 1. Retrieve evidence\\n    retrieved = retrieve(question, all_documents, top_k=3)\\n\\n    # 2. Build model context\\n    prompt = build_prompt(question, retrieved)\\n\\n    # 3. Generate the answer\\n    answer = call_llm(prompt)\\n\\n    return {\\n        \\\"answer\\\": answer,\\n        \\\"sources\\\": [doc[\\\"id\\\"] for doc in retrieved]\\n    }\"},\"type\":\"code\"},{\"data\":{\"text\":\"The function receives \u003Ccode>call_llm\u003C\u002Fcode> as a dependency on purpose. Retrieval should not care whether generation is performed by a cloud model, a local model or another provider. The data path belongs to the application.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Optional Generator: OpenAI Responses API\",\"level\":3},\"type\":\"header\"},{\"data\":{\"text\":\"One possible generator is the OpenAI Responses API. Keeping the model name in an environment variable avoids hard-coding a particular model into the RAG architecture.\"},\"type\":\"paragraph\"},{\"data\":{\"code\":\"# pip install openai\\n\\nimport os\\nfrom openai import OpenAI\\n\\nclient = OpenAI()\\n\\ndef call_llm(prompt):\\n    response = client.responses.create(\\n        model=os.environ[\\\"OPENAI_MODEL\\\"],\\n        input=prompt,\\n    )\\n    return response.output_text\"},\"type\":\"code\"},{\"data\":{\"text\":\"The same retrieval pipeline can be connected to a local inference server. This is an important architectural point: \u003Cb>RAG is not owned by the LLM provider.\u003C\u002Fb> The application owns the source, retrieval and context assembly.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"The Whole Architecture in One View\",\"level\":3},\"type\":\"header\"},{\"data\":{\"code\":\"USER QUESTION\\n     |\\n     v\\n+-------------+\\n|  Retriever  |\\n+-------------+\\n   |       |\\n   |       +----> SQL \u002F API \u002F state query\\n   |\\n   +------------> keyword \u002F full-text search\\n   |\\n   +------------> embedding \u002F vector search\\n                     |\\n                     v\\n              relevant evidence\\n                     |\\n                     v\\n+-----------------------------------+\\n| question + evidence + instructions |\\n+-----------------------------------+\\n                     |\\n                     v\\n                   LLM\\n                     |\\n                     v\\n                  answer\"},\"type\":\"code\"},{\"data\":{\"text\":\"This data-flow model is more durable than memorizing one framework. Libraries, databases and model vendors will change; the responsibility boundaries remain.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Common Misconceptions and Failure Modes\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"“RAG means vector database.”\",\"level\":3},\"type\":\"header\"},{\"data\":{\"text\":\"No. Vector search is one retrieval method. RAG can use full-text search, SQL, APIs, knowledge graphs, vector search or combinations of them. The defining pattern is retrieval of external information for generation.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"“If the data is in PostgreSQL, I must embed the whole database.”\",\"level\":3},\"type\":\"header\"},{\"data\":{\"text\":\"No. Structured records should usually remain queryable as structured records. Embeddings are useful for semantic relevance, not as a replacement for deterministic queries.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"“More chunks means a better answer.”\",\"level\":3},\"type\":\"header\"},{\"data\":{\"text\":\"Not necessarily. Extra context can introduce noise, conflicting versions and irrelevant material. Retrieval should optimize for useful evidence, not maximum volume.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"“A high similarity score proves the answer.”\",\"level\":3},\"type\":\"header\"},{\"data\":{\"text\":\"No. Similarity measures relevance, not truth or applicability. A highly similar passage can be outdated, from the wrong product version or valid only under conditions that do not match the question.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"“Once the correct chunk is retrieved, hallucination is solved.”\",\"level\":3},\"type\":\"header\"},{\"data\":{\"text\":\"No. Retrieval improves grounding but does not guarantee faithful reasoning. Generation still needs evaluation, and high-risk workflows may require deterministic validation or human review.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"“The model failed, so change the model.”\",\"level\":3},\"type\":\"header\"},{\"data\":{\"text\":\"Not necessarily. The correct source may have been missing, parsed incorrectly, split badly, filtered out, ranked too low or omitted from the assembled context. Model replacement should not be the first diagnostic step.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Edge Cases\",\"level\":2},\"type\":\"header\"},{\"data\":{\"items\":[\"\u003Cb>Conflicting documents:\u003C\u002Fb> two sources may disagree because versions, jurisdictions or products differ.\",\"\u003Cb>Time-sensitive facts:\u003C\u002Fb> a semantically relevant source may already be stale.\",\"\u003Cb>Permissions:\u003C\u002Fb> a retriever must not return documents the current user is not authorized to access.\",\"\u003Cb>Multi-language collections:\u003C\u002Fb> the embedding model and retrieval strategy must support the languages actually used.\",\"\u003Cb>Tables and source code:\u003C\u002Fb> plain paragraph chunking can destroy structure that is essential to the answer.\",\"\u003Cb>Very short identifiers:\u003C\u002Fb> semantic retrieval can be weaker than exact matching for SKUs, IDs, error codes or acronyms.\",\"\u003Cb>Long questions requiring several facts:\u003C\u002Fb> retrieval may need decomposition, several searches or reranking rather than one top-k query.\",\"\u003Cb>Source hierarchy:\u003C\u002Fb> an official current policy may need to outrank an older but semantically closer discussion document.\"],\"style\":\"unordered\"},\"type\":\"list\"},{\"data\":{\"text\":\"Limitations\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"The Python examples intentionally optimize for transparency, not scale. The keyword retriever is naive, the chunker uses character length, the SQLite examples do not include production connection management, and the semantic example keeps all embeddings in memory.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"A production system may require vector indexes, rerankers, hybrid retrieval, document parsers, caching, incremental indexing, source versioning, access-control filters, observability, evaluation datasets and failure handling. None of those additions change the core architecture; they make each boundary more reliable.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"RAG also cannot create evidence that is absent from the source set. If the source is wrong, incomplete or stale, a better embedding model cannot turn it into authoritative knowledge.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"What Would Change This Answer?\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"The architecture changes when the task requires more than knowledge lookup. A live order status needs current state. A financial calculation may need deterministic code. A web-research task may need active search. A workflow may need tools that can write data back to another system. An autonomous agent may need planning, permissions and execution control in addition to retrieval.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"RAG is therefore best understood as \u003Cb>one evidence-acquisition layer inside a larger AI system\u003C\u002Fb>. It is powerful precisely because it has a narrow job: find useful external information and place it in the model’s working context.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Conclusion\",\"level\":2},\"type\":\"header\"},{\"data\":{\"text\":\"RAG becomes much easier to understand when the technology names are removed. A file is a source. A database is a source. An API is a source. A search function retrieves evidence. A prompt carries that evidence to the model. The LLM then interprets it and produces language.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"The hard part of production RAG is not calling an embedding model. It is building a trustworthy evidence path from the original source to the final claim: preserving provenance, selecting the right retrieval method, keeping information current, controlling access, evaluating retrieval separately from generation, and knowing when a direct database or tool call is better than semantic search.\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"That is the practical continuation of the basic RAG model: \u003Cb>first understand the roles, then make the data path explicit.\u003C\u002Fb>\"},\"type\":\"paragraph\"},{\"data\":{\"text\":\"Primary Sources\",\"level\":2},\"type\":\"header\"},{\"data\":{\"items\":[\"\u003Ca href=\\\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2005.11401\\\">Lewis et al. — Retrieval-Augmented Generation for Knowledge-Intensive NLP Tasks\u003C\u002Fa> — the 2020 paper introducing the RAG formulation that combines generation with retrieved non-parametric memory.\",\"\u003Ca href=\\\"https:\u002F\u002Fwww.sbert.net\u002Fexamples\u002Fsentence_transformer\u002Fapplications\u002Fsemantic-search\u002FREADME.html\\\">Sentence Transformers — Semantic Search\u003C\u002Fa> — official documentation for semantic retrieval, query embeddings and document embeddings.\",\"\u003Ca href=\\\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fembeddings\\\">OpenAI — Vector Embeddings\u003C\u002Fa> — official documentation describing embeddings as numerical representations used for relatedness and search.\",\"\u003Ca href=\\\"https:\u002F\u002Fwww.sqlite.org\u002Ffts5.html\\\">SQLite — FTS5 Extension\u003C\u002Fa> — official documentation for full-text search and BM25 ranking in SQLite.\",\"\u003Ca href=\\\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Flibraries\\\">OpenAI — SDKs and CLI\u003C\u002Fa> — official Python SDK example for the Responses API used in the optional generator example.\",\"\u003Ca href=\\\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works\\\">What Is RAG? The Simplest Explanation of How It Works\u003C\u002Fa> — the conceptual first part of this series.\"],\"style\":\"unordered\"},\"type\":\"list\"}],\"version\":\"2.31.0\"}",{"time":704,"blocks":705,"version":1143},1790516400000,[706,709,712,716,719,722,725,728,731,734,737,739,742,745,747,750,753,756,759,762,765,768,771,774,777,780,783,786,818,821,824,827,837,840,843,873,876,879,905,908,911,923,926,929,932,935,938,941,944,947,949,952,955,957,960,963,965,968,971,974,976,979,982,985,988,990,993,996,999,1001,1004,1007,1010,1013,1015,1018,1021,1023,1026,1029,1032,1034,1037,1040,1042,1045,1048,1051,1054,1057,1060,1063,1066,1069,1072,1075,1078,1081,1084,1087,1098,1101,1104,1107,1110,1113,1116,1119,1122,1125,1128,1131,1134],{"data":707,"type":217},{"text":708},"The previous article, \u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works\">What Is RAG? The Simplest Explanation of How It Works\u003C\u002Fa>, established the mental model: the LLM writes, RAG retrieves useful knowledge, the application owns current state, and tools perform actions. This article takes the next step: \u003Cb>where does the data actually come from, and what does retrieval look like in Python?\u003C\u002Fb>",{"data":710,"type":217},{"text":711},"The important surprise is that an “LLM data source” is usually nothing exotic. It can be a text file, a folder of Markdown documents, a SQL database, an API response, a product catalog, a support system, or a vector index derived from those sources. The AI does not magically know these systems. Your application has to load, query, search, or retrieve the relevant data and place the result into the model’s context.",{"data":713,"type":226},{"text":714,"caption":715,"alignment":225},"Data source = where information lives. Retrieval = how the application finds useful information. Context = the selected information given to the model. LLM = the component that interprets that context and generates an answer.","The four-part model used throughout this article",{"data":717,"type":42},{"text":718,"level":230},"Question",{"data":720,"type":217},{"text":721},"How does an LLM use external data such as files, databases or APIs, and how can a small Python program implement the essential RAG steps without hiding them behind a framework?",{"data":723,"type":42},{"text":724,"level":230},"What This Really Means",{"data":726,"type":217},{"text":727},"When developers say that an LLM is “connected to company data,” several different operations may be hidden behind that sentence. One application may execute SQL. Another may call an API. Another may run full-text search. Another may calculate embedding similarity over document chunks. All of them can provide external information to an LLM, but they are not the same retrieval method and they should not be treated as interchangeable.",{"data":729,"type":217},{"text":730},"This distinction matters because the best retrieval method depends on the shape of the question. “What is our refund policy?” is a document-retrieval problem. “What is order 4711’s current status?” is usually a structured database lookup. “Which paragraph discusses account recovery?” can be keyword or semantic search. RAG is most useful when the system must \u003Cb>discover relevant knowledge before generation\u003C\u002Fb>.",{"data":732,"type":42},{"text":733,"level":230},"Simplest Example",{"data":735,"type":217},{"text":736},"Start with three strings in ordinary Python. There is no vector database, no framework, and no LLM yet. We only want to make the retrieval step visible.",{"data":738,"type":252},{"code":251},{"data":740,"type":217},{"text":741},"The program prints the first sentence because it contains the term we searched for. This is primitive retrieval, but the architecture is already visible: \u003Cb>question → search → relevant text\u003C\u002Fb>. RAG adds one more major step: pass the retrieved text to a language model together with the question.",{"data":743,"type":217},{"text":744},"A slightly more general version ranks documents by overlapping query terms:",{"data":746,"type":252},{"code":261},{"data":748,"type":217},{"text":749},"This is not a production search engine. It ignores morphology, synonyms, spelling variants, document length and many ranking signals. Its value is educational: \u003Cb>RAG does not begin with a vector database. It begins with retrieval.\u003C\u002Fb>",{"data":751,"type":42},{"text":752,"level":230},"Where the Example Stops Working",{"data":754,"type":217},{"text":755},"Exact or lexical matching becomes weak when the question and the source use different words. A document may say “vehicle maintenance,” while the user asks “how do I repair my car?” A lexical retriever can miss the relationship even though a human sees it immediately. Semantic retrieval addresses this by representing text as vectors and comparing meaning rather than only exact tokens.",{"data":757,"type":217},{"text":758},"Long files create another problem. Searching an entire 80-page manual as one unit is too coarse, but splitting every sentence can destroy useful context. Real RAG systems therefore need decisions about parsing, chunking, metadata, ranking, freshness, permissions and provenance.",{"data":760,"type":217},{"text":761},"The example also says nothing about structured live facts. If the user asks for the current status of order 4711 and the application already has a database key, semantic search is usually the wrong first tool. A deterministic database query is better.",{"data":763,"type":42},{"text":764,"level":230},"Direct Answer",{"data":766,"type":217},{"text":767},"An LLM data source is any external system from which an application can obtain information for the model: files, databases, APIs, search indexes, vector stores or live application state. RAG is the pattern of \u003Cb>retrieving relevant knowledge from such sources before generation\u003C\u002Fb>.",{"data":769,"type":217},{"text":770},"In Python, the essential pipeline can be very small: \u003Cb>load data → create retrievable units → find relevant evidence → assemble context → call the LLM\u003C\u002Fb>. The retrieval method should match the source and the question. Use SQL for exact structured facts, full-text search for lexical matching, embeddings for semantic similarity, and hybrid retrieval when several signals are valuable.",{"data":772,"type":42},{"text":773,"level":230},"Why This Is So",{"data":775,"type":217},{"text":776},"A language model does not automatically receive the contents of your filesystem, PostgreSQL database, CRM, private API or newly edited document. The application decides what external information is accessible and what is placed into the model’s current context.",{"data":778,"type":217},{"text":779},"The original Retrieval-Augmented Generation work by Lewis et al. combined a generative model with external non-parametric memory retrieved from a dense vector index. The broader architectural idea survives beyond that specific implementation: external evidence can be retrieved at inference time instead of expecting all useful knowledge to be encoded in model parameters.",{"data":781,"type":217},{"text":782},"This creates a useful separation of responsibilities: the source stores information, the retriever selects evidence, the context carries that evidence into the request, and the model interprets it. Keeping those boundaries visible makes failures much easier to diagnose.",{"data":784,"type":42},{"text":785,"level":230},"Context: The Main Types of Data Sources",{"data":787,"type":336},{"content":788,"withHeadings":14},[789,793,796,799,803,806,810,814],[790,791,792],"Source","Typical retrieval method","Good for",[309,794,795],"Parsing + lexical or semantic search","Documentation, manuals, articles, notes",[313,797,798],"Structure-aware extraction + search","Policies, reports, contracts, manuals",[800,801,802],"SQL database","SQL query or filtered retrieval","Orders, users, products, structured records",[321,804,805],"HTTP request with parameters","Remote systems and live service data",[807,808,809],"Search index","BM25 \u002F full-text \u002F hybrid search","Large text collections",[811,812,813],"Vector index","Embedding similarity","Semantic document retrieval",[815,816,817],"Application state","Direct state read or tool call","What is true right now",{"data":819,"type":217},{"text":820},"A vector index deserves special attention. In many architectures it is \u003Cb>not the canonical source of truth\u003C\u002Fb>. It is a retrieval index derived from documents or records. The authoritative document may live in object storage, a CMS, Git, PostgreSQL or another system, while embeddings and metadata are stored separately for fast semantic lookup. Some systems do use a vector store as primary storage, but that is an architectural choice rather than a requirement of RAG.",{"data":822,"type":217},{"text":823},"If the boundary between retrieval, persistent memory, current state and model context is still unclear, see \u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context\">AI Agent Memory Is Not RAG\u003C\u002Fa>. Those layers can use some of the same storage technologies while still having different correctness rules.",{"data":825,"type":42},{"text":826,"level":230},"Assumptions",{"data":828,"type":357},{"items":829,"style":356},[830,831,832,833,834,835,836],"The application is allowed to access the external source.","The relevant source contains enough information to answer the question.","The data can be parsed or queried in a form the retrieval layer can use.","The retrieved information is fresh enough for the requested decision.","The model receives the selected evidence in its context.","Authorization is enforced before protected evidence reaches the model.","The generation model can still be wrong even when retrieval is correct.",{"data":838,"type":217},{"text":839},"These assumptions matter because retrieval cannot compensate for missing evidence, stale source versions, broken parsers or unauthorized access. A RAG pipeline can only be as trustworthy as the evidence path that feeds it.",{"data":841,"type":42},{"text":842,"level":230},"Variables",{"data":844,"type":336},{"content":845,"withHeadings":14},[846,849,852,855,858,861,864,867,870],[847,848],"Variable","Why it changes the design",[850,851],"Source structure","A SQL table, legal PDF and source-code repository need different retrieval strategies",[853,854],"Question type","Exact lookup, conceptual search and multi-hop research are different tasks",[856,857],"Freshness requirement","Live state may need direct queries instead of periodically rebuilt indexes",[859,860],"Corpus size","In-memory search may work for hundreds of chunks but not for very large collections",[862,863],"Language","Multilingual retrieval requires models and tokenization suitable for the actual languages",[865,866],"Permissions","Retrieval must filter by the current user’s access rights",[868,869],"Latency and cost","More retrieval stages can improve quality but add runtime and infrastructure cost",[871,872],"Need for provenance","High-trust systems need source IDs, versions and traceable evidence",{"data":874,"type":42},{"text":875,"level":230},"Diagnostic \u002F Decision Method",{"data":877,"type":217},{"text":878},"The first decision is not “Which vector database should I install?” It is: \u003Cb>What kind of fact am I trying to retrieve?\u003C\u002Fb>",{"data":880,"type":336},{"content":881,"withHeadings":14},[882,885,889,893,897,901],[853,883,884],"Preferred first approach","Reason",[886,887,888],"Exact ID or current record","SQL \u002F key lookup \u002F API","Deterministic structured access",[890,891,892],"Exact wording, codes, names","Full-text or keyword search","Lexical precision",[894,895,896],"Conceptual question over documents","Semantic vector search","Meaning can differ from wording",[898,899,900],"Mixed enterprise knowledge","Hybrid retrieval + metadata filters","Combines lexical and semantic signals",[902,903,904],"Current application state","Direct state\u002Ftool access","Freshness matters more than document similarity",{"data":906,"type":217},{"text":907},"A useful test is: \u003Cb>Do I already know which record I need, or must the system discover which passage is relevant?\u003C\u002Fb> If the record is known, query it directly. If relevance must be discovered, search becomes more important.",{"data":909,"type":217},{"text":910},"When an answer is wrong, diagnose the pipeline in order instead of immediately changing the LLM:",{"data":912,"type":357},{"items":913,"style":443},[914,915,916,917,918,919,920,921,922],"\u003Cb>1. Source coverage:\u003C\u002Fb> Does the correct information exist in the accessible source set?","\u003Cb>2. Freshness:\u003C\u002Fb> Is that version current enough for the question?","\u003Cb>3. Parsing:\u003C\u002Fb> Was the relevant content extracted correctly?","\u003Cb>4. Chunking:\u003C\u002Fb> Did the evidence stay together with the conditions that give it meaning?","\u003Cb>5. Retrieval:\u003C\u002Fb> Does the correct chunk appear among the candidates?","\u003Cb>6. Ranking:\u003C\u002Fb> Are stronger sources ranked above weaker or conflicting ones?","\u003Cb>7. Context assembly:\u003C\u002Fb> Did the application actually send the selected evidence to the model?","\u003Cb>8. Generation:\u003C\u002Fb> Did the LLM faithfully use the supplied evidence?","\u003Cb>9. Attribution:\u003C\u002Fb> Can each important claim be traced to a source?",{"data":924,"type":217},{"text":925},"For a deeper production-debugging method, see \u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fblog\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method\">RAG Failed — But Which Layer Actually Failed? A Diagnostic Method\u003C\u002Fa>, which expands this chain into independently testable failure layers.",{"data":927,"type":42},{"text":928,"level":230},"Evidence",{"data":930,"type":217},{"text":931},"The RAG paper by Lewis et al. formalized generation that conditions on retrieved external memory rather than relying only on model parameters. That provides the conceptual foundation for separating the generator from a retrievable knowledge source.",{"data":933,"type":217},{"text":934},"Sentence Transformers documents semantic search as embedding the corpus and the query into a vector space and retrieving items with high semantic similarity. Its current API also distinguishes query encoding from document encoding for retrieval tasks.",{"data":936,"type":217},{"text":937},"SQLite FTS5 demonstrates the other side of the spectrum: mature full-text retrieval can rank documents without embeddings. This matters because lexical search remains valuable for identifiers, exact terminology and many hybrid retrieval designs.",{"data":939,"type":217},{"text":940},"OpenAI’s embeddings documentation describes embeddings as numerical vector representations used for relatedness and search. This is one implementation path for semantic retrieval, not the definition of RAG itself.",{"data":942,"type":42},{"text":943,"level":230},"Real Example 1: A Folder of Text Files",{"data":945,"type":217},{"text":946},"Suppose a directory named \u003Ccode>knowledge\u002F\u003C\u002Fcode> contains ordinary text files. Python can load them with no AI library at all.",{"data":948,"type":252},{"code":470},{"data":950,"type":217},{"text":951},"The filesystem is the data source. The next question is how much text should become one retrievable unit. For long documents, searching one complete file is often too coarse. This is why RAG pipelines commonly create chunks.",{"data":953,"type":42},{"text":954,"level":477},"A Very Simple Chunker",{"data":956,"type":252},{"code":480},{"data":958,"type":217},{"text":959},"This example groups paragraphs until a rough character limit is reached. It is intentionally understandable rather than optimal. Production systems often chunk by tokens, headings, sections, sentence boundaries or document structure. Tables, source code, contracts and API documentation may need different strategies.",{"data":961,"type":42},{"text":962,"level":477},"Preserve Provenance While Chunking",{"data":964,"type":252},{"code":489},{"data":966,"type":217},{"text":967},"A useful chunk carries more than text. Source name, document ID, URL, timestamp, version or section can later support citation, debugging and freshness checks. If provenance is lost during ingestion, it becomes much harder to explain why a particular answer was produced.",{"data":969,"type":42},{"text":970,"level":230},"Real Example 2: Structured Data — Use SQL When SQL Is the Right Tool",{"data":972,"type":217},{"text":973},"Not every external fact should go through semantic search. If the question asks for an exact current record, a direct database query is usually clearer and more deterministic.",{"data":975,"type":252},{"code":501},{"data":977,"type":217},{"text":978},"If the application already knows that the user is asking about order 4711, embedding the entire orders table and asking semantic search to rediscover that row usually adds complexity without benefit. A strong design rule is: \u003Cb>retrieve structured facts with structured queries; retrieve unstructured knowledge with search.\u003C\u002Fb>",{"data":980,"type":217},{"text":981},"The returned database row can still be placed into the model context so the LLM can explain it in natural language. But direct state or record access is conceptually different from searching a knowledge corpus.",{"data":983,"type":42},{"text":984,"level":230},"Real Example 3: Full-Text Search Before Embeddings",{"data":986,"type":217},{"text":987},"Between a naive Python loop and vector search lies a mature class of lexical retrieval systems. SQLite includes FTS5 for full-text search, including BM25 ranking.",{"data":989,"type":252},{"code":516},{"data":991,"type":217},{"text":992},"Lexical search is especially useful when exact terminology, product codes, names, identifiers or domain-specific words matter. Semantic search is not automatically better. Production systems often combine both signals.",{"data":994,"type":42},{"text":995,"level":230},"Real Example 4: Semantic Retrieval With Embeddings",{"data":997,"type":217},{"text":998},"Embeddings turn text into numerical vectors so semantically related passages can be compared even when they do not use identical wording. Sentence Transformers provides a straightforward local implementation.",{"data":1000,"type":252},{"code":528},{"data":1002,"type":217},{"text":1003},"The query does not contain the phrase “vehicle maintenance,” but a semantic model can still rank that passage highly because the concepts are related. This is the practical reason embeddings are common in RAG systems.",{"data":1005,"type":217},{"text":1006},"For small collections, embeddings can stay in memory. Larger systems usually persist them in a vector-capable index or database and perform nearest-neighbor search there. The storage changes, but the logic remains: encode the question, find relevant document representations, return the best evidence.",{"data":1008,"type":42},{"text":1009,"level":230},"Real Example 5: Build the Context for the LLM",{"data":1011,"type":217},{"text":1012},"A retriever should return evidence. The LLM should then receive the question plus that evidence. Keeping retrieval and generation separate makes both easier to inspect and test.",{"data":1014,"type":252},{"code":543},{"data":1016,"type":217},{"text":1017},"The instruction does not make the model infallible. It simply creates an explicit evidence boundary. The model can still misunderstand good evidence, ignore a condition or overgeneralize. That is why retrieval quality and generation quality must be evaluated separately.",{"data":1019,"type":42},{"text":1020,"level":230},"Real Example 6: A Complete Minimal Pipeline",{"data":1022,"type":252},{"code":552},{"data":1024,"type":217},{"text":1025},"The function receives \u003Ccode>call_llm\u003C\u002Fcode> as a dependency on purpose. Retrieval should not care whether generation is performed by a cloud model, a local model or another provider. The data path belongs to the application.",{"data":1027,"type":42},{"text":1028,"level":477},"Optional Generator: OpenAI Responses API",{"data":1030,"type":217},{"text":1031},"One possible generator is the OpenAI Responses API. Keeping the model name in an environment variable avoids hard-coding a particular model into the RAG architecture.",{"data":1033,"type":252},{"code":564},{"data":1035,"type":217},{"text":1036},"The same retrieval pipeline can be connected to a local inference server. This is an important architectural point: \u003Cb>RAG is not owned by the LLM provider.\u003C\u002Fb> The application owns the source, retrieval and context assembly.",{"data":1038,"type":42},{"text":1039,"level":477},"The Whole Architecture in One View",{"data":1041,"type":252},{"code":573},{"data":1043,"type":217},{"text":1044},"This data-flow model is more durable than memorizing one framework. Libraries, databases and model vendors will change; the responsibility boundaries remain.",{"data":1046,"type":42},{"text":1047,"level":230},"Common Misconceptions and Failure Modes",{"data":1049,"type":42},{"text":1050,"level":477},"“RAG means vector database.”",{"data":1052,"type":217},{"text":1053},"No. Vector search is one retrieval method. RAG can use full-text search, SQL, APIs, knowledge graphs, vector search or combinations of them. The defining pattern is retrieval of external information for generation.",{"data":1055,"type":42},{"text":1056,"level":477},"“If the data is in PostgreSQL, I must embed the whole database.”",{"data":1058,"type":217},{"text":1059},"No. Structured records should usually remain queryable as structured records. Embeddings are useful for semantic relevance, not as a replacement for deterministic queries.",{"data":1061,"type":42},{"text":1062,"level":477},"“More chunks means a better answer.”",{"data":1064,"type":217},{"text":1065},"Not necessarily. Extra context can introduce noise, conflicting versions and irrelevant material. Retrieval should optimize for useful evidence, not maximum volume.",{"data":1067,"type":42},{"text":1068,"level":477},"“A high similarity score proves the answer.”",{"data":1070,"type":217},{"text":1071},"No. Similarity measures relevance, not truth or applicability. A highly similar passage can be outdated, from the wrong product version or valid only under conditions that do not match the question.",{"data":1073,"type":42},{"text":1074,"level":477},"“Once the correct chunk is retrieved, hallucination is solved.”",{"data":1076,"type":217},{"text":1077},"No. Retrieval improves grounding but does not guarantee faithful reasoning. Generation still needs evaluation, and high-risk workflows may require deterministic validation or human review.",{"data":1079,"type":42},{"text":1080,"level":477},"“The model failed, so change the model.”",{"data":1082,"type":217},{"text":1083},"Not necessarily. The correct source may have been missing, parsed incorrectly, split badly, filtered out, ranked too low or omitted from the assembled context. Model replacement should not be the first diagnostic step.",{"data":1085,"type":42},{"text":1086,"level":230},"Edge Cases",{"data":1088,"type":357},{"items":1089,"style":356},[1090,1091,1092,1093,1094,1095,1096,1097],"\u003Cb>Conflicting documents:\u003C\u002Fb> two sources may disagree because versions, jurisdictions or products differ.","\u003Cb>Time-sensitive facts:\u003C\u002Fb> a semantically relevant source may already be stale.","\u003Cb>Permissions:\u003C\u002Fb> a retriever must not return documents the current user is not authorized to access.","\u003Cb>Multi-language collections:\u003C\u002Fb> the embedding model and retrieval strategy must support the languages actually used.","\u003Cb>Tables and source code:\u003C\u002Fb> plain paragraph chunking can destroy structure that is essential to the answer.","\u003Cb>Very short identifiers:\u003C\u002Fb> semantic retrieval can be weaker than exact matching for SKUs, IDs, error codes or acronyms.","\u003Cb>Long questions requiring several facts:\u003C\u002Fb> retrieval may need decomposition, several searches or reranking rather than one top-k query.","\u003Cb>Source hierarchy:\u003C\u002Fb> an official current policy may need to outrank an older but semantically closer discussion document.",{"data":1099,"type":42},{"text":1100,"level":230},"Limitations",{"data":1102,"type":217},{"text":1103},"The Python examples intentionally optimize for transparency, not scale. The keyword retriever is naive, the chunker uses character length, the SQLite examples do not include production connection management, and the semantic example keeps all embeddings in memory.",{"data":1105,"type":217},{"text":1106},"A production system may require vector indexes, rerankers, hybrid retrieval, document parsers, caching, incremental indexing, source versioning, access-control filters, observability, evaluation datasets and failure handling. None of those additions change the core architecture; they make each boundary more reliable.",{"data":1108,"type":217},{"text":1109},"RAG also cannot create evidence that is absent from the source set. If the source is wrong, incomplete or stale, a better embedding model cannot turn it into authoritative knowledge.",{"data":1111,"type":42},{"text":1112,"level":230},"What Would Change This Answer?",{"data":1114,"type":217},{"text":1115},"The architecture changes when the task requires more than knowledge lookup. A live order status needs current state. A financial calculation may need deterministic code. A web-research task may need active search. A workflow may need tools that can write data back to another system. An autonomous agent may need planning, permissions and execution control in addition to retrieval.",{"data":1117,"type":217},{"text":1118},"RAG is therefore best understood as \u003Cb>one evidence-acquisition layer inside a larger AI system\u003C\u002Fb>. It is powerful precisely because it has a narrow job: find useful external information and place it in the model’s working context.",{"data":1120,"type":42},{"text":1121,"level":230},"Conclusion",{"data":1123,"type":217},{"text":1124},"RAG becomes much easier to understand when the technology names are removed. A file is a source. A database is a source. An API is a source. A search function retrieves evidence. A prompt carries that evidence to the model. The LLM then interprets it and produces language.",{"data":1126,"type":217},{"text":1127},"The hard part of production RAG is not calling an embedding model. It is building a trustworthy evidence path from the original source to the final claim: preserving provenance, selecting the right retrieval method, keeping information current, controlling access, evaluating retrieval separately from generation, and knowing when a direct database or tool call is better than semantic search.",{"data":1129,"type":217},{"text":1130},"That is the practical continuation of the basic RAG model: \u003Cb>first understand the roles, then make the data path explicit.\u003C\u002Fb>",{"data":1132,"type":42},{"text":1133,"level":230},"Primary Sources",{"data":1135,"type":357},{"items":1136,"style":356},[1137,1138,1139,1140,1141,1142],"\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2005.11401\">Lewis et al. — Retrieval-Augmented Generation for Knowledge-Intensive NLP Tasks\u003C\u002Fa> — the 2020 paper introducing the RAG formulation that combines generation with retrieved non-parametric memory.","\u003Ca href=\"https:\u002F\u002Fwww.sbert.net\u002Fexamples\u002Fsentence_transformer\u002Fapplications\u002Fsemantic-search\u002FREADME.html\">Sentence Transformers — Semantic Search\u003C\u002Fa> — official documentation for semantic retrieval, query embeddings and document embeddings.","\u003Ca href=\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fembeddings\">OpenAI — Vector Embeddings\u003C\u002Fa> — official documentation describing embeddings as numerical representations used for relatedness and search.","\u003Ca href=\"https:\u002F\u002Fwww.sqlite.org\u002Ffts5.html\">SQLite — FTS5 Extension\u003C\u002Fa> — official documentation for full-text search and BM25 ranking in SQLite.","\u003Ca href=\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Flibraries\">OpenAI — SDKs and CLI\u003C\u002Fa> — official Python SDK example for the Responses API used in the optional generator example.","\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works\">What Is RAG? The Simplest Explanation of How It Works\u003C\u002Fa> — the conceptual first part of this series.","2.31.0","An LLM does not magically know your files, databases or APIs. This practical continuation of the RAG series shows, with simple Python, how external data becomes retrievable evidence: from text files and SQL to full-text search, embeddings, context assembly and the final LLM call.",{"lang":7,"title":208,"content":210,"contentJson":1146,"excerpt":676},{"time":212,"blocks":1147,"version":675},[1148,1150,1152,1154,1156,1158,1160,1162,1164,1166,1168,1170,1172,1174,1176,1178,1180,1182,1184,1186,1188,1190,1192,1194,1196,1198,1200,1202,1213,1215,1217,1219,1222,1224,1226,1238,1240,1242,1251,1253,1255,1258,1260,1262,1264,1266,1268,1270,1272,1274,1276,1278,1280,1282,1284,1286,1288,1290,1292,1294,1296,1298,1300,1302,1304,1306,1308,1310,1312,1314,1316,1318,1320,1322,1324,1326,1328,1330,1332,1334,1336,1338,1340,1342,1344,1346,1348,1350,1352,1354,1356,1358,1360,1362,1364,1366,1368,1370,1372,1374,1377,1379,1381,1383,1385,1387,1389,1391,1393,1395,1397,1399,1401],{"data":1149,"type":217},{"text":216},{"data":1151,"type":217},{"text":220},{"data":1153,"type":226},{"text":223,"caption":224,"alignment":225},{"data":1155,"type":42},{"text":229,"level":230},{"data":1157,"type":217},{"text":233},{"data":1159,"type":42},{"text":236,"level":230},{"data":1161,"type":217},{"text":239},{"data":1163,"type":217},{"text":242},{"data":1165,"type":42},{"text":245,"level":230},{"data":1167,"type":217},{"text":248},{"data":1169,"type":252},{"code":251},{"data":1171,"type":217},{"text":255},{"data":1173,"type":217},{"text":258},{"data":1175,"type":252},{"code":261},{"data":1177,"type":217},{"text":264},{"data":1179,"type":42},{"text":267,"level":230},{"data":1181,"type":217},{"text":270},{"data":1183,"type":217},{"text":273},{"data":1185,"type":217},{"text":276},{"data":1187,"type":42},{"text":279,"level":230},{"data":1189,"type":217},{"text":282},{"data":1191,"type":217},{"text":285},{"data":1193,"type":42},{"text":288,"level":230},{"data":1195,"type":217},{"text":291},{"data":1197,"type":217},{"text":294},{"data":1199,"type":217},{"text":297},{"data":1201,"type":42},{"text":300,"level":230},{"data":1203,"type":336},{"content":1204,"withHeadings":14},[1205,1206,1207,1208,1209,1210,1211,1212],[305,306,307],[309,310,311],[313,314,315],[317,318,319],[321,322,323],[325,326,327],[329,330,331],[333,334,335],{"data":1214,"type":217},{"text":339},{"data":1216,"type":217},{"text":342},{"data":1218,"type":42},{"text":345,"level":230},{"data":1220,"type":357},{"items":1221,"style":356},[349,350,351,352,353,354,355],{"data":1223,"type":217},{"text":360},{"data":1225,"type":42},{"text":363,"level":230},{"data":1227,"type":336},{"content":1228,"withHeadings":14},[1229,1230,1231,1232,1233,1234,1235,1236,1237],[363,368],[370,371],[373,374],[376,377],[379,380],[382,383],[385,386],[388,389],[391,392],{"data":1239,"type":42},{"text":395,"level":230},{"data":1241,"type":217},{"text":398},{"data":1243,"type":336},{"content":1244,"withHeadings":14},[1245,1246,1247,1248,1249,1250],[373,403,404],[406,407,408],[410,411,412],[414,415,416],[418,419,420],[422,423,424],{"data":1252,"type":217},{"text":427},{"data":1254,"type":217},{"text":430},{"data":1256,"type":357},{"items":1257,"style":443},[434,435,436,437,438,439,440,441,442],{"data":1259,"type":217},{"text":446},{"data":1261,"type":42},{"text":449,"level":230},{"data":1263,"type":217},{"text":452},{"data":1265,"type":217},{"text":455},{"data":1267,"type":217},{"text":458},{"data":1269,"type":217},{"text":461},{"data":1271,"type":42},{"text":464,"level":230},{"data":1273,"type":217},{"text":467},{"data":1275,"type":252},{"code":470},{"data":1277,"type":217},{"text":473},{"data":1279,"type":42},{"text":476,"level":477},{"data":1281,"type":252},{"code":480},{"data":1283,"type":217},{"text":483},{"data":1285,"type":42},{"text":486,"level":477},{"data":1287,"type":252},{"code":489},{"data":1289,"type":217},{"text":492},{"data":1291,"type":42},{"text":495,"level":230},{"data":1293,"type":217},{"text":498},{"data":1295,"type":252},{"code":501},{"data":1297,"type":217},{"text":504},{"data":1299,"type":217},{"text":507},{"data":1301,"type":42},{"text":510,"level":230},{"data":1303,"type":217},{"text":513},{"data":1305,"type":252},{"code":516},{"data":1307,"type":217},{"text":519},{"data":1309,"type":42},{"text":522,"level":230},{"data":1311,"type":217},{"text":525},{"data":1313,"type":252},{"code":528},{"data":1315,"type":217},{"text":531},{"data":1317,"type":217},{"text":534},{"data":1319,"type":42},{"text":537,"level":230},{"data":1321,"type":217},{"text":540},{"data":1323,"type":252},{"code":543},{"data":1325,"type":217},{"text":546},{"data":1327,"type":42},{"text":549,"level":230},{"data":1329,"type":252},{"code":552},{"data":1331,"type":217},{"text":555},{"data":1333,"type":42},{"text":558,"level":477},{"data":1335,"type":217},{"text":561},{"data":1337,"type":252},{"code":564},{"data":1339,"type":217},{"text":567},{"data":1341,"type":42},{"text":570,"level":477},{"data":1343,"type":252},{"code":573},{"data":1345,"type":217},{"text":576},{"data":1347,"type":42},{"text":579,"level":230},{"data":1349,"type":42},{"text":582,"level":477},{"data":1351,"type":217},{"text":585},{"data":1353,"type":42},{"text":588,"level":477},{"data":1355,"type":217},{"text":591},{"data":1357,"type":42},{"text":594,"level":477},{"data":1359,"type":217},{"text":597},{"data":1361,"type":42},{"text":600,"level":477},{"data":1363,"type":217},{"text":603},{"data":1365,"type":42},{"text":606,"level":477},{"data":1367,"type":217},{"text":609},{"data":1369,"type":42},{"text":612,"level":477},{"data":1371,"type":217},{"text":615},{"data":1373,"type":42},{"text":618,"level":230},{"data":1375,"type":357},{"items":1376,"style":356},[622,623,624,625,626,627,628,629],{"data":1378,"type":42},{"text":632,"level":230},{"data":1380,"type":217},{"text":635},{"data":1382,"type":217},{"text":638},{"data":1384,"type":217},{"text":641},{"data":1386,"type":42},{"text":644,"level":230},{"data":1388,"type":217},{"text":647},{"data":1390,"type":217},{"text":650},{"data":1392,"type":42},{"text":653,"level":230},{"data":1394,"type":217},{"text":656},{"data":1396,"type":217},{"text":659},{"data":1398,"type":217},{"text":662},{"data":1400,"type":42},{"text":665,"level":230},{"data":1402,"type":357},{"items":1403,"style":356},[669,670,671,672,673,674],"Post erfolgreich abgerufen",{"items":1406,"source":1491,"manualIds":1492,"manualMatchedIds":1493},[1407,1414,1421,1428,1435,1442,1449,1456,1463,1470,1477,1484],{"id":1408,"slug":1409,"title":1410,"excerpt":1411,"featuredImage":1412,"publishedAt":1413},"462","the-prompt-is-part-of-the-bias-how-ai-framing-shapes-reasoning","提示词本身就是偏见的一部分：AI的框架设定如何塑造推理","提示词的措辞并非中立。探索框架设定、预设假设、指令遵循和谄媚如何塑造AI推理——以及为什么可靠的结论需要在原始提示之外进行测试。","\u002Fuploads\u002F2026\u002F09\u002Fthe-prompt-is-part-of-the-bias-how-ai-framing-shapes-reasoning-1789804884054-u278vc.webp","2026-09-19T01:04:00.000Z",{"id":1415,"slug":1416,"title":1417,"excerpt":1418,"featuredImage":1419,"publishedAt":1420},"452","zbt-z8102ax-5g-openwrt-router-review-dual-sim-rm500u-ea-and-an-honest-assessment","ZBT Z8102AX 5G OpenWrt路由器评测：双SIM卡、RM500U-EA及真实评估","ZBT Z8102AX是一款独特的5G路由器，采用OpenWrt基础系统、双SIM卡设计以及Quectel RM500U-EA调制解调器。在测试中，它在灵活性、接口和移动连接方面展现出明显优势，但也暴露出厂商定制版OpenWrt固件的典型缺陷。","\u002Fuploads\u002F2026\u002F06\u002Fopenwrt-router-review-dual-sim-01-1781620588908-fwmzj7.webp","2026-06-16T07:34:00.000Z",{"id":1422,"slug":1423,"title":1424,"excerpt":1425,"featuredImage":1426,"publishedAt":1427},"380","streamlining-code-quality-testing-with-eslint-and-prettier","优化代码质量：使用ESLint与Prettier进行测试","在现代软件开发中，保持一致的代码质量和风格至关重要。ESLint与Prettier提供了强大的组合方案，能够自动化这些关键环节，确保代码库的整洁性、可读性，并遵循既定标准。本文将深入探讨这两款工具如何无缝融入测试工作流，从而提升开发者的工作效率与项目的可维护性。","\u002Fuploads\u002F2026\u002F01\u002Ftesting-with-eslint-and-prettier-1769204102989-ezczs0-1769376472926-hwqqkt.webp","2026-01-25T10:26:00.000Z",{"id":1429,"slug":1430,"title":1431,"excerpt":1432,"featuredImage":1433,"publishedAt":1434},"384","new-qwen-3-5-plus","全新Qwen 3.5-Plus：开源AI迈入新纪元","探索阿里巴巴Qwen 3.5-Plus的革命性特性与优势，这款为开发者打造的颠覆性开源人工智能模型。","\u002Fuploads\u002F2026\u002F02\u002Fnew-qwen-3-5-plus-1771515512741-dcbi9p.webp","2026-02-19T10:23:00.000Z",{"id":1436,"slug":1437,"title":1438,"excerpt":1439,"featuredImage":1440,"publishedAt":1441},"451","test-dev-enterprise","Test DEv Enterprise Stajic.de 全面指南：架构与最佳实践","探索使用 Test DEv Enterprise Stajic.de 管理企业级开发和测试环境的架构原则、优势及技术细节。","\u002Fuploads\u002F2026\u002F05\u002Ftest-dev-enterprise-1779534260081-r4dvxn.webp","2026-05-22T23:01:00.000Z",{"id":1443,"slug":1444,"title":1445,"excerpt":1446,"featuredImage":1447,"publishedAt":1448},"434","evaluation-harness","全面评估指南：精通LLM性能评估","本指南详细介绍了评估工具（Evaluation Harness），这是一个在企业级LLMOps流程中严格评估大型语言模型（LLM）能力的关键框架。您将学习其设置方法、最佳实践以及高级技巧，以确保模型基准测试与优化的可靠性。","\u002Fuploads\u002F2026\u002F04\u002Fevaluation-harness-1775466944495-4s0xv2.webp","2026-03-01T17:50:00.000Z",{"id":1450,"slug":1451,"title":1452,"excerpt":1453,"featuredImage":1454,"publishedAt":1455},"448","google-io-2026-android-xr-and-intelligent-eyewear","Google I\u002FO 2026：Android XR、智能眼镜与环境AI界面","Google I\u002FO 2026 将 Android XR 和智能眼镜从概念推向实际平台方向。本文解析了音频眼镜、显示眼镜、Gemini 驱动的上下文感知、开发者影响、隐私风险，以及为何可穿戴 AI 更关乎创造环境辅助界面，而非取代手机。","\u002Fuploads\u002F2026\u002F05\u002Fgoogle-io-2026-android-xr-and-intelligent-eyewear-1779227942270-dtsm9y.webp","2026-05-21T11:05:00.000Z",{"id":1457,"slug":1458,"title":1459,"excerpt":1460,"featuredImage":1461,"publishedAt":1462},"466","the-gpu-is-not-the-product-future-proof-private-ai-architecture","GPU 不是产品：面向未来的私有 AI 架构","私有 AI 基础设施不应围绕单一 GPU 或单一模型来设计。更具韧性的做法是将快速推理 GPU、内存充裕的 AI 系统、物理 AI 节点以及可选的前沿云模型，统一置于一个具备能力感知的路由层之后。","\u002Fuploads\u002F2026\u002F09\u002Fthe-gpu-is-not-the-product-future-proof-private-ai-architecture-1790140878812-8hsl39.webp","2026-09-23T01:19:00.000Z",{"id":1464,"slug":1465,"title":1466,"excerpt":1467,"featuredImage":1468,"publishedAt":1469},"472","why-more-context-can-make-ai-answers-worse","为什么更多上下文会让AI的回答更糟","更大的上下文窗口并不保证更好的答案。本文解释了信号稀释、证据冲突、状态过时、位置敏感性和有损压缩如何降低AI可靠性——并介绍了一种实用的上下文压力测试。","\u002Fuploads\u002F2026\u002F09\u002Fwhy-more-context-can-make-ai-answers-worse-1790351615793-2ntv2v.webp","2026-09-25T11:51:00.000Z",{"id":1471,"slug":1472,"title":1473,"excerpt":1474,"featuredImage":1475,"publishedAt":1476},"471","how-to-know-whether-an-ai-agent-actually-used-the-right-evidence","如何判断一个AI智能体是否真正使用了正确的证据","AI代理可以引用来源，却仍然使用错误的证据。本文介绍一种实用方法，用于核查主张支持、来源权威性、适用性、出处，以及证据是否实际影响了答案。","\u002Fuploads\u002F2026\u002F09\u002Fhow-to-know-whether-an-ai-agent-actually-used-the-right-evidence-1790351317188-o5z9ve.webp","2026-09-25T11:47:00.000Z",{"id":1478,"slug":1479,"title":1480,"excerpt":1481,"featuredImage":1482,"publishedAt":1483},"375","database-marketing","Database Marketing: A Modern Approach to Customer Relationships","Database marketing is essential for modern customer relationship management. Learn how strategic data use, technical expertise, and innovation drive personalized customer interactions and sustainable growth.","\u002Fuploads\u002F2025\u002F01\u002FDatabasemarketing.png-medium.webp","2025-01-06T00:20:00.000Z",{"id":1485,"slug":1486,"title":1487,"excerpt":1488,"featuredImage":1489,"publishedAt":1490},"360","ubuntu-debian-doppelte-apt-paketquellen-entfernen","Remove Duplicate APT Package Sources: Expert Guide for Ubuntu and Debian","A detailed guide for identifying and removing redundant or duplicate APT package sources in Debian and Ubuntu systems to ensure stability and performance.","\u002Fuploads\u002F2022\u002F05\u002FUbuntu-APT-Paketquellen-www.stajic.de_.webp","2025-05-02T09:09:00.000Z","fallback",[],[]]