[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"portal-settings:stajic:zh":3,"public-menus:all":38,"post:vector-databases-embeddings-and-reranking-three-different-parts-of-retrieval:zh":205,"related:post:vector-databases-embeddings-and-reranking-three-different-parts-of-retrieval:zh:1":2658},{"statusCode":4,"data":5,"message":37},200,{"tenantId":6,"lang":7,"defaultLang":8,"siteUrl":9,"contactEmail":10,"brandName":11,"logoUrl":12,"siteName":11,"siteDescription":13,"ogImage":10,"robotsIndex":14,"socialLinks":10,"reservedSlugs":10,"seoPolicy":15},"stajic","zh","de","https:\u002F\u002Fstajic.de",null,"Stajic Platform","\u002FLogo_Planet.svg","Stajic Portal",true,{"branding":16,"relatedContent":17,"crossDomainLinks":18},{"logoUrl":12},{"enabled":14},[19,22,25,28,31,34],{"url":20,"label":21,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Ffigure.rocks","figure.rocks",{"url":23,"label":24,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Floving.rocks","loving.rocks",{"url":26,"label":27,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.com","bazify.com",{"url":29,"label":30,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.de","bazify.de",{"url":32,"label":33,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.at","bazify.at",{"url":35,"label":36,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.ba","bazify.ba","Portal settings resolved",[39,45],{"id":40,"name":41,"location":42,"isActive":14,"isDefault":43,"items":44},1,"main-navigation","header",false,[],{"id":46,"name":47,"location":48,"isActive":14,"isDefault":14,"items":49},4,"main-menu","sidebar",[50,66,79,93,103,118,133],{"id":51,"title":52,"url":60,"target":61,"icon":62,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":64,"portfolioId":10,"children":65},"item-18",{"de":53,"en":54,"es":55,"fr":56,"it":54,"ru":57,"sr":58,"zh":59},"Startseite","Home","Inicio","Accueil","Главная","Почетна","首页","\u002Ffull-stack-web-developer-munich-performance-seo-and-maintainable-builds","_self","i-lucide-home","page",111,[],{"id":67,"title":68,"url":75,"target":61,"icon":76,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":77,"portfolioId":10,"children":78},"item-22",{"de":69,"en":69,"es":70,"fr":69,"it":71,"ru":72,"sr":73,"zh":74},"Vision","Visión","Visione","Видение","Визија","想象","\u002Fueber-uns-webdesign-muenchen-webaplikation","i-lucide-eye",113,[],{"id":80,"title":81,"url":89,"target":61,"icon":90,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":91,"portfolioId":10,"children":92},"item-19",{"de":82,"en":83,"es":84,"fr":83,"it":85,"ru":86,"sr":87,"zh":88},"Leistungen","Services","Servicios","Servizi","Услуги","Услуге","服务","\u002Fservices-dienstleistungen-muenchen","i-lucide-wrench",116,[],{"id":94,"title":95,"url":99,"target":61,"icon":100,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":101,"portfolioId":10,"children":102},"item-23",{"de":96,"en":96,"es":96,"fr":96,"it":96,"ru":97,"sr":97,"zh":98},"Blog","Блог","博客","\u002Fblog","i-lucide-book-open",112,[],{"id":104,"title":105,"url":114,"target":61,"icon":115,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":116,"portfolioId":10,"children":117},"item-32",{"de":106,"en":107,"es":108,"fr":109,"it":110,"ru":111,"sr":112,"zh":113},"Neue Technologien","New Technologies","Nuevas tecnologías","Nouvelles technologies","Nuove tecnologie","Новые технологии","Нове технологије","新技术！","\u002Fneue-webtechnologien","i-lucide-sparkles",122,[],{"id":119,"title":120,"url":129,"target":61,"icon":130,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":131,"portfolioId":10,"children":132},"item-20",{"de":121,"en":122,"es":123,"fr":124,"it":125,"ru":126,"sr":127,"zh":128},"Kontakt","Contact us!","Contacto","Contact","Contatto","Контакт","Контактирајте нас","联系我们！","\u002Fcontact","i-lucide-mail",115,[],{"id":134,"title":135,"url":144,"target":61,"icon":145,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":147},"item-21",{"de":136,"en":137,"es":138,"fr":139,"it":140,"ru":141,"sr":142,"zh":143},"Unsere Arbeit","Our Work","Nuestro trabajo","Nos réalisations","I nostri lavori","Наши работы","Наши радови","文件夹","\u002Fportfolio","i-lucide-briefcase",114,[148,161,175,181,193],{"id":149,"title":150,"url":144,"target":61,"icon":159,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":160},"item-24",{"de":151,"en":152,"es":153,"fr":154,"it":155,"ru":156,"sr":157,"zh":158},"Alle Projekte","All Projects","Todos los proyectos","Tous les projets","Tutti i progetti","Все проекты","Сви пројекти","所有项目","i-lucide-grid-3x3",[],{"id":162,"title":163,"url":171,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":174},"item-29",{"de":164,"en":165,"es":166,"fr":167,"it":168,"ru":169,"sr":170,"zh":143},"Local Roots, Global Reach","Local Roots - Global Reach","Empresa local ","Entreprise locale","Azienda locale","Местная компания","Локално предузеће глобално тржиште","\u002Fportfolio\u002Flocal-roots-global-reach-communication-media-systems-for-modern-business","i-lucide-folder","custom",[],{"id":176,"title":177,"url":179,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":180},"item-28",{"de":178,"en":178,"es":178,"fr":178,"it":178,"ru":178,"sr":178,"zh":178},"Solr Suggester","\u002Fportfolio\u002Fsolr-fuzzy-suggester-und-solr-infix-suggester-abfrage-ueber-ajax-und-filterung",[],{"id":182,"title":183,"url":191,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":192},"item-27",{"de":184,"en":185,"es":186,"fr":187,"it":188,"ru":189,"sr":190,"zh":185},"Firmenwebseite SEO","Company Website SEO","Sitio web corporativo SEO","Site web d’entreprise SEO","Sito web aziendale SEO","Корпоративный сайт SEO","Пословна веб-страница SEO","\u002Fportfolio\u002Fseo-sem-branding-mobile-webseite-muenchen",[],{"id":194,"title":195,"url":203,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":204},"item-31",{"de":196,"en":197,"es":198,"fr":199,"it":200,"ru":201,"sr":202,"zh":197},"Digitalisierungsportal","Digitalization Portal","Portal de digitalización","Portail de numérisation","Portale di digitalizzazione","Портал цифровизации","Портал за дигитализацију","\u002Fportfolio\u002Fdigitalisierungsportal-archiv-museum-bibliothek-ead-lido-mets-mods",[],{"statusCode":4,"data":206,"message":2657},{"id":207,"title":208,"slug":209,"content":210,"contentJson":211,"excerpt":1251,"featuredImage":1252,"featuredImageAlt":1253,"featuredImageCaption":10,"featuredImageTitle":10,"featuredImageCopyright":10,"featuredImageAuthor":10,"featuredImageSourceUrl":10,"featuredImageLicense":10,"featuredImageIsAiGenerated":43,"status":1254,"publishedAt":1255,"createdAt":1256,"updatedAt":1257,"seoLocalePaths":1258,"categories":1267,"author":1280,"translations":1285},"487","向量数据库、嵌入和重排序：检索的三个不同部分","vector-databases-embeddings-and-reranking-three-different-parts-of-retrieval","\u003Cp>嵌入、向量数据库和重排序器是检索的三个不同组成部分。嵌入模型将文本或其他数据转换为数值表示；向量数据库或向量索引存储并搜索这些表示以检索候选项；重排序器接收一个较小的候选集，并使用更昂贵的相关性模型或评分方法对其重新排序。它们经常一起出现在RAG中，但它们都不是RAG本身，也不是每个检索系统都必须具备的。\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--info my-6 rounded-xl border p-5 border-blue-300 bg-blue-50 dark:border-blue-900 dark:bg-blue-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">直接回答\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">\u003Cstrong>嵌入负责表示。向量搜索负责检索。重排序负责精炼。\u003C\u002Fstrong>\u003Cbr>\u003Cbr>一个有用的心智模型是：\u003Cbr>\u003Cstrong>内容 → 嵌入 → 候选检索 → 重排序 → 选定上下文 → 模型\u003C\u002Fstrong>。\u003Cbr>\u003Cbr>这些边界很重要，因为每一层的失败方式不同。糟糕的嵌入会扭曲语义相似度。薄弱的检索索引会遗漏有用的候选。重排序器可以重新排列候选，但无法恢复从未被检索到的相关文档。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Caside class=\"editorjs-callout editorjs-callout--warning my-6 rounded-xl border p-5 border-amber-300 bg-amber-50 dark:border-amber-900 dark:bg-amber-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">不要混淆检索栈的各个层次\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">向量数据库不是嵌入模型。嵌入不是搜索结果。重排序器不是向量数据库。RAG是更广泛的模式，可以在生成之前使用这些组件中的任何一个来检索外部信息。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Caside class=\"editorjs-callout editorjs-callout--note my-6 rounded-xl border p-5 border-gray-300 bg-gray-50 dark:border-gray-700 dark:bg-gray-900\u002F40\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">当前来源说明 — 2026年10月8日\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">尽管产品快速演进，基本架构是稳定的。当前的Qdrant文档将向量、负载元数据、集合和向量索引分开；当前的Elastic指南将语义重排序视为对小型候选集进行的后阶段操作；当前的Cohere文档同样将重排序描述为对词法或语义搜索的第二阶段改进。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Cnav class=\"editorjs-toc\" data-editorjs-toc=\"true\" aria-label=\"目录\">\u003Cstrong class=\"editorjs-toc__title\">目录\u003C\u002Fstrong>\u003Col class=\"editorjs-toc__list editorjs-toc__list--depth-0\">\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-6\" class=\"editorjs-toc__link\">这真正意味着什么\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-10\" class=\"editorjs-toc__link\">最简单的例子\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-15\" class=\"editorjs-toc__link\">简单示例的局限\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-20\" class=\"editorjs-toc__link\">嵌入：表示，而非检索\u003C\u002Fa>\u003Col class=\"editorjs-toc__list editorjs-toc__list--depth-1\">\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-24\" class=\"editorjs-toc__link\">嵌入模型定义了表示空间\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-27\" class=\"editorjs-toc__link\">稠密表示和稀疏表示是不同的\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-30\" class=\"editorjs-toc__link\">相似度函数是表示契约的一部分\u003C\u002Fa>\u003C\u002Fli>\u003C\u002Fol>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-34\" class=\"editorjs-toc__link\">向量数据库和索引：候选检索\u003C\u002Fa>\u003Col class=\"editorjs-toc__list editorjs-toc__list--depth-1\">\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-38\" class=\"editorjs-toc__link\">近似最近邻搜索以精确性换取效率\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-42\" class=\"editorjs-toc__link\">元数据过滤应在候选检索之前或期间进行\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-45\" class=\"editorjs-toc__link\">向量数据库是可选的\u003C\u002Fa>\u003C\u002Fli>\u003C\u002Fol>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-48\" class=\"editorjs-toc__link\">重排序：第二阶段的相关性优化\u003C\u002Fa>\u003Col class=\"editorjs-toc__list editorjs-toc__list--depth-1\">\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-52\" class=\"editorjs-toc__link\">双编码器检索和交叉编码器重排序解决不同的成本问题\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-54\" class=\"editorjs-toc__link\">重排序器无法恢复检索遗漏的内容\u003C\u002Fa>\u003C\u002Fli>\u003C\u002Fol>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-58\" class=\"editorjs-toc__link\">混合检索是一个独立的设计选择\u003C\u002Fa>\u003Col class=\"editorjs-toc__list editorjs-toc__list--depth-1\">\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-62\" class=\"editorjs-toc__link\">BM25 并不会因为嵌入的存在而过时\u003C\u002Fa>\u003C\u002Fli>\u003C\u002Fol>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-65\" class=\"editorjs-toc__link\">分块会改变嵌入和重排序器能够看到的内容\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-68\" class=\"editorjs-toc__link\">不要像比较通用概率那样比较检索分数\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-71\" class=\"editorjs-toc__link\">分别评估检索阶段\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-74\" class=\"editorjs-toc__link\">究竟是哪一层失败了？\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-76\" class=\"editorjs-toc__link\">相关性与事实来源是不同的\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-80\" class=\"editorjs-toc__link\">原始实现证据\u003C\u002Fa>\u003Col class=\"editorjs-toc__list editorjs-toc__list--depth-1\">\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-81\" class=\"editorjs-toc__link\">事实来源研究引擎：词法检索与语义检索是分离的\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-85\" class=\"editorjs-toc__link\">Aaasaasa AI 客户端：Qdrant 是向量基础设施组件\u003C\u002Fa>\u003C\u002Fli>\u003C\u002Fol>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-91\" class=\"editorjs-toc__link\">何时需要每个组件？\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-93\" class=\"editorjs-toc__link\">实用的检索设计流程\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-95\" class=\"editorjs-toc__link\">常见误解\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-97\" class=\"editorjs-toc__link\">边缘情况和限制\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-103\" class=\"editorjs-toc__link\">什么会改变这个答案？\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-107\" class=\"editorjs-toc__link\">相关规范知识\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-113\" class=\"editorjs-toc__link\">常见问题\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-115\" class=\"editorjs-toc__link\">术语表\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-117\" class=\"editorjs-toc__link\">结论\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-121\" class=\"editorjs-toc__link\">主要来源和实现证据\u003C\u002Fa>\u003C\u002Fli>\u003C\u002Fol>\u003C\u002Fnav>\n\u003Ch2 id=\"section-6\">这真正意味着什么\u003C\u002Fh2>\n\u003Cp>搜索系统有两个相互竞争的目标：找到足够多可能相关的材料，并将最好的材料放在靠近顶部的位置。快速的第一阶段检索通常优化候选生成。更强的第二阶段模型则可以花费更多计算来区分最佳候选。\u003C\u002Fp>\n\u003Cp>嵌入、向量索引和重排序器在该过程中占据不同的位置。将它们视为一个功能会掩盖关于召回率、精确率、延迟、存储、元数据过滤和模型成本的重要设计选择。\u003C\u002Fp>\n\u003Cp>这种区分还可以防止一个常见的RAG错误：假设将文档嵌入存储在向量数据库中就会自动创建高质量的检索。检索质量取决于嵌入模型、分块、元数据、查询构建、索引配置、候选数量、混合检索、重排序以及底层来源的权威性。\u003C\u002Fp>\n\u003Ch2 id=\"section-10\">最简单的例子\u003C\u002Fh2>\n\u003Cp>假设一个知识库包含100,000个文档块。用户问：“如何撤销API令牌？”\u003C\u002Fp>\n\u003Cp>首先，嵌入模型可以将查询编码为向量。文档块可能已经存储了自己的嵌入。然后，向量搜索将查询向量与索引的文档向量进行比较，并返回例如30个可能的候选。\u003C\u002Fp>\n\u003Cp>这30个候选随后可以传递给重排序器。重排序器更直接地将查询与每个候选进行比较，并产生新的相关性排序。应用程序可能会保留最好的五个用于模型上下文。\u003C\u002Fp>\n\u003Csection class=\"editorjs-process my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">一个基本的两阶段语义检索流程\u003C\u002Fh3>\u003Cdiv class=\"grid grid-cols-1 md:grid-cols-2 xl:grid-cols-3 gap-4\">\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">1\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">1. 嵌入文档\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">将每个可搜索的块转换为数值表示，通常在摄取时进行。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">2\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">2. 存储\u002F索引向量\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">将向量与文档ID和元数据关联到可搜索的向量索引或数据库中。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">3\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">3. 嵌入查询\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">使用兼容的嵌入模型和查询配置对用户的查询进行编码。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">4\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">4. 检索候选\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">运行向量相似性搜索，通常带有元数据过滤器，以生成更大的top-k候选集。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">5\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">5. 重排序候选\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">对查询和小型候选集应用更强的相关性模型。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">6\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">6. 选择上下文\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">保留最有用的段落，用于下游答案、代理步骤或搜索结果。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-15\">简单示例的局限\u003C\u002Fh2>\n\u003Cp>真实的检索系统根本不必使用密集嵌入。诸如BM25之类的关键词搜索可以作为第一阶段检索器。稀疏学习检索、SQL过滤器、图遍历或应用程序API也可以生成候选。\u003C\u002Fp>\n\u003Cp>重排序器也不关心候选是否来自向量数据库。它可以对BM25结果、混合结果、手动选择的文档或来自多个检索器的候选进行重排序。\u003C\u002Fp>\n\u003Cp>同样，嵌入不需要专门的向量数据库。小数据集可以在内存中或使用通用数据库和向量扩展进行比较。当索引、近似最近邻搜索、过滤、规模、更新行为或操作要求证明其合理性时，专门的向量系统才变得有用。\u003C\u002Fp>\n\u003Csection class=\"editorjs-comparison my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">三种不同的检索组件\u003C\u002Fh3>\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left dark:border-gray-700 dark:bg-gray-900\">\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">嵌入\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">向量数据库 \u002F 索引\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">重排序器\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">主要任务\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">典型输入\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">典型输出\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">成本特征\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">典型故障\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-20\">嵌入：表示，而非检索\u003C\u002Fh2>\n\u003Cp>嵌入是模型生成的数值表示。对于语义检索，含义相关的文本旨在在向量空间中占据有用的位置，以便相似度或距离函数可以比较它们。\u003C\u002Fp>\n\u003Cp>Sentence-BERT 是使句子级语义相似度变得实用的重要一步，它采用了双编码器风格的表示，可以独立计算并高效比较。这一总体思路仍然是现代稠密检索的核心：预计算文档表示，在搜索时计算查询表示，然后进行比较。\u003C\u002Fp>\n\u003Cp>嵌入本身并不搜索语料库。它是由嵌入模型生成的数据。当系统将查询表示与存储的候选进行比较时，检索才开始。\u003C\u002Fp>\n\u003Ch3 id=\"section-24\">嵌入模型定义了表示空间\u003C\u002Fh3>\n\u003Cp>文档和查询向量必须与创建它们时使用的模型和配置兼容。替换嵌入模型可能会改变维度、相似度行为、语言覆盖范围和领域性能。\u003C\u002Fp>\n\u003Cp>这就是为什么嵌入模型迁移不仅仅是更改 API 名称。现有文档可能需要重新嵌入，索引需要重建或版本化。\u003C\u002Fp>\n\u003Ch3 id=\"section-27\">稠密表示和稀疏表示是不同的\u003C\u002Fh3>\n\u003Cp>稠密嵌入通常包含许多非零维度，常用于语义相似度。稀疏表示包含许多零，可以保留更强的词元或词项结构。\u003C\u002Fp>\n\u003Cp>两者都可以支持语义检索，现代搜索系统可以结合稠密、稀疏和词汇信号。因此，“向量搜索”并不总是意味着单一的稠密余弦相似度流水线。\u003C\u002Fp>\n\u003Ch3 id=\"section-30\">相似度函数是表示契约的一部分\u003C\u002Fh3>\n\u003Cp>余弦相似度、点积和欧几里得距离的含义并不相同。正确的度量取决于嵌入模型是如何训练和归一化的。\u003C\u002Fp>\n\u003Cp>例如，当前的 Qdrant 文档要求将距离度量作为向量配置的一部分，并记录了余弦、点积和欧几里得风格的选择。重要的架构规则是将度量视为嵌入\u002F索引契约的一部分，而不是随意选择一种。\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--warning my-6 rounded-xl border p-5 border-amber-300 bg-amber-50 dark:border-amber-900 dark:bg-amber-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">嵌入相似度不是事实支持\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">两个段落可以在语义上接近，而其中一个是过时的、未授权的或错误的。嵌入估计的是表示相似度；它们不决定真源权威、新鲜度或证据有效性。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-34\">向量数据库和索引：候选检索\u003C\u002Fh2>\n\u003Cp>向量数据库或支持向量的搜索系统组织向量表示，以便应用程序能够高效地检索附近的候选。实际系统通常将向量与 ID 和有效负载元数据（如来源、语言、租户、文档类型、时间戳或访问范围）关联起来。\u003C\u002Fp>\n\u003Cp>例如，Qdrant 将数据组织为点的集合，其中点包含向量和可选的有效负载元数据。其文档将基于 HNSW 的相似度搜索和元数据过滤描述为检索层的独立能力。\u003C\u002Fp>\n\u003Cp>这种区分很重要：向量索引解决的是最近邻问题，而有效负载过滤器则强制执行结构性约束，例如租户、文档类别或语言。\u003C\u002Fp>\n\u003Ch3 id=\"section-38\">近似最近邻搜索以精确性换取效率\u003C\u002Fh3>\n\u003Cp>将单个查询向量与每个向量进行比较对于小型集合可能是可行的，但在大规模场景下成本高昂。诸如 HNSW 之类的近似最近邻索引通过遍历索引结构而不是穷举扫描每个向量来降低搜索成本。\u003C\u002Fp>\n\u003Cp>近似搜索引入了召回率与延迟之间的权衡。更快的搜索可能会遗漏精确搜索本应返回的候选结果。因此，索引参数不仅影响基础设施性能，还会影响检索质量。\u003C\u002Fp>\n\u003Cp>Qdrant 同时提供了与 HNSW 相关的参数和精确搜索选项，这表明向量存储和近似检索策略是相互独立的决策。\u003C\u002Fp>\n\u003Ch3 id=\"section-42\">元数据过滤应在候选检索之前或期间进行\u003C\u002Fh3>\n\u003Cp>如果用户只能访问租户 A，那么从租户 B 检索语义相似的片段并试图稍后将其移除，是错误的安​​全边界。授权和硬性资格过滤器应在候选结果能够影响下游处理之前，对候选空间进行约束。\u003C\u002Fp>\n\u003Cp>同样的原则适用于区域设置、文档状态、来源类别、日期、产品版本以及其他确定性约束。相似度应对符合条件的候选结果进行排序；它不应覆盖资格条件。\u003C\u002Fp>\n\u003Ch3 id=\"section-45\">向量数据库是可选的\u003C\u002Fh3>\n\u003Cp>对于小型语料库，暴力余弦比较可能简单且足够。具有向量支持的关系数据库也可能足够。当专用向量数据库的索引、过滤、分布式存储、更新行为或运维特性能够解决实际需求时，它才变得有价值。\u003C\u002Fp>\n\u003Cp>因为“RAG 需要一个向量数据库”而选择向量数据库，颠倒了架构设计流程。应从检索需求和规模出发，然后选择存储\u002F索引技术。\u003C\u002Fp>\n\u003Ch2 id=\"section-48\">重排序：第二阶段的相关性优化\u003C\u002Fh2>\n\u003Cp>重排序器接收一个查询和一组较小的已检索候选结果，然后分配更强的相关性分数或新的排序。它通常比第一阶段检索的计算成本更高，这就是为什么它应用于候选生成之后，而不是针对整个语料库。\u003C\u002Fp>\n\u003Cp>当前 Elastic 的指南将语义重排序描述为针对小型 top-k 集合的最终阶段技术，并指出它可以优化词汇、语义或混合检索。Cohere 记录了相同的架构：先进行第一阶段的词汇或语义搜索，然后进行重排序阶段。\u003C\u002Fp>\n\u003Cp>一种常见的实现使用类似交叉编码器的模型，该模型同时检查查询和每个候选结果。这种更丰富的交互可以比独立的嵌入相似度更精确地区分相关性，但在语料库规模下成本要高得多。\u003C\u002Fp>\n\u003Ch3 id=\"section-52\">双编码器检索和交叉编码器重排序解决不同的成本问题\u003C\u002Fh3>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">属性\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">双编码器 \u002F 嵌入检索\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">交叉编码器式重排序\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">编码方式\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">查询和文档独立表示\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">查询和候选结果联合处理\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">文档计算\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">可在摄取时预先计算\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">通常针对每个查询-候选对重新计算\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">语料库规模搜索\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">适合使用向量索引\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">在整个语料库上通常成本过高\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">典型角色\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">高召回率的候选生成\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">对小型候选集进行高精度排序\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">主要权衡\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">快速且可扩展，但相关性交互被压缩到向量中\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">更丰富的相关性判断，但延迟\u002F成本更高\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch3 id=\"section-54\">重排序器无法恢复检索遗漏的内容\u003C\u002Fh3>\n\u003Cp>如果相关文档不在候选集中，重排序就没有任何可以提升的对象。这就是需要分别评估检索和重排序的核心原因。\u003C\u002Fp>\n\u003Cp>一个流水线可能具有出色的重排序器精确率，但仍然会因为第一阶段召回率较差而失败。提高重排序器质量无法修复缺失的源覆盖、糟糕的分块、限制性过滤器或较弱的候选检索器。\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--success my-6 rounded-xl border p-5 border-emerald-300 bg-emerald-50 dark:border-emerald-900 dark:bg-emerald-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">有用的检索目标\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">第一阶段：\u003Cstrong>不要漏掉有用的候选。\u003C\u002Fstrong>\u003Cbr>第二阶段：\u003Cstrong>把最好的候选放在前面。\u003C\u002Fstrong>\u003Cbr>\u003Cbr>这不是一条通用的数学规则，但它是两阶段检索中一个有用的工程模型。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-58\">混合检索是一个独立的设计选择\u003C\u002Fh2>\n\u003Cp>当查询和文档使用不同措辞但表达相关含义时，稠密语义检索表现很强。当精确术语、标识符、名称、代码或罕见短语很重要时，词法检索表现很强。\u003C\u002Fp>\n\u003Cp>混合检索结合多种候选信号，通常是词法 BM25 和向量相似度，然后使用诸如倒数排名融合或加权分数组合之类的方法合并排名。\u003C\u002Fp>\n\u003Cp>随后，重排序可以在融合后的候选集上运行。因此，混合检索和重排序是互补但不同的阶段。\u003C\u002Fp>\n\u003Ch3 id=\"section-62\">BM25 并不会因为嵌入的存在而过时\u003C\u002Fh3>\n\u003Cp>对于精确标识符、版本号、错误消息、产品代码和专业词汇，关键词搜索可能优于稠密检索。例如，SQLite FTS5 包含用于全文搜索的 BM25 排名函数。\u003C\u002Fp>\n\u003Cp>一个强大的检索架构可以将词法检索作为唯一的第一阶段，将向量检索作为唯一的第一阶段，或者根据语料库和查询分布将两者结合起来。\u003C\u002Fp>\n\u003Ch2 id=\"section-65\">分块会改变嵌入和重排序器能够看到的内容\u003C\u002Fh2>\n\u003Cp>如果文档被糟糕地切分，后续任何检索组件都无法完全重建缺失的语义单元。一个将条件与其例外切开的块，可能会以误导性的方式嵌入，也可能因为候选文本不完整而被错误地重排序。\u003C\u002Fp>\n\u003Cp>因此，块大小、重叠、结构边界和元数据都会影响候选召回和重排序器判断。检索评估应测试从摄取到排名的完整流水线，而不仅仅是嵌入模型。\u003C\u002Fp>\n\u003Ch2 id=\"section-68\">不要像比较通用概率那样比较检索分数\u003C\u002Fh2>\n\u003Cp>余弦相似度、BM25 分数、稀疏向量分数、RRF 排名和重排序器分数具有不同含义。来自一个嵌入模型的 0.82 分，并不自动与来自另一个模型的 0.82 分或与重排序器分数具有可比性。\u003C\u002Fp>\n\u003Cp>阈值应针对实际模型、语料库和任务进行校准。当前 Elastic 指南还指出，嵌入相似度分数可能依赖于查询，这使得通用截断值存在风险。\u003C\u002Fp>\n\u003Ch2 id=\"section-71\">分别评估检索阶段\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">层级\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">有用的问题\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">示例指标或测试\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">源覆盖\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">语料库是否包含所需信息？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">覆盖审计 \u002F 已知答案源集合\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">分块\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">所需证据是否能作为连贯单元被检索到？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">块级支持审查\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">第一阶段检索\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">相关条目是否进入候选集？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Recall@k\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">排名\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">相关证据出现得有多靠前？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">MRR、nDCG、precision@k\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">重排序\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">第二阶段评分是否改善排序？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Delta nDCG \u002F MRR \u002F precision\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">上下文选择\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">最终选定的段落是否包含足够支持？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">上下文相关性 \u002F 覆盖\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">答案阶段\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">模型是否正确使用选定的证据？\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">忠实性 \u002F 主张-证据评估\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Cp>这种分离在操作上很重要。如果 Recall@50 很差，重排序器并不是第一个需要修复的组件。如果 Recall@50 很强，但最佳段落仍排在第 38 位，那么重排序或排名融合就成为一个可能的目标。\u003C\u002Fp>\n\u003Ch2 id=\"section-74\">究竟是哪一层失败了？\u003C\u002Fh2>\n\u003Csection class=\"editorjs-comparison my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">症状与可能的检索层\u003C\u002Fh3>\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left dark:border-gray-700 dark:bg-gray-900\">\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">观察到的症状\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">可能的层\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">首要诊断\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">相关文档从未出现\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">相关文档排名过低\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">语义上良好但被禁止的结果\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">相关但过时的结果\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">检索到正确结果但被提示词遗漏\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-76\">相关性与事实来源是不同的\u003C\u002Fh2>\n\u003Cp>重排序器可以让一份过时的文档看起来极其相关。向量索引可以检索到语义上比原始来源更接近的次级摘要。因此，检索质量不能替代权威规则。\u003C\u002Fp>\n\u003Cp>在来源权威性重要的地方，元数据过滤器、来源类别、版本规则和出处应在结果成为模型上下文之前约束检索。\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--warning my-6 rounded-xl border p-5 border-amber-300 bg-amber-50 dark:border-amber-900 dark:bg-amber-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">重排序无法使非权威来源变得权威\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">相关性回答的是候选结果是否匹配查询。事实来源架构回答的是该候选结果是否被允许确立该主张。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-80\">原始实现证据\u003C\u002Fh2>\n\u003Ch3 id=\"section-81\">事实来源研究引擎：词法检索与语义检索是分离的\u003C\u002Fh3>\n\u003Cp>事实来源研究引擎包含一条使用 SQLite FTS5\u002FBM25 的本地词法检索路径，以及一条使用本地生成嵌入的独立可选语义检索路径。\u003C\u002Fp>\n\u003Cp>其语义搜索实现会计算查询向量，并使用余弦相似度将其与存储的块向量进行比较。该项目有意将语义相似度视为发现信号而非证据：候选结果在支持某项主张之前，仍必须追溯到具体的来源和定位符。\u003C\u002Fp>\n\u003Cp>这对 R01 来说是有用的实现证据，因为同一语料库可以同时支持词法排名和向量相似度，而不会将任一机制与证据权威性相混淆。\u003C\u002Fp>\n\u003Ch3 id=\"section-85\">Aaasaasa AI 客户端：Qdrant 是向量基础设施组件\u003C\u002Fh3>\n\u003Cp>Aaasaasa AI 客户端将 Qdrant\u002F向量基础设施作为独立的本地资源。Electron 架构从受信任的主进程侧暴露 Qdrant 服务，而不是将向量搜索视为模型本身的一部分。\u003C\u002Fp>\n\u003Cp>该仓库包含 Qdrant 客户端适配器、Qdrant 服务配置以及基于 Docker 的 Qdrant 基础设施。这展示了 AI 提供商\u002F模型执行与向量存储\u002F搜索之间的架构分离。\u003C\u002Fp>\n\u003Cp>不应将 Qdrant 支持的存在夸大为完整的生产级 RAG 流水线。这里的证据更为有限：向量基础设施是作为其自身的组件边界实现的。\u003C\u002Fp>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">实现证据\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">它展示了什么\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">事实来源研究引擎中的 SQLite FTS5\u002FBM25\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">词法检索可以独立于嵌入存在。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">本地 Ollama 嵌入\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">表示生成是其自身的阶段。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">存储的语义向量 + 余弦比较\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">语义检索在嵌入生成之后消费它们。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Aaasaasa AI 客户端中的 Qdrant 支持\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">向量存储\u002F搜索是一种独立于模型提供商的基础设施能力。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">事实来源研究引擎中的证据\u002F出处规则\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">检索到的相似度不等于权威性或证明。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">这些实现中没有声称的自定义重排序器\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">重排序被解释为一个架构阶段，而不是被虚假声称已经实现的证据。\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Caside class=\"editorjs-callout editorjs-callout--note my-6 rounded-xl border p-5 border-gray-300 bg-gray-50 dark:border-gray-700 dark:bg-gray-900\u002F40\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">证据边界\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">当前的实现证据确认了词法检索、嵌入、向量搜索基础设施和出处感知检索。本文\u003Cstrong>并不\u003C\u002Fstrong>声称这些项目中已经实现了生产级交叉编码器重排序服务。\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-91\">何时需要每个组件？\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">需求\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">可能的组件\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">不同措辞之间的语义相似性\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">嵌入模型 + 向量相似性搜索\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">在大型向量语料库上进行高效搜索\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">向量索引\u002F数据库或支持向量的搜索引擎\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">精确标识符、错误代码或罕见术语\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">词汇\u002F全文检索，如 BM25\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">既需要精确术语又需要语义含义\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">混合词汇 + 语义检索\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">候选集良好但排序较弱\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">重排序器\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">相关项不在候选集中\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">在重排序之前改进源覆盖、分块、检索器、过滤器或候选数量\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">硬性租户\u002F来源\u002F版本约束\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">确定性元数据\u002F授权过滤\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">小型语料库\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">可能使用简单的暴力相似性计算或通用数据库，而非专用向量数据库\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-93\">实用的检索设计流程\u003C\u002Fh2>\n\u003Csection class=\"editorjs-process my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">从需求出发设计检索，而非从产品名称出发\u003C\u002Fh3>\u003Cdiv class=\"grid grid-cols-1 md:grid-cols-2 xl:grid-cols-3 gap-4\">\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">1\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">1. 定义查询类型\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">识别语义问题、精确查找、标识符、当前状态读取和领域特定模式。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">2\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">2. 定义合格来源\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">应用租户、授权、区域设置、版本、来源类别和新鲜度约束。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">3\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">3. 建立词汇基线\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">衡量简单的全文\u002FBM25 检索是否已经解决了大部分工作负载。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">4\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">4. 在需要语义召回的地方添加嵌入\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">针对代表性领域查询选择并评估嵌入模型。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">5\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">5. 根据规模选择向量存储\u002F索引\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">根据需求使用暴力搜索、数据库向量支持或专用向量引擎。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">6\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">6. 评估第一阶段召回\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">确认相关证据进入足够大的候选集。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">7\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">7. 如果信号互补则添加混合检索\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">当词汇和语义排名都能实质性改善候选生成时，融合两者。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">8\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">8. 如果排序仍是瓶颈则添加重排序\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">仅对候选集应用更强的模型，以证明其成本合理。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">9\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">9. 调整最终上下文选择\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">在生成之前控制冗余、上下文预算、权威性、多样性和证据覆盖。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">10\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">10. 端到端评估\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">分别衡量检索、上下文和答案质量，以便定位失败环节。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-95\">常见误解\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">误解\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">纠正\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“嵌入就是向量数据库。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">嵌入是一种表示；数据库\u002F索引存储并搜索表示。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“向量数据库创造语义含义。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">嵌入模型创建表示；向量系统对其进行索引和比较。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“RAG 需要向量数据库。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">RAG 需要检索，而不是特定的检索技术。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“重排序与向量搜索相同。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">向量搜索生成候选；重排序对候选集重新排序。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“重排序器能修复糟糕的召回。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">它们无法提升从未被检索到的文档。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“密集搜索取代 BM25。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">词汇搜索对于精确术语、标识符和专业词汇仍然有价值。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“相似度越高意味着越权威。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">相似度和来源权威性是不同的维度。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“更大的 top-k 总能改善 RAG。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">更大的候选集可以提高召回，但会增加延迟、噪声和上下文选择负担。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“一个分数阈值适用于所有情况。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">分数取决于模型、查询、语料库和检索方法，必须进行校准。\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">“专用向量数据库总是更先进。”\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">只有当其操作和检索能力符合需求时，才值得使用。\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-97\">边缘情况和限制\u003C\u002Fh2>\n\u003Cp>有些应用不需要语义搜索。精确的数据库查找或结构化 SQL 可能比嵌入检索更正确、更快速且更易于审计。\u003C\u002Fp>\n\u003Cp>有些语料库非常小，全向量扫描是可以接受的。近似索引增加了复杂性却没有有意义的收益。\u003C\u002Fp>\n\u003Cp>有些查询在任何精度优化之前就需要高召回。法律发现、研究和合规审查可能更倾向于广泛的候选检索，然后进行透明的过滤和人工审查。\u003C\u002Fp>\n\u003Cp>多语言和领域特定的检索在不同嵌入模型上的表现可能差异很大。来自公共数据集的基准声明不应被视为对私有语料库的证明。\u003C\u002Fp>\n\u003Cp>重排序延迟随候选数量和长度的增加而增长。因此，候选大小应作为准确性\u002F成本\u002F延迟变量进行调整，而不是从教程中复制。\u003C\u002Fp>\n\u003Ch2 id=\"section-103\">什么会改变这个答案？\u003C\u002Fh2>\n\u003Cp>如果供应商将嵌入生成、向量索引和重排序打包在一个 API 后面，组件边界不会改变。产品可能隐藏这些阶段，但它们在概念上仍然是不同的职责，具有不同的故障模式。\u003C\u002Fp>\n\u003Cp>未来的嵌入或检索模型可能会在某些工作负载中减少对单独重排序的需求，而更强的后期交互或学习稀疏方法可能会模糊传统的密集\u002F词汇类别。架构仍然应该问哪个阶段产生表示、哪个阶段生成候选以及哪个阶段细化排名。\u003C\u002Fp>\n\u003Cp>最佳设计还会随语料库大小、查询组合、语言、领域术语、更新频率、来源权威性、延迟预算和评估结果而变化。\u003C\u002Fp>\n\u003Ch2 id=\"section-107\">相关规范知识\u003C\u002Fh2>\n\u003Cp>R01 假设基本的 RAG 概念已经被理解。RAG 是一种更广泛的模式，其中检索到的外部信息被提供给模型；嵌入、向量搜索和重排序是该模式中的可选检索组件。\u003C\u002Fp>\n\u003Caside class=\"editorjs-referral my-6\">\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works\" class=\"flex flex-col sm:flex-row gap-4 rounded-xl border border-gray-200 dark:border-gray-700 p-4 transition hover:border-primary-500\">\u003Cdiv class=\"min-w-0 flex-1\">\u003Cstrong class=\"block text-lg text-gray-900 dark:text-gray-100\">什么是 RAG？对其工作原理的最简单解释\u003C\u002Fstrong>\u003Cp class=\"mt-2 text-sm text-gray-600 dark:text-gray-300\">用通俗易懂的语言介绍检索如何将外部知识带入模型上下文的基础。\u003C\u002Fp>\u003Cspan class=\"mt-3 inline-flex text-sm font-medium text-primary-600 dark:text-primary-400\">阅读 RAG 基础 →\u003C\u002Fspan>\u003C\u002Fdiv>\u003C\u002Fa>\u003C\u002Faside>\n\u003Cp>当检索失败时，应分别诊断来源覆盖、检索、排序、上下文组装和生成，而不是将整个系统视为一次“RAG 失败”。\u003C\u002Fp>\n\u003Caside class=\"editorjs-referral my-6\">\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method\" class=\"flex flex-col sm:flex-row gap-4 rounded-xl border border-gray-200 dark:border-gray-700 p-4 transition hover:border-primary-500\">\u003Cdiv class=\"min-w-0 flex-1\">\u003Cstrong class=\"block text-lg text-gray-900 dark:text-gray-100\">RAG 失败了——但究竟是哪一层失败了？一种诊断方法\u003C\u002Fstrong>\u003Cp class=\"mt-2 text-sm text-gray-600 dark:text-gray-300\">一种逐层隔离来源覆盖、检索、排序、上下文组装、生成、证据归因和时效性失败的方法。\u003C\u002Fp>\u003Cspan class=\"mt-3 inline-flex text-sm font-medium text-primary-600 dark:text-primary-400\">阅读 RAG 诊断方法 →\u003C\u002Fspan>\u003C\u002Fdiv>\u003C\u002Fa>\u003C\u002Faside>\n\u003Cp>真相来源架构是检索周围的权威层：它决定哪个来源可以确立一项主张，而嵌入和排序只决定哪些候选看起来相关。\u003C\u002Fp>\n\u003Ch2 id=\"section-113\">常见问题\u003C\u002Fh2>\n\u003Csection class=\"editorjs-faq my-6 rounded-xl border border-gray-200 p-5 dark:border-gray-700\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">嵌入、向量数据库和重排序\u003C\u002Fh3>\u003Cdiv id=\"faq1\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">嵌入和向量数据库有什么区别？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">嵌入是由模型生成的数值表示。向量数据库或向量索引将这些表示与 ID 和元数据一起存储和搜索。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq2\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">重排序器做什么？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">重排序器接收已经检索到的候选集，并使用更强的相关性模型或评分方法对这些候选重新评分或重新排序。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq3\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">RAG 需要向量数据库吗？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">不需要。RAG 需要检索外部信息。检索可以使用词法搜索、SQL、API、图、向量搜索、混合搜索或这些的组合。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq4\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">为什么不对整个语料库使用重排序器？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">重排序器通常执行更昂贵的查询-文档交互，因此通常在更快的首阶段检索器之后应用于较小的 top-k 候选集。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq5\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">重排序能修复缺失的文档吗？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">不能。如果相关文档没有被检索到候选集中，重排序就没有任何东西可以提升。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq6\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">余弦相似度是相关性概率吗？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">不是。它是一种相似度度量，其数值含义取决于嵌入模型和语料库。不应将其视为普遍的相关性概率。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq7\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">我应该同时使用 BM25 和向量搜索吗？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">当评估显示词法和语义信号能够恢复互补的相关文档时，使用混合检索。它并非对每个语料库都自动更好。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq8\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">什么时候需要专用的向量数据库？\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">当向量索引、过滤、规模、更新、分布式操作或其他向量特定需求足以证明需要专用系统时。小型工作负载可能不需要。\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-115\">术语表\u003C\u002Fh2>\n\u003Csection class=\"editorjs-glossary my-6 rounded-xl border border-gray-200 dark:border-gray-700 p-5\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">关键检索术语\u003C\u002Fh3>\u003Cdl>\u003Cdiv id=\"embedding\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">嵌入\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">由嵌入模型生成的内容数值表示，用于相似度、聚类、检索或相关任务。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"dense-vector\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">稠密向量\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">一种向量表示，其中许多维度携带非零值，常用于语义检索。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"sparse-vector\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">稀疏向量\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">一种高维表示，其中大多数维度为零，通常保留更强的词元或词项结构。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"vector-index\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">向量索引\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">一种数据结构，用于组织向量以实现高效的相似度或最近邻检索。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"vector-database\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">向量数据库\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">一种存储\u002F搜索系统，旨在管理向量、相关元数据和向量检索工作负载。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"ann\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">ANN\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">近似最近邻搜索，以精确穷举比较换取大规模下更快的检索。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"hnsw\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">HNSW\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">分层可导航小世界，一种基于图的近似最近邻索引方法，广泛用于向量检索。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"bm25\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">BM25\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">一种基于词项出现和语料库统计的词法相关性排序方法，广泛用于全文搜索。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"hybrid-search\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">混合搜索\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">结合多种检索方法（如词法搜索和向量搜索）的结果或分数的检索。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"reranking\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">重排序\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">检索的后续阶段，对已生成的候选集重新评分和重新排序。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"bi-encoder\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">双编码器\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">一种独立编码查询和候选的架构，支持预计算和可扩展的相似度搜索。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"cross-encoder\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">交叉编码器\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">一种联合处理查询和候选文本的模型，通常以更高的计算成本提高相关性判断。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"recall-at-k\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">Recall@k\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">在前 k 个检索到的候选中恢复的相关项比例。\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"ndcg\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">nDCG\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">归一化折损累计增益，一种排序指标，奖励在有序列表中位置更高的相关结果。\u003C\u002Fdd>\u003C\u002Fdiv>\u003C\u002Fdl>\u003C\u002Fsection>\n\u003Ch2 id=\"section-117\">结论\u003C\u002Fh2>\n\u003Cp>清晰的检索模型很简单：嵌入表示含义，向量搜索检索候选，重排序器优化候选排序。\u003C\u002Fp>\n\u003Cp>一旦这些边界明确，架构决策就更容易诊断。缺失候选指向来源覆盖、分块、嵌入、过滤器或首阶段检索。排序不佳指向排序、融合或重排序。然后可以在上下文和生成层分别调查不正确的最终答案。\u003C\u002Fp>\n\u003Cp>最重要的结果不是选择最时髦的检索组件。而是构建一个检索管道，其阶段、权威边界、指标和失败模式可以独立测量。\u003C\u002Fp>\n\u003Ch2 id=\"section-121\">主要来源和实现证据\u003C\u002Fh2>\n\u003Cp>以下外部参考文献记录了本文中使用的表示、向量搜索和重排序机制。项目特定部分是原创实现证据，有意比关于完整生产 RAG 成熟度的声明更窄。\u003C\u002Fp>\n\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F1908.10084\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">Sentence-BERT：使用孪生 BERT 网络的句子嵌入\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">基础论文，展示了可独立计算的句子嵌入，用于高效的语义相似度搜索。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fqdrant.tech\u002Fdocumentation\u002Foverview\u002F\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">Qdrant — 架构和数据结构概述\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">官方文档，描述集合、点、向量、有效负载元数据和基于 HNSW 的相似度索引。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fqdrant.tech\u002Fdocumentation\u002Fsearch\u002Fsearch\u002F\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">Qdrant — 搜索\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">官方向量搜索文档，涵盖相似度查询、过滤、精确与近似搜索以及稠密\u002F稀疏行为。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fwww.elastic.co\u002Fdocs\u002Fsolutions\u002Fsearch\u002Fvector\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">Elastic — 向量搜索\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">关于稠密\u002F稀疏向量检索、词法\u002F向量组合和多阶段搜索管道的最新文档。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fwww.elastic.co\u002Fdocs\u002Fsolutions\u002Fsearch\u002Franking\u002Fsemantic-reranking\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">Elastic — 语义重排序\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">当前指南将语义重排序定义为对较小候选集进行的后期相关性操作。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fdocs.cohere.com\u002Fdocs\u002Freranking-with-cohere\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">Cohere — 使用 Cohere 进行重排序\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">当前文档展示重排序作为对词法或语义第一阶段检索的第二阶段改进。\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fwww.sqlite.org\u002Ffts5.html\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">SQLite FTS5\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">SQLite 官方文档，介绍全文搜索及用作词法检索证据的内置 BM25 排序函数。\u003C\u002Fp>\u003C\u002Fa>",{"time":212,"blocks":213,"version":1250},1791480502684,[214,220,228,235,242,250,255,260,265,270,275,280,285,290,316,321,326,331,336,375,380,385,390,395,400,405,410,415,420,425,430,435,440,446,451,456,461,466,471,476,481,486,491,496,501,506,511,516,521,526,531,536,541,570,575,580,585,592,597,602,607,612,617,622,627,632,637,642,647,652,657,662,699,704,709,745,750,755,760,766,771,776,781,786,791,796,801,806,811,837,843,848,879,884,920,925,963,968,973,978,983,988,993,998,1003,1008,1013,1018,1023,1032,1037,1045,1050,1055,1093,1098,1156,1161,1166,1171,1176,1181,1186,1196,1205,1214,1223,1232,1241],{"id":215,"data":216,"type":218,"tunes":219},"intro",{"text":217},"嵌入、向量数据库和重排序器是检索的三个不同组成部分。嵌入模型将文本或其他数据转换为数值表示；向量数据库或向量索引存储并搜索这些表示以检索候选项；重排序器接收一个较小的候选集，并使用更昂贵的相关性模型或评分方法对其重新排序。它们经常一起出现在RAG中，但它们都不是RAG本身，也不是每个检索系统都必须具备的。","paragraph",{},{"id":221,"data":222,"type":226,"tunes":227},"direct",{"body":223,"title":224,"variant":225},"\u003Cstrong>嵌入负责表示。向量搜索负责检索。重排序负责精炼。\u003C\u002Fstrong>\u003Cbr>\u003Cbr>一个有用的心智模型是：\u003Cbr>\u003Cstrong>内容 → 嵌入 → 候选检索 → 重排序 → 选定上下文 → 模型\u003C\u002Fstrong>。\u003Cbr>\u003Cbr>这些边界很重要，因为每一层的失败方式不同。糟糕的嵌入会扭曲语义相似度。薄弱的检索索引会遗漏有用的候选。重排序器可以重新排列候选，但无法恢复从未被检索到的相关文档。","直接回答","info","callout",{},{"id":229,"data":230,"type":226,"tunes":234},"boundary",{"body":231,"title":232,"variant":233},"向量数据库不是嵌入模型。嵌入不是搜索结果。重排序器不是向量数据库。RAG是更广泛的模式，可以在生成之前使用这些组件中的任何一个来检索外部信息。","不要混淆检索栈的各个层次","warning",{},{"id":236,"data":237,"type":226,"tunes":241},"current",{"body":238,"title":239,"variant":240},"尽管产品快速演进，基本架构是稳定的。当前的Qdrant文档将向量、负载元数据、集合和向量索引分开；当前的Elastic指南将语义重排序视为对小型候选集进行的后阶段操作；当前的Cohere文档同样将重排序描述为对词法或语义搜索的第二阶段改进。","当前来源说明 — 2026年10月8日","note",{},{"id":243,"data":244,"type":248,"tunes":249},"toc",{"title":245,"maxLevel":246,"minLevel":247},"目录",3,2,"tableOfContents",{},{"id":251,"data":252,"type":42,"tunes":254},"h-meaning",{"text":253,"level":247},"这真正意味着什么",{},{"id":256,"data":257,"type":218,"tunes":259},"p-meaning-1",{"text":258},"搜索系统有两个相互竞争的目标：找到足够多可能相关的材料，并将最好的材料放在靠近顶部的位置。快速的第一阶段检索通常优化候选生成。更强的第二阶段模型则可以花费更多计算来区分最佳候选。",{},{"id":261,"data":262,"type":218,"tunes":264},"p-meaning-2",{"text":263},"嵌入、向量索引和重排序器在该过程中占据不同的位置。将它们视为一个功能会掩盖关于召回率、精确率、延迟、存储、元数据过滤和模型成本的重要设计选择。",{},{"id":266,"data":267,"type":218,"tunes":269},"p-meaning-3",{"text":268},"这种区分还可以防止一个常见的RAG错误：假设将文档嵌入存储在向量数据库中就会自动创建高质量的检索。检索质量取决于嵌入模型、分块、元数据、查询构建、索引配置、候选数量、混合检索、重排序以及底层来源的权威性。",{},{"id":271,"data":272,"type":42,"tunes":274},"h-simple",{"text":273,"level":247},"最简单的例子",{},{"id":276,"data":277,"type":218,"tunes":279},"p-simple-1",{"text":278},"假设一个知识库包含100,000个文档块。用户问：“如何撤销API令牌？”",{},{"id":281,"data":282,"type":218,"tunes":284},"p-simple-2",{"text":283},"首先，嵌入模型可以将查询编码为向量。文档块可能已经存储了自己的嵌入。然后，向量搜索将查询向量与索引的文档向量进行比较，并返回例如30个可能的候选。",{},{"id":286,"data":287,"type":218,"tunes":289},"p-simple-3",{"text":288},"这30个候选随后可以传递给重排序器。重排序器更直接地将查询与每个候选进行比较，并产生新的相关性排序。应用程序可能会保留最好的五个用于模型上下文。",{},{"id":291,"data":292,"type":314,"tunes":315},"simple-flow",{"steps":293,"title":312,"orientation":313},[294,297,300,303,306,309],{"label":295,"description":296},"1. 嵌入文档","将每个可搜索的块转换为数值表示，通常在摄取时进行。",{"label":298,"description":299},"2. 存储\u002F索引向量","将向量与文档ID和元数据关联到可搜索的向量索引或数据库中。",{"label":301,"description":302},"3. 嵌入查询","使用兼容的嵌入模型和查询配置对用户的查询进行编码。",{"label":304,"description":305},"4. 检索候选","运行向量相似性搜索，通常带有元数据过滤器，以生成更大的top-k候选集。",{"label":307,"description":308},"5. 重排序候选","对查询和小型候选集应用更强的相关性模型。",{"label":310,"description":311},"6. 选择上下文","保留最有用的段落，用于下游答案、代理步骤或搜索结果。","一个基本的两阶段语义检索流程","auto","processFlow",{},{"id":317,"data":318,"type":42,"tunes":320},"h-stops",{"text":319,"level":247},"简单示例的局限",{},{"id":322,"data":323,"type":218,"tunes":325},"p-stops-1",{"text":324},"真实的检索系统根本不必使用密集嵌入。诸如BM25之类的关键词搜索可以作为第一阶段检索器。稀疏学习检索、SQL过滤器、图遍历或应用程序API也可以生成候选。",{},{"id":327,"data":328,"type":218,"tunes":330},"p-stops-2",{"text":329},"重排序器也不关心候选是否来自向量数据库。它可以对BM25结果、混合结果、手动选择的文档或来自多个检索器的候选进行重排序。",{},{"id":332,"data":333,"type":218,"tunes":335},"p-stops-3",{"text":334},"同样，嵌入不需要专门的向量数据库。小数据集可以在内存中或使用通用数据库和向量扩展进行比较。当索引、近似最近邻搜索、过滤、规模、更新行为或操作要求证明其合理性时，专门的向量系统才变得有用。",{},{"id":337,"data":338,"type":373,"tunes":374},"core-comparison",{"rows":339,"title":361,"layout":362,"columns":363},[340,345,349,353,357],{"id":341,"label":342,"values":343},"job","主要任务",[344,344,344],"",{"id":346,"label":347,"values":348},"input","典型输入",[344,344,344],{"id":350,"label":351,"values":352},"output","典型输出",[344,344,344],{"id":354,"label":355,"values":356},"cost","成本特征",[344,344,344],{"id":358,"label":359,"values":360},"can-miss","典型故障",[344,344,344],"三种不同的检索组件","table",[364,367,370],{"id":365,"label":366},"embedding","嵌入",{"id":368,"label":369},"vector","向量数据库 \u002F 索引",{"id":371,"label":372},"reranker","重排序器","comparison",{},{"id":376,"data":377,"type":42,"tunes":379},"h-embeddings",{"text":378,"level":247},"嵌入：表示，而非检索",{},{"id":381,"data":382,"type":218,"tunes":384},"p-emb-1",{"text":383},"嵌入是模型生成的数值表示。对于语义检索，含义相关的文本旨在在向量空间中占据有用的位置，以便相似度或距离函数可以比较它们。",{},{"id":386,"data":387,"type":218,"tunes":389},"p-emb-2",{"text":388},"Sentence-BERT 是使句子级语义相似度变得实用的重要一步，它采用了双编码器风格的表示，可以独立计算并高效比较。这一总体思路仍然是现代稠密检索的核心：预计算文档表示，在搜索时计算查询表示，然后进行比较。",{},{"id":391,"data":392,"type":218,"tunes":394},"p-emb-3",{"text":393},"嵌入本身并不搜索语料库。它是由嵌入模型生成的数据。当系统将查询表示与存储的候选进行比较时，检索才开始。",{},{"id":396,"data":397,"type":42,"tunes":399},"h-embedding-model",{"text":398,"level":246},"嵌入模型定义了表示空间",{},{"id":401,"data":402,"type":218,"tunes":404},"p-emodel-1",{"text":403},"文档和查询向量必须与创建它们时使用的模型和配置兼容。替换嵌入模型可能会改变维度、相似度行为、语言覆盖范围和领域性能。",{},{"id":406,"data":407,"type":218,"tunes":409},"p-emodel-2",{"text":408},"这就是为什么嵌入模型迁移不仅仅是更改 API 名称。现有文档可能需要重新嵌入，索引需要重建或版本化。",{},{"id":411,"data":412,"type":42,"tunes":414},"h-dense-sparse",{"text":413,"level":246},"稠密表示和稀疏表示是不同的",{},{"id":416,"data":417,"type":218,"tunes":419},"p-dense-1",{"text":418},"稠密嵌入通常包含许多非零维度，常用于语义相似度。稀疏表示包含许多零，可以保留更强的词元或词项结构。",{},{"id":421,"data":422,"type":218,"tunes":424},"p-dense-2",{"text":423},"两者都可以支持语义检索，现代搜索系统可以结合稠密、稀疏和词汇信号。因此，“向量搜索”并不总是意味着单一的稠密余弦相似度流水线。",{},{"id":426,"data":427,"type":42,"tunes":429},"h-distance",{"text":428,"level":246},"相似度函数是表示契约的一部分",{},{"id":431,"data":432,"type":218,"tunes":434},"p-distance-1",{"text":433},"余弦相似度、点积和欧几里得距离的含义并不相同。正确的度量取决于嵌入模型是如何训练和归一化的。",{},{"id":436,"data":437,"type":218,"tunes":439},"p-distance-2",{"text":438},"例如，当前的 Qdrant 文档要求将距离度量作为向量配置的一部分，并记录了余弦、点积和欧几里得风格的选择。重要的架构规则是将度量视为嵌入\u002F索引契约的一部分，而不是随意选择一种。",{},{"id":441,"data":442,"type":226,"tunes":445},"embedding-not-truth",{"body":443,"title":444,"variant":233},"两个段落可以在语义上接近，而其中一个是过时的、未授权的或错误的。嵌入估计的是表示相似度；它们不决定真源权威、新鲜度或证据有效性。","嵌入相似度不是事实支持",{},{"id":447,"data":448,"type":42,"tunes":450},"h-vector-db",{"text":449,"level":247},"向量数据库和索引：候选检索",{},{"id":452,"data":453,"type":218,"tunes":455},"p-vdb-1",{"text":454},"向量数据库或支持向量的搜索系统组织向量表示，以便应用程序能够高效地检索附近的候选。实际系统通常将向量与 ID 和有效负载元数据（如来源、语言、租户、文档类型、时间戳或访问范围）关联起来。",{},{"id":457,"data":458,"type":218,"tunes":460},"p-vdb-2",{"text":459},"例如，Qdrant 将数据组织为点的集合，其中点包含向量和可选的有效负载元数据。其文档将基于 HNSW 的相似度搜索和元数据过滤描述为检索层的独立能力。",{},{"id":462,"data":463,"type":218,"tunes":465},"p-vdb-3",{"text":464},"这种区分很重要：向量索引解决的是最近邻问题，而有效负载过滤器则强制执行结构性约束，例如租户、文档类别或语言。",{},{"id":467,"data":468,"type":42,"tunes":470},"h-ann",{"text":469,"level":246},"近似最近邻搜索以精确性换取效率",{},{"id":472,"data":473,"type":218,"tunes":475},"p-ann-1",{"text":474},"将单个查询向量与每个向量进行比较对于小型集合可能是可行的，但在大规模场景下成本高昂。诸如 HNSW 之类的近似最近邻索引通过遍历索引结构而不是穷举扫描每个向量来降低搜索成本。",{},{"id":477,"data":478,"type":218,"tunes":480},"p-ann-2",{"text":479},"近似搜索引入了召回率与延迟之间的权衡。更快的搜索可能会遗漏精确搜索本应返回的候选结果。因此，索引参数不仅影响基础设施性能，还会影响检索质量。",{},{"id":482,"data":483,"type":218,"tunes":485},"p-ann-3",{"text":484},"Qdrant 同时提供了与 HNSW 相关的参数和精确搜索选项，这表明向量存储和近似检索策略是相互独立的决策。",{},{"id":487,"data":488,"type":42,"tunes":490},"h-filtering",{"text":489,"level":246},"元数据过滤应在候选检索之前或期间进行",{},{"id":492,"data":493,"type":218,"tunes":495},"p-filter-1",{"text":494},"如果用户只能访问租户 A，那么从租户 B 检索语义相似的片段并试图稍后将其移除，是错误的安​​全边界。授权和硬性资格过滤器应在候选结果能够影响下游处理之前，对候选空间进行约束。",{},{"id":497,"data":498,"type":218,"tunes":500},"p-filter-2",{"text":499},"同样的原则适用于区域设置、文档状态、来源类别、日期、产品版本以及其他确定性约束。相似度应对符合条件的候选结果进行排序；它不应覆盖资格条件。",{},{"id":502,"data":503,"type":42,"tunes":505},"h-vector-not-required",{"text":504,"level":246},"向量数据库是可选的",{},{"id":507,"data":508,"type":218,"tunes":510},"p-optional-1",{"text":509},"对于小型语料库，暴力余弦比较可能简单且足够。具有向量支持的关系数据库也可能足够。当专用向量数据库的索引、过滤、分布式存储、更新行为或运维特性能够解决实际需求时，它才变得有价值。",{},{"id":512,"data":513,"type":218,"tunes":515},"p-optional-2",{"text":514},"因为“RAG 需要一个向量数据库”而选择向量数据库，颠倒了架构设计流程。应从检索需求和规模出发，然后选择存储\u002F索引技术。",{},{"id":517,"data":518,"type":42,"tunes":520},"h-rerank",{"text":519,"level":247},"重排序：第二阶段的相关性优化",{},{"id":522,"data":523,"type":218,"tunes":525},"p-rerank-1",{"text":524},"重排序器接收一个查询和一组较小的已检索候选结果，然后分配更强的相关性分数或新的排序。它通常比第一阶段检索的计算成本更高，这就是为什么它应用于候选生成之后，而不是针对整个语料库。",{},{"id":527,"data":528,"type":218,"tunes":530},"p-rerank-2",{"text":529},"当前 Elastic 的指南将语义重排序描述为针对小型 top-k 集合的最终阶段技术，并指出它可以优化词汇、语义或混合检索。Cohere 记录了相同的架构：先进行第一阶段的词汇或语义搜索，然后进行重排序阶段。",{},{"id":532,"data":533,"type":218,"tunes":535},"p-rerank-3",{"text":534},"一种常见的实现使用类似交叉编码器的模型，该模型同时检查查询和每个候选结果。这种更丰富的交互可以比独立的嵌入相似度更精确地区分相关性，但在语料库规模下成本要高得多。",{},{"id":537,"data":538,"type":42,"tunes":540},"h-bi-cross",{"text":539,"level":246},"双编码器检索和交叉编码器重排序解决不同的成本问题",{},{"id":542,"data":543,"type":362,"tunes":569},"encoder-table",{"content":544,"stretched":43,"withHeadings":14},[545,549,553,557,561,565],[546,547,548],"属性","双编码器 \u002F 嵌入检索","交叉编码器式重排序",[550,551,552],"编码方式","查询和文档独立表示","查询和候选结果联合处理",[554,555,556],"文档计算","可在摄取时预先计算","通常针对每个查询-候选对重新计算",[558,559,560],"语料库规模搜索","适合使用向量索引","在整个语料库上通常成本过高",[562,563,564],"典型角色","高召回率的候选生成","对小型候选集进行高精度排序",[566,567,568],"主要权衡","快速且可扩展，但相关性交互被压缩到向量中","更丰富的相关性判断，但延迟\u002F成本更高",{},{"id":571,"data":572,"type":42,"tunes":574},"h-rerank-limit",{"text":573,"level":246},"重排序器无法恢复检索遗漏的内容",{},{"id":576,"data":577,"type":218,"tunes":579},"p-rerank-limit-1",{"text":578},"如果相关文档不在候选集中，重排序就没有任何可以提升的对象。这就是需要分别评估检索和重排序的核心原因。",{},{"id":581,"data":582,"type":218,"tunes":584},"p-rerank-limit-2",{"text":583},"一个流水线可能具有出色的重排序器精确率，但仍然会因为第一阶段召回率较差而失败。提高重排序器质量无法修复缺失的源覆盖、糟糕的分块、限制性过滤器或较弱的候选检索器。",{},{"id":586,"data":587,"type":226,"tunes":591},"recall-precision",{"body":588,"title":589,"variant":590},"第一阶段：\u003Cstrong>不要漏掉有用的候选。\u003C\u002Fstrong>\u003Cbr>第二阶段：\u003Cstrong>把最好的候选放在前面。\u003C\u002Fstrong>\u003Cbr>\u003Cbr>这不是一条通用的数学规则，但它是两阶段检索中一个有用的工程模型。","有用的检索目标","success",{},{"id":593,"data":594,"type":42,"tunes":596},"h-hybrid",{"text":595,"level":247},"混合检索是一个独立的设计选择",{},{"id":598,"data":599,"type":218,"tunes":601},"p-hybrid-1",{"text":600},"当查询和文档使用不同措辞但表达相关含义时，稠密语义检索表现很强。当精确术语、标识符、名称、代码或罕见短语很重要时，词法检索表现很强。",{},{"id":603,"data":604,"type":218,"tunes":606},"p-hybrid-2",{"text":605},"混合检索结合多种候选信号，通常是词法 BM25 和向量相似度，然后使用诸如倒数排名融合或加权分数组合之类的方法合并排名。",{},{"id":608,"data":609,"type":218,"tunes":611},"p-hybrid-3",{"text":610},"随后，重排序可以在融合后的候选集上运行。因此，混合检索和重排序是互补但不同的阶段。",{},{"id":613,"data":614,"type":42,"tunes":616},"h-bm25",{"text":615,"level":246},"BM25 并不会因为嵌入的存在而过时",{},{"id":618,"data":619,"type":218,"tunes":621},"p-bm25-1",{"text":620},"对于精确标识符、版本号、错误消息、产品代码和专业词汇，关键词搜索可能优于稠密检索。例如，SQLite FTS5 包含用于全文搜索的 BM25 排名函数。",{},{"id":623,"data":624,"type":218,"tunes":626},"p-bm25-2",{"text":625},"一个强大的检索架构可以将词法检索作为唯一的第一阶段，将向量检索作为唯一的第一阶段，或者根据语料库和查询分布将两者结合起来。",{},{"id":628,"data":629,"type":42,"tunes":631},"h-chunking",{"text":630,"level":247},"分块会改变嵌入和重排序器能够看到的内容",{},{"id":633,"data":634,"type":218,"tunes":636},"p-chunk-1",{"text":635},"如果文档被糟糕地切分，后续任何检索组件都无法完全重建缺失的语义单元。一个将条件与其例外切开的块，可能会以误导性的方式嵌入，也可能因为候选文本不完整而被错误地重排序。",{},{"id":638,"data":639,"type":218,"tunes":641},"p-chunk-2",{"text":640},"因此，块大小、重叠、结构边界和元数据都会影响候选召回和重排序器判断。检索评估应测试从摄取到排名的完整流水线，而不仅仅是嵌入模型。",{},{"id":643,"data":644,"type":42,"tunes":646},"h-scores",{"text":645,"level":247},"不要像比较通用概率那样比较检索分数",{},{"id":648,"data":649,"type":218,"tunes":651},"p-scores-1",{"text":650},"余弦相似度、BM25 分数、稀疏向量分数、RRF 排名和重排序器分数具有不同含义。来自一个嵌入模型的 0.82 分，并不自动与来自另一个模型的 0.82 分或与重排序器分数具有可比性。",{},{"id":653,"data":654,"type":218,"tunes":656},"p-scores-2",{"text":655},"阈值应针对实际模型、语料库和任务进行校准。当前 Elastic 指南还指出，嵌入相似度分数可能依赖于查询，这使得通用截断值存在风险。",{},{"id":658,"data":659,"type":42,"tunes":661},"h-eval",{"text":660,"level":247},"分别评估检索阶段",{},{"id":663,"data":664,"type":362,"tunes":698},"eval-table",{"content":665,"stretched":43,"withHeadings":14},[666,670,674,678,682,686,690,694],[667,668,669],"层级","有用的问题","示例指标或测试",[671,672,673],"源覆盖","语料库是否包含所需信息？","覆盖审计 \u002F 已知答案源集合",[675,676,677],"分块","所需证据是否能作为连贯单元被检索到？","块级支持审查",[679,680,681],"第一阶段检索","相关条目是否进入候选集？","Recall@k",[683,684,685],"排名","相关证据出现得有多靠前？","MRR、nDCG、precision@k",[687,688,689],"重排序","第二阶段评分是否改善排序？","Delta nDCG \u002F MRR \u002F precision",[691,692,693],"上下文选择","最终选定的段落是否包含足够支持？","上下文相关性 \u002F 覆盖",[695,696,697],"答案阶段","模型是否正确使用选定的证据？","忠实性 \u002F 主张-证据评估",{},{"id":700,"data":701,"type":218,"tunes":703},"p-eval-1",{"text":702},"这种分离在操作上很重要。如果 Recall@50 很差，重排序器并不是第一个需要修复的组件。如果 Recall@50 很强，但最佳段落仍排在第 38 位，那么重排序或排名融合就成为一个可能的目标。",{},{"id":705,"data":706,"type":42,"tunes":708},"h-failure-map",{"text":707,"level":247},"究竟是哪一层失败了？",{},{"id":710,"data":711,"type":373,"tunes":744},"failure-comparison",{"rows":712,"title":733,"layout":362,"columns":734},[713,717,721,725,729],{"id":714,"label":715,"values":716},"missed","相关文档从未出现",[344,344,344],{"id":718,"label":719,"values":720},"lowrank","相关文档排名过低",[344,344,344],{"id":722,"label":723,"values":724},"wrongtenant","语义上良好但被禁止的结果",[344,344,344],{"id":726,"label":727,"values":728},"stale","相关但过时的结果",[344,344,344],{"id":730,"label":731,"values":732},"context","检索到正确结果但被提示词遗漏",[344,344,344],"症状与可能的检索层",[735,738,741],{"id":736,"label":737},"symptom","观察到的症状",{"id":739,"label":740},"likely","可能的层",{"id":742,"label":743},"test","首要诊断",{},{"id":746,"data":747,"type":42,"tunes":749},"h-authority",{"text":748,"level":247},"相关性与事实来源是不同的",{},{"id":751,"data":752,"type":218,"tunes":754},"p-authority-1",{"text":753},"重排序器可以让一份过时的文档看起来极其相关。向量索引可以检索到语义上比原始来源更接近的次级摘要。因此，检索质量不能替代权威规则。",{},{"id":756,"data":757,"type":218,"tunes":759},"p-authority-2",{"text":758},"在来源权威性重要的地方，元数据过滤器、来源类别、版本规则和出处应在结果成为模型上下文之前约束检索。",{},{"id":761,"data":762,"type":226,"tunes":765},"authority-callout",{"body":763,"title":764,"variant":233},"相关性回答的是候选结果是否匹配查询。事实来源架构回答的是该候选结果是否被允许确立该主张。","重排序无法使非权威来源变得权威",{},{"id":767,"data":768,"type":42,"tunes":770},"h-impl",{"text":769,"level":247},"原始实现证据",{},{"id":772,"data":773,"type":42,"tunes":775},"h-sot-engine",{"text":774,"level":246},"事实来源研究引擎：词法检索与语义检索是分离的",{},{"id":777,"data":778,"type":218,"tunes":780},"p-sot-1",{"text":779},"事实来源研究引擎包含一条使用 SQLite FTS5\u002FBM25 的本地词法检索路径，以及一条使用本地生成嵌入的独立可选语义检索路径。",{},{"id":782,"data":783,"type":218,"tunes":785},"p-sot-2",{"text":784},"其语义搜索实现会计算查询向量，并使用余弦相似度将其与存储的块向量进行比较。该项目有意将语义相似度视为发现信号而非证据：候选结果在支持某项主张之前，仍必须追溯到具体的来源和定位符。",{},{"id":787,"data":788,"type":218,"tunes":790},"p-sot-3",{"text":789},"这对 R01 来说是有用的实现证据，因为同一语料库可以同时支持词法排名和向量相似度，而不会将任一机制与证据权威性相混淆。",{},{"id":792,"data":793,"type":42,"tunes":795},"h-client",{"text":794,"level":246},"Aaasaasa AI 客户端：Qdrant 是向量基础设施组件",{},{"id":797,"data":798,"type":218,"tunes":800},"p-client-1",{"text":799},"Aaasaasa AI 客户端将 Qdrant\u002F向量基础设施作为独立的本地资源。Electron 架构从受信任的主进程侧暴露 Qdrant 服务，而不是将向量搜索视为模型本身的一部分。",{},{"id":802,"data":803,"type":218,"tunes":805},"p-client-2",{"text":804},"该仓库包含 Qdrant 客户端适配器、Qdrant 服务配置以及基于 Docker 的 Qdrant 基础设施。这展示了 AI 提供商\u002F模型执行与向量存储\u002F搜索之间的架构分离。",{},{"id":807,"data":808,"type":218,"tunes":810},"p-client-3",{"text":809},"不应将 Qdrant 支持的存在夸大为完整的生产级 RAG 流水线。这里的证据更为有限：向量基础设施是作为其自身的组件边界实现的。",{},{"id":812,"data":813,"type":362,"tunes":836},"impl-table",{"content":814,"stretched":43,"withHeadings":14},[815,818,821,824,827,830,833],[816,817],"实现证据","它展示了什么",[819,820],"事实来源研究引擎中的 SQLite FTS5\u002FBM25","词法检索可以独立于嵌入存在。",[822,823],"本地 Ollama 嵌入","表示生成是其自身的阶段。",[825,826],"存储的语义向量 + 余弦比较","语义检索在嵌入生成之后消费它们。",[828,829],"Aaasaasa AI 客户端中的 Qdrant 支持","向量存储\u002F搜索是一种独立于模型提供商的基础设施能力。",[831,832],"事实来源研究引擎中的证据\u002F出处规则","检索到的相似度不等于权威性或证明。",[834,835],"这些实现中没有声称的自定义重排序器","重排序被解释为一个架构阶段，而不是被虚假声称已经实现的证据。",{},{"id":838,"data":839,"type":226,"tunes":842},"impl-discipline",{"body":840,"title":841,"variant":240},"当前的实现证据确认了词法检索、嵌入、向量搜索基础设施和出处感知检索。本文\u003Cstrong>并不\u003C\u002Fstrong>声称这些项目中已经实现了生产级交叉编码器重排序服务。","证据边界",{},{"id":844,"data":845,"type":42,"tunes":847},"h-decisions",{"text":846,"level":247},"何时需要每个组件？",{},{"id":849,"data":850,"type":362,"tunes":878},"decision-table",{"content":851,"stretched":43,"withHeadings":14},[852,855,858,861,864,867,869,872,875],[853,854],"需求","可能的组件",[856,857],"不同措辞之间的语义相似性","嵌入模型 + 向量相似性搜索",[859,860],"在大型向量语料库上进行高效搜索","向量索引\u002F数据库或支持向量的搜索引擎",[862,863],"精确标识符、错误代码或罕见术语","词汇\u002F全文检索，如 BM25",[865,866],"既需要精确术语又需要语义含义","混合词汇 + 语义检索",[868,372],"候选集良好但排序较弱",[870,871],"相关项不在候选集中","在重排序之前改进源覆盖、分块、检索器、过滤器或候选数量",[873,874],"硬性租户\u002F来源\u002F版本约束","确定性元数据\u002F授权过滤",[876,877],"小型语料库","可能使用简单的暴力相似性计算或通用数据库，而非专用向量数据库",{},{"id":880,"data":881,"type":42,"tunes":883},"h-sequence",{"text":882,"level":247},"实用的检索设计流程",{},{"id":885,"data":886,"type":314,"tunes":919},"design-flow",{"steps":887,"title":918,"orientation":313},[888,891,894,897,900,903,906,909,912,915],{"label":889,"description":890},"1. 定义查询类型","识别语义问题、精确查找、标识符、当前状态读取和领域特定模式。",{"label":892,"description":893},"2. 定义合格来源","应用租户、授权、区域设置、版本、来源类别和新鲜度约束。",{"label":895,"description":896},"3. 建立词汇基线","衡量简单的全文\u002FBM25 检索是否已经解决了大部分工作负载。",{"label":898,"description":899},"4. 在需要语义召回的地方添加嵌入","针对代表性领域查询选择并评估嵌入模型。",{"label":901,"description":902},"5. 根据规模选择向量存储\u002F索引","根据需求使用暴力搜索、数据库向量支持或专用向量引擎。",{"label":904,"description":905},"6. 评估第一阶段召回","确认相关证据进入足够大的候选集。",{"label":907,"description":908},"7. 如果信号互补则添加混合检索","当词汇和语义排名都能实质性改善候选生成时，融合两者。",{"label":910,"description":911},"8. 如果排序仍是瓶颈则添加重排序","仅对候选集应用更强的模型，以证明其成本合理。",{"label":913,"description":914},"9. 调整最终上下文选择","在生成之前控制冗余、上下文预算、权威性、多样性和证据覆盖。",{"label":916,"description":917},"10. 端到端评估","分别衡量检索、上下文和答案质量，以便定位失败环节。","从需求出发设计检索，而非从产品名称出发",{},{"id":921,"data":922,"type":42,"tunes":924},"h-misconceptions",{"text":923,"level":247},"常见误解",{},{"id":926,"data":927,"type":362,"tunes":962},"misconceptions-table",{"content":928,"stretched":43,"withHeadings":14},[929,932,935,938,941,944,947,950,953,956,959],[930,931],"误解","纠正",[933,934],"“嵌入就是向量数据库。”","嵌入是一种表示；数据库\u002F索引存储并搜索表示。",[936,937],"“向量数据库创造语义含义。”","嵌入模型创建表示；向量系统对其进行索引和比较。",[939,940],"“RAG 需要向量数据库。”","RAG 需要检索，而不是特定的检索技术。",[942,943],"“重排序与向量搜索相同。”","向量搜索生成候选；重排序对候选集重新排序。",[945,946],"“重排序器能修复糟糕的召回。”","它们无法提升从未被检索到的文档。",[948,949],"“密集搜索取代 BM25。”","词汇搜索对于精确术语、标识符和专业词汇仍然有价值。",[951,952],"“相似度越高意味着越权威。”","相似度和来源权威性是不同的维度。",[954,955],"“更大的 top-k 总能改善 RAG。”","更大的候选集可以提高召回，但会增加延迟、噪声和上下文选择负担。",[957,958],"“一个分数阈值适用于所有情况。”","分数取决于模型、查询、语料库和检索方法，必须进行校准。",[960,961],"“专用向量数据库总是更先进。”","只有当其操作和检索能力符合需求时，才值得使用。",{},{"id":964,"data":965,"type":42,"tunes":967},"h-edge",{"text":966,"level":247},"边缘情况和限制",{},{"id":969,"data":970,"type":218,"tunes":972},"p-edge-1",{"text":971},"有些应用不需要语义搜索。精确的数据库查找或结构化 SQL 可能比嵌入检索更正确、更快速且更易于审计。",{},{"id":974,"data":975,"type":218,"tunes":977},"p-edge-2",{"text":976},"有些语料库非常小，全向量扫描是可以接受的。近似索引增加了复杂性却没有有意义的收益。",{},{"id":979,"data":980,"type":218,"tunes":982},"p-edge-3",{"text":981},"有些查询在任何精度优化之前就需要高召回。法律发现、研究和合规审查可能更倾向于广泛的候选检索，然后进行透明的过滤和人工审查。",{},{"id":984,"data":985,"type":218,"tunes":987},"p-edge-4",{"text":986},"多语言和领域特定的检索在不同嵌入模型上的表现可能差异很大。来自公共数据集的基准声明不应被视为对私有语料库的证明。",{},{"id":989,"data":990,"type":218,"tunes":992},"p-edge-5",{"text":991},"重排序延迟随候选数量和长度的增加而增长。因此，候选大小应作为准确性\u002F成本\u002F延迟变量进行调整，而不是从教程中复制。",{},{"id":994,"data":995,"type":42,"tunes":997},"h-change",{"text":996,"level":247},"什么会改变这个答案？",{},{"id":999,"data":1000,"type":218,"tunes":1002},"p-change-1",{"text":1001},"如果供应商将嵌入生成、向量索引和重排序打包在一个 API 后面，组件边界不会改变。产品可能隐藏这些阶段，但它们在概念上仍然是不同的职责，具有不同的故障模式。",{},{"id":1004,"data":1005,"type":218,"tunes":1007},"p-change-2",{"text":1006},"未来的嵌入或检索模型可能会在某些工作负载中减少对单独重排序的需求，而更强的后期交互或学习稀疏方法可能会模糊传统的密集\u002F词汇类别。架构仍然应该问哪个阶段产生表示、哪个阶段生成候选以及哪个阶段细化排名。",{},{"id":1009,"data":1010,"type":218,"tunes":1012},"p-change-3",{"text":1011},"最佳设计还会随语料库大小、查询组合、语言、领域术语、更新频率、来源权威性、延迟预算和评估结果而变化。",{},{"id":1014,"data":1015,"type":42,"tunes":1017},"h-related",{"text":1016,"level":247},"相关规范知识",{},{"id":1019,"data":1020,"type":218,"tunes":1022},"p-related-1",{"text":1021},"R01 假设基本的 RAG 概念已经被理解。RAG 是一种更广泛的模式，其中检索到的外部信息被提供给模型；嵌入、向量搜索和重排序是该模式中的可选检索组件。",{},{"id":1024,"data":1025,"type":1030,"tunes":1031},"ref-rag",{"url":1026,"title":1027,"excerpt":1028,"ctaLabel":1029},"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works","什么是 RAG？对其工作原理的最简单解释","用通俗易懂的语言介绍检索如何将外部知识带入模型上下文的基础。","阅读 RAG 基础","referralArticle",{},{"id":1033,"data":1034,"type":218,"tunes":1036},"p-related-2",{"text":1035},"当检索失败时，应分别诊断来源覆盖、检索、排序、上下文组装和生成，而不是将整个系统视为一次“RAG 失败”。",{},{"id":1038,"data":1039,"type":1030,"tunes":1044},"ref-rag-failed",{"url":1040,"title":1041,"excerpt":1042,"ctaLabel":1043},"https:\u002F\u002Fstajic.de\u002Fzh\u002Fblog\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method","RAG 失败了——但究竟是哪一层失败了？一种诊断方法","一种逐层隔离来源覆盖、检索、排序、上下文组装、生成、证据归因和时效性失败的方法。","阅读 RAG 诊断方法",{},{"id":1046,"data":1047,"type":218,"tunes":1049},"p-related-3",{"text":1048},"真相来源架构是检索周围的权威层：它决定哪个来源可以确立一项主张，而嵌入和排序只决定哪些候选看起来相关。",{},{"id":1051,"data":1052,"type":42,"tunes":1054},"h-faq",{"text":1053,"level":247},"常见问题",{},{"id":1056,"data":1057,"type":1056,"tunes":1092},"faq",{"items":1058,"title":1091},[1059,1063,1067,1071,1075,1079,1083,1087],{"id":1060,"answer":1061,"question":1062},"faq1","嵌入是由模型生成的数值表示。向量数据库或向量索引将这些表示与 ID 和元数据一起存储和搜索。","嵌入和向量数据库有什么区别？",{"id":1064,"answer":1065,"question":1066},"faq2","重排序器接收已经检索到的候选集，并使用更强的相关性模型或评分方法对这些候选重新评分或重新排序。","重排序器做什么？",{"id":1068,"answer":1069,"question":1070},"faq3","不需要。RAG 需要检索外部信息。检索可以使用词法搜索、SQL、API、图、向量搜索、混合搜索或这些的组合。","RAG 需要向量数据库吗？",{"id":1072,"answer":1073,"question":1074},"faq4","重排序器通常执行更昂贵的查询-文档交互，因此通常在更快的首阶段检索器之后应用于较小的 top-k 候选集。","为什么不对整个语料库使用重排序器？",{"id":1076,"answer":1077,"question":1078},"faq5","不能。如果相关文档没有被检索到候选集中，重排序就没有任何东西可以提升。","重排序能修复缺失的文档吗？",{"id":1080,"answer":1081,"question":1082},"faq6","不是。它是一种相似度度量，其数值含义取决于嵌入模型和语料库。不应将其视为普遍的相关性概率。","余弦相似度是相关性概率吗？",{"id":1084,"answer":1085,"question":1086},"faq7","当评估显示词法和语义信号能够恢复互补的相关文档时，使用混合检索。它并非对每个语料库都自动更好。","我应该同时使用 BM25 和向量搜索吗？",{"id":1088,"answer":1089,"question":1090},"faq8","当向量索引、过滤、规模、更新、分布式操作或其他向量特定需求足以证明需要专用系统时。小型工作负载可能不需要。","什么时候需要专用的向量数据库？","嵌入、向量数据库和重排序",{},{"id":1094,"data":1095,"type":42,"tunes":1097},"h-glossary",{"text":1096,"level":247},"术语表",{},{"id":1099,"data":1100,"type":1099,"tunes":1155},"glossary",{"title":1101,"entries":1102},"关键检索术语",[1103,1105,1109,1113,1117,1121,1125,1129,1133,1137,1140,1144,1148,1151],{"term":366,"anchor":365,"definition":1104},"由嵌入模型生成的内容数值表示，用于相似度、聚类、检索或相关任务。",{"term":1106,"anchor":1107,"definition":1108},"稠密向量","dense-vector","一种向量表示，其中许多维度携带非零值，常用于语义检索。",{"term":1110,"anchor":1111,"definition":1112},"稀疏向量","sparse-vector","一种高维表示，其中大多数维度为零，通常保留更强的词元或词项结构。",{"term":1114,"anchor":1115,"definition":1116},"向量索引","vector-index","一种数据结构，用于组织向量以实现高效的相似度或最近邻检索。",{"term":1118,"anchor":1119,"definition":1120},"向量数据库","vector-database","一种存储\u002F搜索系统，旨在管理向量、相关元数据和向量检索工作负载。",{"term":1122,"anchor":1123,"definition":1124},"ANN","ann","近似最近邻搜索，以精确穷举比较换取大规模下更快的检索。",{"term":1126,"anchor":1127,"definition":1128},"HNSW","hnsw","分层可导航小世界，一种基于图的近似最近邻索引方法，广泛用于向量检索。",{"term":1130,"anchor":1131,"definition":1132},"BM25","bm25","一种基于词项出现和语料库统计的词法相关性排序方法，广泛用于全文搜索。",{"term":1134,"anchor":1135,"definition":1136},"混合搜索","hybrid-search","结合多种检索方法（如词法搜索和向量搜索）的结果或分数的检索。",{"term":687,"anchor":1138,"definition":1139},"reranking","检索的后续阶段，对已生成的候选集重新评分和重新排序。",{"term":1141,"anchor":1142,"definition":1143},"双编码器","bi-encoder","一种独立编码查询和候选的架构，支持预计算和可扩展的相似度搜索。",{"term":1145,"anchor":1146,"definition":1147},"交叉编码器","cross-encoder","一种联合处理查询和候选文本的模型，通常以更高的计算成本提高相关性判断。",{"term":681,"anchor":1149,"definition":1150},"recall-at-k","在前 k 个检索到的候选中恢复的相关项比例。",{"term":1152,"anchor":1153,"definition":1154},"nDCG","ndcg","归一化折损累计增益，一种排序指标，奖励在有序列表中位置更高的相关结果。",{},{"id":1157,"data":1158,"type":42,"tunes":1160},"h-conclusion",{"text":1159,"level":247},"结论",{},{"id":1162,"data":1163,"type":218,"tunes":1165},"p-conclusion-1",{"text":1164},"清晰的检索模型很简单：嵌入表示含义，向量搜索检索候选，重排序器优化候选排序。",{},{"id":1167,"data":1168,"type":218,"tunes":1170},"p-conclusion-2",{"text":1169},"一旦这些边界明确，架构决策就更容易诊断。缺失候选指向来源覆盖、分块、嵌入、过滤器或首阶段检索。排序不佳指向排序、融合或重排序。然后可以在上下文和生成层分别调查不正确的最终答案。",{},{"id":1172,"data":1173,"type":218,"tunes":1175},"p-conclusion-3",{"text":1174},"最重要的结果不是选择最时髦的检索组件。而是构建一个检索管道，其阶段、权威边界、指标和失败模式可以独立测量。",{},{"id":1177,"data":1178,"type":42,"tunes":1180},"h-sources",{"text":1179,"level":247},"主要来源和实现证据",{},{"id":1182,"data":1183,"type":218,"tunes":1185},"p-sources-note",{"text":1184},"以下外部参考文献记录了本文中使用的表示、向量搜索和重排序机制。项目特定部分是原创实现证据，有意比关于完整生产 RAG 成熟度的声明更窄。",{},{"id":1187,"data":1188,"type":1194,"tunes":1195},"src-sbert",{"link":1189,"meta":1190},"https:\u002F\u002Farxiv.org\u002Fabs\u002F1908.10084",{"image":1191,"title":1192,"description":1193},{"url":344},"Sentence-BERT：使用孪生 BERT 网络的句子嵌入","基础论文，展示了可独立计算的句子嵌入，用于高效的语义相似度搜索。","linkTool",{},{"id":1197,"data":1198,"type":1194,"tunes":1204},"src-qdrant-overview",{"link":1199,"meta":1200},"https:\u002F\u002Fqdrant.tech\u002Fdocumentation\u002Foverview\u002F",{"image":1201,"title":1202,"description":1203},{"url":344},"Qdrant — 架构和数据结构概述","官方文档，描述集合、点、向量、有效负载元数据和基于 HNSW 的相似度索引。",{},{"id":1206,"data":1207,"type":1194,"tunes":1213},"src-qdrant-search",{"link":1208,"meta":1209},"https:\u002F\u002Fqdrant.tech\u002Fdocumentation\u002Fsearch\u002Fsearch\u002F",{"image":1210,"title":1211,"description":1212},{"url":344},"Qdrant — 搜索","官方向量搜索文档，涵盖相似度查询、过滤、精确与近似搜索以及稠密\u002F稀疏行为。",{},{"id":1215,"data":1216,"type":1194,"tunes":1222},"src-elastic-vector",{"link":1217,"meta":1218},"https:\u002F\u002Fwww.elastic.co\u002Fdocs\u002Fsolutions\u002Fsearch\u002Fvector",{"image":1219,"title":1220,"description":1221},{"url":344},"Elastic — 向量搜索","关于稠密\u002F稀疏向量检索、词法\u002F向量组合和多阶段搜索管道的最新文档。",{},{"id":1224,"data":1225,"type":1194,"tunes":1231},"src-elastic-rerank",{"link":1226,"meta":1227},"https:\u002F\u002Fwww.elastic.co\u002Fdocs\u002Fsolutions\u002Fsearch\u002Franking\u002Fsemantic-reranking",{"image":1228,"title":1229,"description":1230},{"url":344},"Elastic — 语义重排序","当前指南将语义重排序定义为对较小候选集进行的后期相关性操作。",{},{"id":1233,"data":1234,"type":1194,"tunes":1240},"src-cohere-rerank",{"link":1235,"meta":1236},"https:\u002F\u002Fdocs.cohere.com\u002Fdocs\u002Freranking-with-cohere",{"image":1237,"title":1238,"description":1239},{"url":344},"Cohere — 使用 Cohere 进行重排序","当前文档展示重排序作为对词法或语义第一阶段检索的第二阶段改进。",{},{"id":1242,"data":1243,"type":1194,"tunes":1249},"src-sqlite-fts5",{"link":1244,"meta":1245},"https:\u002F\u002Fwww.sqlite.org\u002Ffts5.html",{"image":1246,"title":1247,"description":1248},{"url":344},"SQLite FTS5","SQLite 官方文档，介绍全文搜索及用作词法检索证据的内置 BM25 排序函数。",{},"2.31","嵌入表示含义，向量数据库检索候选结果，重排序器则精炼结果。了解这三个检索层在RAG中如何不同并协同工作。","\u002Fuploads\u002F2026\u002F10\u002Fvector-databases-embeddings-and-reranking-three-different-parts-of-retrieval-1791480129884-9dtasz.webp","vector-databases-embeddings-and-reranking-three-different-parts-of-retrieval-1791480129884-9dtasz","PUBLISHED","2026-10-08T11:21:00.000Z","2026-10-08T17:21:30.174Z","2026-10-08T20:06:31.300Z",{"en":1259,"de":1260,"sr":1261,"es":1262,"fr":1263,"it":1264,"ru":1265,"zh":1266},"\u002Fblog\u002Fvector-databases-embeddings-and-reranking-three-different-parts-of-retrieval","\u002Fde\u002Fblog\u002Fvector-databases-embeddings-and-reranking-three-different-parts-of-retrieval","\u002Fsr\u002Fblog\u002Fvector-databases-embeddings-and-reranking-three-different-parts-of-retrieval","\u002Fes\u002Fblog\u002Fvector-databases-embeddings-and-reranking-three-different-parts-of-retrieval","\u002Ffr\u002Fblog\u002Fvector-databases-embeddings-and-reranking-three-different-parts-of-retrieval","\u002Fit\u002Fblog\u002Fvector-databases-embeddings-and-reranking-three-different-parts-of-retrieval","\u002Fru\u002Fblog\u002Fvector-databases-embeddings-and-reranking-three-different-parts-of-retrieval","\u002Fzh\u002Fblog\u002Fvector-databases-embeddings-and-reranking-three-different-parts-of-retrieval",[1268,1272,1276],{"id":1269,"name":1270,"slug":1271},64,"信息架构","information-architecture",{"id":1273,"name":1274,"slug":1275},60,"成本与延迟控制","cost-and-latency",{"id":1277,"name":1278,"slug":1279},57,"数据边界","data-boundaries",{"id":1281,"login":1282,"email":1283,"displayName":1284},"20","rooth8233","aleksandar@stajic.de","Aleksandar Stajić",[1286,2135],{"lang":1287,"title":1288,"content":1289,"contentJson":1290,"excerpt":2134},"en","Vector Databases, Embeddings and Reranking: Three Different Parts of Retrieval","{\"time\":1791489989811,\"blocks\":[{\"id\":\"intro\",\"type\":\"paragraph\",\"data\":{\"text\":\"Embeddings, vector databases and rerankers are three different parts of retrieval. An embedding model converts text or other data into numerical representations; a vector database or vector index stores and searches those representations to retrieve candidate items; a reranker takes a smaller candidate set and reorders it using a more expensive relevance model or scoring method. They often appear together in RAG, but none of them is the same thing as RAG, and none is mandatory in every retrieval system.\"},\"tunes\":{}},{\"id\":\"direct\",\"type\":\"callout\",\"data\":{\"variant\":\"info\",\"title\":\"Direct answer\",\"body\":\"\u003Cstrong>Embeddings represent. Vector search retrieves. Reranking refines.\u003C\u002Fstrong>\u003Cbr>\u003Cbr>A useful mental model is:\u003Cbr>\u003Cstrong>content → embedding → candidate retrieval → reranking → selected context → model\u003C\u002Fstrong>.\u003Cbr>\u003Cbr>The boundaries matter because each layer fails differently. Bad embeddings distort semantic similarity. A weak retrieval index misses useful candidates. A reranker can reorder candidates, but it cannot recover a relevant document that was never retrieved.\"},\"tunes\":{}},{\"id\":\"boundary\",\"type\":\"callout\",\"data\":{\"variant\":\"warning\",\"title\":\"Do not collapse the retrieval stack\",\"body\":\"A vector database is not an embedding model. An embedding is not a search result. A reranker is not a vector database. RAG is the wider pattern that can use any of these components to retrieve external information before generation.\"},\"tunes\":{}},{\"id\":\"current\",\"type\":\"callout\",\"data\":{\"variant\":\"note\",\"title\":\"Current-source note — 8 October 2026\",\"body\":\"The basic architecture is stable even though products evolve rapidly. Current Qdrant documentation separates vectors, payload metadata, collections and vector indexes; current Elastic guidance treats semantic reranking as a later-stage operation over a small candidate set; current Cohere documentation likewise describes reranking as a second-stage improvement over lexical or semantic search.\"},\"tunes\":{}},{\"id\":\"toc\",\"type\":\"tableOfContents\",\"data\":{\"title\":\"Contents\",\"minLevel\":2,\"maxLevel\":3},\"tunes\":{}},{\"id\":\"h-meaning\",\"type\":\"header\",\"data\":{\"text\":\"What this really means\",\"level\":2},\"tunes\":{}},{\"id\":\"p-meaning-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Search systems have two competing goals: find enough potentially relevant material and put the best material near the top. Fast first-stage retrieval usually optimizes candidate generation. A stronger second-stage model can then spend more computation distinguishing the best candidates.\"},\"tunes\":{}},{\"id\":\"p-meaning-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Embeddings, vector indexes and rerankers occupy different positions in that process. Treating them as one feature hides important design choices about recall, precision, latency, storage, metadata filtering and model cost.\"},\"tunes\":{}},{\"id\":\"p-meaning-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The distinction also prevents a common RAG mistake: assuming that storing document embeddings in a vector database automatically creates high-quality retrieval. Retrieval quality depends on the embedding model, chunking, metadata, query construction, index configuration, candidate count, hybrid retrieval, reranking and the authority of the underlying sources.\"},\"tunes\":{}},{\"id\":\"h-simple\",\"type\":\"header\",\"data\":{\"text\":\"The simplest example\",\"level\":2},\"tunes\":{}},{\"id\":\"p-simple-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Suppose a knowledge base contains 100,000 document chunks. A user asks: “How do I revoke an API token?”\"},\"tunes\":{}},{\"id\":\"p-simple-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"First, an embedding model can encode the query into a vector. Document chunks may already have their own stored embeddings. A vector search then compares the query vector to the indexed document vectors and returns, for example, 30 likely candidates.\"},\"tunes\":{}},{\"id\":\"p-simple-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"Those 30 candidates can then be passed to a reranker. The reranker compares the query more directly with each candidate and produces a new relevance ordering. The application might keep the best five for the model context.\"},\"tunes\":{}},{\"id\":\"simple-flow\",\"type\":\"processFlow\",\"data\":{\"title\":\"A basic two-stage semantic retrieval pipeline\",\"orientation\":\"auto\",\"steps\":[{\"label\":\"1. Embed documents\",\"description\":\"Convert each searchable chunk into a numerical representation, usually at ingest time.\"},{\"label\":\"2. Store\u002Findex vectors\",\"description\":\"Associate vectors with document IDs and metadata in a searchable vector index or database.\"},{\"label\":\"3. Embed the query\",\"description\":\"Encode the user's query using the compatible embedding model and query configuration.\"},{\"label\":\"4. Retrieve candidates\",\"description\":\"Run vector similarity search, often with metadata filters, to produce a larger top-k candidate set.\"},{\"label\":\"5. Rerank candidates\",\"description\":\"Apply a stronger relevance model to the query and the small candidate set.\"},{\"label\":\"6. Select context\",\"description\":\"Keep the most useful passages for the downstream answer, agent step or search result.\"}]},\"tunes\":{}},{\"id\":\"h-stops\",\"type\":\"header\",\"data\":{\"text\":\"Where the simple example stops\",\"level\":2},\"tunes\":{}},{\"id\":\"p-stops-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Real retrieval systems do not have to use dense embeddings at all. Keyword search such as BM25 can be the first-stage retriever. Sparse learned retrieval, SQL filters, graph traversal or application APIs can also generate candidates.\"},\"tunes\":{}},{\"id\":\"p-stops-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"A reranker also does not care that the candidates came from a vector database. It can rerank BM25 results, hybrid results, hand-selected documents or candidates from multiple retrievers.\"},\"tunes\":{}},{\"id\":\"p-stops-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"Likewise, embeddings do not require a specialized vector database. Small datasets can be compared in memory or with general-purpose databases and vector extensions. Specialized vector systems become useful when indexing, approximate nearest-neighbor search, filtering, scale, update behavior or operational requirements justify them.\"},\"tunes\":{}},{\"id\":\"core-comparison\",\"type\":\"comparison\",\"data\":{\"title\":\"Three different retrieval components\",\"layout\":\"table\",\"columns\":[{\"id\":\"embedding\",\"label\":\"Embedding\"},{\"id\":\"vector\",\"label\":\"Vector database \u002F index\"},{\"id\":\"reranker\",\"label\":\"Reranker\"}],\"rows\":[{\"id\":\"job\",\"label\":\"Primary job\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"input\",\"label\":\"Typical input\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"output\",\"label\":\"Typical output\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"cost\",\"label\":\"Cost profile\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"can-miss\",\"label\":\"Typical failure\",\"values\":[\"\",\"\",\"\"]}]},\"tunes\":{}},{\"id\":\"h-embeddings\",\"type\":\"header\",\"data\":{\"text\":\"Embeddings: representation, not retrieval\",\"level\":2},\"tunes\":{}},{\"id\":\"p-emb-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"An embedding is a numerical representation produced by a model. For semantic retrieval, texts with related meaning are intended to occupy useful positions in a vector space so that a similarity or distance function can compare them.\"},\"tunes\":{}},{\"id\":\"p-emb-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Sentence-BERT was an influential step in making sentence-level semantic similarity practical with bi-encoder-style representations that can be computed independently and compared efficiently. The general idea remains central to modern dense retrieval: precompute document representations, compute the query representation at search time, then compare them.\"},\"tunes\":{}},{\"id\":\"p-emb-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The embedding itself does not search a corpus. It is data produced by an embedding model. Retrieval begins when the system compares the query representation against stored candidates.\"},\"tunes\":{}},{\"id\":\"h-embedding-model\",\"type\":\"header\",\"data\":{\"text\":\"The embedding model defines the representation space\",\"level\":3},\"tunes\":{}},{\"id\":\"p-emodel-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Document and query vectors must be compatible with the model and configuration used to create them. Replacing an embedding model can change dimensionality, similarity behavior, language coverage and domain performance.\"},\"tunes\":{}},{\"id\":\"p-emodel-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"That is why an embedding-model migration is not merely an API-name change. Existing documents may need to be re-embedded and the index rebuilt or versioned.\"},\"tunes\":{}},{\"id\":\"h-dense-sparse\",\"type\":\"header\",\"data\":{\"text\":\"Dense and sparse representations are different\",\"level\":3},\"tunes\":{}},{\"id\":\"p-dense-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Dense embeddings usually contain many non-zero dimensions and are commonly used for semantic similarity. Sparse representations contain many zeros and can preserve stronger token- or term-like structure.\"},\"tunes\":{}},{\"id\":\"p-dense-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Both can support semantic retrieval, and modern search systems can combine dense, sparse and lexical signals. “Vector search” therefore does not always mean one dense cosine-similarity pipeline.\"},\"tunes\":{}},{\"id\":\"h-distance\",\"type\":\"header\",\"data\":{\"text\":\"Similarity functions are part of the representation contract\",\"level\":3},\"tunes\":{}},{\"id\":\"p-distance-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Cosine similarity, dot product and Euclidean distance do not mean the same thing. The correct metric depends on how the embedding model was trained and normalized.\"},\"tunes\":{}},{\"id\":\"p-distance-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Current Qdrant documentation, for example, requires a distance metric as part of vector configuration and documents cosine, dot-product and Euclidean-style choices. The important architectural rule is to treat the metric as part of the embedding\u002Findex contract rather than choose one arbitrarily.\"},\"tunes\":{}},{\"id\":\"embedding-not-truth\",\"type\":\"callout\",\"data\":{\"variant\":\"warning\",\"title\":\"Embedding similarity is not factual support\",\"body\":\"Two passages can be semantically close while one is stale, unauthorized or wrong. Embeddings estimate representational similarity; they do not determine Source-of-Truth authority, freshness or evidentiary validity.\"},\"tunes\":{}},{\"id\":\"h-vector-db\",\"type\":\"header\",\"data\":{\"text\":\"Vector databases and indexes: candidate retrieval\",\"level\":2},\"tunes\":{}},{\"id\":\"p-vdb-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A vector database or vector-capable search system organizes vector representations so the application can retrieve nearby candidates efficiently. Practical systems usually associate vectors with IDs and payload metadata such as source, language, tenant, document type, timestamp or access scope.\"},\"tunes\":{}},{\"id\":\"p-vdb-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Qdrant, for example, organizes data into collections of points where a point contains a vector and optional payload metadata. Its documentation describes HNSW-based similarity search and metadata filtering as separate capabilities of the retrieval layer.\"},\"tunes\":{}},{\"id\":\"p-vdb-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"That distinction matters: the vector index answers a nearest-neighbor problem, while payload filters enforce structural constraints such as tenant, document class or language.\"},\"tunes\":{}},{\"id\":\"h-ann\",\"type\":\"header\",\"data\":{\"text\":\"Approximate nearest-neighbor search trades exactness for efficiency\",\"level\":3},\"tunes\":{}},{\"id\":\"p-ann-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Comparing one query vector against every vector can be practical for small collections but expensive at large scale. Approximate nearest-neighbor indexes such as HNSW reduce search cost by navigating an index structure instead of exhaustively scanning every vector.\"},\"tunes\":{}},{\"id\":\"p-ann-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Approximate search introduces a recall\u002Flatency trade-off. Faster search can miss candidates that exact search would return. Index parameters therefore affect retrieval quality, not just infrastructure performance.\"},\"tunes\":{}},{\"id\":\"p-ann-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"Qdrant exposes both HNSW-related parameters and an exact-search option, illustrating that vector storage and approximate retrieval policy are separate decisions.\"},\"tunes\":{}},{\"id\":\"h-filtering\",\"type\":\"header\",\"data\":{\"text\":\"Metadata filtering belongs before or during candidate retrieval\",\"level\":3},\"tunes\":{}},{\"id\":\"p-filter-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"If the user may only access tenant A, retrieving semantically similar chunks from tenant B and attempting to remove them later is the wrong security boundary. Authorization and hard eligibility filters should constrain the candidate space before those candidates can influence downstream processing.\"},\"tunes\":{}},{\"id\":\"p-filter-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The same principle applies to locale, document status, source class, date, product version and other deterministic constraints. Similarity should rank eligible candidates; it should not override eligibility.\"},\"tunes\":{}},{\"id\":\"h-vector-not-required\",\"type\":\"header\",\"data\":{\"text\":\"A vector database is optional\",\"level\":3},\"tunes\":{}},{\"id\":\"p-optional-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"For a small corpus, brute-force cosine comparison may be simple and sufficient. A relational database with vector support may also be adequate. A dedicated vector database becomes valuable when its indexing, filtering, distributed storage, update behavior or operational features solve a real requirement.\"},\"tunes\":{}},{\"id\":\"p-optional-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Choosing a vector database because “RAG needs one” reverses the architecture process. Start with retrieval requirements and scale, then select the storage\u002Findex technology.\"},\"tunes\":{}},{\"id\":\"h-rerank\",\"type\":\"header\",\"data\":{\"text\":\"Reranking: second-stage relevance refinement\",\"level\":2},\"tunes\":{}},{\"id\":\"p-rerank-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A reranker receives a query and a smaller set of already retrieved candidates, then assigns stronger relevance scores or a new ordering. It is normally more computationally expensive than first-stage retrieval, which is why it is applied after candidate generation rather than to the entire corpus.\"},\"tunes\":{}},{\"id\":\"p-rerank-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Current Elastic guidance describes semantic reranking as a final-stage technique over a small top-k set and notes that it can refine lexical, semantic or hybrid retrieval. Cohere documents the same architecture: first-stage lexical or semantic search followed by a reranking stage.\"},\"tunes\":{}},{\"id\":\"p-rerank-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"A common implementation uses a cross-encoder-like model that examines the query and each candidate together. That richer interaction can distinguish relevance more precisely than independent embedding similarity, but it is much more expensive at corpus scale.\"},\"tunes\":{}},{\"id\":\"h-bi-cross\",\"type\":\"header\",\"data\":{\"text\":\"Bi-encoder retrieval and cross-encoder reranking solve different cost problems\",\"level\":3},\"tunes\":{}},{\"id\":\"encoder-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Property\",\"Bi-encoder \u002F embedding retrieval\",\"Cross-encoder-style reranking\"],[\"Encoding\",\"Query and documents represented independently\",\"Query and candidate processed jointly\"],[\"Document computation\",\"Can be precomputed at ingest\",\"Normally recomputed per query-candidate pair\"],[\"Corpus-scale search\",\"Suitable with vector indexes\",\"Usually too expensive across the entire corpus\"],[\"Typical role\",\"High-recall candidate generation\",\"High-precision ordering of a small candidate set\"],[\"Main trade-off\",\"Fast and scalable but relevance interaction is compressed into vectors\",\"Richer relevance judgment but higher latency\u002Fcost\"]]},\"tunes\":{}},{\"id\":\"h-rerank-limit\",\"type\":\"header\",\"data\":{\"text\":\"A reranker cannot recover what retrieval missed\",\"level\":3},\"tunes\":{}},{\"id\":\"p-rerank-limit-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"If the relevant document is absent from the candidate set, reranking has nothing to promote. This is the central reason to evaluate retrieval and reranking separately.\"},\"tunes\":{}},{\"id\":\"p-rerank-limit-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"A pipeline can have excellent reranker precision and still fail because first-stage recall is poor. Increasing reranker quality will not repair missing source coverage, bad chunking, restrictive filters or a weak candidate retriever.\"},\"tunes\":{}},{\"id\":\"recall-precision\",\"type\":\"callout\",\"data\":{\"variant\":\"success\",\"title\":\"Useful retrieval objective\",\"body\":\"First stage: \u003Cstrong>do not miss the useful candidates.\u003C\u002Fstrong>\u003Cbr>Second stage: \u003Cstrong>put the best candidates first.\u003C\u002Fstrong>\u003Cbr>\u003Cbr>This is not a universal mathematical rule, but it is a useful engineering model for two-stage retrieval.\"},\"tunes\":{}},{\"id\":\"h-hybrid\",\"type\":\"header\",\"data\":{\"text\":\"Hybrid retrieval is a separate design choice\",\"level\":2},\"tunes\":{}},{\"id\":\"p-hybrid-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Dense semantic retrieval is strong when query and document use different wording but express related meaning. Lexical retrieval is strong when exact terms, identifiers, names, codes or rare phrases matter.\"},\"tunes\":{}},{\"id\":\"p-hybrid-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Hybrid retrieval combines multiple candidate signals, often lexical BM25 and vector similarity, then merges rankings using a method such as Reciprocal Rank Fusion or a weighted score combination.\"},\"tunes\":{}},{\"id\":\"p-hybrid-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"Reranking can then operate on the fused candidate set. Hybrid retrieval and reranking are therefore complementary but distinct stages.\"},\"tunes\":{}},{\"id\":\"h-bm25\",\"type\":\"header\",\"data\":{\"text\":\"BM25 is not obsolete because embeddings exist\",\"level\":3},\"tunes\":{}},{\"id\":\"p-bm25-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Keyword search can outperform dense retrieval for exact identifiers, version numbers, error messages, product codes and specialized vocabulary. SQLite FTS5, for example, includes a BM25 ranking function for full-text search.\"},\"tunes\":{}},{\"id\":\"p-bm25-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"A strong retrieval architecture can use lexical retrieval as the only first stage, vector retrieval as the only first stage, or combine both depending on the corpus and query distribution.\"},\"tunes\":{}},{\"id\":\"h-chunking\",\"type\":\"header\",\"data\":{\"text\":\"Chunking changes what embeddings and rerankers can see\",\"level\":2},\"tunes\":{}},{\"id\":\"p-chunk-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"If a document is split poorly, no later retrieval component can fully reconstruct the missing semantic unit. A chunk that cuts a condition away from its exception may embed misleadingly and may also be reranked incorrectly because the candidate text is incomplete.\"},\"tunes\":{}},{\"id\":\"p-chunk-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Chunk size, overlap, structural boundaries and metadata therefore affect both candidate recall and reranker judgment. Retrieval evaluation should test the complete ingestion-to-ranking pipeline, not only the embedding model.\"},\"tunes\":{}},{\"id\":\"h-scores\",\"type\":\"header\",\"data\":{\"text\":\"Do not compare retrieval scores as if they were universal probabilities\",\"level\":2},\"tunes\":{}},{\"id\":\"p-scores-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Cosine similarity, BM25 scores, sparse-vector scores, RRF ranks and reranker scores have different meanings. A score of 0.82 from one embedding model is not automatically comparable with 0.82 from another model or with a reranker score.\"},\"tunes\":{}},{\"id\":\"p-scores-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Thresholds should be calibrated for the actual model, corpus and task. Current Elastic guidance also notes that embedding similarity scores can be query-dependent, which makes universal cutoffs risky.\"},\"tunes\":{}},{\"id\":\"h-eval\",\"type\":\"header\",\"data\":{\"text\":\"Evaluate retrieval stages separately\",\"level\":2},\"tunes\":{}},{\"id\":\"eval-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Layer\",\"Useful question\",\"Example metric or test\"],[\"Source coverage\",\"Does the corpus contain the needed information?\",\"Coverage audit \u002F known-answer source set\"],[\"Chunking\",\"Is the needed evidence retrievable as a coherent unit?\",\"Chunk-level support review\"],[\"First-stage retrieval\",\"Does the relevant item enter the candidate set?\",\"Recall@k\"],[\"Ranking\",\"How high does relevant evidence appear?\",\"MRR, nDCG, precision@k\"],[\"Reranking\",\"Does second-stage scoring improve ordering?\",\"Delta nDCG \u002F MRR \u002F precision\"],[\"Context selection\",\"Do the final selected passages contain sufficient support?\",\"Context relevance \u002F coverage\"],[\"Answer stage\",\"Does the model use the selected evidence correctly?\",\"Faithfulness \u002F claim-evidence evaluation\"]]},\"tunes\":{}},{\"id\":\"p-eval-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"This separation is operationally important. If Recall@50 is poor, the reranker is not the first component to fix. If Recall@50 is strong but the best passage remains at rank 38, reranking or ranking fusion becomes a plausible target.\"},\"tunes\":{}},{\"id\":\"h-failure-map\",\"type\":\"header\",\"data\":{\"text\":\"Which layer actually failed?\",\"level\":2},\"tunes\":{}},{\"id\":\"failure-comparison\",\"type\":\"comparison\",\"data\":{\"title\":\"Symptoms and likely retrieval layer\",\"layout\":\"table\",\"columns\":[{\"id\":\"symptom\",\"label\":\"Observed symptom\"},{\"id\":\"likely\",\"label\":\"Likely layer\"},{\"id\":\"test\",\"label\":\"First diagnostic\"}],\"rows\":[{\"id\":\"missed\",\"label\":\"Relevant document never appears\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"lowrank\",\"label\":\"Relevant document appears too low\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"wrongtenant\",\"label\":\"Semantically good but forbidden result\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"stale\",\"label\":\"Relevant but outdated result\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"context\",\"label\":\"Correct result retrieved but omitted from prompt\",\"values\":[\"\",\"\",\"\"]}]},\"tunes\":{}},{\"id\":\"h-authority\",\"type\":\"header\",\"data\":{\"text\":\"Relevance and Source of Truth are different\",\"level\":2},\"tunes\":{}},{\"id\":\"p-authority-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A reranker can make a stale document look extremely relevant. A vector index can retrieve a secondary summary that is semantically closer than the primary source. Retrieval quality therefore cannot replace authority rules.\"},\"tunes\":{}},{\"id\":\"p-authority-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Where source authority matters, metadata filters, source classes, version rules and provenance should constrain retrieval before the result becomes model context.\"},\"tunes\":{}},{\"id\":\"authority-callout\",\"type\":\"callout\",\"data\":{\"variant\":\"warning\",\"title\":\"Reranking cannot make a non-authoritative source authoritative\",\"body\":\"Relevance answers whether a candidate fits the query. Source-of-Truth architecture answers whether that candidate is allowed to establish the claim.\"},\"tunes\":{}},{\"id\":\"h-impl\",\"type\":\"header\",\"data\":{\"text\":\"Original implementation evidence\",\"level\":2},\"tunes\":{}},{\"id\":\"h-sot-engine\",\"type\":\"header\",\"data\":{\"text\":\"Source of Truth Research Engine: lexical and semantic retrieval are separate\",\"level\":3},\"tunes\":{}},{\"id\":\"p-sot-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The Source of Truth Research Engine contains a local lexical retrieval path using SQLite FTS5\u002FBM25 and a separate optional semantic retrieval path using locally generated embeddings.\"},\"tunes\":{}},{\"id\":\"p-sot-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Its semantic search implementation computes a query vector and compares it with stored chunk vectors using cosine similarity. The project deliberately treats semantic similarity as a discovery signal rather than evidence: a candidate must still be traced back to a concrete source and locator before it supports a claim.\"},\"tunes\":{}},{\"id\":\"p-sot-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"This is useful implementation evidence for R01 because the same corpus can support lexical ranking and vector similarity without confusing either mechanism with evidentiary authority.\"},\"tunes\":{}},{\"id\":\"h-client\",\"type\":\"header\",\"data\":{\"text\":\"Aaasaasa AI Client: Qdrant is a vector infrastructure component\",\"level\":3},\"tunes\":{}},{\"id\":\"p-client-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Aaasaasa AI Client includes Qdrant\u002Fvector infrastructure as a separate local resource. The Electron architecture exposes Qdrant services from the trusted main-process side rather than treating vector search as part of the model itself.\"},\"tunes\":{}},{\"id\":\"p-client-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The repository contains a Qdrant client adapter, Qdrant service configuration and Docker-based Qdrant infrastructure. This demonstrates the architectural separation between AI provider\u002Fmodel execution and vector storage\u002Fsearch.\"},\"tunes\":{}},{\"id\":\"p-client-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The existence of Qdrant support should not be overstated as a complete production RAG pipeline. The evidence here is narrower: vector infrastructure is implemented as its own component boundary.\"},\"tunes\":{}},{\"id\":\"impl-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Implementation evidence\",\"What it demonstrates\"],[\"SQLite FTS5\u002FBM25 in Source of Truth Research Engine\",\"Lexical retrieval can exist independently of embeddings.\"],[\"Local Ollama embeddings\",\"Representation generation is its own stage.\"],[\"Stored semantic vectors + cosine comparison\",\"Semantic retrieval consumes embeddings after they have been produced.\"],[\"Qdrant support in Aaasaasa AI Client\",\"Vector storage\u002Fsearch is an infrastructure capability separate from the model provider.\"],[\"Evidence\u002Fprovenance rules in Source of Truth Research Engine\",\"Retrieved similarity does not equal authority or proof.\"],[\"No claimed custom reranker in these implementations\",\"Reranking is explained as an architectural stage, not falsely claimed as already implemented evidence.\"]]},\"tunes\":{}},{\"id\":\"impl-discipline\",\"type\":\"callout\",\"data\":{\"variant\":\"note\",\"title\":\"Evidence boundary\",\"body\":\"The current implementation evidence confirms lexical retrieval, embeddings, vector search infrastructure and provenance-aware retrieval. This article does \u003Cstrong>not\u003C\u002Fstrong> claim that a production cross-encoder reranking service is already implemented in these projects.\"},\"tunes\":{}},{\"id\":\"h-decisions\",\"type\":\"header\",\"data\":{\"text\":\"When do you need each component?\",\"level\":2},\"tunes\":{}},{\"id\":\"decision-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Need\",\"Likely component\"],[\"Semantic similarity across different wording\",\"Embedding model + vector similarity search\"],[\"Efficient search over a large vector corpus\",\"Vector index\u002Fdatabase or vector-capable search engine\"],[\"Exact identifiers, error codes or rare terms\",\"Lexical\u002Ffull-text retrieval such as BM25\"],[\"Both exact terminology and semantic meaning\",\"Hybrid lexical + semantic retrieval\"],[\"Candidate set is good but ordering is weak\",\"Reranker\"],[\"Relevant items are absent from candidate set\",\"Improve source coverage, chunking, retriever, filters or candidate count before reranking\"],[\"Hard tenant\u002Fsource\u002Fversion constraints\",\"Deterministic metadata\u002Fauthorization filtering\"],[\"Small corpus\",\"Potentially simple brute-force similarity or general-purpose database rather than dedicated vector DB\"]]},\"tunes\":{}},{\"id\":\"h-sequence\",\"type\":\"header\",\"data\":{\"text\":\"A practical retrieval design sequence\",\"level\":2},\"tunes\":{}},{\"id\":\"design-flow\",\"type\":\"processFlow\",\"data\":{\"title\":\"Design retrieval from requirements, not from product names\",\"orientation\":\"auto\",\"steps\":[{\"label\":\"1. Define the query types\",\"description\":\"Identify semantic questions, exact lookups, identifiers, current-state reads and domain-specific patterns.\"},{\"label\":\"2. Define eligible sources\",\"description\":\"Apply tenant, authorization, locale, version, source class and freshness constraints.\"},{\"label\":\"3. Establish lexical baseline\",\"description\":\"Measure whether simple full-text\u002FBM25 retrieval already solves much of the workload.\"},{\"label\":\"4. Add embeddings where semantic recall is needed\",\"description\":\"Choose and evaluate an embedding model against representative domain queries.\"},{\"label\":\"5. Choose vector storage\u002Findexing based on scale\",\"description\":\"Use brute force, database vector support or a dedicated vector engine according to requirements.\"},{\"label\":\"6. Evaluate first-stage recall\",\"description\":\"Confirm that relevant evidence enters a sufficiently large candidate set.\"},{\"label\":\"7. Add hybrid retrieval if signals are complementary\",\"description\":\"Fuse lexical and semantic rankings when both materially improve candidate generation.\"},{\"label\":\"8. Add reranking if ordering remains the bottleneck\",\"description\":\"Apply the stronger model only to the candidate set where its cost is justified.\"},{\"label\":\"9. Tune final context selection\",\"description\":\"Control redundancy, context budget, authority, diversity and evidence coverage before generation.\"},{\"label\":\"10. Evaluate end-to-end\",\"description\":\"Measure retrieval, context and answer quality separately so failures can be localized.\"}]},\"tunes\":{}},{\"id\":\"h-misconceptions\",\"type\":\"header\",\"data\":{\"text\":\"Common misconceptions\",\"level\":2},\"tunes\":{}},{\"id\":\"misconceptions-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Misconception\",\"Correction\"],[\"“An embedding is a vector database.”\",\"An embedding is a representation; the database\u002Findex stores and searches representations.\"],[\"“A vector database creates semantic meaning.”\",\"The embedding model creates the representation; the vector system indexes and compares it.\"],[\"“RAG requires a vector database.”\",\"RAG requires retrieval, not a specific retrieval technology.\"],[\"“Reranking is the same as vector search.”\",\"Vector search generates candidates; reranking reorders a candidate set.\"],[\"“Rerankers fix poor recall.”\",\"They cannot promote a document that was never retrieved.\"],[\"“Dense search replaces BM25.”\",\"Lexical search remains valuable for exact terms, identifiers and specialized vocabulary.\"],[\"“Higher similarity means more authoritative.”\",\"Similarity and source authority are different dimensions.\"],[\"“More top-k always improves RAG.”\",\"Larger candidate sets can improve recall but add latency, noise and context-selection burden.\"],[\"“One score threshold works everywhere.”\",\"Scores depend on model, query, corpus and retrieval method and must be calibrated.\"],[\"“A dedicated vector DB is always more advanced.”\",\"It is only justified when its operational and retrieval capabilities match the requirements.\"]]},\"tunes\":{}},{\"id\":\"h-edge\",\"type\":\"header\",\"data\":{\"text\":\"Edge cases and limitations\",\"level\":2},\"tunes\":{}},{\"id\":\"p-edge-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Some applications do not need semantic search. Exact database lookup or structured SQL can be more correct, faster and easier to audit than embedding retrieval.\"},\"tunes\":{}},{\"id\":\"p-edge-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Some corpora are so small that a full vector scan is acceptable. Approximate indexing adds complexity without meaningful benefit.\"},\"tunes\":{}},{\"id\":\"p-edge-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"Some queries require high recall before any precision optimization. Legal discovery, research and compliance review may prefer broad candidate retrieval followed by transparent filtering and human review.\"},\"tunes\":{}},{\"id\":\"p-edge-4\",\"type\":\"paragraph\",\"data\":{\"text\":\"Multilingual and domain-specific retrieval can behave very differently across embedding models. Benchmark claims from public datasets should not be treated as proof for a private corpus.\"},\"tunes\":{}},{\"id\":\"p-edge-5\",\"type\":\"paragraph\",\"data\":{\"text\":\"Reranking latency grows with the number and length of candidates. Candidate size should therefore be tuned as an accuracy\u002Fcost\u002Flatency variable rather than copied from a tutorial.\"},\"tunes\":{}},{\"id\":\"h-change\",\"type\":\"header\",\"data\":{\"text\":\"What would change this answer?\",\"level\":2},\"tunes\":{}},{\"id\":\"p-change-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The component boundaries would not change if a vendor packages embedding generation, vector indexing and reranking behind one API. The product may hide the stages, but they remain conceptually different responsibilities with different failure modes.\"},\"tunes\":{}},{\"id\":\"p-change-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Future embedding or retrieval models may reduce the need for separate reranking in some workloads, while stronger late-interaction or learned sparse methods can blur traditional dense\u002Flexical categories. The architecture should still ask which stage produces representations, which stage generates candidates and which stage refines ranking.\"},\"tunes\":{}},{\"id\":\"p-change-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The best design also changes with corpus size, query mix, language, domain terminology, update frequency, source authority, latency budget and evaluation results.\"},\"tunes\":{}},{\"id\":\"h-related\",\"type\":\"header\",\"data\":{\"text\":\"Related canonical knowledge\",\"level\":2},\"tunes\":{}},{\"id\":\"p-related-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"R01 assumes the basic RAG concept is already understood. RAG is the wider pattern in which retrieved external information is supplied to a model; embeddings, vector search and reranking are optional retrieval components inside that pattern.\"},\"tunes\":{}},{\"id\":\"ref-rag\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works\",\"title\":\"What Is RAG? The Simplest Explanation of How It Works\",\"excerpt\":\"A plain-English foundation for how retrieval brings external knowledge into the model context.\",\"ctaLabel\":\"Read the RAG foundation\"},\"tunes\":{}},{\"id\":\"p-related-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"When retrieval fails, diagnose source coverage, retrieval, ranking, context assembly and generation separately rather than treating the whole system as one “RAG failure.”\"},\"tunes\":{}},{\"id\":\"ref-rag-failed\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method\",\"title\":\"RAG Failed — But Which Layer Actually Failed? A Diagnostic Method\",\"excerpt\":\"A layer-by-layer method for isolating source coverage, retrieval, ranking, context assembly, generation, evidence attribution and freshness failures.\",\"ctaLabel\":\"Read the RAG diagnostic method\"},\"tunes\":{}},{\"id\":\"p-related-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"Source-of-Truth architecture is the authority layer around retrieval: it decides which source can establish a claim, while embeddings and ranking only decide which candidates appear relevant.\"},\"tunes\":{}},{\"id\":\"h-faq\",\"type\":\"header\",\"data\":{\"text\":\"Frequently asked questions\",\"level\":2},\"tunes\":{}},{\"id\":\"faq\",\"type\":\"faq\",\"data\":{\"title\":\"Embeddings, vector databases and reranking\",\"items\":[{\"id\":\"faq1\",\"question\":\"What is the difference between embeddings and a vector database?\",\"answer\":\"Embeddings are numerical representations produced by a model. A vector database or vector index stores and searches those representations together with IDs and metadata.\"},{\"id\":\"faq2\",\"question\":\"What does a reranker do?\",\"answer\":\"A reranker takes an already retrieved candidate set and re-scores or reorders those candidates using a stronger relevance model or scoring method.\"},{\"id\":\"faq3\",\"question\":\"Does RAG require a vector database?\",\"answer\":\"No. RAG requires retrieval of external information. Retrieval can use lexical search, SQL, APIs, graphs, vector search, hybrid search or combinations of these.\"},{\"id\":\"faq4\",\"question\":\"Why not use the reranker on the whole corpus?\",\"answer\":\"Rerankers commonly perform more expensive query-document interaction, so they are usually applied to a small top-k candidate set after a faster first-stage retriever.\"},{\"id\":\"faq5\",\"question\":\"Can reranking fix a missing document?\",\"answer\":\"No. If the relevant document was not retrieved into the candidate set, reranking has nothing to promote.\"},{\"id\":\"faq6\",\"question\":\"Is cosine similarity a relevance probability?\",\"answer\":\"No. It is a similarity measure whose numeric meaning depends on the embedding model and corpus. It should not be treated as a universal probability of relevance.\"},{\"id\":\"faq7\",\"question\":\"Should I use BM25 and vector search together?\",\"answer\":\"Use hybrid retrieval when evaluation shows that lexical and semantic signals recover complementary relevant documents. It is not automatically better for every corpus.\"},{\"id\":\"faq8\",\"question\":\"When do I need a dedicated vector database?\",\"answer\":\"When vector indexing, filtering, scale, updates, distributed operation or other vector-specific requirements justify a specialized system. Small workloads may not need one.\"}]},\"tunes\":{}},{\"id\":\"h-glossary\",\"type\":\"header\",\"data\":{\"text\":\"Glossary\",\"level\":2},\"tunes\":{}},{\"id\":\"glossary\",\"type\":\"glossary\",\"data\":{\"title\":\"Key retrieval terms\",\"entries\":[{\"term\":\"Embedding\",\"definition\":\"A numerical representation of content produced by an embedding model for similarity, clustering, retrieval or related tasks.\",\"anchor\":\"embedding\"},{\"term\":\"Dense vector\",\"definition\":\"A vector representation in which many dimensions carry non-zero values, commonly used in semantic retrieval.\",\"anchor\":\"dense-vector\"},{\"term\":\"Sparse vector\",\"definition\":\"A high-dimensional representation in which most dimensions are zero, often preserving stronger token- or term-like structure.\",\"anchor\":\"sparse-vector\"},{\"term\":\"Vector index\",\"definition\":\"A data structure that organizes vectors for efficient similarity or nearest-neighbor retrieval.\",\"anchor\":\"vector-index\"},{\"term\":\"Vector database\",\"definition\":\"A storage\u002Fsearch system designed to manage vectors, associated metadata and vector retrieval workloads.\",\"anchor\":\"vector-database\"},{\"term\":\"ANN\",\"definition\":\"Approximate nearest-neighbor search, which trades exact exhaustive comparison for faster retrieval at scale.\",\"anchor\":\"ann\"},{\"term\":\"HNSW\",\"definition\":\"Hierarchical Navigable Small World, a graph-based approximate nearest-neighbor indexing approach widely used for vector retrieval.\",\"anchor\":\"hnsw\"},{\"term\":\"BM25\",\"definition\":\"A lexical relevance-ranking method based on term occurrence and corpus statistics, widely used in full-text search.\",\"anchor\":\"bm25\"},{\"term\":\"Hybrid search\",\"definition\":\"Retrieval that combines results or scores from multiple retrieval methods such as lexical and vector search.\",\"anchor\":\"hybrid-search\"},{\"term\":\"Reranking\",\"definition\":\"A later retrieval stage that re-scores and reorders an already generated candidate set.\",\"anchor\":\"reranking\"},{\"term\":\"Bi-encoder\",\"definition\":\"An architecture that encodes query and candidate independently, enabling precomputation and scalable similarity search.\",\"anchor\":\"bi-encoder\"},{\"term\":\"Cross-encoder\",\"definition\":\"A model that jointly processes a query and candidate text, often improving relevance judgment at higher computational cost.\",\"anchor\":\"cross-encoder\"},{\"term\":\"Recall@k\",\"definition\":\"The fraction of relevant items recovered within the top k retrieved candidates.\",\"anchor\":\"recall-at-k\"},{\"term\":\"nDCG\",\"definition\":\"Normalized Discounted Cumulative Gain, a ranking metric that rewards relevant results appearing higher in an ordered list.\",\"anchor\":\"ndcg\"}]},\"tunes\":{}},{\"id\":\"h-conclusion\",\"type\":\"header\",\"data\":{\"text\":\"Conclusion\",\"level\":2},\"tunes\":{}},{\"id\":\"p-conclusion-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The clean retrieval model is simple: embeddings represent meaning, vector search retrieves candidates, and rerankers refine candidate ordering.\"},\"tunes\":{}},{\"id\":\"p-conclusion-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Once those boundaries are explicit, architecture decisions become easier to diagnose. Missing candidates point toward source coverage, chunking, embeddings, filters or first-stage retrieval. Poor ordering points toward ranking, fusion or reranking. Incorrect final answers can then be investigated separately at context and generation layers.\"},\"tunes\":{}},{\"id\":\"p-conclusion-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The most important result is not choosing the most fashionable retrieval component. It is building a retrieval pipeline whose stages, authority boundaries, metrics and failure modes can be measured independently.\"},\"tunes\":{}},{\"id\":\"h-sources\",\"type\":\"header\",\"data\":{\"text\":\"Primary sources and implementation evidence\",\"level\":2},\"tunes\":{}},{\"id\":\"p-sources-note\",\"type\":\"paragraph\",\"data\":{\"text\":\"The external references below document the representation, vector-search and reranking mechanisms used in this article. Project-specific sections are original implementation evidence and are intentionally narrower than claims about complete production RAG maturity.\"},\"tunes\":{}},{\"id\":\"src-sbert\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Farxiv.org\u002Fabs\u002F1908.10084\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Sentence-BERT: Sentence Embeddings using Siamese BERT-Networks\",\"description\":\"Foundational paper demonstrating independently computable sentence embeddings for efficient semantic similarity search.\"}},\"tunes\":{}},{\"id\":\"src-qdrant-overview\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fqdrant.tech\u002Fdocumentation\u002Foverview\u002F\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Qdrant — Architecture and data structure overview\",\"description\":\"Official documentation describing collections, points, vectors, payload metadata and HNSW-based similarity indexing.\"}},\"tunes\":{}},{\"id\":\"src-qdrant-search\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fqdrant.tech\u002Fdocumentation\u002Fsearch\u002Fsearch\u002F\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Qdrant — Search\",\"description\":\"Official vector-search documentation covering similarity queries, filtering, exact versus approximate search and dense\u002Fsparse behavior.\"}},\"tunes\":{}},{\"id\":\"src-elastic-vector\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fwww.elastic.co\u002Fdocs\u002Fsolutions\u002Fsearch\u002Fvector\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Elastic — Vector search\",\"description\":\"Current documentation on dense\u002Fsparse vector retrieval, lexical\u002Fvector combinations and multi-stage search pipelines.\"}},\"tunes\":{}},{\"id\":\"src-elastic-rerank\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fwww.elastic.co\u002Fdocs\u002Fsolutions\u002Fsearch\u002Franking\u002Fsemantic-reranking\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Elastic — Semantic reranking\",\"description\":\"Current guidance defining semantic reranking as a later-stage relevance operation over a smaller candidate set.\"}},\"tunes\":{}},{\"id\":\"src-cohere-rerank\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdocs.cohere.com\u002Fdocs\u002Freranking-with-cohere\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Cohere — Reranking with Cohere\",\"description\":\"Current documentation showing reranking as a second-stage improvement over lexical or semantic first-stage retrieval.\"}},\"tunes\":{}},{\"id\":\"src-sqlite-fts5\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fwww.sqlite.org\u002Ffts5.html\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"SQLite FTS5\",\"description\":\"Official SQLite documentation for full-text search and the built-in BM25 ranking function used as lexical retrieval evidence.\"}},\"tunes\":{}}],\"version\":\"2.31.6\"}",{"time":1291,"blocks":1292,"version":2133},1791489989811,[1293,1297,1302,1307,1312,1316,1320,1324,1328,1332,1336,1340,1344,1348,1371,1375,1379,1383,1387,1414,1418,1422,1426,1430,1434,1438,1442,1446,1450,1454,1458,1462,1466,1471,1475,1479,1483,1487,1491,1495,1499,1503,1507,1511,1515,1519,1523,1527,1531,1535,1539,1543,1547,1575,1579,1583,1587,1592,1596,1600,1604,1608,1612,1616,1620,1624,1628,1632,1636,1640,1644,1648,1682,1686,1690,1717,1721,1725,1729,1734,1738,1742,1746,1750,1754,1758,1762,1766,1770,1795,1800,1804,1834,1838,1873,1877,1914,1918,1922,1926,1930,1934,1938,1942,1946,1950,1954,1958,1962,1969,1973,1980,1984,1988,2017,2021,2061,2065,2069,2073,2077,2081,2085,2092,2099,2106,2113,2120,2127],{"id":215,"data":1294,"type":218,"tunes":1296},{"text":1295},"Embeddings, vector databases and rerankers are three different parts of retrieval. An embedding model converts text or other data into numerical representations; a vector database or vector index stores and searches those representations to retrieve candidate items; a reranker takes a smaller candidate set and reorders it using a more expensive relevance model or scoring method. They often appear together in RAG, but none of them is the same thing as RAG, and none is mandatory in every retrieval system.",{},{"id":221,"data":1298,"type":226,"tunes":1301},{"body":1299,"title":1300,"variant":225},"\u003Cstrong>Embeddings represent. Vector search retrieves. Reranking refines.\u003C\u002Fstrong>\u003Cbr>\u003Cbr>A useful mental model is:\u003Cbr>\u003Cstrong>content → embedding → candidate retrieval → reranking → selected context → model\u003C\u002Fstrong>.\u003Cbr>\u003Cbr>The boundaries matter because each layer fails differently. Bad embeddings distort semantic similarity. A weak retrieval index misses useful candidates. A reranker can reorder candidates, but it cannot recover a relevant document that was never retrieved.","Direct answer",{},{"id":229,"data":1303,"type":226,"tunes":1306},{"body":1304,"title":1305,"variant":233},"A vector database is not an embedding model. An embedding is not a search result. A reranker is not a vector database. RAG is the wider pattern that can use any of these components to retrieve external information before generation.","Do not collapse the retrieval stack",{},{"id":236,"data":1308,"type":226,"tunes":1311},{"body":1309,"title":1310,"variant":240},"The basic architecture is stable even though products evolve rapidly. Current Qdrant documentation separates vectors, payload metadata, collections and vector indexes; current Elastic guidance treats semantic reranking as a later-stage operation over a small candidate set; current Cohere documentation likewise describes reranking as a second-stage improvement over lexical or semantic search.","Current-source note — 8 October 2026",{},{"id":243,"data":1313,"type":248,"tunes":1315},{"title":1314,"maxLevel":246,"minLevel":247},"Contents",{},{"id":251,"data":1317,"type":42,"tunes":1319},{"text":1318,"level":247},"What this really means",{},{"id":256,"data":1321,"type":218,"tunes":1323},{"text":1322},"Search systems have two competing goals: find enough potentially relevant material and put the best material near the top. Fast first-stage retrieval usually optimizes candidate generation. A stronger second-stage model can then spend more computation distinguishing the best candidates.",{},{"id":261,"data":1325,"type":218,"tunes":1327},{"text":1326},"Embeddings, vector indexes and rerankers occupy different positions in that process. Treating them as one feature hides important design choices about recall, precision, latency, storage, metadata filtering and model cost.",{},{"id":266,"data":1329,"type":218,"tunes":1331},{"text":1330},"The distinction also prevents a common RAG mistake: assuming that storing document embeddings in a vector database automatically creates high-quality retrieval. Retrieval quality depends on the embedding model, chunking, metadata, query construction, index configuration, candidate count, hybrid retrieval, reranking and the authority of the underlying sources.",{},{"id":271,"data":1333,"type":42,"tunes":1335},{"text":1334,"level":247},"The simplest example",{},{"id":276,"data":1337,"type":218,"tunes":1339},{"text":1338},"Suppose a knowledge base contains 100,000 document chunks. A user asks: “How do I revoke an API token?”",{},{"id":281,"data":1341,"type":218,"tunes":1343},{"text":1342},"First, an embedding model can encode the query into a vector. Document chunks may already have their own stored embeddings. A vector search then compares the query vector to the indexed document vectors and returns, for example, 30 likely candidates.",{},{"id":286,"data":1345,"type":218,"tunes":1347},{"text":1346},"Those 30 candidates can then be passed to a reranker. The reranker compares the query more directly with each candidate and produces a new relevance ordering. The application might keep the best five for the model context.",{},{"id":291,"data":1349,"type":314,"tunes":1370},{"steps":1350,"title":1369,"orientation":313},[1351,1354,1357,1360,1363,1366],{"label":1352,"description":1353},"1. Embed documents","Convert each searchable chunk into a numerical representation, usually at ingest time.",{"label":1355,"description":1356},"2. Store\u002Findex vectors","Associate vectors with document IDs and metadata in a searchable vector index or database.",{"label":1358,"description":1359},"3. Embed the query","Encode the user's query using the compatible embedding model and query configuration.",{"label":1361,"description":1362},"4. Retrieve candidates","Run vector similarity search, often with metadata filters, to produce a larger top-k candidate set.",{"label":1364,"description":1365},"5. Rerank candidates","Apply a stronger relevance model to the query and the small candidate set.",{"label":1367,"description":1368},"6. Select context","Keep the most useful passages for the downstream answer, agent step or search result.","A basic two-stage semantic retrieval pipeline",{},{"id":317,"data":1372,"type":42,"tunes":1374},{"text":1373,"level":247},"Where the simple example stops",{},{"id":322,"data":1376,"type":218,"tunes":1378},{"text":1377},"Real retrieval systems do not have to use dense embeddings at all. Keyword search such as BM25 can be the first-stage retriever. Sparse learned retrieval, SQL filters, graph traversal or application APIs can also generate candidates.",{},{"id":327,"data":1380,"type":218,"tunes":1382},{"text":1381},"A reranker also does not care that the candidates came from a vector database. It can rerank BM25 results, hybrid results, hand-selected documents or candidates from multiple retrievers.",{},{"id":332,"data":1384,"type":218,"tunes":1386},{"text":1385},"Likewise, embeddings do not require a specialized vector database. Small datasets can be compared in memory or with general-purpose databases and vector extensions. Specialized vector systems become useful when indexing, approximate nearest-neighbor search, filtering, scale, update behavior or operational requirements justify them.",{},{"id":337,"data":1388,"type":373,"tunes":1413},{"rows":1389,"title":1405,"layout":362,"columns":1406},[1390,1393,1396,1399,1402],{"id":341,"label":1391,"values":1392},"Primary job",[344,344,344],{"id":346,"label":1394,"values":1395},"Typical input",[344,344,344],{"id":350,"label":1397,"values":1398},"Typical output",[344,344,344],{"id":354,"label":1400,"values":1401},"Cost profile",[344,344,344],{"id":358,"label":1403,"values":1404},"Typical failure",[344,344,344],"Three different retrieval components",[1407,1409,1411],{"id":365,"label":1408},"Embedding",{"id":368,"label":1410},"Vector database \u002F index",{"id":371,"label":1412},"Reranker",{},{"id":376,"data":1415,"type":42,"tunes":1417},{"text":1416,"level":247},"Embeddings: representation, not retrieval",{},{"id":381,"data":1419,"type":218,"tunes":1421},{"text":1420},"An embedding is a numerical representation produced by a model. For semantic retrieval, texts with related meaning are intended to occupy useful positions in a vector space so that a similarity or distance function can compare them.",{},{"id":386,"data":1423,"type":218,"tunes":1425},{"text":1424},"Sentence-BERT was an influential step in making sentence-level semantic similarity practical with bi-encoder-style representations that can be computed independently and compared efficiently. The general idea remains central to modern dense retrieval: precompute document representations, compute the query representation at search time, then compare them.",{},{"id":391,"data":1427,"type":218,"tunes":1429},{"text":1428},"The embedding itself does not search a corpus. It is data produced by an embedding model. Retrieval begins when the system compares the query representation against stored candidates.",{},{"id":396,"data":1431,"type":42,"tunes":1433},{"text":1432,"level":246},"The embedding model defines the representation space",{},{"id":401,"data":1435,"type":218,"tunes":1437},{"text":1436},"Document and query vectors must be compatible with the model and configuration used to create them. Replacing an embedding model can change dimensionality, similarity behavior, language coverage and domain performance.",{},{"id":406,"data":1439,"type":218,"tunes":1441},{"text":1440},"That is why an embedding-model migration is not merely an API-name change. Existing documents may need to be re-embedded and the index rebuilt or versioned.",{},{"id":411,"data":1443,"type":42,"tunes":1445},{"text":1444,"level":246},"Dense and sparse representations are different",{},{"id":416,"data":1447,"type":218,"tunes":1449},{"text":1448},"Dense embeddings usually contain many non-zero dimensions and are commonly used for semantic similarity. Sparse representations contain many zeros and can preserve stronger token- or term-like structure.",{},{"id":421,"data":1451,"type":218,"tunes":1453},{"text":1452},"Both can support semantic retrieval, and modern search systems can combine dense, sparse and lexical signals. “Vector search” therefore does not always mean one dense cosine-similarity pipeline.",{},{"id":426,"data":1455,"type":42,"tunes":1457},{"text":1456,"level":246},"Similarity functions are part of the representation contract",{},{"id":431,"data":1459,"type":218,"tunes":1461},{"text":1460},"Cosine similarity, dot product and Euclidean distance do not mean the same thing. The correct metric depends on how the embedding model was trained and normalized.",{},{"id":436,"data":1463,"type":218,"tunes":1465},{"text":1464},"Current Qdrant documentation, for example, requires a distance metric as part of vector configuration and documents cosine, dot-product and Euclidean-style choices. The important architectural rule is to treat the metric as part of the embedding\u002Findex contract rather than choose one arbitrarily.",{},{"id":441,"data":1467,"type":226,"tunes":1470},{"body":1468,"title":1469,"variant":233},"Two passages can be semantically close while one is stale, unauthorized or wrong. Embeddings estimate representational similarity; they do not determine Source-of-Truth authority, freshness or evidentiary validity.","Embedding similarity is not factual support",{},{"id":447,"data":1472,"type":42,"tunes":1474},{"text":1473,"level":247},"Vector databases and indexes: candidate retrieval",{},{"id":452,"data":1476,"type":218,"tunes":1478},{"text":1477},"A vector database or vector-capable search system organizes vector representations so the application can retrieve nearby candidates efficiently. Practical systems usually associate vectors with IDs and payload metadata such as source, language, tenant, document type, timestamp or access scope.",{},{"id":457,"data":1480,"type":218,"tunes":1482},{"text":1481},"Qdrant, for example, organizes data into collections of points where a point contains a vector and optional payload metadata. Its documentation describes HNSW-based similarity search and metadata filtering as separate capabilities of the retrieval layer.",{},{"id":462,"data":1484,"type":218,"tunes":1486},{"text":1485},"That distinction matters: the vector index answers a nearest-neighbor problem, while payload filters enforce structural constraints such as tenant, document class or language.",{},{"id":467,"data":1488,"type":42,"tunes":1490},{"text":1489,"level":246},"Approximate nearest-neighbor search trades exactness for efficiency",{},{"id":472,"data":1492,"type":218,"tunes":1494},{"text":1493},"Comparing one query vector against every vector can be practical for small collections but expensive at large scale. Approximate nearest-neighbor indexes such as HNSW reduce search cost by navigating an index structure instead of exhaustively scanning every vector.",{},{"id":477,"data":1496,"type":218,"tunes":1498},{"text":1497},"Approximate search introduces a recall\u002Flatency trade-off. Faster search can miss candidates that exact search would return. Index parameters therefore affect retrieval quality, not just infrastructure performance.",{},{"id":482,"data":1500,"type":218,"tunes":1502},{"text":1501},"Qdrant exposes both HNSW-related parameters and an exact-search option, illustrating that vector storage and approximate retrieval policy are separate decisions.",{},{"id":487,"data":1504,"type":42,"tunes":1506},{"text":1505,"level":246},"Metadata filtering belongs before or during candidate retrieval",{},{"id":492,"data":1508,"type":218,"tunes":1510},{"text":1509},"If the user may only access tenant A, retrieving semantically similar chunks from tenant B and attempting to remove them later is the wrong security boundary. Authorization and hard eligibility filters should constrain the candidate space before those candidates can influence downstream processing.",{},{"id":497,"data":1512,"type":218,"tunes":1514},{"text":1513},"The same principle applies to locale, document status, source class, date, product version and other deterministic constraints. Similarity should rank eligible candidates; it should not override eligibility.",{},{"id":502,"data":1516,"type":42,"tunes":1518},{"text":1517,"level":246},"A vector database is optional",{},{"id":507,"data":1520,"type":218,"tunes":1522},{"text":1521},"For a small corpus, brute-force cosine comparison may be simple and sufficient. A relational database with vector support may also be adequate. A dedicated vector database becomes valuable when its indexing, filtering, distributed storage, update behavior or operational features solve a real requirement.",{},{"id":512,"data":1524,"type":218,"tunes":1526},{"text":1525},"Choosing a vector database because “RAG needs one” reverses the architecture process. Start with retrieval requirements and scale, then select the storage\u002Findex technology.",{},{"id":517,"data":1528,"type":42,"tunes":1530},{"text":1529,"level":247},"Reranking: second-stage relevance refinement",{},{"id":522,"data":1532,"type":218,"tunes":1534},{"text":1533},"A reranker receives a query and a smaller set of already retrieved candidates, then assigns stronger relevance scores or a new ordering. It is normally more computationally expensive than first-stage retrieval, which is why it is applied after candidate generation rather than to the entire corpus.",{},{"id":527,"data":1536,"type":218,"tunes":1538},{"text":1537},"Current Elastic guidance describes semantic reranking as a final-stage technique over a small top-k set and notes that it can refine lexical, semantic or hybrid retrieval. Cohere documents the same architecture: first-stage lexical or semantic search followed by a reranking stage.",{},{"id":532,"data":1540,"type":218,"tunes":1542},{"text":1541},"A common implementation uses a cross-encoder-like model that examines the query and each candidate together. That richer interaction can distinguish relevance more precisely than independent embedding similarity, but it is much more expensive at corpus scale.",{},{"id":537,"data":1544,"type":42,"tunes":1546},{"text":1545,"level":246},"Bi-encoder retrieval and cross-encoder reranking solve different cost problems",{},{"id":542,"data":1548,"type":362,"tunes":1574},{"content":1549,"stretched":43,"withHeadings":14},[1550,1554,1558,1562,1566,1570],[1551,1552,1553],"Property","Bi-encoder \u002F embedding retrieval","Cross-encoder-style reranking",[1555,1556,1557],"Encoding","Query and documents represented independently","Query and candidate processed jointly",[1559,1560,1561],"Document computation","Can be precomputed at ingest","Normally recomputed per query-candidate pair",[1563,1564,1565],"Corpus-scale search","Suitable with vector indexes","Usually too expensive across the entire corpus",[1567,1568,1569],"Typical role","High-recall candidate generation","High-precision ordering of a small candidate set",[1571,1572,1573],"Main trade-off","Fast and scalable but relevance interaction is compressed into vectors","Richer relevance judgment but higher latency\u002Fcost",{},{"id":571,"data":1576,"type":42,"tunes":1578},{"text":1577,"level":246},"A reranker cannot recover what retrieval missed",{},{"id":576,"data":1580,"type":218,"tunes":1582},{"text":1581},"If the relevant document is absent from the candidate set, reranking has nothing to promote. This is the central reason to evaluate retrieval and reranking separately.",{},{"id":581,"data":1584,"type":218,"tunes":1586},{"text":1585},"A pipeline can have excellent reranker precision and still fail because first-stage recall is poor. Increasing reranker quality will not repair missing source coverage, bad chunking, restrictive filters or a weak candidate retriever.",{},{"id":586,"data":1588,"type":226,"tunes":1591},{"body":1589,"title":1590,"variant":590},"First stage: \u003Cstrong>do not miss the useful candidates.\u003C\u002Fstrong>\u003Cbr>Second stage: \u003Cstrong>put the best candidates first.\u003C\u002Fstrong>\u003Cbr>\u003Cbr>This is not a universal mathematical rule, but it is a useful engineering model for two-stage retrieval.","Useful retrieval objective",{},{"id":593,"data":1593,"type":42,"tunes":1595},{"text":1594,"level":247},"Hybrid retrieval is a separate design choice",{},{"id":598,"data":1597,"type":218,"tunes":1599},{"text":1598},"Dense semantic retrieval is strong when query and document use different wording but express related meaning. Lexical retrieval is strong when exact terms, identifiers, names, codes or rare phrases matter.",{},{"id":603,"data":1601,"type":218,"tunes":1603},{"text":1602},"Hybrid retrieval combines multiple candidate signals, often lexical BM25 and vector similarity, then merges rankings using a method such as Reciprocal Rank Fusion or a weighted score combination.",{},{"id":608,"data":1605,"type":218,"tunes":1607},{"text":1606},"Reranking can then operate on the fused candidate set. Hybrid retrieval and reranking are therefore complementary but distinct stages.",{},{"id":613,"data":1609,"type":42,"tunes":1611},{"text":1610,"level":246},"BM25 is not obsolete because embeddings exist",{},{"id":618,"data":1613,"type":218,"tunes":1615},{"text":1614},"Keyword search can outperform dense retrieval for exact identifiers, version numbers, error messages, product codes and specialized vocabulary. SQLite FTS5, for example, includes a BM25 ranking function for full-text search.",{},{"id":623,"data":1617,"type":218,"tunes":1619},{"text":1618},"A strong retrieval architecture can use lexical retrieval as the only first stage, vector retrieval as the only first stage, or combine both depending on the corpus and query distribution.",{},{"id":628,"data":1621,"type":42,"tunes":1623},{"text":1622,"level":247},"Chunking changes what embeddings and rerankers can see",{},{"id":633,"data":1625,"type":218,"tunes":1627},{"text":1626},"If a document is split poorly, no later retrieval component can fully reconstruct the missing semantic unit. A chunk that cuts a condition away from its exception may embed misleadingly and may also be reranked incorrectly because the candidate text is incomplete.",{},{"id":638,"data":1629,"type":218,"tunes":1631},{"text":1630},"Chunk size, overlap, structural boundaries and metadata therefore affect both candidate recall and reranker judgment. Retrieval evaluation should test the complete ingestion-to-ranking pipeline, not only the embedding model.",{},{"id":643,"data":1633,"type":42,"tunes":1635},{"text":1634,"level":247},"Do not compare retrieval scores as if they were universal probabilities",{},{"id":648,"data":1637,"type":218,"tunes":1639},{"text":1638},"Cosine similarity, BM25 scores, sparse-vector scores, RRF ranks and reranker scores have different meanings. A score of 0.82 from one embedding model is not automatically comparable with 0.82 from another model or with a reranker score.",{},{"id":653,"data":1641,"type":218,"tunes":1643},{"text":1642},"Thresholds should be calibrated for the actual model, corpus and task. Current Elastic guidance also notes that embedding similarity scores can be query-dependent, which makes universal cutoffs risky.",{},{"id":658,"data":1645,"type":42,"tunes":1647},{"text":1646,"level":247},"Evaluate retrieval stages separately",{},{"id":663,"data":1649,"type":362,"tunes":1681},{"content":1650,"stretched":43,"withHeadings":14},[1651,1655,1659,1663,1666,1670,1673,1677],[1652,1653,1654],"Layer","Useful question","Example metric or test",[1656,1657,1658],"Source coverage","Does the corpus contain the needed information?","Coverage audit \u002F known-answer source set",[1660,1661,1662],"Chunking","Is the needed evidence retrievable as a coherent unit?","Chunk-level support review",[1664,1665,681],"First-stage retrieval","Does the relevant item enter the candidate set?",[1667,1668,1669],"Ranking","How high does relevant evidence appear?","MRR, nDCG, precision@k",[1671,1672,689],"Reranking","Does second-stage scoring improve ordering?",[1674,1675,1676],"Context selection","Do the final selected passages contain sufficient support?","Context relevance \u002F coverage",[1678,1679,1680],"Answer stage","Does the model use the selected evidence correctly?","Faithfulness \u002F claim-evidence evaluation",{},{"id":700,"data":1683,"type":218,"tunes":1685},{"text":1684},"This separation is operationally important. If Recall@50 is poor, the reranker is not the first component to fix. If Recall@50 is strong but the best passage remains at rank 38, reranking or ranking fusion becomes a plausible target.",{},{"id":705,"data":1687,"type":42,"tunes":1689},{"text":1688,"level":247},"Which layer actually failed?",{},{"id":710,"data":1691,"type":373,"tunes":1716},{"rows":1692,"title":1708,"layout":362,"columns":1709},[1693,1696,1699,1702,1705],{"id":714,"label":1694,"values":1695},"Relevant document never appears",[344,344,344],{"id":718,"label":1697,"values":1698},"Relevant document appears too low",[344,344,344],{"id":722,"label":1700,"values":1701},"Semantically good but forbidden result",[344,344,344],{"id":726,"label":1703,"values":1704},"Relevant but outdated result",[344,344,344],{"id":730,"label":1706,"values":1707},"Correct result retrieved but omitted from prompt",[344,344,344],"Symptoms and likely retrieval layer",[1710,1712,1714],{"id":736,"label":1711},"Observed symptom",{"id":739,"label":1713},"Likely layer",{"id":742,"label":1715},"First diagnostic",{},{"id":746,"data":1718,"type":42,"tunes":1720},{"text":1719,"level":247},"Relevance and Source of Truth are different",{},{"id":751,"data":1722,"type":218,"tunes":1724},{"text":1723},"A reranker can make a stale document look extremely relevant. A vector index can retrieve a secondary summary that is semantically closer than the primary source. Retrieval quality therefore cannot replace authority rules.",{},{"id":756,"data":1726,"type":218,"tunes":1728},{"text":1727},"Where source authority matters, metadata filters, source classes, version rules and provenance should constrain retrieval before the result becomes model context.",{},{"id":761,"data":1730,"type":226,"tunes":1733},{"body":1731,"title":1732,"variant":233},"Relevance answers whether a candidate fits the query. Source-of-Truth architecture answers whether that candidate is allowed to establish the claim.","Reranking cannot make a non-authoritative source authoritative",{},{"id":767,"data":1735,"type":42,"tunes":1737},{"text":1736,"level":247},"Original implementation evidence",{},{"id":772,"data":1739,"type":42,"tunes":1741},{"text":1740,"level":246},"Source of Truth Research Engine: lexical and semantic retrieval are separate",{},{"id":777,"data":1743,"type":218,"tunes":1745},{"text":1744},"The Source of Truth Research Engine contains a local lexical retrieval path using SQLite FTS5\u002FBM25 and a separate optional semantic retrieval path using locally generated embeddings.",{},{"id":782,"data":1747,"type":218,"tunes":1749},{"text":1748},"Its semantic search implementation computes a query vector and compares it with stored chunk vectors using cosine similarity. The project deliberately treats semantic similarity as a discovery signal rather than evidence: a candidate must still be traced back to a concrete source and locator before it supports a claim.",{},{"id":787,"data":1751,"type":218,"tunes":1753},{"text":1752},"This is useful implementation evidence for R01 because the same corpus can support lexical ranking and vector similarity without confusing either mechanism with evidentiary authority.",{},{"id":792,"data":1755,"type":42,"tunes":1757},{"text":1756,"level":246},"Aaasaasa AI Client: Qdrant is a vector infrastructure component",{},{"id":797,"data":1759,"type":218,"tunes":1761},{"text":1760},"Aaasaasa AI Client includes Qdrant\u002Fvector infrastructure as a separate local resource. The Electron architecture exposes Qdrant services from the trusted main-process side rather than treating vector search as part of the model itself.",{},{"id":802,"data":1763,"type":218,"tunes":1765},{"text":1764},"The repository contains a Qdrant client adapter, Qdrant service configuration and Docker-based Qdrant infrastructure. This demonstrates the architectural separation between AI provider\u002Fmodel execution and vector storage\u002Fsearch.",{},{"id":807,"data":1767,"type":218,"tunes":1769},{"text":1768},"The existence of Qdrant support should not be overstated as a complete production RAG pipeline. The evidence here is narrower: vector infrastructure is implemented as its own component boundary.",{},{"id":812,"data":1771,"type":362,"tunes":1794},{"content":1772,"stretched":43,"withHeadings":14},[1773,1776,1779,1782,1785,1788,1791],[1774,1775],"Implementation evidence","What it demonstrates",[1777,1778],"SQLite FTS5\u002FBM25 in Source of Truth Research Engine","Lexical retrieval can exist independently of embeddings.",[1780,1781],"Local Ollama embeddings","Representation generation is its own stage.",[1783,1784],"Stored semantic vectors + cosine comparison","Semantic retrieval consumes embeddings after they have been produced.",[1786,1787],"Qdrant support in Aaasaasa AI Client","Vector storage\u002Fsearch is an infrastructure capability separate from the model provider.",[1789,1790],"Evidence\u002Fprovenance rules in Source of Truth Research Engine","Retrieved similarity does not equal authority or proof.",[1792,1793],"No claimed custom reranker in these implementations","Reranking is explained as an architectural stage, not falsely claimed as already implemented evidence.",{},{"id":838,"data":1796,"type":226,"tunes":1799},{"body":1797,"title":1798,"variant":240},"The current implementation evidence confirms lexical retrieval, embeddings, vector search infrastructure and provenance-aware retrieval. This article does \u003Cstrong>not\u003C\u002Fstrong> claim that a production cross-encoder reranking service is already implemented in these projects.","Evidence boundary",{},{"id":844,"data":1801,"type":42,"tunes":1803},{"text":1802,"level":247},"When do you need each component?",{},{"id":849,"data":1805,"type":362,"tunes":1833},{"content":1806,"stretched":43,"withHeadings":14},[1807,1810,1813,1816,1819,1822,1824,1827,1830],[1808,1809],"Need","Likely component",[1811,1812],"Semantic similarity across different wording","Embedding model + vector similarity search",[1814,1815],"Efficient search over a large vector corpus","Vector index\u002Fdatabase or vector-capable search engine",[1817,1818],"Exact identifiers, error codes or rare terms","Lexical\u002Ffull-text retrieval such as BM25",[1820,1821],"Both exact terminology and semantic meaning","Hybrid lexical + semantic retrieval",[1823,1412],"Candidate set is good but ordering is weak",[1825,1826],"Relevant items are absent from candidate set","Improve source coverage, chunking, retriever, filters or candidate count before reranking",[1828,1829],"Hard tenant\u002Fsource\u002Fversion constraints","Deterministic metadata\u002Fauthorization filtering",[1831,1832],"Small corpus","Potentially simple brute-force similarity or general-purpose database rather than dedicated vector DB",{},{"id":880,"data":1835,"type":42,"tunes":1837},{"text":1836,"level":247},"A practical retrieval design sequence",{},{"id":885,"data":1839,"type":314,"tunes":1872},{"steps":1840,"title":1871,"orientation":313},[1841,1844,1847,1850,1853,1856,1859,1862,1865,1868],{"label":1842,"description":1843},"1. Define the query types","Identify semantic questions, exact lookups, identifiers, current-state reads and domain-specific patterns.",{"label":1845,"description":1846},"2. Define eligible sources","Apply tenant, authorization, locale, version, source class and freshness constraints.",{"label":1848,"description":1849},"3. Establish lexical baseline","Measure whether simple full-text\u002FBM25 retrieval already solves much of the workload.",{"label":1851,"description":1852},"4. Add embeddings where semantic recall is needed","Choose and evaluate an embedding model against representative domain queries.",{"label":1854,"description":1855},"5. Choose vector storage\u002Findexing based on scale","Use brute force, database vector support or a dedicated vector engine according to requirements.",{"label":1857,"description":1858},"6. Evaluate first-stage recall","Confirm that relevant evidence enters a sufficiently large candidate set.",{"label":1860,"description":1861},"7. Add hybrid retrieval if signals are complementary","Fuse lexical and semantic rankings when both materially improve candidate generation.",{"label":1863,"description":1864},"8. Add reranking if ordering remains the bottleneck","Apply the stronger model only to the candidate set where its cost is justified.",{"label":1866,"description":1867},"9. Tune final context selection","Control redundancy, context budget, authority, diversity and evidence coverage before generation.",{"label":1869,"description":1870},"10. Evaluate end-to-end","Measure retrieval, context and answer quality separately so failures can be localized.","Design retrieval from requirements, not from product names",{},{"id":921,"data":1874,"type":42,"tunes":1876},{"text":1875,"level":247},"Common misconceptions",{},{"id":926,"data":1878,"type":362,"tunes":1913},{"content":1879,"stretched":43,"withHeadings":14},[1880,1883,1886,1889,1892,1895,1898,1901,1904,1907,1910],[1881,1882],"Misconception","Correction",[1884,1885],"“An embedding is a vector database.”","An embedding is a representation; the database\u002Findex stores and searches representations.",[1887,1888],"“A vector database creates semantic meaning.”","The embedding model creates the representation; the vector system indexes and compares it.",[1890,1891],"“RAG requires a vector database.”","RAG requires retrieval, not a specific retrieval technology.",[1893,1894],"“Reranking is the same as vector search.”","Vector search generates candidates; reranking reorders a candidate set.",[1896,1897],"“Rerankers fix poor recall.”","They cannot promote a document that was never retrieved.",[1899,1900],"“Dense search replaces BM25.”","Lexical search remains valuable for exact terms, identifiers and specialized vocabulary.",[1902,1903],"“Higher similarity means more authoritative.”","Similarity and source authority are different dimensions.",[1905,1906],"“More top-k always improves RAG.”","Larger candidate sets can improve recall but add latency, noise and context-selection burden.",[1908,1909],"“One score threshold works everywhere.”","Scores depend on model, query, corpus and retrieval method and must be calibrated.",[1911,1912],"“A dedicated vector DB is always more advanced.”","It is only justified when its operational and retrieval capabilities match the requirements.",{},{"id":964,"data":1915,"type":42,"tunes":1917},{"text":1916,"level":247},"Edge cases and limitations",{},{"id":969,"data":1919,"type":218,"tunes":1921},{"text":1920},"Some applications do not need semantic search. Exact database lookup or structured SQL can be more correct, faster and easier to audit than embedding retrieval.",{},{"id":974,"data":1923,"type":218,"tunes":1925},{"text":1924},"Some corpora are so small that a full vector scan is acceptable. Approximate indexing adds complexity without meaningful benefit.",{},{"id":979,"data":1927,"type":218,"tunes":1929},{"text":1928},"Some queries require high recall before any precision optimization. Legal discovery, research and compliance review may prefer broad candidate retrieval followed by transparent filtering and human review.",{},{"id":984,"data":1931,"type":218,"tunes":1933},{"text":1932},"Multilingual and domain-specific retrieval can behave very differently across embedding models. Benchmark claims from public datasets should not be treated as proof for a private corpus.",{},{"id":989,"data":1935,"type":218,"tunes":1937},{"text":1936},"Reranking latency grows with the number and length of candidates. Candidate size should therefore be tuned as an accuracy\u002Fcost\u002Flatency variable rather than copied from a tutorial.",{},{"id":994,"data":1939,"type":42,"tunes":1941},{"text":1940,"level":247},"What would change this answer?",{},{"id":999,"data":1943,"type":218,"tunes":1945},{"text":1944},"The component boundaries would not change if a vendor packages embedding generation, vector indexing and reranking behind one API. The product may hide the stages, but they remain conceptually different responsibilities with different failure modes.",{},{"id":1004,"data":1947,"type":218,"tunes":1949},{"text":1948},"Future embedding or retrieval models may reduce the need for separate reranking in some workloads, while stronger late-interaction or learned sparse methods can blur traditional dense\u002Flexical categories. The architecture should still ask which stage produces representations, which stage generates candidates and which stage refines ranking.",{},{"id":1009,"data":1951,"type":218,"tunes":1953},{"text":1952},"The best design also changes with corpus size, query mix, language, domain terminology, update frequency, source authority, latency budget and evaluation results.",{},{"id":1014,"data":1955,"type":42,"tunes":1957},{"text":1956,"level":247},"Related canonical knowledge",{},{"id":1019,"data":1959,"type":218,"tunes":1961},{"text":1960},"R01 assumes the basic RAG concept is already understood. RAG is the wider pattern in which retrieved external information is supplied to a model; embeddings, vector search and reranking are optional retrieval components inside that pattern.",{},{"id":1024,"data":1963,"type":1030,"tunes":1968},{"url":1964,"title":1965,"excerpt":1966,"ctaLabel":1967},"https:\u002F\u002Fstajic.de\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works","What Is RAG? The Simplest Explanation of How It Works","A plain-English foundation for how retrieval brings external knowledge into the model context.","Read the RAG foundation",{},{"id":1033,"data":1970,"type":218,"tunes":1972},{"text":1971},"When retrieval fails, diagnose source coverage, retrieval, ranking, context assembly and generation separately rather than treating the whole system as one “RAG failure.”",{},{"id":1038,"data":1974,"type":1030,"tunes":1979},{"url":1975,"title":1976,"excerpt":1977,"ctaLabel":1978},"https:\u002F\u002Fstajic.de\u002Fblog\u002Frag-failed-but-which-layer-actually-failed-a-diagnostic-method","RAG Failed — But Which Layer Actually Failed? A Diagnostic Method","A layer-by-layer method for isolating source coverage, retrieval, ranking, context assembly, generation, evidence attribution and freshness failures.","Read the RAG diagnostic method",{},{"id":1046,"data":1981,"type":218,"tunes":1983},{"text":1982},"Source-of-Truth architecture is the authority layer around retrieval: it decides which source can establish a claim, while embeddings and ranking only decide which candidates appear relevant.",{},{"id":1051,"data":1985,"type":42,"tunes":1987},{"text":1986,"level":247},"Frequently asked questions",{},{"id":1056,"data":1989,"type":1056,"tunes":2016},{"items":1990,"title":2015},[1991,1994,1997,2000,2003,2006,2009,2012],{"id":1060,"answer":1992,"question":1993},"Embeddings are numerical representations produced by a model. A vector database or vector index stores and searches those representations together with IDs and metadata.","What is the difference between embeddings and a vector database?",{"id":1064,"answer":1995,"question":1996},"A reranker takes an already retrieved candidate set and re-scores or reorders those candidates using a stronger relevance model or scoring method.","What does a reranker do?",{"id":1068,"answer":1998,"question":1999},"No. RAG requires retrieval of external information. Retrieval can use lexical search, SQL, APIs, graphs, vector search, hybrid search or combinations of these.","Does RAG require a vector database?",{"id":1072,"answer":2001,"question":2002},"Rerankers commonly perform more expensive query-document interaction, so they are usually applied to a small top-k candidate set after a faster first-stage retriever.","Why not use the reranker on the whole corpus?",{"id":1076,"answer":2004,"question":2005},"No. If the relevant document was not retrieved into the candidate set, reranking has nothing to promote.","Can reranking fix a missing document?",{"id":1080,"answer":2007,"question":2008},"No. It is a similarity measure whose numeric meaning depends on the embedding model and corpus. It should not be treated as a universal probability of relevance.","Is cosine similarity a relevance probability?",{"id":1084,"answer":2010,"question":2011},"Use hybrid retrieval when evaluation shows that lexical and semantic signals recover complementary relevant documents. It is not automatically better for every corpus.","Should I use BM25 and vector search together?",{"id":1088,"answer":2013,"question":2014},"When vector indexing, filtering, scale, updates, distributed operation or other vector-specific requirements justify a specialized system. Small workloads may not need one.","When do I need a dedicated vector database?","Embeddings, vector databases and reranking",{},{"id":1094,"data":2018,"type":42,"tunes":2020},{"text":2019,"level":247},"Glossary",{},{"id":1099,"data":2022,"type":1099,"tunes":2060},{"title":2023,"entries":2024},"Key retrieval terms",[2025,2027,2030,2033,2036,2039,2041,2043,2045,2048,2050,2053,2056,2058],{"term":1408,"anchor":365,"definition":2026},"A numerical representation of content produced by an embedding model for similarity, clustering, retrieval or related tasks.",{"term":2028,"anchor":1107,"definition":2029},"Dense vector","A vector representation in which many dimensions carry non-zero values, commonly used in semantic retrieval.",{"term":2031,"anchor":1111,"definition":2032},"Sparse vector","A high-dimensional representation in which most dimensions are zero, often preserving stronger token- or term-like structure.",{"term":2034,"anchor":1115,"definition":2035},"Vector index","A data structure that organizes vectors for efficient similarity or nearest-neighbor retrieval.",{"term":2037,"anchor":1119,"definition":2038},"Vector database","A storage\u002Fsearch system designed to manage vectors, associated metadata and vector retrieval workloads.",{"term":1122,"anchor":1123,"definition":2040},"Approximate nearest-neighbor search, which trades exact exhaustive comparison for faster retrieval at scale.",{"term":1126,"anchor":1127,"definition":2042},"Hierarchical Navigable Small World, a graph-based approximate nearest-neighbor indexing approach widely used for vector retrieval.",{"term":1130,"anchor":1131,"definition":2044},"A lexical relevance-ranking method based on term occurrence and corpus statistics, widely used in full-text search.",{"term":2046,"anchor":1135,"definition":2047},"Hybrid search","Retrieval that combines results or scores from multiple retrieval methods such as lexical and vector search.",{"term":1671,"anchor":1138,"definition":2049},"A later retrieval stage that re-scores and reorders an already generated candidate set.",{"term":2051,"anchor":1142,"definition":2052},"Bi-encoder","An architecture that encodes query and candidate independently, enabling precomputation and scalable similarity search.",{"term":2054,"anchor":1146,"definition":2055},"Cross-encoder","A model that jointly processes a query and candidate text, often improving relevance judgment at higher computational cost.",{"term":681,"anchor":1149,"definition":2057},"The fraction of relevant items recovered within the top k retrieved candidates.",{"term":1152,"anchor":1153,"definition":2059},"Normalized Discounted Cumulative Gain, a ranking metric that rewards relevant results appearing higher in an ordered list.",{},{"id":1157,"data":2062,"type":42,"tunes":2064},{"text":2063,"level":247},"Conclusion",{},{"id":1162,"data":2066,"type":218,"tunes":2068},{"text":2067},"The clean retrieval model is simple: embeddings represent meaning, vector search retrieves candidates, and rerankers refine candidate ordering.",{},{"id":1167,"data":2070,"type":218,"tunes":2072},{"text":2071},"Once those boundaries are explicit, architecture decisions become easier to diagnose. Missing candidates point toward source coverage, chunking, embeddings, filters or first-stage retrieval. Poor ordering points toward ranking, fusion or reranking. Incorrect final answers can then be investigated separately at context and generation layers.",{},{"id":1172,"data":2074,"type":218,"tunes":2076},{"text":2075},"The most important result is not choosing the most fashionable retrieval component. It is building a retrieval pipeline whose stages, authority boundaries, metrics and failure modes can be measured independently.",{},{"id":1177,"data":2078,"type":42,"tunes":2080},{"text":2079,"level":247},"Primary sources and implementation evidence",{},{"id":1182,"data":2082,"type":218,"tunes":2084},{"text":2083},"The external references below document the representation, vector-search and reranking mechanisms used in this article. Project-specific sections are original implementation evidence and are intentionally narrower than claims about complete production RAG maturity.",{},{"id":1187,"data":2086,"type":1194,"tunes":2091},{"link":1189,"meta":2087},{"image":2088,"title":2089,"description":2090},{"url":344},"Sentence-BERT: Sentence Embeddings using Siamese BERT-Networks","Foundational paper demonstrating independently computable sentence embeddings for efficient semantic similarity search.",{},{"id":1197,"data":2093,"type":1194,"tunes":2098},{"link":1199,"meta":2094},{"image":2095,"title":2096,"description":2097},{"url":344},"Qdrant — Architecture and data structure overview","Official documentation describing collections, points, vectors, payload metadata and HNSW-based similarity indexing.",{},{"id":1206,"data":2100,"type":1194,"tunes":2105},{"link":1208,"meta":2101},{"image":2102,"title":2103,"description":2104},{"url":344},"Qdrant — Search","Official vector-search documentation covering similarity queries, filtering, exact versus approximate search and dense\u002Fsparse behavior.",{},{"id":1215,"data":2107,"type":1194,"tunes":2112},{"link":1217,"meta":2108},{"image":2109,"title":2110,"description":2111},{"url":344},"Elastic — Vector search","Current documentation on dense\u002Fsparse vector retrieval, lexical\u002Fvector combinations and multi-stage search pipelines.",{},{"id":1224,"data":2114,"type":1194,"tunes":2119},{"link":1226,"meta":2115},{"image":2116,"title":2117,"description":2118},{"url":344},"Elastic — Semantic reranking","Current guidance defining semantic reranking as a later-stage relevance operation over a smaller candidate set.",{},{"id":1233,"data":2121,"type":1194,"tunes":2126},{"link":1235,"meta":2122},{"image":2123,"title":2124,"description":2125},{"url":344},"Cohere — Reranking with Cohere","Current documentation showing reranking as a second-stage improvement over lexical or semantic first-stage retrieval.",{},{"id":1242,"data":2128,"type":1194,"tunes":2132},{"link":1244,"meta":2129},{"image":2130,"title":1247,"description":2131},{"url":344},"Official SQLite documentation for full-text search and the built-in BM25 ranking function used as lexical retrieval evidence.",{},"2.31.6","Embeddings represent meaning, vector databases retrieve candidates, and rerankers refine results. Learn how these three retrieval layers differ and work together in RAG.",{"lang":7,"title":208,"content":210,"contentJson":2136,"excerpt":1251},{"time":212,"blocks":2137,"version":1250},[2138,2141,2144,2147,2150,2153,2156,2159,2162,2165,2168,2171,2174,2177,2187,2190,2193,2196,2199,2217,2220,2223,2226,2229,2232,2235,2238,2241,2244,2247,2250,2253,2256,2259,2262,2265,2268,2271,2274,2277,2280,2283,2286,2289,2292,2295,2298,2301,2304,2307,2310,2313,2316,2326,2329,2332,2335,2338,2341,2344,2347,2350,2353,2356,2359,2362,2365,2368,2371,2374,2377,2380,2392,2395,2398,2416,2419,2422,2425,2428,2431,2434,2437,2440,2443,2446,2449,2452,2455,2466,2469,2472,2485,2488,2502,2505,2520,2523,2526,2529,2532,2535,2538,2541,2544,2547,2550,2553,2556,2559,2562,2565,2568,2571,2583,2586,2604,2607,2610,2613,2616,2619,2622,2627,2632,2637,2642,2647,2652],{"id":215,"data":2139,"type":218,"tunes":2140},{"text":217},{},{"id":221,"data":2142,"type":226,"tunes":2143},{"body":223,"title":224,"variant":225},{},{"id":229,"data":2145,"type":226,"tunes":2146},{"body":231,"title":232,"variant":233},{},{"id":236,"data":2148,"type":226,"tunes":2149},{"body":238,"title":239,"variant":240},{},{"id":243,"data":2151,"type":248,"tunes":2152},{"title":245,"maxLevel":246,"minLevel":247},{},{"id":251,"data":2154,"type":42,"tunes":2155},{"text":253,"level":247},{},{"id":256,"data":2157,"type":218,"tunes":2158},{"text":258},{},{"id":261,"data":2160,"type":218,"tunes":2161},{"text":263},{},{"id":266,"data":2163,"type":218,"tunes":2164},{"text":268},{},{"id":271,"data":2166,"type":42,"tunes":2167},{"text":273,"level":247},{},{"id":276,"data":2169,"type":218,"tunes":2170},{"text":278},{},{"id":281,"data":2172,"type":218,"tunes":2173},{"text":283},{},{"id":286,"data":2175,"type":218,"tunes":2176},{"text":288},{},{"id":291,"data":2178,"type":314,"tunes":2186},{"steps":2179,"title":312,"orientation":313},[2180,2181,2182,2183,2184,2185],{"label":295,"description":296},{"label":298,"description":299},{"label":301,"description":302},{"label":304,"description":305},{"label":307,"description":308},{"label":310,"description":311},{},{"id":317,"data":2188,"type":42,"tunes":2189},{"text":319,"level":247},{},{"id":322,"data":2191,"type":218,"tunes":2192},{"text":324},{},{"id":327,"data":2194,"type":218,"tunes":2195},{"text":329},{},{"id":332,"data":2197,"type":218,"tunes":2198},{"text":334},{},{"id":337,"data":2200,"type":373,"tunes":2216},{"rows":2201,"title":361,"layout":362,"columns":2212},[2202,2204,2206,2208,2210],{"id":341,"label":342,"values":2203},[344,344,344],{"id":346,"label":347,"values":2205},[344,344,344],{"id":350,"label":351,"values":2207},[344,344,344],{"id":354,"label":355,"values":2209},[344,344,344],{"id":358,"label":359,"values":2211},[344,344,344],[2213,2214,2215],{"id":365,"label":366},{"id":368,"label":369},{"id":371,"label":372},{},{"id":376,"data":2218,"type":42,"tunes":2219},{"text":378,"level":247},{},{"id":381,"data":2221,"type":218,"tunes":2222},{"text":383},{},{"id":386,"data":2224,"type":218,"tunes":2225},{"text":388},{},{"id":391,"data":2227,"type":218,"tunes":2228},{"text":393},{},{"id":396,"data":2230,"type":42,"tunes":2231},{"text":398,"level":246},{},{"id":401,"data":2233,"type":218,"tunes":2234},{"text":403},{},{"id":406,"data":2236,"type":218,"tunes":2237},{"text":408},{},{"id":411,"data":2239,"type":42,"tunes":2240},{"text":413,"level":246},{},{"id":416,"data":2242,"type":218,"tunes":2243},{"text":418},{},{"id":421,"data":2245,"type":218,"tunes":2246},{"text":423},{},{"id":426,"data":2248,"type":42,"tunes":2249},{"text":428,"level":246},{},{"id":431,"data":2251,"type":218,"tunes":2252},{"text":433},{},{"id":436,"data":2254,"type":218,"tunes":2255},{"text":438},{},{"id":441,"data":2257,"type":226,"tunes":2258},{"body":443,"title":444,"variant":233},{},{"id":447,"data":2260,"type":42,"tunes":2261},{"text":449,"level":247},{},{"id":452,"data":2263,"type":218,"tunes":2264},{"text":454},{},{"id":457,"data":2266,"type":218,"tunes":2267},{"text":459},{},{"id":462,"data":2269,"type":218,"tunes":2270},{"text":464},{},{"id":467,"data":2272,"type":42,"tunes":2273},{"text":469,"level":246},{},{"id":472,"data":2275,"type":218,"tunes":2276},{"text":474},{},{"id":477,"data":2278,"type":218,"tunes":2279},{"text":479},{},{"id":482,"data":2281,"type":218,"tunes":2282},{"text":484},{},{"id":487,"data":2284,"type":42,"tunes":2285},{"text":489,"level":246},{},{"id":492,"data":2287,"type":218,"tunes":2288},{"text":494},{},{"id":497,"data":2290,"type":218,"tunes":2291},{"text":499},{},{"id":502,"data":2293,"type":42,"tunes":2294},{"text":504,"level":246},{},{"id":507,"data":2296,"type":218,"tunes":2297},{"text":509},{},{"id":512,"data":2299,"type":218,"tunes":2300},{"text":514},{},{"id":517,"data":2302,"type":42,"tunes":2303},{"text":519,"level":247},{},{"id":522,"data":2305,"type":218,"tunes":2306},{"text":524},{},{"id":527,"data":2308,"type":218,"tunes":2309},{"text":529},{},{"id":532,"data":2311,"type":218,"tunes":2312},{"text":534},{},{"id":537,"data":2314,"type":42,"tunes":2315},{"text":539,"level":246},{},{"id":542,"data":2317,"type":362,"tunes":2325},{"content":2318,"stretched":43,"withHeadings":14},[2319,2320,2321,2322,2323,2324],[546,547,548],[550,551,552],[554,555,556],[558,559,560],[562,563,564],[566,567,568],{},{"id":571,"data":2327,"type":42,"tunes":2328},{"text":573,"level":246},{},{"id":576,"data":2330,"type":218,"tunes":2331},{"text":578},{},{"id":581,"data":2333,"type":218,"tunes":2334},{"text":583},{},{"id":586,"data":2336,"type":226,"tunes":2337},{"body":588,"title":589,"variant":590},{},{"id":593,"data":2339,"type":42,"tunes":2340},{"text":595,"level":247},{},{"id":598,"data":2342,"type":218,"tunes":2343},{"text":600},{},{"id":603,"data":2345,"type":218,"tunes":2346},{"text":605},{},{"id":608,"data":2348,"type":218,"tunes":2349},{"text":610},{},{"id":613,"data":2351,"type":42,"tunes":2352},{"text":615,"level":246},{},{"id":618,"data":2354,"type":218,"tunes":2355},{"text":620},{},{"id":623,"data":2357,"type":218,"tunes":2358},{"text":625},{},{"id":628,"data":2360,"type":42,"tunes":2361},{"text":630,"level":247},{},{"id":633,"data":2363,"type":218,"tunes":2364},{"text":635},{},{"id":638,"data":2366,"type":218,"tunes":2367},{"text":640},{},{"id":643,"data":2369,"type":42,"tunes":2370},{"text":645,"level":247},{},{"id":648,"data":2372,"type":218,"tunes":2373},{"text":650},{},{"id":653,"data":2375,"type":218,"tunes":2376},{"text":655},{},{"id":658,"data":2378,"type":42,"tunes":2379},{"text":660,"level":247},{},{"id":663,"data":2381,"type":362,"tunes":2391},{"content":2382,"stretched":43,"withHeadings":14},[2383,2384,2385,2386,2387,2388,2389,2390],[667,668,669],[671,672,673],[675,676,677],[679,680,681],[683,684,685],[687,688,689],[691,692,693],[695,696,697],{},{"id":700,"data":2393,"type":218,"tunes":2394},{"text":702},{},{"id":705,"data":2396,"type":42,"tunes":2397},{"text":707,"level":247},{},{"id":710,"data":2399,"type":373,"tunes":2415},{"rows":2400,"title":733,"layout":362,"columns":2411},[2401,2403,2405,2407,2409],{"id":714,"label":715,"values":2402},[344,344,344],{"id":718,"label":719,"values":2404},[344,344,344],{"id":722,"label":723,"values":2406},[344,344,344],{"id":726,"label":727,"values":2408},[344,344,344],{"id":730,"label":731,"values":2410},[344,344,344],[2412,2413,2414],{"id":736,"label":737},{"id":739,"label":740},{"id":742,"label":743},{},{"id":746,"data":2417,"type":42,"tunes":2418},{"text":748,"level":247},{},{"id":751,"data":2420,"type":218,"tunes":2421},{"text":753},{},{"id":756,"data":2423,"type":218,"tunes":2424},{"text":758},{},{"id":761,"data":2426,"type":226,"tunes":2427},{"body":763,"title":764,"variant":233},{},{"id":767,"data":2429,"type":42,"tunes":2430},{"text":769,"level":247},{},{"id":772,"data":2432,"type":42,"tunes":2433},{"text":774,"level":246},{},{"id":777,"data":2435,"type":218,"tunes":2436},{"text":779},{},{"id":782,"data":2438,"type":218,"tunes":2439},{"text":784},{},{"id":787,"data":2441,"type":218,"tunes":2442},{"text":789},{},{"id":792,"data":2444,"type":42,"tunes":2445},{"text":794,"level":246},{},{"id":797,"data":2447,"type":218,"tunes":2448},{"text":799},{},{"id":802,"data":2450,"type":218,"tunes":2451},{"text":804},{},{"id":807,"data":2453,"type":218,"tunes":2454},{"text":809},{},{"id":812,"data":2456,"type":362,"tunes":2465},{"content":2457,"stretched":43,"withHeadings":14},[2458,2459,2460,2461,2462,2463,2464],[816,817],[819,820],[822,823],[825,826],[828,829],[831,832],[834,835],{},{"id":838,"data":2467,"type":226,"tunes":2468},{"body":840,"title":841,"variant":240},{},{"id":844,"data":2470,"type":42,"tunes":2471},{"text":846,"level":247},{},{"id":849,"data":2473,"type":362,"tunes":2484},{"content":2474,"stretched":43,"withHeadings":14},[2475,2476,2477,2478,2479,2480,2481,2482,2483],[853,854],[856,857],[859,860],[862,863],[865,866],[868,372],[870,871],[873,874],[876,877],{},{"id":880,"data":2486,"type":42,"tunes":2487},{"text":882,"level":247},{},{"id":885,"data":2489,"type":314,"tunes":2501},{"steps":2490,"title":918,"orientation":313},[2491,2492,2493,2494,2495,2496,2497,2498,2499,2500],{"label":889,"description":890},{"label":892,"description":893},{"label":895,"description":896},{"label":898,"description":899},{"label":901,"description":902},{"label":904,"description":905},{"label":907,"description":908},{"label":910,"description":911},{"label":913,"description":914},{"label":916,"description":917},{},{"id":921,"data":2503,"type":42,"tunes":2504},{"text":923,"level":247},{},{"id":926,"data":2506,"type":362,"tunes":2519},{"content":2507,"stretched":43,"withHeadings":14},[2508,2509,2510,2511,2512,2513,2514,2515,2516,2517,2518],[930,931],[933,934],[936,937],[939,940],[942,943],[945,946],[948,949],[951,952],[954,955],[957,958],[960,961],{},{"id":964,"data":2521,"type":42,"tunes":2522},{"text":966,"level":247},{},{"id":969,"data":2524,"type":218,"tunes":2525},{"text":971},{},{"id":974,"data":2527,"type":218,"tunes":2528},{"text":976},{},{"id":979,"data":2530,"type":218,"tunes":2531},{"text":981},{},{"id":984,"data":2533,"type":218,"tunes":2534},{"text":986},{},{"id":989,"data":2536,"type":218,"tunes":2537},{"text":991},{},{"id":994,"data":2539,"type":42,"tunes":2540},{"text":996,"level":247},{},{"id":999,"data":2542,"type":218,"tunes":2543},{"text":1001},{},{"id":1004,"data":2545,"type":218,"tunes":2546},{"text":1006},{},{"id":1009,"data":2548,"type":218,"tunes":2549},{"text":1011},{},{"id":1014,"data":2551,"type":42,"tunes":2552},{"text":1016,"level":247},{},{"id":1019,"data":2554,"type":218,"tunes":2555},{"text":1021},{},{"id":1024,"data":2557,"type":1030,"tunes":2558},{"url":1026,"title":1027,"excerpt":1028,"ctaLabel":1029},{},{"id":1033,"data":2560,"type":218,"tunes":2561},{"text":1035},{},{"id":1038,"data":2563,"type":1030,"tunes":2564},{"url":1040,"title":1041,"excerpt":1042,"ctaLabel":1043},{},{"id":1046,"data":2566,"type":218,"tunes":2567},{"text":1048},{},{"id":1051,"data":2569,"type":42,"tunes":2570},{"text":1053,"level":247},{},{"id":1056,"data":2572,"type":1056,"tunes":2582},{"items":2573,"title":1091},[2574,2575,2576,2577,2578,2579,2580,2581],{"id":1060,"answer":1061,"question":1062},{"id":1064,"answer":1065,"question":1066},{"id":1068,"answer":1069,"question":1070},{"id":1072,"answer":1073,"question":1074},{"id":1076,"answer":1077,"question":1078},{"id":1080,"answer":1081,"question":1082},{"id":1084,"answer":1085,"question":1086},{"id":1088,"answer":1089,"question":1090},{},{"id":1094,"data":2584,"type":42,"tunes":2585},{"text":1096,"level":247},{},{"id":1099,"data":2587,"type":1099,"tunes":2603},{"title":1101,"entries":2588},[2589,2590,2591,2592,2593,2594,2595,2596,2597,2598,2599,2600,2601,2602],{"term":366,"anchor":365,"definition":1104},{"term":1106,"anchor":1107,"definition":1108},{"term":1110,"anchor":1111,"definition":1112},{"term":1114,"anchor":1115,"definition":1116},{"term":1118,"anchor":1119,"definition":1120},{"term":1122,"anchor":1123,"definition":1124},{"term":1126,"anchor":1127,"definition":1128},{"term":1130,"anchor":1131,"definition":1132},{"term":1134,"anchor":1135,"definition":1136},{"term":687,"anchor":1138,"definition":1139},{"term":1141,"anchor":1142,"definition":1143},{"term":1145,"anchor":1146,"definition":1147},{"term":681,"anchor":1149,"definition":1150},{"term":1152,"anchor":1153,"definition":1154},{},{"id":1157,"data":2605,"type":42,"tunes":2606},{"text":1159,"level":247},{},{"id":1162,"data":2608,"type":218,"tunes":2609},{"text":1164},{},{"id":1167,"data":2611,"type":218,"tunes":2612},{"text":1169},{},{"id":1172,"data":2614,"type":218,"tunes":2615},{"text":1174},{},{"id":1177,"data":2617,"type":42,"tunes":2618},{"text":1179,"level":247},{},{"id":1182,"data":2620,"type":218,"tunes":2621},{"text":1184},{},{"id":1187,"data":2623,"type":1194,"tunes":2626},{"link":1189,"meta":2624},{"image":2625,"title":1192,"description":1193},{"url":344},{},{"id":1197,"data":2628,"type":1194,"tunes":2631},{"link":1199,"meta":2629},{"image":2630,"title":1202,"description":1203},{"url":344},{},{"id":1206,"data":2633,"type":1194,"tunes":2636},{"link":1208,"meta":2634},{"image":2635,"title":1211,"description":1212},{"url":344},{},{"id":1215,"data":2638,"type":1194,"tunes":2641},{"link":1217,"meta":2639},{"image":2640,"title":1220,"description":1221},{"url":344},{},{"id":1224,"data":2643,"type":1194,"tunes":2646},{"link":1226,"meta":2644},{"image":2645,"title":1229,"description":1230},{"url":344},{},{"id":1233,"data":2648,"type":1194,"tunes":2651},{"link":1235,"meta":2649},{"image":2650,"title":1238,"description":1239},{"url":344},{},{"id":1242,"data":2653,"type":1194,"tunes":2656},{"link":1244,"meta":2654},{"image":2655,"title":1247,"description":1248},{"url":344},{},"Post erfolgreich abgerufen",{"items":2659,"source":2744,"manualIds":2745,"manualMatchedIds":2746},[2660,2667,2674,2681,2688,2695,2702,2709,2716,2723,2730,2737],{"id":2661,"slug":2662,"title":2663,"excerpt":2664,"featuredImage":2665,"publishedAt":2666},"494","air-gapped-ai-how-ai-systems-work-without-internet-or-cloud-access","气隙AI：AI系统如何在没有互联网或云访问的情况下工作","气隙AI在隔离的安全域内运行模型、RAG和AI应用，无需互联网或云依赖。了解模型、数据、更新和工具如何离线运行。","\u002Fuploads\u002F2026\u002F10\u002Fair-gapped-ai-how-ai-systems-work-without-internet-or-cloud-access-1791487983978-e6xqf0.webp","2026-10-08T11:32:00.000Z",{"id":2668,"slug":2669,"title":2670,"excerpt":2671,"featuredImage":2672,"publishedAt":2673},"363","front-und-backend-entwicklung","前端与后端开发","前端和后端开发是网络开发的重要组成部分，涉及创建网络应用程序和网站。前端开发专注于用户界面，而后端开发则负责编程和管理服务器端。","\u002Fuploads\u002F2026\u002F03\u002Ffront-und-backend-entwicklung-1774872219531-wyu4i1.webp","2023-04-12T11:11:00.000Z",{"id":2675,"slug":2676,"title":2677,"excerpt":2678,"featuredImage":2679,"publishedAt":2680},"472","why-more-context-can-make-ai-answers-worse","为什么更多上下文会让AI的回答更糟","更大的上下文窗口并不保证更好的答案。本文解释了信号稀释、证据冲突、状态过时、位置敏感性和有损压缩如何降低AI可靠性——并介绍了一种实用的上下文压力测试。","\u002Fuploads\u002F2026\u002F09\u002Fwhy-more-context-can-make-ai-answers-worse-1790351615793-2ntv2v.webp","2026-09-25T11:51:00.000Z",{"id":2682,"slug":2683,"title":2684,"excerpt":2685,"featuredImage":2686,"publishedAt":2687},"485","enterprise-ai-architecture-what-changes-when-ai-enters-a-company","企业AI架构：当AI进入公司时会发生什么变化","企业AI架构阐释了AI如何在数据权限、身份、许可、提供商、风险、治理、评估、合规和运营方面改变公司系统。","\u002Fuploads\u002F2026\u002F10\u002Fenterprise-ai-architecture-what-changes-when-ai-enters-a-company-1791478161363-czrwaq.webp","2026-10-08T10:48:00.000Z",{"id":2689,"slug":2690,"title":2691,"excerpt":2692,"featuredImage":2693,"publishedAt":2694},"484","what-is-an-ai-platform-architect-models-data-runtime-security-and-operations","什么是AI平台架构师？模型、数据、运行时、安全与运维","AI平台架构师负责跨模型、提供商、检索、智能体、身份、安全、评估、可观测性和运营设计可复用的AI基础。","\u002Fuploads\u002F2026\u002F10\u002Fwhat-is-an-ai-platform-architect-models-data-runtime-security-and-operations-1791477229171-ou3zcc.webp","2026-10-08T12:32:00.000Z",{"id":2696,"slug":2697,"title":2698,"excerpt":2699,"featuredImage":2700,"publishedAt":2701},"495","sovereign-ai-control-of-models-data-infrastructure-and-dependencies","主权人工智能：模型、数据、基础设施与依赖关系的控制","主权人工智能关乎对模型、数据、基础设施、软件、运营和战略依赖的有效控制——而不仅仅是人工智能模型托管在哪里。","\u002Fuploads\u002F2026\u002F10\u002Fsovereign-ai-control-of-models-data-infrastructure-and-dependencies-1791488833132-niy85x.webp","2026-10-08T15:45:00.000Z",{"id":2703,"slug":2704,"title":2705,"excerpt":2706,"featuredImage":2707,"publishedAt":2708},"490","rbac-vs-tenant-isolation-two-different-security-boundaries","RBAC与租户隔离：两种不同的安全边界","RBAC 控制用户可以做什么；租户隔离控制该操作可以触及哪个租户的资源。了解为什么多租户 SaaS 安全需要这两道边界。","\u002Fuploads\u002F2026\u002F10\u002Frbac-vs-tenant-isolation-two-different-security-boundaries-1791485111528-qqtzby.webp","2026-10-08T14:43:00.000Z",{"id":2710,"slug":2711,"title":2712,"excerpt":2713,"featuredImage":2714,"publishedAt":2715},"480","when-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger","人工智能何时应停止信任自身知识？——检索触发机制","AI 模型并非每个问题都需要检索。重要的问题在于知道何时其内部知识已不再足够。检索触发器是一个实用的决策边界，它决定 AI 系统何时应停止仅依赖模型知识，并在回答前获取外部证据。","\u002Fuploads\u002F2026\u002F09\u002Fwhen-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger-1790574991244-f4rpyg.webp","2026-09-28T01:49:00.000Z",{"id":2717,"slug":2718,"title":2719,"excerpt":2720,"featuredImage":2721,"publishedAt":2722},"467","the-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers","答案有效性边界：相关性到可靠AI答案之间缺失的层级","一个来源可能相关、权威，但对于所提出的问题仍然是错误的。缺失的层次是适用性：答案成立的条件，以及迫使其被重新考虑的变化。本文介绍了“答案有效性边界”这一面向人类、AI搜索和RAG系统的来源设计模式。","\u002Fuploads\u002F2026\u002F09\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers-1790272901306-1g5jly.webp","2026-09-24T11:59:00.000Z",{"id":2724,"slug":2725,"title":2726,"excerpt":2727,"featuredImage":2728,"publishedAt":2729},"489","agentic-ai-explained-when-an-ai-system-can-plan-use-tools-and-act","智能体AI解析：当AI系统能够规划、使用工具并采取行动","代理式AI在多步执行循环中使用模型，这些模型可以在明确的运行时和权限边界内选择工具、观察结果、更新状态并调整其下一步行动。","\u002Fuploads\u002F2026\u002F10\u002Fagentic-ai-explained-when-an-ai-system-can-plan-use-tools-and-act-1791481499084-wnji2a.webp","2026-10-08T11:43:00.000Z",{"id":2731,"slug":2732,"title":2733,"excerpt":2734,"featuredImage":2735,"publishedAt":2736},"468","ai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context","AI代理记忆不是RAG：如何区分记忆、检索、状态和上下文","代理记忆、RAG、状态和上下文经常被当作可以互换的概念来使用。它们并不是。这个实用的架构模型将这四个层次区分开来，展示了每一层各自应处的位置，并解释了当系统将它们合并为一层时会出现什么问题。","\u002Fuploads\u002F2026\u002F09\u002Fai-agent-memory-is-not-rag-how-to-separate-memory-retrieval-state-and-context-1790350560308-np0xy6.webp","2026-09-25T11:34:00.000Z",{"id":2738,"slug":2739,"title":2740,"excerpt":2741,"featuredImage":2742,"publishedAt":2743},"478","what-is-rag-the-simplest-explanation-of-how-it-works","什么是RAG？对其工作原理的最简单解释","RAG听起来很复杂，但想法很简单：在AI回答之前，它先从知识源查找有用的信息，并将该信息提供给语言模型。本指南使用一个简单的思维模型来解释RAG、LLM、状态、记忆和工具。","\u002Fuploads\u002F2026\u002F09\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works-1790377492124-khjagt.webp","2026-09-25T19:03:00.000Z","fallback",[],[]]