From fd1cc3a60dd1a29f93bde1568a56596c0be7fa04 Mon Sep 17 00:00:00 2001 From: mrsibe Date: Mon, 28 Sep 2026 18:14:18 +0800 Subject: [PATCH] feat(retrieval): fusion candidate layer, BM25/BM25+RRF strategies, and the experiment MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit #77 asked whether BM25, hybrid or a reranker beats dense, measured on the frozen chunk baseline. Implementing it exposed a bug that made BM25 silently useless. The experiment (`scripts/eval-retrieval.mjs`, `docs/eval/retrieval-v1.5.md`) runs the real harness once per strategy, chunking held at 1000/100: strategy R@1 R@5 MRR nDCG@10 p95 dense 0.8333 1.0000 0.9278 0.9437 18.35 ms sparse 0.7333 0.8667 0.8056 0.8184 2.73 ms hybrid 0.8667 1.0000 0.9444 0.9561 28.52 ms Hybrid is better on Recall@1, MRR and nDCG@10, but the frozen rule requires Recall@5 to *improve*, and dense is already saturated at 1.0000 — no strategy can meet that condition on this corpus. **Dense stays the default**, and the result is recorded as inconclusive rather than adopted or rejected on a metric that cannot move. The report says this plainly. The bug: `buildFtsMatchQuery` ANDed the terms, which is the right default for a lookup box but wrong for a question. A natural-language question's terms virtually never all appear in one chunk, so sparse scored 0.0000 on every metric and hybrid silently degenerated to dense — a "hybrid" that was dense with extra latency. Terms are now ORed; BM25 still ranks a chunk matching more terms higher. - `candidates.ts`: `rrfFuse` over `chunkId + rank` (cosine and BM25 are not comparable, which is why only ranks are used). - `HybridRetriever` serves dense / sparse / hybrid from one path; `DenseRetriever` gained `candidateHits()` so the dense channel is not duplicated. - `RetrievalRequest.strategy`, `SearchOptions.strategy`, `--eval-retrieval=`. - Reranking is **not evaluated**: a cross-encoder model is not available offline and inventing its numbers would defeat the harness. Stated in the report. Verified: npm run typecheck; npm test (398 pass, incl. RRF tests); npm run check:design; npm run build; `npm run eval` byte-identical to baseline-v1.5.json (dense is still the default); `electron . --smoke-test` PASS (27 checks). --- .prettierignore | 4 + docs/eval/baseline-v1.5.md | 4 +- docs/eval/retrieval-v1.5.json | 48 ++++ docs/eval/retrieval-v1.5.md | 47 ++++ package.json | 3 +- scripts/eval-retrieval.mjs | 224 ++++++++++++++++++ src/main/eval/harness.ts | 8 +- src/main/eval/run.ts | 13 +- src/main/services/KnowledgeService.ts | 14 +- src/main/services/ftsSql.ts | 11 +- src/main/services/retrieval/DenseRetriever.ts | 20 +- .../services/retrieval/HybridRetriever.ts | 69 ++++++ src/main/services/retrieval/candidates.ts | 52 ++++ src/main/services/retrieval/index.ts | 2 + src/main/services/retrieval/types.ts | 10 + test/fts.test.ts | 15 +- test/retrievalFusion.test.ts | 72 ++++++ 17 files changed, 592 insertions(+), 24 deletions(-) create mode 100644 docs/eval/retrieval-v1.5.json create mode 100644 docs/eval/retrieval-v1.5.md create mode 100644 scripts/eval-retrieval.mjs create mode 100644 src/main/services/retrieval/HybridRetriever.ts create mode 100644 src/main/services/retrieval/candidates.ts create mode 100644 test/retrievalFusion.test.ts diff --git a/.prettierignore b/.prettierignore index 06feaef..1a115ff 100644 --- a/.prettierignore +++ b/.prettierignore @@ -18,3 +18,7 @@ docs/eval/baseline-*.md # with `npm run eval:chunking`; the formatter must not be a second writer. docs/eval/chunking-*.json docs/eval/chunking-*.md + +# And the same again for `scripts/eval-retrieval.mjs` (#77). +docs/eval/retrieval-*.json +docs/eval/retrieval-*.md diff --git a/docs/eval/baseline-v1.5.md b/docs/eval/baseline-v1.5.md index ecf8ce9..fde5928 100644 --- a/docs/eval/baseline-v1.5.md +++ b/docs/eval/baseline-v1.5.md @@ -25,8 +25,8 @@ Generated by `npm run eval`. The numbers below are harness output — do not edi | nDCG@10 | 0.9437 | | Evidence precision@5 | 0.2133 | -Timing is informational only and is **not** frozen: indexing 2515 ms, query -p50 13.99 ms, p95 23.62 ms on the +Timing is informational only and is **not** frozen: indexing 2235 ms, query +p50 15.12 ms, p95 26.41 ms on the machine that produced this file. Timing and index size depend on hardware and on the corpus, so they must never be the reason two runs differ. diff --git a/docs/eval/retrieval-v1.5.json b/docs/eval/retrieval-v1.5.json new file mode 100644 index 0000000..a26faf6 --- /dev/null +++ b/docs/eval/retrieval-v1.5.json @@ -0,0 +1,48 @@ +{ + "baseline": "dense", + "chunking": "1000/100", + "strategies": [ + { + "id": "dense", + "label": "dense (vector)", + "chunking": "1000/100", + "chunkCount": 19, + "recallAt1": 0.833333, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 0.927778, + "ndcgAt10": 0.94375, + "evidencePrecisionAt5": 0.213333, + "indexingMs": 2041, + "latencyP95Ms": 30.33 + }, + { + "id": "sparse", + "label": "sparse (BM25)", + "chunking": "1000/100", + "chunkCount": 19, + "recallAt1": 0.733333, + "recallAt5": 0.866667, + "recallAt10": 0.866667, + "mrr": 0.805556, + "ndcgAt10": 0.818355, + "evidencePrecisionAt5": 0.186667, + "indexingMs": 2070, + "latencyP95Ms": 2.5 + }, + { + "id": "hybrid", + "label": "hybrid (RRF of dense + BM25)", + "chunking": "1000/100", + "chunkCount": 19, + "recallAt1": 0.866667, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 0.944444, + "ndcgAt10": 0.956053, + "evidencePrecisionAt5": 0.213333, + "indexingMs": 2134, + "latencyP95Ms": 41.23 + } + ] +} diff --git a/docs/eval/retrieval-v1.5.md b/docs/eval/retrieval-v1.5.md new file mode 100644 index 0000000..8a3d64e --- /dev/null +++ b/docs/eval/retrieval-v1.5.md @@ -0,0 +1,47 @@ +# Retrieval experiments — v1.5 (#77) + +Generated by `node scripts/eval-retrieval.mjs`. Numbers are harness output; do not edit them by hand. + +## What was measured + +Every strategy runs the real RAG eval harness against the same corpus and the same 30 +questions as `baseline-v1.5.json`, with chunking held fixed at 1000/100. Only the retrieval strategy changes. + +| Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | Evidence P@5 | Query p95 | +| --- | --- | --- | --- | --- | --- | --- | +| dense (vector) | 0.8333 | 1.0000 | 0.9278 | 0.9437 | 0.2133 | 30.33 ms | +| sparse (BM25) | 0.7333 | 0.8667 | 0.8056 | 0.8184 | 0.1867 | 2.50 ms | +| hybrid (RRF of dense + BM25) | 0.8667 | 1.0000 | 0.9444 | 0.9561 | 0.2133 | 41.23 ms | + +## Not evaluated + +**Reranking.** The issue lists "hybrid + reranker" as a step, but a cross-encoder +model is not available offline and inventing its numbers would defeat the point of +the harness. It stays open until a model can be pinned the way the embedding model +is. + +## Corpus limitation + +The rule's Recall@5 condition is **saturated** on this corpus: dense already scores +1.0000, so no strategy can improve it and the rule can therefore never be met here. +The metrics that still discriminate are Recall@1, MRR and nDCG@10. A hybrid result +that is better on all three but equal on Recall@5 is therefore *inconclusive*, not a +negative result, and the default is left unchanged until the comparison can run on a +corpus where Recall@5 is not already perfect. + +## Adoption rule + +> Adopt a change only if Recall@5 improves and nDCG@10 does not regress. A change +> that trades a large latency increase for a marginal recall gain is a product +> decision, not an automatic win. + +## Outcome + +Recall@5 is saturated at 1.0000, so the rule's first condition cannot be met by any strategy. **Dense stays the default**, and the non-dense strategies are reported as inconclusive rather than adopted or rejected on a metric that cannot move. + +## Reproduce + +```bash +npm run eval:prepare # one-time, networked model bootstrap +npm run eval:retrieval # offline; runs every strategy and rewrites this file +``` diff --git a/package.json b/package.json index d5ebbca..2bc9bec 100644 --- a/package.json +++ b/package.json @@ -38,7 +38,8 @@ "db:generate": "drizzle-kit generate", "db:migrate": "drizzle-kit migrate", "db:push": "drizzle-kit push", - "db:studio": "drizzle-kit studio" + "db:studio": "drizzle-kit studio", + "eval:retrieval": "npm run build && node scripts/eval-retrieval.mjs" }, "//test": [ "`node --test` strips TypeScript types rather than compiling them, and strip-only", diff --git a/scripts/eval-retrieval.mjs b/scripts/eval-retrieval.mjs new file mode 100644 index 0000000..d573e3b --- /dev/null +++ b/scripts/eval-retrieval.mjs @@ -0,0 +1,224 @@ +#!/usr/bin/env node +/** + * Retrieval experiments for #77. + * + * Runs the real RAG eval harness once per retrieval strategy against the frozen + * chunk baseline (`baseline-v1.5.json`, 1000/100), holding chunking fixed, and + * writes the comparison the issue asks for as a delta against dense. + * + * The harness does the measuring; this script only orchestrates and tabulates. + * + * Usage: + * node scripts/eval-retrieval.mjs + * + * The embedding model must already be prepared (`npm run eval:prepare`). + */ + +import { spawn } from 'node:child_process' +import { existsSync, mkdtempSync, mkdirSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join, resolve } from 'node:path' + +const OUT_MD = resolve(readArg('--out=', 'docs/eval/retrieval-v1.5.md')) +const OUT_JSON = OUT_MD.replace(/\.md$/, '.json') + +/** + * `dense` is the shipped default and the comparison baseline. `sparse` is BM25 over + * the shared FTS index (#96); `hybrid` is RRF of the two. A reranker is not + * evaluated: it needs a cross-encoder model, which this offline harness does not + * have, and pretending otherwise would be a number invented by the script. + */ +const STRATEGIES = [ + { id: 'dense', label: 'dense (vector)' }, + { id: 'sparse', label: 'sparse (BM25)' }, + { id: 'hybrid', label: 'hybrid (RRF of dense + BM25)' } +] + +function readArg(prefix, fallback) { + const arg = process.argv.find((value) => value.startsWith(prefix)) + return arg ? arg.slice(prefix.length) : fallback +} + +const executable = resolve( + 'node_modules/.bin', + process.platform === 'win32' ? 'electron.cmd' : 'electron' +) + +if (!existsSync(executable)) { + console.error('[retrieval] could not find the electron binary. Run `npm install` first.') + process.exit(1) +} + +function runStrategy(strategy, outDir) { + return new Promise((resolvePromise, reject) => { + const args = [ + '.', + '--eval-harness', + '--eval-baseline=v1.5', + `--eval-out=${outDir}`, + `--eval-retrieval=${strategy.id}` + ] + + const isRoot = typeof process.getuid === 'function' && process.getuid() === 0 + if (isRoot || process.env.CI) args.push('--no-sandbox') + + const child = spawn(executable, args, { + stdio: ['ignore', 'pipe', 'pipe'], + env: { ...process.env, ELECTRON_DISABLE_SECURITY_WARNINGS: '1' } + }) + + let stdout = '' + child.stdout.on('data', (data) => { + stdout += data.toString() + }) + child.stderr.on('data', () => {}) + + child.on('error', reject) + child.on('exit', (code) => { + if (code !== 0) { + reject(new Error(`strategy ${strategy.id} exited with code ${code}`)) + return + } + const metricsLine = /\[eval\] metrics (\{.*\})/.exec(stdout) + if (!metricsLine) { + reject(new Error(`strategy ${strategy.id} printed no metrics line`)) + return + } + try { + resolvePromise(JSON.parse(metricsLine[1])) + } catch (error) { + reject( + new Error(`strategy ${strategy.id} printed an unreadable metrics line: ${error.message}`) + ) + } + }) + }) +} + +/** Throughput and p95 are informational and excluded from the deterministic JSON. */ +function readTiming(mdPath) { + if (!existsSync(mdPath)) return { indexingMs: null, latencyP95Ms: null } + const text = readFileSync(mdPath, 'utf8') + const indexing = /indexing (\d+) ms/.exec(text) + const p95 = /p95 ([\d.]+) ms/.exec(text) + return { + indexingMs: indexing ? Number(indexing[1]) : null, + latencyP95Ms: p95 ? Number(p95[1]) : null + } +} + +const workDir = mkdtempSync(join(tmpdir(), 'knownote-retrieval-')) +const results = [] + +try { + for (const strategy of STRATEGIES) { + const outDir = join(workDir, strategy.id) + mkdirSync(outDir, { recursive: true }) + console.log(`[retrieval] running ${strategy.label}`) + const metrics = await runStrategy(strategy, outDir) + + const report = JSON.parse(readFileSync(join(outDir, 'baseline-v1.5.json'), 'utf8')) + results.push({ + id: strategy.id, + label: strategy.label, + chunking: `${report.config.chunking.chunkSize}/${report.config.chunking.chunkOverlap}`, + chunkCount: report.config.chunkCount, + ...metrics, + ...readTiming(join(outDir, 'baseline-v1.5.md')) + }) + } +} finally { + rmSync(workDir, { recursive: true, force: true }) +} + +const baseline = results.find((result) => result.id === 'dense') +if (!baseline) throw new Error('the dense strategy did not run') + +const format4 = (value) => value.toFixed(4) + +const rows = results.map( + (result) => + `| ${result.label} | ${format4(result.recallAt1)} | ${format4(result.recallAt5)} | ${format4(result.mrr)} | ${format4(result.ndcgAt10)} | ${format4(result.evidencePrecisionAt5)} | ${result.latencyP95Ms?.toFixed(2)} ms |` +) + +/** + * The rule frozen in the baseline report: Recall@5 must improve and nDCG@10 must + * not regress. Latency is reported so a win that costs 5x latency is stated as a + * trade-off, not hidden. + */ +const adopted = results.filter( + (result) => + result.id !== 'dense' && + result.recallAt5 > baseline.recallAt5 && + result.ndcgAt10 >= baseline.ndcgAt10 +) +const winner = + adopted.sort((a, b) => b.recallAt5 - a.recallAt5 || b.ndcgAt10 - a.ndcgAt10)[0] ?? null + +let outcome +if (winner) { + outcome = `\`${winner.label}\` clears the rule (Recall@5 ${format4(winner.recallAt5)} vs dense ${format4(baseline.recallAt5)}, nDCG@10 ${format4(winner.ndcgAt10)} vs ${format4(baseline.ndcgAt10)}).` +} else if (baseline.recallAt5 === 1) { + outcome = `Recall@5 is saturated at 1.0000, so the rule's first condition cannot be met by any strategy. **Dense stays the default**, and the non-dense strategies are reported as inconclusive rather than adopted or rejected on a metric that cannot move.` +} else { + outcome = `No strategy cleared the rule. **Dense stays the default.** A negative result is the point of the experiment: it is the measurement that says the extra machinery is not worth its cost on this corpus, not a failure to deliver.` +} + +const markdown = `# Retrieval experiments — v1.5 (#77) + +Generated by \`node scripts/eval-retrieval.mjs\`. Numbers are harness output; do not edit them by hand. + +## What was measured + +Every strategy runs the real RAG eval harness against the same corpus and the same 30 +questions as \`baseline-v1.5.json\`, with chunking held fixed at ${baseline.chunking}. Only the retrieval strategy changes. + +| Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | Evidence P@5 | Query p95 | +| --- | --- | --- | --- | --- | --- | --- | +${rows.join('\n')} + +## Not evaluated + +**Reranking.** The issue lists "hybrid + reranker" as a step, but a cross-encoder +model is not available offline and inventing its numbers would defeat the point of +the harness. It stays open until a model can be pinned the way the embedding model +is. + +## Corpus limitation + +The rule's Recall@5 condition is **saturated** on this corpus: dense already scores +1.0000, so no strategy can improve it and the rule can therefore never be met here. +The metrics that still discriminate are Recall@1, MRR and nDCG@10. A hybrid result +that is better on all three but equal on Recall@5 is therefore *inconclusive*, not a +negative result, and the default is left unchanged until the comparison can run on a +corpus where Recall@5 is not already perfect. + +## Adoption rule + +> Adopt a change only if Recall@5 improves and nDCG@10 does not regress. A change +> that trades a large latency increase for a marginal recall gain is a product +> decision, not an automatic win. + +## Outcome + +${outcome} + +## Reproduce + +\`\`\`bash +npm run eval:prepare # one-time, networked model bootstrap +npm run eval:retrieval # offline; runs every strategy and rewrites this file +\`\`\` +` + +mkdirSync(resolve(OUT_MD, '..'), { recursive: true }) +writeFileSync( + OUT_JSON, + `${JSON.stringify({ baseline: baseline.id, chunking: baseline.chunking, strategies: results }, null, 2)}\n` +) +writeFileSync(OUT_MD, markdown) + +console.log(`[retrieval] wrote ${OUT_JSON} and ${OUT_MD}`) +console.log( + `[retrieval] ${winner ? `best clearing strategy: ${winner.label}` : 'no strategy cleared the rule; keep dense'}` +) diff --git a/src/main/eval/harness.ts b/src/main/eval/harness.ts index 713a4ed..a992415 100644 --- a/src/main/eval/harness.ts +++ b/src/main/eval/harness.ts @@ -18,6 +18,7 @@ import { documentBlocks, notebooks, chunks } from '../db/schema' import type { getDatabase } from '../db' import type { KnowledgeService } from '../services/KnowledgeService' import { DEFAULT_CHUNK_OPTIONS, type ChunkOptions } from '../services/ChunkingService' +import type { RetrievalStrategy } from '../services/retrieval' import { LOCAL_EMBEDDING_MODEL } from '../embedding/localModel' import { evidencePrecisionAtK, @@ -54,6 +55,8 @@ export interface EvalHarnessOptions { evidenceK: number /** 分块配置(#78)。实验变体通过它选择策略;缺省时用生产默认值。 */ chunkOptions: ChunkOptions + /** 检索策略(#77):dense / sparse(BM25) / hybrid(RRF)。 */ + strategy: RetrievalStrategy } const NOTEBOOK_ID = 'eval-notebook' @@ -189,7 +192,8 @@ export async function runEvalHarness( const started = performance.now() const results = await knowledgeService.search(NOTEBOOK_ID, question.question, { topK: options.topK, - threshold: options.threshold + threshold: options.threshold, + strategy: options.strategy }) const latencyMs = performance.now() - started latencies.push(latencyMs) @@ -235,7 +239,7 @@ export async function runEvalHarness( allowSpanPages: chunking.allowSpanPages, respectHeadings: chunking.respectHeadings }, - retrieval: 'dense', + retrieval: options.strategy, topK: options.topK, threshold: options.threshold, evidenceK: options.evidenceK, diff --git a/src/main/eval/run.ts b/src/main/eval/run.ts index 1548608..10857a6 100644 --- a/src/main/eval/run.ts +++ b/src/main/eval/run.ts @@ -21,6 +21,7 @@ import { EmbeddingService } from '../services/EmbeddingService' import { KnowledgeService } from '../services/KnowledgeService' import { isModelInstalled } from '../embedding/ModelRegistry' import { DEFAULT_CHUNK_OPTIONS, type ChunkOptions } from '../services/ChunkingService' +import type { RetrievalStrategy } from '../services/retrieval' import { runEvalHarness, stabilize, @@ -61,6 +62,15 @@ function readBoolOption(argv: readonly string[], prefix: string, fallback: boole return value === 'true' } +/** 检索策略(#77)。默认 dense,所以不带 flag 的 `npm run eval` 仍量的是生产默认。 */ +function readRetrievalStrategy(argv: readonly string[]): RetrievalStrategy { + const raw = readOption(argv, '--eval-retrieval=', 'dense') + if (raw !== 'dense' && raw !== 'sparse' && raw !== 'hybrid') { + throw new Error(`--eval-retrieval expects dense, sparse or hybrid, got ${JSON.stringify(raw)}`) + } + return raw +} + /** * Chunking config for one run (#78). The defaults are the production defaults, so * `npm run eval` with no flags still measures what ships. @@ -160,7 +170,8 @@ export async function runEvalCli(argv: readonly string[] = process.argv): Promis topK: 10, threshold: 0, evidenceK: 5, - chunkOptions: readChunkOptions(argv) + chunkOptions: readChunkOptions(argv), + strategy: readRetrievalStrategy(argv) } const report = stabilize(await runEvalHarness(getDatabase(), knowledgeService, options)) diff --git a/src/main/services/KnowledgeService.ts b/src/main/services/KnowledgeService.ts index 3da76a6..597a301 100644 --- a/src/main/services/KnowledgeService.ts +++ b/src/main/services/KnowledgeService.ts @@ -43,7 +43,7 @@ import { type IdentifiedBlockDraft } from './blocks/documentBlocks' import { resolveChunkProvenance, type ChunkProvenance } from './chunkProvenance' -import { DenseRetriever, hydrateEvidence } from './retrieval' +import { HybridRetriever, hydrateEvidence, type RetrievalStrategy } from './retrieval' import { advanceRun, completeRun, @@ -60,12 +60,7 @@ import type { Retriever } from './retrieval' import { WebFetchService } from './WebFetchService' -import { - deleteDocumentChunksFts, - ensureChunksFts, - indexChunksFts, - searchChunksFts -} from './fts' +import { deleteDocumentChunksFts, ensureChunksFts, indexChunksFts, searchChunksFts } from './fts' import { vectorStoreManager } from '../vectorstore' import Logger from '../../shared/utils/logger' @@ -93,6 +88,8 @@ export interface SearchOptions { includeContent?: boolean // 是否包含 chunk 内容,默认 true /** 只在这些来源里检索(#94);为空/缺省表示整个 notebook。 */ documentIds?: string[] + /** 检索策略(#77);缺省 `dense`,与引入策略之前一致。 */ + strategy?: RetrievalStrategy } /** @@ -163,7 +160,7 @@ export class KnowledgeService { constructor(embeddingService: EmbeddingService) { this.embeddingService = embeddingService this.chunkingService = new ChunkingService() - this.retriever = new DenseRetriever(embeddingService) + this.retriever = new HybridRetriever(embeddingService) this.fileParserService = new FileParserService() this.webFetchService = new WebFetchService() // 知识库文件存储目录 @@ -920,6 +917,7 @@ export class KnowledgeService { query, topK: options.topK, threshold: options.threshold, + strategy: options.strategy, filter: options.documentIds ? { documentIds: options.documentIds } : undefined }) diff --git a/src/main/services/ftsSql.ts b/src/main/services/ftsSql.ts index 2fa9e02..5f28844 100644 --- a/src/main/services/ftsSql.ts +++ b/src/main/services/ftsSql.ts @@ -48,8 +48,13 @@ export function deleteChunksFtsSql(column: 'chunk_id' | 'document_id', count: nu * Every whitespace-separated term is quoted, so the user's text can never be * parsed as FTS operators (`-`, `*`, `:`, `NEAR`, unbalanced quotes) — a query * that is a syntax error is a query that returns nothing, which reads as "no - * results" rather than "your search is malformed". Terms are ANDed, which is - * FTS5's default for space-separated tokens. + * results" rather than "your search is malformed". + * + * Terms are **ORed**. AND was the first version and it made BM25 useless for chat + * retrieval (#77): a natural-language question's terms almost never all appear in + * one chunk, so sparse returned nothing and hybrid silently degenerated to dense. + * BM25 already ranks a chunk that matches more terms higher, so OR keeps the best + * passages first while still finding them. * * Returns null when there is nothing to search for. */ @@ -59,7 +64,7 @@ export function buildFtsMatchQuery(raw: string): string | null { .split(/\s+/) .filter((term) => term.length > 0) if (terms.length === 0) return null - return terms.map((term) => `"${term.replace(/"/g, '""')}"`).join(' ') + return terms.map((term) => `"${term.replace(/"/g, '""')}"`).join(' OR ') } export interface FtsSearchSqlOptions { diff --git a/src/main/services/retrieval/DenseRetriever.ts b/src/main/services/retrieval/DenseRetriever.ts index cf3cf47..a6187a7 100644 --- a/src/main/services/retrieval/DenseRetriever.ts +++ b/src/main/services/retrieval/DenseRetriever.ts @@ -9,6 +9,7 @@ import { getDatabase } from '../../db' import { vectorStoreManager } from '../../vectorstore' import type { EmbeddingService } from '../EmbeddingService' +import type { CandidateHit } from './candidates' import { hydrateEvidence } from './evidence' import { buildRetrievalTrace } from './trace' import type { RetrievalRequest, RetrievalResult, Retriever } from './types' @@ -18,10 +19,16 @@ const STRATEGY = 'dense' export class DenseRetriever implements Retriever { constructor(private readonly embeddingService: EmbeddingService) {} - async search(request: RetrievalRequest): Promise { + /** + * 只做“向量命中”,不补齐证据。 + * + * 抽出来是给 hybrid 用的(#77):融合需要的是 `chunkId + score`,不是已经 join 好 + * 文档/块/偏移的证据。公开方法让 `HybridRetriever` 复用同一条 dense 路径,而不是把 + * embed + KNN 抄一遍。 + */ + async candidateHits(request: RetrievalRequest): Promise { const topK = request.topK ?? 5 const threshold = request.threshold ?? 0.5 - const startedAt = performance.now() // E5 要求 query 前缀,与索引时的 document 前缀区分 await this.embeddingService.ensureReady() @@ -35,6 +42,15 @@ export class DenseRetriever implements Retriever { filter: request.filter }) + return hits.map((hit) => ({ chunkId: hit.chunkId, score: hit.score })) + } + + async search(request: RetrievalRequest): Promise { + const topK = request.topK ?? 5 + const threshold = request.threshold ?? 0.5 + const startedAt = performance.now() + + const hits = await this.candidateHits(request) const evidence = hits.length === 0 ? [] : hydrateEvidence(getDatabase(), hits) return { diff --git a/src/main/services/retrieval/HybridRetriever.ts b/src/main/services/retrieval/HybridRetriever.ts new file mode 100644 index 0000000..bb899f6 --- /dev/null +++ b/src/main/services/retrieval/HybridRetriever.ts @@ -0,0 +1,69 @@ +import { getDatabase } from '../../db' +import type { EmbeddingService } from '../EmbeddingService' +import { searchChunksFts } from '../fts' +import { rrfFuse, type CandidateHit } from './candidates' +import { DenseRetriever } from './DenseRetriever' +import { hydrateEvidence } from './evidence' +import { buildRetrievalTrace } from './trace' +import type { RetrievalRequest, RetrievalResult, RetrievalStrategy, Retriever } from './types' + +/** + * HybridRetriever (#77) + * + * 三条策略共用一条路径,区别只在候选怎么产生: + * + * dense 释义相近的段落(向量) + * sparse 字面出现的段落(BM25 over chunks_fts,#96) + * hybrid RRF 融合两者 + * + * 融合在候选层完成(`chunkId + rank`),证据只补齐一次 —— 见 `candidates.ts`。 + * 两个通道的分数(cosine 与 BM25)不可比,所以只用 rank,这正是 RRF 的意义。 + */ +export class HybridRetriever implements Retriever { + private readonly dense: DenseRetriever + + constructor(embeddingService: EmbeddingService) { + this.dense = new DenseRetriever(embeddingService) + } + + async search(request: RetrievalRequest): Promise { + const strategy: RetrievalStrategy = request.strategy ?? 'dense' + const topK = request.topK ?? 5 + const startedAt = performance.now() + + let hits: CandidateHit[] + // BM25 没有「相似度阈值」这个概念,所以 sparse 的 trace 里 threshold 保持缺省, + // 而不是拿 dense 的 0.5 冒充。 + let threshold: number | undefined + + if (strategy === 'sparse') { + hits = searchChunksFts(request.notebookId, request.query, { + limit: topK, + documentIds: request.filter?.documentIds + }) + } else if (strategy === 'hybrid') { + const denseHits = await this.dense.candidateHits(request) + const sparseHits = searchChunksFts(request.notebookId, request.query, { + limit: topK, + documentIds: request.filter?.documentIds + }) + hits = rrfFuse([denseHits, sparseHits]).slice(0, topK) + } else { + threshold = request.threshold ?? 0.5 + hits = await this.dense.candidateHits({ ...request, threshold }) + } + + const evidence = hits.length === 0 ? [] : hydrateEvidence(getDatabase(), hits) + + return { + evidence, + trace: buildRetrievalTrace({ + strategy, + filter: request.filter, + topK, + threshold, + durationMs: performance.now() - startedAt + }) + } + } +} diff --git a/src/main/services/retrieval/candidates.ts b/src/main/services/retrieval/candidates.ts new file mode 100644 index 0000000..156d403 --- /dev/null +++ b/src/main/services/retrieval/candidates.ts @@ -0,0 +1,52 @@ +/** + * Retrieval candidate layer (#77). + * + * Fusion combines `chunkId + rank`, not hydrated evidence: RRF only needs to know + * where each channel ranked a chunk. Hydrating first would join documents, blocks + * and offsets for every channel and then throw most of it away, so candidates are + * fused as ids and hydrated once at the end. + */ + +/** One channel's hit, before fusion. */ +export interface CandidateHit { + chunkId: string + score: number +} + +/** A fused hit with its final rank. */ +export interface ScoredChunkHit { + chunkId: string + score: number + rank: number +} + +/** The RRF constant from the original paper; it damps the top ranks' advantage. */ +export const RRF_K = 60 + +/** + * Reciprocal Rank Fusion across channels. + * + * A chunk's fused score is `sum(1 / (k + rank))` over the channels that returned + * it, so a chunk ranked well by one channel and absent from another still scores, + * and agreement between channels is what moves a chunk up. Scores are not + * comparable across channels (cosine vs BM25), which is exactly why only the ranks + * are used. + */ +export function rrfFuse( + channels: readonly (readonly CandidateHit[])[], + k: number = RRF_K +): ScoredChunkHit[] { + const fused = new Map() + + for (const channel of channels) { + channel.forEach((hit, index) => { + const contribution = 1 / (k + index + 1) + fused.set(hit.chunkId, (fused.get(hit.chunkId) ?? 0) + contribution) + }) + } + + return [...fused.entries()] + .map(([chunkId, score]) => ({ chunkId, score, rank: 0 })) + .sort((a, b) => b.score - a.score || a.chunkId.localeCompare(b.chunkId)) + .map((hit, index) => ({ ...hit, rank: index + 1 })) +} diff --git a/src/main/services/retrieval/index.ts b/src/main/services/retrieval/index.ts index 9d6eee7..d368f1e 100644 --- a/src/main/services/retrieval/index.ts +++ b/src/main/services/retrieval/index.ts @@ -4,5 +4,7 @@ export * from './types' export { DenseRetriever } from './DenseRetriever' +export { HybridRetriever } from './HybridRetriever' +export { rrfFuse, RRF_K, type CandidateHit, type ScoredChunkHit } from './candidates' export { hydrateEvidence, assembleEvidence, type EvidenceHit } from './evidence' export { buildRetrievalTrace, type RetrievalTraceInput } from './trace' diff --git a/src/main/services/retrieval/types.ts b/src/main/services/retrieval/types.ts index 1145994..060134b 100644 --- a/src/main/services/retrieval/types.ts +++ b/src/main/services/retrieval/types.ts @@ -50,6 +50,14 @@ export interface RetrievalFilter { documentIds?: string[] } +/** + * 检索策略(#77)。 + * + * `dense` 是 v1.4 的默认;`sparse` 是 BM25 over `chunks_fts`(#96);`hybrid` 用 RRF + * 融合两者。语义解释:dense 找释义,sparse 找字面,hybrid 两者都要。 + */ +export type RetrievalStrategy = 'dense' | 'sparse' | 'hybrid' + /** * 一次检索请求。 * @@ -62,6 +70,8 @@ export interface RetrievalRequest { topK?: number threshold?: number filter?: RetrievalFilter + /** 缺省时为 `dense`,与引入策略之前的默认一致。 */ + strategy?: RetrievalStrategy } /** diff --git a/test/fts.test.ts b/test/fts.test.ts index 5392c17..410979e 100644 --- a/test/fts.test.ts +++ b/test/fts.test.ts @@ -17,20 +17,25 @@ import { */ test('a search query is quoted term by term, never parsed as FTS syntax', () => { - assert.equal(buildFtsMatchQuery('hello world'), '"hello" "world"') - assert.equal(buildFtsMatchQuery(' spaced out '), '"spaced" "out"') + assert.equal(buildFtsMatchQuery('hello world'), '"hello" OR "world"') + assert.equal(buildFtsMatchQuery(' spaced out '), '"spaced" OR "out"') // Operators and unbalanced quotes would be a syntax error if passed through, and // a syntax error reads to the user as "no results" rather than "bad query". - assert.equal(buildFtsMatchQuery('a -b'), '"a" "-b"') - assert.equal(buildFtsMatchQuery('NEAR(a b)'), '"NEAR(a" "b)"') - assert.equal(buildFtsMatchQuery('say "hi"'), '"say" """hi"""') + assert.equal(buildFtsMatchQuery('a -b'), '"a" OR "-b"') + assert.equal(buildFtsMatchQuery('NEAR(a b)'), '"NEAR(a" OR "b)"') + assert.equal(buildFtsMatchQuery('say "hi"'), '"say" OR """hi"""') assert.equal(buildFtsMatchQuery('c*'), '"c*"') assert.equal(buildFtsMatchQuery(' '), null) assert.equal(buildFtsMatchQuery(''), null) }) +test('terms are ORed, so a full question still matches some chunks', () => { + // AND made BM25 return nothing for a natural-language question (#77). + assert.match(buildFtsMatchQuery('what does section 3 say about attention') ?? '', / OR /) +}) + test('delete is parameterised by count and refuses an empty list', () => { assert.equal(deleteChunksFtsSql('chunk_id', 2), 'DELETE FROM chunks_fts WHERE chunk_id IN (?, ?)') assert.match(deleteChunksFtsSql('document_id', 1), /WHERE document_id IN \(\?\)/) diff --git a/test/retrievalFusion.test.ts b/test/retrievalFusion.test.ts new file mode 100644 index 0000000..ebacc62 --- /dev/null +++ b/test/retrievalFusion.test.ts @@ -0,0 +1,72 @@ +import { test } from 'node:test' +import assert from 'node:assert/strict' +import { rrfFuse, RRF_K, type CandidateHit } from '../src/main/services/retrieval/candidates.ts' + +/** + * Retrieval fusion (#77). RRF combines ranks, not scores, because a cosine + * similarity and a BM25 score are not on the same scale. These pin the two + * properties the hybrid retriever depends on: agreement between channels wins, and + * a chunk only one channel found still ranks. + */ + +const hit = (chunkId: string, score = 0): CandidateHit => ({ chunkId, score }) + +test('a chunk both channels rank beats one only a single channel found', () => { + const dense = [hit('a'), hit('shared')] + const sparse = [hit('shared'), hit('b')] + + const fused = rrfFuse([dense, sparse]) + + assert.equal(fused[0].chunkId, 'shared') + assert.deepEqual( + fused.map((entry) => entry.rank), + [1, 2, 3] + ) +}) + +test('a chunk found by only one channel still ranks', () => { + const fused = rrfFuse([[hit('only-dense')], [hit('only-sparse')]]) + + assert.equal(fused.length, 2) + // Ranked 1 in its own channel, so both contribute the same amount. + assert.equal(fused[0].score, fused[1].score) +}) + +test('the fused score is the sum of reciprocal ranks', () => { + const fused = rrfFuse([[hit('a'), hit('b')], [hit('b')]]) + + const byId = new Map(fused.map((entry) => [entry.chunkId, entry.score])) + assert.ok(Math.abs((byId.get('a') as number) - 1 / (RRF_K + 1)) < 1e-12) + assert.ok(Math.abs((byId.get('b') as number) - (1 / (RRF_K + 2) + 1 / (RRF_K + 1))) < 1e-12) +}) + +test('an empty channel contributes nothing and does not throw', () => { + const fused = rrfFuse([[], [hit('a')]]) + assert.deepEqual( + fused.map((entry) => entry.chunkId), + ['a'] + ) + assert.deepEqual(rrfFuse([]), []) + assert.deepEqual(rrfFuse([[], []]), []) +}) + +test('scores are not used, only order — a low-scoring rank-1 still wins its channel', () => { + const highScoreButLast = [hit('first', 0.01), hit('second', 0.99)] + const fused = rrfFuse([highScoreButLast]) + + assert.deepEqual( + fused.map((entry) => entry.chunkId), + ['first', 'second'] + ) +}) + +test('a chunk present in both channels outranks a chunk ranked first by one', () => { + // The point of fusion: agreement is stronger evidence than a single channel's + // confidence, even when that channel put its pick first. + const dense = [hit('dense-top'), hit('agreed')] + const sparse = [hit('sparse-top'), hit('agreed')] + + const fused = rrfFuse([dense, sparse]) + + assert.equal(fused[0].chunkId, 'agreed') +})