diff --git a/.prettierignore b/.prettierignore index f6f4ae5..da32ca2 100644 --- a/.prettierignore +++ b/.prettierignore @@ -29,6 +29,10 @@ docs/eval/threshold-*.md docs/eval/sweep-*.json docs/eval/sweep-*.md +# And by `scripts/eval-scores.mjs`. Regenerate with `npm run eval:scores`. +docs/eval/scores-*.json +docs/eval/scores-*.md + # And the same again for `scripts/eval-retrieval.mjs` (#77). docs/eval/retrieval-*.json docs/eval/retrieval-*.md diff --git a/docs/eval/scores-v1.6.json b/docs/eval/scores-v1.6.json new file mode 100644 index 0000000..163c779 --- /dev/null +++ b/docs/eval/scores-v1.6.json @@ -0,0 +1,342 @@ +{ + "baseline": "v1.6", + "split": "validation", + "candidateK": 500, + "threshold": 0, + "indexSize": 53, + "counts": { + "answerable": 19, + "unanswerable": 5 + }, + "distributions": { + "bestRelevant": { + "p0": 0.880777, + "p10": 0.89056, + "p25": 0.912001, + "p50": 0.935891, + "p75": 0.942776, + "p90": 0.953038, + "p100": 0.953664 + }, + "worstRelevant": { + "p0": 0.880777, + "p10": 0.89056, + "p25": 0.908589, + "p50": 0.935467, + "p75": 0.942776, + "p90": 0.953038, + "p100": 0.953664 + }, + "bestNonRelevant": { + "p0": 0.890443, + "p10": 0.908132, + "p25": 0.917792, + "p50": 0.924346, + "p75": 0.933555, + "p90": 0.9386, + "p100": 0.945918 + }, + "margin": { + "p0": -0.02942099999999992, + "p10": -0.027232000000000034, + "p25": -0.00922400000000001, + "p50": 0.0026120000000000587, + "p75": 0.00807100000000005, + "p90": 0.032869999999999955, + "p100": 0.06226200000000004 + }, + "unanswerableMax": { + "p0": 0.897848, + "p10": 0.897848, + "p25": 0.912751, + "p50": 0.925936, + "p75": 0.934276, + "p90": 0.942441, + "p100": 0.942441 + } + }, + "byType": { + "exact": { + "questions": 4, + "bestRelevant": { + "p0": 0.912001, + "p10": 0.912001, + "p25": 0.912001, + "p50": 0.929376, + "p75": 0.936747, + "p90": 0.953038, + "p100": 0.953038 + }, + "worstRelevant": { + "p0": 0.912001, + "p10": 0.912001, + "p25": 0.912001, + "p50": 0.929376, + "p75": 0.936747, + "p90": 0.953038, + "p100": 0.953038 + } + }, + "semantic": { + "questions": 6, + "bestRelevant": { + "p0": 0.920784, + "p10": 0.920784, + "p25": 0.935467, + "p50": 0.939466, + "p75": 0.945637, + "p90": 0.953664, + "p100": 0.953664 + }, + "worstRelevant": { + "p0": 0.920784, + "p10": 0.920784, + "p25": 0.935467, + "p50": 0.939466, + "p75": 0.945637, + "p90": 0.953664, + "p100": 0.953664 + } + }, + "multi-hop": { + "questions": 2, + "bestRelevant": { + "p0": 0.924784, + "p10": 0.924784, + "p25": 0.924784, + "p50": 0.924784, + "p75": 0.93756, + "p90": 0.93756, + "p100": 0.93756 + }, + "worstRelevant": { + "p0": 0.903304, + "p10": 0.903304, + "p25": 0.903304, + "p50": 0.903304, + "p75": 0.933129, + "p90": 0.933129, + "p100": 0.933129 + } + }, + "zh": { + "questions": 1, + "bestRelevant": { + "p0": 0.952705, + "p10": 0.952705, + "p25": 0.952705, + "p50": 0.952705, + "p75": 0.952705, + "p90": 0.952705, + "p100": 0.952705 + }, + "worstRelevant": { + "p0": 0.952705, + "p10": 0.952705, + "p25": 0.952705, + "p50": 0.952705, + "p75": 0.952705, + "p90": 0.952705, + "p100": 0.952705 + } + }, + "cross-lingual": { + "questions": 4, + "bestRelevant": { + "p0": 0.880777, + "p10": 0.880777, + "p25": 0.880777, + "p50": 0.89056, + "p75": 0.904196, + "p90": 0.908589, + "p100": 0.908589 + }, + "worstRelevant": { + "p0": 0.880777, + "p10": 0.880777, + "p25": 0.880777, + "p50": 0.89056, + "p75": 0.904196, + "p90": 0.908589, + "p100": 0.908589 + } + }, + "hard-negative": { + "questions": 2, + "bestRelevant": { + "p0": 0.935891, + "p10": 0.935891, + "p25": 0.935891, + "p50": 0.935891, + "p75": 0.942776, + "p90": 0.942776, + "p100": 0.942776 + }, + "worstRelevant": { + "p0": 0.935891, + "p10": 0.935891, + "p25": 0.935891, + "p50": 0.935891, + "p75": 0.942776, + "p90": 0.942776, + "p100": 0.942776 + } + } + }, + "overlap": { + "worstRelevantP10": 0.89056, + "bestNonRelevantP90": 0.9386, + "unanswerableMaxP50": 0.925936, + "unanswerableMaxP90": 0.942441 + }, + "separable": false, + "curve": [ + { + "threshold": 0.5, + "cosine": 0, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.525, + "cosine": 0.05, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.55, + "cosine": 0.1, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.575, + "cosine": 0.15, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.6, + "cosine": 0.2, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.625, + "cosine": 0.25, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.65, + "cosine": 0.3, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.675, + "cosine": 0.35, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.7, + "cosine": 0.4, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.725, + "cosine": 0.45, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.75, + "cosine": 0.5, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.775, + "cosine": 0.55, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.8, + "cosine": 0.6, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.825, + "cosine": 0.65, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.85, + "cosine": 0.7, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.875, + "cosine": 0.75, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.9, + "cosine": 0.8, + "hitRate": 0.8947368421052632, + "fullRecallRate": 0.8947368421052632, + "abstentionRate": 0.2 + }, + { + "threshold": 0.925, + "cosine": 0.85, + "hitRate": 0.631578947368421, + "fullRecallRate": 0.631578947368421, + "abstentionRate": 0.4 + }, + { + "threshold": 0.95, + "cosine": 0.9, + "hitRate": 0.15789473684210525, + "fullRecallRate": 0.15789473684210525, + "abstentionRate": 1 + }, + { + "threshold": 0.975, + "cosine": 0.95, + "hitRate": 0, + "fullRecallRate": 0, + "abstentionRate": 1 + }, + { + "threshold": 1, + "cosine": 1, + "hitRate": 0, + "fullRecallRate": 0, + "abstentionRate": 1 + } + ] +} diff --git a/docs/eval/scores-v1.6.md b/docs/eval/scores-v1.6.md new file mode 100644 index 0000000..18d8928 --- /dev/null +++ b/docs/eval/scores-v1.6.md @@ -0,0 +1,128 @@ +# Dense score diagnostics — v1.6 (#192) + +Generated by `node scripts/eval-scores.mjs`. Numbers are harness output; do not edit them by hand. + +## Read this first: the score is not a cosine + +`SQLiteVectorStore` computes + +```text +score = 1 - distance / 2 +distance = 1 - cosine (sqlite-vec, distance_metric=cosine) +=> score = (1 + cosine) / 2 +``` + +so the configured `threshold` is an **affine map of the cosine**, not the cosine: + +| threshold | raw cosine | +| --- | --- | +| 0.3 | -0.4 | +| 0.4 | -0.2 | +| **0.5 (shipped)** | **0.0** | +| 0.6 | 0.2 | +| 0.8 | 0.6 | +| 1.0 | 1.0 | + +The shipped `threshold = 0.5` means **cosine ≥ 0**, which is very permissive. Every +threshold row below carries both columns so the two never get confused again, and every +mention of "the sweep looked too low" has to be read through this table. + +## What was measured + +Dense only, `validation` split only, `threshold = 0`, `candidateK = 500` (above the +index size, so every chunk is scored for every query), `contextK = 3`. + +- answerable questions: 19 +- unanswerable questions: 5 +- index size: 53 chunks + +Hybrid is deliberately excluded: its `score` is an RRF value (`1 / (60 + rank)`) and is not +on the same scale as a normalised cosine. + +## Distributions (score, with raw cosine in brackets) + +`best relevant` is the highest-scoring passage that covers ground truth; `worst relevant` +is the lowest one that still has to survive for the question to be fully answered; +`best non-relevant` is the highest-scoring passage that covers nothing; `margin` is the +first minus the third. + +The margin is an **oracle** quantity: at runtime nothing knows which result is relevant, so +it describes how much the score separates the two — it is not a signal a product could use. +Reading it as a candidate mechanism is the mistake the runtime-signal evaluation exists to +avoid. + +Where a row shows a raw cosine in brackets: an **absolute** score maps as +`cosine = 2·score − 1`, while a **margin** maps as `Δcosine = 2·Δscore` because the +`+1` cancels. The `margin` row uses the latter, the others the former. + +| Distribution | n | min | p10 | p25 | p50 | p75 | p90 | max | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | +| best relevant | 19 | 0.8808 (0.762) | 0.8906 (0.781) | 0.9120 (0.824) | 0.9359 (0.872) | 0.9428 (0.886) | 0.9530 (0.906) | 0.9537 (0.907) | +| worst relevant | 19 | 0.8808 (0.762) | 0.8906 (0.781) | 0.9086 (0.817) | 0.9355 (0.871) | 0.9428 (0.886) | 0.9530 (0.906) | 0.9537 (0.907) | +| best non-relevant | 19 | 0.8904 (0.781) | 0.9081 (0.816) | 0.9178 (0.836) | 0.9243 (0.849) | 0.9336 (0.867) | 0.9386 (0.877) | 0.9459 (0.892) | +| margin (best rel − best non-rel) | 19 | -0.0294 (-0.059) | -0.0272 (-0.054) | -0.0092 (-0.018) | 0.0026 (0.005) | 0.0081 (0.016) | 0.0329 (0.066) | 0.0623 (0.125) | +| unanswerable max candidate | 5 | 0.8978 (0.796) | 0.8978 (0.796) | 0.9128 (0.826) | 0.9259 (0.852) | 0.9343 (0.869) | 0.9424 (0.885) | 0.9424 (0.885) | + +## By query type + +`best relevant` p50 and p10, and `worst relevant` p10 — the last is the one that decides +whether a cross-lingual question survives a threshold that a semantic question tolerates. + +| Type | n | best rel p50 | best rel p10 | worst rel p10 | +| --- | --- | --- | --- | --- | +| cross-lingual | 4 | 0.8906 | 0.8808 | 0.8808 | +| exact | 4 | 0.9294 | 0.9120 | 0.9120 | +| hard-negative | 2 | 0.9359 | 0.9359 | 0.9359 | +| multi-hop | 2 | 0.9248 | 0.9248 | 0.9033 | +| semantic | 6 | 0.9395 | 0.9208 | 0.9208 | +| zh | 1 | 0.9527 | 0.9527 | 0.9527 | + +## Threshold curve + +For each candidate threshold: **hit** = share of answerable questions that still have some +relevant passage; **full recall** = share whose every ground-truth block is still covered; +**abstain** = share of unanswerable questions that now return nothing. + +| threshold (score) | raw cosine | answerable hit | answerable full recall | unanswerable abstain | +| --- | --- | --- | --- | --- | +| 0.500 | 0.000 | 1.0000 | 1.0000 | 0.0000 | +| 0.525 | 0.050 | 1.0000 | 1.0000 | 0.0000 | +| 0.550 | 0.100 | 1.0000 | 1.0000 | 0.0000 | +| 0.575 | 0.150 | 1.0000 | 1.0000 | 0.0000 | +| 0.600 | 0.200 | 1.0000 | 1.0000 | 0.0000 | +| 0.625 | 0.250 | 1.0000 | 1.0000 | 0.0000 | +| 0.650 | 0.300 | 1.0000 | 1.0000 | 0.0000 | +| 0.675 | 0.350 | 1.0000 | 1.0000 | 0.0000 | +| 0.700 | 0.400 | 1.0000 | 1.0000 | 0.0000 | +| 0.725 | 0.450 | 1.0000 | 1.0000 | 0.0000 | +| 0.750 | 0.500 | 1.0000 | 1.0000 | 0.0000 | +| 0.775 | 0.550 | 1.0000 | 1.0000 | 0.0000 | +| 0.800 | 0.600 | 1.0000 | 1.0000 | 0.0000 | +| 0.825 | 0.650 | 1.0000 | 1.0000 | 0.0000 | +| 0.850 | 0.700 | 1.0000 | 1.0000 | 0.0000 | +| 0.875 | 0.750 | 1.0000 | 1.0000 | 0.0000 | +| 0.900 | 0.800 | 0.8947 | 0.8947 | 0.2000 | +| 0.925 | 0.850 | 0.6316 | 0.6316 | 0.4000 | +| 0.950 | 0.900 | 0.1579 | 0.1579 | 1.0000 | +| 0.975 | 0.950 | 0.0000 | 0.0000 | 1.0000 | +| 1.000 | 1.000 | 0.0000 | 0.0000 | 1.0000 | + +## The separation question + +The threshold decision turns on whether these overlap: + +| Landmark | score | raw cosine | +| --- | --- | --- | +| worst relevant, p10 | 0.8906 | 0.781 | +| best non-relevant, p90 | 0.9386 | 0.877 | +| unanswerable max, p50 | 0.9259 | 0.852 | +| unanswerable max, p90 | 0.9424 | 0.885 | + +**The distributions **overlap**, so a higher threshold buys abstention by giving up required relevant passages. If the curve above shows abstention rising only as full recall falls, then the honest conclusion is that **a single dense similarity threshold cannot carry both recall and abstention** — and the next mechanism to evaluate is not a finer threshold grid but a different signal (reranker score, top1−top2 margin, per-query thresholds, or claim-level answerability). + +## Reproduce + +```bash +npm run eval:prepare # one-time, networked model bootstrap +npm run eval:scores # offline; rewrites this file +``` diff --git a/eval/README.md b/eval/README.md index db93d5d..299eca7 100644 --- a/eval/README.md +++ b/eval/README.md @@ -12,9 +12,22 @@ npm run eval # offline and deterministic: run the harness, rewrite th npm run eval:retrieval # strategy comparison (#77); validation selects, test reports npm run eval:threshold # derive the similarity threshold on validation, report on test npm run eval:sweep # bounded grid over strategy × candidateK × contextK, one dashboard +npm run eval:scores # dense score distribution: can one threshold separate relevant from not? npm run eval:blocks eval/corpus/foo.md # print the block ordinals ground truth must use ``` +### `threshold` is not a cosine + +The vector store returns \`score = 1 - distance / 2\` and sqlite-vec's cosine distance is +\`1 - cosine\`, so: + +```text +score = (1 + cosine) / 2 score 0.5 == cosine 0.0 +``` + +The shipped \`threshold = 0.5\` therefore means **cosine ≥ 0**, which is very permissive. +\`npm run eval:scores\` reports both columns side by side so the two never get conflated. + ### The harness runs the production configuration From v1.6 the harness defaults to the parameters the app ships, so its numbers describe diff --git a/package.json b/package.json index 77a6406..418c579 100644 --- a/package.json +++ b/package.json @@ -44,7 +44,8 @@ "eval:retrieval": "npm run build && node --experimental-transform-types --disable-warning=ExperimentalWarning --disable-warning=MODULE_TYPELESS_PACKAGE_JSON scripts/eval-retrieval.mjs", "eval:threshold": "npm run build && node scripts/eval-threshold.mjs", "eval:sweep": "npm run build && node scripts/eval-sweep.mjs", - "eval:blocks": "node --experimental-transform-types --import ./test/ts-resolve.mjs --disable-warning=ExperimentalWarning --disable-warning=MODULE_TYPELESS_PACKAGE_JSON scripts/eval-blocks.mjs" + "eval:blocks": "node --experimental-transform-types --import ./test/ts-resolve.mjs --disable-warning=ExperimentalWarning --disable-warning=MODULE_TYPELESS_PACKAGE_JSON scripts/eval-blocks.mjs", + "eval:scores": "npm run build && node scripts/eval-scores.mjs" }, "//test": [ "`node --test` strips TypeScript types rather than compiling them, and strip-only", diff --git a/scripts/eval-scores.mjs b/scripts/eval-scores.mjs new file mode 100644 index 0000000..43e5bbf --- /dev/null +++ b/scripts/eval-scores.mjs @@ -0,0 +1,416 @@ +#!/usr/bin/env node +/** + * Dense score diagnostics for #192. + * + * Answers one question before any threshold is chosen: **is a single dense similarity + * threshold even capable of separating "this answers the question" from "this does not"?** + * + * Three constraints, all deliberate: + * + * - **dense only.** A hybrid `score` is an RRF value (`1 / (60 + rank)`) and is not on the + * same scale as a normalised cosine. Mixing them would manufacture a new misleading + * number, which is what the threshold discussion is trying to avoid. + * - **validation only.** The distribution is used to choose where to sweep, so looking at + * `test` first would be tuning on the reporting side. + * - **the whole corpus per query** (`candidateK` ≥ index size) and `threshold = 0`, so the + * distribution is not already truncated by the parameter being investigated. + * + * Usage: + * node scripts/eval-scores.mjs + */ + +import { spawn } from 'node:child_process' +import { existsSync, mkdtempSync, mkdirSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join, resolve } from 'node:path' + +const OUT_MD = resolve(readArg('--out=', 'docs/eval/scores-v1.6.md')) +const OUT_JSON = OUT_MD.replace(/\.md$/, '.json') + +/** The shipped dense pipeline. Only the split and the candidate depth are changed. */ +const SPLIT = 'validation' +const CONTEXT_K = 3 +/** Above the index size, so every chunk is scored for every query. */ +const CANDIDATE_K = 500 + +function readArg(prefix, fallback) { + const arg = process.argv.find((value) => value.startsWith(prefix)) + return arg ? arg.slice(prefix.length) : fallback +} + +// Node 24 refuses to spawn a `.cmd`/`.bat` without `shell: true` (EINVAL), and the +// `.bin` entry is exactly that on Windows. Use the real binary the wrapper runs. +const { default: electronBinary } = await import('electron') +const executable = resolve(electronBinary) + +if (!existsSync(executable)) { + console.error('[scores] could not find the electron binary. Run `npm install` first.') + process.exit(1) +} + +function runHarness(outDir) { + return new Promise((resolvePromise, reject) => { + const args = [ + '.', + '--eval-harness', + '--eval-baseline=scores', + `--eval-out=${outDir}`, + `--eval-split=${SPLIT}`, + '--eval-retrieval=dense', + `--eval-candidate-k=${CANDIDATE_K}`, + `--eval-context-k=${CONTEXT_K}`, + '--eval-threshold=0', + '--eval-scores' + ] + + const isRoot = typeof process.getuid === 'function' && process.getuid() === 0 + if (isRoot || process.env.CI) args.push('--no-sandbox') + + const child = spawn(executable, args, { + stdio: ['ignore', 'pipe', 'pipe'], + env: { ...process.env, ELECTRON_DISABLE_SECURITY_WARNINGS: '1' } + }) + + child.on('error', reject) + child.on('exit', (code) => { + if (code !== 0) { + reject(new Error(`diagnostics harness exited with code ${code}`)) + return + } + const reportPath = join(outDir, 'baseline-scores.json') + if (!existsSync(reportPath)) { + reject(new Error('diagnostics harness wrote no report')) + return + } + resolvePromise(JSON.parse(readFileSync(reportPath, 'utf8'))) + }) + }) +} + +/** Nearest-rank quantile, matching `src/main/eval/metrics.ts`. */ +function quantile(values, p) { + if (values.length === 0) return null + const sorted = [...values].sort((a, b) => a - b) + const rank = Math.ceil((p / 100) * sorted.length) + return sorted[Math.min(Math.max(rank - 1, 0), sorted.length - 1)] +} + +/** + * `score = (1 + cosine) / 2`, from `SQLiteVectorStore`. Reported everywhere alongside the + * score because `threshold: 0.5` has been read as "cosine ≥ 0.5" and is really "cosine ≥ 0". + */ +const toCosine = (score) => (score === null ? null : 2 * score - 1) + +/** + * A difference of two scores is not an affine map of a difference of two cosines: the + * `+1` cancels, so `Δcosine = 2 * Δscore`. Applying the absolute transform to a margin + * would have produced `2Δscore - 1`, which for a near-zero margin reports a cosine margin + * near −1 — a sign flip on top of a scale error. + */ +const toCosineMargin = (margin) => (margin === null ? null : 2 * margin) + +const QUANTILES = [0, 10, 25, 50, 75, 90, 100] + +function describe(values) { + if (values.length === 0) return null + const out = {} + for (const p of QUANTILES) out[`p${p}`] = quantile(values, p) + return out +} + +const workDir = mkdtempSync(join(tmpdir(), 'knownote-scores-')) +let report +try { + console.log('[scores] running the dense harness on validation, threshold 0') + report = await runHarness(workDir) +} finally { + rmSync(workDir, { recursive: true, force: true }) +} + +const perQuestion = report.perQuestion ?? [] +const answerable = perQuestion.filter((q) => q.answerable) +const unanswerable = perQuestion.filter((q) => !q.answerable) + +if (perQuestion.some((q) => !q.retrievedScores)) { + console.error('[scores] the report carries no scores; did --eval-scores reach the harness?') + process.exit(1) +} + +/** Per-question score landmarks. */ +function landmarks(question) { + const scores = question.retrievedScores + const relevant = [] + const nonRelevant = [] + scores.forEach((score, index) => { + if ((question.matchesByRank[index] ?? []).length > 0) relevant.push(score) + else nonRelevant.push(score) + }) + if (relevant.length === 0) return null + const bestRelevant = Math.max(...relevant) + const worstRelevant = Math.min(...relevant) + const bestNonRelevant = nonRelevant.length > 0 ? Math.max(...nonRelevant) : null + return { + id: question.id, + type: question.type, + bestRelevant, + worstRelevant, + bestNonRelevant, + margin: bestNonRelevant === null ? null : bestRelevant - bestNonRelevant + } +} + +const answerableLandmarks = answerable.map(landmarks).filter((entry) => entry !== null) +const unanswerableMax = unanswerable.map((q) => Math.max(...q.retrievedScores)) + +const byType = {} +for (const entry of answerableLandmarks) { + byType[entry.type] = byType[entry.type] ?? [] + byType[entry.type].push(entry) +} + +/** + * The curve that decides whether one threshold can do both jobs. For each candidate `t`: + * + * - **hit** — share of answerable questions that still have *some* relevant passage; + * - **fullRecall** — share whose *every* ground-truth block is still covered; + * - **abstain** — share of unanswerable questions that now return nothing. + * + * A useful threshold needs `abstain` to rise while `fullRecall` holds. If every `t` that + * raises abstention also drops full recall, no single threshold can carry the job. + */ +const thresholdGrid = [] +for (let t = 0.5; t <= 1.0001; t += 0.025) thresholdGrid.push(Number(t.toFixed(3))) + +const curve = thresholdGrid.map((threshold) => { + let hit = 0 + let fullRecall = 0 + for (const question of answerable) { + const covered = new Set() + let coveredAbove = 0 + question.matchesByRank.forEach((matches, index) => { + if (question.retrievedScores[index] < threshold) return + for (const match of matches) { + if (!covered.has(match)) { + covered.add(match) + coveredAbove += 1 + } + } + }) + if (coveredAbove > 0) hit += 1 + if (question.relevantCount > 0 && coveredAbove === question.relevantCount) fullRecall += 1 + } + + const abstained = unanswerable.filter((q) => Math.max(...q.retrievedScores) < threshold).length + return { + threshold, + cosine: Number(toCosine(threshold).toFixed(3)), + hitRate: answerable.length === 0 ? 0 : hit / answerable.length, + fullRecallRate: answerable.length === 0 ? 0 : fullRecall / answerable.length, + abstentionRate: unanswerable.length === 0 ? 0 : abstained / unanswerable.length + } +}) + +const format4 = (value) => (value === null ? '—' : value.toFixed(4)) +/** + * `transform` maps a value to its cosine counterpart. It defaults to the absolute-score + * transform and is overridden for margins, because the two do not share one. + */ +const quantileRow = (label, values, transform = toCosine) => { + const d = describe(values) + if (!d) return `| ${label} | 0 | ${QUANTILES.map(() => '—').join(' | ')} |` + const cells = QUANTILES.map((p) => { + const value = d[`p${p}`] + return `${value.toFixed(4)} (${transform(value).toFixed(3)})` + }) + return `| ${label} | ${values.length} | ${cells.join(' | ')} |` +} + +const relevantBest = answerableLandmarks.map((entry) => entry.bestRelevant) +const relevantWorst = answerableLandmarks.map((entry) => entry.worstRelevant) +const nonRelevantBest = answerableLandmarks + .map((entry) => entry.bestNonRelevant) + .filter((value) => value !== null) +const margins = answerableLandmarks.map((entry) => entry.margin).filter((value) => value !== null) + +const typeRows = Object.entries(byType) + .sort(([a], [b]) => a.localeCompare(b)) + .map(([type, entries]) => { + const best = describe(entries.map((entry) => entry.bestRelevant)) + const worst = describe(entries.map((entry) => entry.worstRelevant)) + return `| ${type} | ${entries.length} | ${format4(best.p50)} | ${format4(best.p10)} | ${format4(worst.p10)} |` + }) + .join('\n') + +const curveRows = curve + .map( + (row) => + `| ${row.threshold.toFixed(3)} | ${row.cosine.toFixed(3)} | ${format4(row.hitRate)} | ` + + `${format4(row.fullRecallRate)} | ${format4(row.abstentionRate)} |` + ) + .join('\n') + +/** Where the three distributions overlap, which is what the threshold decision turns on. */ +const overlap = { + worstRelevantP10: quantile(relevantWorst, 10), + bestNonRelevantP90: quantile(nonRelevantBest, 90), + unanswerableMaxP50: quantile(unanswerableMax, 50), + unanswerableMaxP90: quantile(unanswerableMax, 90) +} +const separable = + overlap.worstRelevantP10 !== null && + overlap.unanswerableMaxP90 !== null && + overlap.worstRelevantP10 > overlap.unanswerableMaxP90 + +const markdown = `# Dense score diagnostics — v1.6 (#192) + +Generated by \`node scripts/eval-scores.mjs\`. Numbers are harness output; do not edit them by hand. + +## Read this first: the score is not a cosine + +\`SQLiteVectorStore\` computes + +\`\`\`text +score = 1 - distance / 2 +distance = 1 - cosine (sqlite-vec, distance_metric=cosine) +=> score = (1 + cosine) / 2 +\`\`\` + +so the configured \`threshold\` is an **affine map of the cosine**, not the cosine: + +| threshold | raw cosine | +| --- | --- | +| 0.3 | -0.4 | +| 0.4 | -0.2 | +| **0.5 (shipped)** | **0.0** | +| 0.6 | 0.2 | +| 0.8 | 0.6 | +| 1.0 | 1.0 | + +The shipped \`threshold = 0.5\` means **cosine ≥ 0**, which is very permissive. Every +threshold row below carries both columns so the two never get confused again, and every +mention of "the sweep looked too low" has to be read through this table. + +## What was measured + +Dense only, \`${SPLIT}\` split only, \`threshold = 0\`, \`candidateK = ${CANDIDATE_K}\` (above the +index size, so every chunk is scored for every query), \`contextK = ${CONTEXT_K}\`. + +- answerable questions: ${answerable.length} +- unanswerable questions: ${unanswerable.length} +- index size: ${report.config.chunkCount} chunks + +Hybrid is deliberately excluded: its \`score\` is an RRF value (\`1 / (60 + rank)\`) and is not +on the same scale as a normalised cosine. + +## Distributions (score, with raw cosine in brackets) + +\`best relevant\` is the highest-scoring passage that covers ground truth; \`worst relevant\` +is the lowest one that still has to survive for the question to be fully answered; +\`best non-relevant\` is the highest-scoring passage that covers nothing; \`margin\` is the +first minus the third. + +The margin is an **oracle** quantity: at runtime nothing knows which result is relevant, so +it describes how much the score separates the two — it is not a signal a product could use. +Reading it as a candidate mechanism is the mistake the runtime-signal evaluation exists to +avoid. + +Where a row shows a raw cosine in brackets: an **absolute** score maps as +\`cosine = 2·score − 1\`, while a **margin** maps as \`Δcosine = 2·Δscore\` because the +\`+1\` cancels. The \`margin\` row uses the latter, the others the former. + +| Distribution | n | min | p10 | p25 | p50 | p75 | p90 | max | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | +${quantileRow('best relevant', relevantBest)} +${quantileRow('worst relevant', relevantWorst)} +${quantileRow('best non-relevant', nonRelevantBest)} +${quantileRow('margin (best rel − best non-rel)', margins, toCosineMargin)} +${quantileRow('unanswerable max candidate', unanswerableMax)} + +## By query type + +\`best relevant\` p50 and p10, and \`worst relevant\` p10 — the last is the one that decides +whether a cross-lingual question survives a threshold that a semantic question tolerates. + +| Type | n | best rel p50 | best rel p10 | worst rel p10 | +| --- | --- | --- | --- | --- | +${typeRows} + +## Threshold curve + +For each candidate threshold: **hit** = share of answerable questions that still have some +relevant passage; **full recall** = share whose every ground-truth block is still covered; +**abstain** = share of unanswerable questions that now return nothing. + +| threshold (score) | raw cosine | answerable hit | answerable full recall | unanswerable abstain | +| --- | --- | --- | --- | --- | +${curveRows} + +## The separation question + +The threshold decision turns on whether these overlap: + +| Landmark | score | raw cosine | +| --- | --- | --- | +| worst relevant, p10 | ${format4(overlap.worstRelevantP10)} | ${overlap.worstRelevantP10 === null ? '—' : toCosine(overlap.worstRelevantP10).toFixed(3)} | +| best non-relevant, p90 | ${format4(overlap.bestNonRelevantP90)} | ${overlap.bestNonRelevantP90 === null ? '—' : toCosine(overlap.bestNonRelevantP90).toFixed(3)} | +| unanswerable max, p50 | ${format4(overlap.unanswerableMaxP50)} | ${overlap.unanswerableMaxP50 === null ? '—' : toCosine(overlap.unanswerableMaxP50).toFixed(3)} | +| unanswerable max, p90 | ${format4(overlap.unanswerableMaxP90)} | ${overlap.unanswerableMaxP90 === null ? '—' : toCosine(overlap.unanswerableMaxP90).toFixed(3)} | + +**${ + separable + ? 'The distributions separate at the p10/p90 landmarks, so a single threshold is a plausible mechanism on this corpus. Read the grid above for where to sweep.' + : 'The distributions **overlap**, so a higher threshold buys abstention by giving up required relevant passages. If the curve above shows abstention rising only as full recall falls, then the honest conclusion is that **a single dense similarity threshold cannot carry both recall and abstention** — and the next mechanism to evaluate is not a finer threshold grid but a different signal (reranker score, top1−top2 margin, per-query thresholds, or claim-level answerability).' +} + +## Reproduce + +\`\`\`bash +npm run eval:prepare # one-time, networked model bootstrap +npm run eval:scores # offline; rewrites this file +\`\`\` +` + +mkdirSync(resolve(OUT_MD, '..'), { recursive: true }) +writeFileSync( + OUT_JSON, + `${JSON.stringify( + { + baseline: 'v1.6', + split: SPLIT, + candidateK: CANDIDATE_K, + threshold: 0, + indexSize: report.config.chunkCount, + counts: { answerable: answerable.length, unanswerable: unanswerable.length }, + distributions: { + bestRelevant: describe(relevantBest), + worstRelevant: describe(relevantWorst), + bestNonRelevant: describe(nonRelevantBest), + margin: describe(margins), + unanswerableMax: describe(unanswerableMax) + }, + byType: Object.fromEntries( + Object.entries(byType).map(([type, entries]) => [ + type, + { + questions: entries.length, + bestRelevant: describe(entries.map((entry) => entry.bestRelevant)), + worstRelevant: describe(entries.map((entry) => entry.worstRelevant)) + } + ]) + ), + overlap, + separable, + curve + }, + null, + 2 + )}\n` +) +writeFileSync(OUT_MD, markdown) + +console.log(`[scores] answerable ${answerable.length}, unanswerable ${unanswerable.length}, index ${report.config.chunkCount}`) +console.log( + `[scores] worst relevant p10 = ${format4(overlap.worstRelevantP10)} (cosine ${overlap.worstRelevantP10 === null ? '—' : toCosine(overlap.worstRelevantP10).toFixed(3)}), unanswerable max p90 = ${format4(overlap.unanswerableMaxP90)}` +) +console.log(`[scores] separable = ${separable}`) +console.log(`[scores] wrote ${OUT_JSON} and ${OUT_MD}`) diff --git a/src/main/eval/harness.ts b/src/main/eval/harness.ts index 44427de..f34ee5b 100644 --- a/src/main/eval/harness.ts +++ b/src/main/eval/harness.ts @@ -72,6 +72,13 @@ export interface EvalHarnessOptions { threshold: number /** 分块配置(#78)。实验变体通过它选择策略;缺省时用生产默认值。 */ chunkOptions: ChunkOptions + /** + * 把每个 rank 的检索分数也写进报告(`--eval-scores`)。 + * + * 缺省关闭:分数序列会让基线膨胀一倍,而基线是 CI 逐字节 diff 的文件。score + * diagnostics(#192)需要它,生产基线不需要。 + */ + includeScores?: boolean /** 检索策略(#77):dense / sparse(BM25) / hybrid(RRF)。 */ strategy: RetrievalStrategy } @@ -298,7 +305,7 @@ export async function runEvalHarness( .filter((index) => index >= 0) }) - perQuestion.push({ + const questionReport: QuestionReport = { id: question.id, question: question.question, type: question.type ?? UNTAGGED, @@ -310,7 +317,10 @@ export async function runEvalHarness( .slice(0, options.contextK) .reduce((total, result) => total + result.content.length, 0), matchesByRank - }) + } + // Only when asked: the score series doubles the size of the committed baseline. + if (options.includeScores) questionReport.retrievedScores = results.map((r) => r.score) + perQuestion.push(questionReport) } // 不可答的问题不进排名指标:它们没有 ground truth,`recallAtK` 对它们返回的是 0/0 @@ -396,6 +406,14 @@ function roundMetrics(metrics: EvalMetrics): EvalMetrics { } export function stabilize(report: EvalReport): EvalReport { + // Scores are extra data, not metrics; they are rounded the same way so two runs of the + // same diagnostic diff cleanly. + const perQuestion = report.perQuestion.map((question) => + question.retrievedScores + ? { ...question, retrievedScores: question.retrievedScores.map(roundMetric) } + : question + ) + return { ...report, metrics: roundMetrics(report.metrics), @@ -407,7 +425,7 @@ export function stabilize(report: EvalReport): EvalReport { // full report keeps it for the #78 comparison. indexingMs: Math.round(report.timing.indexingMs) }, - perQuestion: report.perQuestion + perQuestion } } diff --git a/src/main/eval/run.ts b/src/main/eval/run.ts index 20c13fa..ad45091 100644 --- a/src/main/eval/run.ts +++ b/src/main/eval/run.ts @@ -200,6 +200,9 @@ export async function runEvalCli(argv: readonly string[] = process.argv): Promis contextK: readNumberOption(argv, '--eval-context-k=', 3), threshold: readNumberOption(argv, '--eval-threshold=', 0.5), chunkOptions: readChunkOptions(argv), + // `--eval-scores` puts the per-rank retrieval scores in the report. Off by default: + // the committed baseline is a CI-diffed file and the series doubles it. + includeScores: readBoolOption(argv, '--eval-scores=', argv.includes('--eval-scores')), strategy: readRetrievalStrategy(argv) } diff --git a/src/main/eval/types.ts b/src/main/eval/types.ts index 715396a..6a33005 100644 --- a/src/main/eval/types.ts +++ b/src/main/eval/types.ts @@ -197,6 +197,15 @@ export interface QuestionReport { contextChars: number /** Ground-truth indices matched by each retrieved rank, in rank order. */ matchesByRank: number[][] + /** + * 每个 rank 的检索分数,与 `matchesByRank` 同序。 + * + * 只在 `--eval-scores` 时出现:它是 score diagnostics(#192)需要的数据,而基线 + * JSON 默认不带它——分数序列会让基线膨胀一倍,而基线是 CI 要逐字节 diff 的文件。 + * + * 注意它不是 cosine:见 `SQLiteVectorStore`,`score = (1 + cosine) / 2`。 + */ + retrievedScores?: number[] } export interface EvalReport { diff --git a/src/main/vectorstore/SQLiteVectorStore.ts b/src/main/vectorstore/SQLiteVectorStore.ts index 8fc300b..8f696f1 100644 --- a/src/main/vectorstore/SQLiteVectorStore.ts +++ b/src/main/vectorstore/SQLiteVectorStore.ts @@ -171,7 +171,11 @@ export class SQLiteVectorStore implements VectorStore { // 转换结果 const queryResults: QueryResult[] = results.map((row) => { - // cosine 距离转相似度(0-1) + // `score = 1 - distance / 2 = (1 + cosine) / 2`。 + // + // **这不是 cosine 本身**,而是仿射映射:score 0.5 对应 cosine 0,score 0.6 对应 + // cosine 0.2,score 1.0 才是 cosine 1.0。配置里的 `threshold` 就是这个 score, + // 把 0.5 读成“cosine ≥ 0.5”会把门槛高估很多(#192 评审)。 const score = 1 - row.distance / 2 return {