From 08fa7c112dc024ce708b68a9c169b4be121c6996 Mon Sep 17 00:00:00 2001 From: MrSibe Date: Wed, 30 Sep 2026 18:10:23 +0800 Subject: [PATCH 1/2] feat(eval): dense score diagnostics, and the answer is not a threshold MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Adds `npm run eval:scores`, which measures whether a single dense similarity threshold is capable of separating "this passage answers the question" from "this one does not" — before any threshold grid is chosen. Child 2 of #192, and it changes what the next step should be. ## First, the score is not a cosine `SQLiteVectorStore` computes `score = 1 - distance / 2`, and sqlite-vec's cosine distance is `1 - cosine`, so: ``` score = (1 + cosine) / 2 => threshold 0.5 == cosine 0.0 ``` The shipped `threshold = 0.5` is therefore **cosine ≥ 0**, not "cosine ≥ 0.5". Every guard that mattered here — the sweep grid, the threshold report, the source comment — was describing an affine map, not the cosine. Fixed in the store's comment, in `eval/README.md`, and reported as two columns throughout the new diagnostics. ## The measurement Dense only (hybrid's `score` is an RRF value, `1 / (60 + rank)`, and not on the same scale), `validation` split only (the distribution is used to choose where to sweep, so it must not see `test`), `threshold = 0` and `candidateK = 500` (above the 53-chunk index, so every chunk is scored for every query). ``` best relevant p10 0.8906 (cosine 0.781) p50 0.9359 (0.872) best non-relevant p50 0.9243 (cosine 0.849) p90 0.9386 (0.877) margin p10 -0.0272 p50 +0.0026 unanswerable max p50 0.9259 (cosine 0.852) p90 0.9424 (0.885) ``` Three things fall out of that: 1. **The best non-relevant passage outranks the best relevant one about half the time.** The margin's p50 is +0.0026 and its p10 is −0.0272. Dense similarity is barely informative about relevance on this corpus — the hard-negative clusters are doing exactly what they were built for. 2. **The unanswerable max sits above the worst required relevant passage** (p90 0.9424 vs p10 0.8906), so the two distributions are not separable. 3. **`cross-lingual` is the lowest of every type** (best relevant p10 0.8808) against `semantic` at 0.9208 — a threshold tuned on the aggregate would cut cross-lingual first. ## The threshold curve is the answer | threshold | raw cosine | answerable hit | answerable full recall | unanswerable abstain | | --- | --- | --- | --- | --- | | 0.875 | 0.75 | 1.0000 | 1.0000 | 0.0000 | | 0.900 | 0.80 | 0.8947 | 0.8947 | 0.2000 | | 0.925 | 0.85 | 0.6316 | 0.6316 | 0.4000 | | 0.950 | 0.900 | 0.1579 | 0.1579 | 1.0000 | **Abstention never rises without full recall falling.** Every threshold that refuses an unanswerable question refuses required relevant passages at the same rate. There is no operating point that buys the first without paying the second. So the conclusion is not "sweep 0.7–0.8 instead of 0–0.6". It is: **a single dense similarity threshold cannot carry both recall and abstention on this corpus**, and the next mechanism to evaluate is a different signal — reranker score, top1−top2 margin, per-query thresholds, or claim-level answerability. That is a much more useful finding than a best threshold would have been, and it is why this PR was worth doing before more clusters. ## What this PR does not do - It does not choose a threshold. That is now blocked on a signal that can separate the distributions. - The `eval:threshold` sweep grid is left as it is. Re-gridding it would move a knob whose curve is flat where it matters and precipitous where it does not — the diagnostic is the replacement for re-gridding, not a preamble to it. ## Testing - `npm run typecheck` — clean - `npm test` — 504 pass - `npm run eval:scores` — writes `docs/eval/scores-v1.6.{json,md}` - The committed baseline is **unchanged**: `--eval-scores` is off by default, so the CI-diffed JSON does not grow a score series it has no use for Part of #192 (child 2: corpus diagnostics). --- .prettierignore | 4 + docs/eval/scores-v1.6.json | 342 +++++++++++++++++++ docs/eval/scores-v1.6.md | 119 +++++++ eval/README.md | 13 + package.json | 3 +- scripts/eval-scores.mjs | 392 ++++++++++++++++++++++ src/main/eval/harness.ts | 24 +- src/main/eval/run.ts | 3 + src/main/eval/types.ts | 9 + src/main/vectorstore/SQLiteVectorStore.ts | 6 +- 10 files changed, 910 insertions(+), 5 deletions(-) create mode 100644 docs/eval/scores-v1.6.json create mode 100644 docs/eval/scores-v1.6.md create mode 100644 scripts/eval-scores.mjs diff --git a/.prettierignore b/.prettierignore index f6f4ae5..da32ca2 100644 --- a/.prettierignore +++ b/.prettierignore @@ -29,6 +29,10 @@ docs/eval/threshold-*.md docs/eval/sweep-*.json docs/eval/sweep-*.md +# And by `scripts/eval-scores.mjs`. Regenerate with `npm run eval:scores`. +docs/eval/scores-*.json +docs/eval/scores-*.md + # And the same again for `scripts/eval-retrieval.mjs` (#77). docs/eval/retrieval-*.json docs/eval/retrieval-*.md diff --git a/docs/eval/scores-v1.6.json b/docs/eval/scores-v1.6.json new file mode 100644 index 0000000..163c779 --- /dev/null +++ b/docs/eval/scores-v1.6.json @@ -0,0 +1,342 @@ +{ + "baseline": "v1.6", + "split": "validation", + "candidateK": 500, + "threshold": 0, + "indexSize": 53, + "counts": { + "answerable": 19, + "unanswerable": 5 + }, + "distributions": { + "bestRelevant": { + "p0": 0.880777, + "p10": 0.89056, + "p25": 0.912001, + "p50": 0.935891, + "p75": 0.942776, + "p90": 0.953038, + "p100": 0.953664 + }, + "worstRelevant": { + "p0": 0.880777, + "p10": 0.89056, + "p25": 0.908589, + "p50": 0.935467, + "p75": 0.942776, + "p90": 0.953038, + "p100": 0.953664 + }, + "bestNonRelevant": { + "p0": 0.890443, + "p10": 0.908132, + "p25": 0.917792, + "p50": 0.924346, + "p75": 0.933555, + "p90": 0.9386, + "p100": 0.945918 + }, + "margin": { + "p0": -0.02942099999999992, + "p10": -0.027232000000000034, + "p25": -0.00922400000000001, + "p50": 0.0026120000000000587, + "p75": 0.00807100000000005, + "p90": 0.032869999999999955, + "p100": 0.06226200000000004 + }, + "unanswerableMax": { + "p0": 0.897848, + "p10": 0.897848, + "p25": 0.912751, + "p50": 0.925936, + "p75": 0.934276, + "p90": 0.942441, + "p100": 0.942441 + } + }, + "byType": { + "exact": { + "questions": 4, + "bestRelevant": { + "p0": 0.912001, + "p10": 0.912001, + "p25": 0.912001, + "p50": 0.929376, + "p75": 0.936747, + "p90": 0.953038, + "p100": 0.953038 + }, + "worstRelevant": { + "p0": 0.912001, + "p10": 0.912001, + "p25": 0.912001, + "p50": 0.929376, + "p75": 0.936747, + "p90": 0.953038, + "p100": 0.953038 + } + }, + "semantic": { + "questions": 6, + "bestRelevant": { + "p0": 0.920784, + "p10": 0.920784, + "p25": 0.935467, + "p50": 0.939466, + "p75": 0.945637, + "p90": 0.953664, + "p100": 0.953664 + }, + "worstRelevant": { + "p0": 0.920784, + "p10": 0.920784, + "p25": 0.935467, + "p50": 0.939466, + "p75": 0.945637, + "p90": 0.953664, + "p100": 0.953664 + } + }, + "multi-hop": { + "questions": 2, + "bestRelevant": { + "p0": 0.924784, + "p10": 0.924784, + "p25": 0.924784, + "p50": 0.924784, + "p75": 0.93756, + "p90": 0.93756, + "p100": 0.93756 + }, + "worstRelevant": { + "p0": 0.903304, + "p10": 0.903304, + "p25": 0.903304, + "p50": 0.903304, + "p75": 0.933129, + "p90": 0.933129, + "p100": 0.933129 + } + }, + "zh": { + "questions": 1, + "bestRelevant": { + "p0": 0.952705, + "p10": 0.952705, + "p25": 0.952705, + "p50": 0.952705, + "p75": 0.952705, + "p90": 0.952705, + "p100": 0.952705 + }, + "worstRelevant": { + "p0": 0.952705, + "p10": 0.952705, + "p25": 0.952705, + "p50": 0.952705, + "p75": 0.952705, + "p90": 0.952705, + "p100": 0.952705 + } + }, + "cross-lingual": { + "questions": 4, + "bestRelevant": { + "p0": 0.880777, + "p10": 0.880777, + "p25": 0.880777, + "p50": 0.89056, + "p75": 0.904196, + "p90": 0.908589, + "p100": 0.908589 + }, + "worstRelevant": { + "p0": 0.880777, + "p10": 0.880777, + "p25": 0.880777, + "p50": 0.89056, + "p75": 0.904196, + "p90": 0.908589, + "p100": 0.908589 + } + }, + "hard-negative": { + "questions": 2, + "bestRelevant": { + "p0": 0.935891, + "p10": 0.935891, + "p25": 0.935891, + "p50": 0.935891, + "p75": 0.942776, + "p90": 0.942776, + "p100": 0.942776 + }, + "worstRelevant": { + "p0": 0.935891, + "p10": 0.935891, + "p25": 0.935891, + "p50": 0.935891, + "p75": 0.942776, + "p90": 0.942776, + "p100": 0.942776 + } + } + }, + "overlap": { + "worstRelevantP10": 0.89056, + "bestNonRelevantP90": 0.9386, + "unanswerableMaxP50": 0.925936, + "unanswerableMaxP90": 0.942441 + }, + "separable": false, + "curve": [ + { + "threshold": 0.5, + "cosine": 0, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.525, + "cosine": 0.05, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.55, + "cosine": 0.1, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.575, + "cosine": 0.15, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.6, + "cosine": 0.2, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.625, + "cosine": 0.25, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.65, + "cosine": 0.3, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.675, + "cosine": 0.35, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.7, + "cosine": 0.4, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.725, + "cosine": 0.45, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.75, + "cosine": 0.5, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.775, + "cosine": 0.55, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.8, + "cosine": 0.6, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.825, + "cosine": 0.65, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.85, + "cosine": 0.7, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.875, + "cosine": 0.75, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.9, + "cosine": 0.8, + "hitRate": 0.8947368421052632, + "fullRecallRate": 0.8947368421052632, + "abstentionRate": 0.2 + }, + { + "threshold": 0.925, + "cosine": 0.85, + "hitRate": 0.631578947368421, + "fullRecallRate": 0.631578947368421, + "abstentionRate": 0.4 + }, + { + "threshold": 0.95, + "cosine": 0.9, + "hitRate": 0.15789473684210525, + "fullRecallRate": 0.15789473684210525, + "abstentionRate": 1 + }, + { + "threshold": 0.975, + "cosine": 0.95, + "hitRate": 0, + "fullRecallRate": 0, + "abstentionRate": 1 + }, + { + "threshold": 1, + "cosine": 1, + "hitRate": 0, + "fullRecallRate": 0, + "abstentionRate": 1 + } + ] +} diff --git a/docs/eval/scores-v1.6.md b/docs/eval/scores-v1.6.md new file mode 100644 index 0000000..2ad38b7 --- /dev/null +++ b/docs/eval/scores-v1.6.md @@ -0,0 +1,119 @@ +# Dense score diagnostics — v1.6 (#192) + +Generated by `node scripts/eval-scores.mjs`. Numbers are harness output; do not edit them by hand. + +## Read this first: the score is not a cosine + +`SQLiteVectorStore` computes + +```text +score = 1 - distance / 2 +distance = 1 - cosine (sqlite-vec, distance_metric=cosine) +=> score = (1 + cosine) / 2 +``` + +so the configured `threshold` is an **affine map of the cosine**, not the cosine: + +| threshold | raw cosine | +| --- | --- | +| 0.3 | -0.4 | +| 0.4 | -0.2 | +| **0.5 (shipped)** | **0.0** | +| 0.6 | 0.2 | +| 0.8 | 0.6 | +| 1.0 | 1.0 | + +The shipped `threshold = 0.5` means **cosine ≥ 0**, which is very permissive. Every +threshold row below carries both columns so the two never get confused again, and every +mention of "the sweep looked too low" has to be read through this table. + +## What was measured + +Dense only, `validation` split only, `threshold = 0`, `candidateK = 500` (above the +index size, so every chunk is scored for every query), `contextK = 3`. + +- answerable questions: 19 +- unanswerable questions: 5 +- index size: 53 chunks + +Hybrid is deliberately excluded: its `score` is an RRF value (`1 / (60 + rank)`) and is not +on the same scale as a normalised cosine. + +## Distributions (score, with raw cosine in brackets) + +`best relevant` is the highest-scoring passage that covers ground truth; `worst relevant` +is the lowest one that still has to survive for the question to be fully answered; +`best non-relevant` is the highest-scoring passage that covers nothing; `margin` is the +first minus the third. + +| Distribution | n | min | p10 | p25 | p50 | p75 | p90 | max | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | +| best relevant | 19 | 0.8808 (0.762) | 0.8906 (0.781) | 0.9120 (0.824) | 0.9359 (0.872) | 0.9428 (0.886) | 0.9530 (0.906) | 0.9537 (0.907) | +| worst relevant | 19 | 0.8808 (0.762) | 0.8906 (0.781) | 0.9086 (0.817) | 0.9355 (0.871) | 0.9428 (0.886) | 0.9530 (0.906) | 0.9537 (0.907) | +| best non-relevant | 19 | 0.8904 (0.781) | 0.9081 (0.816) | 0.9178 (0.836) | 0.9243 (0.849) | 0.9336 (0.867) | 0.9386 (0.877) | 0.9459 (0.892) | +| margin (best rel − best non-rel) | 19 | -0.0294 (-1.059) | -0.0272 (-1.054) | -0.0092 (-1.018) | 0.0026 (-0.995) | 0.0081 (-0.984) | 0.0329 (-0.934) | 0.0623 (-0.875) | +| unanswerable max candidate | 5 | 0.8978 (0.796) | 0.8978 (0.796) | 0.9128 (0.826) | 0.9259 (0.852) | 0.9343 (0.869) | 0.9424 (0.885) | 0.9424 (0.885) | + +## By query type + +`best relevant` p50 and p10, and `worst relevant` p10 — the last is the one that decides +whether a cross-lingual question survives a threshold that a semantic question tolerates. + +| Type | n | best rel p50 | best rel p10 | worst rel p10 | +| --- | --- | --- | --- | --- | +| cross-lingual | 4 | 0.8906 | 0.8808 | 0.8808 | +| exact | 4 | 0.9294 | 0.9120 | 0.9120 | +| hard-negative | 2 | 0.9359 | 0.9359 | 0.9359 | +| multi-hop | 2 | 0.9248 | 0.9248 | 0.9033 | +| semantic | 6 | 0.9395 | 0.9208 | 0.9208 | +| zh | 1 | 0.9527 | 0.9527 | 0.9527 | + +## Threshold curve + +For each candidate threshold: **hit** = share of answerable questions that still have some +relevant passage; **full recall** = share whose every ground-truth block is still covered; +**abstain** = share of unanswerable questions that now return nothing. + +| threshold (score) | raw cosine | answerable hit | answerable full recall | unanswerable abstain | +| --- | --- | --- | --- | --- | +| 0.500 | 0.000 | 1.0000 | 1.0000 | 0.0000 | +| 0.525 | 0.050 | 1.0000 | 1.0000 | 0.0000 | +| 0.550 | 0.100 | 1.0000 | 1.0000 | 0.0000 | +| 0.575 | 0.150 | 1.0000 | 1.0000 | 0.0000 | +| 0.600 | 0.200 | 1.0000 | 1.0000 | 0.0000 | +| 0.625 | 0.250 | 1.0000 | 1.0000 | 0.0000 | +| 0.650 | 0.300 | 1.0000 | 1.0000 | 0.0000 | +| 0.675 | 0.350 | 1.0000 | 1.0000 | 0.0000 | +| 0.700 | 0.400 | 1.0000 | 1.0000 | 0.0000 | +| 0.725 | 0.450 | 1.0000 | 1.0000 | 0.0000 | +| 0.750 | 0.500 | 1.0000 | 1.0000 | 0.0000 | +| 0.775 | 0.550 | 1.0000 | 1.0000 | 0.0000 | +| 0.800 | 0.600 | 1.0000 | 1.0000 | 0.0000 | +| 0.825 | 0.650 | 1.0000 | 1.0000 | 0.0000 | +| 0.850 | 0.700 | 1.0000 | 1.0000 | 0.0000 | +| 0.875 | 0.750 | 1.0000 | 1.0000 | 0.0000 | +| 0.900 | 0.800 | 0.8947 | 0.8947 | 0.2000 | +| 0.925 | 0.850 | 0.6316 | 0.6316 | 0.4000 | +| 0.950 | 0.900 | 0.1579 | 0.1579 | 1.0000 | +| 0.975 | 0.950 | 0.0000 | 0.0000 | 1.0000 | +| 1.000 | 1.000 | 0.0000 | 0.0000 | 1.0000 | + +## The separation question + +The threshold decision turns on whether these overlap: + +| Landmark | score | raw cosine | +| --- | --- | --- | +| worst relevant, p10 | 0.8906 | 0.781 | +| best non-relevant, p90 | 0.9386 | 0.877 | +| unanswerable max, p50 | 0.9259 | 0.852 | +| unanswerable max, p90 | 0.9424 | 0.885 | + +**The distributions **overlap**, so a higher threshold buys abstention by giving up required relevant passages. If the curve above shows abstention rising only as full recall falls, then the honest conclusion is that **a single dense similarity threshold cannot carry both recall and abstention** — and the next mechanism to evaluate is not a finer threshold grid but a different signal (reranker score, top1−top2 margin, per-query thresholds, or claim-level answerability). + +## Reproduce + +```bash +npm run eval:prepare # one-time, networked model bootstrap +npm run eval:scores # offline; rewrites this file +``` diff --git a/eval/README.md b/eval/README.md index db93d5d..299eca7 100644 --- a/eval/README.md +++ b/eval/README.md @@ -12,9 +12,22 @@ npm run eval # offline and deterministic: run the harness, rewrite th npm run eval:retrieval # strategy comparison (#77); validation selects, test reports npm run eval:threshold # derive the similarity threshold on validation, report on test npm run eval:sweep # bounded grid over strategy × candidateK × contextK, one dashboard +npm run eval:scores # dense score distribution: can one threshold separate relevant from not? npm run eval:blocks eval/corpus/foo.md # print the block ordinals ground truth must use ``` +### `threshold` is not a cosine + +The vector store returns \`score = 1 - distance / 2\` and sqlite-vec's cosine distance is +\`1 - cosine\`, so: + +```text +score = (1 + cosine) / 2 score 0.5 == cosine 0.0 +``` + +The shipped \`threshold = 0.5\` therefore means **cosine ≥ 0**, which is very permissive. +\`npm run eval:scores\` reports both columns side by side so the two never get conflated. + ### The harness runs the production configuration From v1.6 the harness defaults to the parameters the app ships, so its numbers describe diff --git a/package.json b/package.json index 77a6406..418c579 100644 --- a/package.json +++ b/package.json @@ -44,7 +44,8 @@ "eval:retrieval": "npm run build && node --experimental-transform-types --disable-warning=ExperimentalWarning --disable-warning=MODULE_TYPELESS_PACKAGE_JSON scripts/eval-retrieval.mjs", "eval:threshold": "npm run build && node scripts/eval-threshold.mjs", "eval:sweep": "npm run build && node scripts/eval-sweep.mjs", - "eval:blocks": "node --experimental-transform-types --import ./test/ts-resolve.mjs --disable-warning=ExperimentalWarning --disable-warning=MODULE_TYPELESS_PACKAGE_JSON scripts/eval-blocks.mjs" + "eval:blocks": "node --experimental-transform-types --import ./test/ts-resolve.mjs --disable-warning=ExperimentalWarning --disable-warning=MODULE_TYPELESS_PACKAGE_JSON scripts/eval-blocks.mjs", + "eval:scores": "npm run build && node scripts/eval-scores.mjs" }, "//test": [ "`node --test` strips TypeScript types rather than compiling them, and strip-only", diff --git a/scripts/eval-scores.mjs b/scripts/eval-scores.mjs new file mode 100644 index 0000000..1356f2a --- /dev/null +++ b/scripts/eval-scores.mjs @@ -0,0 +1,392 @@ +#!/usr/bin/env node +/** + * Dense score diagnostics for #192. + * + * Answers one question before any threshold is chosen: **is a single dense similarity + * threshold even capable of separating "this answers the question" from "this does not"?** + * + * Three constraints, all deliberate: + * + * - **dense only.** A hybrid `score` is an RRF value (`1 / (60 + rank)`) and is not on the + * same scale as a normalised cosine. Mixing them would manufacture a new misleading + * number, which is what the threshold discussion is trying to avoid. + * - **validation only.** The distribution is used to choose where to sweep, so looking at + * `test` first would be tuning on the reporting side. + * - **the whole corpus per query** (`candidateK` ≥ index size) and `threshold = 0`, so the + * distribution is not already truncated by the parameter being investigated. + * + * Usage: + * node scripts/eval-scores.mjs + */ + +import { spawn } from 'node:child_process' +import { existsSync, mkdtempSync, mkdirSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join, resolve } from 'node:path' + +const OUT_MD = resolve(readArg('--out=', 'docs/eval/scores-v1.6.md')) +const OUT_JSON = OUT_MD.replace(/\.md$/, '.json') + +/** The shipped dense pipeline. Only the split and the candidate depth are changed. */ +const SPLIT = 'validation' +const CONTEXT_K = 3 +/** Above the index size, so every chunk is scored for every query. */ +const CANDIDATE_K = 500 + +function readArg(prefix, fallback) { + const arg = process.argv.find((value) => value.startsWith(prefix)) + return arg ? arg.slice(prefix.length) : fallback +} + +// Node 24 refuses to spawn a `.cmd`/`.bat` without `shell: true` (EINVAL), and the +// `.bin` entry is exactly that on Windows. Use the real binary the wrapper runs. +const { default: electronBinary } = await import('electron') +const executable = resolve(electronBinary) + +if (!existsSync(executable)) { + console.error('[scores] could not find the electron binary. Run `npm install` first.') + process.exit(1) +} + +function runHarness(outDir) { + return new Promise((resolvePromise, reject) => { + const args = [ + '.', + '--eval-harness', + '--eval-baseline=scores', + `--eval-out=${outDir}`, + `--eval-split=${SPLIT}`, + '--eval-retrieval=dense', + `--eval-candidate-k=${CANDIDATE_K}`, + `--eval-context-k=${CONTEXT_K}`, + '--eval-threshold=0', + '--eval-scores' + ] + + const isRoot = typeof process.getuid === 'function' && process.getuid() === 0 + if (isRoot || process.env.CI) args.push('--no-sandbox') + + const child = spawn(executable, args, { + stdio: ['ignore', 'pipe', 'pipe'], + env: { ...process.env, ELECTRON_DISABLE_SECURITY_WARNINGS: '1' } + }) + + child.on('error', reject) + child.on('exit', (code) => { + if (code !== 0) { + reject(new Error(`diagnostics harness exited with code ${code}`)) + return + } + const reportPath = join(outDir, 'baseline-scores.json') + if (!existsSync(reportPath)) { + reject(new Error('diagnostics harness wrote no report')) + return + } + resolvePromise(JSON.parse(readFileSync(reportPath, 'utf8'))) + }) + }) +} + +/** Nearest-rank quantile, matching `src/main/eval/metrics.ts`. */ +function quantile(values, p) { + if (values.length === 0) return null + const sorted = [...values].sort((a, b) => a - b) + const rank = Math.ceil((p / 100) * sorted.length) + return sorted[Math.min(Math.max(rank - 1, 0), sorted.length - 1)] +} + +/** + * `score = (1 + cosine) / 2`, from `SQLiteVectorStore`. Reported everywhere alongside the + * score because `threshold: 0.5` has been read as "cosine ≥ 0.5" and is really "cosine ≥ 0". + */ +const toCosine = (score) => (score === null ? null : 2 * score - 1) + +const QUANTILES = [0, 10, 25, 50, 75, 90, 100] + +function describe(values) { + if (values.length === 0) return null + const out = {} + for (const p of QUANTILES) out[`p${p}`] = quantile(values, p) + return out +} + +const workDir = mkdtempSync(join(tmpdir(), 'knownote-scores-')) +let report +try { + console.log('[scores] running the dense harness on validation, threshold 0') + report = await runHarness(workDir) +} finally { + rmSync(workDir, { recursive: true, force: true }) +} + +const perQuestion = report.perQuestion ?? [] +const answerable = perQuestion.filter((q) => q.answerable) +const unanswerable = perQuestion.filter((q) => !q.answerable) + +if (perQuestion.some((q) => !q.retrievedScores)) { + console.error('[scores] the report carries no scores; did --eval-scores reach the harness?') + process.exit(1) +} + +/** Per-question score landmarks. */ +function landmarks(question) { + const scores = question.retrievedScores + const relevant = [] + const nonRelevant = [] + scores.forEach((score, index) => { + if ((question.matchesByRank[index] ?? []).length > 0) relevant.push(score) + else nonRelevant.push(score) + }) + if (relevant.length === 0) return null + const bestRelevant = Math.max(...relevant) + const worstRelevant = Math.min(...relevant) + const bestNonRelevant = nonRelevant.length > 0 ? Math.max(...nonRelevant) : null + return { + id: question.id, + type: question.type, + bestRelevant, + worstRelevant, + bestNonRelevant, + margin: bestNonRelevant === null ? null : bestRelevant - bestNonRelevant + } +} + +const answerableLandmarks = answerable.map(landmarks).filter((entry) => entry !== null) +const unanswerableMax = unanswerable.map((q) => Math.max(...q.retrievedScores)) + +const byType = {} +for (const entry of answerableLandmarks) { + byType[entry.type] = byType[entry.type] ?? [] + byType[entry.type].push(entry) +} + +/** + * The curve that decides whether one threshold can do both jobs. For each candidate `t`: + * + * - **hit** — share of answerable questions that still have *some* relevant passage; + * - **fullRecall** — share whose *every* ground-truth block is still covered; + * - **abstain** — share of unanswerable questions that now return nothing. + * + * A useful threshold needs `abstain` to rise while `fullRecall` holds. If every `t` that + * raises abstention also drops full recall, no single threshold can carry the job. + */ +const thresholdGrid = [] +for (let t = 0.5; t <= 1.0001; t += 0.025) thresholdGrid.push(Number(t.toFixed(3))) + +const curve = thresholdGrid.map((threshold) => { + let hit = 0 + let fullRecall = 0 + for (const question of answerable) { + const covered = new Set() + let coveredAbove = 0 + question.matchesByRank.forEach((matches, index) => { + if (question.retrievedScores[index] < threshold) return + for (const match of matches) { + if (!covered.has(match)) { + covered.add(match) + coveredAbove += 1 + } + } + }) + if (coveredAbove > 0) hit += 1 + if (question.relevantCount > 0 && coveredAbove === question.relevantCount) fullRecall += 1 + } + + const abstained = unanswerable.filter((q) => Math.max(...q.retrievedScores) < threshold).length + return { + threshold, + cosine: Number(toCosine(threshold).toFixed(3)), + hitRate: answerable.length === 0 ? 0 : hit / answerable.length, + fullRecallRate: answerable.length === 0 ? 0 : fullRecall / answerable.length, + abstentionRate: unanswerable.length === 0 ? 0 : abstained / unanswerable.length + } +}) + +const format4 = (value) => (value === null ? '—' : value.toFixed(4)) +const quantileRow = (label, values) => { + const d = describe(values) + if (!d) return `| ${label} | 0 | ${QUANTILES.map(() => '—').join(' | ')} |` + const cells = QUANTILES.map((p) => `${d[`p${p}`].toFixed(4)} (${toCosine(d[`p${p}`]).toFixed(3)})`) + return `| ${label} | ${values.length} | ${cells.join(' | ')} |` +} + +const relevantBest = answerableLandmarks.map((entry) => entry.bestRelevant) +const relevantWorst = answerableLandmarks.map((entry) => entry.worstRelevant) +const nonRelevantBest = answerableLandmarks + .map((entry) => entry.bestNonRelevant) + .filter((value) => value !== null) +const margins = answerableLandmarks.map((entry) => entry.margin).filter((value) => value !== null) + +const typeRows = Object.entries(byType) + .sort(([a], [b]) => a.localeCompare(b)) + .map(([type, entries]) => { + const best = describe(entries.map((entry) => entry.bestRelevant)) + const worst = describe(entries.map((entry) => entry.worstRelevant)) + return `| ${type} | ${entries.length} | ${format4(best.p50)} | ${format4(best.p10)} | ${format4(worst.p10)} |` + }) + .join('\n') + +const curveRows = curve + .map( + (row) => + `| ${row.threshold.toFixed(3)} | ${row.cosine.toFixed(3)} | ${format4(row.hitRate)} | ` + + `${format4(row.fullRecallRate)} | ${format4(row.abstentionRate)} |` + ) + .join('\n') + +/** Where the three distributions overlap, which is what the threshold decision turns on. */ +const overlap = { + worstRelevantP10: quantile(relevantWorst, 10), + bestNonRelevantP90: quantile(nonRelevantBest, 90), + unanswerableMaxP50: quantile(unanswerableMax, 50), + unanswerableMaxP90: quantile(unanswerableMax, 90) +} +const separable = + overlap.worstRelevantP10 !== null && + overlap.unanswerableMaxP90 !== null && + overlap.worstRelevantP10 > overlap.unanswerableMaxP90 + +const markdown = `# Dense score diagnostics — v1.6 (#192) + +Generated by \`node scripts/eval-scores.mjs\`. Numbers are harness output; do not edit them by hand. + +## Read this first: the score is not a cosine + +\`SQLiteVectorStore\` computes + +\`\`\`text +score = 1 - distance / 2 +distance = 1 - cosine (sqlite-vec, distance_metric=cosine) +=> score = (1 + cosine) / 2 +\`\`\` + +so the configured \`threshold\` is an **affine map of the cosine**, not the cosine: + +| threshold | raw cosine | +| --- | --- | +| 0.3 | -0.4 | +| 0.4 | -0.2 | +| **0.5 (shipped)** | **0.0** | +| 0.6 | 0.2 | +| 0.8 | 0.6 | +| 1.0 | 1.0 | + +The shipped \`threshold = 0.5\` means **cosine ≥ 0**, which is very permissive. Every +threshold row below carries both columns so the two never get confused again, and every +mention of "the sweep looked too low" has to be read through this table. + +## What was measured + +Dense only, \`${SPLIT}\` split only, \`threshold = 0\`, \`candidateK = ${CANDIDATE_K}\` (above the +index size, so every chunk is scored for every query), \`contextK = ${CONTEXT_K}\`. + +- answerable questions: ${answerable.length} +- unanswerable questions: ${unanswerable.length} +- index size: ${report.config.chunkCount} chunks + +Hybrid is deliberately excluded: its \`score\` is an RRF value (\`1 / (60 + rank)\`) and is not +on the same scale as a normalised cosine. + +## Distributions (score, with raw cosine in brackets) + +\`best relevant\` is the highest-scoring passage that covers ground truth; \`worst relevant\` +is the lowest one that still has to survive for the question to be fully answered; +\`best non-relevant\` is the highest-scoring passage that covers nothing; \`margin\` is the +first minus the third. + +| Distribution | n | min | p10 | p25 | p50 | p75 | p90 | max | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | +${quantileRow('best relevant', relevantBest)} +${quantileRow('worst relevant', relevantWorst)} +${quantileRow('best non-relevant', nonRelevantBest)} +${quantileRow('margin (best rel − best non-rel)', margins)} +${quantileRow('unanswerable max candidate', unanswerableMax)} + +## By query type + +\`best relevant\` p50 and p10, and \`worst relevant\` p10 — the last is the one that decides +whether a cross-lingual question survives a threshold that a semantic question tolerates. + +| Type | n | best rel p50 | best rel p10 | worst rel p10 | +| --- | --- | --- | --- | --- | +${typeRows} + +## Threshold curve + +For each candidate threshold: **hit** = share of answerable questions that still have some +relevant passage; **full recall** = share whose every ground-truth block is still covered; +**abstain** = share of unanswerable questions that now return nothing. + +| threshold (score) | raw cosine | answerable hit | answerable full recall | unanswerable abstain | +| --- | --- | --- | --- | --- | +${curveRows} + +## The separation question + +The threshold decision turns on whether these overlap: + +| Landmark | score | raw cosine | +| --- | --- | --- | +| worst relevant, p10 | ${format4(overlap.worstRelevantP10)} | ${overlap.worstRelevantP10 === null ? '—' : toCosine(overlap.worstRelevantP10).toFixed(3)} | +| best non-relevant, p90 | ${format4(overlap.bestNonRelevantP90)} | ${overlap.bestNonRelevantP90 === null ? '—' : toCosine(overlap.bestNonRelevantP90).toFixed(3)} | +| unanswerable max, p50 | ${format4(overlap.unanswerableMaxP50)} | ${overlap.unanswerableMaxP50 === null ? '—' : toCosine(overlap.unanswerableMaxP50).toFixed(3)} | +| unanswerable max, p90 | ${format4(overlap.unanswerableMaxP90)} | ${overlap.unanswerableMaxP90 === null ? '—' : toCosine(overlap.unanswerableMaxP90).toFixed(3)} | + +**${ + separable + ? 'The distributions separate at the p10/p90 landmarks, so a single threshold is a plausible mechanism on this corpus. Read the grid above for where to sweep.' + : 'The distributions **overlap**, so a higher threshold buys abstention by giving up required relevant passages. If the curve above shows abstention rising only as full recall falls, then the honest conclusion is that **a single dense similarity threshold cannot carry both recall and abstention** — and the next mechanism to evaluate is not a finer threshold grid but a different signal (reranker score, top1−top2 margin, per-query thresholds, or claim-level answerability).' +} + +## Reproduce + +\`\`\`bash +npm run eval:prepare # one-time, networked model bootstrap +npm run eval:scores # offline; rewrites this file +\`\`\` +` + +mkdirSync(resolve(OUT_MD, '..'), { recursive: true }) +writeFileSync( + OUT_JSON, + `${JSON.stringify( + { + baseline: 'v1.6', + split: SPLIT, + candidateK: CANDIDATE_K, + threshold: 0, + indexSize: report.config.chunkCount, + counts: { answerable: answerable.length, unanswerable: unanswerable.length }, + distributions: { + bestRelevant: describe(relevantBest), + worstRelevant: describe(relevantWorst), + bestNonRelevant: describe(nonRelevantBest), + margin: describe(margins), + unanswerableMax: describe(unanswerableMax) + }, + byType: Object.fromEntries( + Object.entries(byType).map(([type, entries]) => [ + type, + { + questions: entries.length, + bestRelevant: describe(entries.map((entry) => entry.bestRelevant)), + worstRelevant: describe(entries.map((entry) => entry.worstRelevant)) + } + ]) + ), + overlap, + separable, + curve + }, + null, + 2 + )}\n` +) +writeFileSync(OUT_MD, markdown) + +console.log(`[scores] answerable ${answerable.length}, unanswerable ${unanswerable.length}, index ${report.config.chunkCount}`) +console.log( + `[scores] worst relevant p10 = ${format4(overlap.worstRelevantP10)} (cosine ${overlap.worstRelevantP10 === null ? '—' : toCosine(overlap.worstRelevantP10).toFixed(3)}), unanswerable max p90 = ${format4(overlap.unanswerableMaxP90)}` +) +console.log(`[scores] separable = ${separable}`) +console.log(`[scores] wrote ${OUT_JSON} and ${OUT_MD}`) diff --git a/src/main/eval/harness.ts b/src/main/eval/harness.ts index 44427de..f34ee5b 100644 --- a/src/main/eval/harness.ts +++ b/src/main/eval/harness.ts @@ -72,6 +72,13 @@ export interface EvalHarnessOptions { threshold: number /** 分块配置(#78)。实验变体通过它选择策略;缺省时用生产默认值。 */ chunkOptions: ChunkOptions + /** + * 把每个 rank 的检索分数也写进报告(`--eval-scores`)。 + * + * 缺省关闭:分数序列会让基线膨胀一倍,而基线是 CI 逐字节 diff 的文件。score + * diagnostics(#192)需要它,生产基线不需要。 + */ + includeScores?: boolean /** 检索策略(#77):dense / sparse(BM25) / hybrid(RRF)。 */ strategy: RetrievalStrategy } @@ -298,7 +305,7 @@ export async function runEvalHarness( .filter((index) => index >= 0) }) - perQuestion.push({ + const questionReport: QuestionReport = { id: question.id, question: question.question, type: question.type ?? UNTAGGED, @@ -310,7 +317,10 @@ export async function runEvalHarness( .slice(0, options.contextK) .reduce((total, result) => total + result.content.length, 0), matchesByRank - }) + } + // Only when asked: the score series doubles the size of the committed baseline. + if (options.includeScores) questionReport.retrievedScores = results.map((r) => r.score) + perQuestion.push(questionReport) } // 不可答的问题不进排名指标:它们没有 ground truth,`recallAtK` 对它们返回的是 0/0 @@ -396,6 +406,14 @@ function roundMetrics(metrics: EvalMetrics): EvalMetrics { } export function stabilize(report: EvalReport): EvalReport { + // Scores are extra data, not metrics; they are rounded the same way so two runs of the + // same diagnostic diff cleanly. + const perQuestion = report.perQuestion.map((question) => + question.retrievedScores + ? { ...question, retrievedScores: question.retrievedScores.map(roundMetric) } + : question + ) + return { ...report, metrics: roundMetrics(report.metrics), @@ -407,7 +425,7 @@ export function stabilize(report: EvalReport): EvalReport { // full report keeps it for the #78 comparison. indexingMs: Math.round(report.timing.indexingMs) }, - perQuestion: report.perQuestion + perQuestion } } diff --git a/src/main/eval/run.ts b/src/main/eval/run.ts index 20c13fa..ad45091 100644 --- a/src/main/eval/run.ts +++ b/src/main/eval/run.ts @@ -200,6 +200,9 @@ export async function runEvalCli(argv: readonly string[] = process.argv): Promis contextK: readNumberOption(argv, '--eval-context-k=', 3), threshold: readNumberOption(argv, '--eval-threshold=', 0.5), chunkOptions: readChunkOptions(argv), + // `--eval-scores` puts the per-rank retrieval scores in the report. Off by default: + // the committed baseline is a CI-diffed file and the series doubles it. + includeScores: readBoolOption(argv, '--eval-scores=', argv.includes('--eval-scores')), strategy: readRetrievalStrategy(argv) } diff --git a/src/main/eval/types.ts b/src/main/eval/types.ts index 715396a..6a33005 100644 --- a/src/main/eval/types.ts +++ b/src/main/eval/types.ts @@ -197,6 +197,15 @@ export interface QuestionReport { contextChars: number /** Ground-truth indices matched by each retrieved rank, in rank order. */ matchesByRank: number[][] + /** + * 每个 rank 的检索分数,与 `matchesByRank` 同序。 + * + * 只在 `--eval-scores` 时出现:它是 score diagnostics(#192)需要的数据,而基线 + * JSON 默认不带它——分数序列会让基线膨胀一倍,而基线是 CI 要逐字节 diff 的文件。 + * + * 注意它不是 cosine:见 `SQLiteVectorStore`,`score = (1 + cosine) / 2`。 + */ + retrievedScores?: number[] } export interface EvalReport { diff --git a/src/main/vectorstore/SQLiteVectorStore.ts b/src/main/vectorstore/SQLiteVectorStore.ts index 8fc300b..8f696f1 100644 --- a/src/main/vectorstore/SQLiteVectorStore.ts +++ b/src/main/vectorstore/SQLiteVectorStore.ts @@ -171,7 +171,11 @@ export class SQLiteVectorStore implements VectorStore { // 转换结果 const queryResults: QueryResult[] = results.map((row) => { - // cosine 距离转相似度(0-1) + // `score = 1 - distance / 2 = (1 + cosine) / 2`。 + // + // **这不是 cosine 本身**,而是仿射映射:score 0.5 对应 cosine 0,score 0.6 对应 + // cosine 0.2,score 1.0 才是 cosine 1.0。配置里的 `threshold` 就是这个 score, + // 把 0.5 读成“cosine ≥ 0.5”会把门槛高估很多(#192 评审)。 const score = 1 - row.distance / 2 return { From 036a95ae71f554a9f427a8cb43ca0129e1ce7f63 Mon Sep 17 00:00:00 2001 From: MrSibe Date: Wed, 30 Sep 2026 18:13:54 +0800 Subject: [PATCH 2/2] =?UTF-8?q?fix(eval):=20convert=20a=20score=20margin?= =?UTF-8?q?=20with=20=C3=972,=20not=20with=20the=20absolute=20affine=20tra?= =?UTF-8?q?nsform?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `score = (1 + cosine) / 2` is an affine map, so an **absolute** score converts as `cosine = 2·score − 1`. A **difference** of scores does not: the `+1` cancels and `Δcosine = 2·Δscore`. The margin row applied the absolute transform, which reported `2Δscore − 1` — for a margin near zero that is a cosine margin near **−1**, a sign flip on top of a scale error: ``` before after margin p10 -0.0272 (-1.054) -0.0272 (-0.054) margin p50 +0.0026 (-0.995) +0.0026 (+0.005) ``` `quantileRow` now takes the cosine transform as an argument and the margin row passes `toCosineMargin`, so the two cannot be confused at the call site again. The report also says which mapping applies to which row. This does not change the separability conclusion — the p90 for the best non-relevant passage is 0.9386 against a worst-required-relevant p10 of 0.8906, and that comparison uses absolute scores, which were always converted correctly. The margin itself is also now labelled in the report as an **oracle** quantity: at runtime nothing knows which result is relevant, so it measures how much the score separates the two, and it is not a signal the product could use. Part of #192 (correction to the score diagnostics). --- docs/eval/scores-v1.6.md | 11 ++++++++++- scripts/eval-scores.mjs | 30 +++++++++++++++++++++++++++--- 2 files changed, 37 insertions(+), 4 deletions(-) diff --git a/docs/eval/scores-v1.6.md b/docs/eval/scores-v1.6.md index 2ad38b7..18d8928 100644 --- a/docs/eval/scores-v1.6.md +++ b/docs/eval/scores-v1.6.md @@ -46,12 +46,21 @@ is the lowest one that still has to survive for the question to be fully answere `best non-relevant` is the highest-scoring passage that covers nothing; `margin` is the first minus the third. +The margin is an **oracle** quantity: at runtime nothing knows which result is relevant, so +it describes how much the score separates the two — it is not a signal a product could use. +Reading it as a candidate mechanism is the mistake the runtime-signal evaluation exists to +avoid. + +Where a row shows a raw cosine in brackets: an **absolute** score maps as +`cosine = 2·score − 1`, while a **margin** maps as `Δcosine = 2·Δscore` because the +`+1` cancels. The `margin` row uses the latter, the others the former. + | Distribution | n | min | p10 | p25 | p50 | p75 | p90 | max | | --- | --- | --- | --- | --- | --- | --- | --- | --- | | best relevant | 19 | 0.8808 (0.762) | 0.8906 (0.781) | 0.9120 (0.824) | 0.9359 (0.872) | 0.9428 (0.886) | 0.9530 (0.906) | 0.9537 (0.907) | | worst relevant | 19 | 0.8808 (0.762) | 0.8906 (0.781) | 0.9086 (0.817) | 0.9355 (0.871) | 0.9428 (0.886) | 0.9530 (0.906) | 0.9537 (0.907) | | best non-relevant | 19 | 0.8904 (0.781) | 0.9081 (0.816) | 0.9178 (0.836) | 0.9243 (0.849) | 0.9336 (0.867) | 0.9386 (0.877) | 0.9459 (0.892) | -| margin (best rel − best non-rel) | 19 | -0.0294 (-1.059) | -0.0272 (-1.054) | -0.0092 (-1.018) | 0.0026 (-0.995) | 0.0081 (-0.984) | 0.0329 (-0.934) | 0.0623 (-0.875) | +| margin (best rel − best non-rel) | 19 | -0.0294 (-0.059) | -0.0272 (-0.054) | -0.0092 (-0.018) | 0.0026 (0.005) | 0.0081 (0.016) | 0.0329 (0.066) | 0.0623 (0.125) | | unanswerable max candidate | 5 | 0.8978 (0.796) | 0.8978 (0.796) | 0.9128 (0.826) | 0.9259 (0.852) | 0.9343 (0.869) | 0.9424 (0.885) | 0.9424 (0.885) | ## By query type diff --git a/scripts/eval-scores.mjs b/scripts/eval-scores.mjs index 1356f2a..43e5bbf 100644 --- a/scripts/eval-scores.mjs +++ b/scripts/eval-scores.mjs @@ -101,6 +101,14 @@ function quantile(values, p) { */ const toCosine = (score) => (score === null ? null : 2 * score - 1) +/** + * A difference of two scores is not an affine map of a difference of two cosines: the + * `+1` cancels, so `Δcosine = 2 * Δscore`. Applying the absolute transform to a margin + * would have produced `2Δscore - 1`, which for a near-zero margin reports a cosine margin + * near −1 — a sign flip on top of a scale error. + */ +const toCosineMargin = (margin) => (margin === null ? null : 2 * margin) + const QUANTILES = [0, 10, 25, 50, 75, 90, 100] function describe(values) { @@ -203,10 +211,17 @@ const curve = thresholdGrid.map((threshold) => { }) const format4 = (value) => (value === null ? '—' : value.toFixed(4)) -const quantileRow = (label, values) => { +/** + * `transform` maps a value to its cosine counterpart. It defaults to the absolute-score + * transform and is overridden for margins, because the two do not share one. + */ +const quantileRow = (label, values, transform = toCosine) => { const d = describe(values) if (!d) return `| ${label} | 0 | ${QUANTILES.map(() => '—').join(' | ')} |` - const cells = QUANTILES.map((p) => `${d[`p${p}`].toFixed(4)} (${toCosine(d[`p${p}`]).toFixed(3)})`) + const cells = QUANTILES.map((p) => { + const value = d[`p${p}`] + return `${value.toFixed(4)} (${transform(value).toFixed(3)})` + }) return `| ${label} | ${values.length} | ${cells.join(' | ')} |` } @@ -294,12 +309,21 @@ is the lowest one that still has to survive for the question to be fully answere \`best non-relevant\` is the highest-scoring passage that covers nothing; \`margin\` is the first minus the third. +The margin is an **oracle** quantity: at runtime nothing knows which result is relevant, so +it describes how much the score separates the two — it is not a signal a product could use. +Reading it as a candidate mechanism is the mistake the runtime-signal evaluation exists to +avoid. + +Where a row shows a raw cosine in brackets: an **absolute** score maps as +\`cosine = 2·score − 1\`, while a **margin** maps as \`Δcosine = 2·Δscore\` because the +\`+1\` cancels. The \`margin\` row uses the latter, the others the former. + | Distribution | n | min | p10 | p25 | p50 | p75 | p90 | max | | --- | --- | --- | --- | --- | --- | --- | --- | --- | ${quantileRow('best relevant', relevantBest)} ${quantileRow('worst relevant', relevantWorst)} ${quantileRow('best non-relevant', nonRelevantBest)} -${quantileRow('margin (best rel − best non-rel)', margins)} +${quantileRow('margin (best rel − best non-rel)', margins, toCosineMargin)} ${quantileRow('unanswerable max candidate', unanswerableMax)} ## By query type