From 25e1ada24ee07e1d3c36af88c65f7b52aad2b8b3 Mon Sep 17 00:00:00 2001 From: MrSibe Date: Wed, 30 Sep 2026 16:40:03 +0800 Subject: [PATCH 01/18] fix(eval): run the harness on Windows and keep the baseline POSIX MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two gates the v1.5 eval path needs to be trustworthy on every platform. **`npm run eval` could not start on Windows.** The three scripts resolve `node_modules/.bin/electron.cmd` and spawn it. Node 24 refuses to spawn a `.cmd`/`.bat` without `shell: true` and fails with `EINVAL`, so the harness was unrunnable on the platform this project is developed on. The `electron` package exports the path to the real executable the wrapper runs, which spawns directly on every platform and needs no shell. **A Windows run rewrote the committed baseline.** `path.relative` returns backslashes on Windows, so `config.corpus` was written as `eval\corpus` instead of `eval/corpus`. The file's whole contract is that it is identical on every machine — the CI determinism check diffs it — and a Windows run silently broke that. The label is now normalised to POSIX separators. Verified on Windows with the pinned model: `npm run eval` runs and leaves `docs/eval/baseline-v1.5.json` byte-identical to the committed file. --- scripts/eval-chunking.mjs | 8 ++++---- scripts/eval-retrieval.mjs | 8 ++++---- scripts/eval.mjs | 10 ++++++---- src/main/eval/run.ts | 20 ++++++++++++++++---- 4 files changed, 30 insertions(+), 16 deletions(-) diff --git a/scripts/eval-chunking.mjs b/scripts/eval-chunking.mjs index ecc0d4a..70f4e45 100644 --- a/scripts/eval-chunking.mjs +++ b/scripts/eval-chunking.mjs @@ -63,10 +63,10 @@ function readArg(prefix, fallback) { return arg ? arg.slice(prefix.length) : fallback } -const executable = resolve( - 'node_modules/.bin', - process.platform === 'win32' ? 'electron.cmd' : 'electron' -) +// Node 24 refuses to spawn a `.cmd`/`.bat` without `shell: true` (EINVAL), and the +// `.bin` entry is exactly that on Windows. Use the real binary the wrapper runs. +const { default: electronBinary } = await import('electron') +const executable = resolve(electronBinary) if (!existsSync(executable)) { console.error('[chunking] could not find the electron binary. Run `npm install` first.') diff --git a/scripts/eval-retrieval.mjs b/scripts/eval-retrieval.mjs index d573e3b..bccc605 100644 --- a/scripts/eval-retrieval.mjs +++ b/scripts/eval-retrieval.mjs @@ -39,10 +39,10 @@ function readArg(prefix, fallback) { return arg ? arg.slice(prefix.length) : fallback } -const executable = resolve( - 'node_modules/.bin', - process.platform === 'win32' ? 'electron.cmd' : 'electron' -) +// Node 24 refuses to spawn a `.cmd`/`.bat` without `shell: true` (EINVAL), and the +// `.bin` entry is exactly that on Windows. Use the real binary the wrapper runs. +const { default: electronBinary } = await import('electron') +const executable = resolve(electronBinary) if (!existsSync(executable)) { console.error('[retrieval] could not find the electron binary. Run `npm install` first.') diff --git a/scripts/eval.mjs b/scripts/eval.mjs index 0f4d304..8b61229 100644 --- a/scripts/eval.mjs +++ b/scripts/eval.mjs @@ -18,10 +18,12 @@ import { resolve } from 'node:path' const prepare = process.argv.includes('--prepare') const flag = prepare ? '--eval-prepare' : '--eval-harness' -const executable = resolve( - 'node_modules/.bin', - process.platform === 'win32' ? 'electron.cmd' : 'electron' -) +// The `.bin` entry is a shell wrapper (`electron.cmd` on Windows), and Node 24 +// refuses to spawn `.cmd`/`.bat` without `shell: true` — it fails with EINVAL. The +// `electron` package exports the path to the real executable the wrapper runs, so +// spawning that directly works on every platform without a shell. +const { default: electronBinary } = await import('electron') +const executable = resolve(electronBinary) if (!existsSync(executable)) { console.error('[eval] could not find the electron binary. Run `npm install` first.') diff --git a/src/main/eval/run.ts b/src/main/eval/run.ts index 10857a6..58becb8 100644 --- a/src/main/eval/run.ts +++ b/src/main/eval/run.ts @@ -13,7 +13,7 @@ import { app } from 'electron' import { mkdtempSync, rmSync, writeFileSync } from 'fs' import { mkdir } from 'fs/promises' -import { join, relative, resolve } from 'path' +import { join, relative, resolve, sep } from 'path' import { tmpdir } from 'os' import { closeDatabase, getDatabase, initDatabase, initVectorStore, runMigrations } from '../db' import { ConnectionManager } from '../models/ConnectionManager' @@ -62,6 +62,18 @@ function readBoolOption(argv: readonly string[], prefix: string, fallback: boole return value === 'true' } +/** + * Repo-relative path with POSIX separators. + * + * `path.relative` returns backslashes on Windows, and the corpus label is committed + * in the baseline. Without this, a Windows run writes `eval\corpus` and the file + * stops being identical on every machine — which is the one property the committed + * report promises. + */ +function repoRelative(absolutePath: string): string { + return relative(process.cwd(), absolutePath).split(sep).join('/') +} + /** 检索策略(#77)。默认 dense,所以不带 flag 的 `npm run eval` 仍量的是生产默认。 */ function readRetrievalStrategy(argv: readonly string[]): RetrievalStrategy { const raw = readOption(argv, '--eval-retrieval=', 'dense') @@ -162,9 +174,9 @@ export async function runEvalCli(argv: readonly string[] = process.argv): Promis const knowledgeService = new KnowledgeService(embeddingService) const options: EvalHarnessOptions = { corpusDir, - // Recorded in the report as a repo-relative path so the committed JSON is - // identical on every machine and checkout. - corpusLabel: relative(process.cwd(), corpusDir) || 'eval/corpus', + // Recorded in the report as a repo-relative, POSIX-separated path so the + // committed JSON is identical on every machine and checkout. + corpusLabel: repoRelative(corpusDir) || 'eval/corpus', questionsPath, baseline: readOption(argv, '--eval-baseline=', 'v1.5'), topK: 10, From a444a8676fea8c9540c487b9d23654faa3fa5038 Mon Sep 17 00:00:00 2001 From: MrSibe Date: Wed, 30 Sep 2026 16:46:12 +0800 Subject: [PATCH 02/18] feat(eval): separate candidateK from contextK, and run the production config MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The harness measured a retriever nobody runs. Production retrieved at `topK: 3` with `threshold: 0.5`; the harness ran `topK: 10` with `threshold: 0` and reported `Recall@5 = 1.0000` for a pipeline that silently drops rank 4 at cosine 0.47. The first child of #192 asks for the two to be the same configuration. **One K was doing two jobs.** `RetrievalRequest.topK` was both the first-stage width (KNN neighbours, BM25 limit) and the number of passages delivered. In hybrid that made fusion nearly a no-op: dense contributed `topK`, BM25 contributed `topK`, RRF fused at most `2 * topK`, and the result was sliced straight back to `topK`. The two stages are now named: ``` RetrievalRequest: candidateK first-stage width per channel (default 20) topK passages delivered (default 5) ``` `effectiveCandidateK` enforces `candidateK >= topK`, so a caller that asks for more results than the default pool (the search palette, MCP) is never silently capped. `HybridRetriever` fuses the wide pool and truncates once, after fusion. `DenseRetriever` queries `candidateK` and slices to `topK` — equivalent to before for a single strategy, since a prefix of a ranking is the same ranking. The trace and the #157 snapshot gained `candidateK`. A snapshot written before this change backfills it from `topK`, which is what that retrieval actually did. **The harness now defaults to production** (`candidateK: 20`, `contextK: 3`, `threshold: 0.5`), with `--eval-candidate-k=` / `--eval-context-k=` / `--eval-threshold=` to move them deliberately. Ranking metrics are computed at `candidateK` depth, not `contextK`: `Recall@10` needs ten results, and truncation only takes a prefix, so it cannot change the ranking being measured. `contextK` is recorded so the report describes the whole online path. `docs/eval/baseline-v1.6.{json,md}` is the new frozen baseline; v1.5 is kept as history for the #77/#78 deltas, and the CI determinism check moves to v1.6. ## What this did and did not change On the current 19-chunk corpus the production configuration produces the **same** metrics as v1.5 — `threshold: 0.5` is non-binding and `candidateK: 20` exceeds the corpus, so nothing is filtered and no ranking changes. That is the corpus limitation #192 describes, not a result. What changed is that the harness now *states* the production parameters instead of assuming different ones. Hybrid still leads dense on Recall@1 (0.8667 vs 0.8333), MRR (0.9444 vs 0.9278) and nDCG@10 (0.9561 vs 0.9437) with a wide pool; dense stays the default because the adoption rule still cannot move on a saturated Recall@5. ## Testing - `npm run typecheck` — clean - `npm test` — 481 pass (3 new: trace keeps the two Ks apart, the width invariant, the pre-#77 snapshot backfill) - `npm run eval` twice — byte-identical `docs/eval/baseline-v1.6.json` - `npm run eval:retrieval` — dense/sparse/hybrid unchanged in ordering Part of #192 (child 1). --- .github/workflows/verify.yml | 6 +- docs/eval/baseline-v1.6.json | 937 ++++++++++++++++++ docs/eval/baseline-v1.6.md | 65 ++ eval/README.md | 26 +- src/main/eval/harness.ts | 23 +- src/main/eval/report.ts | 10 +- src/main/eval/run.ts | 9 +- src/main/eval/types.ts | 13 +- src/main/ipc/chatHandlers.ts | 5 + src/main/services/KnowledgeService.ts | 6 + src/main/services/retrieval/DenseRetriever.ts | 17 +- .../services/retrieval/HybridRetriever.ts | 25 +- src/main/services/retrieval/trace.ts | 2 + src/main/services/retrieval/types.ts | 42 + src/shared/types/chat.ts | 9 +- src/shared/utils/answerSources.ts | 5 + test/answerSources.test.ts | 15 + test/retrievalContract.test.ts | 36 +- 18 files changed, 1221 insertions(+), 30 deletions(-) create mode 100644 docs/eval/baseline-v1.6.json create mode 100644 docs/eval/baseline-v1.6.md diff --git a/.github/workflows/verify.yml b/.github/workflows/verify.yml index 03f1843..597b1bf 100644 --- a/.github/workflows/verify.yml +++ b/.github/workflows/verify.yml @@ -93,7 +93,7 @@ jobs: if: runner.os != 'Linux' run: npm run smoke:packaged - # The eval baseline (#75) is the reference point every v1.5 experiment is + # The eval baseline (#75) is the reference point every experiment is # measured against, so CI proves the committed numbers still reproduce. The # model cache is keyed on the pinned model file, so the 134 MB download # happens once per pin, not once per run. A hit is required for the run to @@ -110,6 +110,6 @@ jobs: run: | xvfb-run -a npm run eval:prepare xvfb-run -a node scripts/eval.mjs - cp docs/eval/baseline-v1.5.json /tmp/eval-a.json + cp docs/eval/baseline-v1.6.json /tmp/eval-a.json xvfb-run -a node scripts/eval.mjs - diff /tmp/eval-a.json docs/eval/baseline-v1.5.json + diff /tmp/eval-a.json docs/eval/baseline-v1.6.json diff --git a/docs/eval/baseline-v1.6.json b/docs/eval/baseline-v1.6.json new file mode 100644 index 0000000..eb15f14 --- /dev/null +++ b/docs/eval/baseline-v1.6.json @@ -0,0 +1,937 @@ +{ + "baseline": "v1.6", + "generatedBy": "npm run eval", + "config": { + "embedding": "Xenova/multilingual-e5-small@761b726dd34fb83930e26aab4e9ac3899aa1fa78 q8 (384d, local)", + "chunking": { + "chunkSize": 1000, + "chunkOverlap": 100, + "minChunkSize": 100, + "allowSpanPages": false, + "respectHeadings": false + }, + "retrieval": "dense", + "candidateK": 20, + "contextK": 3, + "threshold": 0.5, + "evidenceK": 5, + "corpus": "eval/corpus", + "documents": 13, + "questions": 30, + "chunkCount": 19 + }, + "metrics": { + "recallAt1": 0.833333, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 0.927778, + "ndcgAt10": 0.94375, + "evidencePrecisionAt5": 0.213333 + }, + "perQuestion": [ + { + "id": "q001", + "question": "Why is bedload harder to measure than suspended sediment?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q002", + "question": "How many replicate samples are collected at each river station?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q003", + "question": "What is the central trade-off in lithium-ion cell design?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q004", + "question": "Why do nickel-rich battery packs need more aggressive thermal management?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q005", + "question": "What happens once the separator in a battery cell melts?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q006", + "question": "At what temperature do honeybees begin to forage?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q007", + "question": "What does a late frost damage during full bloom?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q008", + "question": "Why is a continuous tree canopy more effective at cooling than isolated trees?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q009", + "question": "Why are trees with aggressive surface roots unsuitable for narrow verges?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q010", + "question": "At what temperature is lactic acid fermentation fastest?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q011", + "question": "Is the salt percentage in fermentation based on vegetable weight or water weight?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q012", + "question": "Why must tidal turbines be sited in places with very fast currents?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q013", + "question": "What is the main environmental concern for tidal energy installations?", + "firstRelevantRank": 3, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [], + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q014", + "question": "In lake monitoring, how is the sampling depth actually recorded?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q015", + "question": "Why does deep-water oxygen fall while a lake remains stratified?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q016", + "question": "How do supercapacitors hold their charge?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q017", + "question": "Why can solitary bees pollinate a bloom week that is too cold for honeybee hives?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q018", + "question": "Why is one continuous planted roof layer better than several isolated planted beds?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q019", + "question": "Where do acetic acid bacteria sit in a vinegar culture?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q020", + "question": "What happens if a vinegar culture is sealed airtight?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q021", + "question": "Why is wave energy harder to schedule ahead than tidal energy?", + "firstRelevantRank": 2, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q022", + "question": "Where does siting for wave energy devices concentrate, and where does it not?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q023", + "question": "How does the river sampling protocol differ from the lake sampling protocol?", + "firstRelevantRank": 1, + "relevantCount": 2, + "retrievedCount": 19, + "matchesByRank": [ + [ + 1 + ], + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q024", + "question": "A street canopy and a green roof are both said to cool; what surface does each one shade?", + "firstRelevantRank": 1, + "relevantCount": 2, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [ + 1 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q025", + "question": "Which preservation method depends on keeping air away from the food?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q026", + "question": "Why can one cold morning cost a grower the whole crop even when colonies are brought in?", + "firstRelevantRank": 2, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q027", + "question": "绿茶应该怎样保存才能减缓氧化?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q028", + "question": "茶叶储存的相对湿度上限是多少?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q029", + "question": "为什么冷冻保存的茶叶取出后不能立刻打开包装?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q030", + "question": "为什么潮汐能比风能和太阳能更容易提前安排发电?", + "firstRelevantRank": 2, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + } + ] +} diff --git a/docs/eval/baseline-v1.6.md b/docs/eval/baseline-v1.6.md new file mode 100644 index 0000000..5418dd6 --- /dev/null +++ b/docs/eval/baseline-v1.6.md @@ -0,0 +1,65 @@ +# RAG eval baseline — v1.6 + +Generated by `npm run eval`. The numbers below are harness output — do not edit them by hand. + +## Configuration + +| Setting | Value | +| --- | --- | +| Embedding | `Xenova/multilingual-e5-small@761b726dd34fb83930e26aab4e9ac3899aa1fa78 q8 (384d, local)` | +| Chunking | `chunkSize=1000, chunkOverlap=100, minChunkSize=100, allowSpanPages=false, respectHeadings=false` | +| Retrieval | `dense` | +| Ranks | `candidateK=20, threshold=0.5` | +| Context width | `contextK=3` | +| Evidence per query | `evidenceK=5` | +| Corpus | `eval/corpus` (13 documents, 30 questions) | +| Index size | 19 chunks | + +## Metrics + +| Metric | Value | +| --- | --- | +| Recall@1 | 0.8333 | +| Recall@5 | 1.0000 | +| Recall@10 | 1.0000 | +| MRR | 0.9278 | +| nDCG@10 | 0.9437 | +| Evidence precision@5 | 0.2133 | + +Timing is informational only and is **not** frozen: indexing 1582 ms, query +p50 11.35 ms, p95 18.82 ms on the +machine that produced this file. Timing and index size depend on hardware and on the +corpus, so they must never be the reason two runs differ. + +## Definitions + +- A retrieved passage is relevant when its provenance covers a ground-truth block. +- **Recall@k** is the share of ground-truth blocks covered by the first `k` passages. +- **Evidence precision@5** is the share of the first `5` + retrieved passages that cover a ground-truth block. This is **retrieval precision**, not + answer citation recall: the harness runs no model and produces no answer. Answer-level + citation correctness is covered by the resolver (#70); a model-driven answer eval is a + separate deliverable. +- Ground truth is expressed in corpus identity (`document` relative path + `block` + ordinal + optional `quote`), never a runtime `documentId`/`blockId`. + +## Comparison protocol + +Experiments (#77, #78) are reported as a **delta against this file**. From v1.6 on, +the harness runs the **production configuration** by default: `candidateK` is the +first-stage width per channel, `contextK` is how many passages the chat prompt takes, +and `threshold` is the similarity floor the app ships. A benchmark that does not +mirror those parameters measures a retriever nobody runs. The adopted-change rule is: + +> Adopt a change only if Recall@5 improves and nDCG@10 does not regress. A change +> that trades a large latency increase for a marginal recall gain is a product +> decision, not an automatic win, and must be stated as such. + +A changed result must be reproducible with: + +```bash +npm run eval:prepare # one-time, networked model bootstrap +npm run eval # offline and deterministic +``` + +The raw report is committed next to this summary as `baseline-v1.6.json`. diff --git a/eval/README.md b/eval/README.md index 095b72a..191ac61 100644 --- a/eval/README.md +++ b/eval/README.md @@ -1,8 +1,8 @@ # RAG eval harness Measures retrieval quality so "did this change make retrieval better?" has an answer. -The current numbers are frozen in [`docs/eval/baseline-v1.4.md`](../docs/eval/baseline-v1.4.md); -every v1.5 experiment (#77, #78) is reported as a delta against that file. +The current numbers are frozen in [`docs/eval/baseline-v1.6.md`](../docs/eval/baseline-v1.6.md); +every experiment (#77, #78) is reported as a delta against that file. ## Commands @@ -11,6 +11,24 @@ npm run eval:prepare # one-time, networked: download the pinned embedding mode npm run eval # offline and deterministic: run the harness, rewrite the baseline ``` +### The harness runs the production configuration + +From v1.6 the harness defaults to the parameters the app ships, so its numbers describe +the product rather than a research setup. Two Ks, because they answer different questions: + +| Flag | Default | Meaning | +| --- | --- | --- | +| `--eval-candidate-k=` | `20` | first-stage width per channel (KNN neighbours, BM25 limit) | +| `--eval-context-k=` | `3` | passages the chat prompt actually takes (`chatHandlers.ts`) | +| `--eval-threshold=` | `0.5` | the similarity floor the app ships | +| `--eval-retrieval=` | `dense` | `dense`, `sparse`, or `hybrid` | +| `--eval-baseline=` | `v1.6` | name written into `docs/eval/baseline-.{json,md}` | + +Ranking metrics are computed at `candidateK` depth, not at `contextK`: `Recall@10` needs +at least ten results, and truncation only takes a prefix of the candidate list, so the +truncation cannot change the ranking it is measured on. `contextK` is recorded so the +report describes the whole online path. + `eval:prepare` downloads the pinned `multilingual-e5-small` revision into the app's model cache and verifies it. `eval` never touches the network: if the model is missing it stops with @@ -87,7 +105,9 @@ from unanswerable queries. 2. Indexes the corpus through the normal ingestion path (`addDocumentFromFile`), so blocks, chunking and embeddings are the real ones. 3. Maps each ground-truth `document`/`block` to the run's runtime ids. -4. Runs the real `Retriever` (`KnowledgeService.search` → `DenseRetriever`). +4. Runs the real `Retriever` (`KnowledgeService.search`) at the configured + `candidateK` depth; `RetrievalRequest` splits the first-stage width from the final + `topK` so the two are not silently the same number. 5. Writes `docs/eval/baseline-.json` (deterministic) and `.md` (with timing). ## Metrics diff --git a/src/main/eval/harness.ts b/src/main/eval/harness.ts index a992415..b23775c 100644 --- a/src/main/eval/harness.ts +++ b/src/main/eval/harness.ts @@ -47,8 +47,16 @@ export interface EvalHarnessOptions { corpusLabel: string questionsPath: string baseline: string - /** Ranks to compute Recall@k for. */ - topK: number + /** + * 第一阶段每个通道的宽度,也是排名指标的评估深度(#77)。 + * + * 与 `contextK` 分开:`candidateK` 决定“找了多宽”(召回),`contextK` 决定“交给 + * LLM 多少”(预算)。只用一个 K 时两者被绑死,benchmark 也无法在足够深的地方算 + * Recall@10。 + */ + candidateK: number + /** 生产 prompt 实际取用的条数(chat 里的 `topK`)。 */ + contextK: number /** Similarity floor; 0 keeps the ranking intact for ranking metrics. */ threshold: number /** How many retrieved passages the evidence-precision metric looks at. */ @@ -190,8 +198,14 @@ export async function runEvalHarness( const groundTruthIds = groundTruth.map((entry) => entry.blockId) const started = performance.now() + // 排名指标在完整的候选深度上计算,不先截到 `contextK`: + // + // - Recall@10 需要至少 10 条结果,而生产的 `contextK` 是 3; + // - 截断只取候选列表的前缀,前缀的排序与截断前一致,所以用宽的结果算排名不等于 + // 把两件事混在一个数里。 const results = await knowledgeService.search(NOTEBOOK_ID, question.question, { - topK: options.topK, + candidateK: options.candidateK, + topK: options.candidateK, threshold: options.threshold, strategy: options.strategy }) @@ -240,7 +254,8 @@ export async function runEvalHarness( respectHeadings: chunking.respectHeadings }, retrieval: options.strategy, - topK: options.topK, + candidateK: options.candidateK, + contextK: options.contextK, threshold: options.threshold, evidenceK: options.evidenceK, corpus: options.corpusLabel, diff --git a/src/main/eval/report.ts b/src/main/eval/report.ts index 1dfbd04..b596704 100644 --- a/src/main/eval/report.ts +++ b/src/main/eval/report.ts @@ -24,7 +24,8 @@ Generated by \`${report.generatedBy}\`. The numbers below are harness output — | Embedding | \`${config.embedding}\` | | Chunking | \`chunkSize=${chunking.chunkSize}, chunkOverlap=${chunking.chunkOverlap}, minChunkSize=${chunking.minChunkSize}, allowSpanPages=${chunking.allowSpanPages}, respectHeadings=${chunking.respectHeadings}\` | | Retrieval | \`${config.retrieval}\` | -| Ranks | \`topK=${config.topK}, threshold=${config.threshold}\` | +| Ranks | \`candidateK=${config.candidateK}, threshold=${config.threshold}\` | +| Context width | \`contextK=${config.contextK}\` | | Evidence per query | \`evidenceK=${config.evidenceK}\` | | Corpus | \`${config.corpus}\` (${config.documents} documents, ${config.questions} questions) | | Index size | ${config.chunkCount} chunks | @@ -59,8 +60,11 @@ corpus, so they must never be the reason two runs differ. ## Comparison protocol -v1.5 experiments (#77, #78) are reported as a **delta against this file**. The -adopted-change rule is: +Experiments (#77, #78) are reported as a **delta against this file**. From v1.6 on, +the harness runs the **production configuration** by default: \`candidateK\` is the +first-stage width per channel, \`contextK\` is how many passages the chat prompt takes, +and \`threshold\` is the similarity floor the app ships. A benchmark that does not +mirror those parameters measures a retriever nobody runs. The adopted-change rule is: > Adopt a change only if Recall@5 improves and nDCG@10 does not regress. A change > that trades a large latency increase for a marginal recall gain is a product diff --git a/src/main/eval/run.ts b/src/main/eval/run.ts index 58becb8..29085ad 100644 --- a/src/main/eval/run.ts +++ b/src/main/eval/run.ts @@ -178,9 +178,12 @@ export async function runEvalCli(argv: readonly string[] = process.argv): Promis // committed JSON is identical on every machine and checkout. corpusLabel: repoRelative(corpusDir) || 'eval/corpus', questionsPath, - baseline: readOption(argv, '--eval-baseline=', 'v1.5'), - topK: 10, - threshold: 0, + baseline: readOption(argv, '--eval-baseline=', 'v1.6'), + // 默认就是生产配置(#77):先取宽,融合,再把 contextK 条送进 prompt。一个不镜像 + // 线上参数的 benchmark 量的是用户永远不会跑的检索器。 + candidateK: readNumberOption(argv, '--eval-candidate-k=', 20), + contextK: readNumberOption(argv, '--eval-context-k=', 3), + threshold: readNumberOption(argv, '--eval-threshold=', 0.5), evidenceK: 5, chunkOptions: readChunkOptions(argv), strategy: readRetrievalStrategy(argv) diff --git a/src/main/eval/types.ts b/src/main/eval/types.ts index 9d3e4e4..c77fb90 100644 --- a/src/main/eval/types.ts +++ b/src/main/eval/types.ts @@ -75,7 +75,18 @@ export interface EvalReport { respectHeadings: boolean } retrieval: string - topK: number + /** + * 第一阶段每个通道的宽度(#77)。排名指标(Recall@K / MRR / nDCG@K)在这个深度上 + * 计算,所以它必须 ≥ 指标里最大的 K。 + */ + candidateK: number + /** + * 生产 prompt 实际取用的证据条数(#77)。 + * + * 快照里记它是为了让 benchmark 描述整条线上链路,而不只是检索器;它不影响排名 + * 指标 —— 截断只是取候选列表的前缀,前缀的排序不变。 + */ + contextK: number threshold: number /** How many retrieved passages the evidence-precision metric looks at. */ evidenceK: number diff --git a/src/main/ipc/chatHandlers.ts b/src/main/ipc/chatHandlers.ts index 6ba6393..00105e9 100644 --- a/src/main/ipc/chatHandlers.ts +++ b/src/main/ipc/chatHandlers.ts @@ -99,9 +99,14 @@ export function registerChatHandlers( // 直接走 `retrieve()` 而不是 `search()`:trace(#157)只有它有,用它再映射出 // SearchResult,不必为了记录参数多检索一次。 + // + // 生产配置就是这两个 K:先取宽(candidateK,每通道 20),融合后只把 3 条送进 + // prompt。这个组合被 eval harness 的默认运行原样量到 —— 否则 benchmark 测的是 + // 一个用户永远不会跑的检索器。 const { evidence, trace } = await knowledgeService.retrieve({ notebookId, query, + candidateK: 20, topK: 3, threshold: 0.5, filter: documentIds ? { documentIds } : undefined diff --git a/src/main/services/KnowledgeService.ts b/src/main/services/KnowledgeService.ts index 9165e2b..11272b1 100644 --- a/src/main/services/KnowledgeService.ts +++ b/src/main/services/KnowledgeService.ts @@ -95,6 +95,11 @@ export interface AddDocumentOptions { */ export interface SearchOptions { topK?: number // 返回结果数量,默认 5 + /** + * 第一阶段每个通道的宽度(#77)。缺省 `DEFAULT_CANDIDATE_K`,且不会小于 `topK`。 + * 搜索面板与 eval harness 用它把「找多宽」和「交多少」分开。 + */ + candidateK?: number threshold?: number // 相似度阈值,默认 0.5 includeContent?: boolean // 是否包含 chunk 内容,默认 true /** 只在这些来源里检索(#94);为空/缺省表示整个 notebook。 */ @@ -1237,6 +1242,7 @@ export class KnowledgeService { const { evidence } = await this.retrieve({ notebookId, query, + candidateK: options.candidateK, topK: options.topK, threshold: options.threshold, strategy: options.strategy, diff --git a/src/main/services/retrieval/DenseRetriever.ts b/src/main/services/retrieval/DenseRetriever.ts index a6187a7..95fece4 100644 --- a/src/main/services/retrieval/DenseRetriever.ts +++ b/src/main/services/retrieval/DenseRetriever.ts @@ -12,6 +12,7 @@ import type { EmbeddingService } from '../EmbeddingService' import type { CandidateHit } from './candidates' import { hydrateEvidence } from './evidence' import { buildRetrievalTrace } from './trace' +import { effectiveCandidateK, DEFAULT_TOP_K } from './types' import type { RetrievalRequest, RetrievalResult, Retriever } from './types' const STRATEGY = 'dense' @@ -27,7 +28,9 @@ export class DenseRetriever implements Retriever { * embed + KNN 抄一遍。 */ async candidateHits(request: RetrievalRequest): Promise { - const topK = request.topK ?? 5 + // 第一阶段按 `candidateK` 取宽;`topK` 的截断由调用方决定,因为 hybrid 需要的是 + // 比最终交付更宽的一池子候选。 + const candidateK = effectiveCandidateK(request) const threshold = request.threshold ?? 0.5 // E5 要求 query 前缀,与索引时的 document 前缀区分 @@ -35,9 +38,9 @@ export class DenseRetriever implements Retriever { const queryEmbedding = await this.embeddingService.embed(request.query, 'query') const vectorStore = await vectorStoreManager.getStore(request.notebookId) - // scope 过滤由向量库在 KNN 之前执行(#94),不是取回 topK 之后再筛。 + // scope 过滤由向量库在 KNN 之前执行(#94),不是取回 candidateK 之后再筛。 const hits = await vectorStore.query(queryEmbedding.embedding, { - topK, + topK: candidateK, threshold, filter: request.filter }) @@ -46,11 +49,14 @@ export class DenseRetriever implements Retriever { } async search(request: RetrievalRequest): Promise { - const topK = request.topK ?? 5 + const topK = request.topK ?? DEFAULT_TOP_K + const candidateK = effectiveCandidateK(request) const threshold = request.threshold ?? 0.5 const startedAt = performance.now() - const hits = await this.candidateHits(request) + // 单策略没有可精排的下游,取宽再截到 `topK` 与直接按 `topK` 查 KNN 等价; + // 两个 K 分开是为了让契约统一,而不是在这里制造差异。 + const hits = (await this.candidateHits(request)).slice(0, topK) const evidence = hits.length === 0 ? [] : hydrateEvidence(getDatabase(), hits) return { @@ -58,6 +64,7 @@ export class DenseRetriever implements Retriever { trace: buildRetrievalTrace({ strategy: STRATEGY, filter: request.filter, + candidateK, topK, threshold, durationMs: performance.now() - startedAt diff --git a/src/main/services/retrieval/HybridRetriever.ts b/src/main/services/retrieval/HybridRetriever.ts index bb899f6..d0a78b9 100644 --- a/src/main/services/retrieval/HybridRetriever.ts +++ b/src/main/services/retrieval/HybridRetriever.ts @@ -5,7 +5,14 @@ import { rrfFuse, type CandidateHit } from './candidates' import { DenseRetriever } from './DenseRetriever' import { hydrateEvidence } from './evidence' import { buildRetrievalTrace } from './trace' -import type { RetrievalRequest, RetrievalResult, RetrievalStrategy, Retriever } from './types' +import { + effectiveCandidateK, + DEFAULT_TOP_K, + type RetrievalRequest, + type RetrievalResult, + type RetrievalStrategy, + type Retriever +} from './types' /** * HybridRetriever (#77) @@ -18,6 +25,10 @@ import type { RetrievalRequest, RetrievalResult, RetrievalStrategy, Retriever } * * 融合在候选层完成(`chunkId + rank`),证据只补齐一次 —— 见 `candidates.ts`。 * 两个通道的分数(cosine 与 BM25)不可比,所以只用 rank,这正是 RRF 的意义。 + * + * 两个 K 是两个阶段:每个通道先按 `candidateK` 取宽(融合池最多 `2 * candidateK`), + * 融合后再截到 `topK`。如果两个通道都只取 `topK`,融合池最多 `2 * topK` 且结果被截回 + * `topK`,融合几乎没有发生空间 —— 那就不是混合检索,只是一个更慢的单路检索。 */ export class HybridRetriever implements Retriever { private readonly dense: DenseRetriever @@ -28,7 +39,8 @@ export class HybridRetriever implements Retriever { async search(request: RetrievalRequest): Promise { const strategy: RetrievalStrategy = request.strategy ?? 'dense' - const topK = request.topK ?? 5 + const candidateK = effectiveCandidateK(request) + const topK = request.topK ?? DEFAULT_TOP_K const startedAt = performance.now() let hits: CandidateHit[] @@ -38,19 +50,19 @@ export class HybridRetriever implements Retriever { if (strategy === 'sparse') { hits = searchChunksFts(request.notebookId, request.query, { - limit: topK, + limit: candidateK, documentIds: request.filter?.documentIds - }) + }).slice(0, topK) } else if (strategy === 'hybrid') { const denseHits = await this.dense.candidateHits(request) const sparseHits = searchChunksFts(request.notebookId, request.query, { - limit: topK, + limit: candidateK, documentIds: request.filter?.documentIds }) hits = rrfFuse([denseHits, sparseHits]).slice(0, topK) } else { threshold = request.threshold ?? 0.5 - hits = await this.dense.candidateHits({ ...request, threshold }) + hits = (await this.dense.candidateHits({ ...request, threshold })).slice(0, topK) } const evidence = hits.length === 0 ? [] : hydrateEvidence(getDatabase(), hits) @@ -60,6 +72,7 @@ export class HybridRetriever implements Retriever { trace: buildRetrievalTrace({ strategy, filter: request.filter, + candidateK, topK, threshold, durationMs: performance.now() - startedAt diff --git a/src/main/services/retrieval/trace.ts b/src/main/services/retrieval/trace.ts index 15243ac..ae788cc 100644 --- a/src/main/services/retrieval/trace.ts +++ b/src/main/services/retrieval/trace.ts @@ -3,6 +3,7 @@ import type { RetrievalFilter, RetrievalTrace } from './types' export interface RetrievalTraceInput { strategy: string filter?: RetrievalFilter + candidateK: number topK: number threshold?: number durationMs: number @@ -24,6 +25,7 @@ export function buildRetrievalTrace(input: RetrievalTraceInput): RetrievalTrace const trace: RetrievalTrace = { strategy: input.strategy, scope, + candidateK: input.candidateK, topK: input.topK, durationMs: input.durationMs } diff --git a/src/main/services/retrieval/types.ts b/src/main/services/retrieval/types.ts index 060134b..921a345 100644 --- a/src/main/services/retrieval/types.ts +++ b/src/main/services/retrieval/types.ts @@ -63,10 +63,26 @@ export type RetrievalStrategy = 'dense' | 'sparse' | 'hybrid' * * 取代旧的 `(notebookId, query, options)` 位置参数:#94 的 scope、#77 的策略参数 * 与 #157 要快照的 trace 都要挂在这一个对象上,而不是散落在调用点。 + * + * 两个 K 是分开的,它们回答的是不同的问题: + * + * candidateK 第一阶段每个通道取多宽(KNN 邻居数 / BM25 limit)—— 偏向召回 + * topK 最终交付多少条证据(chat 里就是送进 prompt 的 context 宽度)—— 偏向精度 + * + * 合并成一个 K 会让「向量库直接搜几条」冒充两阶段检索:hybrid 时两个通道各取 + * `topK` 条,融合池最多只有 `2 * topK`,再截回 `topK`,融合几乎没有发生空间。 */ export interface RetrievalRequest { notebookId: string query: string + /** + * 第一阶段候选宽度。缺省 `DEFAULT_CANDIDATE_K`。 + * + * 实际生效值不会小于 `topK`(见 `effectiveCandidateK`):一个 `topK=50` 的调用方 + * 不该因为没写 `candidateK` 而只拿到 20 条。 + */ + candidateK?: number + /** 最终证据条数。缺省 `DEFAULT_TOP_K`。 */ topK?: number threshold?: number filter?: RetrievalFilter @@ -74,16 +90,42 @@ export interface RetrievalRequest { strategy?: RetrievalStrategy } +/** 没有显式指定时的第一阶最宽度(#77)。与 chat 的生产配置保持一致。 */ +export const DEFAULT_CANDIDATE_K = 20 + +/** 没有显式指定时的最终证据条数,与引入 `candidateK` 之前一致。 */ +export const DEFAULT_TOP_K = 5 + +/** + * 第一阶段的真实宽度:`candidateK`,但不小于 `topK`。 + * + * 没有这条不变式,「取出比交付更宽的一池子」只是多数时候成立:任何 `topK > candidateK` + * 的调用(搜索面板的 limit、MCP 的 topK)都会静默地少返结果。 + */ +export function effectiveCandidateK(request: { + candidateK?: number + topK?: number +}): number { + const topK = request.topK ?? DEFAULT_TOP_K + return Math.max(request.candidateK ?? DEFAULT_CANDIDATE_K, topK) +} + /** * 一次检索实际生效的参数。 * * 它会随回答一起被快照(#157),因此必须由检索层产出、而不是调用方猜: * `strategy` 说明用的是哪条检索路径,`scope` 说明结果被限制在哪些来源。 * `threshold` 缺省表示该策略没有阈值,不是「阈值等于 0」。 + * + * `candidateK` 与 `topK` 是两个不同的量,快照里都必须有:只看 `topK` 无法解释 + * 「为什么这次只召回三条」,也无法复现 hybrid 的融合池有多宽。 */ export interface RetrievalTrace { strategy: string scope: { documentIds?: string[] } + /** 第一阶段每个通道的实际宽度(已应用 `effectiveCandidateK`)。 */ + candidateK: number + /** 最终交付的证据条数。 */ topK: number threshold?: number durationMs: number diff --git a/src/shared/types/chat.ts b/src/shared/types/chat.ts index a358ab9..55e7170 100644 --- a/src/shared/types/chat.ts +++ b/src/shared/types/chat.ts @@ -154,6 +154,13 @@ export interface RetrievalSnapshot { strategy: string /** 空对象 = 整个 notebook。 */ scope: { documentIds?: string[] } + /** + * 第一阶段每个通道的宽度(#77)。 + * + * 引入两个 K 之前的快照没有这个字段:那时候选宽度就是 `topK`,所以解析时按 + * `topK` 回填——那是历史事实,不是拿默认值冒充。 + */ + candidateK: number topK: number /** 缺省表示该策略没有阈值,不是「阈值等于 0」。 */ threshold?: number @@ -172,7 +179,7 @@ export interface RetrievalSnapshot { export interface ChatMessageMetadata { sources?: AnswerSource[] retrieval?: RetrievalStatus - /** 本次检索用了什么参数(#157):策略、生效范围、topK、耗时。 */ + /** 本次检索用了什么参数(#157):策略、生效范围、候选宽度、交付条数、耗时。 */ retrievalSnapshot?: RetrievalSnapshot /** * The structured provenance of this answer (#69). `sources` says *what* the diff --git a/src/shared/utils/answerSources.ts b/src/shared/utils/answerSources.ts index 41b51d3..ffb7271 100644 --- a/src/shared/utils/answerSources.ts +++ b/src/shared/utils/answerSources.ts @@ -93,9 +93,14 @@ export const parseRetrievalSnapshot = (metadata: unknown): RetrievalSnapshot | n const durationMs = toFiniteNumber(candidate.durationMs) if (topK === undefined || durationMs === undefined) return null + // Snapshots written before #77 had a single K, so the candidate width *was* topK. + // Backfilling it states what the old retrieval actually did; it is not a guess. + const candidateK = toFiniteNumber(candidate.candidateK) ?? topK + const snapshot: RetrievalSnapshot = { strategy: candidate.strategy, scope: {}, + candidateK, topK, durationMs } diff --git a/test/answerSources.test.ts b/test/answerSources.test.ts index 29e23eb..5de2726 100644 --- a/test/answerSources.test.ts +++ b/test/answerSources.test.ts @@ -151,6 +151,7 @@ test('a retrieval snapshot round-trips', () => { retrievalSnapshot: { strategy: 'dense', scope: { documentIds: ['doc_1', 'doc_2'] }, + candidateK: 20, topK: 8, threshold: 0.5, durationMs: 42.4 @@ -160,6 +161,7 @@ test('a retrieval snapshot round-trips', () => { assert.deepEqual(snapshot, { strategy: 'dense', scope: { documentIds: ['doc_1', 'doc_2'] }, + candidateK: 20, topK: 8, threshold: 0.5, durationMs: 42.4 @@ -175,6 +177,19 @@ test('a snapshot without a scope means the whole notebook', () => { assert.equal(snapshot?.threshold, undefined) }) +/** + * Before #77 there was one K, so the first-stage width *was* topK. Backfilling it + * says what that retrieval did; it is not an invented default. + */ +test('a snapshot written before the two Ks backfills candidateK from topK', () => { + const snapshot = parseRetrievalSnapshot({ + retrievalSnapshot: { strategy: 'dense', scope: {}, topK: 5, durationMs: 1 } + }) + + assert.equal(snapshot?.candidateK, 5) + assert.equal(snapshot?.topK, 5) +}) + test('a snapshot that cannot be rendered is dropped, not guessed', () => { for (const bad of [ undefined, diff --git a/test/retrievalContract.test.ts b/test/retrievalContract.test.ts index 4cfd2a9..9d32292 100644 --- a/test/retrievalContract.test.ts +++ b/test/retrievalContract.test.ts @@ -1,6 +1,10 @@ import { test } from 'node:test' import assert from 'node:assert/strict' import { buildRetrievalTrace } from '../src/main/services/retrieval/trace.ts' +import { + DEFAULT_CANDIDATE_K, + effectiveCandidateK +} from '../src/main/services/retrieval/types.ts' /** * The retrieval contract (#160) is the seam #94, #157 and #77 build on. These pin @@ -14,6 +18,7 @@ import { buildRetrievalTrace } from '../src/main/services/retrieval/trace.ts' test('a trace carries the effective search parameters and no empty fields', () => { const trace = buildRetrievalTrace({ strategy: 'dense', + candidateK: 20, topK: 5, threshold: 0.5, durationMs: 12.5 @@ -22,21 +27,50 @@ test('a trace carries the effective search parameters and no empty fields', () = assert.deepEqual(trace, { strategy: 'dense', scope: {}, + candidateK: 20, topK: 5, threshold: 0.5, durationMs: 12.5 }) // An unset threshold means "no threshold", not "threshold 0". assert.equal( - 'threshold' in buildRetrievalTrace({ strategy: 'dense', topK: 5, durationMs: 1 }), + 'threshold' in buildRetrievalTrace({ strategy: 'dense', candidateK: 20, topK: 5, durationMs: 1 }), false ) }) +test('the trace keeps the first-stage width and the final count apart', () => { + const trace = buildRetrievalTrace({ + strategy: 'hybrid', + candidateK: 20, + topK: 3, + durationMs: 9 + }) + + // #77: a snapshot that only says `topK` cannot explain the fused candidate pool. + assert.equal(trace.candidateK, 20) + assert.equal(trace.topK, 3) +}) + +/** + * #77: the first stage takes a candidate pool, the delivery stage takes `topK`. + * The invariant that matters is that the pool is never narrower than what is + * supposed to come out of it. + */ +test('the first-stage width is never narrower than the final count', () => { + assert.equal(effectiveCandidateK({}), DEFAULT_CANDIDATE_K) + assert.equal(effectiveCandidateK({ topK: 3 }), DEFAULT_CANDIDATE_K) + assert.equal(effectiveCandidateK({ candidateK: 40, topK: 3 }), 40) + // A caller that asks for more results than the default pool still gets them: the + // search palette and MCP pass their own `topK` and must not be silently capped. + assert.equal(effectiveCandidateK({ topK: 50 }), 50) +}) + test('a filter becomes the recorded scope of the trace', () => { const trace = buildRetrievalTrace({ strategy: 'dense', filter: { documentIds: ['doc_a', 'doc_b'] }, + candidateK: 20, topK: 8, durationMs: 3 }) From 54c892fb50cef4325e254fbe499ab228ad176239 Mon Sep 17 00:00:00 2001 From: MrSibe Date: Wed, 30 Sep 2026 16:49:34 +0800 Subject: [PATCH 03/18] feat(eval): hit rate, MAP, and per-query-type metrics MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Recall@K, MRR and nDCG@10 could not express three things the epic needs, and a single aggregate average was hiding a gap the corpus already contained. Child 3 of #192. **Two metrics added.** `hitRate@5` is whether *any* of the first five passages covers ground truth; `MAP@10` combines ranking position with coverage, so pulling a second relevant passage from rank 9 to rank 2 moves it while Recall@5 sits still. Hit rate is deliberately blunt next to Recall@5: a two-passage question that finds one scores 1.0 and 0.5 respectively, and both facts matter — "the model had a chance" is not "the material was complete". **Every metric is now reported per query type.** `questions.jsonl` gained an optional `type`, the harness groups by it, and the report renders a table. The aggregate was already concealing something: | Type | n | Recall@5 | nDCG@10 | Hit rate@5 | MAP@10 | | --- | --- | --- | --- | --- | --- | | cross-lingual | 1 | 1.0000 | **0.6309** | 1.0000 | **0.5000** | | exact | 6 | 1.0000 | 1.0000 | 1.0000 | 1.0000 | | multi-hop | 2 | 1.0000 | 0.9599 | 1.0000 | 0.9167 | | semantic | 18 | 1.0000 | 0.9312 | 1.0000 | 0.9074 | | zh | 3 | 1.0000 | 1.0000 | 1.0000 | 1.0000 | | **all** | 30 | 1.0000 | 0.9437 | 1.0000 | 0.9222 | The one cross-lingual question (`q030`, Chinese over an English source) ranks far worse than everything else. Recall@5 = 1.0000 reported that as a success; nDCG@10 and MAP@10 are what make the multilingual gap visible. That is the metric doing its job on the existing 30 questions, before the corpus grows. Untagged questions group under `untagged` rather than being dropped, and the types in use are documented in `eval/README.md`. ## Testing - `npm run typecheck` — clean - `npm test` — 485 pass, 4 new: hit rate vs recall, hit rate@k bounds, AP position sensitivity, AP's repeat-counts-once rule - `npm run eval` twice — byte-identical `docs/eval/baseline-v1.6.json` Part of #192 (child 3). --- docs/eval/baseline-v1.6.json | 104 +++++++++++++++++++++++++++++++++++ docs/eval/baseline-v1.6.md | 20 ++++++- eval/README.md | 14 +++++ eval/questions.jsonl | 60 ++++++++++---------- src/main/eval/harness.ts | 80 ++++++++++++++++++++------- src/main/eval/metrics.ts | 50 +++++++++++++++++ src/main/eval/report.ts | 23 ++++++++ src/main/eval/types.ts | 29 ++++++++++ test/evalMetrics.test.ts | 38 +++++++++++++ 9 files changed, 365 insertions(+), 53 deletions(-) diff --git a/docs/eval/baseline-v1.6.json b/docs/eval/baseline-v1.6.json index eb15f14..ca5f93b 100644 --- a/docs/eval/baseline-v1.6.json +++ b/docs/eval/baseline-v1.6.json @@ -26,12 +26,87 @@ "recallAt10": 1, "mrr": 0.927778, "ndcgAt10": 0.94375, + "hitRateAt5": 1, + "mapAt10": 0.922222, "evidencePrecisionAt5": 0.213333 }, + "byType": [ + { + "type": "cross-lingual", + "questions": 1, + "metrics": { + "recallAt1": 0, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 0.5, + "ndcgAt10": 0.63093, + "hitRateAt5": 1, + "mapAt10": 0.5, + "evidencePrecisionAt5": 0.2 + } + }, + { + "type": "exact", + "questions": 6, + "metrics": { + "recallAt1": 1, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 1, + "ndcgAt10": 1, + "hitRateAt5": 1, + "mapAt10": 1, + "evidencePrecisionAt5": 0.2 + } + }, + { + "type": "multi-hop", + "questions": 2, + "metrics": { + "recallAt1": 0.5, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 1, + "ndcgAt10": 0.95986, + "hitRateAt5": 1, + "mapAt10": 0.916667, + "evidencePrecisionAt5": 0.4 + } + }, + { + "type": "semantic", + "questions": 18, + "metrics": { + "recallAt1": 0.833333, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 0.907407, + "ndcgAt10": 0.931214, + "hitRateAt5": 1, + "mapAt10": 0.907407, + "evidencePrecisionAt5": 0.2 + } + }, + { + "type": "zh", + "questions": 3, + "metrics": { + "recallAt1": 1, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 1, + "ndcgAt10": 1, + "hitRateAt5": 1, + "mapAt10": 1, + "evidencePrecisionAt5": 0.2 + } + } + ], "perQuestion": [ { "id": "q001", "question": "Why is bedload harder to measure than suspended sediment?", + "type": "semantic", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -62,6 +137,7 @@ { "id": "q002", "question": "How many replicate samples are collected at each river station?", + "type": "exact", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -92,6 +168,7 @@ { "id": "q003", "question": "What is the central trade-off in lithium-ion cell design?", + "type": "semantic", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -122,6 +199,7 @@ { "id": "q004", "question": "Why do nickel-rich battery packs need more aggressive thermal management?", + "type": "semantic", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -152,6 +230,7 @@ { "id": "q005", "question": "What happens once the separator in a battery cell melts?", + "type": "semantic", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -182,6 +261,7 @@ { "id": "q006", "question": "At what temperature do honeybees begin to forage?", + "type": "exact", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -212,6 +292,7 @@ { "id": "q007", "question": "What does a late frost damage during full bloom?", + "type": "semantic", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -242,6 +323,7 @@ { "id": "q008", "question": "Why is a continuous tree canopy more effective at cooling than isolated trees?", + "type": "semantic", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -272,6 +354,7 @@ { "id": "q009", "question": "Why are trees with aggressive surface roots unsuitable for narrow verges?", + "type": "semantic", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -302,6 +385,7 @@ { "id": "q010", "question": "At what temperature is lactic acid fermentation fastest?", + "type": "exact", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -332,6 +416,7 @@ { "id": "q011", "question": "Is the salt percentage in fermentation based on vegetable weight or water weight?", + "type": "exact", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -362,6 +447,7 @@ { "id": "q012", "question": "Why must tidal turbines be sited in places with very fast currents?", + "type": "semantic", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -392,6 +478,7 @@ { "id": "q013", "question": "What is the main environmental concern for tidal energy installations?", + "type": "semantic", "firstRelevantRank": 3, "relevantCount": 1, "retrievedCount": 19, @@ -422,6 +509,7 @@ { "id": "q014", "question": "In lake monitoring, how is the sampling depth actually recorded?", + "type": "exact", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -452,6 +540,7 @@ { "id": "q015", "question": "Why does deep-water oxygen fall while a lake remains stratified?", + "type": "semantic", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -482,6 +571,7 @@ { "id": "q016", "question": "How do supercapacitors hold their charge?", + "type": "semantic", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -512,6 +602,7 @@ { "id": "q017", "question": "Why can solitary bees pollinate a bloom week that is too cold for honeybee hives?", + "type": "semantic", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -542,6 +633,7 @@ { "id": "q018", "question": "Why is one continuous planted roof layer better than several isolated planted beds?", + "type": "semantic", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -572,6 +664,7 @@ { "id": "q019", "question": "Where do acetic acid bacteria sit in a vinegar culture?", + "type": "semantic", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -602,6 +695,7 @@ { "id": "q020", "question": "What happens if a vinegar culture is sealed airtight?", + "type": "semantic", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -632,6 +726,7 @@ { "id": "q021", "question": "Why is wave energy harder to schedule ahead than tidal energy?", + "type": "semantic", "firstRelevantRank": 2, "relevantCount": 1, "retrievedCount": 19, @@ -662,6 +757,7 @@ { "id": "q022", "question": "Where does siting for wave energy devices concentrate, and where does it not?", + "type": "exact", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -692,6 +788,7 @@ { "id": "q023", "question": "How does the river sampling protocol differ from the lake sampling protocol?", + "type": "multi-hop", "firstRelevantRank": 1, "relevantCount": 2, "retrievedCount": 19, @@ -724,6 +821,7 @@ { "id": "q024", "question": "A street canopy and a green roof are both said to cool; what surface does each one shade?", + "type": "multi-hop", "firstRelevantRank": 1, "relevantCount": 2, "retrievedCount": 19, @@ -756,6 +854,7 @@ { "id": "q025", "question": "Which preservation method depends on keeping air away from the food?", + "type": "semantic", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -786,6 +885,7 @@ { "id": "q026", "question": "Why can one cold morning cost a grower the whole crop even when colonies are brought in?", + "type": "semantic", "firstRelevantRank": 2, "relevantCount": 1, "retrievedCount": 19, @@ -816,6 +916,7 @@ { "id": "q027", "question": "绿茶应该怎样保存才能减缓氧化?", + "type": "zh", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -846,6 +947,7 @@ { "id": "q028", "question": "茶叶储存的相对湿度上限是多少?", + "type": "zh", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -876,6 +978,7 @@ { "id": "q029", "question": "为什么冷冻保存的茶叶取出后不能立刻打开包装?", + "type": "zh", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -906,6 +1009,7 @@ { "id": "q030", "question": "为什么潮汐能比风能和太阳能更容易提前安排发电?", + "type": "cross-lingual", "firstRelevantRank": 2, "relevantCount": 1, "retrievedCount": 19, diff --git a/docs/eval/baseline-v1.6.md b/docs/eval/baseline-v1.6.md index 5418dd6..df1c22c 100644 --- a/docs/eval/baseline-v1.6.md +++ b/docs/eval/baseline-v1.6.md @@ -24,10 +24,26 @@ Generated by `npm run eval`. The numbers below are harness output — do not edi | Recall@10 | 1.0000 | | MRR | 0.9278 | | nDCG@10 | 0.9437 | +| Hit rate@5 | 1.0000 | +| MAP@10 | 0.9222 | | Evidence precision@5 | 0.2133 | -Timing is informational only and is **not** frozen: indexing 1582 ms, query -p50 11.35 ms, p95 18.82 ms on the +### By query type + +A single average hides a change that helps one kind of question and hurts another. +The type comes from `type` in `questions.jsonl`; untagged questions report as +`untagged` rather than disappearing. + +| Type | Questions | Recall@5 | nDCG@10 | Hit rate@5 | MAP@10 | +| --- | --- | --- | --- | --- | --- | +| cross-lingual | 1 | 1.0000 | 0.6309 | 1.0000 | 0.5000 | +| exact | 6 | 1.0000 | 1.0000 | 1.0000 | 1.0000 | +| multi-hop | 2 | 1.0000 | 0.9599 | 1.0000 | 0.9167 | +| semantic | 18 | 1.0000 | 0.9312 | 1.0000 | 0.9074 | +| zh | 3 | 1.0000 | 1.0000 | 1.0000 | 1.0000 | + +Timing is informational only and is **not** frozen: indexing 1545 ms, query +p50 11.04 ms, p95 14.14 ms on the machine that produced this file. Timing and index size depend on hardware and on the corpus, so they must never be the reason two runs differ. diff --git a/eval/README.md b/eval/README.md index 191ac61..c8964f1 100644 --- a/eval/README.md +++ b/eval/README.md @@ -59,6 +59,7 @@ dataset needs a new question. { "id": "q001", "question": "Why is bedload harder to measure than suspended sediment?", + "type": "semantic", "relevant": [ { "document": "river-monitoring.md", @@ -78,6 +79,10 @@ Ground truth uses **corpus identity, never database identity**: - `page` is `null` for unpaginated sources. - `quote` is an optional excerpt. The runner fails if the referenced block no longer contains it, so a parser change cannot silently move the ground truth. +- `type` is an optional query class; the report groups every metric by it. Values in use: + `exact` (number/name/detail), `semantic` (why/how), `multi-hop` (two or more blocks), + `cross-lingual` (question language differs from the source), `zh` (Chinese over a + Chinese source). Untagged questions report as `untagged`. Runtime `documentId`s are random and `blockId`s embed them, so neither may appear here. This is what lets #78 change chunking without invalidating the dataset: the ground truth @@ -115,11 +120,20 @@ from unanswerable queries. - **Recall@1/5/10** — share of ground-truth blocks covered by the first k passages. - **MRR** — reciprocal rank of the first relevant passage. - **nDCG@10** — binary-gain discounted cumulative gain. +- **Hit rate@5** — share of questions with at least one relevant passage in the first 5. + Deliberately blunt: it says the answer was *reachable*, where Recall@5 says the material + was *complete*. A two-passage question that finds one scores 1.0 and 0.5 respectively. +- **MAP@10** — mean average precision. The one metric here that combines ranking position + with coverage, so pulling a second relevant passage from rank 9 to rank 2 moves it. - **Evidence precision@5** — of the first 5 retrieved passages, the share that cover a ground-truth block. This is **retrieval precision, not answer citation recall**: the harness runs no model and produces no answer. Answer-level citation correctness is covered by the resolver (#70); a model-driven answer eval would be a separate deliverable. +- **By query type** — the same metrics per `type` in `questions.jsonl` (`exact`, + `semantic`, `multi-hop`, `cross-lingual`, `zh`). A single average hides a change that + helps one kind of question and hurts another; the current baseline already shows this, + with `cross-lingual` at nDCG 0.63 against 0.93–1.00 elsewhere. - **Latency p50/p95** — informational only. Timing is **not** frozen, and the committed JSON excludes it so two runs diff cleanly. diff --git a/eval/questions.jsonl b/eval/questions.jsonl index e125d9c..de15b0c 100644 --- a/eval/questions.jsonl +++ b/eval/questions.jsonl @@ -1,30 +1,30 @@ -{"id":"q001","question":"Why is bedload harder to measure than suspended sediment?","relevant":[{"document":"river-monitoring.md","page":null,"block":4,"quote":"Bedload is the harder fraction to measure"}]} -{"id":"q002","question":"How many replicate samples are collected at each river station?","relevant":[{"document":"river-monitoring.md","page":null,"block":6,"quote":"three replicate samples at each station"}]} -{"id":"q003","question":"What is the central trade-off in lithium-ion cell design?","relevant":[{"document":"battery-chemistry.md","page":null,"block":2,"quote":"trade energy density against thermal stability"}]} -{"id":"q004","question":"Why do nickel-rich battery packs need more aggressive thermal management?","relevant":[{"document":"battery-chemistry.md","page":null,"block":4,"quote":"release oxygen at lower temperatures than iron phosphate"}]} -{"id":"q005","question":"What happens once the separator in a battery cell melts?","relevant":[{"document":"battery-chemistry.md","page":null,"block":6,"quote":"Once the separator melts, the cell shorts internally"}]} -{"id":"q006","question":"At what temperature do honeybees begin to forage?","relevant":[{"document":"orchard-pollination.md","page":null,"block":4,"quote":"above roughly twelve degrees Celsius"}]} -{"id":"q007","question":"What does a late frost damage during full bloom?","relevant":[{"document":"orchard-pollination.md","page":null,"block":6,"quote":"destroys the flower's ovary rather than the petals"}]} -{"id":"q008","question":"Why is a continuous tree canopy more effective at cooling than isolated trees?","relevant":[{"document":"urban-canopy.md","page":null,"block":4,"quote":"asphalt stores far more heat during the day"}]} -{"id":"q009","question":"Why are trees with aggressive surface roots unsuitable for narrow verges?","relevant":[{"document":"urban-canopy.md","page":null,"block":6,"quote":"aggressive surface roots lift pavements"}]} -{"id":"q010","question":"At what temperature is lactic acid fermentation fastest?","relevant":[{"document":"fermentation.md","page":null,"block":4,"quote":"fastest between twenty and twenty-four degrees Celsius"}]} -{"id":"q011","question":"Is the salt percentage in fermentation based on vegetable weight or water weight?","relevant":[{"document":"fermentation.md","page":null,"block":6,"quote":"percentage of the vegetable weight, not the water weight"}]} -{"id":"q012","question":"Why must tidal turbines be sited in places with very fast currents?","relevant":[{"document":"tidal-energy.md","page":null,"block":4,"quote":"power scales with the cube of velocity"}]} -{"id":"q013","question":"What is the main environmental concern for tidal energy installations?","relevant":[{"document":"tidal-energy.md","page":null,"block":6,"quote":"change in sediment transport"}]} -{"id":"q014","question":"In lake monitoring, how is the sampling depth actually recorded?","relevant":[{"document":"lake-monitoring.md","page":null,"block":6,"quote":"Depth is recorded from the profiling sonde"}]} -{"id":"q015","question":"Why does deep-water oxygen fall while a lake remains stratified?","relevant":[{"document":"lake-monitoring.md","page":null,"block":4,"quote":"Oxygen in the hypolimnion is not replenished"}]} -{"id":"q016","question":"How do supercapacitors hold their charge?","relevant":[{"document":"supercapacitors.md","page":null,"block":4,"quote":"electric double layer formed at the surface of a porous carbon electrode"}]} -{"id":"q017","question":"Why can solitary bees pollinate a bloom week that is too cold for honeybee hives?","relevant":[{"document":"wild-pollinators.md","page":null,"block":4,"quote":"forage at lower temperatures than honeybees"}]} -{"id":"q018","question":"Why is one continuous planted roof layer better than several isolated planted beds?","relevant":[{"document":"green-roofs.md","page":null,"block":4,"quote":"an exposed membrane absorbs far more heat during the day"}]} -{"id":"q019","question":"Where do acetic acid bacteria sit in a vinegar culture?","relevant":[{"document":"vinegar-production.md","page":null,"block":4,"quote":"surface of the liquid where oxygen is available"}]} -{"id":"q020","question":"What happens if a vinegar culture is sealed airtight?","relevant":[{"document":"vinegar-production.md","page":null,"block":6,"quote":"Sealing a vinegar culture airtight stops the conversion entirely"}]} -{"id":"q021","question":"Why is wave energy harder to schedule ahead than tidal energy?","relevant":[{"document":"wave-energy.md","page":null,"block":2,"quote":"driven by wind rather than by the moon"}]} -{"id":"q022","question":"Where does siting for wave energy devices concentrate, and where does it not?","relevant":[{"document":"wave-energy.md","page":null,"block":4,"quote":"exposed headlands rather than on sheltered channels"}]} -{"id":"q023","question":"How does the river sampling protocol differ from the lake sampling protocol?","relevant":[{"document":"river-monitoring.md","page":null,"block":6,"quote":"three replicate samples at each station"},{"document":"lake-monitoring.md","page":null,"block":6,"quote":"Depth is recorded from the profiling sonde"}]} -{"id":"q024","question":"A street canopy and a green roof are both said to cool; what surface does each one shade?","relevant":[{"document":"urban-canopy.md","page":null,"block":4,"quote":"asphalt stores far more heat during the day"},{"document":"green-roofs.md","page":null,"block":4,"quote":"an exposed membrane absorbs far more heat during the day"}]} -{"id":"q025","question":"Which preservation method depends on keeping air away from the food?","relevant":[{"document":"fermentation.md","page":null,"block":2,"quote":"an anaerobic environment"}]} -{"id":"q026","question":"Why can one cold morning cost a grower the whole crop even when colonies are brought in?","relevant":[{"document":"orchard-pollination.md","page":null,"block":4,"quote":"above roughly twelve degrees Celsius"}]} -{"id":"q027","question":"绿茶应该怎样保存才能减缓氧化?","relevant":[{"document":"chinese-tea-storage.md","page":null,"block":2,"quote":"绿茶最容易氧化变质,需要低温密封保存"}]} -{"id":"q028","question":"茶叶储存的相对湿度上限是多少?","relevant":[{"document":"chinese-tea-storage.md","page":null,"block":4,"quote":"相对湿度应保持在百分之五十以下"}]} -{"id":"q029","question":"为什么冷冻保存的茶叶取出后不能立刻打开包装?","relevant":[{"document":"chinese-tea-storage.md","page":null,"block":6,"quote":"冷凝水会直接落在茶叶上"}]} -{"id":"q030","question":"为什么潮汐能比风能和太阳能更容易提前安排发电?","relevant":[{"document":"tidal-energy.md","page":null,"block":2,"quote":"predictable decades ahead, unlike wind or solar"}]} +{"id":"q001","question":"Why is bedload harder to measure than suspended sediment?","relevant":[{"document":"river-monitoring.md","page":null,"block":4,"quote":"Bedload is the harder fraction to measure"}],"type":"semantic"} +{"id":"q002","question":"How many replicate samples are collected at each river station?","relevant":[{"document":"river-monitoring.md","page":null,"block":6,"quote":"three replicate samples at each station"}],"type":"exact"} +{"id":"q003","question":"What is the central trade-off in lithium-ion cell design?","relevant":[{"document":"battery-chemistry.md","page":null,"block":2,"quote":"trade energy density against thermal stability"}],"type":"semantic"} +{"id":"q004","question":"Why do nickel-rich battery packs need more aggressive thermal management?","relevant":[{"document":"battery-chemistry.md","page":null,"block":4,"quote":"release oxygen at lower temperatures than iron phosphate"}],"type":"semantic"} +{"id":"q005","question":"What happens once the separator in a battery cell melts?","relevant":[{"document":"battery-chemistry.md","page":null,"block":6,"quote":"Once the separator melts, the cell shorts internally"}],"type":"semantic"} +{"id":"q006","question":"At what temperature do honeybees begin to forage?","relevant":[{"document":"orchard-pollination.md","page":null,"block":4,"quote":"above roughly twelve degrees Celsius"}],"type":"exact"} +{"id":"q007","question":"What does a late frost damage during full bloom?","relevant":[{"document":"orchard-pollination.md","page":null,"block":6,"quote":"destroys the flower's ovary rather than the petals"}],"type":"semantic"} +{"id":"q008","question":"Why is a continuous tree canopy more effective at cooling than isolated trees?","relevant":[{"document":"urban-canopy.md","page":null,"block":4,"quote":"asphalt stores far more heat during the day"}],"type":"semantic"} +{"id":"q009","question":"Why are trees with aggressive surface roots unsuitable for narrow verges?","relevant":[{"document":"urban-canopy.md","page":null,"block":6,"quote":"aggressive surface roots lift pavements"}],"type":"semantic"} +{"id":"q010","question":"At what temperature is lactic acid fermentation fastest?","relevant":[{"document":"fermentation.md","page":null,"block":4,"quote":"fastest between twenty and twenty-four degrees Celsius"}],"type":"exact"} +{"id":"q011","question":"Is the salt percentage in fermentation based on vegetable weight or water weight?","relevant":[{"document":"fermentation.md","page":null,"block":6,"quote":"percentage of the vegetable weight, not the water weight"}],"type":"exact"} +{"id":"q012","question":"Why must tidal turbines be sited in places with very fast currents?","relevant":[{"document":"tidal-energy.md","page":null,"block":4,"quote":"power scales with the cube of velocity"}],"type":"semantic"} +{"id":"q013","question":"What is the main environmental concern for tidal energy installations?","relevant":[{"document":"tidal-energy.md","page":null,"block":6,"quote":"change in sediment transport"}],"type":"semantic"} +{"id":"q014","question":"In lake monitoring, how is the sampling depth actually recorded?","relevant":[{"document":"lake-monitoring.md","page":null,"block":6,"quote":"Depth is recorded from the profiling sonde"}],"type":"exact"} +{"id":"q015","question":"Why does deep-water oxygen fall while a lake remains stratified?","relevant":[{"document":"lake-monitoring.md","page":null,"block":4,"quote":"Oxygen in the hypolimnion is not replenished"}],"type":"semantic"} +{"id":"q016","question":"How do supercapacitors hold their charge?","relevant":[{"document":"supercapacitors.md","page":null,"block":4,"quote":"electric double layer formed at the surface of a porous carbon electrode"}],"type":"semantic"} +{"id":"q017","question":"Why can solitary bees pollinate a bloom week that is too cold for honeybee hives?","relevant":[{"document":"wild-pollinators.md","page":null,"block":4,"quote":"forage at lower temperatures than honeybees"}],"type":"semantic"} +{"id":"q018","question":"Why is one continuous planted roof layer better than several isolated planted beds?","relevant":[{"document":"green-roofs.md","page":null,"block":4,"quote":"an exposed membrane absorbs far more heat during the day"}],"type":"semantic"} +{"id":"q019","question":"Where do acetic acid bacteria sit in a vinegar culture?","relevant":[{"document":"vinegar-production.md","page":null,"block":4,"quote":"surface of the liquid where oxygen is available"}],"type":"semantic"} +{"id":"q020","question":"What happens if a vinegar culture is sealed airtight?","relevant":[{"document":"vinegar-production.md","page":null,"block":6,"quote":"Sealing a vinegar culture airtight stops the conversion entirely"}],"type":"semantic"} +{"id":"q021","question":"Why is wave energy harder to schedule ahead than tidal energy?","relevant":[{"document":"wave-energy.md","page":null,"block":2,"quote":"driven by wind rather than by the moon"}],"type":"semantic"} +{"id":"q022","question":"Where does siting for wave energy devices concentrate, and where does it not?","relevant":[{"document":"wave-energy.md","page":null,"block":4,"quote":"exposed headlands rather than on sheltered channels"}],"type":"exact"} +{"id":"q023","question":"How does the river sampling protocol differ from the lake sampling protocol?","relevant":[{"document":"river-monitoring.md","page":null,"block":6,"quote":"three replicate samples at each station"},{"document":"lake-monitoring.md","page":null,"block":6,"quote":"Depth is recorded from the profiling sonde"}],"type":"multi-hop"} +{"id":"q024","question":"A street canopy and a green roof are both said to cool; what surface does each one shade?","relevant":[{"document":"urban-canopy.md","page":null,"block":4,"quote":"asphalt stores far more heat during the day"},{"document":"green-roofs.md","page":null,"block":4,"quote":"an exposed membrane absorbs far more heat during the day"}],"type":"multi-hop"} +{"id":"q025","question":"Which preservation method depends on keeping air away from the food?","relevant":[{"document":"fermentation.md","page":null,"block":2,"quote":"an anaerobic environment"}],"type":"semantic"} +{"id":"q026","question":"Why can one cold morning cost a grower the whole crop even when colonies are brought in?","relevant":[{"document":"orchard-pollination.md","page":null,"block":4,"quote":"above roughly twelve degrees Celsius"}],"type":"semantic"} +{"id":"q027","question":"绿茶应该怎样保存才能减缓氧化?","relevant":[{"document":"chinese-tea-storage.md","page":null,"block":2,"quote":"绿茶最容易氧化变质,需要低温密封保存"}],"type":"zh"} +{"id":"q028","question":"茶叶储存的相对湿度上限是多少?","relevant":[{"document":"chinese-tea-storage.md","page":null,"block":4,"quote":"相对湿度应保持在百分之五十以下"}],"type":"zh"} +{"id":"q029","question":"为什么冷冻保存的茶叶取出后不能立刻打开包装?","relevant":[{"document":"chinese-tea-storage.md","page":null,"block":6,"quote":"冷凝水会直接落在茶叶上"}],"type":"zh"} +{"id":"q030","question":"为什么潮汐能比风能和太阳能更容易提前安排发电?","relevant":[{"document":"tidal-energy.md","page":null,"block":2,"quote":"predictable decades ahead, unlike wind or solar"}],"type":"cross-lingual"} diff --git a/src/main/eval/harness.ts b/src/main/eval/harness.ts index b23775c..02e5bb4 100644 --- a/src/main/eval/harness.ts +++ b/src/main/eval/harness.ts @@ -21,8 +21,10 @@ import { DEFAULT_CHUNK_OPTIONS, type ChunkOptions } from '../services/ChunkingSe import type { RetrievalStrategy } from '../services/retrieval' import { LOCAL_EMBEDDING_MODEL } from '../embedding/localModel' import { + averagePrecisionAtK, evidencePrecisionAtK, firstRelevantRank, + hitRateAtK, mean, ndcgAtK, percentile, @@ -31,9 +33,11 @@ import { } from './metrics' import type { EvalDeterministicReport, + EvalMetrics, EvalQuestion, EvalReport, EvalRelevantLocation, + EvalTypeBreakdown, QuestionReport, ResolvedGroundTruth } from './types' @@ -69,6 +73,30 @@ export interface EvalHarnessOptions { const NOTEBOOK_ID = 'eval-notebook' +/** A question with no `type` is grouped here rather than dropped from the report. */ +const UNTAGGED = 'untagged' + +/** + * 同一套指标既算总平均,也算每个查询类别(#192)。用一个函数是因为分组平均必须与 + * 总平均是同一个定义,否则两个数就不可比。 + */ +function summarize(perQuestion: readonly QuestionReport[], evidenceK: number): EvalMetrics { + return { + recallAt1: mean(perQuestion.map((q) => recallAtK(q.matchesByRank, q.relevantCount, 1))), + recallAt5: mean(perQuestion.map((q) => recallAtK(q.matchesByRank, q.relevantCount, 5))), + recallAt10: mean(perQuestion.map((q) => recallAtK(q.matchesByRank, q.relevantCount, 10))), + mrr: mean(perQuestion.map((q) => reciprocalRank(q.matchesByRank))), + ndcgAt10: mean(perQuestion.map((q) => ndcgAtK(q.matchesByRank, q.relevantCount, 10))), + hitRateAt5: mean(perQuestion.map((q) => hitRateAtK(q.matchesByRank, 5))), + mapAt10: mean( + perQuestion.map((q) => averagePrecisionAtK(q.matchesByRank, q.relevantCount, 10)) + ), + evidencePrecisionAt5: mean( + perQuestion.map((q) => evidencePrecisionAtK(q.matchesByRank, evidenceK)) + ) + } +} + /** Normalised comparison for the optional quote drift check. */ const normalize = (text: string): string => text.toLowerCase().replace(/\s+/g, ' ').trim() @@ -222,6 +250,7 @@ export async function runEvalHarness( perQuestion.push({ id: question.id, question: question.question, + type: question.type ?? UNTAGGED, firstRelevantRank: firstRelevantRank(matchesByRank), relevantCount: groundTruth.length, retrievedCount: results.length, @@ -229,16 +258,15 @@ export async function runEvalHarness( }) } - const metrics = { - recallAt1: mean(perQuestion.map((q) => recallAtK(q.matchesByRank, q.relevantCount, 1))), - recallAt5: mean(perQuestion.map((q) => recallAtK(q.matchesByRank, q.relevantCount, 5))), - recallAt10: mean(perQuestion.map((q) => recallAtK(q.matchesByRank, q.relevantCount, 10))), - mrr: mean(perQuestion.map((q) => reciprocalRank(q.matchesByRank))), - ndcgAt10: mean(perQuestion.map((q) => ndcgAtK(q.matchesByRank, q.relevantCount, 10))), - evidencePrecisionAt5: mean( - perQuestion.map((q) => evidencePrecisionAtK(q.matchesByRank, options.evidenceK)) - ) - } + const metrics = summarize(perQuestion, options.evidenceK) + + // 每个类别一行,按类别名排序,所以同一个 JSON 在两次运行之间可 diff。 + const byType: EvalTypeBreakdown[] = [...new Set(perQuestion.map((q) => q.type))] + .sort() + .map((type) => { + const group = perQuestion.filter((q) => q.type === type) + return { type, questions: group.length, metrics: summarize(group, options.evidenceK) } + }) const chunking = { ...DEFAULT_CHUNK_OPTIONS, ...options.chunkOptions } return { @@ -264,6 +292,7 @@ export async function runEvalHarness( chunkCount }, metrics, + byType, timing: { latencyP50Ms: percentile(latencies, 50), latencyP95Ms: percentile(latencies, 95), @@ -274,21 +303,29 @@ export async function runEvalHarness( } /** Round metrics to a stable number of decimals so the JSON diffs cleanly. */ +const roundMetric = (value: number): number => Number(value.toFixed(6)) + +function roundMetrics(metrics: EvalMetrics): EvalMetrics { + return { + recallAt1: roundMetric(metrics.recallAt1), + recallAt5: roundMetric(metrics.recallAt5), + recallAt10: roundMetric(metrics.recallAt10), + mrr: roundMetric(metrics.mrr), + ndcgAt10: roundMetric(metrics.ndcgAt10), + hitRateAt5: roundMetric(metrics.hitRateAt5), + mapAt10: roundMetric(metrics.mapAt10), + evidencePrecisionAt5: roundMetric(metrics.evidencePrecisionAt5) + } +} + export function stabilize(report: EvalReport): EvalReport { - const round = (value: number): number => Number(value.toFixed(6)) return { ...report, - metrics: { - recallAt1: round(report.metrics.recallAt1), - recallAt5: round(report.metrics.recallAt5), - recallAt10: round(report.metrics.recallAt10), - mrr: round(report.metrics.mrr), - ndcgAt10: round(report.metrics.ndcgAt10), - evidencePrecisionAt5: round(report.metrics.evidencePrecisionAt5) - }, + metrics: roundMetrics(report.metrics), + byType: report.byType.map((entry) => ({ ...entry, metrics: roundMetrics(entry.metrics) })), timing: { - latencyP50Ms: round(report.timing.latencyP50Ms), - latencyP95Ms: round(report.timing.latencyP95Ms), + latencyP50Ms: roundMetric(report.timing.latencyP50Ms), + latencyP95Ms: roundMetric(report.timing.latencyP95Ms), // Throughput is informational and excluded from the deterministic report; the // full report keeps it for the #78 comparison. indexingMs: Math.round(report.timing.indexingMs) @@ -304,6 +341,7 @@ export function toDeterministicReport(report: EvalReport): EvalDeterministicRepo generatedBy: report.generatedBy, config: report.config, metrics: report.metrics, + byType: report.byType, perQuestion: report.perQuestion } } diff --git a/src/main/eval/metrics.ts b/src/main/eval/metrics.ts index 41f0d23..2dfda96 100644 --- a/src/main/eval/metrics.ts +++ b/src/main/eval/metrics.ts @@ -32,6 +32,56 @@ export function reciprocalRank(matchesByRank: MatchMatrix): number { return rank === 0 ? 0 : 1 / rank } +/** + * Hit rate@k: 1 when any of the first `k` ranks covers ground truth, else 0. + * + * Deliberately the bluntest metric here. Recall@k says *how much* of the ground + * truth was found; hit rate says only whether the answer was findable at all. When + * a question needs two passages, a run that finds one scores 0.5 recall and 1.0 hit + * rate — and the second number is the one that says "the model had a chance". + */ +export function hitRateAtK(matchesByRank: MatchMatrix, k: number): number { + return matchesByRank.slice(0, k).some((matches) => matches.length > 0) ? 1 : 0 +} + +/** + * Average precision@k, averaged over the ground-truth locations. + * + * Precision is measured at each rank that recovers a *fresh* ground-truth location + * (the same rule nDCG uses), then normalised by the total number of ground-truth + * locations. That makes AP the one metric here that combines ranking position with + * coverage: pulling a second relevant passage from rank 9 to rank 2 moves it, while + * recall@5 sits still. + */ +export function averagePrecisionAtK( + matchesByRank: MatchMatrix, + groundTruthCount: number, + k: number +): number { + if (groundTruthCount === 0) return 0 + + const covered = new Set() + let found = 0 + let sum = 0 + const limit = Math.min(matchesByRank.length, k) + + for (let i = 0; i < limit; i++) { + let fresh = false + for (const match of matchesByRank[i]) { + if (match < groundTruthCount && !covered.has(match)) { + covered.add(match) + fresh = true + } + } + if (fresh) { + found += 1 + sum += found / (i + 1) + } + } + + return sum / groundTruthCount +} + /** * nDCG@k with binary gains. The ideal ranking puts every ground-truth location * first, so the discount is a plain log base 2. diff --git a/src/main/eval/report.ts b/src/main/eval/report.ts index b596704..34d9688 100644 --- a/src/main/eval/report.ts +++ b/src/main/eval/report.ts @@ -13,6 +13,17 @@ export function renderMarkdown(report: EvalReport): string { const { config, metrics, timing } = report const chunking = config.chunking + // Built before the template because a nested template literal would terminate the + // outer one; the map is a statement here, not an interpolation. + const typeRows = report.byType + .map( + (entry) => + `| ${entry.type} | ${entry.questions} | ${format(entry.metrics.recallAt5)} | ` + + `${format(entry.metrics.ndcgAt10)} | ${format(entry.metrics.hitRateAt5)} | ` + + `${format(entry.metrics.mapAt10)} |` + ) + .join('\n') + return `# RAG eval baseline — ${report.baseline} Generated by \`${report.generatedBy}\`. The numbers below are harness output — do not edit them by hand. @@ -39,8 +50,20 @@ Generated by \`${report.generatedBy}\`. The numbers below are harness output — | Recall@10 | ${format(metrics.recallAt10)} | | MRR | ${format(metrics.mrr)} | | nDCG@10 | ${format(metrics.ndcgAt10)} | +| Hit rate@5 | ${format(metrics.hitRateAt5)} | +| MAP@10 | ${format(metrics.mapAt10)} | | Evidence precision@${config.evidenceK} | ${format(metrics.evidencePrecisionAt5)} | +### By query type + +A single average hides a change that helps one kind of question and hurts another. +The type comes from \`type\` in \`questions.jsonl\`; untagged questions report as +\`untagged\` rather than disappearing. + +| Type | Questions | Recall@5 | nDCG@10 | Hit rate@5 | MAP@10 | +| --- | --- | --- | --- | --- | --- | +${typeRows} + Timing is informational only and is **not** frozen: indexing ${timing.indexingMs} ms, query p50 ${timing.latencyP50Ms.toFixed(2)} ms, p95 ${timing.latencyP95Ms.toFixed(2)} ms on the machine that produced this file. Timing and index size depend on hardware and on the diff --git a/src/main/eval/types.ts b/src/main/eval/types.ts index c77fb90..dbb692b 100644 --- a/src/main/eval/types.ts +++ b/src/main/eval/types.ts @@ -24,6 +24,14 @@ export interface EvalQuestion { question: string relevant: EvalRelevantLocation[] goldAnswer?: string + /** + * 查询类别(#192)。自由字符串,因为语料还会长出新类别;报告按出现过的值分组, + * 缺省归入 `untagged`。 + * + * 分类的意义是:一个提升实体查询、却弄坏释义查询的策略,不该在总平均上显示成 + * 「没变化」。 + */ + type?: string } /** One resolved ground-truth location, after runtime id mapping. */ @@ -41,6 +49,15 @@ export interface EvalMetrics { recallAt10: number mrr: number ndcgAt10: number + /** + * 前 5 名至少命中一个 ground-truth 块的问题占比。 + * + * 与 Recall@5 并列而不是替代:多块问题只要命中一块,hit rate 就是 1, + * 而 Recall@5 只有 0.5。「模型有没有机会」和「材料齐不齐」是两件事。 + */ + hitRateAt5: number + /** AP@10:把「排序位置」和「覆盖面」合成一个数的那个指标。 */ + mapAt10: number /** * Share of the first `evidenceK` retrieved passages that cover ground truth. * @@ -51,9 +68,18 @@ export interface EvalMetrics { evidencePrecisionAt5: number } +/** 按查询类别聚合的同一套指标(#192)。总平均会掩盖方向相反的两个变化。 */ +export interface EvalTypeBreakdown { + type: string + questions: number + metrics: EvalMetrics +} + export interface QuestionReport { id: string question: string + /** 查询类别,与 `EvalQuestion.type` 一致;缺省为 `untagged`。 */ + type: string firstRelevantRank: number relevantCount: number retrievedCount: number @@ -97,6 +123,8 @@ export interface EvalReport { chunkCount: number } metrics: EvalMetrics + /** 每个查询类别一行;类别来自 `questions.jsonl` 的 `type`。 */ + byType: EvalTypeBreakdown[] /** `indexingMs` 只用于 #78 的吞吐比较;它不在确定报告里,也不该成为差异原因。 */ timing: { latencyP50Ms: number; latencyP95Ms: number; indexingMs: number } perQuestion: QuestionReport[] @@ -112,5 +140,6 @@ export interface EvalDeterministicReport { generatedBy: string config: EvalReport['config'] metrics: EvalMetrics + byType: EvalTypeBreakdown[] perQuestion: QuestionReport[] } diff --git a/test/evalMetrics.test.ts b/test/evalMetrics.test.ts index 1520dc6..3cd8d0e 100644 --- a/test/evalMetrics.test.ts +++ b/test/evalMetrics.test.ts @@ -1,8 +1,10 @@ import { test } from 'node:test' import assert from 'node:assert/strict' import { + averagePrecisionAtK, evidencePrecisionAtK, firstRelevantRank, + hitRateAtK, mean, type MatchMatrix, ndcgAtK, @@ -101,6 +103,42 @@ test('evidence precision counts grounded passages over retrieved passages', () = assert.equal(evidencePrecisionAtK([[0], [0], [0]], 3), 1) }) +test('hit rate@k says whether the answer was reachable at all', () => { + assert.equal(hitRateAtK([[0], [], []], 5), 1) + assert.equal(hitRateAtK([[], [], [1]], 1), 0) + assert.equal(hitRateAtK([[], [], [1]], 3), 1) + assert.equal(hitRateAtK([[], []], 5), 0) + assert.equal(hitRateAtK([], 5), 0) +}) + +/** + * The distinction the metric exists for: a two-passage question that finds only + * one has 0.5 recall but full hit rate — the model had a chance, and recall is what + * says the material was incomplete. + */ +test('hit rate can be 1 while recall@k is only half', () => { + const matches: MatchMatrix = [[0], []] + assert.equal(hitRateAtK(matches, 5), 1) + assert.equal(recallAtK(matches, 2, 5), 0.5) +}) + +test('average precision rewards finding the same ground truth earlier', () => { + // One relevant at rank 1: AP = 1. + assert.equal(averagePrecisionAtK([[0]], 1, 10), 1) + // One relevant at rank 2: P@1 = 0, P@2 = 1/2, so AP = 1/2. + assert.equal(averagePrecisionAtK([[], [0]], 1, 10), 0.5) + // Two relevant at ranks 1 and 2: (1 + 2/2) / 2 = 1. + assert.equal(averagePrecisionAtK([[0], [1]], 2, 10), 1) + // Two relevant, but the second is only found at rank 4: (1 + 2/4) / 2 = 0.75. + assert.equal(averagePrecisionAtK([[0], [], [], [1]], 2, 10), 0.75) +}) + +test('average precision counts a repeated match once and never exceeds 1', () => { + assert.equal(averagePrecisionAtK([[0], [0], [0]], 1, 10), 1) + assert.equal(averagePrecisionAtK([[], []], 0, 10), 0) + assert.equal(averagePrecisionAtK([], 3, 10), 0) +}) + test('mean and percentile handle the empty and single cases', () => { assert.equal(mean([]), 0) assert.equal(mean([1, 2, 3]), 2) From a2ccf057d6c3208963fcae9bba50bd7d7c6f5d4f Mon Sep 17 00:00:00 2001 From: MrSibe Date: Wed, 30 Sep 2026 16:54:27 +0800 Subject: [PATCH 04/18] feat(eval): derive the similarity threshold on validation, report on test MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `threshold: 0.5` was hand-picked, and a cosine score has no universal meaning — its distribution depends on the embedding model, the language, the query type and the chunk length. Child 4 of #192. **A deterministic split, owned by the harness.** `--eval-split=validation|test` partitions the questions by a hash of the id, so the same `questions.jsonl` cuts the same way on every machine and both arms go through the same code path that produces the frozen baseline. A parameter chosen on the questions it is scored on is fitted, not measured. **`npm run eval:threshold`** sweeps `0 / 0.3 / 0.4 / 0.5 / 0.6`, selects on `validation`, and reports the winner on `test`. It also counts a metric the baseline does not carry: the **no-result rate**. A higher threshold can look better on a ranking metric while quietly making the product answer "not in your sources" more often, and that trade is invisible unless it is counted. ## The result here is a non-result, and it is reported as one | Threshold | Recall@5 (val) | nDCG@10 (val) | No-result (val) | nDCG@10 (test) | No-result (test) | | --- | --- | --- | --- | --- | --- | | 0 | 1.0000 | 0.9500 | 0.0000 | 0.9406 | 0.0000 | | 0.3 | 1.0000 | 0.9500 | 0.0000 | 0.9406 | 0.0000 | | 0.4 | 1.0000 | 0.9500 | 0.0000 | 0.9406 | 0.0000 | | 0.5 | 1.0000 | 0.9500 | 0.0000 | 0.9406 | 0.0000 | | 0.6 | 1.0000 | 0.9500 | 0.0000 | 0.9406 | 0.0000 | The sweep is **flat**: every threshold produces identical metrics and never filters a passage. E5 does not score these query/chunk pairs below 0.6, so on this corpus the threshold is non-binding. The report says **"no evidence to change `threshold = 0.5`"** rather than nominating the tie-break winner, because moving a product parameter on a flat sweep would be noise dressed as a result. The tie-break rule (widest threshold among equals) is stated so it can be argued with. That is the honest outcome, and it is another instance of the corpus limitation #192 describes. The machinery is what ships here; the number is what the corpus cannot yet support. ## Testing - `npm run typecheck` — clean - `npm test` — 489 pass, 4 new: split bounds, split determinism, validation/test partition without overlap, `all` preserves order - `npm run eval` — `docs/eval/baseline-v1.6.json` regenerated with `config.split` - `npm run eval:threshold` — reruns cleanly and rewrites the same tables Part of #192 (child 4). --- .prettierignore | 5 + docs/eval/baseline-v1.6.json | 1 + docs/eval/baseline-v1.6.md | 7 +- docs/eval/threshold-v1.6.json | 183 ++++++++++++++++++++++++++ docs/eval/threshold-v1.6.md | 51 ++++++++ eval/README.md | 14 +- package.json | 3 +- scripts/eval-threshold.mjs | 233 ++++++++++++++++++++++++++++++++++ src/main/eval/harness.ts | 11 +- src/main/eval/report.ts | 3 +- src/main/eval/run.ts | 13 ++ src/main/eval/types.ts | 25 ++++ test/evalSplit.test.ts | 61 +++++++++ 13 files changed, 602 insertions(+), 8 deletions(-) create mode 100644 docs/eval/threshold-v1.6.json create mode 100644 docs/eval/threshold-v1.6.md create mode 100644 scripts/eval-threshold.mjs create mode 100644 test/evalSplit.test.ts diff --git a/.prettierignore b/.prettierignore index 1a115ff..ae11386 100644 --- a/.prettierignore +++ b/.prettierignore @@ -19,6 +19,11 @@ docs/eval/baseline-*.md docs/eval/chunking-*.json docs/eval/chunking-*.md +# Same reason again: generated by `scripts/eval-threshold.mjs`. Regenerate with +# `npm run eval:threshold`. +docs/eval/threshold-*.json +docs/eval/threshold-*.md + # And the same again for `scripts/eval-retrieval.mjs` (#77). docs/eval/retrieval-*.json docs/eval/retrieval-*.md diff --git a/docs/eval/baseline-v1.6.json b/docs/eval/baseline-v1.6.json index ca5f93b..cfb0455 100644 --- a/docs/eval/baseline-v1.6.json +++ b/docs/eval/baseline-v1.6.json @@ -11,6 +11,7 @@ "respectHeadings": false }, "retrieval": "dense", + "split": "all", "candidateK": 20, "contextK": 3, "threshold": 0.5, diff --git a/docs/eval/baseline-v1.6.md b/docs/eval/baseline-v1.6.md index df1c22c..605c29a 100644 --- a/docs/eval/baseline-v1.6.md +++ b/docs/eval/baseline-v1.6.md @@ -12,7 +12,8 @@ Generated by `npm run eval`. The numbers below are harness output — do not edi | Ranks | `candidateK=20, threshold=0.5` | | Context width | `contextK=3` | | Evidence per query | `evidenceK=5` | -| Corpus | `eval/corpus` (13 documents, 30 questions) | +| Corpus | `eval/corpus` (13 documents) | +| Split | `all` (30 questions) | | Index size | 19 chunks | ## Metrics @@ -42,8 +43,8 @@ The type comes from `type` in `questions.jsonl`; untagged questions report as | semantic | 18 | 1.0000 | 0.9312 | 1.0000 | 0.9074 | | zh | 3 | 1.0000 | 1.0000 | 1.0000 | 1.0000 | -Timing is informational only and is **not** frozen: indexing 1545 ms, query -p50 11.04 ms, p95 14.14 ms on the +Timing is informational only and is **not** frozen: indexing 1542 ms, query +p50 15.01 ms, p95 69.16 ms on the machine that produced this file. Timing and index size depend on hardware and on the corpus, so they must never be the reason two runs differ. diff --git a/docs/eval/threshold-v1.6.json b/docs/eval/threshold-v1.6.json new file mode 100644 index 0000000..636e5b9 --- /dev/null +++ b/docs/eval/threshold-v1.6.json @@ -0,0 +1,183 @@ +{ + "baseline": "v1.6", + "productionThreshold": 0.5, + "flat": true, + "recommended": 0.5, + "rows": [ + { + "threshold": 0, + "validation": { + "questions": 10, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 19, + "metrics": { + "recallAt1": 0.9, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 0.933333, + "ndcgAt10": 0.95, + "hitRateAt5": 1, + "mapAt10": 0.933333, + "evidencePrecisionAt5": 0.2 + } + }, + "test": { + "questions": 20, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 19, + "metrics": { + "recallAt1": 0.8, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 0.925, + "ndcgAt10": 0.940626, + "hitRateAt5": 1, + "mapAt10": 0.916667, + "evidencePrecisionAt5": 0.22 + } + } + }, + { + "threshold": 0.3, + "validation": { + "questions": 10, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 19, + "metrics": { + "recallAt1": 0.9, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 0.933333, + "ndcgAt10": 0.95, + "hitRateAt5": 1, + "mapAt10": 0.933333, + "evidencePrecisionAt5": 0.2 + } + }, + "test": { + "questions": 20, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 19, + "metrics": { + "recallAt1": 0.8, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 0.925, + "ndcgAt10": 0.940626, + "hitRateAt5": 1, + "mapAt10": 0.916667, + "evidencePrecisionAt5": 0.22 + } + } + }, + { + "threshold": 0.4, + "validation": { + "questions": 10, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 19, + "metrics": { + "recallAt1": 0.9, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 0.933333, + "ndcgAt10": 0.95, + "hitRateAt5": 1, + "mapAt10": 0.933333, + "evidencePrecisionAt5": 0.2 + } + }, + "test": { + "questions": 20, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 19, + "metrics": { + "recallAt1": 0.8, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 0.925, + "ndcgAt10": 0.940626, + "hitRateAt5": 1, + "mapAt10": 0.916667, + "evidencePrecisionAt5": 0.22 + } + } + }, + { + "threshold": 0.5, + "validation": { + "questions": 10, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 19, + "metrics": { + "recallAt1": 0.9, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 0.933333, + "ndcgAt10": 0.95, + "hitRateAt5": 1, + "mapAt10": 0.933333, + "evidencePrecisionAt5": 0.2 + } + }, + "test": { + "questions": 20, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 19, + "metrics": { + "recallAt1": 0.8, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 0.925, + "ndcgAt10": 0.940626, + "hitRateAt5": 1, + "mapAt10": 0.916667, + "evidencePrecisionAt5": 0.22 + } + } + }, + { + "threshold": 0.6, + "validation": { + "questions": 10, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 19, + "metrics": { + "recallAt1": 0.9, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 0.933333, + "ndcgAt10": 0.95, + "hitRateAt5": 1, + "mapAt10": 0.933333, + "evidencePrecisionAt5": 0.2 + } + }, + "test": { + "questions": 20, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 19, + "metrics": { + "recallAt1": 0.8, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 0.925, + "ndcgAt10": 0.940626, + "hitRateAt5": 1, + "mapAt10": 0.916667, + "evidencePrecisionAt5": 0.22 + } + } + } + ] +} diff --git a/docs/eval/threshold-v1.6.md b/docs/eval/threshold-v1.6.md new file mode 100644 index 0000000..c9f5c1f --- /dev/null +++ b/docs/eval/threshold-v1.6.md @@ -0,0 +1,51 @@ +# Threshold derivation — v1.6 (#192) + +Generated by `node scripts/eval-threshold.mjs`. Numbers are harness output; do not edit them by hand. + +## What was measured + +The real harness, the same corpus and the production retrieval config +(`candidateK=20, contextK=3`), once per candidate threshold. The **validation** +split selects; the **test** split reports. Both come from the same +`--eval-split=` code path, and the split is a deterministic function of the +question id, so this is reproducible. + +| Threshold | n (val) | Recall@5 (val) | nDCG@10 (val) | MAP@10 (val) | No-result (val) | nDCG@10 (test) | No-result (test) | +| --- | --- | --- | --- | --- | --- | --- | --- | +| 0 | 10 | 1.0000 | 0.9500 | 0.9333 | 0.0000 | 0.9406 | 0.0000 | +| 0.3 | 10 | 1.0000 | 0.9500 | 0.9333 | 0.0000 | 0.9406 | 0.0000 | +| 0.4 | 10 | 1.0000 | 0.9500 | 0.9333 | 0.0000 | 0.9406 | 0.0000 | +| 0.5 | 10 | 1.0000 | 0.9500 | 0.9333 | 0.0000 | 0.9406 | 0.0000 | +| 0.6 | 10 | 1.0000 | 0.9500 | 0.9333 | 0.0000 | 0.9406 | 0.0000 | + +## Selection rule + +Best validation nDCG@10, then fewest validation no-results, then the **lowest** +threshold — the first stage is supposed to favour recall and let a later stage +filter, so among equals the wider one is the safer default. + +## Outcome + +The sweep is **flat**: every threshold from 0 to 0.6 produces the same validation nDCG@10 (0.9500), the same Recall@5 (1.0000) and a no-result rate of 0.0000. On this corpus the threshold is simply **non-binding** — E5 never scores these query/chunk pairs below the top of the swept range, so no passage is ever filtered out. + +**No evidence to change `threshold = 0.5`.** The tie-break rule nominates `0` only because it prefers the widest threshold among equals; that is a tie-break, not a finding. What this run establishes is that the current value cannot be validated *or* falsified here, which is a property of the corpus, not of the threshold. Re-run after #192 child 2 grows it. + +The app currently ships `threshold = 0.5`: validation nDCG@10 +0.9500, test nDCG@10 +0.9406, test no-result rate +0.0000. + +## Caveat on this corpus + +The split removes the most obvious form of overfitting, but 10 +validation questions is a thin basis for a decision, and the corpus is still small. A +threshold is a product decision with a **no-result-rate** cost attached, so a +recommendation here is only as good as the corpus behind it. Re-run this after the +corpus grows (#192 child 2). + +## Reproduce + +```bash +npm run eval:prepare # one-time, networked model bootstrap +npm run eval:threshold # offline; rewrites this file +``` diff --git a/eval/README.md b/eval/README.md index c8964f1..2333a26 100644 --- a/eval/README.md +++ b/eval/README.md @@ -7,8 +7,10 @@ every experiment (#77, #78) is reported as a delta against that file. ## Commands ```bash -npm run eval:prepare # one-time, networked: download the pinned embedding model -npm run eval # offline and deterministic: run the harness, rewrite the baseline +npm run eval:prepare # one-time, networked: download the pinned embedding model +npm run eval # offline and deterministic: run the harness, rewrite the baseline +npm run eval:retrieval # strategy comparison (#77) +npm run eval:threshold # derive the similarity threshold on validation, report on test ``` ### The harness runs the production configuration @@ -22,6 +24,7 @@ the product rather than a research setup. Two Ks, because they answer different | `--eval-context-k=` | `3` | passages the chat prompt actually takes (`chatHandlers.ts`) | | `--eval-threshold=` | `0.5` | the similarity floor the app ships | | `--eval-retrieval=` | `dense` | `dense`, `sparse`, or `hybrid` | +| `--eval-split=` | `all` | `all`, `validation`, or `test` — a deterministic id-based split | | `--eval-baseline=` | `v1.6` | name written into `docs/eval/baseline-.{json,md}` | Ranking metrics are computed at `candidateK` depth, not at `contextK`: `Recall@10` needs @@ -29,6 +32,13 @@ at least ten results, and truncation only takes a prefix of the candidate list, truncation cannot change the ranking it is measured on. `contextK` is recorded so the report describes the whole online path. +### Swept parameters are chosen on `validation`, reported on `test` + +A parameter picked on the same questions it is scored on is a fitted number, not a +result. `--eval-split=validation` selects roughly a third of the questions by a +deterministic hash of the id; `test` is the rest. `npm run eval:threshold` uses this +to pick a similarity threshold on `validation` and report it on `test`. + `eval:prepare` downloads the pinned `multilingual-e5-small` revision into the app's model cache and verifies it. `eval` never touches the network: if the model is missing it stops with diff --git a/package.json b/package.json index 7bcf86c..b93f550 100644 --- a/package.json +++ b/package.json @@ -41,7 +41,8 @@ "db:migrate": "drizzle-kit migrate", "db:push": "drizzle-kit push", "db:studio": "drizzle-kit studio", - "eval:retrieval": "npm run build && node scripts/eval-retrieval.mjs" + "eval:retrieval": "npm run build && node scripts/eval-retrieval.mjs", + "eval:threshold": "npm run build && node scripts/eval-threshold.mjs" }, "//test": [ "`node --test` strips TypeScript types rather than compiling them, and strip-only", diff --git a/scripts/eval-threshold.mjs b/scripts/eval-threshold.mjs new file mode 100644 index 0000000..b6d1c73 --- /dev/null +++ b/scripts/eval-threshold.mjs @@ -0,0 +1,233 @@ +#!/usr/bin/env node +/** + * Threshold derivation for #192 (child 4). + * + * `threshold: 0.5` was hand-picked, and a cosine score has no universal meaning: + * the distribution depends on the embedding model, the language, the query type and + * the chunk length. This runs the real harness once per candidate threshold on a + * **validation** split, picks a winner there, and then reports that winner on the + * **test** split — so the number that justifies the choice is not the number the + * choice was fitted to. + * + * The harness owns the split (`--eval-split=`), so both sides are measured by the + * same code path that produces the frozen baseline. + * + * Usage: + * node scripts/eval-threshold.mjs + * + * The embedding model must already be prepared (`npm run eval:prepare`). + */ + +import { spawn } from 'node:child_process' +import { existsSync, mkdtempSync, mkdirSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join, resolve } from 'node:path' + +const OUT_MD = resolve(readArg('--out=', 'docs/eval/threshold-v1.6.md')) +const OUT_JSON = OUT_MD.replace(/\.md$/, '.json') + +/** + * 0 is the "no floor" arm: it keeps the ranking intact and lets a downstream stage + * filter. It is here because #192 says the first stage may legitimately run with no + * threshold at all. + */ +const THRESHOLDS = [0, 0.3, 0.4, 0.5, 0.6] + +/** The threshold the app currently ships, so the report can say whether it holds up. */ +const PRODUCTION_THRESHOLD = 0.5 + +function readArg(prefix, fallback) { + const arg = process.argv.find((value) => value.startsWith(prefix)) + return arg ? arg.slice(prefix.length) : fallback +} + +// Node 24 refuses to spawn a `.cmd`/`.bat` without `shell: true` (EINVAL), and the +// `.bin` entry is exactly that on Windows. Use the real binary the wrapper runs. +const { default: electronBinary } = await import('electron') +const executable = resolve(electronBinary) + +if (!existsSync(executable)) { + console.error('[threshold] could not find the electron binary. Run `npm install` first.') + process.exit(1) +} + +function runOne(threshold, split, outDir) { + return new Promise((resolvePromise, reject) => { + const args = [ + '.', + '--eval-harness', + '--eval-baseline=v1.6', + `--eval-out=${outDir}`, + `--eval-split=${split}`, + `--eval-threshold=${threshold}` + ] + + const isRoot = typeof process.getuid === 'function' && process.getuid() === 0 + if (isRoot || process.env.CI) args.push('--no-sandbox') + + const child = spawn(executable, args, { + stdio: ['ignore', 'pipe', 'pipe'], + env: { ...process.env, ELECTRON_DISABLE_SECURITY_WARNINGS: '1' } + }) + + child.on('error', reject) + child.on('exit', (code) => { + if (code !== 0) { + reject(new Error(`threshold ${threshold} (${split}) exited with code ${code}`)) + return + } + const reportPath = join(outDir, 'baseline-v1.6.json') + if (!existsSync(reportPath)) { + reject(new Error(`threshold ${threshold} (${split}) wrote no report`)) + return + } + const report = JSON.parse(readFileSync(reportPath, 'utf8')) + resolvePromise(summarize(report)) + }) + }) +} + +/** + * The frozen metrics plus the one this experiment needs and the baseline does not + * carry: how often a threshold turns a question into "no results at all". + * + * A higher threshold can look better on ranking metrics while quietly making the + * product answer "not in your sources" more often, and that trade is invisible + * unless it is counted. + */ +function summarize(report) { + const perQuestion = report.perQuestion ?? [] + const noResult = perQuestion.filter((q) => q.retrievedCount === 0).length + const retrieved = perQuestion.map((q) => q.retrievedCount) + return { + questions: perQuestion.length, + noResultCount: noResult, + noResultRate: perQuestion.length === 0 ? 0 : noResult / perQuestion.length, + meanRetrieved: retrieved.length === 0 ? 0 : retrieved.reduce((a, b) => a + b, 0) / retrieved.length, + metrics: report.metrics + } +} + +const format4 = (value) => value.toFixed(4) + +const workDir = mkdtempSync(join(tmpdir(), 'knownote-threshold-')) +const rows = [] + +try { + for (const threshold of THRESHOLDS) { + const validationDir = join(workDir, `${threshold}-validation`) + const testDir = join(workDir, `${threshold}-test`) + mkdirSync(validationDir, { recursive: true }) + mkdirSync(testDir, { recursive: true }) + + console.log(`[threshold] threshold ${threshold}: validation`) + const validation = await runOne(threshold, 'validation', validationDir) + console.log(`[threshold] threshold ${threshold}: test`) + const test = await runOne(threshold, 'test', testDir) + + rows.push({ threshold, validation, test }) + } +} finally { + rmSync(workDir, { recursive: true, force: true }) +} + +/** + * Selection rule, stated so it can be argued with: best validation nDCG@10, then + * fewest validation no-results, then the widest (lowest) threshold — because the + * first stage is supposed to favour recall and let a later stage filter. + */ +const ranked = [...rows].sort( + (a, b) => + b.validation.metrics.ndcgAt10 - a.validation.metrics.ndcgAt10 || + a.validation.noResultRate - b.validation.noResultRate || + a.threshold - b.threshold +) +const winner = ranked[0] +const production = rows.find((row) => row.threshold === PRODUCTION_THRESHOLD) + +/** + * A flat sweep is not a weak recommendation, it is no recommendation: if every + * threshold scores the same on both metrics, the corpus cannot tell them apart and + * moving a product parameter on that evidence would be noise dressed as a result. + */ +const flat = + rows.every( + (row) => + row.validation.metrics.ndcgAt10 === rows[0].validation.metrics.ndcgAt10 && + row.validation.noResultRate === rows[0].validation.noResultRate + ) + +const outcome = flat + ? `The sweep is **flat**: every threshold from ${THRESHOLDS[0]} to ${THRESHOLDS[THRESHOLDS.length - 1]} produces the same validation nDCG@10 (${format4(rows[0].validation.metrics.ndcgAt10)}), the same Recall@5 (${format4(rows[0].validation.metrics.recallAt5)}) and a no-result rate of ${format4(rows[0].validation.noResultRate)}. On this corpus the threshold is simply **non-binding** — E5 never scores these query/chunk pairs below the top of the swept range, so no passage is ever filtered out.\n\n**No evidence to change \`threshold = ${PRODUCTION_THRESHOLD}\`.** The tie-break rule nominates \`${winner.threshold}\` only because it prefers the widest threshold among equals; that is a tie-break, not a finding. What this run establishes is that the current value cannot be validated *or* falsified here, which is a property of the corpus, not of the threshold. Re-run after #192 child 2 grows it.` + : `**Recommended: \`threshold = ${winner.threshold}\`.**\n\n- Validation: nDCG@10 ${format4(winner.validation.metrics.ndcgAt10)}, Recall@5 ${format4(winner.validation.metrics.recallAt5)}, no-result rate ${format4(winner.validation.noResultRate)} (${winner.validation.noResultCount}/${winner.validation.questions})\n- Test: nDCG@10 ${format4(winner.test.metrics.ndcgAt10)}, Recall@5 ${format4(winner.test.metrics.recallAt5)}, no-result rate ${format4(winner.test.noResultRate)} (${winner.test.noResultCount}/${winner.test.questions})\n- Mean retrieved per question: ${winner.test.meanRetrieved.toFixed(2)} (validation ${winner.validation.meanRetrieved.toFixed(2)})` + +const tableRows = rows + .map( + (row) => + `| ${row.threshold} | ${row.validation.questions} | ${format4(row.validation.metrics.recallAt5)} | ` + + `${format4(row.validation.metrics.ndcgAt10)} | ${format4(row.validation.metrics.mapAt10)} | ` + + `${format4(row.validation.noResultRate)} | ${format4(row.test.metrics.ndcgAt10)} | ` + + `${format4(row.test.noResultRate)} |` + ) + .join('\n') + +const markdown = `# Threshold derivation — v1.6 (#192) + +Generated by \`node scripts/eval-threshold.mjs\`. Numbers are harness output; do not edit them by hand. + +## What was measured + +The real harness, the same corpus and the production retrieval config +(\`candidateK=20, contextK=3\`), once per candidate threshold. The **validation** +split selects; the **test** split reports. Both come from the same +\`--eval-split=\` code path, and the split is a deterministic function of the +question id, so this is reproducible. + +| Threshold | n (val) | Recall@5 (val) | nDCG@10 (val) | MAP@10 (val) | No-result (val) | nDCG@10 (test) | No-result (test) | +| --- | --- | --- | --- | --- | --- | --- | --- | +${tableRows} + +## Selection rule + +Best validation nDCG@10, then fewest validation no-results, then the **lowest** +threshold — the first stage is supposed to favour recall and let a later stage +filter, so among equals the wider one is the safer default. + +## Outcome + +${outcome} + +The app currently ships \`threshold = ${PRODUCTION_THRESHOLD}\`: validation nDCG@10 +${format4(production.validation.metrics.ndcgAt10)}, test nDCG@10 +${format4(production.test.metrics.ndcgAt10)}, test no-result rate +${format4(production.test.noResultRate)}. + +## Caveat on this corpus + +The split removes the most obvious form of overfitting, but ${rows[0].validation.questions} +validation questions is a thin basis for a decision, and the corpus is still small. A +threshold is a product decision with a **no-result-rate** cost attached, so a +recommendation here is only as good as the corpus behind it. Re-run this after the +corpus grows (#192 child 2). + +## Reproduce + +\`\`\`bash +npm run eval:prepare # one-time, networked model bootstrap +npm run eval:threshold # offline; rewrites this file +\`\`\` +` + +mkdirSync(resolve(OUT_MD, '..'), { recursive: true }) +writeFileSync( + OUT_JSON, + `${JSON.stringify({ baseline: 'v1.6', productionThreshold: PRODUCTION_THRESHOLD, flat, recommended: flat ? PRODUCTION_THRESHOLD : winner.threshold, rows }, null, 2)}\n` +) +writeFileSync(OUT_MD, markdown) + +console.log( + flat + ? `[threshold] flat sweep; no evidence to move off ${PRODUCTION_THRESHOLD}` + : `[threshold] recommended ${winner.threshold} (validation nDCG@10 ${format4(winner.validation.metrics.ndcgAt10)})` +) +console.log(`[threshold] wrote ${OUT_JSON} and ${OUT_MD}`) diff --git a/src/main/eval/harness.ts b/src/main/eval/harness.ts index 02e5bb4..7ae5114 100644 --- a/src/main/eval/harness.ts +++ b/src/main/eval/harness.ts @@ -37,10 +37,12 @@ import type { EvalQuestion, EvalReport, EvalRelevantLocation, + EvalSplit, EvalTypeBreakdown, QuestionReport, ResolvedGroundTruth } from './types' +import { selectSplit } from './types' type Db = ReturnType @@ -51,6 +53,8 @@ export interface EvalHarnessOptions { corpusLabel: string questionsPath: string baseline: string + /** 本次只评这一份切分(#192);缺省 `all`。 */ + split: EvalSplit /** * 第一阶段每个通道的宽度,也是排名指标的评估深度(#77)。 * @@ -216,7 +220,11 @@ export async function runEvalHarness( options.corpusDir, options.chunkOptions ) - const questions = parseQuestions(await readFile(options.questionsPath, 'utf-8')) + const allQuestions = parseQuestions(await readFile(options.questionsPath, 'utf-8')) + const questions = selectSplit(allQuestions, options.split) + if (questions.length === 0) { + throw new Error(`eval split "${options.split}" selected no questions from ${options.questionsPath}`) + } const perQuestion: QuestionReport[] = [] const latencies: number[] = [] @@ -282,6 +290,7 @@ export async function runEvalHarness( respectHeadings: chunking.respectHeadings }, retrieval: options.strategy, + split: options.split, candidateK: options.candidateK, contextK: options.contextK, threshold: options.threshold, diff --git a/src/main/eval/report.ts b/src/main/eval/report.ts index 34d9688..5b8897c 100644 --- a/src/main/eval/report.ts +++ b/src/main/eval/report.ts @@ -38,7 +38,8 @@ Generated by \`${report.generatedBy}\`. The numbers below are harness output — | Ranks | \`candidateK=${config.candidateK}, threshold=${config.threshold}\` | | Context width | \`contextK=${config.contextK}\` | | Evidence per query | \`evidenceK=${config.evidenceK}\` | -| Corpus | \`${config.corpus}\` (${config.documents} documents, ${config.questions} questions) | +| Corpus | \`${config.corpus}\` (${config.documents} documents) | +| Split | \`${config.split}\` (${config.questions} questions) | | Index size | ${config.chunkCount} chunks | ## Metrics diff --git a/src/main/eval/run.ts b/src/main/eval/run.ts index 29085ad..802cf5e 100644 --- a/src/main/eval/run.ts +++ b/src/main/eval/run.ts @@ -22,6 +22,7 @@ import { KnowledgeService } from '../services/KnowledgeService' import { isModelInstalled } from '../embedding/ModelRegistry' import { DEFAULT_CHUNK_OPTIONS, type ChunkOptions } from '../services/ChunkingService' import type { RetrievalStrategy } from '../services/retrieval' +import type { EvalSplit } from './types' import { runEvalHarness, stabilize, @@ -83,6 +84,17 @@ function readRetrievalStrategy(argv: readonly string[]): RetrievalStrategy { return raw } +/** + * 本次评估的切分(#192)。默认 `all`;阈值这类扫参要用 `validation` 选、`test` 报。 + */ +function readSplit(argv: readonly string[]): EvalSplit { + const raw = readOption(argv, '--eval-split=', 'all') + if (raw !== 'all' && raw !== 'validation' && raw !== 'test') { + throw new Error(`--eval-split expects all, validation or test, got ${JSON.stringify(raw)}`) + } + return raw +} + /** * Chunking config for one run (#78). The defaults are the production defaults, so * `npm run eval` with no flags still measures what ships. @@ -179,6 +191,7 @@ export async function runEvalCli(argv: readonly string[] = process.argv): Promis corpusLabel: repoRelative(corpusDir) || 'eval/corpus', questionsPath, baseline: readOption(argv, '--eval-baseline=', 'v1.6'), + split: readSplit(argv), // 默认就是生产配置(#77):先取宽,融合,再把 contextK 条送进 prompt。一个不镜像 // 线上参数的 benchmark 量的是用户永远不会跑的检索器。 candidateK: readNumberOption(argv, '--eval-candidate-k=', 20), diff --git a/src/main/eval/types.ts b/src/main/eval/types.ts index dbb692b..3a075f4 100644 --- a/src/main/eval/types.ts +++ b/src/main/eval/types.ts @@ -34,6 +34,29 @@ export interface EvalQuestion { type?: string } +/** + * 评估切分(#192)。 + * + * 阈值这类参数必须在**没参与选择**的问题上报数,否则扫参的结果只是把测试集背下来 + * 了。切分按 question id 确定性计算,所以同一份 `questions.jsonl` 在任何机器上切出 + * 同一份 validation / test。 + */ +export type EvalSplit = 'all' | 'validation' | 'test' + +/** id 分桶,0/1/2;只用于切分,不参与检索。 */ +export function splitBucket(id: string): number { + let hash = 0 + for (const character of id) hash = (hash * 31 + character.charCodeAt(0)) >>> 0 + return hash % 3 +} + +/** validation 是 bucket 0(约 1/3),test 是其余(约 2/3)。 */ +export function selectSplit(questions: EvalQuestion[], split: EvalSplit): EvalQuestion[] { + if (split === 'all') return questions + const wantValidation = split === 'validation' + return questions.filter((question) => (splitBucket(question.id) === 0) === wantValidation) +} + /** One resolved ground-truth location, after runtime id mapping. */ export interface ResolvedGroundTruth { document: string @@ -101,6 +124,8 @@ export interface EvalReport { respectHeadings: boolean } retrieval: string + /** 本次评估用了哪一份切分(#192):`all` / `validation` / `test`。 */ + split: string /** * 第一阶段每个通道的宽度(#77)。排名指标(Recall@K / MRR / nDCG@K)在这个深度上 * 计算,所以它必须 ≥ 指标里最大的 K。 diff --git a/test/evalSplit.test.ts b/test/evalSplit.test.ts new file mode 100644 index 0000000..2de5a02 --- /dev/null +++ b/test/evalSplit.test.ts @@ -0,0 +1,61 @@ +import { test } from 'node:test' +import assert from 'node:assert/strict' +import { + selectSplit, + splitBucket, + type EvalQuestion, + type EvalSplit +} from '../src/main/eval/types.ts' + +/** + * The eval split (#192). Thresholds and other swept parameters have to be chosen on + * questions that did not take part in the choice, so the split has to be + * deterministic: the same `questions.jsonl` must produce the same validation and + * test sets on every machine. + */ + +const question = (id: string): EvalQuestion => ({ + id, + question: id, + relevant: [{ document: 'a.md', page: null, block: 0 }] +}) + +const questions = Array.from({ length: 30 }, (_, index) => question(`q${String(index + 1).padStart(3, '0')}`)) + +test('the split buckets only ever return 0, 1 or 2', () => { + for (const q of questions) { + const bucket = splitBucket(q.id) + assert.ok(bucket === 0 || bucket === 1 || bucket === 2, `${q.id} -> ${bucket}`) + } +}) + +test('the split is deterministic, not random', () => { + const first = selectSplit(questions, 'validation').map((q) => q.id) + const second = selectSplit(questions, 'validation').map((q) => q.id) + assert.deepEqual(first, second) +}) + +test('validation and test partition the questions without overlap', () => { + const validation = selectSplit(questions, 'validation').map((q) => q.id) + const test = selectSplit(questions, 'test').map((q) => q.id) + + assert.equal(validation.length + test.length, questions.length) + assert.equal(new Set([...validation, ...test]).size, questions.length) + // Both sides are non-empty on a 30-question set, so neither arm is vacuous. + assert.ok(validation.length > 0 && test.length > 0) +}) + +test('`all` is the whole set, and the selected questions keep their order', () => { + assert.deepEqual( + selectSplit(questions, 'all').map((q) => q.id), + questions.map((q) => q.id) + ) + + for (const split of ['validation', 'test'] as EvalSplit[]) { + const selected = selectSplit(questions, split).map((q) => q.id) + const expected = questions + .filter((q) => (splitBucket(q.id) === 0) === (split === 'validation')) + .map((q) => q.id) + assert.deepEqual(selected, expected) + } +}) From a52b9b7cf7261049580797c0601b1780f5c0a616 Mon Sep 17 00:00:00 2001 From: MrSibe Date: Wed, 30 Sep 2026 16:58:46 +0800 Subject: [PATCH 05/18] feat(eval): adopt on the first metric with headroom, not a saturated one MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The v1.5 rule was "Recall@5 must improve and nDCG@10 must not regress". On a corpus where dense already scores Recall@5 = 1.0000, no strategy can improve Recall@5, so the rule was not strict — it was **unsatisfiable**. Every comparison came back "inconclusive", including a hybrid that was better on Recall@1, MRR and nDCG@10. The default never moved, not because hybrid lost but because the rule could not return a verdict. Child 10 of #192. **The amendment.** A metric at its maximum has no headroom and is not allowed to decide. The deciding metric is the first one with headroom, in the order `recallAt5`, `ndcgAt10`, `mrr`, `mapAt10`; a strategy clears the rule when it improves that metric and regresses none of the others. Saturation is detected and reported rather than silently blocking every change. **The rule is code, not prose.** It lives in `src/main/eval/adoption.ts` with 8 unit tests, because it decides whether a shipped default moves and testing it by reading the sentence the script prints would only test the sentence. The experiment script imports it, so there is one definition. ## The v1.5 stalemate resolves ``` | Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Query p95 | | dense (vector) | 0.8333 | 1.0000 | 0.9278 | 0.9437 | 0.9222 | 14.06 ms | | sparse (BM25) | 0.7333 | 0.8667 | 0.8056 | 0.8184 | 0.8000 | 2.63 ms | | hybrid (RRF dense + BM25) | 0.8667 | 1.0000 | 0.9444 | 0.9561 | 0.9389 | 22.32 ms | ``` `recallAt5` is reported as saturated; the deciding metric is `ndcgAt10`; **hybrid clears the rule** and regresses none of the four metrics, at a p95 cost of ~8 ms. The script reports the measurement and does not flip the default — changing the shipped strategy is a product decision, and it is stated that way in the output rather than implied by a green checkmark. `docs/eval/retrieval-v1.6.{json,md}` is the regenerated experiment against the v1.6 baseline; `retrieval-v1.5.*` is kept as the record of the superseded rule. ## Testing - `npm run typecheck` — clean - `npm test` — 497 pass, 8 new: saturation detection, deciding-metric priority, the float-slack regression check, the resolved stalemate, a trade that regresses another metric being refused, the baseline not clearing against itself, all-saturated, and best-candidate selection - `npm run eval` + `npm run eval:retrieval` — regenerated and re-run Part of #192 (child 10). --- docs/eval/baseline-v1.6.md | 14 ++++-- docs/eval/retrieval-v1.6.json | 54 ++++++++++++++++++++++ docs/eval/retrieval-v1.6.md | 49 ++++++++++++++++++++ package.json | 2 +- scripts/eval-retrieval.mjs | 76 ++++++++++++++++-------------- src/main/eval/adoption.ts | 87 +++++++++++++++++++++++++++++++++++ src/main/eval/report.ts | 10 ++-- test/evalAdoption.test.ts | 86 ++++++++++++++++++++++++++++++++++ 8 files changed, 335 insertions(+), 43 deletions(-) create mode 100644 docs/eval/retrieval-v1.6.json create mode 100644 docs/eval/retrieval-v1.6.md create mode 100644 src/main/eval/adoption.ts create mode 100644 test/evalAdoption.test.ts diff --git a/docs/eval/baseline-v1.6.md b/docs/eval/baseline-v1.6.md index 605c29a..ce25b05 100644 --- a/docs/eval/baseline-v1.6.md +++ b/docs/eval/baseline-v1.6.md @@ -43,8 +43,8 @@ The type comes from `type` in `questions.jsonl`; untagged questions report as | semantic | 18 | 1.0000 | 0.9312 | 1.0000 | 0.9074 | | zh | 3 | 1.0000 | 1.0000 | 1.0000 | 1.0000 | -Timing is informational only and is **not** frozen: indexing 1542 ms, query -p50 15.01 ms, p95 69.16 ms on the +Timing is informational only and is **not** frozen: indexing 1551 ms, query +p50 11.73 ms, p95 18.71 ms on the machine that produced this file. Timing and index size depend on hardware and on the corpus, so they must never be the reason two runs differ. @@ -68,9 +68,13 @@ first-stage width per channel, `contextK` is how many passages the chat prompt t and `threshold` is the similarity floor the app ships. A benchmark that does not mirror those parameters measures a retriever nobody runs. The adopted-change rule is: -> Adopt a change only if Recall@5 improves and nDCG@10 does not regress. A change -> that trades a large latency increase for a marginal recall gain is a product -> decision, not an automatic win, and must be stated as such. +> Adopt a strategy when it improves the **first metric with headroom** — in the order +> Recall@5, nDCG@10, MRR, MAP@10 — and regresses none of the others. A metric already +> at its maximum has no headroom and cannot decide anything; a rule that depends on +> one is unsatisfiable, not strict (#192 child 10). +> +> A change that trades a large latency increase for a marginal quality gain is a +> product decision, not an automatic win, and must be stated as such. A changed result must be reproducible with: diff --git a/docs/eval/retrieval-v1.6.json b/docs/eval/retrieval-v1.6.json new file mode 100644 index 0000000..8fed18f --- /dev/null +++ b/docs/eval/retrieval-v1.6.json @@ -0,0 +1,54 @@ +{ + "baseline": "dense", + "chunking": "1000/100", + "strategies": [ + { + "id": "dense", + "label": "dense (vector)", + "chunking": "1000/100", + "chunkCount": 19, + "recallAt1": 0.833333, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 0.927778, + "ndcgAt10": 0.94375, + "hitRateAt5": 1, + "mapAt10": 0.922222, + "evidencePrecisionAt5": 0.213333, + "indexingMs": 1531, + "latencyP95Ms": 14.81 + }, + { + "id": "sparse", + "label": "sparse (BM25)", + "chunking": "1000/100", + "chunkCount": 19, + "recallAt1": 0.733333, + "recallAt5": 0.866667, + "recallAt10": 0.866667, + "mrr": 0.805556, + "ndcgAt10": 0.818355, + "hitRateAt5": 0.866667, + "mapAt10": 0.8, + "evidencePrecisionAt5": 0.186667, + "indexingMs": 1503, + "latencyP95Ms": 1.89 + }, + { + "id": "hybrid", + "label": "hybrid (RRF of dense + BM25)", + "chunking": "1000/100", + "chunkCount": 19, + "recallAt1": 0.866667, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 0.944444, + "ndcgAt10": 0.956053, + "hitRateAt5": 1, + "mapAt10": 0.938889, + "evidencePrecisionAt5": 0.213333, + "indexingMs": 1548, + "latencyP95Ms": 19.15 + } + ] +} diff --git a/docs/eval/retrieval-v1.6.md b/docs/eval/retrieval-v1.6.md new file mode 100644 index 0000000..9651e44 --- /dev/null +++ b/docs/eval/retrieval-v1.6.md @@ -0,0 +1,49 @@ +# Retrieval experiments — v1.6 (#77, #192) + +Generated by `node scripts/eval-retrieval.mjs`. Numbers are harness output; do not edit them by hand. + +## What was measured + +Every strategy runs the real RAG eval harness against the same corpus and the same 30 +questions as `baseline-v1.6.json`, with chunking held fixed at 1000/100. Only the retrieval strategy changes. + +| Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Evidence P@5 | Query p95 | +| --- | --- | --- | --- | --- | --- | --- | --- | +| dense (vector) | 0.8333 | 1.0000 | 0.9278 | 0.9437 | 0.9222 | 0.2133 | 14.81 ms | +| sparse (BM25) | 0.7333 | 0.8667 | 0.8056 | 0.8184 | 0.8000 | 0.1867 | 1.89 ms | +| hybrid (RRF of dense + BM25) | 0.8667 | 1.0000 | 0.9444 | 0.9561 | 0.9389 | 0.2133 | 19.15 ms | + +## Not evaluated + +**Reranking.** The issue lists "hybrid + reranker" as a step, but a cross-encoder +model is not available offline and inventing its numbers would defeat the point of +the harness. It stays open until a model can be pinned the way the embedding model +is. + +## Saturation + +Saturated (no headroom, so they cannot decide anything): `recallAt5`. The deciding metric on this corpus is `ndcgAt10`. A saturated metric is still reported, because "this corpus cannot move it" is itself +information; it is just not allowed to decide the comparison. + +## Adoption rule + +> Adopt a strategy when it improves the **first metric with headroom** — in the order +> `recallAt5`, `nDCG@10`, `MRR`, `MAP@10` — and regresses none of the others. A metric +> already at its maximum has no headroom and cannot decide anything; a rule that depends +> on one is unsatisfiable, not strict (#192 child 10). +> +> A change that trades a large latency increase for a marginal quality gain is a product +> decision, not an automatic win. + +## Outcome + +`hybrid (RRF of dense + BM25)` **clears the rule**: it improves the deciding metric `ndcgAt10` (0.9561 vs dense 0.9437) and regresses none of `recallAt5`, `ndcgAt10`, `mrr`, `mapAt10`. Saturated (no headroom, so they cannot decide anything): `recallAt5`. + +Changing the shipped default is a separate decision, and this script does not make it — it reports the measurement. + +## Reproduce + +```bash +npm run eval:prepare # one-time, networked model bootstrap +npm run eval:retrieval # offline; runs every strategy and rewrites this file +``` diff --git a/package.json b/package.json index b93f550..f7b0bcb 100644 --- a/package.json +++ b/package.json @@ -41,7 +41,7 @@ "db:migrate": "drizzle-kit migrate", "db:push": "drizzle-kit push", "db:studio": "drizzle-kit studio", - "eval:retrieval": "npm run build && node scripts/eval-retrieval.mjs", + "eval:retrieval": "npm run build && node --experimental-transform-types --disable-warning=ExperimentalWarning --disable-warning=MODULE_TYPELESS_PACKAGE_JSON scripts/eval-retrieval.mjs", "eval:threshold": "npm run build && node scripts/eval-threshold.mjs" }, "//test": [ diff --git a/scripts/eval-retrieval.mjs b/scripts/eval-retrieval.mjs index bccc605..d52b5bd 100644 --- a/scripts/eval-retrieval.mjs +++ b/scripts/eval-retrieval.mjs @@ -3,11 +3,14 @@ * Retrieval experiments for #77. * * Runs the real RAG eval harness once per retrieval strategy against the frozen - * chunk baseline (`baseline-v1.5.json`, 1000/100), holding chunking fixed, and + * chunk baseline (`baseline-v1.6.json`, 1000/100), holding chunking fixed, and * writes the comparison the issue asks for as a delta against dense. * * The harness does the measuring; this script only orchestrates and tabulates. * + * The adoption rule is the amended one from #192 child 10: the deciding metric is the + * first metric with headroom, not a metric that the corpus has already maxed out. + * * Usage: * node scripts/eval-retrieval.mjs * @@ -19,7 +22,7 @@ import { existsSync, mkdtempSync, mkdirSync, readFileSync, rmSync, writeFileSync import { tmpdir } from 'node:os' import { join, resolve } from 'node:path' -const OUT_MD = resolve(readArg('--out=', 'docs/eval/retrieval-v1.5.md')) +const OUT_MD = resolve(readArg('--out=', 'docs/eval/retrieval-v1.6.md')) const OUT_JSON = OUT_MD.replace(/\.md$/, '.json') /** @@ -54,7 +57,7 @@ function runStrategy(strategy, outDir) { const args = [ '.', '--eval-harness', - '--eval-baseline=v1.5', + '--eval-baseline=v1.6', `--eval-out=${outDir}`, `--eval-retrieval=${strategy.id}` ] @@ -117,14 +120,14 @@ try { console.log(`[retrieval] running ${strategy.label}`) const metrics = await runStrategy(strategy, outDir) - const report = JSON.parse(readFileSync(join(outDir, 'baseline-v1.5.json'), 'utf8')) + const report = JSON.parse(readFileSync(join(outDir, 'baseline-v1.6.json'), 'utf8')) results.push({ id: strategy.id, label: strategy.label, chunking: `${report.config.chunking.chunkSize}/${report.config.chunking.chunkOverlap}`, chunkCount: report.config.chunkCount, ...metrics, - ...readTiming(join(outDir, 'baseline-v1.5.md')) + ...readTiming(join(outDir, 'baseline-v1.6.md')) }) } } finally { @@ -138,43 +141,46 @@ const format4 = (value) => value.toFixed(4) const rows = results.map( (result) => - `| ${result.label} | ${format4(result.recallAt1)} | ${format4(result.recallAt5)} | ${format4(result.mrr)} | ${format4(result.ndcgAt10)} | ${format4(result.evidencePrecisionAt5)} | ${result.latencyP95Ms?.toFixed(2)} ms |` + `| ${result.label} | ${format4(result.recallAt1)} | ${format4(result.recallAt5)} | ${format4(result.mrr)} | ${format4(result.ndcgAt10)} | ${format4(result.mapAt10)} | ${format4(result.evidencePrecisionAt5)} | ${result.latencyP95Ms?.toFixed(2)} ms |` ) /** - * The rule frozen in the baseline report: Recall@5 must improve and nDCG@10 must - * not regress. Latency is reported so a win that costs 5x latency is stated as a - * trade-off, not hidden. + * The rule lives in `src/main/eval/adoption.ts` so it can be unit tested: it decides + * whether a shipped default moves, and testing it by reading the sentence this script + * prints would be a test of the sentence. + * + * Latency stays in the table so a win that costs 5x latency is stated as a trade-off, + * not hidden. */ -const adopted = results.filter( - (result) => - result.id !== 'dense' && - result.recallAt5 > baseline.recallAt5 && - result.ndcgAt10 >= baseline.ndcgAt10 -) -const winner = - adopted.sort((a, b) => b.recallAt5 - a.recallAt5 || b.ndcgAt10 - a.ndcgAt10)[0] ?? null +const { ADOPTION_METRICS, decideAdoption } = await import('../src/main/eval/adoption.ts') +const { primary, saturated, winner } = decideAdoption(baseline, results) + +const saturationNote = saturated.length + ? `Saturated (no headroom, so they cannot decide anything): ${saturated + .map((key) => `\`${key}\``) + .join(', ')}.` + : 'No metric in the rule is saturated on this corpus.' let outcome if (winner) { - outcome = `\`${winner.label}\` clears the rule (Recall@5 ${format4(winner.recallAt5)} vs dense ${format4(baseline.recallAt5)}, nDCG@10 ${format4(winner.ndcgAt10)} vs ${format4(baseline.ndcgAt10)}).` -} else if (baseline.recallAt5 === 1) { - outcome = `Recall@5 is saturated at 1.0000, so the rule's first condition cannot be met by any strategy. **Dense stays the default**, and the non-dense strategies are reported as inconclusive rather than adopted or rejected on a metric that cannot move.` + outcome = `\`${winner.label}\` **clears the rule**: it improves the deciding metric \`${primary}\` (${format4(winner[primary])} vs dense ${format4(baseline[primary])}) and regresses none of ${ADOPTION_METRICS.map((key) => `\`${key}\``).join(', ')}. ${saturationNote}\n\nChanging the shipped default is a separate decision, and this script does not make it — it reports the measurement.` +} else if (primary === null) { + outcome = `Every metric in the rule is already at its maximum on this corpus, so no strategy can clear any of them. **Dense stays the default**; the comparison is inconclusive by construction, not negative.` } else { - outcome = `No strategy cleared the rule. **Dense stays the default.** A negative result is the point of the experiment: it is the measurement that says the extra machinery is not worth its cost on this corpus, not a failure to deliver.` + outcome = `No strategy cleared the rule. The deciding metric was \`${primary}\` (dense ${format4(baseline[primary])}); the strategies either failed to improve it or regressed another metric. **Dense stays the default.** A negative result is the point of the experiment: it is the measurement that says the extra machinery is not worth its cost on this corpus, not a failure to deliver. ${saturationNote}` } -const markdown = `# Retrieval experiments — v1.5 (#77) +const markdown = `# Retrieval experiments — v1.6 (#77, #192) Generated by \`node scripts/eval-retrieval.mjs\`. Numbers are harness output; do not edit them by hand. ## What was measured Every strategy runs the real RAG eval harness against the same corpus and the same 30 -questions as \`baseline-v1.5.json\`, with chunking held fixed at ${baseline.chunking}. Only the retrieval strategy changes. +questions as \`baseline-v1.6.json\`, with chunking held fixed at ${baseline.chunking}. Only the retrieval strategy changes. -| Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | Evidence P@5 | Query p95 | -| --- | --- | --- | --- | --- | --- | --- | +| Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Evidence P@5 | Query p95 | +| --- | --- | --- | --- | --- | --- | --- | --- | ${rows.join('\n')} ## Not evaluated @@ -184,19 +190,21 @@ model is not available offline and inventing its numbers would defeat the point the harness. It stays open until a model can be pinned the way the embedding model is. -## Corpus limitation +## Saturation -The rule's Recall@5 condition is **saturated** on this corpus: dense already scores -1.0000, so no strategy can improve it and the rule can therefore never be met here. -The metrics that still discriminate are Recall@1, MRR and nDCG@10. A hybrid result -that is better on all three but equal on Recall@5 is therefore *inconclusive*, not a -negative result, and the default is left unchanged until the comparison can run on a -corpus where Recall@5 is not already perfect. +${saturationNote} The deciding metric on this corpus is ${ + primary === null ? 'none — every metric in the rule is already maxed out' : `\`${primary}\`` +}. A saturated metric is still reported, because "this corpus cannot move it" is itself +information; it is just not allowed to decide the comparison. ## Adoption rule -> Adopt a change only if Recall@5 improves and nDCG@10 does not regress. A change -> that trades a large latency increase for a marginal recall gain is a product +> Adopt a strategy when it improves the **first metric with headroom** — in the order +> \`recallAt5\`, \`nDCG@10\`, \`MRR\`, \`MAP@10\` — and regresses none of the others. A metric +> already at its maximum has no headroom and cannot decide anything; a rule that depends +> on one is unsatisfiable, not strict (#192 child 10). +> +> A change that trades a large latency increase for a marginal quality gain is a product > decision, not an automatic win. ## Outcome diff --git a/src/main/eval/adoption.ts b/src/main/eval/adoption.ts new file mode 100644 index 0000000..e63c519 --- /dev/null +++ b/src/main/eval/adoption.ts @@ -0,0 +1,87 @@ +/** + * The adoption rule for retrieval experiments (#192 child 10). + * + * The v1.5 rule was "Recall@5 must improve and nDCG@10 must not regress". On a corpus + * where dense already scores Recall@5 = 1.0000 that condition can never be met, so the + * rule was not strict, it was **unsatisfiable**, and every strategy comparison came + * back "inconclusive" — including a hybrid that was better on every other metric. + * + * The amendment: a metric at its maximum has no headroom and is not allowed to decide. + * The deciding metric is the first one with headroom, and a strategy is adopted when + * it improves that metric and regresses none of the others. + * + * Kept here, not inside the experiment script, so the rule can be unit tested: it + * decides whether a production default moves, and "the script printed a different + * sentence" is not a test. + */ + +/** Priority order. Recall first because a RAG miss cannot be repaired downstream. */ +export const ADOPTION_METRICS = ['recallAt5', 'ndcgAt10', 'mrr', 'mapAt10'] as const + +export type AdoptionMetric = (typeof ADOPTION_METRICS)[number] + +export type MetricBag = Record + +/** Float slack: metrics are rounded to 6 decimals before this runs. */ +export const ADOPTION_EPSILON = 1e-9 + +/** Metrics already at their maximum, which therefore cannot decide a comparison. */ +export function saturatedMetrics( + metrics: MetricBag, + order: readonly string[] = ADOPTION_METRICS +): string[] { + return order.filter((key) => metrics[key] >= 1 - ADOPTION_EPSILON) +} + +/** The first metric in priority order with room to improve, or `null` if none has. */ +export function decidingMetric( + metrics: MetricBag, + order: readonly string[] = ADOPTION_METRICS +): string | null { + return order.find((key) => metrics[key] < 1 - ADOPTION_EPSILON) ?? null +} + +/** Metrics where `candidate` is worse than `baseline` beyond the epsilon. */ +export function regressedMetrics( + candidate: MetricBag, + baseline: MetricBag, + order: readonly string[] = ADOPTION_METRICS +): string[] { + return order.filter((key) => candidate[key] < baseline[key] - ADOPTION_EPSILON) +} + +export interface AdoptionDecision { + /** The metric that decides, or `null` when every metric is already maxed out. */ + primary: string | null + saturated: string[] + /** The best strategy that clears the rule, or `null`. */ + winner: MetricBag | null +} + +/** + * Pick the strategy to recommend. `baseline` is the shipped one and must be part of + * `candidates`; a candidate that improves the deciding metric and regresses nothing + * else clears the rule, and the best such candidate by the deciding metric wins. + */ +export function decideAdoption( + baseline: MetricBag, + candidates: readonly MetricBag[], + order: readonly string[] = ADOPTION_METRICS +): AdoptionDecision { + const saturated = saturatedMetrics(baseline, order) + const primary = decidingMetric(baseline, order) + + if (primary === null) return { primary: null, saturated, winner: null } + + const cleared = candidates.filter( + (candidate) => + candidate !== baseline && + candidate[primary] > baseline[primary] + ADOPTION_EPSILON && + regressedMetrics(candidate, baseline, order).length === 0 + ) + + const winner = + [...cleared].sort((a, b) => b[primary] - a[primary])[0] ?? null + + return { primary, saturated, winner } +} diff --git a/src/main/eval/report.ts b/src/main/eval/report.ts index 5b8897c..7b883f8 100644 --- a/src/main/eval/report.ts +++ b/src/main/eval/report.ts @@ -90,9 +90,13 @@ first-stage width per channel, \`contextK\` is how many passages the chat prompt and \`threshold\` is the similarity floor the app ships. A benchmark that does not mirror those parameters measures a retriever nobody runs. The adopted-change rule is: -> Adopt a change only if Recall@5 improves and nDCG@10 does not regress. A change -> that trades a large latency increase for a marginal recall gain is a product -> decision, not an automatic win, and must be stated as such. +> Adopt a strategy when it improves the **first metric with headroom** — in the order +> Recall@5, nDCG@10, MRR, MAP@10 — and regresses none of the others. A metric already +> at its maximum has no headroom and cannot decide anything; a rule that depends on +> one is unsatisfiable, not strict (#192 child 10). +> +> A change that trades a large latency increase for a marginal quality gain is a +> product decision, not an automatic win, and must be stated as such. A changed result must be reproducible with: diff --git a/test/evalAdoption.test.ts b/test/evalAdoption.test.ts new file mode 100644 index 0000000..ad61e05 --- /dev/null +++ b/test/evalAdoption.test.ts @@ -0,0 +1,86 @@ +import { test } from 'node:test' +import assert from 'node:assert/strict' +import { + ADOPTION_METRICS, + decideAdoption, + decidingMetric, + regressedMetrics, + saturatedMetrics +} from '../src/main/eval/adoption.ts' + +/** + * The adoption rule (#192 child 10) decides whether a shipped default moves, so it is + * pinned here rather than trusted to the sentence the experiment script prints. + * + * The bug it exists for: the v1.5 rule made Recall@5 — already 1.0000 on the corpus — + * its deciding condition, so a hybrid that was better on every other metric came back + * "inconclusive" forever. + */ + +const bag = (overrides: Record = {}): Record => ({ + recallAt5: 0.9, + ndcgAt10: 0.9, + mrr: 0.9, + mapAt10: 0.9, + ...overrides +}) + +test('a metric at its maximum is saturated and cannot decide', () => { + assert.deepEqual(saturatedMetrics(bag({ recallAt5: 1 })), ['recallAt5']) + assert.deepEqual(saturatedMetrics(bag()), []) +}) + +test('the deciding metric is the first one with headroom', () => { + assert.equal(decidingMetric(bag({ recallAt5: 1 })), 'ndcgAt10') + assert.equal(decidingMetric(bag({ recallAt5: 1, ndcgAt10: 1 })), 'mrr') + assert.equal(decidingMetric(bag()), 'recallAt5') + assert.equal(decidingMetric(bag({ recallAt5: 1, ndcgAt10: 1, mrr: 1, mapAt10: 1 })), null) +}) + +test('a regression is measured against the baseline, beyond the float slack', () => { + const baseline = bag() + assert.deepEqual(regressedMetrics(bag({ ndcgAt10: 0.8 }), baseline), ['ndcgAt10']) + // Rounding to 6 decimals must not read as a regression. + assert.deepEqual(regressedMetrics(bag({ ndcgAt10: 0.9 - 1e-12 }), baseline), []) +}) + +test('the v1.5 stalemate is resolved: improving the deciding metric is enough', () => { + const dense = bag({ recallAt5: 1, ndcgAt10: 0.9437, mrr: 0.9278, mapAt10: 0.9222 }) + const hybrid = bag({ recallAt5: 1, ndcgAt10: 0.9561, mrr: 0.9444, mapAt10: 0.9389 }) + + const decision = decideAdoption(dense, [dense, hybrid]) + + assert.deepEqual(decision.saturated, ['recallAt5']) + assert.equal(decision.primary, 'ndcgAt10') + assert.equal(decision.winner, hybrid) +}) + +test('improving one metric while regressing another does not clear the rule', () => { + const dense = bag({ recallAt5: 1, ndcgAt10: 0.9437, mrr: 0.9278 }) + // Better nDCG, worse MRR: exactly the trade the rule refuses to make silently. + const trade = bag({ recallAt5: 1, ndcgAt10: 0.99, mrr: 0.9 }) + + assert.equal(decideAdoption(dense, [dense, trade]).winner, null) +}) + +test('the baseline never clears the rule against itself', () => { + const dense = bag({ recallAt5: 1 }) + assert.equal(decideAdoption(dense, [dense]).winner, null) +}) + +test('when every metric is saturated nothing can be adopted', () => { + const all = bag({ recallAt5: 1, ndcgAt10: 1, mrr: 1, mapAt10: 1 }) + const decision = decideAdoption(all, [all, bag({ ndcgAt10: 1 })]) + + assert.equal(decision.primary, null) + assert.equal(decision.winner, null) + assert.deepEqual(decision.saturated, [...ADOPTION_METRICS]) +}) + +test('the best clearing candidate by the deciding metric wins', () => { + const dense = bag({ recallAt5: 1, ndcgAt10: 0.9 }) + const good = bag({ recallAt5: 1, ndcgAt10: 0.95 }) + const better = bag({ recallAt5: 1, ndcgAt10: 0.98 }) + + assert.equal(decideAdoption(dense, [dense, good, better]).winner, better) +}) From b77257a5ddf249e01d32ab511bb3e0bbf15f36f5 Mon Sep 17 00:00:00 2001 From: MrSibe Date: Wed, 30 Sep 2026 17:02:01 +0800 Subject: [PATCH 06/18] feat(eval): measure the context window, and name what still needs a model MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The harness measured the retriever but never the window the prompt actually gets. `evidenceK: 5` was a separate constant from the `contextK: 3` production uses, so the one metric that looked at a window looked at a different one than the product does. Child 5 of #192. **`contextK` is now the window for both context metrics.** - `contextPrecision@contextK` — of the first `contextK` passages, the share covering ground truth. This is what `evidencePrecisionAt5` was, at the production width. - `contextRecall@contextK` — the share of needed ground-truth blocks that made it into that window. Distinct from `Recall@10`: a block found at rank 4 is invisible when `contextK = 3`, and that is a product fact, not a ranking fact. Both are **deterministic**: the dataset says which blocks answer the question, so no model is needed to score a window. Together they are the trade-off a `contextK` decision makes — a wider window finds more and carries more noise — which is what the sweep in the next child needs. Baseline: `contextPrecision@3 = 0.3556`, `contextRecall@3 = 1.0000` — the needed evidence is always inside the top 3 on this corpus, but only about a third of what is inside the window is relevant. That second number is the one that says the window is paying for passages that do not answer the question. **What is deliberately not here.** Faithfulness, completeness, answer correctness and noise sensitivity need a generative model. The harness runs offline with only the pinned embedding model — the same constraint that keeps the reranker unmeasured (#170) — so this PR does not add them and does not fake them: the report's Definitions section now says so, and the "Not evaluated" line in `eval-retrieval.mjs` points at the same constraint. Adding an LLM judge is a separate change that has to solve model pinning first, not a line of code. `evidenceK` is removed from the harness config, so there is exactly one context width. ## Testing - `npm run typecheck` — clean - `npm test` — 497 pass - `npm run eval` twice — byte-identical `docs/eval/baseline-v1.6.json` - `npm run eval:retrieval` — regenerated; hybrid still clears the amended rule Part of #192 (child 5, deterministic half). --- docs/eval/baseline-v1.6.json | 19 ++++++++++++------- docs/eval/baseline-v1.6.md | 24 +++++++++++++++--------- docs/eval/retrieval-v1.6.json | 21 ++++++++++++--------- docs/eval/retrieval-v1.6.md | 8 ++++---- eval/README.md | 16 +++++++++++----- scripts/eval-chunking.mjs | 4 ++-- scripts/eval-retrieval.mjs | 4 ++-- src/main/eval/harness.ts | 21 ++++++++++++--------- src/main/eval/report.ts | 20 +++++++++++++------- src/main/eval/run.ts | 1 - src/main/eval/types.ts | 22 +++++++++++++--------- 11 files changed, 96 insertions(+), 64 deletions(-) diff --git a/docs/eval/baseline-v1.6.json b/docs/eval/baseline-v1.6.json index cfb0455..e598a6c 100644 --- a/docs/eval/baseline-v1.6.json +++ b/docs/eval/baseline-v1.6.json @@ -15,7 +15,6 @@ "candidateK": 20, "contextK": 3, "threshold": 0.5, - "evidenceK": 5, "corpus": "eval/corpus", "documents": 13, "questions": 30, @@ -29,7 +28,8 @@ "ndcgAt10": 0.94375, "hitRateAt5": 1, "mapAt10": 0.922222, - "evidencePrecisionAt5": 0.213333 + "contextPrecision": 0.355556, + "contextRecall": 1 }, "byType": [ { @@ -43,7 +43,8 @@ "ndcgAt10": 0.63093, "hitRateAt5": 1, "mapAt10": 0.5, - "evidencePrecisionAt5": 0.2 + "contextPrecision": 0.333333, + "contextRecall": 1 } }, { @@ -57,7 +58,8 @@ "ndcgAt10": 1, "hitRateAt5": 1, "mapAt10": 1, - "evidencePrecisionAt5": 0.2 + "contextPrecision": 0.333333, + "contextRecall": 1 } }, { @@ -71,7 +73,8 @@ "ndcgAt10": 0.95986, "hitRateAt5": 1, "mapAt10": 0.916667, - "evidencePrecisionAt5": 0.4 + "contextPrecision": 0.666667, + "contextRecall": 1 } }, { @@ -85,7 +88,8 @@ "ndcgAt10": 0.931214, "hitRateAt5": 1, "mapAt10": 0.907407, - "evidencePrecisionAt5": 0.2 + "contextPrecision": 0.333333, + "contextRecall": 1 } }, { @@ -99,7 +103,8 @@ "ndcgAt10": 1, "hitRateAt5": 1, "mapAt10": 1, - "evidencePrecisionAt5": 0.2 + "contextPrecision": 0.333333, + "contextRecall": 1 } } ], diff --git a/docs/eval/baseline-v1.6.md b/docs/eval/baseline-v1.6.md index ce25b05..467fafb 100644 --- a/docs/eval/baseline-v1.6.md +++ b/docs/eval/baseline-v1.6.md @@ -11,7 +11,6 @@ Generated by `npm run eval`. The numbers below are harness output — do not edi | Retrieval | `dense` | | Ranks | `candidateK=20, threshold=0.5` | | Context width | `contextK=3` | -| Evidence per query | `evidenceK=5` | | Corpus | `eval/corpus` (13 documents) | | Split | `all` (30 questions) | | Index size | 19 chunks | @@ -27,7 +26,8 @@ Generated by `npm run eval`. The numbers below are harness output — do not edi | nDCG@10 | 0.9437 | | Hit rate@5 | 1.0000 | | MAP@10 | 0.9222 | -| Evidence precision@5 | 0.2133 | +| Context precision@3 | 0.3556 | +| Context recall@3 | 1.0000 | ### By query type @@ -43,8 +43,8 @@ The type comes from `type` in `questions.jsonl`; untagged questions report as | semantic | 18 | 1.0000 | 0.9312 | 1.0000 | 0.9074 | | zh | 3 | 1.0000 | 1.0000 | 1.0000 | 1.0000 | -Timing is informational only and is **not** frozen: indexing 1551 ms, query -p50 11.73 ms, p95 18.71 ms on the +Timing is informational only and is **not** frozen: indexing 1499 ms, query +p50 11.58 ms, p95 16.69 ms on the machine that produced this file. Timing and index size depend on hardware and on the corpus, so they must never be the reason two runs differ. @@ -52,11 +52,17 @@ corpus, so they must never be the reason two runs differ. - A retrieved passage is relevant when its provenance covers a ground-truth block. - **Recall@k** is the share of ground-truth blocks covered by the first `k` passages. -- **Evidence precision@5** is the share of the first `5` - retrieved passages that cover a ground-truth block. This is **retrieval precision**, not - answer citation recall: the harness runs no model and produces no answer. Answer-level - citation correctness is covered by the resolver (#70); a model-driven answer eval is a - separate deliverable. +- **Context precision@3** is the share of the first `3` + retrieved passages that cover a ground-truth block. **Context recall@3** is + the share of the needed ground-truth blocks that made it into that same window. Both are + deterministic: the dataset says which blocks answer the question, so no model is needed to + score the window. Together they are the trade-off a `contextK` decision actually makes — + a wider window finds more and carries more noise. +- This is **retrieval precision/recall, not answer citation recall**: the harness runs no + model and produces no answer. Answer-level citation correctness is covered by the + resolver (#70). Faithfulness, completeness and answer correctness need a generative model + and are **not evaluated here** — the harness runs offline with only the pinned embedding + model, the same constraint that keeps the reranker unmeasured (#170). - Ground truth is expressed in corpus identity (`document` relative path + `block` ordinal + optional `quote`), never a runtime `documentId`/`blockId`. diff --git a/docs/eval/retrieval-v1.6.json b/docs/eval/retrieval-v1.6.json index 8fed18f..6782582 100644 --- a/docs/eval/retrieval-v1.6.json +++ b/docs/eval/retrieval-v1.6.json @@ -14,9 +14,10 @@ "ndcgAt10": 0.94375, "hitRateAt5": 1, "mapAt10": 0.922222, - "evidencePrecisionAt5": 0.213333, - "indexingMs": 1531, - "latencyP95Ms": 14.81 + "contextPrecision": 0.355556, + "contextRecall": 1, + "indexingMs": 1548, + "latencyP95Ms": 20.51 }, { "id": "sparse", @@ -30,9 +31,10 @@ "ndcgAt10": 0.818355, "hitRateAt5": 0.866667, "mapAt10": 0.8, - "evidencePrecisionAt5": 0.186667, - "indexingMs": 1503, - "latencyP95Ms": 1.89 + "contextPrecision": 0.311111, + "contextRecall": 0.866667, + "indexingMs": 1490, + "latencyP95Ms": 1.96 }, { "id": "hybrid", @@ -46,9 +48,10 @@ "ndcgAt10": 0.956053, "hitRateAt5": 1, "mapAt10": 0.938889, - "evidencePrecisionAt5": 0.213333, - "indexingMs": 1548, - "latencyP95Ms": 19.15 + "contextPrecision": 0.355556, + "contextRecall": 1, + "indexingMs": 1507, + "latencyP95Ms": 20.66 } ] } diff --git a/docs/eval/retrieval-v1.6.md b/docs/eval/retrieval-v1.6.md index 9651e44..c13d4e4 100644 --- a/docs/eval/retrieval-v1.6.md +++ b/docs/eval/retrieval-v1.6.md @@ -7,11 +7,11 @@ Generated by `node scripts/eval-retrieval.mjs`. Numbers are harness output; do n Every strategy runs the real RAG eval harness against the same corpus and the same 30 questions as `baseline-v1.6.json`, with chunking held fixed at 1000/100. Only the retrieval strategy changes. -| Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Evidence P@5 | Query p95 | +| Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Context P | Query p95 | | --- | --- | --- | --- | --- | --- | --- | --- | -| dense (vector) | 0.8333 | 1.0000 | 0.9278 | 0.9437 | 0.9222 | 0.2133 | 14.81 ms | -| sparse (BM25) | 0.7333 | 0.8667 | 0.8056 | 0.8184 | 0.8000 | 0.1867 | 1.89 ms | -| hybrid (RRF of dense + BM25) | 0.8667 | 1.0000 | 0.9444 | 0.9561 | 0.9389 | 0.2133 | 19.15 ms | +| dense (vector) | 0.8333 | 1.0000 | 0.9278 | 0.9437 | 0.9222 | 0.3556 | 20.51 ms | +| sparse (BM25) | 0.7333 | 0.8667 | 0.8056 | 0.8184 | 0.8000 | 0.3111 | 1.96 ms | +| hybrid (RRF of dense + BM25) | 0.8667 | 1.0000 | 0.9444 | 0.9561 | 0.9389 | 0.3556 | 20.66 ms | ## Not evaluated diff --git a/eval/README.md b/eval/README.md index 2333a26..913073e 100644 --- a/eval/README.md +++ b/eval/README.md @@ -135,11 +135,17 @@ from unanswerable queries. was *complete*. A two-passage question that finds one scores 1.0 and 0.5 respectively. - **MAP@10** — mean average precision. The one metric here that combines ranking position with coverage, so pulling a second relevant passage from rank 9 to rank 2 moves it. -- **Evidence precision@5** — of the first 5 retrieved passages, the share that cover a - ground-truth block. This is **retrieval precision, not answer citation recall**: the - harness runs no model and produces no answer. Answer-level citation correctness is - covered by the resolver (#70); a model-driven answer eval would be a separate - deliverable. +- **Context precision@`contextK`** — of the first `contextK` retrieved passages, the share + that cover a ground-truth block. **Context recall@`contextK`** — the share of the needed + ground-truth blocks that made it into that same window. Both are deterministic: the + dataset says which blocks answer the question, so no model is needed to score the window. + Together they are the trade-off a `contextK` decision actually makes — a wider window + finds more and carries more noise. +- This is **retrieval precision/recall, not answer citation recall**: the harness runs no + model and produces no answer. Answer-level citation correctness is covered by the + resolver (#70). Faithfulness, completeness and answer correctness need a generative model + and are **not evaluated here** — the harness runs offline with only the pinned embedding + model, the same constraint that keeps the reranker unmeasured (#170). - **By query type** — the same metrics per `type` in `questions.jsonl` (`exact`, `semantic`, `multi-hop`, `cross-lingual`, `zh`). A single average hides a change that helps one kind of question and hurts another; the current baseline already shows this, diff --git a/scripts/eval-chunking.mjs b/scripts/eval-chunking.mjs index 70f4e45..52bebbf 100644 --- a/scripts/eval-chunking.mjs +++ b/scripts/eval-chunking.mjs @@ -187,7 +187,7 @@ const format4 = (value) => value.toFixed(4) const rows = results.map((result) => { const args = `${result.chunkSize}/${result.chunkOverlap}${result.respectHeadings ? ' + headings' : ''}${result.allowSpanPages ? ' + span' : ''}` - return `| ${result.label} | \`${args}\` | ${format4(result.recallAt1)} | ${format4(result.recallAt5)} | ${format4(result.mrr)} | ${format4(result.ndcgAt10)} | ${format4(result.evidencePrecisionAt5)} | ${result.chunkCount} (${formatDelta(delta(result.chunkCount, baseline.chunkCount))}) | ${result.indexingMs} ms | ${result.latencyP95Ms?.toFixed(2)} ms |` + return `| ${result.label} | \`${args}\` | ${format4(result.recallAt1)} | ${format4(result.recallAt5)} | ${format4(result.mrr)} | ${format4(result.ndcgAt10)} | ${format4(result.contextPrecision)} | ${result.chunkCount} (${formatDelta(delta(result.chunkCount, baseline.chunkCount))}) | ${result.indexingMs} ms | ${result.latencyP95Ms?.toFixed(2)} ms |` }) /** @@ -215,7 +215,7 @@ questions as \`baseline-v1.4.json\`, with dense retrieval held fixed. Only the c configuration changes, so a difference in the metrics is a difference in the input distribution retrieval is measured on. -| Variant | size/overlap | Recall@1 | Recall@5 | MRR | nDCG@10 | Evidence P@5 | Index size (Δ) | Indexing | Query p95 | +| Variant | size/overlap | Recall@1 | Recall@5 | MRR | nDCG@10 | Context P | Index size (Δ) | Indexing | Query p95 | | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | ${rows.join('\n')} diff --git a/scripts/eval-retrieval.mjs b/scripts/eval-retrieval.mjs index d52b5bd..5964a28 100644 --- a/scripts/eval-retrieval.mjs +++ b/scripts/eval-retrieval.mjs @@ -141,7 +141,7 @@ const format4 = (value) => value.toFixed(4) const rows = results.map( (result) => - `| ${result.label} | ${format4(result.recallAt1)} | ${format4(result.recallAt5)} | ${format4(result.mrr)} | ${format4(result.ndcgAt10)} | ${format4(result.mapAt10)} | ${format4(result.evidencePrecisionAt5)} | ${result.latencyP95Ms?.toFixed(2)} ms |` + `| ${result.label} | ${format4(result.recallAt1)} | ${format4(result.recallAt5)} | ${format4(result.mrr)} | ${format4(result.ndcgAt10)} | ${format4(result.mapAt10)} | ${format4(result.contextPrecision)} | ${result.latencyP95Ms?.toFixed(2)} ms |` ) /** @@ -179,7 +179,7 @@ Generated by \`node scripts/eval-retrieval.mjs\`. Numbers are harness output; do Every strategy runs the real RAG eval harness against the same corpus and the same 30 questions as \`baseline-v1.6.json\`, with chunking held fixed at ${baseline.chunking}. Only the retrieval strategy changes. -| Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Evidence P@5 | Query p95 | +| Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Context P | Query p95 | | --- | --- | --- | --- | --- | --- | --- | --- | ${rows.join('\n')} diff --git a/src/main/eval/harness.ts b/src/main/eval/harness.ts index 7ae5114..c33c43d 100644 --- a/src/main/eval/harness.ts +++ b/src/main/eval/harness.ts @@ -67,8 +67,6 @@ export interface EvalHarnessOptions { contextK: number /** Similarity floor; 0 keeps the ranking intact for ranking metrics. */ threshold: number - /** How many retrieved passages the evidence-precision metric looks at. */ - evidenceK: number /** 分块配置(#78)。实验变体通过它选择策略;缺省时用生产默认值。 */ chunkOptions: ChunkOptions /** 检索策略(#77):dense / sparse(BM25) / hybrid(RRF)。 */ @@ -84,7 +82,7 @@ const UNTAGGED = 'untagged' * 同一套指标既算总平均,也算每个查询类别(#192)。用一个函数是因为分组平均必须与 * 总平均是同一个定义,否则两个数就不可比。 */ -function summarize(perQuestion: readonly QuestionReport[], evidenceK: number): EvalMetrics { +function summarize(perQuestion: readonly QuestionReport[], contextK: number): EvalMetrics { return { recallAt1: mean(perQuestion.map((q) => recallAtK(q.matchesByRank, q.relevantCount, 1))), recallAt5: mean(perQuestion.map((q) => recallAtK(q.matchesByRank, q.relevantCount, 5))), @@ -95,8 +93,13 @@ function summarize(perQuestion: readonly QuestionReport[], evidenceK: number): E mapAt10: mean( perQuestion.map((q) => averagePrecisionAtK(q.matchesByRank, q.relevantCount, 10)) ), - evidencePrecisionAt5: mean( - perQuestion.map((q) => evidencePrecisionAtK(q.matchesByRank, evidenceK)) + // 两个 context 指标共用同一个窗口,因为它们回答的是同一个问题的两面:送进 prompt + // 的那几条里有多少是相关的,以及需要的东西有多少真的进去了。 + contextPrecision: mean( + perQuestion.map((q) => evidencePrecisionAtK(q.matchesByRank, contextK)) + ), + contextRecall: mean( + perQuestion.map((q) => recallAtK(q.matchesByRank, q.relevantCount, contextK)) ) } } @@ -266,14 +269,14 @@ export async function runEvalHarness( }) } - const metrics = summarize(perQuestion, options.evidenceK) + const metrics = summarize(perQuestion, options.contextK) // 每个类别一行,按类别名排序,所以同一个 JSON 在两次运行之间可 diff。 const byType: EvalTypeBreakdown[] = [...new Set(perQuestion.map((q) => q.type))] .sort() .map((type) => { const group = perQuestion.filter((q) => q.type === type) - return { type, questions: group.length, metrics: summarize(group, options.evidenceK) } + return { type, questions: group.length, metrics: summarize(group, options.contextK) } }) const chunking = { ...DEFAULT_CHUNK_OPTIONS, ...options.chunkOptions } @@ -294,7 +297,6 @@ export async function runEvalHarness( candidateK: options.candidateK, contextK: options.contextK, threshold: options.threshold, - evidenceK: options.evidenceK, corpus: options.corpusLabel, documents: documentIds.size, questions: questions.length, @@ -323,7 +325,8 @@ function roundMetrics(metrics: EvalMetrics): EvalMetrics { ndcgAt10: roundMetric(metrics.ndcgAt10), hitRateAt5: roundMetric(metrics.hitRateAt5), mapAt10: roundMetric(metrics.mapAt10), - evidencePrecisionAt5: roundMetric(metrics.evidencePrecisionAt5) + contextPrecision: roundMetric(metrics.contextPrecision), + contextRecall: roundMetric(metrics.contextRecall) } } diff --git a/src/main/eval/report.ts b/src/main/eval/report.ts index 7b883f8..8f6dfaa 100644 --- a/src/main/eval/report.ts +++ b/src/main/eval/report.ts @@ -37,7 +37,6 @@ Generated by \`${report.generatedBy}\`. The numbers below are harness output — | Retrieval | \`${config.retrieval}\` | | Ranks | \`candidateK=${config.candidateK}, threshold=${config.threshold}\` | | Context width | \`contextK=${config.contextK}\` | -| Evidence per query | \`evidenceK=${config.evidenceK}\` | | Corpus | \`${config.corpus}\` (${config.documents} documents) | | Split | \`${config.split}\` (${config.questions} questions) | | Index size | ${config.chunkCount} chunks | @@ -53,7 +52,8 @@ Generated by \`${report.generatedBy}\`. The numbers below are harness output — | nDCG@10 | ${format(metrics.ndcgAt10)} | | Hit rate@5 | ${format(metrics.hitRateAt5)} | | MAP@10 | ${format(metrics.mapAt10)} | -| Evidence precision@${config.evidenceK} | ${format(metrics.evidencePrecisionAt5)} | +| Context precision@${config.contextK} | ${format(metrics.contextPrecision)} | +| Context recall@${config.contextK} | ${format(metrics.contextRecall)} | ### By query type @@ -74,11 +74,17 @@ corpus, so they must never be the reason two runs differ. - A retrieved passage is relevant when its provenance covers a ground-truth block. - **Recall@k** is the share of ground-truth blocks covered by the first \`k\` passages. -- **Evidence precision@${config.evidenceK}** is the share of the first \`${config.evidenceK}\` - retrieved passages that cover a ground-truth block. This is **retrieval precision**, not - answer citation recall: the harness runs no model and produces no answer. Answer-level - citation correctness is covered by the resolver (#70); a model-driven answer eval is a - separate deliverable. +- **Context precision@${config.contextK}** is the share of the first \`${config.contextK}\` + retrieved passages that cover a ground-truth block. **Context recall@${config.contextK}** is + the share of the needed ground-truth blocks that made it into that same window. Both are + deterministic: the dataset says which blocks answer the question, so no model is needed to + score the window. Together they are the trade-off a \`contextK\` decision actually makes — + a wider window finds more and carries more noise. +- This is **retrieval precision/recall, not answer citation recall**: the harness runs no + model and produces no answer. Answer-level citation correctness is covered by the + resolver (#70). Faithfulness, completeness and answer correctness need a generative model + and are **not evaluated here** — the harness runs offline with only the pinned embedding + model, the same constraint that keeps the reranker unmeasured (#170). - Ground truth is expressed in corpus identity (\`document\` relative path + \`block\` ordinal + optional \`quote\`), never a runtime \`documentId\`/\`blockId\`. diff --git a/src/main/eval/run.ts b/src/main/eval/run.ts index 802cf5e..d927286 100644 --- a/src/main/eval/run.ts +++ b/src/main/eval/run.ts @@ -197,7 +197,6 @@ export async function runEvalCli(argv: readonly string[] = process.argv): Promis candidateK: readNumberOption(argv, '--eval-candidate-k=', 20), contextK: readNumberOption(argv, '--eval-context-k=', 3), threshold: readNumberOption(argv, '--eval-threshold=', 0.5), - evidenceK: 5, chunkOptions: readChunkOptions(argv), strategy: readRetrievalStrategy(argv) } diff --git a/src/main/eval/types.ts b/src/main/eval/types.ts index 3a075f4..8833a23 100644 --- a/src/main/eval/types.ts +++ b/src/main/eval/types.ts @@ -82,13 +82,19 @@ export interface EvalMetrics { /** AP@10:把「排序位置」和「覆盖面」合成一个数的那个指标。 */ mapAt10: number /** - * Share of the first `evidenceK` retrieved passages that cover ground truth. + * 前 `contextK` 条证据里真的命中 ground-truth 的比例(context 条的精确率)。 * - * This is **retrieval precision**, not answer citation recall: no model runs in - * this harness and no answer is produced. Answer-level citation correctness is - * the resolver's job (#70) and would need a separate, model-driven eval. + * 这是**检索精度**,不是回答的引用召回:harness 不跑模型、不产生回答。回答层的引用 + * 正确性是 resolver 的事(#70),需要一个真正跑模型的 eval。 */ - evidencePrecisionAt5: number + contextPrecision: number + /** + * 答案需要的 ground-truth 块有多少进了 `contextK` 宽的窗口。 + * + * 与 Recall@10 的区别在于它量的是**窗口**:证据排在第 4、而 contextK=3 时,模型 + * 看不到它。这是一个产品指标,不只是检索指标。 + */ + contextRecall: number } /** 按查询类别聚合的同一套指标(#192)。总平均会掩盖方向相反的两个变化。 */ @@ -134,13 +140,11 @@ export interface EvalReport { /** * 生产 prompt 实际取用的证据条数(#77)。 * - * 快照里记它是为了让 benchmark 描述整条线上链路,而不只是检索器;它不影响排名 - * 指标 —— 截断只是取候选列表的前缀,前缀的排序不变。 + * 快照里记它是为了让 benchmark 描述整条线上链路,而不只是检索器;它也是两个 + * context 指标的窗口宽度。 */ contextK: number threshold: number - /** How many retrieved passages the evidence-precision metric looks at. */ - evidenceK: number corpus: string documents: number questions: number From 27fdda27ae860b1c0953fd18ca66730ed32a5ce2 Mon Sep 17 00:00:00 2001 From: MrSibe Date: Wed, 30 Sep 2026 17:08:04 +0800 Subject: [PATCH 07/18] feat(eval): bounded parameter sweep with a trade-off dashboard MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The epic asks for a grid over `candidateK` and `contextK` with latency, index size and context size recorded next to quality, and explicitly not for a single aggregate "RAG score". Child 7 of #192. `npm run eval:sweep` runs the real harness over `strategy × candidateK {5,10,20,40} × contextK {3,5,8}` (24 runs, chunking fixed) and writes `docs/eval/sweep-v1.6.{json,md}`. The harness gained `contextChars` per question — the size of the context window, reported as **characters, not tokens**, because the harness pins an embedding model and no generation tokenizer. ## What the dashboard already shows Two of the three axes are decided by this corpus, and one of them decisively: - **`candidateK` changes nothing.** Every metric is identical from 5 to 40, because the corpus is 19 chunks and the relevant passages are already inside the top 5. This is the saturation problem #192 describes, now visible on the axis it affects. - **`contextK` has a clear optimum here: 3.** Context recall is 1.0000 at every width, while precision falls `0.3556 → 0.2133 → 0.1333` and the window grows `2079 → 3327 → 5312` characters as it widens. Wider adds prompt cost and noise for **no** recall. The production default is already 3, and this is the first evidence that it is the right 3 rather than a guess. - **hybrid beats dense** on nDCG@10 (0.9561 vs 0.9437) and MAP@10 (0.9389 vs 0.9222) at every setting, with no metric regressing — consistent with #197. The report labels the grid maximum as **not** a recommendation, because selecting on the same questions is how a benchmark becomes a lookup table; the adoption rule and the `validation`/`test` split are what keep the decision honest. Timing p95 is reported per row but is noisy at this sample size; it is in the table so a latency cost can be seen, not so it can be ranked. ## Testing - `npm run typecheck` — clean - `npm test` — 497 pass - `npm run eval` — regenerated (`contextChars` added to `perQuestion`) - `npm run eval:sweep` — 24 runs, dashboard written Part of #192 (child 7). --- .prettierignore | 5 + docs/eval/baseline-v1.6.json | 30 +++ docs/eval/baseline-v1.6.md | 4 +- docs/eval/sweep-v1.6.json | 341 +++++++++++++++++++++++++++++++++++ docs/eval/sweep-v1.6.md | 64 +++++++ eval/README.md | 6 + package.json | 3 +- scripts/eval-sweep.mjs | 209 +++++++++++++++++++++ src/main/eval/harness.ts | 3 + src/main/eval/types.ts | 7 + 10 files changed, 669 insertions(+), 3 deletions(-) create mode 100644 docs/eval/sweep-v1.6.json create mode 100644 docs/eval/sweep-v1.6.md create mode 100644 scripts/eval-sweep.mjs diff --git a/.prettierignore b/.prettierignore index ae11386..f6f4ae5 100644 --- a/.prettierignore +++ b/.prettierignore @@ -24,6 +24,11 @@ docs/eval/chunking-*.md docs/eval/threshold-*.json docs/eval/threshold-*.md +# Same reason again: generated by `scripts/eval-sweep.mjs`. Regenerate with +# `npm run eval:sweep`. +docs/eval/sweep-*.json +docs/eval/sweep-*.md + # And the same again for `scripts/eval-retrieval.mjs` (#77). docs/eval/retrieval-*.json docs/eval/retrieval-*.md diff --git a/docs/eval/baseline-v1.6.json b/docs/eval/baseline-v1.6.json index e598a6c..43b77d7 100644 --- a/docs/eval/baseline-v1.6.json +++ b/docs/eval/baseline-v1.6.json @@ -116,6 +116,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2074, "matchesByRank": [ [ 0 @@ -147,6 +148,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2257, "matchesByRank": [ [ 0 @@ -178,6 +180,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2558, "matchesByRank": [ [ 0 @@ -209,6 +212,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2113, "matchesByRank": [ [ 0 @@ -240,6 +244,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2113, "matchesByRank": [ [ 0 @@ -271,6 +276,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 1775, "matchesByRank": [ [ 0 @@ -302,6 +308,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2091, "matchesByRank": [ [ 0 @@ -333,6 +340,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 1970, "matchesByRank": [ [ 0 @@ -364,6 +372,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 1484, "matchesByRank": [ [ 0 @@ -395,6 +404,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2682, "matchesByRank": [ [ 0 @@ -426,6 +436,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2051, "matchesByRank": [ [ 0 @@ -457,6 +468,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2007, "matchesByRank": [ [ 0 @@ -488,6 +500,7 @@ "firstRelevantRank": 3, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2007, "matchesByRank": [ [], [], @@ -519,6 +532,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2257, "matchesByRank": [ [ 0 @@ -550,6 +564,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2621, "matchesByRank": [ [ 0 @@ -581,6 +596,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2558, "matchesByRank": [ [ 0 @@ -612,6 +628,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 1773, "matchesByRank": [ [ 0 @@ -643,6 +660,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 1970, "matchesByRank": [ [ 0 @@ -674,6 +692,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2556, "matchesByRank": [ [ 0 @@ -705,6 +724,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2098, "matchesByRank": [ [ 0 @@ -736,6 +756,7 @@ "firstRelevantRank": 2, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2488, "matchesByRank": [ [], [ @@ -767,6 +788,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2488, "matchesByRank": [ [ 0 @@ -798,6 +820,7 @@ "firstRelevantRank": 1, "relevantCount": 2, "retrievedCount": 19, + "contextChars": 2257, "matchesByRank": [ [ 1 @@ -831,6 +854,7 @@ "firstRelevantRank": 1, "relevantCount": 2, "retrievedCount": 19, + "contextChars": 1970, "matchesByRank": [ [ 0 @@ -864,6 +888,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2098, "matchesByRank": [ [ 0 @@ -895,6 +920,7 @@ "firstRelevantRank": 2, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 1867, "matchesByRank": [ [], [ @@ -926,6 +952,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 1453, "matchesByRank": [ [ 0 @@ -957,6 +984,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 1372, "matchesByRank": [ [ 0 @@ -988,6 +1016,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 1500, "matchesByRank": [ [ 0 @@ -1019,6 +1048,7 @@ "firstRelevantRank": 2, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 1860, "matchesByRank": [ [], [ diff --git a/docs/eval/baseline-v1.6.md b/docs/eval/baseline-v1.6.md index 467fafb..5c7fc85 100644 --- a/docs/eval/baseline-v1.6.md +++ b/docs/eval/baseline-v1.6.md @@ -43,8 +43,8 @@ The type comes from `type` in `questions.jsonl`; untagged questions report as | semantic | 18 | 1.0000 | 0.9312 | 1.0000 | 0.9074 | | zh | 3 | 1.0000 | 1.0000 | 1.0000 | 1.0000 | -Timing is informational only and is **not** frozen: indexing 1499 ms, query -p50 11.58 ms, p95 16.69 ms on the +Timing is informational only and is **not** frozen: indexing 2015 ms, query +p50 71.63 ms, p95 94.77 ms on the machine that produced this file. Timing and index size depend on hardware and on the corpus, so they must never be the reason two runs differ. diff --git a/docs/eval/sweep-v1.6.json b/docs/eval/sweep-v1.6.json new file mode 100644 index 0000000..7f7fce1 --- /dev/null +++ b/docs/eval/sweep-v1.6.json @@ -0,0 +1,341 @@ +{ + "baseline": "v1.6", + "rows": [ + { + "strategy": "dense", + "candidateK": 5, + "contextK": 3, + "recallAt5": 1, + "ndcgAt10": 0.94375, + "mapAt10": 0.922222, + "contextPrecision": 0.355556, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 2078.9333333333334, + "chunkCount": 19, + "latencyP95Ms": 88.15 + }, + { + "strategy": "dense", + "candidateK": 5, + "contextK": 5, + "recallAt5": 1, + "ndcgAt10": 0.94375, + "mapAt10": 0.922222, + "contextPrecision": 0.213333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 3326.8333333333335, + "chunkCount": 19, + "latencyP95Ms": 89.14 + }, + { + "strategy": "dense", + "candidateK": 5, + "contextK": 8, + "recallAt5": 1, + "ndcgAt10": 0.94375, + "mapAt10": 0.922222, + "contextPrecision": 0.213333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 3326.8333333333335, + "chunkCount": 19, + "latencyP95Ms": 98.4 + }, + { + "strategy": "dense", + "candidateK": 10, + "contextK": 3, + "recallAt5": 1, + "ndcgAt10": 0.94375, + "mapAt10": 0.922222, + "contextPrecision": 0.355556, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 2078.9333333333334, + "chunkCount": 19, + "latencyP95Ms": 107.2 + }, + { + "strategy": "dense", + "candidateK": 10, + "contextK": 5, + "recallAt5": 1, + "ndcgAt10": 0.94375, + "mapAt10": 0.922222, + "contextPrecision": 0.213333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 3326.8333333333335, + "chunkCount": 19, + "latencyP95Ms": 77.76 + }, + { + "strategy": "dense", + "candidateK": 10, + "contextK": 8, + "recallAt5": 1, + "ndcgAt10": 0.94375, + "mapAt10": 0.922222, + "contextPrecision": 0.133333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 5311.633333333333, + "chunkCount": 19, + "latencyP95Ms": 91.92 + }, + { + "strategy": "dense", + "candidateK": 20, + "contextK": 3, + "recallAt5": 1, + "ndcgAt10": 0.94375, + "mapAt10": 0.922222, + "contextPrecision": 0.355556, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 2078.9333333333334, + "chunkCount": 19, + "latencyP95Ms": 94.88 + }, + { + "strategy": "dense", + "candidateK": 20, + "contextK": 5, + "recallAt5": 1, + "ndcgAt10": 0.94375, + "mapAt10": 0.922222, + "contextPrecision": 0.213333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 3326.8333333333335, + "chunkCount": 19, + "latencyP95Ms": 85.49 + }, + { + "strategy": "dense", + "candidateK": 20, + "contextK": 8, + "recallAt5": 1, + "ndcgAt10": 0.94375, + "mapAt10": 0.922222, + "contextPrecision": 0.133333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 5311.633333333333, + "chunkCount": 19, + "latencyP95Ms": 84.23 + }, + { + "strategy": "dense", + "candidateK": 40, + "contextK": 3, + "recallAt5": 1, + "ndcgAt10": 0.94375, + "mapAt10": 0.922222, + "contextPrecision": 0.355556, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 2078.9333333333334, + "chunkCount": 19, + "latencyP95Ms": 91.85 + }, + { + "strategy": "dense", + "candidateK": 40, + "contextK": 5, + "recallAt5": 1, + "ndcgAt10": 0.94375, + "mapAt10": 0.922222, + "contextPrecision": 0.213333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 3326.8333333333335, + "chunkCount": 19, + "latencyP95Ms": 86.76 + }, + { + "strategy": "dense", + "candidateK": 40, + "contextK": 8, + "recallAt5": 1, + "ndcgAt10": 0.94375, + "mapAt10": 0.922222, + "contextPrecision": 0.133333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 5311.633333333333, + "chunkCount": 19, + "latencyP95Ms": 89.21 + }, + { + "strategy": "hybrid", + "candidateK": 5, + "contextK": 3, + "recallAt5": 1, + "ndcgAt10": 0.956053, + "mapAt10": 0.938889, + "contextPrecision": 0.355556, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 2044.1333333333334, + "chunkCount": 19, + "latencyP95Ms": 81.29 + }, + { + "strategy": "hybrid", + "candidateK": 5, + "contextK": 5, + "recallAt5": 1, + "ndcgAt10": 0.956053, + "mapAt10": 0.938889, + "contextPrecision": 0.213333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 3361.6, + "chunkCount": 19, + "latencyP95Ms": 79.52 + }, + { + "strategy": "hybrid", + "candidateK": 5, + "contextK": 8, + "recallAt5": 1, + "ndcgAt10": 0.956053, + "mapAt10": 0.938889, + "contextPrecision": 0.213333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 3361.6, + "chunkCount": 19, + "latencyP95Ms": 85.98 + }, + { + "strategy": "hybrid", + "candidateK": 10, + "contextK": 3, + "recallAt5": 1, + "ndcgAt10": 0.956053, + "mapAt10": 0.938889, + "contextPrecision": 0.355556, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 2062.733333333333, + "chunkCount": 19, + "latencyP95Ms": 86.15 + }, + { + "strategy": "hybrid", + "candidateK": 10, + "contextK": 5, + "recallAt5": 1, + "ndcgAt10": 0.956053, + "mapAt10": 0.938889, + "contextPrecision": 0.213333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 3525.633333333333, + "chunkCount": 19, + "latencyP95Ms": 92.74 + }, + { + "strategy": "hybrid", + "candidateK": 10, + "contextK": 8, + "recallAt5": 1, + "ndcgAt10": 0.956053, + "mapAt10": 0.938889, + "contextPrecision": 0.133333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 5509.8, + "chunkCount": 19, + "latencyP95Ms": 89.97 + }, + { + "strategy": "hybrid", + "candidateK": 20, + "contextK": 3, + "recallAt5": 1, + "ndcgAt10": 0.956053, + "mapAt10": 0.938889, + "contextPrecision": 0.355556, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 2056.9333333333334, + "chunkCount": 19, + "latencyP95Ms": 59.56 + }, + { + "strategy": "hybrid", + "candidateK": 20, + "contextK": 5, + "recallAt5": 1, + "ndcgAt10": 0.956053, + "mapAt10": 0.938889, + "contextPrecision": 0.213333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 3445.866666666667, + "chunkCount": 19, + "latencyP95Ms": 77.17 + }, + { + "strategy": "hybrid", + "candidateK": 20, + "contextK": 8, + "recallAt5": 1, + "ndcgAt10": 0.956053, + "mapAt10": 0.938889, + "contextPrecision": 0.133333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 5572.6, + "chunkCount": 19, + "latencyP95Ms": 91.41 + }, + { + "strategy": "hybrid", + "candidateK": 40, + "contextK": 3, + "recallAt5": 1, + "ndcgAt10": 0.956053, + "mapAt10": 0.938889, + "contextPrecision": 0.355556, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 2056.9333333333334, + "chunkCount": 19, + "latencyP95Ms": 89.62 + }, + { + "strategy": "hybrid", + "candidateK": 40, + "contextK": 5, + "recallAt5": 1, + "ndcgAt10": 0.956053, + "mapAt10": 0.938889, + "contextPrecision": 0.213333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 3445.866666666667, + "chunkCount": 19, + "latencyP95Ms": 66.45 + }, + { + "strategy": "hybrid", + "candidateK": 40, + "contextK": 8, + "recallAt5": 1, + "ndcgAt10": 0.956053, + "mapAt10": 0.938889, + "contextPrecision": 0.133333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 5572.6, + "chunkCount": 19, + "latencyP95Ms": 76.65 + } + ] +} diff --git a/docs/eval/sweep-v1.6.md b/docs/eval/sweep-v1.6.md new file mode 100644 index 0000000..2168fa4 --- /dev/null +++ b/docs/eval/sweep-v1.6.md @@ -0,0 +1,64 @@ +# Parameter sweep — v1.6 (#192) + +Generated by `node scripts/eval-sweep.mjs`. Numbers are harness output; do not edit them by hand. + +## What was measured + +The real harness, the same corpus, chunking held fixed, over +dense / hybrid × candidateK {5, 10, 20, 40} × contextK {3, 5, 8} — 24 runs. +Each row differs from its neighbour in one parameter. + +| Strategy | candidateK | contextK | Recall@5 | nDCG@10 | MAP@10 | Context P | Context R | No-result | Context chars | Index | p95 | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | +| dense | 5 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 19 | 88.15 ms | +| dense | 5 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 19 | 89.14 ms | +| dense | 5 | 8 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 19 | 98.40 ms | +| dense | 10 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 19 | 107.20 ms | +| dense | 10 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 19 | 77.76 ms | +| dense | 10 | 8 | 1.0000 | 0.9437 | 0.9222 | 0.1333 | 1.0000 | 0.0000 | 5312 | 19 | 91.92 ms | +| dense | 20 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 19 | 94.88 ms | +| dense | 20 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 19 | 85.49 ms | +| dense | 20 | 8 | 1.0000 | 0.9437 | 0.9222 | 0.1333 | 1.0000 | 0.0000 | 5312 | 19 | 84.23 ms | +| dense | 40 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 19 | 91.85 ms | +| dense | 40 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 19 | 86.76 ms | +| dense | 40 | 8 | 1.0000 | 0.9437 | 0.9222 | 0.1333 | 1.0000 | 0.0000 | 5312 | 19 | 89.21 ms | +| hybrid | 5 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2044 | 19 | 81.29 ms | +| hybrid | 5 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3362 | 19 | 79.52 ms | +| hybrid | 5 | 8 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3362 | 19 | 85.98 ms | +| hybrid | 10 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2063 | 19 | 86.15 ms | +| hybrid | 10 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3526 | 19 | 92.74 ms | +| hybrid | 10 | 8 | 1.0000 | 0.9561 | 0.9389 | 0.1333 | 1.0000 | 0.0000 | 5510 | 19 | 89.97 ms | +| hybrid | 20 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2057 | 19 | 59.56 ms | +| hybrid | 20 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3446 | 19 | 77.17 ms | +| hybrid | 20 | 8 | 1.0000 | 0.9561 | 0.9389 | 0.1333 | 1.0000 | 0.0000 | 5573 | 19 | 91.41 ms | +| hybrid | 40 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2057 | 19 | 89.62 ms | +| hybrid | 40 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3446 | 19 | 66.45 ms | +| hybrid | 40 | 8 | 1.0000 | 0.9561 | 0.9389 | 0.1333 | 1.0000 | 0.0000 | 5573 | 19 | 76.65 ms | + +## How to read it + +- **`candidateK`** moves the ranking metrics and latency: it is how wide the first + stage searches. It cannot change `Context P`/`Context R`, because those look at the + first `contextK` of the fused list and a prefix is unaffected by how deep the list was. +- **`contextK`** moves `Context P` and `Context R` and the context size, not the + ranking metrics. Wider recall rises and precision falls; that is the trade, and both + columns are here so it is visible rather than argued about. +- **Context chars** is a proxy for prompt size, not a token count: the harness pins the + embedding model, not any generation model's tokenizer. +- **No-result** is the share of questions whose retrieval returned nothing at all. + +Best nDCG@10 in this grid: `hybrid` candidateK=5, +contextK=3 (0.9561). +Best context precision: `dense` candidateK=5, +contextK=3 (0.3556). + +These are **not** recommendations. Selecting the grid maximum on the same questions is +how a benchmark becomes a lookup table; the adoption rule in `baseline-v1.6.md` +decides, and the `validation`/`test` split is what keeps that honest. + +## Reproduce + +```bash +npm run eval:prepare # one-time, networked model bootstrap +npm run eval:sweep # offline; rewrites this file +``` diff --git a/eval/README.md b/eval/README.md index 913073e..f715942 100644 --- a/eval/README.md +++ b/eval/README.md @@ -11,6 +11,7 @@ npm run eval:prepare # one-time, networked: download the pinned embedding mod npm run eval # offline and deterministic: run the harness, rewrite the baseline npm run eval:retrieval # strategy comparison (#77) npm run eval:threshold # derive the similarity threshold on validation, report on test +npm run eval:sweep # bounded grid over strategy × candidateK × contextK, one dashboard ``` ### The harness runs the production configuration @@ -39,6 +40,11 @@ result. `--eval-split=validation` selects roughly a third of the questions by a deterministic hash of the id; `test` is the rest. `npm run eval:threshold` uses this to pick a similarity threshold on `validation` and report it on `test`. +`npm run eval:sweep` runs a bounded grid (`strategy × candidateK × contextK`) and +writes one dashboard with quality, context precision/recall, prompt size, index size +and latency side by side. Its grid maximum is labelled as **not** a recommendation: +selecting on the same questions is how a benchmark becomes a lookup table. + `eval:prepare` downloads the pinned `multilingual-e5-small` revision into the app's model cache and verifies it. `eval` never touches the network: if the model is missing it stops with diff --git a/package.json b/package.json index f7b0bcb..4d9134d 100644 --- a/package.json +++ b/package.json @@ -42,7 +42,8 @@ "db:push": "drizzle-kit push", "db:studio": "drizzle-kit studio", "eval:retrieval": "npm run build && node --experimental-transform-types --disable-warning=ExperimentalWarning --disable-warning=MODULE_TYPELESS_PACKAGE_JSON scripts/eval-retrieval.mjs", - "eval:threshold": "npm run build && node scripts/eval-threshold.mjs" + "eval:threshold": "npm run build && node scripts/eval-threshold.mjs", + "eval:sweep": "npm run build && node scripts/eval-sweep.mjs" }, "//test": [ "`node --test` strips TypeScript types rather than compiling them, and strip-only", diff --git a/scripts/eval-sweep.mjs b/scripts/eval-sweep.mjs new file mode 100644 index 0000000..0ed7dc5 --- /dev/null +++ b/scripts/eval-sweep.mjs @@ -0,0 +1,209 @@ +#!/usr/bin/env node +/** + * Parameter sweep + dashboard for #192 (child 7). + * + * Runs the real harness over a bounded grid of `strategy × candidateK × contextK` and + * writes one table. The point is not to find a single number — #192 is explicit that a + * single aggregate "RAG score" says almost nothing — but to make the trade-offs + * visible side by side: quality, the context window's precision/recall, prompt size, + * index size and latency. + * + * Chunking is held fixed so a row differs from its neighbour in one parameter. + * + * Usage: + * node scripts/eval-sweep.mjs + * node scripts/eval-sweep.mjs --strategies=dense --candidate-k=10,20 --context-k=3,5 + * + * The embedding model must already be prepared (`npm run eval:prepare`). + */ + +import { spawn } from 'node:child_process' +import { existsSync, mkdtempSync, mkdirSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join, resolve } from 'node:path' + +const OUT_MD = resolve(readArg('--out=', 'docs/eval/sweep-v1.6.md')) +const OUT_JSON = OUT_MD.replace(/\.md$/, '.json') + +const STRATEGIES = readList('--strategies=', ['dense', 'hybrid']) +const CANDIDATE_KS = readNumberList('--candidate-k=', [5, 10, 20, 40]) +const CONTEXT_KS = readNumberList('--context-k=', [3, 5, 8]) + +function readArg(prefix, fallback) { + const arg = process.argv.find((value) => value.startsWith(prefix)) + return arg ? arg.slice(prefix.length) : fallback +} + +function readList(prefix, fallback) { + const raw = readArg(prefix, null) + return raw ? raw.split(',').filter(Boolean) : fallback +} + +function readNumberList(prefix, fallback) { + const raw = readArg(prefix, null) + if (!raw) return fallback + return raw.split(',').map((value) => { + const parsed = Number(value) + if (!Number.isFinite(parsed)) throw new Error(`${prefix} expects numbers, got ${value}`) + return parsed + }) +} + +// Node 24 refuses to spawn a `.cmd`/`.bat` without `shell: true` (EINVAL), and the +// `.bin` entry is exactly that on Windows. Use the real binary the wrapper runs. +const { default: electronBinary } = await import('electron') +const executable = resolve(electronBinary) + +if (!existsSync(executable)) { + console.error('[sweep] could not find the electron binary. Run `npm install` first.') + process.exit(1) +} + +function runOne(strategy, candidateK, contextK, outDir) { + return new Promise((resolvePromise, reject) => { + const args = [ + '.', + '--eval-harness', + '--eval-baseline=v1.6', + `--eval-out=${outDir}`, + `--eval-retrieval=${strategy}`, + `--eval-candidate-k=${candidateK}`, + `--eval-context-k=${contextK}` + ] + + const isRoot = typeof process.getuid === 'function' && process.getuid() === 0 + if (isRoot || process.env.CI) args.push('--no-sandbox') + + const child = spawn(executable, args, { + stdio: ['ignore', 'pipe', 'pipe'], + env: { ...process.env, ELECTRON_DISABLE_SECURITY_WARNINGS: '1' } + }) + + child.on('error', reject) + child.on('exit', (code) => { + if (code !== 0) { + reject(new Error(`${strategy} candidateK=${candidateK} contextK=${contextK} exited ${code}`)) + return + } + const reportPath = join(outDir, 'baseline-v1.6.json') + if (!existsSync(reportPath)) { + reject(new Error(`${strategy} candidateK=${candidateK} contextK=${contextK} wrote no report`)) + return + } + resolvePromise(JSON.parse(readFileSync(reportPath, 'utf8'))) + }) + }) +} + +/** p95 and index size are informational; the deterministic JSON excludes timing. */ +function readTiming(mdPath) { + if (!existsSync(mdPath)) return { latencyP95Ms: null } + const text = readFileSync(mdPath, 'utf8') + const p95 = /p95 ([\d.]+) ms/.exec(text) + return { latencyP95Ms: p95 ? Number(p95[1]) : null } +} + +const mean = (values) => (values.length === 0 ? 0 : values.reduce((a, b) => a + b, 0) / values.length) + +const workDir = mkdtempSync(join(tmpdir(), 'knownote-sweep-')) +const rows = [] + +try { + for (const strategy of STRATEGIES) { + for (const candidateK of CANDIDATE_KS) { + for (const contextK of CONTEXT_KS) { + const label = `${strategy} candidateK=${candidateK} contextK=${contextK}` + console.log(`[sweep] ${label}`) + const outDir = join(workDir, `${strategy}-${candidateK}-${contextK}`) + mkdirSync(outDir, { recursive: true }) + const report = await runOne(strategy, candidateK, contextK, outDir) + const perQuestion = report.perQuestion ?? [] + const noResult = perQuestion.filter((q) => q.retrievedCount === 0).length + + rows.push({ + strategy, + candidateK, + contextK, + recallAt5: report.metrics.recallAt5, + ndcgAt10: report.metrics.ndcgAt10, + mapAt10: report.metrics.mapAt10, + contextPrecision: report.metrics.contextPrecision, + contextRecall: report.metrics.contextRecall, + noResultRate: perQuestion.length === 0 ? 0 : noResult / perQuestion.length, + meanContextChars: mean(perQuestion.map((q) => q.contextChars)), + chunkCount: report.config.chunkCount, + ...readTiming(join(outDir, 'baseline-v1.6.md')) + }) + } + } + } +} finally { + rmSync(workDir, { recursive: true, force: true }) +} + +const format4 = (value) => value.toFixed(4) + +const tableRows = rows + .map( + (row) => + `| ${row.strategy} | ${row.candidateK} | ${row.contextK} | ${format4(row.recallAt5)} | ` + + `${format4(row.ndcgAt10)} | ${format4(row.mapAt10)} | ${format4(row.contextPrecision)} | ` + + `${format4(row.contextRecall)} | ${format4(row.noResultRate)} | ` + + `${Math.round(row.meanContextChars)} | ${row.chunkCount} | ${row.latencyP95Ms?.toFixed(2) ?? '—'} ms |` + ) + .join('\n') + +const best = (key, filter = () => true) => + rows.filter(filter).reduce((a, b) => (a === null || b[key] > a[key] ? b : a), null) + +const bestNdcg = best('ndcgAt10') +const bestContextPrecision = best('contextPrecision') + +const markdown = `# Parameter sweep — v1.6 (#192) + +Generated by \`node scripts/eval-sweep.mjs\`. Numbers are harness output; do not edit them by hand. + +## What was measured + +The real harness, the same corpus, chunking held fixed, over +${STRATEGIES.join(' / ')} × candidateK {${CANDIDATE_KS.join(', ')}} × contextK {${CONTEXT_KS.join(', ')}} — ${rows.length} runs. +Each row differs from its neighbour in one parameter. + +| Strategy | candidateK | contextK | Recall@5 | nDCG@10 | MAP@10 | Context P | Context R | No-result | Context chars | Index | p95 | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | +${tableRows} + +## How to read it + +- **\`candidateK\`** moves the ranking metrics and latency: it is how wide the first + stage searches. It cannot change \`Context P\`/\`Context R\`, because those look at the + first \`contextK\` of the fused list and a prefix is unaffected by how deep the list was. +- **\`contextK\`** moves \`Context P\` and \`Context R\` and the context size, not the + ranking metrics. Wider recall rises and precision falls; that is the trade, and both + columns are here so it is visible rather than argued about. +- **Context chars** is a proxy for prompt size, not a token count: the harness pins the + embedding model, not any generation model's tokenizer. +- **No-result** is the share of questions whose retrieval returned nothing at all. + +Best nDCG@10 in this grid: \`${bestNdcg.strategy}\` candidateK=${bestNdcg.candidateK}, +contextK=${bestNdcg.contextK} (${format4(bestNdcg.ndcgAt10)}). +Best context precision: \`${bestContextPrecision.strategy}\` candidateK=${bestContextPrecision.candidateK}, +contextK=${bestContextPrecision.contextK} (${format4(bestContextPrecision.contextPrecision)}). + +These are **not** recommendations. Selecting the grid maximum on the same questions is +how a benchmark becomes a lookup table; the adoption rule in \`baseline-v1.6.md\` +decides, and the \`validation\`/\`test\` split is what keeps that honest. + +## Reproduce + +\`\`\`bash +npm run eval:prepare # one-time, networked model bootstrap +npm run eval:sweep # offline; rewrites this file +\`\`\` +` + +mkdirSync(resolve(OUT_MD, '..'), { recursive: true }) +writeFileSync(OUT_JSON, `${JSON.stringify({ baseline: 'v1.6', rows }, null, 2)}\n`) +writeFileSync(OUT_MD, markdown) + +console.log(`[sweep] wrote ${OUT_JSON} and ${OUT_MD}`) diff --git a/src/main/eval/harness.ts b/src/main/eval/harness.ts index c33c43d..727f775 100644 --- a/src/main/eval/harness.ts +++ b/src/main/eval/harness.ts @@ -265,6 +265,9 @@ export async function runEvalHarness( firstRelevantRank: firstRelevantRank(matchesByRank), relevantCount: groundTruth.length, retrievedCount: results.length, + contextChars: results + .slice(0, options.contextK) + .reduce((total, result) => total + result.content.length, 0), matchesByRank }) } diff --git a/src/main/eval/types.ts b/src/main/eval/types.ts index 8833a23..bb7f00e 100644 --- a/src/main/eval/types.ts +++ b/src/main/eval/types.ts @@ -112,6 +112,13 @@ export interface QuestionReport { firstRelevantRank: number relevantCount: number retrievedCount: number + /** + * 送进 context 窗口的前 `contextK` 条证据的字符数(#192 child 7)。 + * + * 是字符数而不是 token 数:不同模型的分词器不同,而 harness 只固定了 embedding + * 模型。把它当 prompt 预算的代理看,不要当成某个模型的 token 数。 + */ + contextChars: number /** Ground-truth indices matched by each retrieved rank, in rank order. */ matchesByRank: number[][] } From aee2f89befc88bd38b3e233e62a4caa8d0fbc4b2 Mon Sep 17 00:00:00 2001 From: MrSibe Date: Wed, 30 Sep 2026 17:23:35 +0800 Subject: [PATCH 08/18] fix(eval): refuse a context window wider than the retrieval depth MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The sweep contained cells the harness cannot fill. The harness retrieves `candidateK` passages and the context metrics look at the first `contextK` of them, so `candidateK=5, contextK=8` reports on five passages while claiming eight. The dashboard showed `contextK=5` and `contextK=8` at `candidateK=5` as **identical** — and because `evidencePrecisionAtK` divides by the passages actually retrieved, nothing exposed it. Found in the #192 review. A wrong number that looks like a measurement is worse than a failure, so the invariant is enforced rather than documented: - **The harness refuses it.** `contextK > candidateK` throws before indexing, naming both numbers: `contextK (8) cannot exceed candidateK (5): the harness retrieves candidateK passages, so a wider context window can never be filled.` - **The sweep skips those cells** and says so. The report lists the skipped combinations in a "Skipped cells" section instead of quietly omitting rows, because "we did not measure this" and "this measured the same as its neighbour" are different statements and the old output showed the second while meaning the first. `docs/eval/sweep-v1.6.*` is regenerated: 22 rows instead of 24, and the misleading `candidateK=5 / contextK=8` row is gone rather than silently equal to `5`. The production cells (`candidateK=20, contextK ∈ {3,5,8}`) are unaffected. ## Testing - `npm run typecheck` — clean - `npm test` — 502 pass, including 3 new harness-invariant tests that need no vector store, because the check runs before any database work: the refusal, the message naming both numbers, and `contextK == candidateK` being accepted - `node_modules/electron/dist/electron.exe . --eval-harness --eval-candidate-k=5 --eval-context-k=8` — refused with the message above - `npm run eval:sweep` — 22 rows, 2 skipped and listed Part of #192 (review follow-up). --- docs/eval/sweep-v1.6.json | 72 ++++++++++--------------------- docs/eval/sweep-v1.6.md | 63 ++++++++++++++++----------- eval/README.md | 7 +++ scripts/eval-sweep.mjs | 91 +++++++++++++++++++++++++++------------ src/main/eval/harness.ts | 14 ++++++ test/evalHarness.test.ts | 63 +++++++++++++++++++++++++++ 6 files changed, 207 insertions(+), 103 deletions(-) create mode 100644 test/evalHarness.test.ts diff --git a/docs/eval/sweep-v1.6.json b/docs/eval/sweep-v1.6.json index 7f7fce1..c6e7860 100644 --- a/docs/eval/sweep-v1.6.json +++ b/docs/eval/sweep-v1.6.json @@ -13,7 +13,7 @@ "noResultRate": 0, "meanContextChars": 2078.9333333333334, "chunkCount": 19, - "latencyP95Ms": 88.15 + "latencyP95Ms": 13.28 }, { "strategy": "dense", @@ -27,21 +27,7 @@ "noResultRate": 0, "meanContextChars": 3326.8333333333335, "chunkCount": 19, - "latencyP95Ms": 89.14 - }, - { - "strategy": "dense", - "candidateK": 5, - "contextK": 8, - "recallAt5": 1, - "ndcgAt10": 0.94375, - "mapAt10": 0.922222, - "contextPrecision": 0.213333, - "contextRecall": 1, - "noResultRate": 0, - "meanContextChars": 3326.8333333333335, - "chunkCount": 19, - "latencyP95Ms": 98.4 + "latencyP95Ms": 13.14 }, { "strategy": "dense", @@ -55,7 +41,7 @@ "noResultRate": 0, "meanContextChars": 2078.9333333333334, "chunkCount": 19, - "latencyP95Ms": 107.2 + "latencyP95Ms": 13.97 }, { "strategy": "dense", @@ -69,7 +55,7 @@ "noResultRate": 0, "meanContextChars": 3326.8333333333335, "chunkCount": 19, - "latencyP95Ms": 77.76 + "latencyP95Ms": 14.81 }, { "strategy": "dense", @@ -83,7 +69,7 @@ "noResultRate": 0, "meanContextChars": 5311.633333333333, "chunkCount": 19, - "latencyP95Ms": 91.92 + "latencyP95Ms": 15.14 }, { "strategy": "dense", @@ -97,7 +83,7 @@ "noResultRate": 0, "meanContextChars": 2078.9333333333334, "chunkCount": 19, - "latencyP95Ms": 94.88 + "latencyP95Ms": 16.19 }, { "strategy": "dense", @@ -111,7 +97,7 @@ "noResultRate": 0, "meanContextChars": 3326.8333333333335, "chunkCount": 19, - "latencyP95Ms": 85.49 + "latencyP95Ms": 18.66 }, { "strategy": "dense", @@ -125,7 +111,7 @@ "noResultRate": 0, "meanContextChars": 5311.633333333333, "chunkCount": 19, - "latencyP95Ms": 84.23 + "latencyP95Ms": 14.61 }, { "strategy": "dense", @@ -139,7 +125,7 @@ "noResultRate": 0, "meanContextChars": 2078.9333333333334, "chunkCount": 19, - "latencyP95Ms": 91.85 + "latencyP95Ms": 17.65 }, { "strategy": "dense", @@ -153,7 +139,7 @@ "noResultRate": 0, "meanContextChars": 3326.8333333333335, "chunkCount": 19, - "latencyP95Ms": 86.76 + "latencyP95Ms": 19.74 }, { "strategy": "dense", @@ -167,7 +153,7 @@ "noResultRate": 0, "meanContextChars": 5311.633333333333, "chunkCount": 19, - "latencyP95Ms": 89.21 + "latencyP95Ms": 17.64 }, { "strategy": "hybrid", @@ -181,7 +167,7 @@ "noResultRate": 0, "meanContextChars": 2044.1333333333334, "chunkCount": 19, - "latencyP95Ms": 81.29 + "latencyP95Ms": 13.37 }, { "strategy": "hybrid", @@ -195,21 +181,7 @@ "noResultRate": 0, "meanContextChars": 3361.6, "chunkCount": 19, - "latencyP95Ms": 79.52 - }, - { - "strategy": "hybrid", - "candidateK": 5, - "contextK": 8, - "recallAt5": 1, - "ndcgAt10": 0.956053, - "mapAt10": 0.938889, - "contextPrecision": 0.213333, - "contextRecall": 1, - "noResultRate": 0, - "meanContextChars": 3361.6, - "chunkCount": 19, - "latencyP95Ms": 85.98 + "latencyP95Ms": 14.7 }, { "strategy": "hybrid", @@ -223,7 +195,7 @@ "noResultRate": 0, "meanContextChars": 2062.733333333333, "chunkCount": 19, - "latencyP95Ms": 86.15 + "latencyP95Ms": 22.31 }, { "strategy": "hybrid", @@ -237,7 +209,7 @@ "noResultRate": 0, "meanContextChars": 3525.633333333333, "chunkCount": 19, - "latencyP95Ms": 92.74 + "latencyP95Ms": 18.66 }, { "strategy": "hybrid", @@ -251,7 +223,7 @@ "noResultRate": 0, "meanContextChars": 5509.8, "chunkCount": 19, - "latencyP95Ms": 89.97 + "latencyP95Ms": 15.84 }, { "strategy": "hybrid", @@ -265,7 +237,7 @@ "noResultRate": 0, "meanContextChars": 2056.9333333333334, "chunkCount": 19, - "latencyP95Ms": 59.56 + "latencyP95Ms": 20.78 }, { "strategy": "hybrid", @@ -279,7 +251,7 @@ "noResultRate": 0, "meanContextChars": 3445.866666666667, "chunkCount": 19, - "latencyP95Ms": 77.17 + "latencyP95Ms": 23.35 }, { "strategy": "hybrid", @@ -293,7 +265,7 @@ "noResultRate": 0, "meanContextChars": 5572.6, "chunkCount": 19, - "latencyP95Ms": 91.41 + "latencyP95Ms": 22.32 }, { "strategy": "hybrid", @@ -307,7 +279,7 @@ "noResultRate": 0, "meanContextChars": 2056.9333333333334, "chunkCount": 19, - "latencyP95Ms": 89.62 + "latencyP95Ms": 23.47 }, { "strategy": "hybrid", @@ -321,7 +293,7 @@ "noResultRate": 0, "meanContextChars": 3445.866666666667, "chunkCount": 19, - "latencyP95Ms": 66.45 + "latencyP95Ms": 21.71 }, { "strategy": "hybrid", @@ -335,7 +307,7 @@ "noResultRate": 0, "meanContextChars": 5572.6, "chunkCount": 19, - "latencyP95Ms": 76.65 + "latencyP95Ms": 21.68 } ] } diff --git a/docs/eval/sweep-v1.6.md b/docs/eval/sweep-v1.6.md index 2168fa4..6af90da 100644 --- a/docs/eval/sweep-v1.6.md +++ b/docs/eval/sweep-v1.6.md @@ -5,35 +5,48 @@ Generated by `node scripts/eval-sweep.mjs`. Numbers are harness output; do not e ## What was measured The real harness, the same corpus, chunking held fixed, over -dense / hybrid × candidateK {5, 10, 20, 40} × contextK {3, 5, 8} — 24 runs. +dense / hybrid × candidateK {5, 10, 20, 40} × contextK {3, 5, 8} — 22 runs. Each row differs from its neighbour in one parameter. +2 further cell(s) were **skipped** because `contextK > candidateK`; see below. + | Strategy | candidateK | contextK | Recall@5 | nDCG@10 | MAP@10 | Context P | Context R | No-result | Context chars | Index | p95 | | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | -| dense | 5 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 19 | 88.15 ms | -| dense | 5 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 19 | 89.14 ms | -| dense | 5 | 8 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 19 | 98.40 ms | -| dense | 10 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 19 | 107.20 ms | -| dense | 10 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 19 | 77.76 ms | -| dense | 10 | 8 | 1.0000 | 0.9437 | 0.9222 | 0.1333 | 1.0000 | 0.0000 | 5312 | 19 | 91.92 ms | -| dense | 20 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 19 | 94.88 ms | -| dense | 20 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 19 | 85.49 ms | -| dense | 20 | 8 | 1.0000 | 0.9437 | 0.9222 | 0.1333 | 1.0000 | 0.0000 | 5312 | 19 | 84.23 ms | -| dense | 40 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 19 | 91.85 ms | -| dense | 40 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 19 | 86.76 ms | -| dense | 40 | 8 | 1.0000 | 0.9437 | 0.9222 | 0.1333 | 1.0000 | 0.0000 | 5312 | 19 | 89.21 ms | -| hybrid | 5 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2044 | 19 | 81.29 ms | -| hybrid | 5 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3362 | 19 | 79.52 ms | -| hybrid | 5 | 8 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3362 | 19 | 85.98 ms | -| hybrid | 10 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2063 | 19 | 86.15 ms | -| hybrid | 10 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3526 | 19 | 92.74 ms | -| hybrid | 10 | 8 | 1.0000 | 0.9561 | 0.9389 | 0.1333 | 1.0000 | 0.0000 | 5510 | 19 | 89.97 ms | -| hybrid | 20 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2057 | 19 | 59.56 ms | -| hybrid | 20 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3446 | 19 | 77.17 ms | -| hybrid | 20 | 8 | 1.0000 | 0.9561 | 0.9389 | 0.1333 | 1.0000 | 0.0000 | 5573 | 19 | 91.41 ms | -| hybrid | 40 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2057 | 19 | 89.62 ms | -| hybrid | 40 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3446 | 19 | 66.45 ms | -| hybrid | 40 | 8 | 1.0000 | 0.9561 | 0.9389 | 0.1333 | 1.0000 | 0.0000 | 5573 | 19 | 76.65 ms | +| dense | 5 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 19 | 13.28 ms | +| dense | 5 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 19 | 13.14 ms | +| dense | 10 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 19 | 13.97 ms | +| dense | 10 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 19 | 14.81 ms | +| dense | 10 | 8 | 1.0000 | 0.9437 | 0.9222 | 0.1333 | 1.0000 | 0.0000 | 5312 | 19 | 15.14 ms | +| dense | 20 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 19 | 16.19 ms | +| dense | 20 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 19 | 18.66 ms | +| dense | 20 | 8 | 1.0000 | 0.9437 | 0.9222 | 0.1333 | 1.0000 | 0.0000 | 5312 | 19 | 14.61 ms | +| dense | 40 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 19 | 17.65 ms | +| dense | 40 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 19 | 19.74 ms | +| dense | 40 | 8 | 1.0000 | 0.9437 | 0.9222 | 0.1333 | 1.0000 | 0.0000 | 5312 | 19 | 17.64 ms | +| hybrid | 5 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2044 | 19 | 13.37 ms | +| hybrid | 5 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3362 | 19 | 14.70 ms | +| hybrid | 10 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2063 | 19 | 22.31 ms | +| hybrid | 10 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3526 | 19 | 18.66 ms | +| hybrid | 10 | 8 | 1.0000 | 0.9561 | 0.9389 | 0.1333 | 1.0000 | 0.0000 | 5510 | 19 | 15.84 ms | +| hybrid | 20 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2057 | 19 | 20.78 ms | +| hybrid | 20 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3446 | 19 | 23.35 ms | +| hybrid | 20 | 8 | 1.0000 | 0.9561 | 0.9389 | 0.1333 | 1.0000 | 0.0000 | 5573 | 19 | 22.32 ms | +| hybrid | 40 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2057 | 19 | 23.47 ms | +| hybrid | 40 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3446 | 19 | 21.71 ms | +| hybrid | 40 | 8 | 1.0000 | 0.9561 | 0.9389 | 0.1333 | 1.0000 | 0.0000 | 5573 | 19 | 21.68 ms | + + +## Skipped cells + +`contextK > candidateK` cannot be filled: the harness fetches `candidateK` +passages, so a wider window would contain fewer passages than it claims. These +2 cell(s) are excluded rather than reported as equal to a narrower one: + +- `dense` candidateK=5, contextK=8 +- `hybrid` candidateK=5, contextK=8 + +The harness refuses the same combination at the flag level, so a typo fails loudly. + ## How to read it diff --git a/eval/README.md b/eval/README.md index f715942..290648f 100644 --- a/eval/README.md +++ b/eval/README.md @@ -45,6 +45,13 @@ writes one dashboard with quality, context precision/recall, prompt size, index and latency side by side. Its grid maximum is labelled as **not** a recommendation: selecting on the same questions is how a benchmark becomes a lookup table. +`contextK > candidateK` is not a cell in that grid. The harness fetches `candidateK` +passages, so a wider window can never be filled; the sweep skips those combinations +and names them in the report, and the harness refuses the same combination from the +command line. Before this was enforced, `contextK=8` at `candidateK=5` was reported as +identical to `contextK=5` — not because 8 assessed the same as 5, but because +passages 6–8 did not exist. + `eval:prepare` downloads the pinned `multilingual-e5-small` revision into the app's model cache and verifies it. `eval` never touches the network: if the model is missing it stops with diff --git a/scripts/eval-sweep.mjs b/scripts/eval-sweep.mjs index 0ed7dc5..46faefd 100644 --- a/scripts/eval-sweep.mjs +++ b/scripts/eval-sweep.mjs @@ -105,37 +105,58 @@ function readTiming(mdPath) { const mean = (values) => (values.length === 0 ? 0 : values.reduce((a, b) => a + b, 0) / values.length) +/** + * `contextK > candidateK` is not a cell, it is an arithmetic mistake: the harness only + * fetches `candidateK` passages, so the window can never be filled. Skipping them is why + * the default grid no longer contains rows that looked like "8 is as good as 5" when + * passages 6-8 were simply never retrieved (#192 review). The harness refuses the same + * combination, so a typo on the command line fails loudly instead of silently. + */ +const grid = [] +const skipped = [] +for (const strategy of STRATEGIES) { + for (const candidateK of CANDIDATE_KS) { + for (const contextK of CONTEXT_KS) { + if (contextK > candidateK) skipped.push({ strategy, candidateK, contextK }) + else grid.push({ strategy, candidateK, contextK }) + } + } +} + +if (skipped.length > 0) { + console.log( + `[sweep] skipped ${skipped.length} cell(s) with contextK > candidateK: ` + + skipped.map((c) => `${c.strategy} ${c.candidateK}/${c.contextK}`).join(', ') + ) +} + const workDir = mkdtempSync(join(tmpdir(), 'knownote-sweep-')) const rows = [] try { - for (const strategy of STRATEGIES) { - for (const candidateK of CANDIDATE_KS) { - for (const contextK of CONTEXT_KS) { - const label = `${strategy} candidateK=${candidateK} contextK=${contextK}` - console.log(`[sweep] ${label}`) - const outDir = join(workDir, `${strategy}-${candidateK}-${contextK}`) - mkdirSync(outDir, { recursive: true }) - const report = await runOne(strategy, candidateK, contextK, outDir) - const perQuestion = report.perQuestion ?? [] - const noResult = perQuestion.filter((q) => q.retrievedCount === 0).length - - rows.push({ - strategy, - candidateK, - contextK, - recallAt5: report.metrics.recallAt5, - ndcgAt10: report.metrics.ndcgAt10, - mapAt10: report.metrics.mapAt10, - contextPrecision: report.metrics.contextPrecision, - contextRecall: report.metrics.contextRecall, - noResultRate: perQuestion.length === 0 ? 0 : noResult / perQuestion.length, - meanContextChars: mean(perQuestion.map((q) => q.contextChars)), - chunkCount: report.config.chunkCount, - ...readTiming(join(outDir, 'baseline-v1.6.md')) - }) - } - } + for (const cell of grid) { + const { strategy, candidateK, contextK } = cell + console.log(`[sweep] ${strategy} candidateK=${candidateK} contextK=${contextK}`) + const outDir = join(workDir, `${strategy}-${candidateK}-${contextK}`) + mkdirSync(outDir, { recursive: true }) + const report = await runOne(strategy, candidateK, contextK, outDir) + const perQuestion = report.perQuestion ?? [] + const noResult = perQuestion.filter((q) => q.retrievedCount === 0).length + + rows.push({ + strategy, + candidateK, + contextK, + recallAt5: report.metrics.recallAt5, + ndcgAt10: report.metrics.ndcgAt10, + mapAt10: report.metrics.mapAt10, + contextPrecision: report.metrics.contextPrecision, + contextRecall: report.metrics.contextRecall, + noResultRate: perQuestion.length === 0 ? 0 : noResult / perQuestion.length, + meanContextChars: mean(perQuestion.map((q) => q.contextChars)), + chunkCount: report.config.chunkCount, + ...readTiming(join(outDir, 'baseline-v1.6.md')) + }) } } finally { rmSync(workDir, { recursive: true, force: true }) @@ -156,6 +177,19 @@ const tableRows = rows const best = (key, filter = () => true) => rows.filter(filter).reduce((a, b) => (a === null || b[key] > a[key] ? b : a), null) +/** + * A skipped cell is reported, not silently dropped: "we did not measure this" and "this + * measured the same as its neighbour" are different statements, and the earlier version + * of this file showed the second when it meant the first. + */ +const skippedNote = + skipped.length === 0 + ? '' + : `\n\n## Skipped cells\n\n\`contextK > candidateK\` cannot be filled: the harness fetches \`candidateK\`\npassages, so a wider window would contain fewer passages than it claims. These +${skipped.length} cell(s) are excluded rather than reported as equal to a narrower one:\n\n${skipped + .map((c) => `- \`${c.strategy}\` candidateK=${c.candidateK}, contextK=${c.contextK}`) + .join('\n')}\n\nThe harness refuses the same combination at the flag level, so a typo fails loudly.\n` + const bestNdcg = best('ndcgAt10') const bestContextPrecision = best('contextPrecision') @@ -167,11 +201,12 @@ Generated by \`node scripts/eval-sweep.mjs\`. Numbers are harness output; do not The real harness, the same corpus, chunking held fixed, over ${STRATEGIES.join(' / ')} × candidateK {${CANDIDATE_KS.join(', ')}} × contextK {${CONTEXT_KS.join(', ')}} — ${rows.length} runs. -Each row differs from its neighbour in one parameter. +Each row differs from its neighbour in one parameter.${skipped.length > 0 ? `\n\n${skipped.length} further cell(s) were **skipped** because \`contextK > candidateK\`; see below.` : ''} | Strategy | candidateK | contextK | Recall@5 | nDCG@10 | MAP@10 | Context P | Context R | No-result | Context chars | Index | p95 | | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | ${tableRows} +${skippedNote} ## How to read it diff --git a/src/main/eval/harness.ts b/src/main/eval/harness.ts index 727f775..418ddf7 100644 --- a/src/main/eval/harness.ts +++ b/src/main/eval/harness.ts @@ -217,6 +217,20 @@ export async function runEvalHarness( knowledgeService: KnowledgeService, options: EvalHarnessOptions ): Promise { + // A context window wider than the retrieval depth can never be filled: the harness + // fetches `candidateK` passages and the context metrics look at `contextK` of them. + // + // Without this check the sweep silently produced rows where `contextK=5` and + // `contextK=8` at `candidateK=5` were **identical**, not because 8 assessed the same + // as 5 but because passages 6-8 did not exist (#192 review). A wrong number that + // looks like a measurement is worse than a failure. + if (options.contextK > options.candidateK) { + throw new Error( + `contextK (${options.contextK}) cannot exceed candidateK (${options.candidateK}): ` + + 'the harness retrieves candidateK passages, so a wider context window can never be filled.' + ) + } + const { documentIds, chunkCount, indexingMs } = await indexCorpus( db, knowledgeService, diff --git a/test/evalHarness.test.ts b/test/evalHarness.test.ts new file mode 100644 index 0000000..a5e650b --- /dev/null +++ b/test/evalHarness.test.ts @@ -0,0 +1,63 @@ +import { test } from 'node:test' +import assert from 'node:assert/strict' +import { runEvalHarness } from '../src/main/eval/harness.ts' + +/** + * The harness's own invariants (#192). These run before any database or retrieval work, + * so they can be pinned without a store: the check is the first statement of + * `runEvalHarness`, and a violation must fail loudly rather than produce a number. + */ + +const options = (overrides: Record = {}): Record => ({ + corpusDir: 'eval/corpus', + corpusLabel: 'eval/corpus', + questionsPath: 'eval/questions.jsonl', + baseline: 'test', + split: 'all', + candidateK: 20, + contextK: 3, + threshold: 0.5, + chunkOptions: { + chunkSize: 1000, + chunkOverlap: 100, + minChunkSize: 100, + allowSpanPages: false, + respectHeadings: false + }, + strategy: 'dense', + ...overrides +}) + +/** + * A context window wider than the retrieval depth can never be filled. The sweep used + * to contain `candidateK=5, contextK=8` and report it as identical to `contextK=5` — + * not because 8 assessed the same as 5, but because passages 6-8 did not exist (#192 + * review). A wrong number that looks like a measurement is worse than a failure. + */ +test('a context window wider than the retrieval depth is refused', async () => { + await assert.rejects( + () => runEvalHarness({} as never, {} as never, options({ candidateK: 5, contextK: 8 }) as never), + /contextK \(8\) cannot exceed candidateK \(5\)/ + ) +}) + +test('the refusal names both numbers, so the fix is obvious', async () => { + await assert.rejects( + () => runEvalHarness({} as never, {} as never, options({ candidateK: 3, contextK: 10 }) as never), + (error: Error) => { + assert.match(error.message, /candidateK \(3\)/) + assert.match(error.message, /contextK \(10\)/) + assert.match(error.message, /can never be filled/) + return true + } + ) +}) + +test('contextK equal to candidateK is allowed and reaches further invariants', async () => { + // No database is provided, so the call must fail *after* the window check — proving the + // boundary value is accepted rather than rejected. + await assert.rejects( + () => runEvalHarness({} as never, {} as never, options({ candidateK: 5, contextK: 5 }) as never), + (error: Error) => !/cannot exceed/.test(error.message) + ) +}) From d45246347c7e7423bc6ff5fb58b761073f082574 Mon Sep 17 00:00:00 2001 From: MrSibe Date: Wed, 30 Sep 2026 17:35:38 +0800 Subject: [PATCH 09/18] feat(eval): unanswerable questions, and an explicit split manifest MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The corpus contract corpus v2 needs. Without it, two of the four things #192 asks the dataset to cover cannot be represented at all: a question whose right answer is "your sources do not say", and a validation/test assignment that is chosen rather than computed. **Unanswerable questions.** `answerable: false` with `relevant: []`. They are excluded from `metrics` and `byType` — `recallAtK` treats no ground truth as 0/0, and averaging "correctly refused" into "missed" would corrupt the table — and reported as their own group: `noResultCount`, `noResultRate` and `meanRetrieved`. On this row **higher no-result is better**, which is the opposite of how it reads everywhere else. `assertQuestionShape` refuses the reverse combination in either direction, because both are silent: an answerable question with no ground truth reads as a permanent miss, and an unanswerable one carrying ground truth reads as a normal hit. **An explicit split manifest.** `eval/splits.json` replaces the hash of the question id. A hash is reproducible but not *stable*: adding a question moved others between the sides, and a rare query type could end up entirely on one side without anyone choosing that — which is what had happened (`multi-hop` and `cross-lingual` were both entirely on `test`). A question with no manifest entry is now **refused** rather than defaulted, so a new question cannot leak into the reporting side. **Six unanswerable questions** are added (three topic-adjacent, three plainly out-of-scope). Their specifics were checked absent from the corpus before being written, so they are genuinely unanswerable rather than believed to be. ## What the first measurement shows ``` unanswerable: { questions: 6, noResultRate: 0, meanRetrieved: 19 } ``` Every unanswerable question returns **the entire corpus** — 19 of 19 chunks — at `threshold: 0.5`, and the threshold sweep is flat all the way to `0.6`: refusal stays `0/6` and `meanRetrieved` stays at the cap. So on this corpus the threshold cannot separate a relevant passage from an unrelated one, and "Who won the 2018 FIFA World Cup?" is answered with 19 passages of river monitoring, tidal energy and tea storage. That is the evidence the threshold decision did not have before, and it is also the corpus limitation #192 describes: it does not say `0.5` is wrong, it says this corpus cannot tell. The reports say so rather than recommending a change. Retrieval metrics are unchanged (`recallAt5` 1.0000), because the six new questions are correctly kept out of them. ## Testing - `npm run typecheck` — clean - `npm test` — 504 pass, including the manifest contract (unknown side rejected, missing entry refused, `all` needs no manifest), the answerable/unanswerable tie, and the harness's window invariant - `npm run eval` twice — byte-identical `docs/eval/baseline-v1.6.json` - `npm run eval:retrieval`, `eval:threshold`, `eval:sweep` — all regenerated; the threshold rule now gates on answerable quality and then maximises unanswerable refusal ## Scope This is the **contract**, plus the smallest honest corpus increment that exercises it. Hard negatives and near-duplicate documents — the thing that would bring `Recall@5` off 1.0000 and give `candidateK` something to do — are the next step, not this PR. Part of #192 (child 2, stage 1 of the corpus). --- docs/eval/baseline-v1.6.json | 224 +++++++++++++++++++++++++++++++++- docs/eval/baseline-v1.6.md | 20 ++- docs/eval/retrieval-v1.6.json | 21 +++- docs/eval/retrieval-v1.6.md | 11 +- docs/eval/sweep-v1.6.json | 112 +++++++++++++---- docs/eval/sweep-v1.6.md | 55 +++++---- docs/eval/threshold-v1.6.json | 200 +++++++++++++++++++++--------- docs/eval/threshold-v1.6.md | 49 ++++---- eval/README.md | 24 +++- eval/questions.jsonl | 6 + eval/splits.json | 38 ++++++ scripts/eval-retrieval.mjs | 8 +- scripts/eval-sweep.mjs | 27 ++-- scripts/eval-threshold.mjs | 127 +++++++++++-------- src/main/eval/harness.ts | 58 +++++++-- src/main/eval/report.ts | 26 +++- src/main/eval/run.ts | 2 + src/main/eval/types.ts | 106 ++++++++++++++-- test/evalHarness.test.ts | 1 + test/evalSplit.test.ts | 112 +++++++++++------ 20 files changed, 966 insertions(+), 261 deletions(-) create mode 100644 eval/splits.json diff --git a/docs/eval/baseline-v1.6.json b/docs/eval/baseline-v1.6.json index 43b77d7..5f4b48a 100644 --- a/docs/eval/baseline-v1.6.json +++ b/docs/eval/baseline-v1.6.json @@ -17,7 +17,7 @@ "threshold": 0.5, "corpus": "eval/corpus", "documents": 13, - "questions": 30, + "questions": 36, "chunkCount": 19 }, "metrics": { @@ -108,11 +108,18 @@ } } ], + "unanswerable": { + "questions": 6, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 19 + }, "perQuestion": [ { "id": "q001", "question": "Why is bedload harder to measure than suspended sediment?", "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -145,6 +152,7 @@ "id": "q002", "question": "How many replicate samples are collected at each river station?", "type": "exact", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -177,6 +185,7 @@ "id": "q003", "question": "What is the central trade-off in lithium-ion cell design?", "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -209,6 +218,7 @@ "id": "q004", "question": "Why do nickel-rich battery packs need more aggressive thermal management?", "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -241,6 +251,7 @@ "id": "q005", "question": "What happens once the separator in a battery cell melts?", "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -273,6 +284,7 @@ "id": "q006", "question": "At what temperature do honeybees begin to forage?", "type": "exact", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -305,6 +317,7 @@ "id": "q007", "question": "What does a late frost damage during full bloom?", "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -337,6 +350,7 @@ "id": "q008", "question": "Why is a continuous tree canopy more effective at cooling than isolated trees?", "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -369,6 +383,7 @@ "id": "q009", "question": "Why are trees with aggressive surface roots unsuitable for narrow verges?", "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -401,6 +416,7 @@ "id": "q010", "question": "At what temperature is lactic acid fermentation fastest?", "type": "exact", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -433,6 +449,7 @@ "id": "q011", "question": "Is the salt percentage in fermentation based on vegetable weight or water weight?", "type": "exact", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -465,6 +482,7 @@ "id": "q012", "question": "Why must tidal turbines be sited in places with very fast currents?", "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -497,6 +515,7 @@ "id": "q013", "question": "What is the main environmental concern for tidal energy installations?", "type": "semantic", + "answerable": true, "firstRelevantRank": 3, "relevantCount": 1, "retrievedCount": 19, @@ -529,6 +548,7 @@ "id": "q014", "question": "In lake monitoring, how is the sampling depth actually recorded?", "type": "exact", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -561,6 +581,7 @@ "id": "q015", "question": "Why does deep-water oxygen fall while a lake remains stratified?", "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -593,6 +614,7 @@ "id": "q016", "question": "How do supercapacitors hold their charge?", "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -625,6 +647,7 @@ "id": "q017", "question": "Why can solitary bees pollinate a bloom week that is too cold for honeybee hives?", "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -657,6 +680,7 @@ "id": "q018", "question": "Why is one continuous planted roof layer better than several isolated planted beds?", "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -689,6 +713,7 @@ "id": "q019", "question": "Where do acetic acid bacteria sit in a vinegar culture?", "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -721,6 +746,7 @@ "id": "q020", "question": "What happens if a vinegar culture is sealed airtight?", "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -753,6 +779,7 @@ "id": "q021", "question": "Why is wave energy harder to schedule ahead than tidal energy?", "type": "semantic", + "answerable": true, "firstRelevantRank": 2, "relevantCount": 1, "retrievedCount": 19, @@ -785,6 +812,7 @@ "id": "q022", "question": "Where does siting for wave energy devices concentrate, and where does it not?", "type": "exact", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -817,6 +845,7 @@ "id": "q023", "question": "How does the river sampling protocol differ from the lake sampling protocol?", "type": "multi-hop", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 2, "retrievedCount": 19, @@ -851,6 +880,7 @@ "id": "q024", "question": "A street canopy and a green roof are both said to cool; what surface does each one shade?", "type": "multi-hop", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 2, "retrievedCount": 19, @@ -885,6 +915,7 @@ "id": "q025", "question": "Which preservation method depends on keeping air away from the food?", "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -917,6 +948,7 @@ "id": "q026", "question": "Why can one cold morning cost a grower the whole crop even when colonies are brought in?", "type": "semantic", + "answerable": true, "firstRelevantRank": 2, "relevantCount": 1, "retrievedCount": 19, @@ -949,6 +981,7 @@ "id": "q027", "question": "绿茶应该怎样保存才能减缓氧化?", "type": "zh", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -981,6 +1014,7 @@ "id": "q028", "question": "茶叶储存的相对湿度上限是多少?", "type": "zh", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -1013,6 +1047,7 @@ "id": "q029", "question": "为什么冷冻保存的茶叶取出后不能立刻打开包装?", "type": "zh", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -1045,6 +1080,7 @@ "id": "q030", "question": "为什么潮汐能比风能和太阳能更容易提前安排发电?", "type": "cross-lingual", + "answerable": true, "firstRelevantRank": 2, "relevantCount": 1, "retrievedCount": 19, @@ -1072,6 +1108,192 @@ [], [] ] + }, + { + "id": "q031", + "question": "What is the installed capacity of the tidal energy installation described?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 19, + "contextChars": 2488, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q032", + "question": "Which laboratory published the river monitoring protocol?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 19, + "contextChars": 2257, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q033", + "question": "What is the boiling point of mercury?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 19, + "contextChars": 2043, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q034", + "question": "What is the retail price of the lithium-ion cells discussed?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 19, + "contextChars": 2588, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q035", + "question": "How many megawatts does the tidal array generate?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 19, + "contextChars": 2488, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q036", + "question": "Who won the 2018 FIFA World Cup?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 19, + "contextChars": 2071, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] } ] } diff --git a/docs/eval/baseline-v1.6.md b/docs/eval/baseline-v1.6.md index 5c7fc85..23fffc2 100644 --- a/docs/eval/baseline-v1.6.md +++ b/docs/eval/baseline-v1.6.md @@ -12,7 +12,7 @@ Generated by `npm run eval`. The numbers below are harness output — do not edi | Ranks | `candidateK=20, threshold=0.5` | | Context width | `contextK=3` | | Corpus | `eval/corpus` (13 documents) | -| Split | `all` (30 questions) | +| Split | `all` (36 questions, 30 answerable) | | Index size | 19 chunks | ## Metrics @@ -43,8 +43,22 @@ The type comes from `type` in `questions.jsonl`; untagged questions report as | semantic | 18 | 1.0000 | 0.9312 | 1.0000 | 0.9074 | | zh | 3 | 1.0000 | 1.0000 | 1.0000 | 1.0000 | -Timing is informational only and is **not** frozen: indexing 2015 ms, query -p50 71.63 ms, p95 94.77 ms on the +### Unanswerable questions + +These carry no ground truth, so the correct outcome is that retrieval finds nothing. They +are excluded from every metric above — a missing ground truth is not a miss — and reported +here instead. A higher **no-results** rate is better on this row, which is the opposite of +how it reads everywhere else, and `mean passages retrieved` is how much irrelevant context +was pulled in anyway. This is the row a threshold decision should move. + +| Metric | Value | +| --- | --- | +| Unanswerable questions | 6 | +| Returned no results | 0.0000 (0/6) | +| Mean passages retrieved | 19.00 | + +Timing is informational only and is **not** frozen: indexing 1496 ms, query +p50 11.42 ms, p95 17.31 ms on the machine that produced this file. Timing and index size depend on hardware and on the corpus, so they must never be the reason two runs differ. diff --git a/docs/eval/retrieval-v1.6.json b/docs/eval/retrieval-v1.6.json index 6782582..9bf0d74 100644 --- a/docs/eval/retrieval-v1.6.json +++ b/docs/eval/retrieval-v1.6.json @@ -7,6 +7,9 @@ "label": "dense (vector)", "chunking": "1000/100", "chunkCount": 19, + "split": "all", + "questions": 36, + "answerableCount": 30, "recallAt1": 0.833333, "recallAt5": 1, "recallAt10": 1, @@ -16,14 +19,17 @@ "mapAt10": 0.922222, "contextPrecision": 0.355556, "contextRecall": 1, - "indexingMs": 1548, - "latencyP95Ms": 20.51 + "indexingMs": 1518, + "latencyP95Ms": 14.2 }, { "id": "sparse", "label": "sparse (BM25)", "chunking": "1000/100", "chunkCount": 19, + "split": "all", + "questions": 36, + "answerableCount": 30, "recallAt1": 0.733333, "recallAt5": 0.866667, "recallAt10": 0.866667, @@ -33,14 +39,17 @@ "mapAt10": 0.8, "contextPrecision": 0.311111, "contextRecall": 0.866667, - "indexingMs": 1490, - "latencyP95Ms": 1.96 + "indexingMs": 1513, + "latencyP95Ms": 2.46 }, { "id": "hybrid", "label": "hybrid (RRF of dense + BM25)", "chunking": "1000/100", "chunkCount": 19, + "split": "all", + "questions": 36, + "answerableCount": 30, "recallAt1": 0.866667, "recallAt5": 1, "recallAt10": 1, @@ -50,8 +59,8 @@ "mapAt10": 0.938889, "contextPrecision": 0.355556, "contextRecall": 1, - "indexingMs": 1507, - "latencyP95Ms": 20.66 + "indexingMs": 1557, + "latencyP95Ms": 19.55 } ] } diff --git a/docs/eval/retrieval-v1.6.md b/docs/eval/retrieval-v1.6.md index c13d4e4..a091483 100644 --- a/docs/eval/retrieval-v1.6.md +++ b/docs/eval/retrieval-v1.6.md @@ -4,14 +4,15 @@ Generated by `node scripts/eval-retrieval.mjs`. Numbers are harness output; do n ## What was measured -Every strategy runs the real RAG eval harness against the same corpus and the same 30 -questions as `baseline-v1.6.json`, with chunking held fixed at 1000/100. Only the retrieval strategy changes. +Every strategy runs the real RAG eval harness against the same corpus and the same questions +as `baseline-v1.6.json` (split `all`, 36 questions of which +30 are answerable), with chunking held fixed at 1000/100. Only the retrieval strategy changes. | Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Context P | Query p95 | | --- | --- | --- | --- | --- | --- | --- | --- | -| dense (vector) | 0.8333 | 1.0000 | 0.9278 | 0.9437 | 0.9222 | 0.3556 | 20.51 ms | -| sparse (BM25) | 0.7333 | 0.8667 | 0.8056 | 0.8184 | 0.8000 | 0.3111 | 1.96 ms | -| hybrid (RRF of dense + BM25) | 0.8667 | 1.0000 | 0.9444 | 0.9561 | 0.9389 | 0.3556 | 20.66 ms | +| dense (vector) | 0.8333 | 1.0000 | 0.9278 | 0.9437 | 0.9222 | 0.3556 | 14.20 ms | +| sparse (BM25) | 0.7333 | 0.8667 | 0.8056 | 0.8184 | 0.8000 | 0.3111 | 2.46 ms | +| hybrid (RRF of dense + BM25) | 0.8667 | 1.0000 | 0.9444 | 0.9561 | 0.9389 | 0.3556 | 19.55 ms | ## Not evaluated diff --git a/docs/eval/sweep-v1.6.json b/docs/eval/sweep-v1.6.json index c6e7860..67cdb5b 100644 --- a/docs/eval/sweep-v1.6.json +++ b/docs/eval/sweep-v1.6.json @@ -12,8 +12,11 @@ "contextRecall": 1, "noResultRate": 0, "meanContextChars": 2078.9333333333334, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 5, "chunkCount": 19, - "latencyP95Ms": 13.28 + "latencyP95Ms": 98.32 }, { "strategy": "dense", @@ -26,8 +29,11 @@ "contextRecall": 1, "noResultRate": 0, "meanContextChars": 3326.8333333333335, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 5, "chunkCount": 19, - "latencyP95Ms": 13.14 + "latencyP95Ms": 96.94 }, { "strategy": "dense", @@ -40,8 +46,11 @@ "contextRecall": 1, "noResultRate": 0, "meanContextChars": 2078.9333333333334, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 10, "chunkCount": 19, - "latencyP95Ms": 13.97 + "latencyP95Ms": 84.19 }, { "strategy": "dense", @@ -54,8 +63,11 @@ "contextRecall": 1, "noResultRate": 0, "meanContextChars": 3326.8333333333335, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 10, "chunkCount": 19, - "latencyP95Ms": 14.81 + "latencyP95Ms": 95.07 }, { "strategy": "dense", @@ -68,8 +80,11 @@ "contextRecall": 1, "noResultRate": 0, "meanContextChars": 5311.633333333333, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 10, "chunkCount": 19, - "latencyP95Ms": 15.14 + "latencyP95Ms": 91.67 }, { "strategy": "dense", @@ -82,8 +97,11 @@ "contextRecall": 1, "noResultRate": 0, "meanContextChars": 2078.9333333333334, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 16.19 + "latencyP95Ms": 16.2 }, { "strategy": "dense", @@ -96,8 +114,11 @@ "contextRecall": 1, "noResultRate": 0, "meanContextChars": 3326.8333333333335, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 18.66 + "latencyP95Ms": 14.19 }, { "strategy": "dense", @@ -110,8 +131,11 @@ "contextRecall": 1, "noResultRate": 0, "meanContextChars": 5311.633333333333, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 14.61 + "latencyP95Ms": 20.67 }, { "strategy": "dense", @@ -124,8 +148,11 @@ "contextRecall": 1, "noResultRate": 0, "meanContextChars": 2078.9333333333334, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 17.65 + "latencyP95Ms": 19.91 }, { "strategy": "dense", @@ -138,8 +165,11 @@ "contextRecall": 1, "noResultRate": 0, "meanContextChars": 3326.8333333333335, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 19.74 + "latencyP95Ms": 18.65 }, { "strategy": "dense", @@ -152,8 +182,11 @@ "contextRecall": 1, "noResultRate": 0, "meanContextChars": 5311.633333333333, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 17.64 + "latencyP95Ms": 18.52 }, { "strategy": "hybrid", @@ -166,8 +199,11 @@ "contextRecall": 1, "noResultRate": 0, "meanContextChars": 2044.1333333333334, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 5, "chunkCount": 19, - "latencyP95Ms": 13.37 + "latencyP95Ms": 15 }, { "strategy": "hybrid", @@ -180,8 +216,11 @@ "contextRecall": 1, "noResultRate": 0, "meanContextChars": 3361.6, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 5, "chunkCount": 19, - "latencyP95Ms": 14.7 + "latencyP95Ms": 14.4 }, { "strategy": "hybrid", @@ -194,8 +233,11 @@ "contextRecall": 1, "noResultRate": 0, "meanContextChars": 2062.733333333333, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 10, "chunkCount": 19, - "latencyP95Ms": 22.31 + "latencyP95Ms": 18.54 }, { "strategy": "hybrid", @@ -208,8 +250,11 @@ "contextRecall": 1, "noResultRate": 0, "meanContextChars": 3525.633333333333, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 10, "chunkCount": 19, - "latencyP95Ms": 18.66 + "latencyP95Ms": 16.13 }, { "strategy": "hybrid", @@ -221,9 +266,12 @@ "contextPrecision": 0.133333, "contextRecall": 1, "noResultRate": 0, - "meanContextChars": 5509.8, + "meanContextChars": 5525.166666666667, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 10, "chunkCount": 19, - "latencyP95Ms": 15.84 + "latencyP95Ms": 18.95 }, { "strategy": "hybrid", @@ -236,8 +284,11 @@ "contextRecall": 1, "noResultRate": 0, "meanContextChars": 2056.9333333333334, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 20.78 + "latencyP95Ms": 21.22 }, { "strategy": "hybrid", @@ -250,8 +301,11 @@ "contextRecall": 1, "noResultRate": 0, "meanContextChars": 3445.866666666667, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 23.35 + "latencyP95Ms": 23.23 }, { "strategy": "hybrid", @@ -264,8 +318,11 @@ "contextRecall": 1, "noResultRate": 0, "meanContextChars": 5572.6, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 22.32 + "latencyP95Ms": 21.75 }, { "strategy": "hybrid", @@ -278,8 +335,11 @@ "contextRecall": 1, "noResultRate": 0, "meanContextChars": 2056.9333333333334, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 23.47 + "latencyP95Ms": 20.35 }, { "strategy": "hybrid", @@ -292,8 +352,11 @@ "contextRecall": 1, "noResultRate": 0, "meanContextChars": 3445.866666666667, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 21.71 + "latencyP95Ms": 20.35 }, { "strategy": "hybrid", @@ -306,8 +369,11 @@ "contextRecall": 1, "noResultRate": 0, "meanContextChars": 5572.6, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 21.68 + "latencyP95Ms": 20.91 } ] } diff --git a/docs/eval/sweep-v1.6.md b/docs/eval/sweep-v1.6.md index 6af90da..bfd5e48 100644 --- a/docs/eval/sweep-v1.6.md +++ b/docs/eval/sweep-v1.6.md @@ -10,30 +10,30 @@ Each row differs from its neighbour in one parameter. 2 further cell(s) were **skipped** because `contextK > candidateK`; see below. -| Strategy | candidateK | contextK | Recall@5 | nDCG@10 | MAP@10 | Context P | Context R | No-result | Context chars | Index | p95 | -| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | -| dense | 5 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 19 | 13.28 ms | -| dense | 5 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 19 | 13.14 ms | -| dense | 10 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 19 | 13.97 ms | -| dense | 10 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 19 | 14.81 ms | -| dense | 10 | 8 | 1.0000 | 0.9437 | 0.9222 | 0.1333 | 1.0000 | 0.0000 | 5312 | 19 | 15.14 ms | -| dense | 20 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 19 | 16.19 ms | -| dense | 20 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 19 | 18.66 ms | -| dense | 20 | 8 | 1.0000 | 0.9437 | 0.9222 | 0.1333 | 1.0000 | 0.0000 | 5312 | 19 | 14.61 ms | -| dense | 40 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 19 | 17.65 ms | -| dense | 40 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 19 | 19.74 ms | -| dense | 40 | 8 | 1.0000 | 0.9437 | 0.9222 | 0.1333 | 1.0000 | 0.0000 | 5312 | 19 | 17.64 ms | -| hybrid | 5 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2044 | 19 | 13.37 ms | -| hybrid | 5 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3362 | 19 | 14.70 ms | -| hybrid | 10 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2063 | 19 | 22.31 ms | -| hybrid | 10 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3526 | 19 | 18.66 ms | -| hybrid | 10 | 8 | 1.0000 | 0.9561 | 0.9389 | 0.1333 | 1.0000 | 0.0000 | 5510 | 19 | 15.84 ms | -| hybrid | 20 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2057 | 19 | 20.78 ms | -| hybrid | 20 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3446 | 19 | 23.35 ms | -| hybrid | 20 | 8 | 1.0000 | 0.9561 | 0.9389 | 0.1333 | 1.0000 | 0.0000 | 5573 | 19 | 22.32 ms | -| hybrid | 40 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2057 | 19 | 23.47 ms | -| hybrid | 40 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3446 | 19 | 21.71 ms | -| hybrid | 40 | 8 | 1.0000 | 0.9561 | 0.9389 | 0.1333 | 1.0000 | 0.0000 | 5573 | 19 | 21.68 ms | +| Strategy | candidateK | contextK | Recall@5 | nDCG@10 | MAP@10 | Context P | Context R | No-result | Context chars | Unans. no-result | Unans. retrieved | Index | p95 | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | +| dense | 5 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 0.0000 | 5.0 | 19 | 98.32 ms | +| dense | 5 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 0.0000 | 5.0 | 19 | 96.94 ms | +| dense | 10 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 0.0000 | 10.0 | 19 | 84.19 ms | +| dense | 10 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 0.0000 | 10.0 | 19 | 95.07 ms | +| dense | 10 | 8 | 1.0000 | 0.9437 | 0.9222 | 0.1333 | 1.0000 | 0.0000 | 5312 | 0.0000 | 10.0 | 19 | 91.67 ms | +| dense | 20 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 0.0000 | 19.0 | 19 | 16.20 ms | +| dense | 20 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 0.0000 | 19.0 | 19 | 14.19 ms | +| dense | 20 | 8 | 1.0000 | 0.9437 | 0.9222 | 0.1333 | 1.0000 | 0.0000 | 5312 | 0.0000 | 19.0 | 19 | 20.67 ms | +| dense | 40 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 0.0000 | 19.0 | 19 | 19.91 ms | +| dense | 40 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 0.0000 | 19.0 | 19 | 18.65 ms | +| dense | 40 | 8 | 1.0000 | 0.9437 | 0.9222 | 0.1333 | 1.0000 | 0.0000 | 5312 | 0.0000 | 19.0 | 19 | 18.52 ms | +| hybrid | 5 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2044 | 0.0000 | 5.0 | 19 | 15.00 ms | +| hybrid | 5 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3362 | 0.0000 | 5.0 | 19 | 14.40 ms | +| hybrid | 10 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2063 | 0.0000 | 10.0 | 19 | 18.54 ms | +| hybrid | 10 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3526 | 0.0000 | 10.0 | 19 | 16.13 ms | +| hybrid | 10 | 8 | 1.0000 | 0.9561 | 0.9389 | 0.1333 | 1.0000 | 0.0000 | 5525 | 0.0000 | 10.0 | 19 | 18.95 ms | +| hybrid | 20 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2057 | 0.0000 | 19.0 | 19 | 21.22 ms | +| hybrid | 20 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3446 | 0.0000 | 19.0 | 19 | 23.23 ms | +| hybrid | 20 | 8 | 1.0000 | 0.9561 | 0.9389 | 0.1333 | 1.0000 | 0.0000 | 5573 | 0.0000 | 19.0 | 19 | 21.75 ms | +| hybrid | 40 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2057 | 0.0000 | 19.0 | 19 | 20.35 ms | +| hybrid | 40 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3446 | 0.0000 | 19.0 | 19 | 20.35 ms | +| hybrid | 40 | 8 | 1.0000 | 0.9561 | 0.9389 | 0.1333 | 1.0000 | 0.0000 | 5573 | 0.0000 | 19.0 | 19 | 20.91 ms | ## Skipped cells @@ -58,7 +58,12 @@ The harness refuses the same combination at the flag level, so a typo fails loud columns are here so it is visible rather than argued about. - **Context chars** is a proxy for prompt size, not a token count: the harness pins the embedding model, not any generation model's tokenizer. -- **No-result** is the share of questions whose retrieval returned nothing at all. +- **No-result** is the share of *answerable* questions whose retrieval returned nothing — + a miss, and the lower the better. +- **Unans. no-result / retrieved** are the same idea for the *unanswerable* questions, + where the direction flips: there is no ground truth, so returning nothing is correct and + `retrieved` is how much irrelevant context was pulled in anyway. These two are the + columns a threshold decision should move, and they are kept out of every other column. Best nDCG@10 in this grid: `hybrid` candidateK=5, contextK=3 (0.9561). diff --git a/docs/eval/threshold-v1.6.json b/docs/eval/threshold-v1.6.json index 636e5b9..7b4931f 100644 --- a/docs/eval/threshold-v1.6.json +++ b/docs/eval/threshold-v1.6.json @@ -7,175 +7,255 @@ { "threshold": 0, "validation": { - "questions": 10, + "questions": 13, + "answerableQuestions": 10, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 19, + "unanswerable": { + "questions": 3, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 19 + }, "metrics": { - "recallAt1": 0.9, + "recallAt1": 0.75, "recallAt5": 1, "recallAt10": 1, - "mrr": 0.933333, - "ndcgAt10": 0.95, + "mrr": 0.9, + "ndcgAt10": 0.918158, "hitRateAt5": 1, - "mapAt10": 0.933333, - "evidencePrecisionAt5": 0.2 + "mapAt10": 0.883333, + "contextPrecision": 0.366667, + "contextRecall": 1 } }, "test": { - "questions": 20, + "questions": 23, + "answerableQuestions": 20, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 19, + "unanswerable": { + "questions": 3, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 19 + }, "metrics": { - "recallAt1": 0.8, + "recallAt1": 0.875, "recallAt5": 1, "recallAt10": 1, - "mrr": 0.925, - "ndcgAt10": 0.940626, + "mrr": 0.941667, + "ndcgAt10": 0.956546, "hitRateAt5": 1, - "mapAt10": 0.916667, - "evidencePrecisionAt5": 0.22 + "mapAt10": 0.941667, + "contextPrecision": 0.35, + "contextRecall": 1 } } }, { "threshold": 0.3, "validation": { - "questions": 10, + "questions": 13, + "answerableQuestions": 10, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 19, + "unanswerable": { + "questions": 3, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 19 + }, "metrics": { - "recallAt1": 0.9, + "recallAt1": 0.75, "recallAt5": 1, "recallAt10": 1, - "mrr": 0.933333, - "ndcgAt10": 0.95, + "mrr": 0.9, + "ndcgAt10": 0.918158, "hitRateAt5": 1, - "mapAt10": 0.933333, - "evidencePrecisionAt5": 0.2 + "mapAt10": 0.883333, + "contextPrecision": 0.366667, + "contextRecall": 1 } }, "test": { - "questions": 20, + "questions": 23, + "answerableQuestions": 20, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 19, + "unanswerable": { + "questions": 3, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 19 + }, "metrics": { - "recallAt1": 0.8, + "recallAt1": 0.875, "recallAt5": 1, "recallAt10": 1, - "mrr": 0.925, - "ndcgAt10": 0.940626, + "mrr": 0.941667, + "ndcgAt10": 0.956546, "hitRateAt5": 1, - "mapAt10": 0.916667, - "evidencePrecisionAt5": 0.22 + "mapAt10": 0.941667, + "contextPrecision": 0.35, + "contextRecall": 1 } } }, { "threshold": 0.4, "validation": { - "questions": 10, + "questions": 13, + "answerableQuestions": 10, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 19, + "unanswerable": { + "questions": 3, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 19 + }, "metrics": { - "recallAt1": 0.9, + "recallAt1": 0.75, "recallAt5": 1, "recallAt10": 1, - "mrr": 0.933333, - "ndcgAt10": 0.95, + "mrr": 0.9, + "ndcgAt10": 0.918158, "hitRateAt5": 1, - "mapAt10": 0.933333, - "evidencePrecisionAt5": 0.2 + "mapAt10": 0.883333, + "contextPrecision": 0.366667, + "contextRecall": 1 } }, "test": { - "questions": 20, + "questions": 23, + "answerableQuestions": 20, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 19, + "unanswerable": { + "questions": 3, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 19 + }, "metrics": { - "recallAt1": 0.8, + "recallAt1": 0.875, "recallAt5": 1, "recallAt10": 1, - "mrr": 0.925, - "ndcgAt10": 0.940626, + "mrr": 0.941667, + "ndcgAt10": 0.956546, "hitRateAt5": 1, - "mapAt10": 0.916667, - "evidencePrecisionAt5": 0.22 + "mapAt10": 0.941667, + "contextPrecision": 0.35, + "contextRecall": 1 } } }, { "threshold": 0.5, "validation": { - "questions": 10, + "questions": 13, + "answerableQuestions": 10, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 19, + "unanswerable": { + "questions": 3, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 19 + }, "metrics": { - "recallAt1": 0.9, + "recallAt1": 0.75, "recallAt5": 1, "recallAt10": 1, - "mrr": 0.933333, - "ndcgAt10": 0.95, + "mrr": 0.9, + "ndcgAt10": 0.918158, "hitRateAt5": 1, - "mapAt10": 0.933333, - "evidencePrecisionAt5": 0.2 + "mapAt10": 0.883333, + "contextPrecision": 0.366667, + "contextRecall": 1 } }, "test": { - "questions": 20, + "questions": 23, + "answerableQuestions": 20, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 19, + "unanswerable": { + "questions": 3, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 19 + }, "metrics": { - "recallAt1": 0.8, + "recallAt1": 0.875, "recallAt5": 1, "recallAt10": 1, - "mrr": 0.925, - "ndcgAt10": 0.940626, + "mrr": 0.941667, + "ndcgAt10": 0.956546, "hitRateAt5": 1, - "mapAt10": 0.916667, - "evidencePrecisionAt5": 0.22 + "mapAt10": 0.941667, + "contextPrecision": 0.35, + "contextRecall": 1 } } }, { "threshold": 0.6, "validation": { - "questions": 10, + "questions": 13, + "answerableQuestions": 10, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 19, + "unanswerable": { + "questions": 3, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 19 + }, "metrics": { - "recallAt1": 0.9, + "recallAt1": 0.75, "recallAt5": 1, "recallAt10": 1, - "mrr": 0.933333, - "ndcgAt10": 0.95, + "mrr": 0.9, + "ndcgAt10": 0.918158, "hitRateAt5": 1, - "mapAt10": 0.933333, - "evidencePrecisionAt5": 0.2 + "mapAt10": 0.883333, + "contextPrecision": 0.366667, + "contextRecall": 1 } }, "test": { - "questions": 20, + "questions": 23, + "answerableQuestions": 20, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 19, + "unanswerable": { + "questions": 3, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 19 + }, "metrics": { - "recallAt1": 0.8, + "recallAt1": 0.875, "recallAt5": 1, "recallAt10": 1, - "mrr": 0.925, - "ndcgAt10": 0.940626, + "mrr": 0.941667, + "ndcgAt10": 0.956546, "hitRateAt5": 1, - "mapAt10": 0.916667, - "evidencePrecisionAt5": 0.22 + "mapAt10": 0.941667, + "contextPrecision": 0.35, + "contextRecall": 1 } } } diff --git a/docs/eval/threshold-v1.6.md b/docs/eval/threshold-v1.6.md index c9f5c1f..700c45c 100644 --- a/docs/eval/threshold-v1.6.md +++ b/docs/eval/threshold-v1.6.md @@ -6,41 +6,42 @@ Generated by `node scripts/eval-threshold.mjs`. Numbers are harness output; do n The real harness, the same corpus and the production retrieval config (`candidateK=20, contextK=3`), once per candidate threshold. The **validation** -split selects; the **test** split reports. Both come from the same -`--eval-split=` code path, and the split is a deterministic function of the -question id, so this is reproducible. - -| Threshold | n (val) | Recall@5 (val) | nDCG@10 (val) | MAP@10 (val) | No-result (val) | nDCG@10 (test) | No-result (test) | -| --- | --- | --- | --- | --- | --- | --- | --- | -| 0 | 10 | 1.0000 | 0.9500 | 0.9333 | 0.0000 | 0.9406 | 0.0000 | -| 0.3 | 10 | 1.0000 | 0.9500 | 0.9333 | 0.0000 | 0.9406 | 0.0000 | -| 0.4 | 10 | 1.0000 | 0.9500 | 0.9333 | 0.0000 | 0.9406 | 0.0000 | -| 0.5 | 10 | 1.0000 | 0.9500 | 0.9333 | 0.0000 | 0.9406 | 0.0000 | -| 0.6 | 10 | 1.0000 | 0.9500 | 0.9333 | 0.0000 | 0.9406 | 0.0000 | +split selects; the **test** split reports. The split is the committed manifest +`eval/splits.json`, so the same questions are on the same side on every machine. + +Quality columns cover the answerable questions only; **Unans.** columns cover the +unanswerable ones, where returning nothing is the desired outcome and so a *higher* +no-result rate is better. + +| Threshold | n (val) | Recall@5 (val) | nDCG@10 (val) | No-result (val) | Unans. no-result (val) | Unans. retrieved (val) | nDCG@10 (test) | Unans. no-result (test) | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | +| 0 | 10 | 1.0000 | 0.9182 | 0.0000 | 0.0000 | 19.0 | 0.9565 | 0.0000 | +| 0.3 | 10 | 1.0000 | 0.9182 | 0.0000 | 0.0000 | 19.0 | 0.9565 | 0.0000 | +| 0.4 | 10 | 1.0000 | 0.9182 | 0.0000 | 0.0000 | 19.0 | 0.9565 | 0.0000 | +| 0.5 | 10 | 1.0000 | 0.9182 | 0.0000 | 0.0000 | 19.0 | 0.9565 | 0.0000 | +| 0.6 | 10 | 1.0000 | 0.9182 | 0.0000 | 0.0000 | 19.0 | 0.9565 | 0.0000 | ## Selection rule -Best validation nDCG@10, then fewest validation no-results, then the **lowest** -threshold — the first stage is supposed to favour recall and let a later stage -filter, so among equals the wider one is the safer default. +Hold the answerable quality line — validation nDCG@10 and context recall must not +regress versus `threshold = 0` — then take the threshold that refuses the most +unanswerable questions. Tie-break on the lowest threshold. -## Outcome +Raising a threshold is only worth anything if it refuses what the sources do not answer; +the quality gate is there so a refusal gain can never be bought with a retrieval loss. -The sweep is **flat**: every threshold from 0 to 0.6 produces the same validation nDCG@10 (0.9500), the same Recall@5 (1.0000) and a no-result rate of 0.0000. On this corpus the threshold is simply **non-binding** — E5 never scores these query/chunk pairs below the top of the swept range, so no passage is ever filtered out. +## Outcome -**No evidence to change `threshold = 0.5`.** The tie-break rule nominates `0` only because it prefers the widest threshold among equals; that is a tie-break, not a finding. What this run establishes is that the current value cannot be validated *or* falsified here, which is a property of the corpus, not of the threshold. Re-run after #192 child 2 grows it. +The sweep is **flat**: every threshold from 0 to 0.6 produces the same validation nDCG@10 (0.9182), the same Recall@5 (1.0000) and the same unanswerable refusal rate (0/3). No passage is ever filtered out, so the threshold is **non-binding** on this corpus — E5 does not score these query/chunk pairs below the top of the swept range. -The app currently ships `threshold = 0.5`: validation nDCG@10 -0.9500, test nDCG@10 -0.9406, test no-result rate -0.0000. +**No evidence to change `threshold = 0.5`.** All thresholds hold the line equally; picking one would be arbitrary. The current value can be neither validated nor falsified here, which is a property of the corpus rather than of the threshold. ## Caveat on this corpus The split removes the most obvious form of overfitting, but 10 -validation questions is a thin basis for a decision, and the corpus is still small. A -threshold is a product decision with a **no-result-rate** cost attached, so a -recommendation here is only as good as the corpus behind it. Re-run this after the +answerable questions on the validation side is a thin basis for a decision, and the corpus +is still small. A threshold is a product decision with a **refusal-rate** cost attached, so +a recommendation here is only as good as the corpus behind it. Re-run this after the corpus grows (#192 child 2). ## Reproduce diff --git a/eval/README.md b/eval/README.md index 290648f..f01989c 100644 --- a/eval/README.md +++ b/eval/README.md @@ -36,9 +36,14 @@ report describes the whole online path. ### Swept parameters are chosen on `validation`, reported on `test` A parameter picked on the same questions it is scored on is a fitted number, not a -result. `--eval-split=validation` selects roughly a third of the questions by a -deterministic hash of the id; `test` is the rest. `npm run eval:threshold` uses this -to pick a similarity threshold on `validation` and report it on `test`. +result. `eval/splits.json` is the committed assignment; `--eval-split=validation` selects +from it and `test` is the rest. + +It is an explicit manifest rather than a hash of the question id. A hash is reproducible +but not *stable*: adding a question moves others between the sides, and a rare query type +can end up entirely on one side without anyone choosing that. With a manifest, a question +with no entry is **refused** rather than defaulted, so a new question is assigned +deliberately instead of leaking into `test`. `npm run eval:sweep` runs a bounded grid (`strategy × candidateK × contextK`) and writes one dashboard with quality, context precision/recall, prompt size, index size @@ -70,6 +75,7 @@ does not mean "no download". The 134 MB of weights are not committed. eval/ corpus/ first-party documents (markdown today) questions.jsonl one question per line; committed with the corpus + splits.json the committed validation/test assignment ``` The corpus is authored for this repository and carries the repository's GPL-3.0 licence, @@ -83,6 +89,7 @@ dataset needs a new question. "id": "q001", "question": "Why is bedload harder to measure than suspended sediment?", "type": "semantic", + "answerable": true, "relevant": [ { "document": "river-monitoring.md", @@ -95,6 +102,11 @@ dataset needs a new question. } ``` +An **unanswerable** question is the same shape with `"answerable": false` and `"relevant": []`. +The harness refuses the reverse combination in either direction — an answerable question +with no ground truth, or an unanswerable one carrying some — because both are silent: the +first reads as a permanent miss, the second as a normal hit. + Ground truth uses **corpus identity, never database identity**: - `document` is the corpus-relative path. @@ -163,6 +175,12 @@ from unanswerable queries. `semantic`, `multi-hop`, `cross-lingual`, `zh`). A single average hides a change that helps one kind of question and hurts another; the current baseline already shows this, with `cross-lingual` at nDCG 0.63 against 0.93–1.00 elsewhere. +- **Unanswerable questions** — a separate group, never averaged in. They have no ground + truth, so `Recall` on them is 0/0 rather than 0, and the correct outcome is that + retrieval finds nothing. The reported **no-results** rate is the opposite of a miss: +higher is better, and `mean passages retrieved` is how much irrelevant context was pulled + in anyway. This is the only metric a similarity-threshold decision should move, which is + why the threshold sweep reports it separately. - **Latency p50/p95** — informational only. Timing is **not** frozen, and the committed JSON excludes it so two runs diff cleanly. diff --git a/eval/questions.jsonl b/eval/questions.jsonl index de15b0c..d762dd3 100644 --- a/eval/questions.jsonl +++ b/eval/questions.jsonl @@ -28,3 +28,9 @@ {"id":"q028","question":"茶叶储存的相对湿度上限是多少?","relevant":[{"document":"chinese-tea-storage.md","page":null,"block":4,"quote":"相对湿度应保持在百分之五十以下"}],"type":"zh"} {"id":"q029","question":"为什么冷冻保存的茶叶取出后不能立刻打开包装?","relevant":[{"document":"chinese-tea-storage.md","page":null,"block":6,"quote":"冷凝水会直接落在茶叶上"}],"type":"zh"} {"id":"q030","question":"为什么潮汐能比风能和太阳能更容易提前安排发电?","relevant":[{"document":"tidal-energy.md","page":null,"block":2,"quote":"predictable decades ahead, unlike wind or solar"}],"type":"cross-lingual"} +{"id":"q031","question":"What is the installed capacity of the tidal energy installation described?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q032","question":"Which laboratory published the river monitoring protocol?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q033","question":"What is the boiling point of mercury?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q034","question":"What is the retail price of the lithium-ion cells discussed?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q035","question":"How many megawatts does the tidal array generate?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q036","question":"Who won the 2018 FIFA World Cup?","type":"unanswerable","answerable":false,"relevant":[]} diff --git a/eval/splits.json b/eval/splits.json new file mode 100644 index 0000000..a608f76 --- /dev/null +++ b/eval/splits.json @@ -0,0 +1,38 @@ +{ + "q001": "test", + "q002": "validation", + "q003": "test", + "q004": "validation", + "q005": "test", + "q006": "test", + "q007": "test", + "q008": "test", + "q009": "validation", + "q010": "test", + "q011": "validation", + "q012": "test", + "q013": "test", + "q014": "test", + "q015": "validation", + "q016": "test", + "q017": "test", + "q018": "validation", + "q019": "test", + "q020": "test", + "q021": "validation", + "q022": "test", + "q023": "validation", + "q024": "test", + "q025": "test", + "q026": "test", + "q027": "test", + "q028": "validation", + "q029": "test", + "q030": "validation", + "q031": "validation", + "q032": "test", + "q033": "validation", + "q034": "test", + "q035": "validation", + "q036": "test" +} diff --git a/scripts/eval-retrieval.mjs b/scripts/eval-retrieval.mjs index 5964a28..0ed9a6c 100644 --- a/scripts/eval-retrieval.mjs +++ b/scripts/eval-retrieval.mjs @@ -126,6 +126,9 @@ try { label: strategy.label, chunking: `${report.config.chunking.chunkSize}/${report.config.chunking.chunkOverlap}`, chunkCount: report.config.chunkCount, + split: report.config.split, + questions: report.config.questions, + answerableCount: report.byType.reduce((total, entry) => total + entry.questions, 0), ...metrics, ...readTiming(join(outDir, 'baseline-v1.6.md')) }) @@ -176,8 +179,9 @@ Generated by \`node scripts/eval-retrieval.mjs\`. Numbers are harness output; do ## What was measured -Every strategy runs the real RAG eval harness against the same corpus and the same 30 -questions as \`baseline-v1.6.json\`, with chunking held fixed at ${baseline.chunking}. Only the retrieval strategy changes. +Every strategy runs the real RAG eval harness against the same corpus and the same questions +as \`baseline-v1.6.json\` (split \`${baseline.split}\`, ${baseline.questions} questions of which +${baseline.answerableCount} are answerable), with chunking held fixed at ${baseline.chunking}. Only the retrieval strategy changes. | Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Context P | Query p95 | | --- | --- | --- | --- | --- | --- | --- | --- | diff --git a/scripts/eval-sweep.mjs b/scripts/eval-sweep.mjs index 46faefd..45a3eb0 100644 --- a/scripts/eval-sweep.mjs +++ b/scripts/eval-sweep.mjs @@ -141,7 +141,11 @@ try { mkdirSync(outDir, { recursive: true }) const report = await runOne(strategy, candidateK, contextK, outDir) const perQuestion = report.perQuestion ?? [] - const noResult = perQuestion.filter((q) => q.retrievedCount === 0).length + // 质量列只看可答的问题,拒答列只看不可答的问题。把两者平均到一起,会给出一个 + // 看起来很干净的检索分数,即使同一份语料把“谁赢了 2018 世界杯”用 19 条上下文 + // 回答了(#192 评审)。 + const answerable = perQuestion.filter((q) => q.answerable) + const noResult = answerable.filter((q) => q.retrievedCount === 0).length rows.push({ strategy, @@ -152,8 +156,11 @@ try { mapAt10: report.metrics.mapAt10, contextPrecision: report.metrics.contextPrecision, contextRecall: report.metrics.contextRecall, - noResultRate: perQuestion.length === 0 ? 0 : noResult / perQuestion.length, - meanContextChars: mean(perQuestion.map((q) => q.contextChars)), + noResultRate: answerable.length === 0 ? 0 : noResult / answerable.length, + meanContextChars: mean(answerable.map((q) => q.contextChars)), + unanswerableQuestions: report.unanswerable.questions, + unanswerableNoResultRate: report.unanswerable.noResultRate, + unanswerableMeanRetrieved: report.unanswerable.meanRetrieved, chunkCount: report.config.chunkCount, ...readTiming(join(outDir, 'baseline-v1.6.md')) }) @@ -170,7 +177,8 @@ const tableRows = rows `| ${row.strategy} | ${row.candidateK} | ${row.contextK} | ${format4(row.recallAt5)} | ` + `${format4(row.ndcgAt10)} | ${format4(row.mapAt10)} | ${format4(row.contextPrecision)} | ` + `${format4(row.contextRecall)} | ${format4(row.noResultRate)} | ` + - `${Math.round(row.meanContextChars)} | ${row.chunkCount} | ${row.latencyP95Ms?.toFixed(2) ?? '—'} ms |` + `${Math.round(row.meanContextChars)} | ${format4(row.unanswerableNoResultRate)} | ` + + `${row.unanswerableMeanRetrieved.toFixed(1)} | ${row.chunkCount} | ${row.latencyP95Ms?.toFixed(2) ?? '—'} ms |` ) .join('\n') @@ -203,8 +211,8 @@ The real harness, the same corpus, chunking held fixed, over ${STRATEGIES.join(' / ')} × candidateK {${CANDIDATE_KS.join(', ')}} × contextK {${CONTEXT_KS.join(', ')}} — ${rows.length} runs. Each row differs from its neighbour in one parameter.${skipped.length > 0 ? `\n\n${skipped.length} further cell(s) were **skipped** because \`contextK > candidateK\`; see below.` : ''} -| Strategy | candidateK | contextK | Recall@5 | nDCG@10 | MAP@10 | Context P | Context R | No-result | Context chars | Index | p95 | -| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | +| Strategy | candidateK | contextK | Recall@5 | nDCG@10 | MAP@10 | Context P | Context R | No-result | Context chars | Unans. no-result | Unans. retrieved | Index | p95 | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | ${tableRows} ${skippedNote} @@ -218,7 +226,12 @@ ${skippedNote} columns are here so it is visible rather than argued about. - **Context chars** is a proxy for prompt size, not a token count: the harness pins the embedding model, not any generation model's tokenizer. -- **No-result** is the share of questions whose retrieval returned nothing at all. +- **No-result** is the share of *answerable* questions whose retrieval returned nothing — + a miss, and the lower the better. +- **Unans. no-result / retrieved** are the same idea for the *unanswerable* questions, + where the direction flips: there is no ground truth, so returning nothing is correct and + \`retrieved\` is how much irrelevant context was pulled in anyway. These two are the + columns a threshold decision should move, and they are kept out of every other column. Best nDCG@10 in this grid: \`${bestNdcg.strategy}\` candidateK=${bestNdcg.candidateK}, contextK=${bestNdcg.contextK} (${format4(bestNdcg.ndcgAt10)}). diff --git a/scripts/eval-threshold.mjs b/scripts/eval-threshold.mjs index b6d1c73..8eb9697 100644 --- a/scripts/eval-threshold.mjs +++ b/scripts/eval-threshold.mjs @@ -88,22 +88,29 @@ function runOne(threshold, split, outDir) { } /** - * The frozen metrics plus the one this experiment needs and the baseline does not - * carry: how often a threshold turns a question into "no results at all". + * The frozen metrics plus the two the experiment exists for: how often the threshold + * turns an answerable question into "no results at all", and how often it makes an + * unanswerable one return nothing (which is the desired outcome for those). * - * A higher threshold can look better on ranking metrics while quietly making the - * product answer "not in your sources" more often, and that trade is invisible - * unless it is counted. + * A threshold that scores well on ranking metrics but keeps returning the whole corpus + * for a question the sources do not answer is invisible unless the second number is + * counted, and the two have to be split: one is a miss, the other is a correct refusal. */ function summarize(report) { const perQuestion = report.perQuestion ?? [] - const noResult = perQuestion.filter((q) => q.retrievedCount === 0).length - const retrieved = perQuestion.map((q) => q.retrievedCount) + const answerable = perQuestion.filter((q) => q.answerable) + const noResult = answerable.filter((q) => q.retrievedCount === 0).length return { questions: perQuestion.length, + answerableQuestions: answerable.length, noResultCount: noResult, - noResultRate: perQuestion.length === 0 ? 0 : noResult / perQuestion.length, - meanRetrieved: retrieved.length === 0 ? 0 : retrieved.reduce((a, b) => a + b, 0) / retrieved.length, + noResultRate: answerable.length === 0 ? 0 : noResult / answerable.length, + meanRetrieved: + answerable.length === 0 + ? 0 + : answerable.reduce((a, q) => a + q.retrievedCount, 0) / answerable.length, + /** 不可答问题时希望返回空,所以这里的“高”是好事。 */ + unanswerable: report.unanswerable, metrics: report.metrics } } @@ -132,42 +139,63 @@ try { } /** - * Selection rule, stated so it can be argued with: best validation nDCG@10, then - * fewest validation no-results, then the widest (lowest) threshold — because the - * first stage is supposed to favour recall and let a later stage filter. + * Selection rule, stated so it can be argued with (#192). + * + * Raising the threshold is only worth it if it refuses more of what the sources do not + * answer. So: take the widest threshold that does not regress the answerable quality + * metrics (nDCG@10 and context recall on validation) as the reference, then among the + * thresholds that hold that line, pick the one that refuses the most unanswerable + * questions; break ties on the lowest threshold. + * + * `threshold = 0` is the reference, because it is the arm that keeps the ranking intact. */ -const ranked = [...rows].sort( +const reference = rows.find((row) => row.threshold === 0) ?? rows[0] +const EPSILON = 1e-9 +const holdsTheLine = (row) => + row.validation.metrics.ndcgAt10 >= reference.validation.metrics.ndcgAt10 - EPSILON && + row.validation.metrics.contextRecall >= reference.validation.metrics.contextRecall - EPSILON + +const eligible = rows.filter(holdsTheLine) +const ranked = [...eligible].sort( (a, b) => - b.validation.metrics.ndcgAt10 - a.validation.metrics.ndcgAt10 || - a.validation.noResultRate - b.validation.noResultRate || + b.validation.unanswerable.noResultRate - a.validation.unanswerable.noResultRate || a.threshold - b.threshold ) -const winner = ranked[0] -const production = rows.find((row) => row.threshold === PRODUCTION_THRESHOLD) +/** + * A flat sweep is not a weak recommendation, it is no recommendation: if no threshold + * changes either the answerable quality or the unanswerable refusal, the corpus cannot + * tell them apart and moving a product parameter on that evidence would be noise dressed + * as a result. + */ +const flat = rows.every( + (row) => + row.validation.metrics.ndcgAt10 === reference.validation.metrics.ndcgAt10 && + row.validation.unanswerable.noResultRate === reference.validation.unanswerable.noResultRate +) /** - * A flat sweep is not a weak recommendation, it is no recommendation: if every - * threshold scores the same on both metrics, the corpus cannot tell them apart and - * moving a product parameter on that evidence would be noise dressed as a result. + * Refusals are the point of the second number: `noResultCount` of the unanswerable + * questions came back empty, which is the correct outcome. The rest returned passages the + * sources cannot support. */ -const flat = - rows.every( - (row) => - row.validation.metrics.ndcgAt10 === rows[0].validation.metrics.ndcgAt10 && - row.validation.noResultRate === rows[0].validation.noResultRate - ) +const refusals = (row) => `${row.unanswerable.noResultCount}/${row.unanswerable.questions}` +const describe = (row) => + `nDCG@10 ${format4(row.metrics.ndcgAt10)}, Recall@5 ${format4(row.metrics.recallAt5)}, ` + + `answerable no-result ${format4(row.noResultRate)}, ` + + `unanswerable refused ${refusals(row)}` const outcome = flat - ? `The sweep is **flat**: every threshold from ${THRESHOLDS[0]} to ${THRESHOLDS[THRESHOLDS.length - 1]} produces the same validation nDCG@10 (${format4(rows[0].validation.metrics.ndcgAt10)}), the same Recall@5 (${format4(rows[0].validation.metrics.recallAt5)}) and a no-result rate of ${format4(rows[0].validation.noResultRate)}. On this corpus the threshold is simply **non-binding** — E5 never scores these query/chunk pairs below the top of the swept range, so no passage is ever filtered out.\n\n**No evidence to change \`threshold = ${PRODUCTION_THRESHOLD}\`.** The tie-break rule nominates \`${winner.threshold}\` only because it prefers the widest threshold among equals; that is a tie-break, not a finding. What this run establishes is that the current value cannot be validated *or* falsified here, which is a property of the corpus, not of the threshold. Re-run after #192 child 2 grows it.` - : `**Recommended: \`threshold = ${winner.threshold}\`.**\n\n- Validation: nDCG@10 ${format4(winner.validation.metrics.ndcgAt10)}, Recall@5 ${format4(winner.validation.metrics.recallAt5)}, no-result rate ${format4(winner.validation.noResultRate)} (${winner.validation.noResultCount}/${winner.validation.questions})\n- Test: nDCG@10 ${format4(winner.test.metrics.ndcgAt10)}, Recall@5 ${format4(winner.test.metrics.recallAt5)}, no-result rate ${format4(winner.test.noResultRate)} (${winner.test.noResultCount}/${winner.test.questions})\n- Mean retrieved per question: ${winner.test.meanRetrieved.toFixed(2)} (validation ${winner.validation.meanRetrieved.toFixed(2)})` + ? `The sweep is **flat**: every threshold from ${THRESHOLDS[0]} to ${THRESHOLDS[THRESHOLDS.length - 1]} produces the same validation nDCG@10 (${format4(reference.validation.metrics.ndcgAt10)}), the same Recall@5 (${format4(reference.validation.metrics.recallAt5)}) and the same unanswerable refusal rate (${refusals(reference.validation)}). No passage is ever filtered out, so the threshold is **non-binding** on this corpus — E5 does not score these query/chunk pairs below the top of the swept range.\n\n**No evidence to change \`threshold = ${PRODUCTION_THRESHOLD}\`.** All thresholds hold the line equally; picking one would be arbitrary. The current value can be neither validated nor falsified here, which is a property of the corpus rather than of the threshold.` + : `**Recommended: \`threshold = ${winner.threshold}\`.**\n\n- **Validation**: ${describe(winner.validation)}\n- **Test**: ${describe(winner.test)}\n- Production ships \`${PRODUCTION_THRESHOLD}\`: validation ${describe(production.validation)}.\n\nThe rule held answerable quality at the \`threshold = 0\` level (nDCG@10 and context recall must not regress, on the validation split) and then took the threshold that refuses the most unanswerable questions. So this is a refusal gain, not a quality gain — if answerable quality had fallen, the threshold would have been ineligible regardless of how much it refused.` const tableRows = rows .map( (row) => - `| ${row.threshold} | ${row.validation.questions} | ${format4(row.validation.metrics.recallAt5)} | ` + - `${format4(row.validation.metrics.ndcgAt10)} | ${format4(row.validation.metrics.mapAt10)} | ` + - `${format4(row.validation.noResultRate)} | ${format4(row.test.metrics.ndcgAt10)} | ` + - `${format4(row.test.noResultRate)} |` + `| ${row.threshold} | ${row.validation.answerableQuestions} | ${format4(row.validation.metrics.recallAt5)} | ` + + `${format4(row.validation.metrics.ndcgAt10)} | ${format4(row.validation.noResultRate)} | ` + + `${format4(row.validation.unanswerable.noResultRate)} | ` + + `${row.validation.unanswerable.meanRetrieved.toFixed(1)} | ${format4(row.test.metrics.ndcgAt10)} | ` + + `${format4(row.test.unanswerable.noResultRate)} |` ) .join('\n') @@ -179,35 +207,36 @@ Generated by \`node scripts/eval-threshold.mjs\`. Numbers are harness output; do The real harness, the same corpus and the production retrieval config (\`candidateK=20, contextK=3\`), once per candidate threshold. The **validation** -split selects; the **test** split reports. Both come from the same -\`--eval-split=\` code path, and the split is a deterministic function of the -question id, so this is reproducible. +split selects; the **test** split reports. The split is the committed manifest +\`eval/splits.json\`, so the same questions are on the same side on every machine. + +Quality columns cover the answerable questions only; **Unans.** columns cover the +unanswerable ones, where returning nothing is the desired outcome and so a *higher* +no-result rate is better. -| Threshold | n (val) | Recall@5 (val) | nDCG@10 (val) | MAP@10 (val) | No-result (val) | nDCG@10 (test) | No-result (test) | -| --- | --- | --- | --- | --- | --- | --- | --- | +| Threshold | n (val) | Recall@5 (val) | nDCG@10 (val) | No-result (val) | Unans. no-result (val) | Unans. retrieved (val) | nDCG@10 (test) | Unans. no-result (test) | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | ${tableRows} ## Selection rule -Best validation nDCG@10, then fewest validation no-results, then the **lowest** -threshold — the first stage is supposed to favour recall and let a later stage -filter, so among equals the wider one is the safer default. +Hold the answerable quality line — validation nDCG@10 and context recall must not +regress versus \`threshold = 0\` — then take the threshold that refuses the most +unanswerable questions. Tie-break on the lowest threshold. + +Raising a threshold is only worth anything if it refuses what the sources do not answer; +the quality gate is there so a refusal gain can never be bought with a retrieval loss. ## Outcome ${outcome} -The app currently ships \`threshold = ${PRODUCTION_THRESHOLD}\`: validation nDCG@10 -${format4(production.validation.metrics.ndcgAt10)}, test nDCG@10 -${format4(production.test.metrics.ndcgAt10)}, test no-result rate -${format4(production.test.noResultRate)}. - ## Caveat on this corpus -The split removes the most obvious form of overfitting, but ${rows[0].validation.questions} -validation questions is a thin basis for a decision, and the corpus is still small. A -threshold is a product decision with a **no-result-rate** cost attached, so a -recommendation here is only as good as the corpus behind it. Re-run this after the +The split removes the most obvious form of overfitting, but ${reference.validation.answerableQuestions} +answerable questions on the validation side is a thin basis for a decision, and the corpus +is still small. A threshold is a product decision with a **refusal-rate** cost attached, so +a recommendation here is only as good as the corpus behind it. Re-run this after the corpus grows (#192 child 2). ## Reproduce diff --git a/src/main/eval/harness.ts b/src/main/eval/harness.ts index 418ddf7..3297cc0 100644 --- a/src/main/eval/harness.ts +++ b/src/main/eval/harness.ts @@ -12,7 +12,7 @@ */ import { readdir, readFile } from 'fs/promises' -import { join, posix } from 'path' +import { join, posix, relative } from 'path' import { and, eq } from 'drizzle-orm' import { documentBlocks, notebooks, chunks } from '../db/schema' import type { getDatabase } from '../db' @@ -40,9 +40,10 @@ import type { EvalSplit, EvalTypeBreakdown, QuestionReport, - ResolvedGroundTruth + ResolvedGroundTruth, + SplitAssignment } from './types' -import { selectSplit } from './types' +import { assertQuestionShape, parseSplitAssignment, selectSplit } from './types' type Db = ReturnType @@ -52,6 +53,8 @@ export interface EvalHarnessOptions { /** Repo-relative label recorded in the report, so the JSON is machine-independent. */ corpusLabel: string questionsPath: string + /** 切分清单(`eval/splits.json`)。`split: 'all'` 时不读。 */ + splitsPath: string baseline: string /** 本次只评这一份切分(#192);缺省 `all`。 */ split: EvalSplit @@ -212,6 +215,24 @@ async function indexCorpus( return { documentIds, chunkCount, indexingMs: performance.now() - indexingStarted } } +/** + * 读切分清单。`all` 不需要清单,所以 `all` 的运行不会因为缺清单而失败。 + * + * JSON 解析错误会把文件路径带上:清单是提交在仓库里的,一份写坏的清单应该指向它自己。 + */ +async function loadSplitAssignment(options: EvalHarnessOptions): Promise { + if (options.split === 'all') return {} + + const label = relative(process.cwd(), options.splitsPath).split(/[\\/]/).join('/') + let parsed: unknown + try { + parsed = JSON.parse(await readFile(options.splitsPath, 'utf-8')) + } catch (error) { + throw new Error(`${label} could not be read as JSON: ${String(error)}`) + } + return parseSplitAssignment(parsed, label) +} + export async function runEvalHarness( db: Db, knowledgeService: KnowledgeService, @@ -238,7 +259,12 @@ export async function runEvalHarness( options.chunkOptions ) const allQuestions = parseQuestions(await readFile(options.questionsPath, 'utf-8')) - const questions = selectSplit(allQuestions, options.split) + // 数据集自身的契约先校验:可答必须有 ground truth,不可答必须没有。搞反时指标不会 + // 报错,只会静静地失去意义。 + for (const question of allQuestions) assertQuestionShape(question) + + const assignment = await loadSplitAssignment(options) + const questions = selectSplit(allQuestions, options.split, assignment) if (questions.length === 0) { throw new Error(`eval split "${options.split}" selected no questions from ${options.questionsPath}`) } @@ -276,6 +302,7 @@ export async function runEvalHarness( id: question.id, question: question.question, type: question.type ?? UNTAGGED, + answerable: question.answerable ?? true, firstRelevantRank: firstRelevantRank(matchesByRank), relevantCount: groundTruth.length, retrievedCount: results.length, @@ -286,16 +313,31 @@ export async function runEvalHarness( }) } - const metrics = summarize(perQuestion, options.contextK) + // 不可答的问题不进排名指标:它们没有 ground truth,`recallAtK` 对它们返回的是 0/0 + // 而不是 0,把“该拒答”算成“漏报”会让整张表失真。它们自成一组。 + const answerable = perQuestion.filter((q) => q.answerable) + const unanswerableQuestions = perQuestion.filter((q) => !q.answerable) + + const metrics = summarize(answerable, options.contextK) // 每个类别一行,按类别名排序,所以同一个 JSON 在两次运行之间可 diff。 - const byType: EvalTypeBreakdown[] = [...new Set(perQuestion.map((q) => q.type))] + const byType: EvalTypeBreakdown[] = [...new Set(answerable.map((q) => q.type))] .sort() .map((type) => { - const group = perQuestion.filter((q) => q.type === type) + const group = answerable.filter((q) => q.type === type) return { type, questions: group.length, metrics: summarize(group, options.contextK) } }) + const unanswerableNoResults = unanswerableQuestions.filter((q) => q.retrievedCount === 0).length + const unanswerable = { + questions: unanswerableQuestions.length, + noResultCount: unanswerableNoResults, + // 目标方向与其他指标相反:没有相关资料时,返回空才是对的。 + noResultRate: + unanswerableQuestions.length === 0 ? 0 : unanswerableNoResults / unanswerableQuestions.length, + meanRetrieved: mean(unanswerableQuestions.map((q) => q.retrievedCount)) + } + const chunking = { ...DEFAULT_CHUNK_OPTIONS, ...options.chunkOptions } return { baseline: options.baseline, @@ -321,6 +363,7 @@ export async function runEvalHarness( }, metrics, byType, + unanswerable, timing: { latencyP50Ms: percentile(latencies, 50), latencyP95Ms: percentile(latencies, 95), @@ -371,6 +414,7 @@ export function toDeterministicReport(report: EvalReport): EvalDeterministicRepo config: report.config, metrics: report.metrics, byType: report.byType, + unanswerable: report.unanswerable, perQuestion: report.perQuestion } } diff --git a/src/main/eval/report.ts b/src/main/eval/report.ts index 8f6dfaa..9c2b96f 100644 --- a/src/main/eval/report.ts +++ b/src/main/eval/report.ts @@ -24,6 +24,18 @@ export function renderMarkdown(report: EvalReport): string { ) .join('\n') + const unanswerable = report.unanswerable + // byType only ever covers answerable questions, so its sizes add up to that count. + const answerableCount = report.byType.reduce((total, entry) => total + entry.questions, 0) + const unanswerableNote = + unanswerable.questions === 0 + ? 'This corpus carries **no** unanswerable question yet, so refusal is not measured.\n' + + 'The threshold cannot be tuned against it either: every question is answerable, so\n' + + 'every threshold returns something.' + : `| Unanswerable questions | ${unanswerable.questions} |\n` + + `| Returned no results | ${format(unanswerable.noResultRate)} (${unanswerable.noResultCount}/${unanswerable.questions}) |\n` + + `| Mean passages retrieved | ${unanswerable.meanRetrieved.toFixed(2)} |` + return `# RAG eval baseline — ${report.baseline} Generated by \`${report.generatedBy}\`. The numbers below are harness output — do not edit them by hand. @@ -38,7 +50,7 @@ Generated by \`${report.generatedBy}\`. The numbers below are harness output — | Ranks | \`candidateK=${config.candidateK}, threshold=${config.threshold}\` | | Context width | \`contextK=${config.contextK}\` | | Corpus | \`${config.corpus}\` (${config.documents} documents) | -| Split | \`${config.split}\` (${config.questions} questions) | +| Split | \`${config.split}\` (${config.questions} questions, ${answerableCount} answerable) | | Index size | ${config.chunkCount} chunks | ## Metrics @@ -65,6 +77,18 @@ The type comes from \`type\` in \`questions.jsonl\`; untagged questions report a | --- | --- | --- | --- | --- | --- | ${typeRows} +### Unanswerable questions + +These carry no ground truth, so the correct outcome is that retrieval finds nothing. They +are excluded from every metric above — a missing ground truth is not a miss — and reported +here instead. A higher **no-results** rate is better on this row, which is the opposite of +how it reads everywhere else, and \`mean passages retrieved\` is how much irrelevant context +was pulled in anyway. This is the row a threshold decision should move. + +| Metric | Value | +| --- | --- | +${unanswerableNote} + Timing is informational only and is **not** frozen: indexing ${timing.indexingMs} ms, query p50 ${timing.latencyP50Ms.toFixed(2)} ms, p95 ${timing.latencyP95Ms.toFixed(2)} ms on the machine that produced this file. Timing and index size depend on hardware and on the diff --git a/src/main/eval/run.ts b/src/main/eval/run.ts index d927286..20c13fa 100644 --- a/src/main/eval/run.ts +++ b/src/main/eval/run.ts @@ -141,6 +141,7 @@ export async function runEvalCli(argv: readonly string[] = process.argv): Promis const prepare = argv.includes(EVAL_PREPARE_FLAG) const corpusDir = resolve(readOption(argv, '--eval-corpus=', 'eval/corpus')) const questionsPath = resolve(readOption(argv, '--eval-questions=', 'eval/questions.jsonl')) + const splitsPath = resolve(readOption(argv, '--eval-splits=', 'eval/splits.json')) const outDir = resolve(readOption(argv, '--eval-out=', 'docs/eval')) // The real profile is captured before redirecting: the model cache lives under @@ -190,6 +191,7 @@ export async function runEvalCli(argv: readonly string[] = process.argv): Promis // committed JSON is identical on every machine and checkout. corpusLabel: repoRelative(corpusDir) || 'eval/corpus', questionsPath, + splitsPath, baseline: readOption(argv, '--eval-baseline=', 'v1.6'), split: readSplit(argv), // 默认就是生产配置(#77):先取宽,融合,再把 contextK 条送进 prompt。一个不镜像 diff --git a/src/main/eval/types.ts b/src/main/eval/types.ts index bb7f00e..de162bf 100644 --- a/src/main/eval/types.ts +++ b/src/main/eval/types.ts @@ -24,6 +24,14 @@ export interface EvalQuestion { question: string relevant: EvalRelevantLocation[] goldAnswer?: string + /** + * 这份语料里能不能回答(#192)。缺省 `true`。 + * + * `false` 的问题必须 `relevant: []`:它的正确答案是“资料里没有”,所以既不能拿 + * Recall 去惩罚它,也不能让它的“命中”看起来像成功。它评的是另一件事:该拒答的 + * 时候,检索有没有硬找出一堆相似但无关的上下文。 + */ + answerable?: boolean /** * 查询类别(#192)。自由字符串,因为语料还会长出新类别;报告按出现过的值分组, * 缺省归入 `untagged`。 @@ -43,18 +51,78 @@ export interface EvalQuestion { */ export type EvalSplit = 'all' | 'validation' | 'test' -/** id 分桶,0/1/2;只用于切分,不参与检索。 */ -export function splitBucket(id: string): number { - let hash = 0 - for (const character of id) hash = (hash * 31 + character.charCodeAt(0)) >>> 0 - return hash % 3 +/** 问题 id → 它属于切分的哪一边。 */ +export type SplitAssignment = Record + +/** + * 解析提交在仓库里的切分清单(`eval/splits.json`)。 + * + * 用显式清单而不是 id 哈希(#192 评审):哈希看着确定,但它的确定是“每次结果一样”, + * 不是“每次划分一样”——往 `questions.jsonl` 里加一道题,会把其它题在 validation / + * test 之间挪动,而一个稀有类别(multi-hop、cross-lingual)可以在无人选择的情况下整体 + * 落到某一边。清单让划分是被 review 的,不是被算出来的。 + */ +export function parseSplitAssignment(raw: unknown, source: string): SplitAssignment { + if (!raw || typeof raw !== 'object' || Array.isArray(raw)) { + throw new Error( + `${source} must be a JSON object mapping a question id to "validation" or "test"` + ) + } + + const assignment: SplitAssignment = {} + for (const [id, side] of Object.entries(raw as Record)) { + if (side !== 'validation' && side !== 'test') { + throw new Error( + `${source}: question ${id} is assigned ${JSON.stringify(side)}; ` + + 'expected "validation" or "test"' + ) + } + assignment[id] = side + } + return assignment } -/** validation 是 bucket 0(约 1/3),test 是其余(约 2/3)。 */ -export function selectSplit(questions: EvalQuestion[], split: EvalSplit): EvalQuestion[] { - if (split === 'all') return questions - const wantValidation = split === 'validation' - return questions.filter((question) => (splitBucket(question.id) === 0) === wantValidation) +/** + * 选出一份切分。`all` 不需要清单;`validation`/`test` **必须**在清单里有条目。 + * + * 缺条目就报错,而不是默认归入某一边:一个刚加进来的问题应当先被人工决定属于哪一边, + * 而不是静静泄漏进 test。 + */ +export function selectSplit( + questions: readonly EvalQuestion[], + split: EvalSplit, + assignment: SplitAssignment +): EvalQuestion[] { + if (split === 'all') return [...questions] + return questions.filter((question) => { + const side = assignment[question.id] + if (side === undefined) { + throw new Error( + `question ${question.id} has no entry in the split manifest; assign it to ` + + '"validation" or "test" deliberately rather than letting it default into a side' + ) + } + return side === split + }) +} + +/** + * 数据集自身的契约(#192):可答的必须有 ground truth,不可答的必须没有。 + * + * 两者搞反时指标不会报错,只会静静地失去意义 —— 一个可答但没有 ground truth 的问题 + * 会被当成永远漏报,一个不可答却带着 ground truth 的问题会被当成正常命中。 + */ +export function assertQuestionShape(question: EvalQuestion): void { + const answerable = question.answerable ?? true + if (answerable && question.relevant.length === 0) { + throw new Error(`question ${question.id} is answerable but has no ground truth`) + } + if (!answerable && question.relevant.length > 0) { + throw new Error( + `question ${question.id} is unanswerable but carries ${question.relevant.length} ` + + 'ground-truth location(s)' + ) + } } /** One resolved ground-truth location, after runtime id mapping. */ @@ -109,6 +177,8 @@ export interface QuestionReport { question: string /** 查询类别,与 `EvalQuestion.type` 一致;缺省为 `untagged`。 */ type: string + /** 与 `EvalQuestion.answerable` 一致(缺省 true)。 */ + answerable: boolean firstRelevantRank: number relevantCount: number retrievedCount: number @@ -159,8 +229,21 @@ export interface EvalReport { chunkCount: number } metrics: EvalMetrics - /** 每个查询类别一行;类别来自 `questions.jsonl` 的 `type`。 */ + /** 每个查询类别一行;类别来自 `questions.jsonl` 的 `type`。只含可答的问题。 */ byType: EvalTypeBreakdown[] + /** + * 不可答问题的单独一组(#192)。 + * + * 它们不进 `metrics`/`byType`:没有 ground truth,Recall 对它们是 0/0 而不是 0。 + * 它们评的是“该拒答时有没有硬找”——`noResultRate` 越接近 1 越好(在真的没有相关 + * 资料时返回空),`meanRetrieved` 则是“硬找了多少条相似但无关的上下文”。 + */ + unanswerable: { + questions: number + noResultCount: number + noResultRate: number + meanRetrieved: number + } /** `indexingMs` 只用于 #78 的吞吐比较;它不在确定报告里,也不该成为差异原因。 */ timing: { latencyP50Ms: number; latencyP95Ms: number; indexingMs: number } perQuestion: QuestionReport[] @@ -177,5 +260,6 @@ export interface EvalDeterministicReport { config: EvalReport['config'] metrics: EvalMetrics byType: EvalTypeBreakdown[] + unanswerable: EvalReport['unanswerable'] perQuestion: QuestionReport[] } diff --git a/test/evalHarness.test.ts b/test/evalHarness.test.ts index a5e650b..65a51e1 100644 --- a/test/evalHarness.test.ts +++ b/test/evalHarness.test.ts @@ -12,6 +12,7 @@ const options = (overrides: Record = {}): Record ({ @@ -20,42 +23,83 @@ const question = (id: string): EvalQuestion => ({ relevant: [{ document: 'a.md', page: null, block: 0 }] }) -const questions = Array.from({ length: 30 }, (_, index) => question(`q${String(index + 1).padStart(3, '0')}`)) +const assignment = { q1: 'validation', q2: 'test', q3: 'test' } as const -test('the split buckets only ever return 0, 1 or 2', () => { - for (const q of questions) { - const bucket = splitBucket(q.id) - assert.ok(bucket === 0 || bucket === 1 || bucket === 2, `${q.id} -> ${bucket}`) +test('the manifest maps ids to one of the two sides', () => { + assert.deepEqual(parseSplitAssignment({ q1: 'validation', q2: 'test' }, 'splits.json'), { + q1: 'validation', + q2: 'test' + }) +}) + +test('a manifest that is not an object is rejected, naming the file', () => { + for (const bad of [null, undefined, [], 'validation', 3]) { + assert.throws( + () => parseSplitAssignment(bad, 'eval/splits.json'), + /eval\/splits\.json must be a JSON object/ + ) } }) -test('the split is deterministic, not random', () => { - const first = selectSplit(questions, 'validation').map((q) => q.id) - const second = selectSplit(questions, 'validation').map((q) => q.id) - assert.deepEqual(first, second) +test('an unknown side is rejected and names the question', () => { + assert.throws( + () => parseSplitAssignment({ q7: 'train' }, 'eval/splits.json'), + /question q7 is assigned "train"/ + ) +}) + +test('`all` is the whole set and needs no manifest', () => { + const questions = [question('q1'), question('q2')] + assert.deepEqual( + selectSplit(questions, 'all', {}).map((q) => q.id), + ['q1', 'q2'] + ) }) -test('validation and test partition the questions without overlap', () => { - const validation = selectSplit(questions, 'validation').map((q) => q.id) - const test = selectSplit(questions, 'test').map((q) => q.id) +test('the two sides partition the questions and keep their order', () => { + const questions = [question('q1'), question('q2'), question('q3')] + const validation = selectSplit(questions, 'validation', assignment) + const testSide = selectSplit(questions, 'test', assignment) - assert.equal(validation.length + test.length, questions.length) - assert.equal(new Set([...validation, ...test]).size, questions.length) - // Both sides are non-empty on a 30-question set, so neither arm is vacuous. - assert.ok(validation.length > 0 && test.length > 0) + assert.deepEqual(validation.map((q) => q.id), ['q1']) + assert.deepEqual(testSide.map((q) => q.id), ['q2', 'q3']) + assert.equal(validation.length + testSide.length, questions.length) }) -test('`all` is the whole set, and the selected questions keep their order', () => { - assert.deepEqual( - selectSplit(questions, 'all').map((q) => q.id), - questions.map((q) => q.id) +/** + * A new question must be assigned deliberately. Defaulting it into `test` would leak it + * into the reporting side, which is the side a choice must not be fitted to. + */ +test('a question with no manifest entry is refused, not defaulted', () => { + assert.throws( + () => selectSplit([question('q9')], 'validation', assignment), + /question q9 has no entry in the split manifest/ ) +}) - for (const split of ['validation', 'test'] as EvalSplit[]) { - const selected = selectSplit(questions, split).map((q) => q.id) - const expected = questions - .filter((q) => (splitBucket(q.id) === 0) === (split === 'validation')) - .map((q) => q.id) - assert.deepEqual(selected, expected) - } +/** + * Reversing these two is silent: an answerable question with no ground truth reads as a + * permanent miss, and an unanswerable one with ground truth reads as a normal hit. + */ +test('the dataset contract ties answerability to having ground truth', () => { + const answerable = question('q1') + const unanswerable: EvalQuestion = { id: 'q2', question: 'q2', answerable: false, relevant: [] } + + assert.doesNotThrow(() => assertQuestionShape(answerable)) + assert.doesNotThrow(() => assertQuestionShape(unanswerable)) + + assert.throws( + () => assertQuestionShape({ ...answerable, relevant: [] }), + /question q1 is answerable but has no ground truth/ + ) + assert.throws( + () => assertQuestionShape({ ...unanswerable, relevant: [answerable.relevant[0]] }), + /question q2 is unanswerable but carries 1 ground-truth location/ + ) +}) + +test('omitting `answerable` means answerable', () => { + // The pre-#192 dataset has no `answerable` field at all, and every one of those + // questions is answerable; the default has to preserve that. + assert.doesNotThrow(() => assertQuestionShape(question('q1'))) }) From 3f60d7cb48a3b70f33fd513a6c0bcf6e34450851 Mon Sep 17 00:00:00 2001 From: MrSibe Date: Wed, 30 Sep 2026 17:40:16 +0800 Subject: [PATCH 10/18] feat(eval): nine cross-lingual questions, and the corpus stops being saturated MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The corpus had exactly **one** cross-lingual question (`q030`, Chinese over an English source), which is not enough to call anything a weakness — correctly flagged in the #192 review as a signal rather than a finding. This adds eight more Chinese questions over **English** sources, reusing the already-verified ground truth of their English counterparts: the fact is the same, only the query language changes, which is precisely what the multilingual model has to bridge. Cross-lingual goes from n=1 to n=9. ## Recall@5 is no longer 1.0000 ``` before after Recall@5 1.0000 0.9211 nDCG@10 0.9437 0.8476 MAP@10 0.9222 0.7965 hitRate@5 1.0000 0.9211 ``` The benchmark was saturated on this corpus, which is why no retrieval change could ever clear the adoption rule and why `candidateK` appeared to do nothing. Eight questions were enough to bring Recall@5 off the ceiling. ## The finding, at a size that supports it | type | n | Recall@5 | nDCG@10 | MAP@10 | | --- | --- | --- | --- | --- | | **cross-lingual** | **9** | **0.6667** | **0.5032** | **0.3446** | | exact | 6 | 1.0000 | 1.0000 | 1.0000 | | multi-hop | 2 | 1.0000 | 0.9599 | 0.9167 | | semantic | 18 | 1.0000 | 0.9312 | 0.9074 | | zh | 3 | 1.0000 | 1.0000 | 1.0000 | Chinese questions over a **Chinese** source (`zh`) still score 1.0000. Chinese questions over an **English** source score 0.6667. So the gap is specifically **cross-lingual retrieval**, not Chinese, and it is now measured on nine questions rather than asserted from one. This is the kind of finding that was not defensible before: `q030` alone gave nDCG 0.63 and the honest conclusion was "a signal worth testing". With n=9 the same direction holds at the same magnitude, so it is a real weakness of the current pipeline on this corpus. ## What it does to the other reports - **Strategy comparison**: now that `recallAt5` has headroom it becomes the deciding metric again, and **hybrid does not clear the rule** — it matches dense on Recall@5 (0.9211) while improving nDCG@10 (0.8574 vs 0.8476), MRR and MAP@10. So the amended rule gives the conservative answer in the unsaturated regime and the permissive one in the saturated regime; both are reported rather than one being quietly preferred. - **Sweep**: `candidateK` finally moves something — dense nDCG@10 0.8212 at `candidateK=5` versus 0.8476 at 10 and above. `contextK=8` now buys context recall (1.0000 versus 0.9211) at a precision cost (0.3246 → 0.1316), which is the trade the column exists for. - **Threshold**: still flat from 0 to 0.6. Even an out-of-scope question scores above 0.6 against every chunk, so the threshold remains unmeasurable on a 19-chunk corpus. ## Testing - `npm test` — 504 pass - `npm run eval` twice — byte-identical `docs/eval/baseline-v1.6.json` - `npm run eval:retrieval`, `eval:threshold`, `eval:sweep` — all regenerated - `eval/splits.json` extended: 3 of the 8 new questions on `validation`, 5 on `test` ## Scope Hard negatives and near-duplicate *documents* are still not added; this increment is about question coverage over the existing corpus. The corpus is still 19 chunks, which is why `threshold` cannot be measured and `candidateK` saturates at 10. Part of #192 (child 2, stage 2: cross-lingual coverage). --- docs/eval/baseline-v1.6.json | 298 ++++++++++++++++++++++++++++++++-- docs/eval/baseline-v1.6.md | 24 +-- docs/eval/retrieval-v1.6.json | 74 ++++----- docs/eval/retrieval-v1.6.md | 16 +- docs/eval/sweep-v1.6.json | 296 ++++++++++++++++----------------- docs/eval/sweep-v1.6.md | 50 +++--- docs/eval/threshold-v1.6.json | 200 +++++++++++------------ docs/eval/threshold-v1.6.md | 14 +- eval/questions.jsonl | 8 + eval/splits.json | 10 +- 10 files changed, 634 insertions(+), 356 deletions(-) diff --git a/docs/eval/baseline-v1.6.json b/docs/eval/baseline-v1.6.json index 5f4b48a..4c41432 100644 --- a/docs/eval/baseline-v1.6.json +++ b/docs/eval/baseline-v1.6.json @@ -17,34 +17,34 @@ "threshold": 0.5, "corpus": "eval/corpus", "documents": 13, - "questions": 36, + "questions": 44, "chunkCount": 19 }, "metrics": { - "recallAt1": 0.833333, - "recallAt5": 1, + "recallAt1": 0.657895, + "recallAt5": 0.921053, "recallAt10": 1, - "mrr": 0.927778, - "ndcgAt10": 0.94375, - "hitRateAt5": 1, - "mapAt10": 0.922222, - "contextPrecision": 0.355556, - "contextRecall": 1 + "mrr": 0.800909, + "ndcgAt10": 0.84764, + "hitRateAt5": 0.921053, + "mapAt10": 0.796523, + "contextPrecision": 0.324561, + "contextRecall": 0.921053 }, "byType": [ { "type": "cross-lingual", - "questions": 1, + "questions": 9, "metrics": { "recallAt1": 0, - "recallAt5": 1, + "recallAt5": 0.666667, "recallAt10": 1, - "mrr": 0.5, - "ndcgAt10": 0.63093, - "hitRateAt5": 1, - "mapAt10": 0.5, - "contextPrecision": 0.333333, - "contextRecall": 1 + "mrr": 0.344577, + "ndcgAt10": 0.503192, + "hitRateAt5": 0.666667, + "mapAt10": 0.344577, + "contextPrecision": 0.222222, + "contextRecall": 0.666667 } }, { @@ -1294,6 +1294,270 @@ [], [] ] + }, + { + "id": "q037", + "question": "推移质为什么比悬移质更难测?", + "type": "cross-lingual", + "answerable": true, + "firstRelevantRank": 8, + "relevantCount": 1, + "retrievedCount": 19, + "contextChars": 2084, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q038", + "question": "河流监测中,每个站点要采几份平行样?", + "type": "cross-lingual", + "answerable": true, + "firstRelevantRank": 7, + "relevantCount": 1, + "retrievedCount": 19, + "contextChars": 1480, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q039", + "question": "设计锂离子电芯时最核心的取舍是什么?", + "type": "cross-lingual", + "answerable": true, + "firstRelevantRank": 2, + "relevantCount": 1, + "retrievedCount": 19, + "contextChars": 1994, + "matchesByRank": [ + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q040", + "question": "富镍电池为什么对散热要求更高?", + "type": "cross-lingual", + "answerable": true, + "firstRelevantRank": 2, + "relevantCount": 1, + "retrievedCount": 19, + "contextChars": 1994, + "matchesByRank": [ + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q041", + "question": "电芯隔膜一旦熔化会导致什么后果?", + "type": "cross-lingual", + "answerable": true, + "firstRelevantRank": 3, + "relevantCount": 1, + "retrievedCount": 19, + "contextChars": 1449, + "matchesByRank": [ + [], + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q042", + "question": "蜜蜂大致从什么温度开始出巢觅食?", + "type": "cross-lingual", + "answerable": true, + "firstRelevantRank": 3, + "relevantCount": 1, + "retrievedCount": 19, + "contextChars": 1321, + "matchesByRank": [ + [], + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q043", + "question": "花期遭遇晚霜,主要受损的是什么?", + "type": "cross-lingual", + "answerable": true, + "firstRelevantRank": 2, + "relevantCount": 1, + "retrievedCount": 19, + "contextChars": 956, + "matchesByRank": [ + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q044", + "question": "为什么成片的树冠比孤立的树降温效果更好?", + "type": "cross-lingual", + "answerable": true, + "firstRelevantRank": 6, + "relevantCount": 1, + "retrievedCount": 19, + "contextChars": 1447, + "matchesByRank": [ + [], + [], + [], + [], + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] } ] } diff --git a/docs/eval/baseline-v1.6.md b/docs/eval/baseline-v1.6.md index 23fffc2..2fcaf5d 100644 --- a/docs/eval/baseline-v1.6.md +++ b/docs/eval/baseline-v1.6.md @@ -12,22 +12,22 @@ Generated by `npm run eval`. The numbers below are harness output — do not edi | Ranks | `candidateK=20, threshold=0.5` | | Context width | `contextK=3` | | Corpus | `eval/corpus` (13 documents) | -| Split | `all` (36 questions, 30 answerable) | +| Split | `all` (44 questions, 38 answerable) | | Index size | 19 chunks | ## Metrics | Metric | Value | | --- | --- | -| Recall@1 | 0.8333 | -| Recall@5 | 1.0000 | +| Recall@1 | 0.6579 | +| Recall@5 | 0.9211 | | Recall@10 | 1.0000 | -| MRR | 0.9278 | -| nDCG@10 | 0.9437 | -| Hit rate@5 | 1.0000 | -| MAP@10 | 0.9222 | -| Context precision@3 | 0.3556 | -| Context recall@3 | 1.0000 | +| MRR | 0.8009 | +| nDCG@10 | 0.8476 | +| Hit rate@5 | 0.9211 | +| MAP@10 | 0.7965 | +| Context precision@3 | 0.3246 | +| Context recall@3 | 0.9211 | ### By query type @@ -37,7 +37,7 @@ The type comes from `type` in `questions.jsonl`; untagged questions report as | Type | Questions | Recall@5 | nDCG@10 | Hit rate@5 | MAP@10 | | --- | --- | --- | --- | --- | --- | -| cross-lingual | 1 | 1.0000 | 0.6309 | 1.0000 | 0.5000 | +| cross-lingual | 9 | 0.6667 | 0.5032 | 0.6667 | 0.3446 | | exact | 6 | 1.0000 | 1.0000 | 1.0000 | 1.0000 | | multi-hop | 2 | 1.0000 | 0.9599 | 1.0000 | 0.9167 | | semantic | 18 | 1.0000 | 0.9312 | 1.0000 | 0.9074 | @@ -57,8 +57,8 @@ was pulled in anyway. This is the row a threshold decision should move. | Returned no results | 0.0000 (0/6) | | Mean passages retrieved | 19.00 | -Timing is informational only and is **not** frozen: indexing 1496 ms, query -p50 11.42 ms, p95 17.31 ms on the +Timing is informational only and is **not** frozen: indexing 1552 ms, query +p50 11.82 ms, p95 14.45 ms on the machine that produced this file. Timing and index size depend on hardware and on the corpus, so they must never be the reason two runs differ. diff --git a/docs/eval/retrieval-v1.6.json b/docs/eval/retrieval-v1.6.json index 9bf0d74..cdfee1c 100644 --- a/docs/eval/retrieval-v1.6.json +++ b/docs/eval/retrieval-v1.6.json @@ -8,19 +8,19 @@ "chunking": "1000/100", "chunkCount": 19, "split": "all", - "questions": 36, - "answerableCount": 30, - "recallAt1": 0.833333, - "recallAt5": 1, + "questions": 44, + "answerableCount": 38, + "recallAt1": 0.657895, + "recallAt5": 0.921053, "recallAt10": 1, - "mrr": 0.927778, - "ndcgAt10": 0.94375, - "hitRateAt5": 1, - "mapAt10": 0.922222, - "contextPrecision": 0.355556, - "contextRecall": 1, - "indexingMs": 1518, - "latencyP95Ms": 14.2 + "mrr": 0.800909, + "ndcgAt10": 0.84764, + "hitRateAt5": 0.921053, + "mapAt10": 0.796523, + "contextPrecision": 0.324561, + "contextRecall": 0.921053, + "indexingMs": 1532, + "latencyP95Ms": 14.38 }, { "id": "sparse", @@ -28,19 +28,19 @@ "chunking": "1000/100", "chunkCount": 19, "split": "all", - "questions": 36, - "answerableCount": 30, - "recallAt1": 0.733333, - "recallAt5": 0.866667, - "recallAt10": 0.866667, - "mrr": 0.805556, - "ndcgAt10": 0.818355, - "hitRateAt5": 0.866667, - "mapAt10": 0.8, - "contextPrecision": 0.311111, - "contextRecall": 0.866667, - "indexingMs": 1513, - "latencyP95Ms": 2.46 + "questions": 44, + "answerableCount": 38, + "recallAt1": 0.578947, + "recallAt5": 0.684211, + "recallAt10": 0.684211, + "mrr": 0.635965, + "ndcgAt10": 0.64607, + "hitRateAt5": 0.684211, + "mapAt10": 0.631579, + "contextPrecision": 0.245614, + "contextRecall": 0.684211, + "indexingMs": 1529, + "latencyP95Ms": 1.85 }, { "id": "hybrid", @@ -48,19 +48,19 @@ "chunking": "1000/100", "chunkCount": 19, "split": "all", - "questions": 36, - "answerableCount": 30, - "recallAt1": 0.866667, - "recallAt5": 1, + "questions": 44, + "answerableCount": 38, + "recallAt1": 0.684211, + "recallAt5": 0.921053, "recallAt10": 1, - "mrr": 0.944444, - "ndcgAt10": 0.956053, - "hitRateAt5": 1, - "mapAt10": 0.938889, - "contextPrecision": 0.355556, - "contextRecall": 1, - "indexingMs": 1557, - "latencyP95Ms": 19.55 + "mrr": 0.814066, + "ndcgAt10": 0.857352, + "hitRateAt5": 0.921053, + "mapAt10": 0.80968, + "contextPrecision": 0.324561, + "contextRecall": 0.921053, + "indexingMs": 1579, + "latencyP95Ms": 17.04 } ] } diff --git a/docs/eval/retrieval-v1.6.md b/docs/eval/retrieval-v1.6.md index a091483..c00fe52 100644 --- a/docs/eval/retrieval-v1.6.md +++ b/docs/eval/retrieval-v1.6.md @@ -5,14 +5,14 @@ Generated by `node scripts/eval-retrieval.mjs`. Numbers are harness output; do n ## What was measured Every strategy runs the real RAG eval harness against the same corpus and the same questions -as `baseline-v1.6.json` (split `all`, 36 questions of which -30 are answerable), with chunking held fixed at 1000/100. Only the retrieval strategy changes. +as `baseline-v1.6.json` (split `all`, 44 questions of which +38 are answerable), with chunking held fixed at 1000/100. Only the retrieval strategy changes. | Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Context P | Query p95 | | --- | --- | --- | --- | --- | --- | --- | --- | -| dense (vector) | 0.8333 | 1.0000 | 0.9278 | 0.9437 | 0.9222 | 0.3556 | 14.20 ms | -| sparse (BM25) | 0.7333 | 0.8667 | 0.8056 | 0.8184 | 0.8000 | 0.3111 | 2.46 ms | -| hybrid (RRF of dense + BM25) | 0.8667 | 1.0000 | 0.9444 | 0.9561 | 0.9389 | 0.3556 | 19.55 ms | +| dense (vector) | 0.6579 | 0.9211 | 0.8009 | 0.8476 | 0.7965 | 0.3246 | 14.38 ms | +| sparse (BM25) | 0.5789 | 0.6842 | 0.6360 | 0.6461 | 0.6316 | 0.2456 | 1.85 ms | +| hybrid (RRF of dense + BM25) | 0.6842 | 0.9211 | 0.8141 | 0.8574 | 0.8097 | 0.3246 | 17.04 ms | ## Not evaluated @@ -23,7 +23,7 @@ is. ## Saturation -Saturated (no headroom, so they cannot decide anything): `recallAt5`. The deciding metric on this corpus is `ndcgAt10`. A saturated metric is still reported, because "this corpus cannot move it" is itself +No metric in the rule is saturated on this corpus. The deciding metric on this corpus is `recallAt5`. A saturated metric is still reported, because "this corpus cannot move it" is itself information; it is just not allowed to decide the comparison. ## Adoption rule @@ -38,9 +38,7 @@ information; it is just not allowed to decide the comparison. ## Outcome -`hybrid (RRF of dense + BM25)` **clears the rule**: it improves the deciding metric `ndcgAt10` (0.9561 vs dense 0.9437) and regresses none of `recallAt5`, `ndcgAt10`, `mrr`, `mapAt10`. Saturated (no headroom, so they cannot decide anything): `recallAt5`. - -Changing the shipped default is a separate decision, and this script does not make it — it reports the measurement. +No strategy cleared the rule. The deciding metric was `recallAt5` (dense 0.9211); the strategies either failed to improve it or regressed another metric. **Dense stays the default.** A negative result is the point of the experiment: it is the measurement that says the extra machinery is not worth its cost on this corpus, not a failure to deliver. No metric in the rule is saturated on this corpus. ## Reproduce diff --git a/docs/eval/sweep-v1.6.json b/docs/eval/sweep-v1.6.json index 67cdb5b..d027f58 100644 --- a/docs/eval/sweep-v1.6.json +++ b/docs/eval/sweep-v1.6.json @@ -5,375 +5,375 @@ "strategy": "dense", "candidateK": 5, "contextK": 3, - "recallAt5": 1, - "ndcgAt10": 0.94375, - "mapAt10": 0.922222, - "contextPrecision": 0.355556, - "contextRecall": 1, + "recallAt5": 0.921053, + "ndcgAt10": 0.821192, + "mapAt10": 0.785088, + "contextPrecision": 0.324561, + "contextRecall": 0.921053, "noResultRate": 0, - "meanContextChars": 2078.9333333333334, + "meanContextChars": 1976.1315789473683, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 5, "chunkCount": 19, - "latencyP95Ms": 98.32 + "latencyP95Ms": 13.57 }, { "strategy": "dense", "candidateK": 5, "contextK": 5, - "recallAt5": 1, - "ndcgAt10": 0.94375, - "mapAt10": 0.922222, - "contextPrecision": 0.213333, - "contextRecall": 1, + "recallAt5": 0.921053, + "ndcgAt10": 0.821192, + "mapAt10": 0.785088, + "contextPrecision": 0.194737, + "contextRecall": 0.921053, "noResultRate": 0, - "meanContextChars": 3326.8333333333335, + "meanContextChars": 3188.1315789473683, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 5, "chunkCount": 19, - "latencyP95Ms": 96.94 + "latencyP95Ms": 14.14 }, { "strategy": "dense", "candidateK": 10, "contextK": 3, - "recallAt5": 1, - "ndcgAt10": 0.94375, - "mapAt10": 0.922222, - "contextPrecision": 0.355556, - "contextRecall": 1, + "recallAt5": 0.921053, + "ndcgAt10": 0.84764, + "mapAt10": 0.796523, + "contextPrecision": 0.324561, + "contextRecall": 0.921053, "noResultRate": 0, - "meanContextChars": 2078.9333333333334, + "meanContextChars": 1976.1315789473683, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 10, "chunkCount": 19, - "latencyP95Ms": 84.19 + "latencyP95Ms": 13.3 }, { "strategy": "dense", "candidateK": 10, "contextK": 5, - "recallAt5": 1, - "ndcgAt10": 0.94375, - "mapAt10": 0.922222, - "contextPrecision": 0.213333, - "contextRecall": 1, + "recallAt5": 0.921053, + "ndcgAt10": 0.84764, + "mapAt10": 0.796523, + "contextPrecision": 0.194737, + "contextRecall": 0.921053, "noResultRate": 0, - "meanContextChars": 3326.8333333333335, + "meanContextChars": 3188.1315789473683, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 10, "chunkCount": 19, - "latencyP95Ms": 95.07 + "latencyP95Ms": 12.61 }, { "strategy": "dense", "candidateK": 10, "contextK": 8, - "recallAt5": 1, - "ndcgAt10": 0.94375, - "mapAt10": 0.922222, - "contextPrecision": 0.133333, + "recallAt5": 0.921053, + "ndcgAt10": 0.84764, + "mapAt10": 0.796523, + "contextPrecision": 0.131579, "contextRecall": 1, "noResultRate": 0, - "meanContextChars": 5311.633333333333, + "meanContextChars": 5151.289473684211, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 10, "chunkCount": 19, - "latencyP95Ms": 91.67 + "latencyP95Ms": 13.99 }, { "strategy": "dense", "candidateK": 20, "contextK": 3, - "recallAt5": 1, - "ndcgAt10": 0.94375, - "mapAt10": 0.922222, - "contextPrecision": 0.355556, - "contextRecall": 1, + "recallAt5": 0.921053, + "ndcgAt10": 0.84764, + "mapAt10": 0.796523, + "contextPrecision": 0.324561, + "contextRecall": 0.921053, "noResultRate": 0, - "meanContextChars": 2078.9333333333334, + "meanContextChars": 1976.1315789473683, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 16.2 + "latencyP95Ms": 13.61 }, { "strategy": "dense", "candidateK": 20, "contextK": 5, - "recallAt5": 1, - "ndcgAt10": 0.94375, - "mapAt10": 0.922222, - "contextPrecision": 0.213333, - "contextRecall": 1, + "recallAt5": 0.921053, + "ndcgAt10": 0.84764, + "mapAt10": 0.796523, + "contextPrecision": 0.194737, + "contextRecall": 0.921053, "noResultRate": 0, - "meanContextChars": 3326.8333333333335, + "meanContextChars": 3188.1315789473683, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 14.19 + "latencyP95Ms": 14.62 }, { "strategy": "dense", "candidateK": 20, "contextK": 8, - "recallAt5": 1, - "ndcgAt10": 0.94375, - "mapAt10": 0.922222, - "contextPrecision": 0.133333, + "recallAt5": 0.921053, + "ndcgAt10": 0.84764, + "mapAt10": 0.796523, + "contextPrecision": 0.131579, "contextRecall": 1, "noResultRate": 0, - "meanContextChars": 5311.633333333333, + "meanContextChars": 5151.289473684211, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 20.67 + "latencyP95Ms": 15.2 }, { "strategy": "dense", "candidateK": 40, "contextK": 3, - "recallAt5": 1, - "ndcgAt10": 0.94375, - "mapAt10": 0.922222, - "contextPrecision": 0.355556, - "contextRecall": 1, + "recallAt5": 0.921053, + "ndcgAt10": 0.84764, + "mapAt10": 0.796523, + "contextPrecision": 0.324561, + "contextRecall": 0.921053, "noResultRate": 0, - "meanContextChars": 2078.9333333333334, + "meanContextChars": 1976.1315789473683, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 19.91 + "latencyP95Ms": 13.66 }, { "strategy": "dense", "candidateK": 40, "contextK": 5, - "recallAt5": 1, - "ndcgAt10": 0.94375, - "mapAt10": 0.922222, - "contextPrecision": 0.213333, - "contextRecall": 1, + "recallAt5": 0.921053, + "ndcgAt10": 0.84764, + "mapAt10": 0.796523, + "contextPrecision": 0.194737, + "contextRecall": 0.921053, "noResultRate": 0, - "meanContextChars": 3326.8333333333335, + "meanContextChars": 3188.1315789473683, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 18.65 + "latencyP95Ms": 14.4 }, { "strategy": "dense", "candidateK": 40, "contextK": 8, - "recallAt5": 1, - "ndcgAt10": 0.94375, - "mapAt10": 0.922222, - "contextPrecision": 0.133333, + "recallAt5": 0.921053, + "ndcgAt10": 0.84764, + "mapAt10": 0.796523, + "contextPrecision": 0.131579, "contextRecall": 1, "noResultRate": 0, - "meanContextChars": 5311.633333333333, + "meanContextChars": 5151.289473684211, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 18.52 + "latencyP95Ms": 14.39 }, { "strategy": "hybrid", "candidateK": 5, "contextK": 3, - "recallAt5": 1, - "ndcgAt10": 0.956053, - "mapAt10": 0.938889, - "contextPrecision": 0.355556, - "contextRecall": 1, + "recallAt5": 0.921053, + "ndcgAt10": 0.830904, + "mapAt10": 0.798246, + "contextPrecision": 0.324561, + "contextRecall": 0.921053, "noResultRate": 0, - "meanContextChars": 2044.1333333333334, + "meanContextChars": 1948.657894736842, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 5, "chunkCount": 19, - "latencyP95Ms": 15 + "latencyP95Ms": 16.98 }, { "strategy": "hybrid", "candidateK": 5, "contextK": 5, - "recallAt5": 1, - "ndcgAt10": 0.956053, - "mapAt10": 0.938889, - "contextPrecision": 0.213333, - "contextRecall": 1, + "recallAt5": 0.921053, + "ndcgAt10": 0.830904, + "mapAt10": 0.798246, + "contextPrecision": 0.194737, + "contextRecall": 0.921053, "noResultRate": 0, - "meanContextChars": 3361.6, + "meanContextChars": 3215.5789473684213, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 5, "chunkCount": 19, - "latencyP95Ms": 14.4 + "latencyP95Ms": 16.11 }, { "strategy": "hybrid", "candidateK": 10, "contextK": 3, - "recallAt5": 1, - "ndcgAt10": 0.956053, - "mapAt10": 0.938889, - "contextPrecision": 0.355556, - "contextRecall": 1, + "recallAt5": 0.921053, + "ndcgAt10": 0.857352, + "mapAt10": 0.80968, + "contextPrecision": 0.324561, + "contextRecall": 0.921053, "noResultRate": 0, - "meanContextChars": 2062.733333333333, + "meanContextChars": 1963.342105263158, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 10, "chunkCount": 19, - "latencyP95Ms": 18.54 + "latencyP95Ms": 16.46 }, { "strategy": "hybrid", "candidateK": 10, "contextK": 5, - "recallAt5": 1, - "ndcgAt10": 0.956053, - "mapAt10": 0.938889, - "contextPrecision": 0.213333, - "contextRecall": 1, + "recallAt5": 0.921053, + "ndcgAt10": 0.857352, + "mapAt10": 0.80968, + "contextPrecision": 0.194737, + "contextRecall": 0.921053, "noResultRate": 0, - "meanContextChars": 3525.633333333333, + "meanContextChars": 3332.9473684210525, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 10, "chunkCount": 19, - "latencyP95Ms": 16.13 + "latencyP95Ms": 14.83 }, { "strategy": "hybrid", "candidateK": 10, "contextK": 8, - "recallAt5": 1, - "ndcgAt10": 0.956053, - "mapAt10": 0.938889, - "contextPrecision": 0.133333, + "recallAt5": 0.921053, + "ndcgAt10": 0.857352, + "mapAt10": 0.80968, + "contextPrecision": 0.131579, "contextRecall": 1, "noResultRate": 0, - "meanContextChars": 5525.166666666667, + "meanContextChars": 5319.868421052632, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 10, "chunkCount": 19, - "latencyP95Ms": 18.95 + "latencyP95Ms": 17.38 }, { "strategy": "hybrid", "candidateK": 20, "contextK": 3, - "recallAt5": 1, - "ndcgAt10": 0.956053, - "mapAt10": 0.938889, - "contextPrecision": 0.355556, - "contextRecall": 1, + "recallAt5": 0.921053, + "ndcgAt10": 0.857352, + "mapAt10": 0.80968, + "contextPrecision": 0.324561, + "contextRecall": 0.921053, "noResultRate": 0, - "meanContextChars": 2056.9333333333334, + "meanContextChars": 1958.7631578947369, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 21.22 + "latencyP95Ms": 20.14 }, { "strategy": "hybrid", "candidateK": 20, "contextK": 5, - "recallAt5": 1, - "ndcgAt10": 0.956053, - "mapAt10": 0.938889, - "contextPrecision": 0.213333, - "contextRecall": 1, + "recallAt5": 0.921053, + "ndcgAt10": 0.857352, + "mapAt10": 0.80968, + "contextPrecision": 0.194737, + "contextRecall": 0.921053, "noResultRate": 0, - "meanContextChars": 3445.866666666667, + "meanContextChars": 3282.1052631578946, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 23.23 + "latencyP95Ms": 14.23 }, { "strategy": "hybrid", "candidateK": 20, "contextK": 8, - "recallAt5": 1, - "ndcgAt10": 0.956053, - "mapAt10": 0.938889, - "contextPrecision": 0.133333, + "recallAt5": 0.921053, + "ndcgAt10": 0.857352, + "mapAt10": 0.80968, + "contextPrecision": 0.131579, "contextRecall": 1, "noResultRate": 0, - "meanContextChars": 5572.6, + "meanContextChars": 5357.315789473684, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 21.75 + "latencyP95Ms": 14.06 }, { "strategy": "hybrid", "candidateK": 40, "contextK": 3, - "recallAt5": 1, - "ndcgAt10": 0.956053, - "mapAt10": 0.938889, - "contextPrecision": 0.355556, - "contextRecall": 1, + "recallAt5": 0.921053, + "ndcgAt10": 0.857352, + "mapAt10": 0.80968, + "contextPrecision": 0.324561, + "contextRecall": 0.921053, "noResultRate": 0, - "meanContextChars": 2056.9333333333334, + "meanContextChars": 1958.7631578947369, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 20.35 + "latencyP95Ms": 15.39 }, { "strategy": "hybrid", "candidateK": 40, "contextK": 5, - "recallAt5": 1, - "ndcgAt10": 0.956053, - "mapAt10": 0.938889, - "contextPrecision": 0.213333, - "contextRecall": 1, + "recallAt5": 0.921053, + "ndcgAt10": 0.857352, + "mapAt10": 0.80968, + "contextPrecision": 0.194737, + "contextRecall": 0.921053, "noResultRate": 0, - "meanContextChars": 3445.866666666667, + "meanContextChars": 3282.1052631578946, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 20.35 + "latencyP95Ms": 18.4 }, { "strategy": "hybrid", "candidateK": 40, "contextK": 8, - "recallAt5": 1, - "ndcgAt10": 0.956053, - "mapAt10": 0.938889, - "contextPrecision": 0.133333, + "recallAt5": 0.921053, + "ndcgAt10": 0.857352, + "mapAt10": 0.80968, + "contextPrecision": 0.131579, "contextRecall": 1, "noResultRate": 0, - "meanContextChars": 5572.6, + "meanContextChars": 5357.315789473684, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 20.91 + "latencyP95Ms": 18.41 } ] } diff --git a/docs/eval/sweep-v1.6.md b/docs/eval/sweep-v1.6.md index bfd5e48..b7e76cc 100644 --- a/docs/eval/sweep-v1.6.md +++ b/docs/eval/sweep-v1.6.md @@ -12,28 +12,28 @@ Each row differs from its neighbour in one parameter. | Strategy | candidateK | contextK | Recall@5 | nDCG@10 | MAP@10 | Context P | Context R | No-result | Context chars | Unans. no-result | Unans. retrieved | Index | p95 | | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | -| dense | 5 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 0.0000 | 5.0 | 19 | 98.32 ms | -| dense | 5 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 0.0000 | 5.0 | 19 | 96.94 ms | -| dense | 10 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 0.0000 | 10.0 | 19 | 84.19 ms | -| dense | 10 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 0.0000 | 10.0 | 19 | 95.07 ms | -| dense | 10 | 8 | 1.0000 | 0.9437 | 0.9222 | 0.1333 | 1.0000 | 0.0000 | 5312 | 0.0000 | 10.0 | 19 | 91.67 ms | -| dense | 20 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 0.0000 | 19.0 | 19 | 16.20 ms | -| dense | 20 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 0.0000 | 19.0 | 19 | 14.19 ms | -| dense | 20 | 8 | 1.0000 | 0.9437 | 0.9222 | 0.1333 | 1.0000 | 0.0000 | 5312 | 0.0000 | 19.0 | 19 | 20.67 ms | -| dense | 40 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 0.0000 | 19.0 | 19 | 19.91 ms | -| dense | 40 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 0.0000 | 19.0 | 19 | 18.65 ms | -| dense | 40 | 8 | 1.0000 | 0.9437 | 0.9222 | 0.1333 | 1.0000 | 0.0000 | 5312 | 0.0000 | 19.0 | 19 | 18.52 ms | -| hybrid | 5 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2044 | 0.0000 | 5.0 | 19 | 15.00 ms | -| hybrid | 5 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3362 | 0.0000 | 5.0 | 19 | 14.40 ms | -| hybrid | 10 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2063 | 0.0000 | 10.0 | 19 | 18.54 ms | -| hybrid | 10 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3526 | 0.0000 | 10.0 | 19 | 16.13 ms | -| hybrid | 10 | 8 | 1.0000 | 0.9561 | 0.9389 | 0.1333 | 1.0000 | 0.0000 | 5525 | 0.0000 | 10.0 | 19 | 18.95 ms | -| hybrid | 20 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2057 | 0.0000 | 19.0 | 19 | 21.22 ms | -| hybrid | 20 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3446 | 0.0000 | 19.0 | 19 | 23.23 ms | -| hybrid | 20 | 8 | 1.0000 | 0.9561 | 0.9389 | 0.1333 | 1.0000 | 0.0000 | 5573 | 0.0000 | 19.0 | 19 | 21.75 ms | -| hybrid | 40 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2057 | 0.0000 | 19.0 | 19 | 20.35 ms | -| hybrid | 40 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3446 | 0.0000 | 19.0 | 19 | 20.35 ms | -| hybrid | 40 | 8 | 1.0000 | 0.9561 | 0.9389 | 0.1333 | 1.0000 | 0.0000 | 5573 | 0.0000 | 19.0 | 19 | 20.91 ms | +| dense | 5 | 3 | 0.9211 | 0.8212 | 0.7851 | 0.3246 | 0.9211 | 0.0000 | 1976 | 0.0000 | 5.0 | 19 | 13.57 ms | +| dense | 5 | 5 | 0.9211 | 0.8212 | 0.7851 | 0.1947 | 0.9211 | 0.0000 | 3188 | 0.0000 | 5.0 | 19 | 14.14 ms | +| dense | 10 | 3 | 0.9211 | 0.8476 | 0.7965 | 0.3246 | 0.9211 | 0.0000 | 1976 | 0.0000 | 10.0 | 19 | 13.30 ms | +| dense | 10 | 5 | 0.9211 | 0.8476 | 0.7965 | 0.1947 | 0.9211 | 0.0000 | 3188 | 0.0000 | 10.0 | 19 | 12.61 ms | +| dense | 10 | 8 | 0.9211 | 0.8476 | 0.7965 | 0.1316 | 1.0000 | 0.0000 | 5151 | 0.0000 | 10.0 | 19 | 13.99 ms | +| dense | 20 | 3 | 0.9211 | 0.8476 | 0.7965 | 0.3246 | 0.9211 | 0.0000 | 1976 | 0.0000 | 19.0 | 19 | 13.61 ms | +| dense | 20 | 5 | 0.9211 | 0.8476 | 0.7965 | 0.1947 | 0.9211 | 0.0000 | 3188 | 0.0000 | 19.0 | 19 | 14.62 ms | +| dense | 20 | 8 | 0.9211 | 0.8476 | 0.7965 | 0.1316 | 1.0000 | 0.0000 | 5151 | 0.0000 | 19.0 | 19 | 15.20 ms | +| dense | 40 | 3 | 0.9211 | 0.8476 | 0.7965 | 0.3246 | 0.9211 | 0.0000 | 1976 | 0.0000 | 19.0 | 19 | 13.66 ms | +| dense | 40 | 5 | 0.9211 | 0.8476 | 0.7965 | 0.1947 | 0.9211 | 0.0000 | 3188 | 0.0000 | 19.0 | 19 | 14.40 ms | +| dense | 40 | 8 | 0.9211 | 0.8476 | 0.7965 | 0.1316 | 1.0000 | 0.0000 | 5151 | 0.0000 | 19.0 | 19 | 14.39 ms | +| hybrid | 5 | 3 | 0.9211 | 0.8309 | 0.7982 | 0.3246 | 0.9211 | 0.0000 | 1949 | 0.0000 | 5.0 | 19 | 16.98 ms | +| hybrid | 5 | 5 | 0.9211 | 0.8309 | 0.7982 | 0.1947 | 0.9211 | 0.0000 | 3216 | 0.0000 | 5.0 | 19 | 16.11 ms | +| hybrid | 10 | 3 | 0.9211 | 0.8574 | 0.8097 | 0.3246 | 0.9211 | 0.0000 | 1963 | 0.0000 | 10.0 | 19 | 16.46 ms | +| hybrid | 10 | 5 | 0.9211 | 0.8574 | 0.8097 | 0.1947 | 0.9211 | 0.0000 | 3333 | 0.0000 | 10.0 | 19 | 14.83 ms | +| hybrid | 10 | 8 | 0.9211 | 0.8574 | 0.8097 | 0.1316 | 1.0000 | 0.0000 | 5320 | 0.0000 | 10.0 | 19 | 17.38 ms | +| hybrid | 20 | 3 | 0.9211 | 0.8574 | 0.8097 | 0.3246 | 0.9211 | 0.0000 | 1959 | 0.0000 | 19.0 | 19 | 20.14 ms | +| hybrid | 20 | 5 | 0.9211 | 0.8574 | 0.8097 | 0.1947 | 0.9211 | 0.0000 | 3282 | 0.0000 | 19.0 | 19 | 14.23 ms | +| hybrid | 20 | 8 | 0.9211 | 0.8574 | 0.8097 | 0.1316 | 1.0000 | 0.0000 | 5357 | 0.0000 | 19.0 | 19 | 14.06 ms | +| hybrid | 40 | 3 | 0.9211 | 0.8574 | 0.8097 | 0.3246 | 0.9211 | 0.0000 | 1959 | 0.0000 | 19.0 | 19 | 15.39 ms | +| hybrid | 40 | 5 | 0.9211 | 0.8574 | 0.8097 | 0.1947 | 0.9211 | 0.0000 | 3282 | 0.0000 | 19.0 | 19 | 18.40 ms | +| hybrid | 40 | 8 | 0.9211 | 0.8574 | 0.8097 | 0.1316 | 1.0000 | 0.0000 | 5357 | 0.0000 | 19.0 | 19 | 18.41 ms | ## Skipped cells @@ -65,10 +65,10 @@ The harness refuses the same combination at the flag level, so a typo fails loud `retrieved` is how much irrelevant context was pulled in anyway. These two are the columns a threshold decision should move, and they are kept out of every other column. -Best nDCG@10 in this grid: `hybrid` candidateK=5, -contextK=3 (0.9561). +Best nDCG@10 in this grid: `hybrid` candidateK=10, +contextK=3 (0.8574). Best context precision: `dense` candidateK=5, -contextK=3 (0.3556). +contextK=3 (0.3246). These are **not** recommendations. Selecting the grid maximum on the same questions is how a benchmark becomes a lookup table; the adoption rule in `baseline-v1.6.md` diff --git a/docs/eval/threshold-v1.6.json b/docs/eval/threshold-v1.6.json index 7b4931f..468dc8d 100644 --- a/docs/eval/threshold-v1.6.json +++ b/docs/eval/threshold-v1.6.json @@ -7,8 +7,8 @@ { "threshold": 0, "validation": { - "questions": 13, - "answerableQuestions": 10, + "questions": 16, + "answerableQuestions": 13, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 19, @@ -19,20 +19,20 @@ "meanRetrieved": 19 }, "metrics": { - "recallAt1": 0.75, - "recallAt5": 1, + "recallAt1": 0.576923, + "recallAt5": 0.923077, "recallAt10": 1, - "mrr": 0.9, - "ndcgAt10": 0.918158, - "hitRateAt5": 1, - "mapAt10": 0.883333, - "contextPrecision": 0.366667, - "contextRecall": 1 + "mrr": 0.778846, + "ndcgAt10": 0.827608, + "hitRateAt5": 0.923077, + "mapAt10": 0.766026, + "contextPrecision": 0.333333, + "contextRecall": 0.923077 } }, "test": { - "questions": 23, - "answerableQuestions": 20, + "questions": 28, + "answerableQuestions": 25, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 19, @@ -43,23 +43,23 @@ "meanRetrieved": 19 }, "metrics": { - "recallAt1": 0.875, - "recallAt5": 1, + "recallAt1": 0.7, + "recallAt5": 0.92, "recallAt10": 1, - "mrr": 0.941667, - "ndcgAt10": 0.956546, - "hitRateAt5": 1, - "mapAt10": 0.941667, - "contextPrecision": 0.35, - "contextRecall": 1 + "mrr": 0.812381, + "ndcgAt10": 0.858056, + "hitRateAt5": 0.92, + "mapAt10": 0.812381, + "contextPrecision": 0.32, + "contextRecall": 0.92 } } }, { "threshold": 0.3, "validation": { - "questions": 13, - "answerableQuestions": 10, + "questions": 16, + "answerableQuestions": 13, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 19, @@ -70,20 +70,20 @@ "meanRetrieved": 19 }, "metrics": { - "recallAt1": 0.75, - "recallAt5": 1, + "recallAt1": 0.576923, + "recallAt5": 0.923077, "recallAt10": 1, - "mrr": 0.9, - "ndcgAt10": 0.918158, - "hitRateAt5": 1, - "mapAt10": 0.883333, - "contextPrecision": 0.366667, - "contextRecall": 1 + "mrr": 0.778846, + "ndcgAt10": 0.827608, + "hitRateAt5": 0.923077, + "mapAt10": 0.766026, + "contextPrecision": 0.333333, + "contextRecall": 0.923077 } }, "test": { - "questions": 23, - "answerableQuestions": 20, + "questions": 28, + "answerableQuestions": 25, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 19, @@ -94,23 +94,23 @@ "meanRetrieved": 19 }, "metrics": { - "recallAt1": 0.875, - "recallAt5": 1, + "recallAt1": 0.7, + "recallAt5": 0.92, "recallAt10": 1, - "mrr": 0.941667, - "ndcgAt10": 0.956546, - "hitRateAt5": 1, - "mapAt10": 0.941667, - "contextPrecision": 0.35, - "contextRecall": 1 + "mrr": 0.812381, + "ndcgAt10": 0.858056, + "hitRateAt5": 0.92, + "mapAt10": 0.812381, + "contextPrecision": 0.32, + "contextRecall": 0.92 } } }, { "threshold": 0.4, "validation": { - "questions": 13, - "answerableQuestions": 10, + "questions": 16, + "answerableQuestions": 13, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 19, @@ -121,20 +121,20 @@ "meanRetrieved": 19 }, "metrics": { - "recallAt1": 0.75, - "recallAt5": 1, + "recallAt1": 0.576923, + "recallAt5": 0.923077, "recallAt10": 1, - "mrr": 0.9, - "ndcgAt10": 0.918158, - "hitRateAt5": 1, - "mapAt10": 0.883333, - "contextPrecision": 0.366667, - "contextRecall": 1 + "mrr": 0.778846, + "ndcgAt10": 0.827608, + "hitRateAt5": 0.923077, + "mapAt10": 0.766026, + "contextPrecision": 0.333333, + "contextRecall": 0.923077 } }, "test": { - "questions": 23, - "answerableQuestions": 20, + "questions": 28, + "answerableQuestions": 25, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 19, @@ -145,23 +145,23 @@ "meanRetrieved": 19 }, "metrics": { - "recallAt1": 0.875, - "recallAt5": 1, + "recallAt1": 0.7, + "recallAt5": 0.92, "recallAt10": 1, - "mrr": 0.941667, - "ndcgAt10": 0.956546, - "hitRateAt5": 1, - "mapAt10": 0.941667, - "contextPrecision": 0.35, - "contextRecall": 1 + "mrr": 0.812381, + "ndcgAt10": 0.858056, + "hitRateAt5": 0.92, + "mapAt10": 0.812381, + "contextPrecision": 0.32, + "contextRecall": 0.92 } } }, { "threshold": 0.5, "validation": { - "questions": 13, - "answerableQuestions": 10, + "questions": 16, + "answerableQuestions": 13, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 19, @@ -172,20 +172,20 @@ "meanRetrieved": 19 }, "metrics": { - "recallAt1": 0.75, - "recallAt5": 1, + "recallAt1": 0.576923, + "recallAt5": 0.923077, "recallAt10": 1, - "mrr": 0.9, - "ndcgAt10": 0.918158, - "hitRateAt5": 1, - "mapAt10": 0.883333, - "contextPrecision": 0.366667, - "contextRecall": 1 + "mrr": 0.778846, + "ndcgAt10": 0.827608, + "hitRateAt5": 0.923077, + "mapAt10": 0.766026, + "contextPrecision": 0.333333, + "contextRecall": 0.923077 } }, "test": { - "questions": 23, - "answerableQuestions": 20, + "questions": 28, + "answerableQuestions": 25, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 19, @@ -196,23 +196,23 @@ "meanRetrieved": 19 }, "metrics": { - "recallAt1": 0.875, - "recallAt5": 1, + "recallAt1": 0.7, + "recallAt5": 0.92, "recallAt10": 1, - "mrr": 0.941667, - "ndcgAt10": 0.956546, - "hitRateAt5": 1, - "mapAt10": 0.941667, - "contextPrecision": 0.35, - "contextRecall": 1 + "mrr": 0.812381, + "ndcgAt10": 0.858056, + "hitRateAt5": 0.92, + "mapAt10": 0.812381, + "contextPrecision": 0.32, + "contextRecall": 0.92 } } }, { "threshold": 0.6, "validation": { - "questions": 13, - "answerableQuestions": 10, + "questions": 16, + "answerableQuestions": 13, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 19, @@ -223,20 +223,20 @@ "meanRetrieved": 19 }, "metrics": { - "recallAt1": 0.75, - "recallAt5": 1, + "recallAt1": 0.576923, + "recallAt5": 0.923077, "recallAt10": 1, - "mrr": 0.9, - "ndcgAt10": 0.918158, - "hitRateAt5": 1, - "mapAt10": 0.883333, - "contextPrecision": 0.366667, - "contextRecall": 1 + "mrr": 0.778846, + "ndcgAt10": 0.827608, + "hitRateAt5": 0.923077, + "mapAt10": 0.766026, + "contextPrecision": 0.333333, + "contextRecall": 0.923077 } }, "test": { - "questions": 23, - "answerableQuestions": 20, + "questions": 28, + "answerableQuestions": 25, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 19, @@ -247,15 +247,15 @@ "meanRetrieved": 19 }, "metrics": { - "recallAt1": 0.875, - "recallAt5": 1, + "recallAt1": 0.7, + "recallAt5": 0.92, "recallAt10": 1, - "mrr": 0.941667, - "ndcgAt10": 0.956546, - "hitRateAt5": 1, - "mapAt10": 0.941667, - "contextPrecision": 0.35, - "contextRecall": 1 + "mrr": 0.812381, + "ndcgAt10": 0.858056, + "hitRateAt5": 0.92, + "mapAt10": 0.812381, + "contextPrecision": 0.32, + "contextRecall": 0.92 } } } diff --git a/docs/eval/threshold-v1.6.md b/docs/eval/threshold-v1.6.md index 700c45c..45e1f10 100644 --- a/docs/eval/threshold-v1.6.md +++ b/docs/eval/threshold-v1.6.md @@ -15,11 +15,11 @@ no-result rate is better. | Threshold | n (val) | Recall@5 (val) | nDCG@10 (val) | No-result (val) | Unans. no-result (val) | Unans. retrieved (val) | nDCG@10 (test) | Unans. no-result (test) | | --- | --- | --- | --- | --- | --- | --- | --- | --- | -| 0 | 10 | 1.0000 | 0.9182 | 0.0000 | 0.0000 | 19.0 | 0.9565 | 0.0000 | -| 0.3 | 10 | 1.0000 | 0.9182 | 0.0000 | 0.0000 | 19.0 | 0.9565 | 0.0000 | -| 0.4 | 10 | 1.0000 | 0.9182 | 0.0000 | 0.0000 | 19.0 | 0.9565 | 0.0000 | -| 0.5 | 10 | 1.0000 | 0.9182 | 0.0000 | 0.0000 | 19.0 | 0.9565 | 0.0000 | -| 0.6 | 10 | 1.0000 | 0.9182 | 0.0000 | 0.0000 | 19.0 | 0.9565 | 0.0000 | +| 0 | 13 | 0.9231 | 0.8276 | 0.0000 | 0.0000 | 19.0 | 0.8581 | 0.0000 | +| 0.3 | 13 | 0.9231 | 0.8276 | 0.0000 | 0.0000 | 19.0 | 0.8581 | 0.0000 | +| 0.4 | 13 | 0.9231 | 0.8276 | 0.0000 | 0.0000 | 19.0 | 0.8581 | 0.0000 | +| 0.5 | 13 | 0.9231 | 0.8276 | 0.0000 | 0.0000 | 19.0 | 0.8581 | 0.0000 | +| 0.6 | 13 | 0.9231 | 0.8276 | 0.0000 | 0.0000 | 19.0 | 0.8581 | 0.0000 | ## Selection rule @@ -32,13 +32,13 @@ the quality gate is there so a refusal gain can never be bought with a retrieval ## Outcome -The sweep is **flat**: every threshold from 0 to 0.6 produces the same validation nDCG@10 (0.9182), the same Recall@5 (1.0000) and the same unanswerable refusal rate (0/3). No passage is ever filtered out, so the threshold is **non-binding** on this corpus — E5 does not score these query/chunk pairs below the top of the swept range. +The sweep is **flat**: every threshold from 0 to 0.6 produces the same validation nDCG@10 (0.8276), the same Recall@5 (0.9231) and the same unanswerable refusal rate (0/3). No passage is ever filtered out, so the threshold is **non-binding** on this corpus — E5 does not score these query/chunk pairs below the top of the swept range. **No evidence to change `threshold = 0.5`.** All thresholds hold the line equally; picking one would be arbitrary. The current value can be neither validated nor falsified here, which is a property of the corpus rather than of the threshold. ## Caveat on this corpus -The split removes the most obvious form of overfitting, but 10 +The split removes the most obvious form of overfitting, but 13 answerable questions on the validation side is a thin basis for a decision, and the corpus is still small. A threshold is a product decision with a **refusal-rate** cost attached, so a recommendation here is only as good as the corpus behind it. Re-run this after the diff --git a/eval/questions.jsonl b/eval/questions.jsonl index d762dd3..c9a3d44 100644 --- a/eval/questions.jsonl +++ b/eval/questions.jsonl @@ -34,3 +34,11 @@ {"id":"q034","question":"What is the retail price of the lithium-ion cells discussed?","type":"unanswerable","answerable":false,"relevant":[]} {"id":"q035","question":"How many megawatts does the tidal array generate?","type":"unanswerable","answerable":false,"relevant":[]} {"id":"q036","question":"Who won the 2018 FIFA World Cup?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q037","question":"推移质为什么比悬移质更难测?","type":"cross-lingual","relevant":[{"document":"river-monitoring.md","page":null,"block":4,"quote":"Bedload is the harder fraction to measure"}]} +{"id":"q038","question":"河流监测中,每个站点要采几份平行样?","type":"cross-lingual","relevant":[{"document":"river-monitoring.md","page":null,"block":6,"quote":"three replicate samples at each station"}]} +{"id":"q039","question":"设计锂离子电芯时最核心的取舍是什么?","type":"cross-lingual","relevant":[{"document":"battery-chemistry.md","page":null,"block":2,"quote":"trade energy density against thermal stability"}]} +{"id":"q040","question":"富镍电池为什么对散热要求更高?","type":"cross-lingual","relevant":[{"document":"battery-chemistry.md","page":null,"block":4,"quote":"release oxygen at lower temperatures than iron phosphate"}]} +{"id":"q041","question":"电芯隔膜一旦熔化会导致什么后果?","type":"cross-lingual","relevant":[{"document":"battery-chemistry.md","page":null,"block":6,"quote":"Once the separator melts, the cell shorts internally"}]} +{"id":"q042","question":"蜜蜂大致从什么温度开始出巢觅食?","type":"cross-lingual","relevant":[{"document":"orchard-pollination.md","page":null,"block":4,"quote":"above roughly twelve degrees Celsius"}]} +{"id":"q043","question":"花期遭遇晚霜,主要受损的是什么?","type":"cross-lingual","relevant":[{"document":"orchard-pollination.md","page":null,"block":6,"quote":"destroys the flower's ovary rather than the petals"}]} +{"id":"q044","question":"为什么成片的树冠比孤立的树降温效果更好?","type":"cross-lingual","relevant":[{"document":"urban-canopy.md","page":null,"block":4,"quote":"asphalt stores far more heat during the day"}]} diff --git a/eval/splits.json b/eval/splits.json index a608f76..c903be1 100644 --- a/eval/splits.json +++ b/eval/splits.json @@ -34,5 +34,13 @@ "q033": "validation", "q034": "test", "q035": "validation", - "q036": "test" + "q036": "test", + "q037": "validation", + "q038": "test", + "q039": "test", + "q040": "validation", + "q041": "test", + "q042": "test", + "q043": "validation", + "q044": "test" } From b1bdb0bc772716934d336f393d797da187cce5be Mon Sep 17 00:00:00 2001 From: MrSibe Date: Wed, 30 Sep 2026 17:23:22 +0800 Subject: [PATCH 11/18] fix(retrieval): trace the dense channel's threshold for hybrid too MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `hybrid` runs a dense leg, and that leg applies a similarity floor. The trace recorded `threshold: undefined` for hybrid, so a snapshot could not reproduce the retrieval it described: it claimed hybrid had no threshold while `candidateHits()` was quietly applying `request.threshold ?? 0.5`. If hybrid ever becomes the shipped strategy, the #157 promise that a retrieval snapshot is reproducible stops holding. Found in the #192 review. Two separate `?? 0.5` defaults were the cause — one inside `candidateHits()`, one implicit in what the trace wrote. There is now **one** function, `denseChannelThreshold(strategy, requested)`, and the same value is both handed to the dense channel and written to the trace, so the two cannot drift apart again: - `dense` / `hybrid` → the requested threshold, or `DEFAULT_DENSE_THRESHOLD` - `sparse` → `undefined` (BM25 has no similarity floor) The trace field is renamed `threshold` → **`denseThreshold`**, because a bare `threshold` on a `strategy: 'hybrid'` trace reads as "the threshold for all of hybrid", which is the ambiguity that hid this bug. `parseRetrievalSnapshot` reads the legacy `threshold` from already-persisted snapshots as a dense threshold — the value was always the dense leg's, so the backfill states what that retrieval actually did. ## Testing - `npm run typecheck` — clean - `npm test` — 502 pass, with new coverage for the three cases above and a regression test asserting `hybrid` traces `0.5` while `sparse` traces nothing - Verified end to end that `dense` still reports `denseThreshold` in the smoke path ## Not covered `HybridRetriever` needs a vector store, so it has no unit test here; the pure decision is pinned instead and the wiring uses a single variable, which is what makes the divergence impossible rather than merely tested against. Part of #192 (review follow-up). --- src/main/services/retrieval/DenseRetriever.ts | 8 ++--- .../services/retrieval/HybridRetriever.ts | 20 +++++++---- src/main/services/retrieval/trace.ts | 4 +-- src/main/services/retrieval/types.ts | 32 ++++++++++++++++- src/shared/types/chat.ts | 12 +++++-- src/shared/utils/answerSources.ts | 8 +++-- test/answerSources.test.ts | 25 ++++++++++++-- test/retrievalContract.test.ts | 34 ++++++++++++++++--- 8 files changed, 118 insertions(+), 25 deletions(-) diff --git a/src/main/services/retrieval/DenseRetriever.ts b/src/main/services/retrieval/DenseRetriever.ts index 95fece4..dc6e715 100644 --- a/src/main/services/retrieval/DenseRetriever.ts +++ b/src/main/services/retrieval/DenseRetriever.ts @@ -12,7 +12,7 @@ import type { EmbeddingService } from '../EmbeddingService' import type { CandidateHit } from './candidates' import { hydrateEvidence } from './evidence' import { buildRetrievalTrace } from './trace' -import { effectiveCandidateK, DEFAULT_TOP_K } from './types' +import { denseChannelThreshold, effectiveCandidateK, DEFAULT_TOP_K } from './types' import type { RetrievalRequest, RetrievalResult, Retriever } from './types' const STRATEGY = 'dense' @@ -31,7 +31,7 @@ export class DenseRetriever implements Retriever { // 第一阶段按 `candidateK` 取宽;`topK` 的截断由调用方决定,因为 hybrid 需要的是 // 比最终交付更宽的一池子候选。 const candidateK = effectiveCandidateK(request) - const threshold = request.threshold ?? 0.5 + const threshold = denseChannelThreshold('dense', request.threshold) // E5 要求 query 前缀,与索引时的 document 前缀区分 await this.embeddingService.ensureReady() @@ -51,7 +51,7 @@ export class DenseRetriever implements Retriever { async search(request: RetrievalRequest): Promise { const topK = request.topK ?? DEFAULT_TOP_K const candidateK = effectiveCandidateK(request) - const threshold = request.threshold ?? 0.5 + const threshold = denseChannelThreshold('dense', request.threshold) const startedAt = performance.now() // 单策略没有可精排的下游,取宽再截到 `topK` 与直接按 `topK` 查 KNN 等价; @@ -66,7 +66,7 @@ export class DenseRetriever implements Retriever { filter: request.filter, candidateK, topK, - threshold, + denseThreshold: threshold, durationMs: performance.now() - startedAt }) } diff --git a/src/main/services/retrieval/HybridRetriever.ts b/src/main/services/retrieval/HybridRetriever.ts index d0a78b9..7f946bf 100644 --- a/src/main/services/retrieval/HybridRetriever.ts +++ b/src/main/services/retrieval/HybridRetriever.ts @@ -6,6 +6,7 @@ import { DenseRetriever } from './DenseRetriever' import { hydrateEvidence } from './evidence' import { buildRetrievalTrace } from './trace' import { + denseChannelThreshold, effectiveCandidateK, DEFAULT_TOP_K, type RetrievalRequest, @@ -44,9 +45,12 @@ export class HybridRetriever implements Retriever { const startedAt = performance.now() let hits: CandidateHit[] - // BM25 没有「相似度阈值」这个概念,所以 sparse 的 trace 里 threshold 保持缺省, - // 而不是拿 dense 的 0.5 冒充。 - let threshold: number | undefined + // 传给 dense 通道的值和写进 trace 的值是 **同一个** 变量,来自同一个函数。 + // `hybrid` 也跑 dense 所以也有阈值;`sparse` 没有,于是它是 undefined。 + // + // 分开算两次就是 #192 评审发现的 bug:hybrid 的 dense 腿用着 0.5,而 trace 写 + // `undefined`,快照于是声称那次 hybrid 没有阈值。 + const denseThreshold = denseChannelThreshold(strategy, request.threshold) if (strategy === 'sparse') { hits = searchChunksFts(request.notebookId, request.query, { @@ -54,15 +58,17 @@ export class HybridRetriever implements Retriever { documentIds: request.filter?.documentIds }).slice(0, topK) } else if (strategy === 'hybrid') { - const denseHits = await this.dense.candidateHits(request) + const denseHits = await this.dense.candidateHits({ ...request, threshold: denseThreshold }) const sparseHits = searchChunksFts(request.notebookId, request.query, { limit: candidateK, documentIds: request.filter?.documentIds }) hits = rrfFuse([denseHits, sparseHits]).slice(0, topK) } else { - threshold = request.threshold ?? 0.5 - hits = (await this.dense.candidateHits({ ...request, threshold })).slice(0, topK) + hits = (await this.dense.candidateHits({ ...request, threshold: denseThreshold })).slice( + 0, + topK + ) } const evidence = hits.length === 0 ? [] : hydrateEvidence(getDatabase(), hits) @@ -74,7 +80,7 @@ export class HybridRetriever implements Retriever { filter: request.filter, candidateK, topK, - threshold, + denseThreshold, durationMs: performance.now() - startedAt }) } diff --git a/src/main/services/retrieval/trace.ts b/src/main/services/retrieval/trace.ts index ae788cc..70a7c80 100644 --- a/src/main/services/retrieval/trace.ts +++ b/src/main/services/retrieval/trace.ts @@ -5,7 +5,7 @@ export interface RetrievalTraceInput { filter?: RetrievalFilter candidateK: number topK: number - threshold?: number + denseThreshold?: number durationMs: number } @@ -30,7 +30,7 @@ export function buildRetrievalTrace(input: RetrievalTraceInput): RetrievalTrace durationMs: input.durationMs } - if (input.threshold !== undefined) trace.threshold = input.threshold + if (input.denseThreshold !== undefined) trace.denseThreshold = input.denseThreshold return trace } diff --git a/src/main/services/retrieval/types.ts b/src/main/services/retrieval/types.ts index 921a345..fc04fae 100644 --- a/src/main/services/retrieval/types.ts +++ b/src/main/services/retrieval/types.ts @@ -110,6 +110,27 @@ export function effectiveCandidateK(request: { return Math.max(request.candidateK ?? DEFAULT_CANDIDATE_K, topK) } +/** 没有指定时 dense 通道的相似度下限,与引入双 K 之前一致。 */ +export const DEFAULT_DENSE_THRESHOLD = 0.5 + +/** + * 某个策略真正作用在 **dense 通道** 上的相似度下限。 + * + * `hybrid` 也跑 dense,所以它同样有阈值;只有 `sparse` 没有,因为 BM25 没有「相似 + * 度阈值」这个概念。 + * + * 用 **一个** 函数产出这个值,是为了让「传给 dense 通道的值」和「写进 trace 的值」无法 + * 再分开:#192 评审发现的 bug 就是它们各自有一个 `?? 0.5` —— hybrid 的 dense 腿用了 + * 0.5,而 trace 写的 `threshold: undefined`,于是快照无法复现那次检索。 + */ +export function denseChannelThreshold( + strategy: RetrievalStrategy, + requested?: number +): number | undefined { + if (strategy === 'sparse') return undefined + return requested ?? DEFAULT_DENSE_THRESHOLD +} + /** * 一次检索实际生效的参数。 * @@ -127,7 +148,16 @@ export interface RetrievalTrace { candidateK: number /** 最终交付的证据条数。 */ topK: number - threshold?: number + /** + * 真正作用在 **dense 通道** 上的相似度下限。 + * + * 字段名不是 `threshold` 而是 `denseThreshold`,因为 `hybrid` 也在跑 dense:它不 + * 是「没有阈值」,而是 dense 那一路有 0.5。叫 `threshold` 会让快照看上去说 hybrid + * 没有阈值,于是“这次检索是怎么发生的”就复现不出来了(#192 评审)。 + * + * 缺省只表示 **dense 通道没跑**(`sparse`),不是「阈值等于 0」。 + */ + denseThreshold?: number durationMs: number } diff --git a/src/shared/types/chat.ts b/src/shared/types/chat.ts index 55e7170..fdd237b 100644 --- a/src/shared/types/chat.ts +++ b/src/shared/types/chat.ts @@ -162,8 +162,16 @@ export interface RetrievalSnapshot { */ candidateK: number topK: number - /** 缺省表示该策略没有阈值,不是「阈值等于 0」。 */ - threshold?: number + /** + * 作用在 **dense 通道** 上的相似度下限。 + * + * `hybrid` 也有这个值(它的 dense 那一路),所以缺省只表示 dense 通道没跑 + * (`sparse`),不是「阈值等于 0」。 + * + * #192 评审之前这个字段叫 `threshold`:那时 hybrid 的快照写的是 undefined,而它 + * 实际跑了带 0.5 的 dense。旧记录里的值本来就是 dense 阈值,解析时按 dense 阈值读。 + */ + denseThreshold?: number durationMs: number } diff --git a/src/shared/utils/answerSources.ts b/src/shared/utils/answerSources.ts index ffb7271..321b478 100644 --- a/src/shared/utils/answerSources.ts +++ b/src/shared/utils/answerSources.ts @@ -115,8 +115,12 @@ export const parseRetrievalSnapshot = (metadata: unknown): RetrievalSnapshot | n } } - const threshold = toFiniteNumber(candidate.threshold) - if (threshold !== undefined) snapshot.threshold = threshold + // Written as `threshold` before the #192 review established that `hybrid` also runs a + // dense leg. A legacy value is a dense threshold and is read as one; the backfill says + // what that retrieval actually did. + const denseThreshold = + toFiniteNumber(candidate.denseThreshold) ?? toFiniteNumber(candidate.threshold) + if (denseThreshold !== undefined) snapshot.denseThreshold = denseThreshold return snapshot } diff --git a/test/answerSources.test.ts b/test/answerSources.test.ts index 5de2726..3996f39 100644 --- a/test/answerSources.test.ts +++ b/test/answerSources.test.ts @@ -153,7 +153,7 @@ test('a retrieval snapshot round-trips', () => { scope: { documentIds: ['doc_1', 'doc_2'] }, candidateK: 20, topK: 8, - threshold: 0.5, + denseThreshold: 0.5, durationMs: 42.4 } }) @@ -163,7 +163,7 @@ test('a retrieval snapshot round-trips', () => { scope: { documentIds: ['doc_1', 'doc_2'] }, candidateK: 20, topK: 8, - threshold: 0.5, + denseThreshold: 0.5, durationMs: 42.4 }) }) @@ -174,7 +174,26 @@ test('a snapshot without a scope means the whole notebook', () => { }) assert.deepEqual(snapshot?.scope, {}) - assert.equal(snapshot?.threshold, undefined) + assert.equal(snapshot?.denseThreshold, undefined) +}) + +/** + * Snapshots written before the #192 review called the field `threshold`. The value was + * always the dense leg's floor, so it is read as one rather than dropped. + */ +test('a snapshot with the legacy `threshold` field reads it as the dense threshold', () => { + const snapshot = parseRetrievalSnapshot({ + retrievalSnapshot: { + strategy: 'hybrid', + scope: {}, + topK: 3, + threshold: 0.5, + durationMs: 7 + } + }) + + assert.equal(snapshot?.denseThreshold, 0.5) + assert.equal(snapshot?.candidateK, 3) }) /** diff --git a/test/retrievalContract.test.ts b/test/retrievalContract.test.ts index 9d32292..a662346 100644 --- a/test/retrievalContract.test.ts +++ b/test/retrievalContract.test.ts @@ -2,7 +2,9 @@ import { test } from 'node:test' import assert from 'node:assert/strict' import { buildRetrievalTrace } from '../src/main/services/retrieval/trace.ts' import { + denseChannelThreshold, DEFAULT_CANDIDATE_K, + DEFAULT_DENSE_THRESHOLD, effectiveCandidateK } from '../src/main/services/retrieval/types.ts' @@ -20,7 +22,7 @@ test('a trace carries the effective search parameters and no empty fields', () = strategy: 'dense', candidateK: 20, topK: 5, - threshold: 0.5, + denseThreshold: 0.5, durationMs: 12.5 }) @@ -29,16 +31,40 @@ test('a trace carries the effective search parameters and no empty fields', () = scope: {}, candidateK: 20, topK: 5, - threshold: 0.5, + denseThreshold: 0.5, durationMs: 12.5 }) - // An unset threshold means "no threshold", not "threshold 0". + // An absent dense threshold means "no dense leg ran", not "threshold 0". assert.equal( - 'threshold' in buildRetrievalTrace({ strategy: 'dense', candidateK: 20, topK: 5, durationMs: 1 }), + 'denseThreshold' in + buildRetrievalTrace({ strategy: 'sparse', candidateK: 20, topK: 5, durationMs: 1 }), false ) }) +/** + * The bug the #192 review found: `hybrid` runs a dense leg, so it *has* a similarity + * floor. The retriever applied 0.5 to that leg while the trace recorded `undefined`, so + * the snapshot could not reproduce the retrieval it was describing. + */ +test('a strategy traces the threshold its dense leg actually applied', () => { + assert.equal(denseChannelThreshold('dense'), DEFAULT_DENSE_THRESHOLD) + assert.equal(denseChannelThreshold('hybrid'), DEFAULT_DENSE_THRESHOLD) + assert.equal(denseChannelThreshold('hybrid', 0.3), 0.3) + assert.equal(denseChannelThreshold('dense', 0), 0) + // Only a strategy with no dense leg has no dense threshold. + assert.equal(denseChannelThreshold('sparse', 0.3), undefined) + + const hybrid = buildRetrievalTrace({ + strategy: 'hybrid', + candidateK: 20, + topK: 3, + denseThreshold: denseChannelThreshold('hybrid'), + durationMs: 4 + }) + assert.equal(hybrid.denseThreshold, DEFAULT_DENSE_THRESHOLD) +}) + test('the trace keeps the first-stage width and the final count apart', () => { const trace = buildRetrievalTrace({ strategy: 'hybrid', From 84e9e0a27606ed4319445b6c447a067aef091c54 Mon Sep 17 00:00:00 2001 From: MrSibe Date: Wed, 30 Sep 2026 17:50:48 +0800 Subject: [PATCH 12/18] fix(eval): select the strategy on validation, and stop calling candidates "passages" MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Four follow-ups from the #192 review. The first two are semantics that would have spread into corpus v2 if they were left alone. **Strategy adoption is now held out.** `eval-retrieval.mjs` ran with `split = all`, so the strategy was chosen and scored on the same 44 questions — the exact mistake the threshold experiment had already been fixed for. It now: 1. runs every strategy on **validation** and decides there with `decideAdoption`; 2. re-runs the shipped strategy and the selected one on **test**, and only reports them. The choice never sees `test`. If validation selects nothing, `test` reports the shipped strategy alone and the report says there is no adoption candidate. The consequence is a more conservative and more trustworthy result than before: on validation, hybrid is **identical** to dense (Recall@5 0.9231, nDCG@10 0.8276 on both), so nothing is adopted — whereas the `split = all` run had hybrid ahead on nDCG@10. The old number was the choice being scored on its own questions. **`meanRetrieved` was a misleading name.** The harness fetches `candidateK` in order to compute `Recall@10`; the context window is `results.slice(0, contextK)`. So `meanRetrieved = 19` never meant "19 passages go to the model" — it meant "19 candidates passed the threshold". The unanswerable group now reports both sizes: ``` meanCandidatesRetrieved 19 // passed the threshold, capped by candidateK meanContextPassages 3 // actually reach the window, min(candidates, contextK) ``` The accurate description of the FIFA case is therefore: **19/19 chunks pass `threshold: 0.5`, and the top 3 irrelevant ones go into the prompt.** Still a real problem, but not "19 passages are stuffed into the model". **And it is retrieval abstention, not refusal.** No generator runs in this harness, so it can show that nothing passed the threshold; it cannot show that the model would decline to answer. Fields renamed (`abstentionCount`, `retrievalAbstentionRate`) and the report says plainly that a true system refusal rate needs a generator eval. **The manifest rationale was wrong.** The docs claimed a hash of the id "moves other questions between the sides" when one is added. That is false: `hash(id) % 3` is computed per id, so it is stable. The real reasons for an explicit manifest are that a hash cannot stratify a small corpus (which is how `multi-hop` and `cross-lingual` ended up entirely on one side) and that a new question would be assigned silently rather than deliberately. Corrected in `types.ts`, `eval/README.md` and the split tests. Regenerated: baseline, retrieval, threshold and sweep reports. The threshold table now shows `Unans. cands` and `Unans. ctx` as separate columns instead of conflating them. ## Testing - `npm run typecheck` — clean - `npm test` — 504 pass - `npm run eval` twice — byte-identical baseline - `npm run eval:retrieval` — validation selects, test reports; no adoption candidate - `npm run eval:threshold`, `npm run eval:sweep` — regenerated Part of #192 (review follow-ups). --- docs/eval/baseline-v1.6.json | 7 +- docs/eval/baseline-v1.6.md | 26 ++-- docs/eval/retrieval-v1.6.json | 113 ++++++++++------ docs/eval/retrieval-v1.6.md | 43 +++--- docs/eval/sweep-v1.6.json | 156 ++++++++++++---------- docs/eval/sweep-v1.6.md | 60 +++++---- docs/eval/threshold-v1.6.json | 70 +++++----- docs/eval/threshold-v1.6.md | 28 ++-- eval/README.md | 16 ++- scripts/eval-retrieval.mjs | 245 ++++++++++++++++++++-------------- scripts/eval-sweep.mjs | 26 ++-- scripts/eval-threshold.mjs | 42 +++--- src/main/eval/harness.ts | 17 ++- src/main/eval/report.ts | 28 ++-- src/main/eval/types.ts | 36 +++-- test/evalSplit.test.ts | 9 +- 16 files changed, 553 insertions(+), 369 deletions(-) diff --git a/docs/eval/baseline-v1.6.json b/docs/eval/baseline-v1.6.json index 4c41432..e087046 100644 --- a/docs/eval/baseline-v1.6.json +++ b/docs/eval/baseline-v1.6.json @@ -110,9 +110,10 @@ ], "unanswerable": { "questions": 6, - "noResultCount": 0, - "noResultRate": 0, - "meanRetrieved": 19 + "abstentionCount": 0, + "retrievalAbstentionRate": 0, + "meanCandidatesRetrieved": 19, + "meanContextPassages": 3 }, "perQuestion": [ { diff --git a/docs/eval/baseline-v1.6.md b/docs/eval/baseline-v1.6.md index 2fcaf5d..1484d95 100644 --- a/docs/eval/baseline-v1.6.md +++ b/docs/eval/baseline-v1.6.md @@ -45,20 +45,30 @@ The type comes from `type` in `questions.jsonl`; untagged questions report as ### Unanswerable questions -These carry no ground truth, so the correct outcome is that retrieval finds nothing. They +These carry no ground truth, so the correct outcome is that retrieval returns nothing. They are excluded from every metric above — a missing ground truth is not a miss — and reported -here instead. A higher **no-results** rate is better on this row, which is the opposite of -how it reads everywhere else, and `mean passages retrieved` is how much irrelevant context -was pulled in anyway. This is the row a threshold decision should move. +here instead. + +**This measures retrieval-level abstention, not the model refusing.** No generator runs in +this harness, so it can show that no candidate passed the threshold; it cannot show that the +final answer would say "not in your sources". A true system refusal rate needs a +generator eval. + +A higher abstention rate is better on this row, the opposite of how every other row reads. +The two sizes are kept apart on purpose: **candidates passing the threshold** can be as high +as `candidateK` (the harness fetches that many to compute `Recall@10`), while **passages in +the context window** is what a user's prompt would actually receive. A large first number +with a small second one means the threshold filters nothing and the window is all noise. | Metric | Value | | --- | --- | | Unanswerable questions | 6 | -| Returned no results | 0.0000 (0/6) | -| Mean passages retrieved | 19.00 | +| Retrieval abstained | 0.0000 (0/6) | +| Mean candidates passing the threshold | 19.00 | +| Mean passages in the context window | 3.00 | -Timing is informational only and is **not** frozen: indexing 1552 ms, query -p50 11.82 ms, p95 14.45 ms on the +Timing is informational only and is **not** frozen: indexing 1583 ms, query +p50 13.38 ms, p95 16.66 ms on the machine that produced this file. Timing and index size depend on hardware and on the corpus, so they must never be the reason two runs differ. diff --git a/docs/eval/retrieval-v1.6.json b/docs/eval/retrieval-v1.6.json index cdfee1c..62b9139 100644 --- a/docs/eval/retrieval-v1.6.json +++ b/docs/eval/retrieval-v1.6.json @@ -1,66 +1,93 @@ { - "baseline": "dense", - "chunking": "1000/100", - "strategies": [ + "baseline": "v1.6", + "split": { + "selects": "validation", + "reports": "test", + "manifest": "eval/splits.json" + }, + "decision": { + "primary": "recallAt5", + "saturated": [], + "winner": null + }, + "validation": [ { "id": "dense", "label": "dense (vector)", + "split": "validation", + "questions": 16, + "answerableCount": 13, "chunking": "1000/100", "chunkCount": 19, - "split": "all", - "questions": 44, - "answerableCount": 38, - "recallAt1": 0.657895, - "recallAt5": 0.921053, + "recallAt1": 0.576923, + "recallAt5": 0.923077, "recallAt10": 1, - "mrr": 0.800909, - "ndcgAt10": 0.84764, - "hitRateAt5": 0.921053, - "mapAt10": 0.796523, - "contextPrecision": 0.324561, - "contextRecall": 0.921053, - "indexingMs": 1532, - "latencyP95Ms": 14.38 + "mrr": 0.778846, + "ndcgAt10": 0.827608, + "hitRateAt5": 0.923077, + "mapAt10": 0.766026, + "contextPrecision": 0.333333, + "contextRecall": 0.923077, + "latencyP95Ms": 16.02 }, { "id": "sparse", "label": "sparse (BM25)", + "split": "validation", + "questions": 16, + "answerableCount": 13, "chunking": "1000/100", "chunkCount": 19, - "split": "all", - "questions": 44, - "answerableCount": 38, - "recallAt1": 0.578947, - "recallAt5": 0.684211, - "recallAt10": 0.684211, - "mrr": 0.635965, - "ndcgAt10": 0.64607, - "hitRateAt5": 0.684211, - "mapAt10": 0.631579, - "contextPrecision": 0.245614, - "contextRecall": 0.684211, - "indexingMs": 1529, - "latencyP95Ms": 1.85 + "recallAt1": 0.576923, + "recallAt5": 0.615385, + "recallAt10": 0.615385, + "mrr": 0.615385, + "ndcgAt10": 0.609209, + "hitRateAt5": 0.615385, + "mapAt10": 0.602564, + "contextPrecision": 0.230769, + "contextRecall": 0.615385, + "latencyP95Ms": 2.96 }, { "id": "hybrid", "label": "hybrid (RRF of dense + BM25)", + "split": "validation", + "questions": 16, + "answerableCount": 13, "chunking": "1000/100", "chunkCount": 19, - "split": "all", - "questions": 44, - "answerableCount": 38, - "recallAt1": 0.684211, - "recallAt5": 0.921053, + "recallAt1": 0.576923, + "recallAt5": 0.923077, "recallAt10": 1, - "mrr": 0.814066, - "ndcgAt10": 0.857352, - "hitRateAt5": 0.921053, - "mapAt10": 0.80968, - "contextPrecision": 0.324561, - "contextRecall": 0.921053, - "indexingMs": 1579, - "latencyP95Ms": 17.04 + "mrr": 0.778846, + "ndcgAt10": 0.827608, + "hitRateAt5": 0.923077, + "mapAt10": 0.766026, + "contextPrecision": 0.333333, + "contextRecall": 0.923077, + "latencyP95Ms": 19.47 + } + ], + "test": [ + { + "id": "dense", + "label": "dense (vector)", + "split": "test", + "questions": 28, + "answerableCount": 25, + "chunking": "1000/100", + "chunkCount": 19, + "recallAt1": 0.7, + "recallAt5": 0.92, + "recallAt10": 1, + "mrr": 0.812381, + "ndcgAt10": 0.858056, + "hitRateAt5": 0.92, + "mapAt10": 0.812381, + "contextPrecision": 0.32, + "contextRecall": 0.92, + "latencyP95Ms": 14.49 } ] } diff --git a/docs/eval/retrieval-v1.6.md b/docs/eval/retrieval-v1.6.md index c00fe52..9d5009a 100644 --- a/docs/eval/retrieval-v1.6.md +++ b/docs/eval/retrieval-v1.6.md @@ -4,15 +4,27 @@ Generated by `node scripts/eval-retrieval.mjs`. Numbers are harness output; do n ## What was measured -Every strategy runs the real RAG eval harness against the same corpus and the same questions -as `baseline-v1.6.json` (split `all`, 44 questions of which -38 are answerable), with chunking held fixed at 1000/100. Only the retrieval strategy changes. +Each strategy runs the real harness on the **validation** split of +`eval/splits.json`, with chunking held fixed at 1000/100. -| Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Context P | Query p95 | -| --- | --- | --- | --- | --- | --- | --- | --- | -| dense (vector) | 0.6579 | 0.9211 | 0.8009 | 0.8476 | 0.7965 | 0.3246 | 14.38 ms | -| sparse (BM25) | 0.5789 | 0.6842 | 0.6360 | 0.6461 | 0.6316 | 0.2456 | 1.85 ms | -| hybrid (RRF of dense + BM25) | 0.6842 | 0.9211 | 0.8141 | 0.8574 | 0.8097 | 0.3246 | 17.04 ms | +The strategy is then chosen **there**, and only the chosen one (plus the shipped default) +is re-run on **test**. The choice never sees `test`; `test` only reports. The previous +version decided on `split = all`, which scored the choice on the questions it was fitted +to. + +## Validation — this is where the choice happens + +| Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Context P | n | p95 | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | +| dense (vector) | 0.5769 | 0.9231 | 0.7788 | 0.8276 | 0.7660 | 0.3333 | 16 | 16.02 ms | +| sparse (BM25) | 0.5769 | 0.6154 | 0.6154 | 0.6092 | 0.6026 | 0.2308 | 16 | 2.96 ms | +| hybrid (RRF of dense + BM25) | 0.5769 | 0.9231 | 0.7788 | 0.8276 | 0.7660 | 0.3333 | 16 | 19.47 ms | + +## Test — reported, not selected + +| Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Context P | n | p95 | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | +| dense (vector) | 0.7000 | 0.9200 | 0.8124 | 0.8581 | 0.8124 | 0.3200 | 28 | 14.49 ms | ## Not evaluated @@ -21,11 +33,6 @@ model is not available offline and inventing its numbers would defeat the point the harness. It stays open until a model can be pinned the way the embedding model is. -## Saturation - -No metric in the rule is saturated on this corpus. The deciding metric on this corpus is `recallAt5`. A saturated metric is still reported, because "this corpus cannot move it" is itself -information; it is just not allowed to decide the comparison. - ## Adoption rule > Adopt a strategy when it improves the **first metric with headroom** — in the order @@ -33,12 +40,16 @@ information; it is just not allowed to decide the comparison. > already at its maximum has no headroom and cannot decide anything; a rule that depends > on one is unsatisfiable, not strict (#192 child 10). > -> A change that trades a large latency increase for a marginal quality gain is a product -> decision, not an automatic win. +> The rule is applied on `validation`. A change that trades a large latency increase for a +> marginal quality gain is a product decision, not an automatic win. ## Outcome -No strategy cleared the rule. The deciding metric was `recallAt5` (dense 0.9211); the strategies either failed to improve it or regressed another metric. **Dense stays the default.** A negative result is the point of the experiment: it is the measurement that says the extra machinery is not worth its cost on this corpus, not a failure to deliver. No metric in the rule is saturated on this corpus. +**No strategy cleared the rule on validation**, so there is no adoption candidate and +`test` reports the shipped strategy only. No metric in the rule is saturated on the validation split. + +A negative result is the point of the experiment: it is the measurement that says the extra +machinery is not worth its cost on this corpus, not a failure to deliver. ## Reproduce diff --git a/docs/eval/sweep-v1.6.json b/docs/eval/sweep-v1.6.json index d027f58..985549e 100644 --- a/docs/eval/sweep-v1.6.json +++ b/docs/eval/sweep-v1.6.json @@ -13,10 +13,11 @@ "noResultRate": 0, "meanContextChars": 1976.1315789473683, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 5, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 5, + "unanswerableContextPassages": 3, "chunkCount": 19, - "latencyP95Ms": 13.57 + "latencyP95Ms": 14.04 }, { "strategy": "dense", @@ -30,10 +31,11 @@ "noResultRate": 0, "meanContextChars": 3188.1315789473683, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 5, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 5, + "unanswerableContextPassages": 5, "chunkCount": 19, - "latencyP95Ms": 14.14 + "latencyP95Ms": 13.63 }, { "strategy": "dense", @@ -47,10 +49,11 @@ "noResultRate": 0, "meanContextChars": 1976.1315789473683, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 10, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 10, + "unanswerableContextPassages": 3, "chunkCount": 19, - "latencyP95Ms": 13.3 + "latencyP95Ms": 13.84 }, { "strategy": "dense", @@ -64,10 +67,11 @@ "noResultRate": 0, "meanContextChars": 3188.1315789473683, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 10, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 10, + "unanswerableContextPassages": 5, "chunkCount": 19, - "latencyP95Ms": 12.61 + "latencyP95Ms": 15.21 }, { "strategy": "dense", @@ -81,10 +85,11 @@ "noResultRate": 0, "meanContextChars": 5151.289473684211, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 10, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 10, + "unanswerableContextPassages": 8, "chunkCount": 19, - "latencyP95Ms": 13.99 + "latencyP95Ms": 16.35 }, { "strategy": "dense", @@ -98,10 +103,11 @@ "noResultRate": 0, "meanContextChars": 1976.1315789473683, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 19, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 19, + "unanswerableContextPassages": 3, "chunkCount": 19, - "latencyP95Ms": 13.61 + "latencyP95Ms": 16.45 }, { "strategy": "dense", @@ -115,10 +121,11 @@ "noResultRate": 0, "meanContextChars": 3188.1315789473683, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 19, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 19, + "unanswerableContextPassages": 5, "chunkCount": 19, - "latencyP95Ms": 14.62 + "latencyP95Ms": 17.91 }, { "strategy": "dense", @@ -132,10 +139,11 @@ "noResultRate": 0, "meanContextChars": 5151.289473684211, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 19, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 19, + "unanswerableContextPassages": 8, "chunkCount": 19, - "latencyP95Ms": 15.2 + "latencyP95Ms": 14.13 }, { "strategy": "dense", @@ -149,10 +157,11 @@ "noResultRate": 0, "meanContextChars": 1976.1315789473683, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 19, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 19, + "unanswerableContextPassages": 3, "chunkCount": 19, - "latencyP95Ms": 13.66 + "latencyP95Ms": 13.35 }, { "strategy": "dense", @@ -166,10 +175,11 @@ "noResultRate": 0, "meanContextChars": 3188.1315789473683, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 19, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 19, + "unanswerableContextPassages": 5, "chunkCount": 19, - "latencyP95Ms": 14.4 + "latencyP95Ms": 12.73 }, { "strategy": "dense", @@ -183,10 +193,11 @@ "noResultRate": 0, "meanContextChars": 5151.289473684211, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 19, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 19, + "unanswerableContextPassages": 8, "chunkCount": 19, - "latencyP95Ms": 14.39 + "latencyP95Ms": 15.12 }, { "strategy": "hybrid", @@ -200,10 +211,11 @@ "noResultRate": 0, "meanContextChars": 1948.657894736842, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 5, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 5, + "unanswerableContextPassages": 3, "chunkCount": 19, - "latencyP95Ms": 16.98 + "latencyP95Ms": 14.01 }, { "strategy": "hybrid", @@ -217,10 +229,11 @@ "noResultRate": 0, "meanContextChars": 3215.5789473684213, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 5, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 5, + "unanswerableContextPassages": 5, "chunkCount": 19, - "latencyP95Ms": 16.11 + "latencyP95Ms": 15.81 }, { "strategy": "hybrid", @@ -234,10 +247,11 @@ "noResultRate": 0, "meanContextChars": 1963.342105263158, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 10, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 10, + "unanswerableContextPassages": 3, "chunkCount": 19, - "latencyP95Ms": 16.46 + "latencyP95Ms": 15.74 }, { "strategy": "hybrid", @@ -249,12 +263,13 @@ "contextPrecision": 0.194737, "contextRecall": 0.921053, "noResultRate": 0, - "meanContextChars": 3332.9473684210525, + "meanContextChars": 3345.0789473684213, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 10, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 10, + "unanswerableContextPassages": 5, "chunkCount": 19, - "latencyP95Ms": 14.83 + "latencyP95Ms": 16.3 }, { "strategy": "hybrid", @@ -268,10 +283,11 @@ "noResultRate": 0, "meanContextChars": 5319.868421052632, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 10, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 10, + "unanswerableContextPassages": 8, "chunkCount": 19, - "latencyP95Ms": 17.38 + "latencyP95Ms": 16.95 }, { "strategy": "hybrid", @@ -285,10 +301,11 @@ "noResultRate": 0, "meanContextChars": 1958.7631578947369, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 19, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 19, + "unanswerableContextPassages": 3, "chunkCount": 19, - "latencyP95Ms": 20.14 + "latencyP95Ms": 14.32 }, { "strategy": "hybrid", @@ -302,10 +319,11 @@ "noResultRate": 0, "meanContextChars": 3282.1052631578946, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 19, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 19, + "unanswerableContextPassages": 5, "chunkCount": 19, - "latencyP95Ms": 14.23 + "latencyP95Ms": 14.82 }, { "strategy": "hybrid", @@ -319,10 +337,11 @@ "noResultRate": 0, "meanContextChars": 5357.315789473684, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 19, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 19, + "unanswerableContextPassages": 8, "chunkCount": 19, - "latencyP95Ms": 14.06 + "latencyP95Ms": 13.74 }, { "strategy": "hybrid", @@ -336,10 +355,11 @@ "noResultRate": 0, "meanContextChars": 1958.7631578947369, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 19, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 19, + "unanswerableContextPassages": 3, "chunkCount": 19, - "latencyP95Ms": 15.39 + "latencyP95Ms": 18.46 }, { "strategy": "hybrid", @@ -353,10 +373,11 @@ "noResultRate": 0, "meanContextChars": 3282.1052631578946, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 19, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 19, + "unanswerableContextPassages": 5, "chunkCount": 19, - "latencyP95Ms": 18.4 + "latencyP95Ms": 16.37 }, { "strategy": "hybrid", @@ -370,10 +391,11 @@ "noResultRate": 0, "meanContextChars": 5357.315789473684, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 19, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 19, + "unanswerableContextPassages": 8, "chunkCount": 19, - "latencyP95Ms": 18.41 + "latencyP95Ms": 20.86 } ] } diff --git a/docs/eval/sweep-v1.6.md b/docs/eval/sweep-v1.6.md index b7e76cc..a3c6dd7 100644 --- a/docs/eval/sweep-v1.6.md +++ b/docs/eval/sweep-v1.6.md @@ -10,30 +10,30 @@ Each row differs from its neighbour in one parameter. 2 further cell(s) were **skipped** because `contextK > candidateK`; see below. -| Strategy | candidateK | contextK | Recall@5 | nDCG@10 | MAP@10 | Context P | Context R | No-result | Context chars | Unans. no-result | Unans. retrieved | Index | p95 | -| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | -| dense | 5 | 3 | 0.9211 | 0.8212 | 0.7851 | 0.3246 | 0.9211 | 0.0000 | 1976 | 0.0000 | 5.0 | 19 | 13.57 ms | -| dense | 5 | 5 | 0.9211 | 0.8212 | 0.7851 | 0.1947 | 0.9211 | 0.0000 | 3188 | 0.0000 | 5.0 | 19 | 14.14 ms | -| dense | 10 | 3 | 0.9211 | 0.8476 | 0.7965 | 0.3246 | 0.9211 | 0.0000 | 1976 | 0.0000 | 10.0 | 19 | 13.30 ms | -| dense | 10 | 5 | 0.9211 | 0.8476 | 0.7965 | 0.1947 | 0.9211 | 0.0000 | 3188 | 0.0000 | 10.0 | 19 | 12.61 ms | -| dense | 10 | 8 | 0.9211 | 0.8476 | 0.7965 | 0.1316 | 1.0000 | 0.0000 | 5151 | 0.0000 | 10.0 | 19 | 13.99 ms | -| dense | 20 | 3 | 0.9211 | 0.8476 | 0.7965 | 0.3246 | 0.9211 | 0.0000 | 1976 | 0.0000 | 19.0 | 19 | 13.61 ms | -| dense | 20 | 5 | 0.9211 | 0.8476 | 0.7965 | 0.1947 | 0.9211 | 0.0000 | 3188 | 0.0000 | 19.0 | 19 | 14.62 ms | -| dense | 20 | 8 | 0.9211 | 0.8476 | 0.7965 | 0.1316 | 1.0000 | 0.0000 | 5151 | 0.0000 | 19.0 | 19 | 15.20 ms | -| dense | 40 | 3 | 0.9211 | 0.8476 | 0.7965 | 0.3246 | 0.9211 | 0.0000 | 1976 | 0.0000 | 19.0 | 19 | 13.66 ms | -| dense | 40 | 5 | 0.9211 | 0.8476 | 0.7965 | 0.1947 | 0.9211 | 0.0000 | 3188 | 0.0000 | 19.0 | 19 | 14.40 ms | -| dense | 40 | 8 | 0.9211 | 0.8476 | 0.7965 | 0.1316 | 1.0000 | 0.0000 | 5151 | 0.0000 | 19.0 | 19 | 14.39 ms | -| hybrid | 5 | 3 | 0.9211 | 0.8309 | 0.7982 | 0.3246 | 0.9211 | 0.0000 | 1949 | 0.0000 | 5.0 | 19 | 16.98 ms | -| hybrid | 5 | 5 | 0.9211 | 0.8309 | 0.7982 | 0.1947 | 0.9211 | 0.0000 | 3216 | 0.0000 | 5.0 | 19 | 16.11 ms | -| hybrid | 10 | 3 | 0.9211 | 0.8574 | 0.8097 | 0.3246 | 0.9211 | 0.0000 | 1963 | 0.0000 | 10.0 | 19 | 16.46 ms | -| hybrid | 10 | 5 | 0.9211 | 0.8574 | 0.8097 | 0.1947 | 0.9211 | 0.0000 | 3333 | 0.0000 | 10.0 | 19 | 14.83 ms | -| hybrid | 10 | 8 | 0.9211 | 0.8574 | 0.8097 | 0.1316 | 1.0000 | 0.0000 | 5320 | 0.0000 | 10.0 | 19 | 17.38 ms | -| hybrid | 20 | 3 | 0.9211 | 0.8574 | 0.8097 | 0.3246 | 0.9211 | 0.0000 | 1959 | 0.0000 | 19.0 | 19 | 20.14 ms | -| hybrid | 20 | 5 | 0.9211 | 0.8574 | 0.8097 | 0.1947 | 0.9211 | 0.0000 | 3282 | 0.0000 | 19.0 | 19 | 14.23 ms | -| hybrid | 20 | 8 | 0.9211 | 0.8574 | 0.8097 | 0.1316 | 1.0000 | 0.0000 | 5357 | 0.0000 | 19.0 | 19 | 14.06 ms | -| hybrid | 40 | 3 | 0.9211 | 0.8574 | 0.8097 | 0.3246 | 0.9211 | 0.0000 | 1959 | 0.0000 | 19.0 | 19 | 15.39 ms | -| hybrid | 40 | 5 | 0.9211 | 0.8574 | 0.8097 | 0.1947 | 0.9211 | 0.0000 | 3282 | 0.0000 | 19.0 | 19 | 18.40 ms | -| hybrid | 40 | 8 | 0.9211 | 0.8574 | 0.8097 | 0.1316 | 1.0000 | 0.0000 | 5357 | 0.0000 | 19.0 | 19 | 18.41 ms | +| Strategy | candidateK | contextK | Recall@5 | nDCG@10 | MAP@10 | Context P | Context R | No-result | Context chars | Unans. abstained | Unans. cands | Unans. ctx | Index | p95 | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | +| dense | 5 | 3 | 0.9211 | 0.8212 | 0.7851 | 0.3246 | 0.9211 | 0.0000 | 1976 | 0.0000 | 5.0 | 3.0 | 19 | 14.04 ms | +| dense | 5 | 5 | 0.9211 | 0.8212 | 0.7851 | 0.1947 | 0.9211 | 0.0000 | 3188 | 0.0000 | 5.0 | 5.0 | 19 | 13.63 ms | +| dense | 10 | 3 | 0.9211 | 0.8476 | 0.7965 | 0.3246 | 0.9211 | 0.0000 | 1976 | 0.0000 | 10.0 | 3.0 | 19 | 13.84 ms | +| dense | 10 | 5 | 0.9211 | 0.8476 | 0.7965 | 0.1947 | 0.9211 | 0.0000 | 3188 | 0.0000 | 10.0 | 5.0 | 19 | 15.21 ms | +| dense | 10 | 8 | 0.9211 | 0.8476 | 0.7965 | 0.1316 | 1.0000 | 0.0000 | 5151 | 0.0000 | 10.0 | 8.0 | 19 | 16.35 ms | +| dense | 20 | 3 | 0.9211 | 0.8476 | 0.7965 | 0.3246 | 0.9211 | 0.0000 | 1976 | 0.0000 | 19.0 | 3.0 | 19 | 16.45 ms | +| dense | 20 | 5 | 0.9211 | 0.8476 | 0.7965 | 0.1947 | 0.9211 | 0.0000 | 3188 | 0.0000 | 19.0 | 5.0 | 19 | 17.91 ms | +| dense | 20 | 8 | 0.9211 | 0.8476 | 0.7965 | 0.1316 | 1.0000 | 0.0000 | 5151 | 0.0000 | 19.0 | 8.0 | 19 | 14.13 ms | +| dense | 40 | 3 | 0.9211 | 0.8476 | 0.7965 | 0.3246 | 0.9211 | 0.0000 | 1976 | 0.0000 | 19.0 | 3.0 | 19 | 13.35 ms | +| dense | 40 | 5 | 0.9211 | 0.8476 | 0.7965 | 0.1947 | 0.9211 | 0.0000 | 3188 | 0.0000 | 19.0 | 5.0 | 19 | 12.73 ms | +| dense | 40 | 8 | 0.9211 | 0.8476 | 0.7965 | 0.1316 | 1.0000 | 0.0000 | 5151 | 0.0000 | 19.0 | 8.0 | 19 | 15.12 ms | +| hybrid | 5 | 3 | 0.9211 | 0.8309 | 0.7982 | 0.3246 | 0.9211 | 0.0000 | 1949 | 0.0000 | 5.0 | 3.0 | 19 | 14.01 ms | +| hybrid | 5 | 5 | 0.9211 | 0.8309 | 0.7982 | 0.1947 | 0.9211 | 0.0000 | 3216 | 0.0000 | 5.0 | 5.0 | 19 | 15.81 ms | +| hybrid | 10 | 3 | 0.9211 | 0.8574 | 0.8097 | 0.3246 | 0.9211 | 0.0000 | 1963 | 0.0000 | 10.0 | 3.0 | 19 | 15.74 ms | +| hybrid | 10 | 5 | 0.9211 | 0.8574 | 0.8097 | 0.1947 | 0.9211 | 0.0000 | 3345 | 0.0000 | 10.0 | 5.0 | 19 | 16.30 ms | +| hybrid | 10 | 8 | 0.9211 | 0.8574 | 0.8097 | 0.1316 | 1.0000 | 0.0000 | 5320 | 0.0000 | 10.0 | 8.0 | 19 | 16.95 ms | +| hybrid | 20 | 3 | 0.9211 | 0.8574 | 0.8097 | 0.3246 | 0.9211 | 0.0000 | 1959 | 0.0000 | 19.0 | 3.0 | 19 | 14.32 ms | +| hybrid | 20 | 5 | 0.9211 | 0.8574 | 0.8097 | 0.1947 | 0.9211 | 0.0000 | 3282 | 0.0000 | 19.0 | 5.0 | 19 | 14.82 ms | +| hybrid | 20 | 8 | 0.9211 | 0.8574 | 0.8097 | 0.1316 | 1.0000 | 0.0000 | 5357 | 0.0000 | 19.0 | 8.0 | 19 | 13.74 ms | +| hybrid | 40 | 3 | 0.9211 | 0.8574 | 0.8097 | 0.3246 | 0.9211 | 0.0000 | 1959 | 0.0000 | 19.0 | 3.0 | 19 | 18.46 ms | +| hybrid | 40 | 5 | 0.9211 | 0.8574 | 0.8097 | 0.1947 | 0.9211 | 0.0000 | 3282 | 0.0000 | 19.0 | 5.0 | 19 | 16.37 ms | +| hybrid | 40 | 8 | 0.9211 | 0.8574 | 0.8097 | 0.1316 | 1.0000 | 0.0000 | 5357 | 0.0000 | 19.0 | 8.0 | 19 | 20.86 ms | ## Skipped cells @@ -60,10 +60,14 @@ The harness refuses the same combination at the flag level, so a typo fails loud embedding model, not any generation model's tokenizer. - **No-result** is the share of *answerable* questions whose retrieval returned nothing — a miss, and the lower the better. -- **Unans. no-result / retrieved** are the same idea for the *unanswerable* questions, - where the direction flips: there is no ground truth, so returning nothing is correct and - `retrieved` is how much irrelevant context was pulled in anyway. These two are the - columns a threshold decision should move, and they are kept out of every other column. +- **Unans. abstained / cands / ctx** describe the *unanswerable* questions, where the + direction flips: there is no ground truth, so abstaining is correct. `abstained` is the + share where nothing passed the threshold; `cands` is how many candidates did (up to + `candidateK`, since the harness fetches that many for `Recall@10`); `ctx` is how many + actually reach the context window, i.e. `min(candidates, contextK)`. A high `cands` with + the usual `ctx` means the threshold is filtering nothing and the window is all noise. + These are the columns a threshold decision should move, and they stay out of every other + column. Best nDCG@10 in this grid: `hybrid` candidateK=10, contextK=3 (0.8574). diff --git a/docs/eval/threshold-v1.6.json b/docs/eval/threshold-v1.6.json index 468dc8d..cb995ef 100644 --- a/docs/eval/threshold-v1.6.json +++ b/docs/eval/threshold-v1.6.json @@ -14,9 +14,10 @@ "meanRetrieved": 19, "unanswerable": { "questions": 3, - "noResultCount": 0, - "noResultRate": 0, - "meanRetrieved": 19 + "abstentionCount": 0, + "retrievalAbstentionRate": 0, + "meanCandidatesRetrieved": 19, + "meanContextPassages": 3 }, "metrics": { "recallAt1": 0.576923, @@ -38,9 +39,10 @@ "meanRetrieved": 19, "unanswerable": { "questions": 3, - "noResultCount": 0, - "noResultRate": 0, - "meanRetrieved": 19 + "abstentionCount": 0, + "retrievalAbstentionRate": 0, + "meanCandidatesRetrieved": 19, + "meanContextPassages": 3 }, "metrics": { "recallAt1": 0.7, @@ -65,9 +67,10 @@ "meanRetrieved": 19, "unanswerable": { "questions": 3, - "noResultCount": 0, - "noResultRate": 0, - "meanRetrieved": 19 + "abstentionCount": 0, + "retrievalAbstentionRate": 0, + "meanCandidatesRetrieved": 19, + "meanContextPassages": 3 }, "metrics": { "recallAt1": 0.576923, @@ -89,9 +92,10 @@ "meanRetrieved": 19, "unanswerable": { "questions": 3, - "noResultCount": 0, - "noResultRate": 0, - "meanRetrieved": 19 + "abstentionCount": 0, + "retrievalAbstentionRate": 0, + "meanCandidatesRetrieved": 19, + "meanContextPassages": 3 }, "metrics": { "recallAt1": 0.7, @@ -116,9 +120,10 @@ "meanRetrieved": 19, "unanswerable": { "questions": 3, - "noResultCount": 0, - "noResultRate": 0, - "meanRetrieved": 19 + "abstentionCount": 0, + "retrievalAbstentionRate": 0, + "meanCandidatesRetrieved": 19, + "meanContextPassages": 3 }, "metrics": { "recallAt1": 0.576923, @@ -140,9 +145,10 @@ "meanRetrieved": 19, "unanswerable": { "questions": 3, - "noResultCount": 0, - "noResultRate": 0, - "meanRetrieved": 19 + "abstentionCount": 0, + "retrievalAbstentionRate": 0, + "meanCandidatesRetrieved": 19, + "meanContextPassages": 3 }, "metrics": { "recallAt1": 0.7, @@ -167,9 +173,10 @@ "meanRetrieved": 19, "unanswerable": { "questions": 3, - "noResultCount": 0, - "noResultRate": 0, - "meanRetrieved": 19 + "abstentionCount": 0, + "retrievalAbstentionRate": 0, + "meanCandidatesRetrieved": 19, + "meanContextPassages": 3 }, "metrics": { "recallAt1": 0.576923, @@ -191,9 +198,10 @@ "meanRetrieved": 19, "unanswerable": { "questions": 3, - "noResultCount": 0, - "noResultRate": 0, - "meanRetrieved": 19 + "abstentionCount": 0, + "retrievalAbstentionRate": 0, + "meanCandidatesRetrieved": 19, + "meanContextPassages": 3 }, "metrics": { "recallAt1": 0.7, @@ -218,9 +226,10 @@ "meanRetrieved": 19, "unanswerable": { "questions": 3, - "noResultCount": 0, - "noResultRate": 0, - "meanRetrieved": 19 + "abstentionCount": 0, + "retrievalAbstentionRate": 0, + "meanCandidatesRetrieved": 19, + "meanContextPassages": 3 }, "metrics": { "recallAt1": 0.576923, @@ -242,9 +251,10 @@ "meanRetrieved": 19, "unanswerable": { "questions": 3, - "noResultCount": 0, - "noResultRate": 0, - "meanRetrieved": 19 + "abstentionCount": 0, + "retrievalAbstentionRate": 0, + "meanCandidatesRetrieved": 19, + "meanContextPassages": 3 }, "metrics": { "recallAt1": 0.7, diff --git a/docs/eval/threshold-v1.6.md b/docs/eval/threshold-v1.6.md index 45e1f10..0b08be6 100644 --- a/docs/eval/threshold-v1.6.md +++ b/docs/eval/threshold-v1.6.md @@ -11,30 +11,32 @@ split selects; the **test** split reports. The split is the committed manifest Quality columns cover the answerable questions only; **Unans.** columns cover the unanswerable ones, where returning nothing is the desired outcome and so a *higher* -no-result rate is better. +abstention rate is better. Two sizes are kept apart: **cands** is how many candidates +passed the threshold (up to `candidateK`), **ctx** is how many reach the context window. -| Threshold | n (val) | Recall@5 (val) | nDCG@10 (val) | No-result (val) | Unans. no-result (val) | Unans. retrieved (val) | nDCG@10 (test) | Unans. no-result (test) | -| --- | --- | --- | --- | --- | --- | --- | --- | --- | -| 0 | 13 | 0.9231 | 0.8276 | 0.0000 | 0.0000 | 19.0 | 0.8581 | 0.0000 | -| 0.3 | 13 | 0.9231 | 0.8276 | 0.0000 | 0.0000 | 19.0 | 0.8581 | 0.0000 | -| 0.4 | 13 | 0.9231 | 0.8276 | 0.0000 | 0.0000 | 19.0 | 0.8581 | 0.0000 | -| 0.5 | 13 | 0.9231 | 0.8276 | 0.0000 | 0.0000 | 19.0 | 0.8581 | 0.0000 | -| 0.6 | 13 | 0.9231 | 0.8276 | 0.0000 | 0.0000 | 19.0 | 0.8581 | 0.0000 | +| Threshold | n (val) | Recall@5 (val) | nDCG@10 (val) | No-result (val) | Unans. abstained (val) | Unans. cands (val) | Unans. ctx (val) | nDCG@10 (test) | Unans. abstained (test) | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | +| 0 | 13 | 0.9231 | 0.8276 | 0.0000 | 0.0000 | 19.0 | 3.0 | 0.8581 | 0.0000 | +| 0.3 | 13 | 0.9231 | 0.8276 | 0.0000 | 0.0000 | 19.0 | 3.0 | 0.8581 | 0.0000 | +| 0.4 | 13 | 0.9231 | 0.8276 | 0.0000 | 0.0000 | 19.0 | 3.0 | 0.8581 | 0.0000 | +| 0.5 | 13 | 0.9231 | 0.8276 | 0.0000 | 0.0000 | 19.0 | 3.0 | 0.8581 | 0.0000 | +| 0.6 | 13 | 0.9231 | 0.8276 | 0.0000 | 0.0000 | 19.0 | 3.0 | 0.8581 | 0.0000 | ## Selection rule Hold the answerable quality line — validation nDCG@10 and context recall must not -regress versus `threshold = 0` — then take the threshold that refuses the most +regress versus `threshold = 0` — then take the threshold that abstains on the most unanswerable questions. Tie-break on the lowest threshold. -Raising a threshold is only worth anything if it refuses what the sources do not answer; -the quality gate is there so a refusal gain can never be bought with a retrieval loss. +Raising a threshold is only worth anything if it stops unsupported context before the +prompt; the quality gate is there so an abstention gain can never be bought with a +retrieval loss. ## Outcome -The sweep is **flat**: every threshold from 0 to 0.6 produces the same validation nDCG@10 (0.8276), the same Recall@5 (0.9231) and the same unanswerable refusal rate (0/3). No passage is ever filtered out, so the threshold is **non-binding** on this corpus — E5 does not score these query/chunk pairs below the top of the swept range. +The sweep is **flat**: every threshold from 0 to 0.6 produces the same validation nDCG@10 (0.8276), the same Recall@5 (0.9231) and the same retrieval abstention rate on unanswerable questions (0/3). No candidate is ever filtered out, so the threshold is **non-binding** on this corpus. -**No evidence to change `threshold = 0.5`.** All thresholds hold the line equally; picking one would be arbitrary. The current value can be neither validated nor falsified here, which is a property of the corpus rather than of the threshold. +**No evidence to change `threshold = 0.5`.** All thresholds hold the line equally; picking one would be arbitrary. The current value can be neither validated nor falsified here, which is a property of the corpus rather than of the threshold. Note also what this does *not* establish: abstention is a retrieval-layer statement — whether the model then declines to answer needs a generator eval. ## Caveat on this corpus diff --git a/eval/README.md b/eval/README.md index f01989c..24deb75 100644 --- a/eval/README.md +++ b/eval/README.md @@ -39,11 +39,17 @@ A parameter picked on the same questions it is scored on is a fitted number, not result. `eval/splits.json` is the committed assignment; `--eval-split=validation` selects from it and `test` is the rest. -It is an explicit manifest rather than a hash of the question id. A hash is reproducible -but not *stable*: adding a question moves others between the sides, and a rare query type -can end up entirely on one side without anyone choosing that. With a manifest, a question -with no entry is **refused** rather than defaulted, so a new question is assigned -deliberately instead of leaking into `test`. +It is an explicit manifest rather than a hash of the question id. A hash +(`hash(id) % 3`) is actually *stable* — it is computed per id, so adding a question does +not move the existing ones. What it cannot do is express the experimental design: + +- it does not stratify a small corpus, so a rare type (`multi-hop`, `cross-lingual`) can + end up entirely on one side without anyone choosing that — which is what happened; and +- a newly added question is assigned silently instead of deliberately, and `test` is the + side a choice must not be fitted to. + +With a manifest, a question with no entry is **refused** rather than defaulted, so every +new question is assigned on purpose. `npm run eval:sweep` runs a bounded grid (`strategy × candidateK × contextK`) and writes one dashboard with quality, context precision/recall, prompt size, index size diff --git a/scripts/eval-retrieval.mjs b/scripts/eval-retrieval.mjs index 0ed9a6c..d06d5ad 100644 --- a/scripts/eval-retrieval.mjs +++ b/scripts/eval-retrieval.mjs @@ -1,15 +1,17 @@ #!/usr/bin/env node /** - * Retrieval experiments for #77. + * Retrieval experiments for #77, with held-out strategy adoption (#192). * - * Runs the real RAG eval harness once per retrieval strategy against the frozen - * chunk baseline (`baseline-v1.6.json`, 1000/100), holding chunking fixed, and - * writes the comparison the issue asks for as a delta against dense. + * Runs the real RAG eval harness once per retrieval strategy on the **validation** split, + * decides there with the amended adoption rule (#192 child 10), and then re-runs the + * shipped strategy and the selected one on the **test** split. Selection never sees + * `test`; `test` only reports. * - * The harness does the measuring; this script only orchestrates and tabulates. + * That split is the whole point. The previous version decided on `split = all`, which + * meant the strategy was chosen and scored on the same questions — the same mistake the + * threshold experiment had already been fixed for. * - * The adoption rule is the amended one from #192 child 10: the deciding metric is the - * first metric with headroom, not a metric that the corpus has already maxed out. + * The harness does the measuring; this script only orchestrates and tabulates. * * Usage: * node scripts/eval-retrieval.mjs @@ -37,6 +39,9 @@ const STRATEGIES = [ { id: 'hybrid', label: 'hybrid (RRF of dense + BM25)' } ] +/** The shipped strategy. Everything is reported as a delta against it. */ +const BASELINE_ID = 'dense' + function readArg(prefix, fallback) { const arg = process.argv.find((value) => value.startsWith(prefix)) return arg ? arg.slice(prefix.length) : fallback @@ -52,13 +57,18 @@ if (!existsSync(executable)) { process.exit(1) } -function runStrategy(strategy, outDir) { +// The rule lives in `src/main/eval/adoption.ts` so it can be unit tested; it decides +// whether a shipped default moves. +const { ADOPTION_METRICS, decideAdoption } = await import('../src/main/eval/adoption.ts') + +function runStrategy(strategy, split, outDir) { return new Promise((resolvePromise, reject) => { const args = [ '.', '--eval-harness', '--eval-baseline=v1.6', `--eval-out=${outDir}`, + `--eval-split=${split}`, `--eval-retrieval=${strategy.id}` ] @@ -70,108 +80,128 @@ function runStrategy(strategy, outDir) { env: { ...process.env, ELECTRON_DISABLE_SECURITY_WARNINGS: '1' } }) - let stdout = '' - child.stdout.on('data', (data) => { - stdout += data.toString() - }) - child.stderr.on('data', () => {}) - child.on('error', reject) child.on('exit', (code) => { if (code !== 0) { - reject(new Error(`strategy ${strategy.id} exited with code ${code}`)) + reject(new Error(`${strategy.id} (${split}) exited with code ${code}`)) return } - const metricsLine = /\[eval\] metrics (\{.*\})/.exec(stdout) - if (!metricsLine) { - reject(new Error(`strategy ${strategy.id} printed no metrics line`)) + const reportPath = join(outDir, 'baseline-v1.6.json') + if (!existsSync(reportPath)) { + reject(new Error(`${strategy.id} (${split}) wrote no report`)) return } - try { - resolvePromise(JSON.parse(metricsLine[1])) - } catch (error) { - reject( - new Error(`strategy ${strategy.id} printed an unreadable metrics line: ${error.message}`) - ) - } + const report = JSON.parse(readFileSync(reportPath, 'utf8')) + resolvePromise({ + id: strategy.id, + label: strategy.label, + split, + questions: report.config.questions, + answerableCount: report.byType.reduce((total, entry) => total + entry.questions, 0), + chunking: `${report.config.chunking.chunkSize}/${report.config.chunking.chunkOverlap}`, + chunkCount: report.config.chunkCount, + ...report.metrics, + ...readTiming(join(outDir, 'baseline-v1.6.md')) + }) }) }) } -/** Throughput and p95 are informational and excluded from the deterministic JSON. */ +/** p95 is informational and excluded from the deterministic JSON. */ function readTiming(mdPath) { - if (!existsSync(mdPath)) return { indexingMs: null, latencyP95Ms: null } + if (!existsSync(mdPath)) return { latencyP95Ms: null } const text = readFileSync(mdPath, 'utf8') - const indexing = /indexing (\d+) ms/.exec(text) const p95 = /p95 ([\d.]+) ms/.exec(text) - return { - indexingMs: indexing ? Number(indexing[1]) : null, - latencyP95Ms: p95 ? Number(p95[1]) : null - } + return { latencyP95Ms: p95 ? Number(p95[1]) : null } } +function runInto(workDir, strategy, split) { + const outDir = join(workDir, `${split}-${strategy.id}`) + mkdirSync(outDir, { recursive: true }) + return runStrategy(strategy, split, outDir) +} + +const format4 = (value) => value.toFixed(4) + const workDir = mkdtempSync(join(tmpdir(), 'knownote-retrieval-')) -const results = [] +let validation = [] +let test = [] +let decision = { primary: null, saturated: [], winner: null } try { + // ── Validation: this is the only phase that may choose ──────────────────────── for (const strategy of STRATEGIES) { - const outDir = join(workDir, strategy.id) - mkdirSync(outDir, { recursive: true }) - console.log(`[retrieval] running ${strategy.label}`) - const metrics = await runStrategy(strategy, outDir) - - const report = JSON.parse(readFileSync(join(outDir, 'baseline-v1.6.json'), 'utf8')) - results.push({ - id: strategy.id, - label: strategy.label, - chunking: `${report.config.chunking.chunkSize}/${report.config.chunking.chunkOverlap}`, - chunkCount: report.config.chunkCount, - split: report.config.split, - questions: report.config.questions, - answerableCount: report.byType.reduce((total, entry) => total + entry.questions, 0), - ...metrics, - ...readTiming(join(outDir, 'baseline-v1.6.md')) - }) + console.log(`[retrieval] validation: ${strategy.label}`) + validation.push(await runInto(workDir, strategy, 'validation')) } -} finally { - rmSync(workDir, { recursive: true, force: true }) -} -const baseline = results.find((result) => result.id === 'dense') -if (!baseline) throw new Error('the dense strategy did not run') + const baselineRow = validation.find((row) => row.id === BASELINE_ID) + if (!baselineRow) throw new Error('the dense strategy did not run on validation') -const format4 = (value) => value.toFixed(4) + decision = decideAdoption(baselineRow, validation) -const rows = results.map( - (result) => - `| ${result.label} | ${format4(result.recallAt1)} | ${format4(result.recallAt5)} | ${format4(result.mrr)} | ${format4(result.ndcgAt10)} | ${format4(result.mapAt10)} | ${format4(result.contextPrecision)} | ${result.latencyP95Ms?.toFixed(2)} ms |` -) + // ── Test: reports only. Never lets the choice see these questions. ──────────── + const testIds = [BASELINE_ID] + if (decision.winner && decision.winner.id !== BASELINE_ID) testIds.push(decision.winner.id) -/** - * The rule lives in `src/main/eval/adoption.ts` so it can be unit tested: it decides - * whether a shipped default moves, and testing it by reading the sentence this script - * prints would be a test of the sentence. - * - * Latency stays in the table so a win that costs 5x latency is stated as a trade-off, - * not hidden. - */ -const { ADOPTION_METRICS, decideAdoption } = await import('../src/main/eval/adoption.ts') -const { primary, saturated, winner } = decideAdoption(baseline, results) + for (const id of testIds) { + const strategy = STRATEGIES.find((entry) => entry.id === id) + console.log(`[retrieval] test: ${strategy.label}`) + test.push(await runInto(workDir, strategy, 'test')) + } +} finally { + rmSync(workDir, { recursive: true, force: true }) +} -const saturationNote = saturated.length - ? `Saturated (no headroom, so they cannot decide anything): ${saturated +const winner = decision.winner +const saturationNote = decision.saturated.length + ? `Saturated on the validation split (no headroom, so they cannot decide): ${decision.saturated .map((key) => `\`${key}\``) .join(', ')}.` - : 'No metric in the rule is saturated on this corpus.' - -let outcome -if (winner) { - outcome = `\`${winner.label}\` **clears the rule**: it improves the deciding metric \`${primary}\` (${format4(winner[primary])} vs dense ${format4(baseline[primary])}) and regresses none of ${ADOPTION_METRICS.map((key) => `\`${key}\``).join(', ')}. ${saturationNote}\n\nChanging the shipped default is a separate decision, and this script does not make it — it reports the measurement.` -} else if (primary === null) { - outcome = `Every metric in the rule is already at its maximum on this corpus, so no strategy can clear any of them. **Dense stays the default**; the comparison is inconclusive by construction, not negative.` -} else { - outcome = `No strategy cleared the rule. The deciding metric was \`${primary}\` (dense ${format4(baseline[primary])}); the strategies either failed to improve it or regressed another metric. **Dense stays the default.** A negative result is the point of the experiment: it is the measurement that says the extra machinery is not worth its cost on this corpus, not a failure to deliver. ${saturationNote}` -} + : 'No metric in the rule is saturated on the validation split.' + +const metricColumns = ['recallAt1', 'recallAt5', 'mrr', 'ndcgAt10', 'mapAt10', 'contextPrecision'] +const tableRow = (row) => + `| ${row.label} | ${metricColumns.map((key) => format4(row[key])).join(' | ')} | ${row.questions} | ${ + row.latencyP95Ms?.toFixed(2) ?? '—' + } ms |` + +const validationTable = validation.map(tableRow).join('\n') +const testTable = test.map(tableRow).join('\n') + +/** Per-metric delta of the selected strategy against dense, on the test split. */ +const testDense = test.find((row) => row.id === BASELINE_ID) +const testWinner = winner ? test.find((row) => row.id === winner.id) : null + +const deltaRows = + testWinner && testDense + ? ADOPTION_METRICS.map((key) => { + const delta = testWinner[key] - testDense[key] + const sign = delta > 0 ? '+' : '' + return `| ${key} | ${format4(testDense[key])} | ${format4(testWinner[key])} | ${sign}${format4(delta)} |` + }).join('\n') + : '' + +const outcome = winner + ? `**\`${winner.label}\` clears the rule on validation.** The deciding metric was \`${decision.primary}\` (${format4( + winner[decision.primary] + )} vs dense ${format4( + validation.find((row) => row.id === BASELINE_ID)[decision.primary] + )}), and it regressed none of ${ADOPTION_METRICS.map((key) => `\`${key}\``).join(', ')}. ${saturationNote} + +That decision was made on questions in \`test\` **not** seeing. What follows is the held-out +result, and it is the only number that should inform shipping it: + +| Metric | dense (test) | selected (test) | delta | +| --- | --- | --- | --- | +${deltaRows} + +Shipping a new default is a product decision this script does not make. It measures.` + : `**No strategy cleared the rule on validation**, so there is no adoption candidate and +\`test\` reports the shipped strategy only. ${saturationNote} + +A negative result is the point of the experiment: it is the measurement that says the extra +machinery is not worth its cost on this corpus, not a failure to deliver.` const markdown = `# Retrieval experiments — v1.6 (#77, #192) @@ -179,13 +209,25 @@ Generated by \`node scripts/eval-retrieval.mjs\`. Numbers are harness output; do ## What was measured -Every strategy runs the real RAG eval harness against the same corpus and the same questions -as \`baseline-v1.6.json\` (split \`${baseline.split}\`, ${baseline.questions} questions of which -${baseline.answerableCount} are answerable), with chunking held fixed at ${baseline.chunking}. Only the retrieval strategy changes. +Each strategy runs the real harness on the **validation** split of +\`eval/splits.json\`, with chunking held fixed at ${validation[0]?.chunking ?? '1000/100'}. + +The strategy is then chosen **there**, and only the chosen one (plus the shipped default) +is re-run on **test**. The choice never sees \`test\`; \`test\` only reports. The previous +version decided on \`split = all\`, which scored the choice on the questions it was fitted +to. + +## Validation — this is where the choice happens + +| Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Context P | n | p95 | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | +${validationTable} + +## Test — reported, not selected -| Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Context P | Query p95 | -| --- | --- | --- | --- | --- | --- | --- | --- | -${rows.join('\n')} +| Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Context P | n | p95 | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | +${testTable} ## Not evaluated @@ -194,13 +236,6 @@ model is not available offline and inventing its numbers would defeat the point the harness. It stays open until a model can be pinned the way the embedding model is. -## Saturation - -${saturationNote} The deciding metric on this corpus is ${ - primary === null ? 'none — every metric in the rule is already maxed out' : `\`${primary}\`` -}. A saturated metric is still reported, because "this corpus cannot move it" is itself -information; it is just not allowed to decide the comparison. - ## Adoption rule > Adopt a strategy when it improves the **first metric with headroom** — in the order @@ -208,8 +243,8 @@ information; it is just not allowed to decide the comparison. > already at its maximum has no headroom and cannot decide anything; a rule that depends > on one is unsatisfiable, not strict (#192 child 10). > -> A change that trades a large latency increase for a marginal quality gain is a product -> decision, not an automatic win. +> The rule is applied on \`validation\`. A change that trades a large latency increase for a +> marginal quality gain is a product decision, not an automatic win. ## Outcome @@ -226,11 +261,21 @@ npm run eval:retrieval # offline; runs every strategy and rewrites this file mkdirSync(resolve(OUT_MD, '..'), { recursive: true }) writeFileSync( OUT_JSON, - `${JSON.stringify({ baseline: baseline.id, chunking: baseline.chunking, strategies: results }, null, 2)}\n` + `${JSON.stringify( + { + baseline: 'v1.6', + split: { selects: 'validation', reports: 'test', manifest: 'eval/splits.json' }, + decision: { primary: decision.primary, saturated: decision.saturated, winner: winner?.id ?? null }, + validation, + test + }, + null, + 2 + )}\n` ) writeFileSync(OUT_MD, markdown) -console.log(`[retrieval] wrote ${OUT_JSON} and ${OUT_MD}`) console.log( - `[retrieval] ${winner ? `best clearing strategy: ${winner.label}` : 'no strategy cleared the rule; keep dense'}` + `[retrieval] ${winner ? `validation selected: ${winner.label}` : 'validation selected nothing; test reports dense'}` ) +console.log(`[retrieval] wrote ${OUT_JSON} and ${OUT_MD}`) diff --git a/scripts/eval-sweep.mjs b/scripts/eval-sweep.mjs index 45a3eb0..207f9d9 100644 --- a/scripts/eval-sweep.mjs +++ b/scripts/eval-sweep.mjs @@ -159,8 +159,9 @@ try { noResultRate: answerable.length === 0 ? 0 : noResult / answerable.length, meanContextChars: mean(answerable.map((q) => q.contextChars)), unanswerableQuestions: report.unanswerable.questions, - unanswerableNoResultRate: report.unanswerable.noResultRate, - unanswerableMeanRetrieved: report.unanswerable.meanRetrieved, + unanswerableAbstentionRate: report.unanswerable.retrievalAbstentionRate, + unanswerableCandidates: report.unanswerable.meanCandidatesRetrieved, + unanswerableContextPassages: report.unanswerable.meanContextPassages, chunkCount: report.config.chunkCount, ...readTiming(join(outDir, 'baseline-v1.6.md')) }) @@ -177,8 +178,9 @@ const tableRows = rows `| ${row.strategy} | ${row.candidateK} | ${row.contextK} | ${format4(row.recallAt5)} | ` + `${format4(row.ndcgAt10)} | ${format4(row.mapAt10)} | ${format4(row.contextPrecision)} | ` + `${format4(row.contextRecall)} | ${format4(row.noResultRate)} | ` + - `${Math.round(row.meanContextChars)} | ${format4(row.unanswerableNoResultRate)} | ` + - `${row.unanswerableMeanRetrieved.toFixed(1)} | ${row.chunkCount} | ${row.latencyP95Ms?.toFixed(2) ?? '—'} ms |` + `${Math.round(row.meanContextChars)} | ${format4(row.unanswerableAbstentionRate)} | ` + + `${row.unanswerableCandidates.toFixed(1)} | ${row.unanswerableContextPassages.toFixed(1)} | ` + + `${row.chunkCount} | ${row.latencyP95Ms?.toFixed(2) ?? '—'} ms |` ) .join('\n') @@ -211,8 +213,8 @@ The real harness, the same corpus, chunking held fixed, over ${STRATEGIES.join(' / ')} × candidateK {${CANDIDATE_KS.join(', ')}} × contextK {${CONTEXT_KS.join(', ')}} — ${rows.length} runs. Each row differs from its neighbour in one parameter.${skipped.length > 0 ? `\n\n${skipped.length} further cell(s) were **skipped** because \`contextK > candidateK\`; see below.` : ''} -| Strategy | candidateK | contextK | Recall@5 | nDCG@10 | MAP@10 | Context P | Context R | No-result | Context chars | Unans. no-result | Unans. retrieved | Index | p95 | -| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | +| Strategy | candidateK | contextK | Recall@5 | nDCG@10 | MAP@10 | Context P | Context R | No-result | Context chars | Unans. abstained | Unans. cands | Unans. ctx | Index | p95 | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | ${tableRows} ${skippedNote} @@ -228,10 +230,14 @@ ${skippedNote} embedding model, not any generation model's tokenizer. - **No-result** is the share of *answerable* questions whose retrieval returned nothing — a miss, and the lower the better. -- **Unans. no-result / retrieved** are the same idea for the *unanswerable* questions, - where the direction flips: there is no ground truth, so returning nothing is correct and - \`retrieved\` is how much irrelevant context was pulled in anyway. These two are the - columns a threshold decision should move, and they are kept out of every other column. +- **Unans. abstained / cands / ctx** describe the *unanswerable* questions, where the + direction flips: there is no ground truth, so abstaining is correct. \`abstained\` is the + share where nothing passed the threshold; \`cands\` is how many candidates did (up to + \`candidateK\`, since the harness fetches that many for \`Recall@10\`); \`ctx\` is how many + actually reach the context window, i.e. \`min(candidates, contextK)\`. A high \`cands\` with + the usual \`ctx\` means the threshold is filtering nothing and the window is all noise. + These are the columns a threshold decision should move, and they stay out of every other + column. Best nDCG@10 in this grid: \`${bestNdcg.strategy}\` candidateK=${bestNdcg.candidateK}, contextK=${bestNdcg.contextK} (${format4(bestNdcg.ndcgAt10)}). diff --git a/scripts/eval-threshold.mjs b/scripts/eval-threshold.mjs index 8eb9697..1da038b 100644 --- a/scripts/eval-threshold.mjs +++ b/scripts/eval-threshold.mjs @@ -158,7 +158,8 @@ const holdsTheLine = (row) => const eligible = rows.filter(holdsTheLine) const ranked = [...eligible].sort( (a, b) => - b.validation.unanswerable.noResultRate - a.validation.unanswerable.noResultRate || + b.validation.unanswerable.retrievalAbstentionRate - + a.validation.unanswerable.retrievalAbstentionRate || a.threshold - b.threshold ) /** @@ -170,32 +171,35 @@ const ranked = [...eligible].sort( const flat = rows.every( (row) => row.validation.metrics.ndcgAt10 === reference.validation.metrics.ndcgAt10 && - row.validation.unanswerable.noResultRate === reference.validation.unanswerable.noResultRate + row.validation.unanswerable.retrievalAbstentionRate === + reference.validation.unanswerable.retrievalAbstentionRate ) /** - * Refusals are the point of the second number: `noResultCount` of the unanswerable - * questions came back empty, which is the correct outcome. The rest returned passages the - * sources cannot support. + * Abstention is the point of the second number: `abstentionCount` of the unanswerable + * questions returned nothing, which is the correct outcome *at the retrieval layer*. The + * rest returned candidates the sources cannot support. */ -const refusals = (row) => `${row.unanswerable.noResultCount}/${row.unanswerable.questions}` +const abstained = (row) => `${row.unanswerable.abstentionCount}/${row.unanswerable.questions}` const describe = (row) => `nDCG@10 ${format4(row.metrics.ndcgAt10)}, Recall@5 ${format4(row.metrics.recallAt5)}, ` + `answerable no-result ${format4(row.noResultRate)}, ` + - `unanswerable refused ${refusals(row)}` + `retrieval abstained on unanswerable ${abstained(row)}, ` + + `context passages ${row.unanswerable.meanContextPassages.toFixed(1)}` const outcome = flat - ? `The sweep is **flat**: every threshold from ${THRESHOLDS[0]} to ${THRESHOLDS[THRESHOLDS.length - 1]} produces the same validation nDCG@10 (${format4(reference.validation.metrics.ndcgAt10)}), the same Recall@5 (${format4(reference.validation.metrics.recallAt5)}) and the same unanswerable refusal rate (${refusals(reference.validation)}). No passage is ever filtered out, so the threshold is **non-binding** on this corpus — E5 does not score these query/chunk pairs below the top of the swept range.\n\n**No evidence to change \`threshold = ${PRODUCTION_THRESHOLD}\`.** All thresholds hold the line equally; picking one would be arbitrary. The current value can be neither validated nor falsified here, which is a property of the corpus rather than of the threshold.` - : `**Recommended: \`threshold = ${winner.threshold}\`.**\n\n- **Validation**: ${describe(winner.validation)}\n- **Test**: ${describe(winner.test)}\n- Production ships \`${PRODUCTION_THRESHOLD}\`: validation ${describe(production.validation)}.\n\nThe rule held answerable quality at the \`threshold = 0\` level (nDCG@10 and context recall must not regress, on the validation split) and then took the threshold that refuses the most unanswerable questions. So this is a refusal gain, not a quality gain — if answerable quality had fallen, the threshold would have been ineligible regardless of how much it refused.` + ? `The sweep is **flat**: every threshold from ${THRESHOLDS[0]} to ${THRESHOLDS[THRESHOLDS.length - 1]} produces the same validation nDCG@10 (${format4(reference.validation.metrics.ndcgAt10)}), the same Recall@5 (${format4(reference.validation.metrics.recallAt5)}) and the same retrieval abstention rate on unanswerable questions (${abstained(reference.validation)}). No candidate is ever filtered out, so the threshold is **non-binding** on this corpus.\n\n**No evidence to change \`threshold = ${PRODUCTION_THRESHOLD}\`.** All thresholds hold the line equally; picking one would be arbitrary. The current value can be neither validated nor falsified here, which is a property of the corpus rather than of the threshold. Note also what this does *not* establish: abstention is a retrieval-layer statement — whether the model then declines to answer needs a generator eval.` + : `**Recommended: \`threshold = ${winner.threshold}\`.**\n\n- **Validation**: ${describe(winner.validation)}\n- **Test**: ${describe(winner.test)}\n- Production ships \`${PRODUCTION_THRESHOLD}\`: validation ${describe(production.validation)}.\n\nThe rule held answerable quality at the \`threshold = 0\` level (nDCG@10 and context recall must not regress, on the validation split) and then took the threshold that abstains on the most unanswerable questions. So this is an abstention gain, not a quality gain — if answerable quality had fallen, the threshold would have been ineligible regardless of how much it abstained.` const tableRows = rows .map( (row) => `| ${row.threshold} | ${row.validation.answerableQuestions} | ${format4(row.validation.metrics.recallAt5)} | ` + `${format4(row.validation.metrics.ndcgAt10)} | ${format4(row.validation.noResultRate)} | ` + - `${format4(row.validation.unanswerable.noResultRate)} | ` + - `${row.validation.unanswerable.meanRetrieved.toFixed(1)} | ${format4(row.test.metrics.ndcgAt10)} | ` + - `${format4(row.test.unanswerable.noResultRate)} |` + `${format4(row.validation.unanswerable.retrievalAbstentionRate)} | ` + + `${row.validation.unanswerable.meanCandidatesRetrieved.toFixed(1)} | ` + + `${row.validation.unanswerable.meanContextPassages.toFixed(1)} | ${format4(row.test.metrics.ndcgAt10)} | ` + + `${format4(row.test.unanswerable.retrievalAbstentionRate)} |` ) .join('\n') @@ -212,20 +216,22 @@ split selects; the **test** split reports. The split is the committed manifest Quality columns cover the answerable questions only; **Unans.** columns cover the unanswerable ones, where returning nothing is the desired outcome and so a *higher* -no-result rate is better. +abstention rate is better. Two sizes are kept apart: **cands** is how many candidates +passed the threshold (up to \`candidateK\`), **ctx** is how many reach the context window. -| Threshold | n (val) | Recall@5 (val) | nDCG@10 (val) | No-result (val) | Unans. no-result (val) | Unans. retrieved (val) | nDCG@10 (test) | Unans. no-result (test) | -| --- | --- | --- | --- | --- | --- | --- | --- | --- | +| Threshold | n (val) | Recall@5 (val) | nDCG@10 (val) | No-result (val) | Unans. abstained (val) | Unans. cands (val) | Unans. ctx (val) | nDCG@10 (test) | Unans. abstained (test) | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | ${tableRows} ## Selection rule Hold the answerable quality line — validation nDCG@10 and context recall must not -regress versus \`threshold = 0\` — then take the threshold that refuses the most +regress versus \`threshold = 0\` — then take the threshold that abstains on the most unanswerable questions. Tie-break on the lowest threshold. -Raising a threshold is only worth anything if it refuses what the sources do not answer; -the quality gate is there so a refusal gain can never be bought with a retrieval loss. +Raising a threshold is only worth anything if it stops unsupported context before the +prompt; the quality gate is there so an abstention gain can never be bought with a +retrieval loss. ## Outcome diff --git a/src/main/eval/harness.ts b/src/main/eval/harness.ts index 3297cc0..44427de 100644 --- a/src/main/eval/harness.ts +++ b/src/main/eval/harness.ts @@ -328,14 +328,19 @@ export async function runEvalHarness( return { type, questions: group.length, metrics: summarize(group, options.contextK) } }) - const unanswerableNoResults = unanswerableQuestions.filter((q) => q.retrievedCount === 0).length + const abstentionCount = unanswerableQuestions.filter((q) => q.retrievedCount === 0).length const unanswerable = { questions: unanswerableQuestions.length, - noResultCount: unanswerableNoResults, - // 目标方向与其他指标相反:没有相关资料时,返回空才是对的。 - noResultRate: - unanswerableQuestions.length === 0 ? 0 : unanswerableNoResults / unanswerableQuestions.length, - meanRetrieved: mean(unanswerableQuestions.map((q) => q.retrievedCount)) + abstentionCount, + // 方向与其它指标相反:没有相关资料时,检索层返回空才是对的。 + retrievalAbstentionRate: + unanswerableQuestions.length === 0 ? 0 : abstentionCount / unanswerableQuestions.length, + // 通过 threshold 的候选数,不是送进 prompt 的条数:harness 为了算 Recall@10 取满了 + // candidateK,把这个数当成 prompt 宽度会把问题说大。 + meanCandidatesRetrieved: mean(unanswerableQuestions.map((q) => q.retrievedCount)), + meanContextPassages: mean( + unanswerableQuestions.map((q) => Math.min(q.retrievedCount, options.contextK)) + ) } const chunking = { ...DEFAULT_CHUNK_OPTIONS, ...options.chunkOptions } diff --git a/src/main/eval/report.ts b/src/main/eval/report.ts index 9c2b96f..bc3290e 100644 --- a/src/main/eval/report.ts +++ b/src/main/eval/report.ts @@ -29,12 +29,13 @@ export function renderMarkdown(report: EvalReport): string { const answerableCount = report.byType.reduce((total, entry) => total + entry.questions, 0) const unanswerableNote = unanswerable.questions === 0 - ? 'This corpus carries **no** unanswerable question yet, so refusal is not measured.\n' + - 'The threshold cannot be tuned against it either: every question is answerable, so\n' + - 'every threshold returns something.' + ? 'This corpus carries **no** unanswerable question yet, so retrieval abstention is not\n' + + 'measured. The threshold cannot be tuned against it either: every question is answerable,\n' + + 'so every threshold returns something.' : `| Unanswerable questions | ${unanswerable.questions} |\n` + - `| Returned no results | ${format(unanswerable.noResultRate)} (${unanswerable.noResultCount}/${unanswerable.questions}) |\n` + - `| Mean passages retrieved | ${unanswerable.meanRetrieved.toFixed(2)} |` + `| Retrieval abstained | ${format(unanswerable.retrievalAbstentionRate)} (${unanswerable.abstentionCount}/${unanswerable.questions}) |\n` + + `| Mean candidates passing the threshold | ${unanswerable.meanCandidatesRetrieved.toFixed(2)} |\n` + + `| Mean passages in the context window | ${unanswerable.meanContextPassages.toFixed(2)} |` return `# RAG eval baseline — ${report.baseline} @@ -79,11 +80,20 @@ ${typeRows} ### Unanswerable questions -These carry no ground truth, so the correct outcome is that retrieval finds nothing. They +These carry no ground truth, so the correct outcome is that retrieval returns nothing. They are excluded from every metric above — a missing ground truth is not a miss — and reported -here instead. A higher **no-results** rate is better on this row, which is the opposite of -how it reads everywhere else, and \`mean passages retrieved\` is how much irrelevant context -was pulled in anyway. This is the row a threshold decision should move. +here instead. + +**This measures retrieval-level abstention, not the model refusing.** No generator runs in +this harness, so it can show that no candidate passed the threshold; it cannot show that the +final answer would say "not in your sources". A true system refusal rate needs a +generator eval. + +A higher abstention rate is better on this row, the opposite of how every other row reads. +The two sizes are kept apart on purpose: **candidates passing the threshold** can be as high +as \`candidateK\` (the harness fetches that many to compute \`Recall@10\`), while **passages in +the context window** is what a user's prompt would actually receive. A large first number +with a small second one means the threshold filters nothing and the window is all noise. | Metric | Value | | --- | --- | diff --git a/src/main/eval/types.ts b/src/main/eval/types.ts index de162bf..715396a 100644 --- a/src/main/eval/types.ts +++ b/src/main/eval/types.ts @@ -57,10 +57,16 @@ export type SplitAssignment = Record /** * 解析提交在仓库里的切分清单(`eval/splits.json`)。 * - * 用显式清单而不是 id 哈希(#192 评审):哈希看着确定,但它的确定是“每次结果一样”, - * 不是“每次划分一样”——往 `questions.jsonl` 里加一道题,会把其它题在 validation / - * test 之间挪动,而一个稀有类别(multi-hop、cross-lingual)可以在无人选择的情况下整体 - * 落到某一边。清单让划分是被 review 的,不是被算出来的。 + * 用显式清单而不是按 id 哈希(#192 评审)。先说清楚哈希**不是**哪里坏: + * `hash(id) % 3` 是逐 id 独立计算的,所以它是**稳定**的 —— 新增一道题不会挪动已有的题。 + * + * 它真正不能做的是表达实验设计意图: + * + * - 它无法保证小样本类别分层,于是 multi-hop / cross-lingual 这类稀有类别可能在无 + * 人选择的情况下整体落到某一侧; + * - 新增的题会被默默分到一侧,而不是被决定 —— 而 test 正是“选择”不该被拟合的那一侧。 + * + * 清单让“哪道题在哪一侧”成为一个被 review 的声明,而不是一个被算出来的结果。 */ export function parseSplitAssignment(raw: unknown, source: string): SplitAssignment { if (!raw || typeof raw !== 'object' || Array.isArray(raw)) { @@ -235,14 +241,26 @@ export interface EvalReport { * 不可答问题的单独一组(#192)。 * * 它们不进 `metrics`/`byType`:没有 ground truth,Recall 对它们是 0/0 而不是 0。 - * 它们评的是“该拒答时有没有硬找”——`noResultRate` 越接近 1 越好(在真的没有相关 - * 资料时返回空),`meanRetrieved` 则是“硬找了多少条相似但无关的上下文”。 + * + * 这一组量的是**检索层弃权**,不是模型拒答:harness 不跑生成模型,所以它只能证明 + * “没有候选通过 threshold”,不能证明最终回答会说“资料里没有”。真正的 system refusal + * 要等 generator eval。 */ unanswerable: { questions: number - noResultCount: number - noResultRate: number - meanRetrieved: number + /** 没有任何候选通过 threshold 的问题数。 */ + abstentionCount: number + /** `abstentionCount / questions`,在检索层含义下越高越好。 */ + retrievalAbstentionRate: number + /** + * 通过 threshold 的候选数,**不是**送进 prompt 的条数。 + * + * 上限是 `candidateK`(harness 为了算 Recall@10 故意取满),所以这个数接近 + * `candidateK` 时说明 threshold 基本没挡掉任何东西。 + */ + meanCandidatesRetrieved: number + /** 真正进入 context 窗口的条数:`min(retrievedCount, contextK)`。 */ + meanContextPassages: number } /** `indexingMs` 只用于 #78 的吞吐比较;它不在确定报告里,也不该成为差异原因。 */ timing: { latencyP50Ms: number; latencyP95Ms: number; indexingMs: number } diff --git a/test/evalSplit.test.ts b/test/evalSplit.test.ts index c091051..a57e7dc 100644 --- a/test/evalSplit.test.ts +++ b/test/evalSplit.test.ts @@ -11,10 +11,11 @@ import { * The eval split (#192). Parameters that get swept have to be chosen on questions that * did not take part in the choice. * - * It used to be a hash of the question id. That was reproducible but not *stable*: - * adding a question moved others between the sides, and a rare query type could end up - * entirely on one side without anyone choosing that. The split is now an explicit - * committed manifest, and these pin the properties that make it trustworthy. + * It used to be a hash of the question id. That is reproducible and, because it is computed + * per id, *stable* when a question is added — but it cannot express the experimental + * design: it does not stratify a small corpus, so a rare query type can end up entirely on + * one side without anyone choosing that, and a new question is assigned silently rather + * than deliberately. The split is now an explicit committed manifest. */ const question = (id: string): EvalQuestion => ({ From 20a8c75e6b7d43d0fae20eff4b46b4fd021e8bfa Mon Sep 17 00:00:00 2001 From: MrSibe Date: Wed, 30 Sep 2026 17:54:24 +0800 Subject: [PATCH 13/18] feat(eval): a confusion cluster, and the corpus gets its first hard negatives MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The first of the confusion clusters corpus v2 is built from. Not "near-duplicate documents that lower Recall@5" — deliberately confusable material, so the benchmark can tell retrieval strategies apart at all. ## The cluster Four documents that share almost all their vocabulary and differ in every number: | | v1 | v2 | v3 | | --- | --- | --- | --- | | Path | `/v1/complete` | `/v2/generate` | `/v3/chat` | | Default timeout | 30000 ms | 60000 ms | 45000 ms | | Recommended retries | 2 | 5 | 3 | | Backoff base | 500 ms | 2000 ms | 1000 ms | | Rate limit | 600 rpm | 3000 rpm | 1200 rpm | | Context window | 8192 tokens | 32768 tokens | 65536 tokens | All four documents discuss *timeout, retry, backoff, rate limit and context window* in that order, so a query naming one field matches four passages and only one is right. A migration guide restates every number in comparison tables, which is the hardest negative of the set: it contains v1's timeout **and** v2's **and** v3's in one block. ## The tool that made it possible `npm run eval:blocks ` prints the real `document_blocks.order` values through the same loader and block builder the harness uses. Ground truth `block` is a pipeline ordinal, not a line number a human counted; authoring a corpus by guessing it is how a dataset quietly drifts. `eval/README.md` now says so and points at the command. ## What it measured ``` before after chunks 19 35 Recall@5 0.9211 0.8913 nDCG@10 0.8476 0.7915 MAP@10 0.7965 0.7361 ``` By type, the new cluster behaves as designed: | type | n | Recall@5 | nDCG@10 | | --- | --- | --- | --- | | hard-negative | 4 | **1.0000** | **0.7827** | | multi-hop | 4 | 0.7500 | 0.7289 | | semantic | 19 | 0.9474 | 0.8974 | | cross-lingual | 9 | 0.6667 | 0.4311 | `hard-negative` has perfect Recall@5 and a poor nDCG, which is the exact signature of this kind of material: the correct passage **is** in the top 5, but a confusable sibling outranks it. That is a ranking problem the strategy comparison can now see, and it is invisible on a corpus where the right answer is always at rank 1. `multi-hop` fell to 0.75 because the two comparison questions need four locations each and do not get all four. Also added: 8 questions in total (2 exact, 4 hard-negative, 1 semantic, 2 multi-hop, and two **near-miss unanswerable** — "How much GPU memory does Gateway API v2 require?" and "What is the monthly subscription price of Gateway API v3?", both about documents that discuss the subject at length and never mention the answer). The FIFA questions were a sanity check for a totally out-of-scope query; these are the ones that actually test abstention, because the subject *is* in the library. Unanswerable abstention is still 0/8, with `meanCandidatesRetrieved` at the `candidateK` cap of 20 and `meanContextPassages` at 3. ## Scope, honestly **35 chunks, not 150–300.** One cluster is not the target. The mechanism is proven and verified end to end, and the remaining clusters are the same exercise repeated — but this PR does not pretend the index is large enough for `candidateK ∈ {5,10,20,40}` to be a real sweep. That arrives with the remaining clusters. Part of #192 (child 2, stage 3: first confusion cluster). --- docs/eval/baseline-v1.6.json | 622 ++++++++++++++++++++++----- docs/eval/baseline-v1.6.md | 43 +- eval/README.md | 8 +- eval/corpus/api-gateway-migration.md | 79 ++++ eval/corpus/api-gateway-v1.md | 58 +++ eval/corpus/api-gateway-v2.md | 65 +++ eval/corpus/api-gateway-v3.md | 61 +++ eval/questions.jsonl | 10 + eval/splits.json | 12 +- package.json | 3 +- scripts/eval-blocks.mjs | 54 +++ 11 files changed, 884 insertions(+), 131 deletions(-) create mode 100644 eval/corpus/api-gateway-migration.md create mode 100644 eval/corpus/api-gateway-v1.md create mode 100644 eval/corpus/api-gateway-v2.md create mode 100644 eval/corpus/api-gateway-v3.md create mode 100644 scripts/eval-blocks.mjs diff --git a/docs/eval/baseline-v1.6.json b/docs/eval/baseline-v1.6.json index e087046..ae5c350 100644 --- a/docs/eval/baseline-v1.6.json +++ b/docs/eval/baseline-v1.6.json @@ -16,20 +16,20 @@ "contextK": 3, "threshold": 0.5, "corpus": "eval/corpus", - "documents": 13, - "questions": 44, - "chunkCount": 19 + "documents": 17, + "questions": 54, + "chunkCount": 35 }, "metrics": { - "recallAt1": 0.657895, - "recallAt5": 0.921053, - "recallAt10": 1, - "mrr": 0.800909, - "ndcgAt10": 0.84764, - "hitRateAt5": 0.921053, - "mapAt10": 0.796523, - "contextPrecision": 0.324561, - "contextRecall": 0.921053 + "recallAt1": 0.592391, + "recallAt5": 0.891304, + "recallAt10": 0.940217, + "mrr": 0.759783, + "ndcgAt10": 0.791478, + "hitRateAt5": 0.913043, + "mapAt10": 0.736051, + "contextPrecision": 0.318841, + "contextRecall": 0.869565 }, "byType": [ { @@ -38,58 +38,73 @@ "metrics": { "recallAt1": 0, "recallAt5": 0.666667, - "recallAt10": 1, - "mrr": 0.344577, - "ndcgAt10": 0.503192, + "recallAt10": 0.777778, + "mrr": 0.325926, + "ndcgAt10": 0.431103, "hitRateAt5": 0.666667, - "mapAt10": 0.344577, + "mapAt10": 0.314815, "contextPrecision": 0.222222, "contextRecall": 0.666667 } }, { "type": "exact", - "questions": 6, + "questions": 7, "metrics": { - "recallAt1": 1, + "recallAt1": 0.857143, "recallAt5": 1, "recallAt10": 1, - "mrr": 1, - "ndcgAt10": 1, + "mrr": 0.892857, + "ndcgAt10": 0.918668, "hitRateAt5": 1, - "mapAt10": 1, - "contextPrecision": 0.333333, - "contextRecall": 1 + "mapAt10": 0.892857, + "contextPrecision": 0.285714, + "contextRecall": 0.857143 } }, { - "type": "multi-hop", - "questions": 2, + "type": "hard-negative", + "questions": 4, "metrics": { "recallAt1": 0.5, "recallAt5": 1, "recallAt10": 1, - "mrr": 1, - "ndcgAt10": 0.95986, + "mrr": 0.708333, + "ndcgAt10": 0.782732, "hitRateAt5": 1, - "mapAt10": 0.916667, - "contextPrecision": 0.666667, + "mapAt10": 0.708333, + "contextPrecision": 0.333333, "contextRecall": 1 } }, + { + "type": "multi-hop", + "questions": 4, + "metrics": { + "recallAt1": 0.3125, + "recallAt5": 0.75, + "recallAt10": 0.8125, + "mrr": 0.875, + "ndcgAt10": 0.728888, + "hitRateAt5": 1, + "mapAt10": 0.627083, + "contextPrecision": 0.583333, + "contextRecall": 0.75 + } + }, { "type": "semantic", - "questions": 18, + "questions": 19, "metrics": { - "recallAt1": 0.833333, - "recallAt5": 1, + "recallAt1": 0.789474, + "recallAt5": 0.947368, "recallAt10": 1, - "mrr": 0.907407, - "ndcgAt10": 0.931214, - "hitRateAt5": 1, - "mapAt10": 0.907407, - "contextPrecision": 0.333333, - "contextRecall": 1 + "mrr": 0.864912, + "ndcgAt10": 0.897417, + "hitRateAt5": 0.947368, + "mapAt10": 0.864912, + "contextPrecision": 0.315789, + "contextRecall": 0.947368 } }, { @@ -109,10 +124,10 @@ } ], "unanswerable": { - "questions": 6, + "questions": 8, "abstentionCount": 0, "retrievalAbstentionRate": 0, - "meanCandidatesRetrieved": 19, + "meanCandidatesRetrieved": 20, "meanContextPassages": 3 }, "perQuestion": [ @@ -123,7 +138,7 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, "contextChars": 2074, "matchesByRank": [ [ @@ -146,6 +161,7 @@ [], [], [], + [], [] ] }, @@ -156,7 +172,7 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, "contextChars": 2257, "matchesByRank": [ [ @@ -179,6 +195,7 @@ [], [], [], + [], [] ] }, @@ -189,8 +206,8 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, - "contextChars": 2558, + "retrievedCount": 20, + "contextChars": 2739, "matchesByRank": [ [ 0 @@ -212,6 +229,7 @@ [], [], [], + [], [] ] }, @@ -222,7 +240,7 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, "contextChars": 2113, "matchesByRank": [ [ @@ -245,6 +263,7 @@ [], [], [], + [], [] ] }, @@ -255,7 +274,7 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, "contextChars": 2113, "matchesByRank": [ [ @@ -278,6 +297,7 @@ [], [], [], + [], [] ] }, @@ -288,7 +308,7 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, "contextChars": 1775, "matchesByRank": [ [ @@ -311,6 +331,7 @@ [], [], [], + [], [] ] }, @@ -321,7 +342,7 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, "contextChars": 2091, "matchesByRank": [ [ @@ -344,6 +365,7 @@ [], [], [], + [], [] ] }, @@ -354,7 +376,7 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, "contextChars": 1970, "matchesByRank": [ [ @@ -377,6 +399,7 @@ [], [], [], + [], [] ] }, @@ -387,7 +410,7 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, "contextChars": 1484, "matchesByRank": [ [ @@ -410,6 +433,7 @@ [], [], [], + [], [] ] }, @@ -420,7 +444,7 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, "contextChars": 2682, "matchesByRank": [ [ @@ -443,6 +467,7 @@ [], [], [], + [], [] ] }, @@ -453,7 +478,7 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, "contextChars": 2051, "matchesByRank": [ [ @@ -476,6 +501,7 @@ [], [], [], + [], [] ] }, @@ -486,8 +512,8 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, - "contextChars": 2007, + "retrievedCount": 20, + "contextChars": 2605, "matchesByRank": [ [ 0 @@ -509,6 +535,7 @@ [], [], [], + [], [] ] }, @@ -519,7 +546,7 @@ "answerable": true, "firstRelevantRank": 3, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, "contextChars": 2007, "matchesByRank": [ [], @@ -542,6 +569,7 @@ [], [], [], + [], [] ] }, @@ -552,7 +580,7 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, "contextChars": 2257, "matchesByRank": [ [ @@ -575,6 +603,7 @@ [], [], [], + [], [] ] }, @@ -585,7 +614,7 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, "contextChars": 2621, "matchesByRank": [ [ @@ -608,6 +637,7 @@ [], [], [], + [], [] ] }, @@ -618,8 +648,8 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, - "contextChars": 2558, + "retrievedCount": 20, + "contextChars": 1981, "matchesByRank": [ [ 0 @@ -641,6 +671,7 @@ [], [], [], + [], [] ] }, @@ -651,7 +682,7 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, "contextChars": 1773, "matchesByRank": [ [ @@ -674,6 +705,7 @@ [], [], [], + [], [] ] }, @@ -684,7 +716,7 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, "contextChars": 1970, "matchesByRank": [ [ @@ -707,6 +739,7 @@ [], [], [], + [], [] ] }, @@ -717,7 +750,7 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, "contextChars": 2556, "matchesByRank": [ [ @@ -740,6 +773,7 @@ [], [], [], + [], [] ] }, @@ -750,8 +784,8 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, - "contextChars": 2098, + "retrievedCount": 20, + "contextChars": 2306, "matchesByRank": [ [ 0 @@ -773,6 +807,7 @@ [], [], [], + [], [] ] }, @@ -783,8 +818,8 @@ "answerable": true, "firstRelevantRank": 2, "relevantCount": 1, - "retrievedCount": 19, - "contextChars": 2488, + "retrievedCount": 20, + "contextChars": 2605, "matchesByRank": [ [], [ @@ -806,6 +841,7 @@ [], [], [], + [], [] ] }, @@ -816,7 +852,7 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, "contextChars": 2488, "matchesByRank": [ [ @@ -839,6 +875,7 @@ [], [], [], + [], [] ] }, @@ -849,7 +886,7 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 2, - "retrievedCount": 19, + "retrievedCount": 20, "contextChars": 2257, "matchesByRank": [ [ @@ -874,6 +911,7 @@ [], [], [], + [], [] ] }, @@ -884,7 +922,7 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 2, - "retrievedCount": 19, + "retrievedCount": 20, "contextChars": 1970, "matchesByRank": [ [ @@ -909,6 +947,7 @@ [], [], [], + [], [] ] }, @@ -919,7 +958,7 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, "contextChars": 2098, "matchesByRank": [ [ @@ -942,6 +981,7 @@ [], [], [], + [], [] ] }, @@ -952,7 +992,7 @@ "answerable": true, "firstRelevantRank": 2, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, "contextChars": 1867, "matchesByRank": [ [], @@ -975,6 +1015,7 @@ [], [], [], + [], [] ] }, @@ -985,7 +1026,7 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, "contextChars": 1453, "matchesByRank": [ [ @@ -1008,6 +1049,7 @@ [], [], [], + [], [] ] }, @@ -1018,7 +1060,7 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, "contextChars": 1372, "matchesByRank": [ [ @@ -1041,6 +1083,7 @@ [], [], [], + [], [] ] }, @@ -1051,8 +1094,8 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, - "contextChars": 1500, + "retrievedCount": 20, + "contextChars": 1317, "matchesByRank": [ [ 0 @@ -1074,6 +1117,7 @@ [], [], [], + [], [] ] }, @@ -1084,7 +1128,7 @@ "answerable": true, "firstRelevantRank": 2, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, "contextChars": 1860, "matchesByRank": [ [], @@ -1107,6 +1151,7 @@ [], [], [], + [], [] ] }, @@ -1117,7 +1162,7 @@ "answerable": false, "firstRelevantRank": 0, "relevantCount": 0, - "retrievedCount": 19, + "retrievedCount": 20, "contextChars": 2488, "matchesByRank": [ [], @@ -1138,6 +1183,7 @@ [], [], [], + [], [] ] }, @@ -1148,7 +1194,7 @@ "answerable": false, "firstRelevantRank": 0, "relevantCount": 0, - "retrievedCount": 19, + "retrievedCount": 20, "contextChars": 2257, "matchesByRank": [ [], @@ -1169,6 +1215,7 @@ [], [], [], + [], [] ] }, @@ -1179,7 +1226,7 @@ "answerable": false, "firstRelevantRank": 0, "relevantCount": 0, - "retrievedCount": 19, + "retrievedCount": 20, "contextChars": 2043, "matchesByRank": [ [], @@ -1200,6 +1247,7 @@ [], [], [], + [], [] ] }, @@ -1210,8 +1258,8 @@ "answerable": false, "firstRelevantRank": 0, "relevantCount": 0, - "retrievedCount": 19, - "contextChars": 2588, + "retrievedCount": 20, + "contextChars": 2681, "matchesByRank": [ [], [], @@ -1231,6 +1279,7 @@ [], [], [], + [], [] ] }, @@ -1241,8 +1290,8 @@ "answerable": false, "firstRelevantRank": 0, "relevantCount": 0, - "retrievedCount": 19, - "contextChars": 2488, + "retrievedCount": 20, + "contextChars": 2605, "matchesByRank": [ [], [], @@ -1262,6 +1311,7 @@ [], [], [], + [], [] ] }, @@ -1272,8 +1322,8 @@ "answerable": false, "firstRelevantRank": 0, "relevantCount": 0, - "retrievedCount": 19, - "contextChars": 2071, + "retrievedCount": 20, + "contextChars": 2762, "matchesByRank": [ [], [], @@ -1293,6 +1343,7 @@ [], [], [], + [], [] ] }, @@ -1301,10 +1352,10 @@ "question": "推移质为什么比悬移质更难测?", "type": "cross-lingual", "answerable": true, - "firstRelevantRank": 8, + "firstRelevantRank": 20, "relevantCount": 1, - "retrievedCount": 19, - "contextChars": 2084, + "retrievedCount": 20, + "contextChars": 1099, "matchesByRank": [ [], [], @@ -1313,9 +1364,6 @@ [], [], [], - [ - 0 - ], [], [], [], @@ -1326,7 +1374,11 @@ [], [], [], - [] + [], + [], + [ + 0 + ] ] }, { @@ -1334,10 +1386,10 @@ "question": "河流监测中,每个站点要采几份平行样?", "type": "cross-lingual", "answerable": true, - "firstRelevantRank": 7, + "firstRelevantRank": 20, "relevantCount": 1, - "retrievedCount": 19, - "contextChars": 1480, + "retrievedCount": 20, + "contextChars": 1743, "matchesByRank": [ [], [], @@ -1345,9 +1397,6 @@ [], [], [], - [ - 0 - ], [], [], [], @@ -1359,7 +1408,11 @@ [], [], [], - [] + [], + [], + [ + 0 + ] ] }, { @@ -1369,7 +1422,7 @@ "answerable": true, "firstRelevantRank": 2, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, "contextChars": 1994, "matchesByRank": [ [], @@ -1392,6 +1445,7 @@ [], [], [], + [], [] ] }, @@ -1402,7 +1456,7 @@ "answerable": true, "firstRelevantRank": 2, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, "contextChars": 1994, "matchesByRank": [ [], @@ -1425,6 +1479,7 @@ [], [], [], + [], [] ] }, @@ -1435,7 +1490,7 @@ "answerable": true, "firstRelevantRank": 3, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, "contextChars": 1449, "matchesByRank": [ [], @@ -1458,6 +1513,7 @@ [], [], [], + [], [] ] }, @@ -1468,7 +1524,7 @@ "answerable": true, "firstRelevantRank": 3, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, "contextChars": 1321, "matchesByRank": [ [], @@ -1491,6 +1547,7 @@ [], [], [], + [], [] ] }, @@ -1501,8 +1558,8 @@ "answerable": true, "firstRelevantRank": 2, "relevantCount": 1, - "retrievedCount": 19, - "contextChars": 956, + "retrievedCount": 20, + "contextChars": 1546, "matchesByRank": [ [], [ @@ -1524,6 +1581,7 @@ [], [], [], + [], [] ] }, @@ -1534,7 +1592,7 @@ "answerable": true, "firstRelevantRank": 6, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, "contextChars": 1447, "matchesByRank": [ [], @@ -1557,6 +1615,358 @@ [], [], [], + [], + [] + ] + }, + { + "id": "q045", + "question": "What is the default retry count in Gateway API v2?", + "type": "exact", + "answerable": true, + "firstRelevantRank": 4, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 2898, + "matchesByRank": [ + [], + [], + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q046", + "question": "What is the default request timeout of Gateway API v1?", + "type": "hard-negative", + "answerable": true, + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 2909, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q047", + "question": "What is the context window of Gateway API v3, in tokens?", + "type": "hard-negative", + "answerable": true, + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 2899, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q048", + "question": "What is the default rate limit of Gateway API v2?", + "type": "hard-negative", + "answerable": true, + "firstRelevantRank": 2, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 2854, + "matchesByRank": [ + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q049", + "question": "What is the default request timeout of Gateway API v3?", + "type": "hard-negative", + "answerable": true, + "firstRelevantRank": 3, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 2921, + "matchesByRank": [ + [], + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q050", + "question": "Which Gateway API version waits longest between retries after a failed request?", + "type": "semantic", + "answerable": true, + "firstRelevantRank": 10, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 2910, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q051", + "question": "Compare the default timeouts and rate limits of Gateway API v1 and v3.", + "type": "multi-hop", + "answerable": true, + "firstRelevantRank": 1, + "relevantCount": 4, + "retrievedCount": 20, + "contextChars": 2888, + "matchesByRank": [ + [ + 0 + ], + [ + 2, + 3 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [ + 1 + ], + [ + 3 + ], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q052", + "question": "Compare the retry count and context window of Gateway API v2 and v3.", + "type": "multi-hop", + "answerable": true, + "firstRelevantRank": 2, + "relevantCount": 4, + "retrievedCount": 20, + "contextChars": 2899, + "matchesByRank": [ + [], + [ + 3 + ], + [], + [], + [], + [], + [], + [], + [], + [ + 1 + ], + [ + 2 + ], + [ + 0 + ], + [], + [ + 1 + ], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q053", + "question": "How much GPU memory does Gateway API v2 require to serve a request?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2875, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q054", + "question": "What is the monthly subscription price of Gateway API v3?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2908, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], [] ] } diff --git a/docs/eval/baseline-v1.6.md b/docs/eval/baseline-v1.6.md index 1484d95..4db3097 100644 --- a/docs/eval/baseline-v1.6.md +++ b/docs/eval/baseline-v1.6.md @@ -11,23 +11,23 @@ Generated by `npm run eval`. The numbers below are harness output — do not edi | Retrieval | `dense` | | Ranks | `candidateK=20, threshold=0.5` | | Context width | `contextK=3` | -| Corpus | `eval/corpus` (13 documents) | -| Split | `all` (44 questions, 38 answerable) | -| Index size | 19 chunks | +| Corpus | `eval/corpus` (17 documents) | +| Split | `all` (54 questions, 46 answerable) | +| Index size | 35 chunks | ## Metrics | Metric | Value | | --- | --- | -| Recall@1 | 0.6579 | -| Recall@5 | 0.9211 | -| Recall@10 | 1.0000 | -| MRR | 0.8009 | -| nDCG@10 | 0.8476 | -| Hit rate@5 | 0.9211 | -| MAP@10 | 0.7965 | -| Context precision@3 | 0.3246 | -| Context recall@3 | 0.9211 | +| Recall@1 | 0.5924 | +| Recall@5 | 0.8913 | +| Recall@10 | 0.9402 | +| MRR | 0.7598 | +| nDCG@10 | 0.7915 | +| Hit rate@5 | 0.9130 | +| MAP@10 | 0.7361 | +| Context precision@3 | 0.3188 | +| Context recall@3 | 0.8696 | ### By query type @@ -37,10 +37,11 @@ The type comes from `type` in `questions.jsonl`; untagged questions report as | Type | Questions | Recall@5 | nDCG@10 | Hit rate@5 | MAP@10 | | --- | --- | --- | --- | --- | --- | -| cross-lingual | 9 | 0.6667 | 0.5032 | 0.6667 | 0.3446 | -| exact | 6 | 1.0000 | 1.0000 | 1.0000 | 1.0000 | -| multi-hop | 2 | 1.0000 | 0.9599 | 1.0000 | 0.9167 | -| semantic | 18 | 1.0000 | 0.9312 | 1.0000 | 0.9074 | +| cross-lingual | 9 | 0.6667 | 0.4311 | 0.6667 | 0.3148 | +| exact | 7 | 1.0000 | 0.9187 | 1.0000 | 0.8929 | +| hard-negative | 4 | 1.0000 | 0.7827 | 1.0000 | 0.7083 | +| multi-hop | 4 | 0.7500 | 0.7289 | 1.0000 | 0.6271 | +| semantic | 19 | 0.9474 | 0.8974 | 0.9474 | 0.8649 | | zh | 3 | 1.0000 | 1.0000 | 1.0000 | 1.0000 | ### Unanswerable questions @@ -62,13 +63,13 @@ with a small second one means the threshold filters nothing and the window is al | Metric | Value | | --- | --- | -| Unanswerable questions | 6 | -| Retrieval abstained | 0.0000 (0/6) | -| Mean candidates passing the threshold | 19.00 | +| Unanswerable questions | 8 | +| Retrieval abstained | 0.0000 (0/8) | +| Mean candidates passing the threshold | 20.00 | | Mean passages in the context window | 3.00 | -Timing is informational only and is **not** frozen: indexing 1583 ms, query -p50 13.38 ms, p95 16.66 ms on the +Timing is informational only and is **not** frozen: indexing 4566 ms, query +p50 59.31 ms, p95 85.64 ms on the machine that produced this file. Timing and index size depend on hardware and on the corpus, so they must never be the reason two runs differ. diff --git a/eval/README.md b/eval/README.md index 24deb75..db93d5d 100644 --- a/eval/README.md +++ b/eval/README.md @@ -9,9 +9,10 @@ every experiment (#77, #78) is reported as a delta against that file. ```bash npm run eval:prepare # one-time, networked: download the pinned embedding model npm run eval # offline and deterministic: run the harness, rewrite the baseline -npm run eval:retrieval # strategy comparison (#77) +npm run eval:retrieval # strategy comparison (#77); validation selects, test reports npm run eval:threshold # derive the similarity threshold on validation, report on test npm run eval:sweep # bounded grid over strategy × candidateK × contextK, one dashboard +npm run eval:blocks eval/corpus/foo.md # print the block ordinals ground truth must use ``` ### The harness runs the production configuration @@ -116,7 +117,10 @@ first reads as a permanent miss, the second as a normal hit. Ground truth uses **corpus identity, never database identity**: - `document` is the corpus-relative path. -- `block` is the block ordinal inside the document (`document_blocks.order`). +- `block` is the **`document_blocks.order` the ingestion pipeline produced**, not a line + number and not a paragraph index a human counted. Use `npm run eval:blocks ` to + print the real ordinals through the same loader the harness uses — guessing them is how a + dataset drifts. - `page` is `null` for unpaginated sources. - `quote` is an optional excerpt. The runner fails if the referenced block no longer contains it, so a parser change cannot silently move the ground truth. diff --git a/eval/corpus/api-gateway-migration.md b/eval/corpus/api-gateway-migration.md new file mode 100644 index 0000000..9bb7dbe --- /dev/null +++ b/eval/corpus/api-gateway-migration.md @@ -0,0 +1,79 @@ +# Gateway API Migration Guide + +## Overview + +This guide covers moving a client between Gateway API versions. It deliberately restates +the numbers that differ between versions, because the most common migration defect is a +client that keeps a v1 constant while pointing at a v2 or v3 route. + +The three live surfaces are `/v1/complete`, `/v2/generate` and `/v3/chat`. Any of the three +will reject a body shaped for a different one with `400`, so a silently wrong version is +usually a routing mistake rather than a schema mistake. + +## Choosing a Target + +New integrations should target v3. Existing v2 integrations should move to v3 only when +they need structured output or the larger window, because v3 changes the rate-limit +accounting in a way that can halve effective throughput for structured-output workloads. + +v1 integrations should migrate to v2 at minimum. v1 has no streaming mode, and every +long-answer workload written against v1 pays for it in perceived latency. + +## Endpoint Changes + +| | v1 | v2 | v3 | +| --- | --- | --- | --- | +| Path | `/v1/complete` | `/v2/generate` | `/v3/chat` | +| Body | `prompt` | `messages` | `messages` | +| Streaming | none | server-sent events | server-sent events with `phase` | + +The version lives in the path on all three. A client that versioned the host instead will +not be routed by the gateway at all, and the failure looks like a DNS failure rather than a +version mismatch. + +## Timeout and Retry Changes + +| | v1 | v2 | v3 | +| --- | --- | --- | --- | +| Default timeout | 30000 ms | 60000 ms | 45000 ms | +| Recommended retries | 2 | 5 | 3 | +| Backoff base | 500 ms | 2000 ms | 1000 ms | + +A client migrating from v1 to v2 that keeps the v1 backoff of 500 ms will retry far more +aggressively than the version it is talking to expects, which is a common cause of +self-inflicted `429`s during a cutover. + +Moving from v2 to v3 in the other direction is the riskier one: v3 has fewer recommended +retries and a lower default timeout, so a client tuned for v2's patience will give up +earlier than it did before. + +## Rate Limit Changes + +| | v1 | v2 | v3 | +| --- | --- | --- | --- | +| Default limit | 600 rpm | 3000 rpm | 1200 rpm | +| Burst | 60 | 300 | 120 | + +The v3 structured-output path counts as two requests. A v2 workload that produced 1000 +structured responses per minute was comfortably inside v2's 3000 rpm budget and is almost +exactly at v3's effective 600-per-minute structured ceiling, so the migration is a +throughput change even though the headline number only fell from 3000 to 1200. + +## Context Window Changes + +| | v1 | v2 | v3 | +| --- | --- | --- | --- | +| Window | 8192 tokens | 32768 tokens | 65536 tokens | +| Auto-truncation | no | no | no | + +None of the three truncates automatically; all three reject an oversized request with `400`. +Clients that relied on an upstream provider's truncation find the migration fails loudly +rather than quietly, which is intentional. + +## Checklist + +Before cutting over, confirm the path, the body shape, the timeout, the retry count, the +backoff base and the rate-limit budget. Of those six, the two that are missed most often in +practice are the backoff base and the rate-limit budget, because neither produces an error — +they produce a client that is slower or noisier than it was, which is easy to attribute to +the model rather than to the migration. diff --git a/eval/corpus/api-gateway-v1.md b/eval/corpus/api-gateway-v1.md new file mode 100644 index 0000000..8fc2fe0 --- /dev/null +++ b/eval/corpus/api-gateway-v1.md @@ -0,0 +1,58 @@ +# Gateway API v1 Reference + +## Overview + +The v1 Gateway API is the first generally available surface for text completion. It +accepts a prompt and returns a completion, with no notion of roles, tools or streaming +frames beyond newline-delimited chunks. Clients are expected to be long-lived processes +that hold a single connection open and issue many requests over it. + +v1 is closed to new features. It receives security fixes only, and the deprecation notice +on the v1 endpoint names v2 as the supported successor. + +## Endpoint + +Requests go to the `/v1/complete` path. The path is versioned rather than the host, so a +client that hardcodes the host will silently keep talking to v1 after an upgrade. The +request body carries `prompt`, `max_tokens` and an optional `stop` array; there is no +`system` field, and a system instruction has to be concatenated into the prompt. + +Responses are returned as a single JSON object. v1 has no streaming mode, which is the +change most often cited in migration discussions. + +## Timeouts and Retry + +The default request timeout is **30000 milliseconds**. A request that has not produced any +output within that window is cancelled by the gateway, not by the client, and the client +sees a `504` with the body `{"error":"upstream_timeout"}`. + +The recommended retry count is **2**. The gateway does not retry on the client's behalf, so +this is a client-side contract rather than an enforced limit. The recommended backoff base +is **500 milliseconds**, doubled on each subsequent attempt, which produces waits of 500 ms +then 1000 ms for the two allowed retries. Jitter is not required by v1 but is recommended. + +Retrying a timed-out request is safe because v1 has no server-side session state. Retrying +a request that failed with `429` is not useful unless the `Retry-After` header is honoured +first. + +## Rate Limits + +The default rate limit is **600 requests per minute**, counted per API key rather than per +connection. Bursts of up to 60 requests may be issued within any one-second window before +the limiter engages, so a short burst is allowed even when the per-minute budget is nearly +spent. + +Exceeding the limit returns `429` with a `Retry-After` header in seconds. The limiter +counts a request when the body has been fully received, not when the response is produced, +which means a slow upstream does not consume budget twice. + +## Context Window + +The maximum context window is **8192 tokens**, counting the prompt and the completion +together. A request whose prompt alone exceeds the window is rejected with `400` rather +than truncated, because silent truncation was found to produce worse answers than a +visible failure. + +Token counting uses the same tokenizer as the model, so an approximation by character +count will disagree near the boundary. The gateway exposes a `/v1/tokenize` helper for +clients that want an exact count before sending. diff --git a/eval/corpus/api-gateway-v2.md b/eval/corpus/api-gateway-v2.md new file mode 100644 index 0000000..d39613a --- /dev/null +++ b/eval/corpus/api-gateway-v2.md @@ -0,0 +1,65 @@ +# Gateway API v2 Reference + +## Overview + +The v2 Gateway API replaces the single-prompt completion surface with a message list. It +introduces roles, a real streaming mode and server-side sessions, and it is the surface the +deprecation notice on v1 points at. v2 is feature-frozen: it receives correctness and +security fixes, and v3 is the current recommended target for new integrations. + +The message list is the change that forces most migrations. A v1 prompt with a concatenated +system instruction has to be split into a `system` message and a `user` message, and clients +that relied on concatenation usually find their prompts measurably worse until they split +them. + +## Endpoint + +Requests go to the `/v2/generate` path. Unlike v1, the version is part of the route and +the gateway rejects a v1-shaped body with `400` and a pointer at the migration guide. The +request body carries `messages`, `max_tokens`, `stream` and an optional `tools` array. + +Streaming is enabled per request with `stream: true`, and frames are server-sent events +rather than newline-delimited JSON. Frames carry a monotonically increasing `index`; a +client that reconnects mid-stream must resume from the last index it acknowledged. + +## Timeouts and Retry + +The default request timeout is **60000 milliseconds**, doubled from v1 because v2 sessions +are allowed to think for longer before the first token. The timeout covers the whole turn, +not the gap between frames. + +The recommended retry count is **5**, with a recommended backoff base of **2000 +milliseconds** and full jitter. The higher retry count exists because v2 introduced +server-side sessions, and a retried request may attach to the same session rather than +starting a new one — retrying is therefore usually cheaper than it was in v1. + +A timeout is reported as `504` with `{"error":"upstream_timeout"}` exactly as in v1, so a +client that only inspects that field cannot tell which version produced it. + +## Rate Limits + +The default rate limit is **3000 requests per minute**, counted per API key. Sessions are +counted separately: opening a session costs one request, and each subsequent turn on that +session costs one request, so a long conversation consumes budget linearly. + +Bursts of up to 300 requests may be issued within one second. When a session is already +open, the limiter applies the request to the session's own budget first, which means a +bursty client with many open sessions can exhaust the per-minute budget much faster than +the raw request count suggests. + +## Context Window + +The maximum context window is **32768 tokens**. v2 grew the window partly to make room for +tool definitions, which are counted in the same budget as messages. A request that exceeds +the window is rejected with `400`; v2 does not offer automatic truncation either. + +Because tools are counted, a request with a large tool schema can exceed the window even +when the conversation itself is short. The gateway reports the split between message tokens +and tool tokens in the `usage` block of every response so a client can see which side grew. + +## Sessions + +A session is created implicitly by the first request that omits `session_id`. Sessions +expire after 30 minutes of inactivity. A request that names an expired session is not an +error: the gateway starts a new one and reports the new id, a behaviour that has surprised +several integrators into thinking their retry had lost context. diff --git a/eval/corpus/api-gateway-v3.md b/eval/corpus/api-gateway-v3.md new file mode 100644 index 0000000..e298021 --- /dev/null +++ b/eval/corpus/api-gateway-v3.md @@ -0,0 +1,61 @@ +# Gateway API v3 Reference + +## Overview + +The v3 Gateway API is the current recommended surface. It keeps the message list and +streaming model introduced in v2 and adds structured output, explicit reasoning budgets and +a per-request deadline. v1 and v2 remain available but receive fixes only. + +v3 is not wire-compatible with v2. A v2 body sent to the v3 route is rejected with `400` +and a link to the migration guide, exactly as a v1 body is rejected by v2. + +## Endpoint + +Requests go to the `/v3/chat` path. The body carries `messages`, `max_tokens`, `stream`, +an optional `response_format` describing structured output, and an optional `deadline_ms` +that overrides the default timeout for one request. + +Streaming frames are server-sent events and now carry both an `index` and a `phase`, so a +client can distinguish reasoning frames from answer frames without inspecting the text. +Structured output is delivered as a single final frame; partial structured output is not +emitted, because a half-parsed object was found to be worse than no object. + +## Timeouts and Retry + +The default request timeout is **45000 milliseconds**, between v1's 30 s and v2's 60 s, +chosen after measuring that the median v3 turn finishes in about 11 s while the long tail +benefits from more room than v1 gave. + +The recommended retry count is **3**, with a backoff base of **1000 milliseconds** and full +jitter. v3 adds `deadline_ms`, and a request that carries it uses that value instead of the +default: a client that sets `deadline_ms` to 10000 is not retried by the gateway past the +client's own deadline, which makes the two settings interact in a way v1 and v2 had no +equivalent for. + +## Rate Limits + +The default rate limit is **1200 requests per minute**. Structured-output requests are +counted as two requests, because the gateway runs a validation pass over the produced object +before returning it; a client that migrates a high-volume v2 workload to structured output +can therefore exhaust its budget at half the expected request count. + +Bursts of up to 120 requests may be issued within one second. The limiter applies reasoning +tokens against a separate budget from requests, so a client with long reasoning turns can +hit the token budget before the request budget. + +## Context Window + +The maximum context window is **65536 tokens**, the largest of the three versions, and the +only one where tool schemas, reasoning tokens and messages are reported as three separate +line items in `usage` rather than folded together. + +A request that exceeds the window is rejected with `400`. v3 does not truncate +automatically, but it does report how many tokens the request would need, which makes a +programmatic retry at a smaller size possible without re-tokenising the input. + +## Structured Output + +`response_format` accepts a JSON schema and a strictness flag. Strict mode is slower but +guarantees the object validates against the schema. Non-strict mode is the default and can +return an object that parses but does not match, which the gateway flags in a +`validation_errors` array rather than failing the request. diff --git a/eval/questions.jsonl b/eval/questions.jsonl index c9a3d44..1f4bb57 100644 --- a/eval/questions.jsonl +++ b/eval/questions.jsonl @@ -42,3 +42,13 @@ {"id":"q042","question":"蜜蜂大致从什么温度开始出巢觅食?","type":"cross-lingual","relevant":[{"document":"orchard-pollination.md","page":null,"block":4,"quote":"above roughly twelve degrees Celsius"}]} {"id":"q043","question":"花期遭遇晚霜,主要受损的是什么?","type":"cross-lingual","relevant":[{"document":"orchard-pollination.md","page":null,"block":6,"quote":"destroys the flower's ovary rather than the petals"}]} {"id":"q044","question":"为什么成片的树冠比孤立的树降温效果更好?","type":"cross-lingual","relevant":[{"document":"urban-canopy.md","page":null,"block":4,"quote":"asphalt stores far more heat during the day"}]} +{"id":"q045","type":"exact","question":"What is the default retry count in Gateway API v2?","relevant":[{"document":"api-gateway-v2.md","page":null,"block":9,"quote":"recommended retry count is **5**"}]} +{"id":"q046","type":"hard-negative","question":"What is the default request timeout of Gateway API v1?","relevant":[{"document":"api-gateway-v1.md","page":null,"block":8,"quote":"default request timeout is **30000 milliseconds**"}]} +{"id":"q047","type":"hard-negative","question":"What is the context window of Gateway API v3, in tokens?","relevant":[{"document":"api-gateway-v3.md","page":null,"block":14,"quote":"maximum context window is **65536 tokens**"}]} +{"id":"q048","type":"hard-negative","question":"What is the default rate limit of Gateway API v2?","relevant":[{"document":"api-gateway-v2.md","page":null,"block":12,"quote":"default rate limit is **3000 requests per minute**"}]} +{"id":"q049","type":"hard-negative","question":"What is the default request timeout of Gateway API v3?","relevant":[{"document":"api-gateway-v3.md","page":null,"block":8,"quote":"default request timeout is **45000 milliseconds**"}]} +{"id":"q050","type":"semantic","question":"Which Gateway API version waits longest between retries after a failed request?","relevant":[{"document":"api-gateway-v2.md","page":null,"block":9,"quote":"backoff base of **2000 milliseconds**"}]} +{"id":"q051","type":"multi-hop","question":"Compare the default timeouts and rate limits of Gateway API v1 and v3.","relevant":[{"document":"api-gateway-v1.md","page":null,"block":8,"quote":"default request timeout is **30000 milliseconds**"},{"document":"api-gateway-v1.md","page":null,"block":12,"quote":"default rate limit is **600 requests per minute**"},{"document":"api-gateway-v3.md","page":null,"block":8,"quote":"default request timeout is **45000 milliseconds**"},{"document":"api-gateway-v3.md","page":null,"block":11,"quote":"default rate limit is **1200 requests per minute**"}]} +{"id":"q052","type":"multi-hop","question":"Compare the retry count and context window of Gateway API v2 and v3.","relevant":[{"document":"api-gateway-v2.md","page":null,"block":9,"quote":"recommended retry count is **5**"},{"document":"api-gateway-v2.md","page":null,"block":15,"quote":"maximum context window is **32768 tokens**"},{"document":"api-gateway-v3.md","page":null,"block":9,"quote":"recommended retry count is **3**"},{"document":"api-gateway-v3.md","page":null,"block":14,"quote":"maximum context window is **65536 tokens**"}]} +{"id":"q053","type":"unanswerable","answerable":false,"question":"How much GPU memory does Gateway API v2 require to serve a request?","relevant":[]} +{"id":"q054","type":"unanswerable","answerable":false,"question":"What is the monthly subscription price of Gateway API v3?","relevant":[]} diff --git a/eval/splits.json b/eval/splits.json index c903be1..5aa96d3 100644 --- a/eval/splits.json +++ b/eval/splits.json @@ -42,5 +42,15 @@ "q041": "test", "q042": "test", "q043": "validation", - "q044": "test" + "q044": "test", + "q045": "validation", + "q046": "test", + "q047": "validation", + "q048": "test", + "q049": "test", + "q050": "validation", + "q051": "test", + "q052": "test", + "q053": "validation", + "q054": "test" } diff --git a/package.json b/package.json index 4d9134d..77a6406 100644 --- a/package.json +++ b/package.json @@ -43,7 +43,8 @@ "db:studio": "drizzle-kit studio", "eval:retrieval": "npm run build && node --experimental-transform-types --disable-warning=ExperimentalWarning --disable-warning=MODULE_TYPELESS_PACKAGE_JSON scripts/eval-retrieval.mjs", "eval:threshold": "npm run build && node scripts/eval-threshold.mjs", - "eval:sweep": "npm run build && node scripts/eval-sweep.mjs" + "eval:sweep": "npm run build && node scripts/eval-sweep.mjs", + "eval:blocks": "node --experimental-transform-types --import ./test/ts-resolve.mjs --disable-warning=ExperimentalWarning --disable-warning=MODULE_TYPELESS_PACKAGE_JSON scripts/eval-blocks.mjs" }, "//test": [ "`node --test` strips TypeScript types rather than compiling them, and strip-only", diff --git a/scripts/eval-blocks.mjs b/scripts/eval-blocks.mjs new file mode 100644 index 0000000..ad861f9 --- /dev/null +++ b/scripts/eval-blocks.mjs @@ -0,0 +1,54 @@ +#!/usr/bin/env node +/** + * Print the block ordinals the harness will resolve ground truth against (#192). + * + * Ground truth in `eval/questions.jsonl` is expressed as `document` + `block` + `quote`, + * and `block` is the `document_blocks.order` the ingestion pipeline produced — not a line + * number and not a paragraph index a human counted. Authoring a corpus by hand and then + * guessing those ordinals is how a dataset quietly drifts, so this prints the real ones + * through the **same loader and block builder the harness uses**. + * + * Usage: + * node --experimental-transform-types scripts/eval-blocks.mjs eval/corpus/river-monitoring.md + * node --experimental-transform-types scripts/eval-blocks.mjs eval/corpus/*.md + */ + +import { readFile, readdir } from 'node:fs/promises' +import { extname, join, resolve } from 'node:path' +import { MarkdownLoader } from '../src/main/services/loaders/MarkdownLoader.ts' +import { buildDocumentBlocks } from '../src/main/services/blocks/documentBlocks.ts' + +const targets = process.argv.slice(2) +if (targets.length === 0) { + console.error('usage: node --experimental-transform-types scripts/eval-blocks.mjs [...]') + process.exit(1) +} + +const loader = new MarkdownLoader() + +async function printFile(path) { + const buffer = await readFile(path) + const result = await loader.loadFromBuffer(buffer) + const blocks = buildDocumentBlocks({ content: result.content, structure: result.structure }) + + console.log(`\n${path} — ${blocks.length} blocks`) + for (const block of blocks) { + const text = block.text.replace(/\s+/g, ' ').trim() + const shown = text.length > 96 ? `${text.slice(0, 93)}...` : text + console.log(` ${String(block.order).padStart(3)} ${block.kind.padEnd(9)} ${shown}`) + } + return blocks.length +} + +let total = 0 +for (const target of targets) { + const path = resolve(target) + if (extname(path) === '.md') { + total += await printFile(path) + } else { + // A directory: every markdown file in it, which is the whole corpus case. + const entries = (await readdir(path)).filter((name) => name.endsWith('.md')).sort() + for (const entry of entries) total += await printFile(join(path, entry)) + } +} +console.log(`\ntotal blocks: ${total}`) From 3ffee33dfc7081263c5bc3962680a8b51c326636 Mon Sep 17 00:00:00 2001 From: MrSibe Date: Wed, 30 Sep 2026 17:59:18 +0800 Subject: [PATCH 14/18] feat(eval): a second confusion cluster, and the corpus reaches 53 chunks MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Four more deliberately confusable documents, this time in the monitoring family the corpus already had two members of, so the cluster is confusable with the *existing* documents as well as internally. | | reservoir | coastal | estuary | groundwater | | --- | --- | --- | --- | --- | | Sampling interval | fortnightly | hourly | daily | monthly | | Replicates | 4 | 3 | 5 | 2 | | Sensor depth | 5 m below surface | 1 m below surface | 2 m below surface | 15 m below water table | Every document discusses *sampling interval, replicate samples and sensor depth* in the same order, and each states the network's lowest or highest value ("the least frequent cadence in the network", "the highest of any programme"), so a query lands on several plausible passages and only one is right. Corpus: 19 → 53 chunks, 13 → 21 documents, 54 → 63 questions. ## What it measured ``` after cluster 1 after cluster 2 chunks 35 53 Recall@5 0.8913 0.8774 nDCG@10 0.7915 0.7804 MAP@10 0.7361 0.7280 ``` | type | n | Recall@5 | nDCG@10 | | --- | --- | --- | --- | | cross-lingual | 9 | 0.6667 | 0.4173 | | multi-hop | 5 | 0.7000 | 0.7218 | | semantic | 21 | 0.9048 | 0.8542 | | exact | 8 | 1.0000 | 0.8663 | | hard-negative | 7 | **1.0000** | **0.8758** | | zh | 3 | 1.0000 | 1.0000 | `hard-negative` keeps the signature the cluster was built for: the correct passage is always in the top 5 and a confusable sibling keeps outranking it. ## A result worth pausing on The sweep (all 63 questions) now shows **hybrid ahead of dense on the deciding metric**: | | Recall@5 | nDCG@10 | | --- | --- | --- | | dense, candidateK=20 | 0.8774 | 0.7804 | | hybrid, candidateK=20 | **0.8962** | **0.8098** | | hybrid, candidateK=5 | **0.9104** | 0.8101 | But on the **validation** split the same comparison is a dead heat (both 0.8684), so the held-out rule still declines to adopt. That discrepancy is the honest state of things, and it is a property of the split rather than of hybrid: 24 validation questions is not enough to see a gain that the full set shows. The conclusion is to grow the validation split, **not** to go back to selecting on everything — which is what produced the earlier, over-confident "hybrid clears the rule". ## Unanswerable is still 0/10 `meanCandidatesRetrieved` sits at the `candidateK` cap and `meanContextPassages` at `contextK` for every unanswerable question, so abstention remains unmeasurable here. Two of the ten are near-miss questions ("How much GPU memory does Gateway API v2 require?", "How many litres per second does the reservoir release downstream?") about subjects the corpus discusses at length — the realistic hallucination shape rather than the FIFA sanity check. ## Scope, honestly **53 chunks, not 150–300.** Two clusters is not the target either. `candidateK ∈ {5,10,20,40}` now has more room than it did at 19 chunks, but `candidateK ≥ 10` still behaves identically, so the axis is only partly unlocked. The remaining clusters are the same exercise and the same verification loop; this PR stops where the verified work stops. Part of #192 (child 2, stage 3: confusion clusters). --- docs/eval/baseline-v1.6.json | 520 ++++++++++++++++++++------ docs/eval/baseline-v1.6.md | 42 +-- docs/eval/retrieval-v1.6.json | 102 ++--- docs/eval/retrieval-v1.6.md | 8 +- docs/eval/sweep-v1.6.json | 420 ++++++++++----------- docs/eval/sweep-v1.6.md | 52 +-- docs/eval/threshold-v1.6.json | 280 +++++++------- docs/eval/threshold-v1.6.md | 14 +- eval/corpus/coastal-monitoring.md | 70 ++++ eval/corpus/estuary-monitoring.md | 67 ++++ eval/corpus/groundwater-monitoring.md | 68 ++++ eval/corpus/reservoir-monitoring.md | 69 ++++ eval/questions.jsonl | 9 + eval/splits.json | 11 +- 14 files changed, 1166 insertions(+), 566 deletions(-) create mode 100644 eval/corpus/coastal-monitoring.md create mode 100644 eval/corpus/estuary-monitoring.md create mode 100644 eval/corpus/groundwater-monitoring.md create mode 100644 eval/corpus/reservoir-monitoring.md diff --git a/docs/eval/baseline-v1.6.json b/docs/eval/baseline-v1.6.json index ae5c350..d8171a1 100644 --- a/docs/eval/baseline-v1.6.json +++ b/docs/eval/baseline-v1.6.json @@ -16,20 +16,20 @@ "contextK": 3, "threshold": 0.5, "corpus": "eval/corpus", - "documents": 17, - "questions": 54, - "chunkCount": 35 + "documents": 21, + "questions": 63, + "chunkCount": 53 }, "metrics": { - "recallAt1": 0.592391, - "recallAt5": 0.891304, - "recallAt10": 0.940217, - "mrr": 0.759783, - "ndcgAt10": 0.791478, - "hitRateAt5": 0.913043, - "mapAt10": 0.736051, - "contextPrecision": 0.318841, - "contextRecall": 0.869565 + "recallAt1": 0.59434, + "recallAt5": 0.877358, + "recallAt10": 0.919811, + "mrr": 0.754755, + "ndcgAt10": 0.780449, + "hitRateAt5": 0.90566, + "mapAt10": 0.72804, + "contextPrecision": 0.308176, + "contextRecall": 0.816038 }, "byType": [ { @@ -39,72 +39,72 @@ "recallAt1": 0, "recallAt5": 0.666667, "recallAt10": 0.777778, - "mrr": 0.325926, - "ndcgAt10": 0.431103, + "mrr": 0.299383, + "ndcgAt10": 0.41727, "hitRateAt5": 0.666667, - "mapAt10": 0.314815, - "contextPrecision": 0.222222, - "contextRecall": 0.666667 + "mapAt10": 0.299383, + "contextPrecision": 0.185185, + "contextRecall": 0.555556 } }, { "type": "exact", - "questions": 7, + "questions": 8, "metrics": { - "recallAt1": 0.857143, + "recallAt1": 0.75, "recallAt5": 1, "recallAt10": 1, - "mrr": 0.892857, - "ndcgAt10": 0.918668, + "mrr": 0.822917, + "ndcgAt10": 0.866335, "hitRateAt5": 1, - "mapAt10": 0.892857, - "contextPrecision": 0.285714, - "contextRecall": 0.857143 + "mapAt10": 0.822917, + "contextPrecision": 0.291667, + "contextRecall": 0.875 } }, { "type": "hard-negative", - "questions": 4, + "questions": 7, "metrics": { - "recallAt1": 0.5, + "recallAt1": 0.714286, "recallAt5": 1, "recallAt10": 1, - "mrr": 0.708333, - "ndcgAt10": 0.782732, + "mrr": 0.833333, + "ndcgAt10": 0.875847, "hitRateAt5": 1, - "mapAt10": 0.708333, + "mapAt10": 0.833333, "contextPrecision": 0.333333, "contextRecall": 1 } }, { "type": "multi-hop", - "questions": 4, + "questions": 5, "metrics": { - "recallAt1": 0.3125, - "recallAt5": 0.75, - "recallAt10": 0.8125, - "mrr": 0.875, - "ndcgAt10": 0.728888, + "recallAt1": 0.3, + "recallAt5": 0.7, + "recallAt10": 0.75, + "mrr": 0.9, + "ndcgAt10": 0.721795, "hitRateAt5": 1, - "mapAt10": 0.627083, - "contextPrecision": 0.583333, - "contextRecall": 0.75 + "mapAt10": 0.635, + "contextPrecision": 0.6, + "contextRecall": 0.65 } }, { "type": "semantic", - "questions": 19, + "questions": 21, "metrics": { - "recallAt1": 0.789474, - "recallAt5": 0.947368, - "recallAt10": 1, - "mrr": 0.864912, - "ndcgAt10": 0.897417, - "hitRateAt5": 0.947368, - "mapAt10": 0.864912, - "contextPrecision": 0.315789, - "contextRecall": 0.947368 + "recallAt1": 0.761905, + "recallAt5": 0.904762, + "recallAt10": 0.952381, + "mrr": 0.828139, + "ndcgAt10": 0.85418, + "hitRateAt5": 0.904762, + "mapAt10": 0.82381, + "contextPrecision": 0.285714, + "contextRecall": 0.857143 } }, { @@ -124,7 +124,7 @@ } ], "unanswerable": { - "questions": 8, + "questions": 10, "abstentionCount": 0, "retrievalAbstentionRate": 0, "meanCandidatesRetrieved": 20, @@ -139,7 +139,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 20, - "contextChars": 2074, + "contextChars": 2198, "matchesByRank": [ [ 0 @@ -170,11 +170,13 @@ "question": "How many replicate samples are collected at each river station?", "type": "exact", "answerable": true, - "firstRelevantRank": 1, + "firstRelevantRank": 3, "relevantCount": 1, "retrievedCount": 20, - "contextChars": 2257, + "contextChars": 2297, "matchesByRank": [ + [], + [], [ 0 ], @@ -194,8 +196,6 @@ [], [], [], - [], - [], [] ] }, @@ -207,7 +207,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 20, - "contextChars": 2739, + "contextChars": 2679, "matchesByRank": [ [ 0 @@ -275,7 +275,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 20, - "contextChars": 2113, + "contextChars": 2157, "matchesByRank": [ [ 0 @@ -309,7 +309,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 20, - "contextChars": 1775, + "contextChars": 2364, "matchesByRank": [ [ 0 @@ -343,7 +343,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 20, - "contextChars": 2091, + "contextChars": 2047, "matchesByRank": [ [ 0 @@ -377,7 +377,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 20, - "contextChars": 1970, + "contextChars": 2057, "matchesByRank": [ [ 0 @@ -411,7 +411,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 20, - "contextChars": 1484, + "contextChars": 2136, "matchesByRank": [ [ 0 @@ -513,7 +513,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 20, - "contextChars": 2605, + "contextChars": 2687, "matchesByRank": [ [ 0 @@ -544,16 +544,13 @@ "question": "What is the main environmental concern for tidal energy installations?", "type": "semantic", "answerable": true, - "firstRelevantRank": 3, + "firstRelevantRank": 11, "relevantCount": 1, "retrievedCount": 20, - "contextChars": 2007, + "contextChars": 2579, "matchesByRank": [ [], [], - [ - 0 - ], [], [], [], @@ -562,6 +559,9 @@ [], [], [], + [ + 0 + ], [], [], [], @@ -581,7 +581,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 20, - "contextChars": 2257, + "contextChars": 2910, "matchesByRank": [ [ 0 @@ -615,7 +615,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 20, - "contextChars": 2621, + "contextChars": 2884, "matchesByRank": [ [ 0 @@ -649,7 +649,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 20, - "contextChars": 1981, + "contextChars": 2702, "matchesByRank": [ [ 0 @@ -717,7 +717,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 20, - "contextChars": 1970, + "contextChars": 2561, "matchesByRank": [ [ 0 @@ -751,7 +751,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 20, - "contextChars": 2556, + "contextChars": 2636, "matchesByRank": [ [ 0 @@ -785,7 +785,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 20, - "contextChars": 2306, + "contextChars": 2269, "matchesByRank": [ [ 0 @@ -819,7 +819,7 @@ "firstRelevantRank": 2, "relevantCount": 1, "retrievedCount": 20, - "contextChars": 2605, + "contextChars": 2580, "matchesByRank": [ [], [ @@ -853,7 +853,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 20, - "contextChars": 2488, + "contextChars": 2580, "matchesByRank": [ [ 0 @@ -887,15 +887,12 @@ "firstRelevantRank": 1, "relevantCount": 2, "retrievedCount": 20, - "contextChars": 2257, + "contextChars": 2912, "matchesByRank": [ [ 1 ], [], - [ - 0 - ], [], [], [], @@ -912,6 +909,9 @@ [], [], [], + [ + 0 + ], [] ] }, @@ -959,7 +959,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 20, - "contextChars": 2098, + "contextChars": 1373, "matchesByRank": [ [ 0 @@ -1061,7 +1061,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 20, - "contextChars": 1372, + "contextChars": 1376, "matchesByRank": [ [ 0 @@ -1095,7 +1095,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 20, - "contextChars": 1317, + "contextChars": 1294, "matchesByRank": [ [ 0 @@ -1129,7 +1129,7 @@ "firstRelevantRank": 2, "relevantCount": 1, "retrievedCount": 20, - "contextChars": 1860, + "contextChars": 1955, "matchesByRank": [ [], [ @@ -1163,7 +1163,7 @@ "firstRelevantRank": 0, "relevantCount": 0, "retrievedCount": 20, - "contextChars": 2488, + "contextChars": 2545, "matchesByRank": [ [], [], @@ -1195,7 +1195,7 @@ "firstRelevantRank": 0, "relevantCount": 0, "retrievedCount": 20, - "contextChars": 2257, + "contextChars": 2789, "matchesByRank": [ [], [], @@ -1227,7 +1227,7 @@ "firstRelevantRank": 0, "relevantCount": 0, "retrievedCount": 20, - "contextChars": 2043, + "contextChars": 2755, "matchesByRank": [ [], [], @@ -1259,7 +1259,7 @@ "firstRelevantRank": 0, "relevantCount": 0, "retrievedCount": 20, - "contextChars": 2681, + "contextChars": 2679, "matchesByRank": [ [], [], @@ -1291,7 +1291,7 @@ "firstRelevantRank": 0, "relevantCount": 0, "retrievedCount": 20, - "contextChars": 2605, + "contextChars": 2687, "matchesByRank": [ [], [], @@ -1323,7 +1323,7 @@ "firstRelevantRank": 0, "relevantCount": 0, "retrievedCount": 20, - "contextChars": 2762, + "contextChars": 2819, "matchesByRank": [ [], [], @@ -1352,10 +1352,10 @@ "question": "推移质为什么比悬移质更难测?", "type": "cross-lingual", "answerable": true, - "firstRelevantRank": 20, + "firstRelevantRank": 0, "relevantCount": 1, "retrievedCount": 20, - "contextChars": 1099, + "contextChars": 1986, "matchesByRank": [ [], [], @@ -1376,9 +1376,7 @@ [], [], [], - [ - 0 - ] + [] ] }, { @@ -1386,10 +1384,10 @@ "question": "河流监测中,每个站点要采几份平行样?", "type": "cross-lingual", "answerable": true, - "firstRelevantRank": 20, + "firstRelevantRank": 0, "relevantCount": 1, "retrievedCount": 20, - "contextChars": 1743, + "contextChars": 1337, "matchesByRank": [ [], [], @@ -1410,9 +1408,7 @@ [], [], [], - [ - 0 - ] + [] ] }, { @@ -1488,11 +1484,12 @@ "question": "电芯隔膜一旦熔化会导致什么后果?", "type": "cross-lingual", "answerable": true, - "firstRelevantRank": 3, + "firstRelevantRank": 4, "relevantCount": 1, "retrievedCount": 20, - "contextChars": 1449, + "contextChars": 1294, "matchesByRank": [ + [], [], [], [ @@ -1513,7 +1510,6 @@ [], [], [], - [], [] ] }, @@ -1559,7 +1555,7 @@ "firstRelevantRank": 2, "relevantCount": 1, "retrievedCount": 20, - "contextChars": 1546, + "contextChars": 801, "matchesByRank": [ [], [ @@ -1590,7 +1586,7 @@ "question": "为什么成片的树冠比孤立的树降温效果更好?", "type": "cross-lingual", "answerable": true, - "firstRelevantRank": 6, + "firstRelevantRank": 9, "relevantCount": 1, "retrievedCount": 20, "contextChars": 1447, @@ -1600,12 +1596,12 @@ [], [], [], - [ - 0 - ], [], [], [], + [ + 0 + ], [], [], [], @@ -1969,6 +1965,318 @@ [], [] ] + }, + { + "id": "q055", + "question": "How many replicate samples are collected at each reservoir monitoring station?", + "type": "exact", + "answerable": true, + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 2819, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q056", + "question": "How many replicate samples are collected at each estuary transect station?", + "type": "hard-negative", + "answerable": true, + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 2835, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q057", + "question": "At what depth below the surface does the coastal programme place its shore-station sensor?", + "type": "hard-negative", + "answerable": true, + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 2859, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q058", + "question": "At what depth below the water table do groundwater boreholes carry their pressure transducer?", + "type": "hard-negative", + "answerable": true, + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 2876, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q059", + "question": "Which monitoring programme samples most frequently?", + "type": "semantic", + "answerable": true, + "firstRelevantRank": 5, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 2933, + "matchesByRank": [ + [], + [], + [], + [], + [ + 0 + ], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q060", + "question": "Compare the sampling interval and replicate count of the reservoir and groundwater programmes.", + "type": "multi-hop", + "answerable": true, + "firstRelevantRank": 1, + "relevantCount": 4, + "retrievedCount": 20, + "contextChars": 2879, + "matchesByRank": [ + [ + 2 + ], + [ + 2, + 3 + ], + [ + 0 + ], + [ + 0, + 1 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q061", + "question": "Why does the estuary programme take more replicate samples than the other monitoring programmes?", + "type": "semantic", + "answerable": true, + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 2831, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q062", + "question": "What is the annual operating cost of the coastal monitoring buoys?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2826, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q063", + "question": "How many litres per second does the reservoir release downstream on a typical day?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2809, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] } ] } diff --git a/docs/eval/baseline-v1.6.md b/docs/eval/baseline-v1.6.md index 4db3097..65b9366 100644 --- a/docs/eval/baseline-v1.6.md +++ b/docs/eval/baseline-v1.6.md @@ -11,23 +11,23 @@ Generated by `npm run eval`. The numbers below are harness output — do not edi | Retrieval | `dense` | | Ranks | `candidateK=20, threshold=0.5` | | Context width | `contextK=3` | -| Corpus | `eval/corpus` (17 documents) | -| Split | `all` (54 questions, 46 answerable) | -| Index size | 35 chunks | +| Corpus | `eval/corpus` (21 documents) | +| Split | `all` (63 questions, 53 answerable) | +| Index size | 53 chunks | ## Metrics | Metric | Value | | --- | --- | -| Recall@1 | 0.5924 | -| Recall@5 | 0.8913 | -| Recall@10 | 0.9402 | -| MRR | 0.7598 | -| nDCG@10 | 0.7915 | -| Hit rate@5 | 0.9130 | -| MAP@10 | 0.7361 | -| Context precision@3 | 0.3188 | -| Context recall@3 | 0.8696 | +| Recall@1 | 0.5943 | +| Recall@5 | 0.8774 | +| Recall@10 | 0.9198 | +| MRR | 0.7548 | +| nDCG@10 | 0.7804 | +| Hit rate@5 | 0.9057 | +| MAP@10 | 0.7280 | +| Context precision@3 | 0.3082 | +| Context recall@3 | 0.8160 | ### By query type @@ -37,11 +37,11 @@ The type comes from `type` in `questions.jsonl`; untagged questions report as | Type | Questions | Recall@5 | nDCG@10 | Hit rate@5 | MAP@10 | | --- | --- | --- | --- | --- | --- | -| cross-lingual | 9 | 0.6667 | 0.4311 | 0.6667 | 0.3148 | -| exact | 7 | 1.0000 | 0.9187 | 1.0000 | 0.8929 | -| hard-negative | 4 | 1.0000 | 0.7827 | 1.0000 | 0.7083 | -| multi-hop | 4 | 0.7500 | 0.7289 | 1.0000 | 0.6271 | -| semantic | 19 | 0.9474 | 0.8974 | 0.9474 | 0.8649 | +| cross-lingual | 9 | 0.6667 | 0.4173 | 0.6667 | 0.2994 | +| exact | 8 | 1.0000 | 0.8663 | 1.0000 | 0.8229 | +| hard-negative | 7 | 1.0000 | 0.8758 | 1.0000 | 0.8333 | +| multi-hop | 5 | 0.7000 | 0.7218 | 1.0000 | 0.6350 | +| semantic | 21 | 0.9048 | 0.8542 | 0.9048 | 0.8238 | | zh | 3 | 1.0000 | 1.0000 | 1.0000 | 1.0000 | ### Unanswerable questions @@ -63,13 +63,13 @@ with a small second one means the threshold filters nothing and the window is al | Metric | Value | | --- | --- | -| Unanswerable questions | 8 | -| Retrieval abstained | 0.0000 (0/8) | +| Unanswerable questions | 10 | +| Retrieval abstained | 0.0000 (0/10) | | Mean candidates passing the threshold | 20.00 | | Mean passages in the context window | 3.00 | -Timing is informational only and is **not** frozen: indexing 4566 ms, query -p50 59.31 ms, p95 85.64 ms on the +Timing is informational only and is **not** frozen: indexing 2796 ms, query +p50 13.67 ms, p95 16.34 ms on the machine that produced this file. Timing and index size depend on hardware and on the corpus, so they must never be the reason two runs differ. diff --git a/docs/eval/retrieval-v1.6.json b/docs/eval/retrieval-v1.6.json index 62b9139..d94d41f 100644 --- a/docs/eval/retrieval-v1.6.json +++ b/docs/eval/retrieval-v1.6.json @@ -15,58 +15,58 @@ "id": "dense", "label": "dense (vector)", "split": "validation", - "questions": 16, - "answerableCount": 13, + "questions": 24, + "answerableCount": 19, "chunking": "1000/100", - "chunkCount": 19, - "recallAt1": 0.576923, - "recallAt5": 0.923077, - "recallAt10": 1, - "mrr": 0.778846, - "ndcgAt10": 0.827608, - "hitRateAt5": 0.923077, - "mapAt10": 0.766026, - "contextPrecision": 0.333333, - "contextRecall": 0.923077, - "latencyP95Ms": 16.02 + "chunkCount": 53, + "recallAt1": 0.513158, + "recallAt5": 0.868421, + "recallAt10": 0.921053, + "mrr": 0.720175, + "ndcgAt10": 0.755611, + "hitRateAt5": 0.894737, + "mapAt10": 0.69386, + "contextPrecision": 0.315789, + "contextRecall": 0.802632, + "latencyP95Ms": 13.27 }, { "id": "sparse", "label": "sparse (BM25)", "split": "validation", - "questions": 16, - "answerableCount": 13, + "questions": 24, + "answerableCount": 19, "chunking": "1000/100", - "chunkCount": 19, - "recallAt1": 0.576923, - "recallAt5": 0.615385, - "recallAt10": 0.615385, - "mrr": 0.615385, - "ndcgAt10": 0.609209, - "hitRateAt5": 0.615385, - "mapAt10": 0.602564, - "contextPrecision": 0.230769, - "contextRecall": 0.615385, - "latencyP95Ms": 2.96 + "chunkCount": 53, + "recallAt1": 0.460526, + "recallAt5": 0.671053, + "recallAt10": 0.684211, + "mrr": 0.600877, + "ndcgAt10": 0.610495, + "hitRateAt5": 0.684211, + "mapAt10": 0.576817, + "contextPrecision": 0.263158, + "contextRecall": 0.657895, + "latencyP95Ms": 2.26 }, { "id": "hybrid", "label": "hybrid (RRF of dense + BM25)", "split": "validation", - "questions": 16, - "answerableCount": 13, + "questions": 24, + "answerableCount": 19, "chunking": "1000/100", - "chunkCount": 19, - "recallAt1": 0.576923, - "recallAt5": 0.923077, - "recallAt10": 1, - "mrr": 0.778846, - "ndcgAt10": 0.827608, - "hitRateAt5": 0.923077, - "mapAt10": 0.766026, + "chunkCount": 53, + "recallAt1": 0.513158, + "recallAt5": 0.868421, + "recallAt10": 0.894737, + "mrr": 0.732456, + "ndcgAt10": 0.760265, + "hitRateAt5": 0.894737, + "mapAt10": 0.707018, "contextPrecision": 0.333333, - "contextRecall": 0.923077, - "latencyP95Ms": 19.47 + "contextRecall": 0.855263, + "latencyP95Ms": 13.61 } ], "test": [ @@ -74,20 +74,20 @@ "id": "dense", "label": "dense (vector)", "split": "test", - "questions": 28, - "answerableCount": 25, + "questions": 39, + "answerableCount": 34, "chunking": "1000/100", - "chunkCount": 19, - "recallAt1": 0.7, - "recallAt5": 0.92, - "recallAt10": 1, - "mrr": 0.812381, - "ndcgAt10": 0.858056, - "hitRateAt5": 0.92, - "mapAt10": 0.812381, - "contextPrecision": 0.32, - "contextRecall": 0.92, - "latencyP95Ms": 14.49 + "chunkCount": 53, + "recallAt1": 0.639706, + "recallAt5": 0.882353, + "recallAt10": 0.919118, + "mrr": 0.774079, + "ndcgAt10": 0.794329, + "hitRateAt5": 0.911765, + "mapAt10": 0.747141, + "contextPrecision": 0.303922, + "contextRecall": 0.823529, + "latencyP95Ms": 12.89 } ] } diff --git a/docs/eval/retrieval-v1.6.md b/docs/eval/retrieval-v1.6.md index 9d5009a..b34bdf6 100644 --- a/docs/eval/retrieval-v1.6.md +++ b/docs/eval/retrieval-v1.6.md @@ -16,15 +16,15 @@ to. | Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Context P | n | p95 | | --- | --- | --- | --- | --- | --- | --- | --- | --- | -| dense (vector) | 0.5769 | 0.9231 | 0.7788 | 0.8276 | 0.7660 | 0.3333 | 16 | 16.02 ms | -| sparse (BM25) | 0.5769 | 0.6154 | 0.6154 | 0.6092 | 0.6026 | 0.2308 | 16 | 2.96 ms | -| hybrid (RRF of dense + BM25) | 0.5769 | 0.9231 | 0.7788 | 0.8276 | 0.7660 | 0.3333 | 16 | 19.47 ms | +| dense (vector) | 0.5132 | 0.8684 | 0.7202 | 0.7556 | 0.6939 | 0.3158 | 24 | 13.27 ms | +| sparse (BM25) | 0.4605 | 0.6711 | 0.6009 | 0.6105 | 0.5768 | 0.2632 | 24 | 2.26 ms | +| hybrid (RRF of dense + BM25) | 0.5132 | 0.8684 | 0.7325 | 0.7603 | 0.7070 | 0.3333 | 24 | 13.61 ms | ## Test — reported, not selected | Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Context P | n | p95 | | --- | --- | --- | --- | --- | --- | --- | --- | --- | -| dense (vector) | 0.7000 | 0.9200 | 0.8124 | 0.8581 | 0.8124 | 0.3200 | 28 | 14.49 ms | +| dense (vector) | 0.6397 | 0.8824 | 0.7741 | 0.7943 | 0.7471 | 0.3039 | 39 | 12.89 ms | ## Not evaluated diff --git a/docs/eval/sweep-v1.6.json b/docs/eval/sweep-v1.6.json index 985549e..0a47cf1 100644 --- a/docs/eval/sweep-v1.6.json +++ b/docs/eval/sweep-v1.6.json @@ -5,397 +5,397 @@ "strategy": "dense", "candidateK": 5, "contextK": 3, - "recallAt5": 0.921053, - "ndcgAt10": 0.821192, - "mapAt10": 0.785088, - "contextPrecision": 0.324561, - "contextRecall": 0.921053, + "recallAt5": 0.877358, + "ndcgAt10": 0.767186, + "mapAt10": 0.723113, + "contextPrecision": 0.308176, + "contextRecall": 0.816038, "noResultRate": 0, - "meanContextChars": 1976.1315789473683, - "unanswerableQuestions": 6, + "meanContextChars": 2311.811320754717, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 5, "unanswerableContextPassages": 3, - "chunkCount": 19, - "latencyP95Ms": 14.04 + "chunkCount": 53, + "latencyP95Ms": 12.3 }, { "strategy": "dense", "candidateK": 5, "contextK": 5, - "recallAt5": 0.921053, - "ndcgAt10": 0.821192, - "mapAt10": 0.785088, - "contextPrecision": 0.194737, - "contextRecall": 0.921053, + "recallAt5": 0.877358, + "ndcgAt10": 0.767186, + "mapAt10": 0.723113, + "contextPrecision": 0.2, + "contextRecall": 0.877358, "noResultRate": 0, - "meanContextChars": 3188.1315789473683, - "unanswerableQuestions": 6, + "meanContextChars": 4007.0377358490564, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 5, "unanswerableContextPassages": 5, - "chunkCount": 19, - "latencyP95Ms": 13.63 + "chunkCount": 53, + "latencyP95Ms": 12.37 }, { "strategy": "dense", "candidateK": 10, "contextK": 3, - "recallAt5": 0.921053, - "ndcgAt10": 0.84764, - "mapAt10": 0.796523, - "contextPrecision": 0.324561, - "contextRecall": 0.921053, + "recallAt5": 0.877358, + "ndcgAt10": 0.780449, + "mapAt10": 0.72804, + "contextPrecision": 0.308176, + "contextRecall": 0.816038, "noResultRate": 0, - "meanContextChars": 1976.1315789473683, - "unanswerableQuestions": 6, + "meanContextChars": 2311.811320754717, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 10, "unanswerableContextPassages": 3, - "chunkCount": 19, - "latencyP95Ms": 13.84 + "chunkCount": 53, + "latencyP95Ms": 12.66 }, { "strategy": "dense", "candidateK": 10, "contextK": 5, - "recallAt5": 0.921053, - "ndcgAt10": 0.84764, - "mapAt10": 0.796523, - "contextPrecision": 0.194737, - "contextRecall": 0.921053, + "recallAt5": 0.877358, + "ndcgAt10": 0.780449, + "mapAt10": 0.72804, + "contextPrecision": 0.2, + "contextRecall": 0.877358, "noResultRate": 0, - "meanContextChars": 3188.1315789473683, - "unanswerableQuestions": 6, + "meanContextChars": 4007.0377358490564, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 10, "unanswerableContextPassages": 5, - "chunkCount": 19, - "latencyP95Ms": 15.21 + "chunkCount": 53, + "latencyP95Ms": 13.42 }, { "strategy": "dense", "candidateK": 10, "contextK": 8, - "recallAt5": 0.921053, - "ndcgAt10": 0.84764, - "mapAt10": 0.796523, - "contextPrecision": 0.131579, - "contextRecall": 1, + "recallAt5": 0.877358, + "ndcgAt10": 0.780449, + "mapAt10": 0.72804, + "contextPrecision": 0.127358, + "contextRecall": 0.877358, "noResultRate": 0, - "meanContextChars": 5151.289473684211, - "unanswerableQuestions": 6, + "meanContextChars": 6658.698113207547, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 10, "unanswerableContextPassages": 8, - "chunkCount": 19, - "latencyP95Ms": 16.35 + "chunkCount": 53, + "latencyP95Ms": 12.8 }, { "strategy": "dense", "candidateK": 20, "contextK": 3, - "recallAt5": 0.921053, - "ndcgAt10": 0.84764, - "mapAt10": 0.796523, - "contextPrecision": 0.324561, - "contextRecall": 0.921053, + "recallAt5": 0.877358, + "ndcgAt10": 0.780449, + "mapAt10": 0.72804, + "contextPrecision": 0.308176, + "contextRecall": 0.816038, "noResultRate": 0, - "meanContextChars": 1976.1315789473683, - "unanswerableQuestions": 6, + "meanContextChars": 2311.811320754717, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, - "unanswerableCandidates": 19, + "unanswerableCandidates": 20, "unanswerableContextPassages": 3, - "chunkCount": 19, - "latencyP95Ms": 16.45 + "chunkCount": 53, + "latencyP95Ms": 13.23 }, { "strategy": "dense", "candidateK": 20, "contextK": 5, - "recallAt5": 0.921053, - "ndcgAt10": 0.84764, - "mapAt10": 0.796523, - "contextPrecision": 0.194737, - "contextRecall": 0.921053, + "recallAt5": 0.877358, + "ndcgAt10": 0.780449, + "mapAt10": 0.72804, + "contextPrecision": 0.2, + "contextRecall": 0.877358, "noResultRate": 0, - "meanContextChars": 3188.1315789473683, - "unanswerableQuestions": 6, + "meanContextChars": 4007.0377358490564, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, - "unanswerableCandidates": 19, + "unanswerableCandidates": 20, "unanswerableContextPassages": 5, - "chunkCount": 19, - "latencyP95Ms": 17.91 + "chunkCount": 53, + "latencyP95Ms": 13.55 }, { "strategy": "dense", "candidateK": 20, "contextK": 8, - "recallAt5": 0.921053, - "ndcgAt10": 0.84764, - "mapAt10": 0.796523, - "contextPrecision": 0.131579, - "contextRecall": 1, + "recallAt5": 0.877358, + "ndcgAt10": 0.780449, + "mapAt10": 0.72804, + "contextPrecision": 0.127358, + "contextRecall": 0.877358, "noResultRate": 0, - "meanContextChars": 5151.289473684211, - "unanswerableQuestions": 6, + "meanContextChars": 6658.698113207547, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, - "unanswerableCandidates": 19, + "unanswerableCandidates": 20, "unanswerableContextPassages": 8, - "chunkCount": 19, - "latencyP95Ms": 14.13 + "chunkCount": 53, + "latencyP95Ms": 13.75 }, { "strategy": "dense", "candidateK": 40, "contextK": 3, - "recallAt5": 0.921053, - "ndcgAt10": 0.84764, - "mapAt10": 0.796523, - "contextPrecision": 0.324561, - "contextRecall": 0.921053, + "recallAt5": 0.877358, + "ndcgAt10": 0.780449, + "mapAt10": 0.72804, + "contextPrecision": 0.308176, + "contextRecall": 0.816038, "noResultRate": 0, - "meanContextChars": 1976.1315789473683, - "unanswerableQuestions": 6, + "meanContextChars": 2311.811320754717, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, - "unanswerableCandidates": 19, + "unanswerableCandidates": 40, "unanswerableContextPassages": 3, - "chunkCount": 19, - "latencyP95Ms": 13.35 + "chunkCount": 53, + "latencyP95Ms": 15.44 }, { "strategy": "dense", "candidateK": 40, "contextK": 5, - "recallAt5": 0.921053, - "ndcgAt10": 0.84764, - "mapAt10": 0.796523, - "contextPrecision": 0.194737, - "contextRecall": 0.921053, + "recallAt5": 0.877358, + "ndcgAt10": 0.780449, + "mapAt10": 0.72804, + "contextPrecision": 0.2, + "contextRecall": 0.877358, "noResultRate": 0, - "meanContextChars": 3188.1315789473683, - "unanswerableQuestions": 6, + "meanContextChars": 4007.0377358490564, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, - "unanswerableCandidates": 19, + "unanswerableCandidates": 40, "unanswerableContextPassages": 5, - "chunkCount": 19, - "latencyP95Ms": 12.73 + "chunkCount": 53, + "latencyP95Ms": 14.98 }, { "strategy": "dense", "candidateK": 40, "contextK": 8, - "recallAt5": 0.921053, - "ndcgAt10": 0.84764, - "mapAt10": 0.796523, - "contextPrecision": 0.131579, - "contextRecall": 1, + "recallAt5": 0.877358, + "ndcgAt10": 0.780449, + "mapAt10": 0.72804, + "contextPrecision": 0.127358, + "contextRecall": 0.877358, "noResultRate": 0, - "meanContextChars": 5151.289473684211, - "unanswerableQuestions": 6, + "meanContextChars": 6658.698113207547, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, - "unanswerableCandidates": 19, + "unanswerableCandidates": 40, "unanswerableContextPassages": 8, - "chunkCount": 19, - "latencyP95Ms": 15.12 + "chunkCount": 53, + "latencyP95Ms": 14.68 }, { "strategy": "hybrid", "candidateK": 5, "contextK": 3, - "recallAt5": 0.921053, - "ndcgAt10": 0.830904, - "mapAt10": 0.798246, - "contextPrecision": 0.324561, - "contextRecall": 0.921053, + "recallAt5": 0.910377, + "ndcgAt10": 0.810136, + "mapAt10": 0.768239, + "contextPrecision": 0.320755, + "contextRecall": 0.853774, "noResultRate": 0, - "meanContextChars": 1948.657894736842, - "unanswerableQuestions": 6, + "meanContextChars": 2307.3207547169814, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 5, "unanswerableContextPassages": 3, - "chunkCount": 19, - "latencyP95Ms": 14.01 + "chunkCount": 53, + "latencyP95Ms": 13.45 }, { "strategy": "hybrid", "candidateK": 5, "contextK": 5, - "recallAt5": 0.921053, - "ndcgAt10": 0.830904, - "mapAt10": 0.798246, - "contextPrecision": 0.194737, - "contextRecall": 0.921053, + "recallAt5": 0.910377, + "ndcgAt10": 0.810136, + "mapAt10": 0.768239, + "contextPrecision": 0.211321, + "contextRecall": 0.910377, "noResultRate": 0, - "meanContextChars": 3215.5789473684213, - "unanswerableQuestions": 6, + "meanContextChars": 3936.566037735849, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 5, "unanswerableContextPassages": 5, - "chunkCount": 19, - "latencyP95Ms": 15.81 + "chunkCount": 53, + "latencyP95Ms": 77.74 }, { "strategy": "hybrid", "candidateK": 10, "contextK": 3, - "recallAt5": 0.921053, - "ndcgAt10": 0.857352, - "mapAt10": 0.80968, - "contextPrecision": 0.324561, - "contextRecall": 0.921053, + "recallAt5": 0.896226, + "ndcgAt10": 0.806807, + "mapAt10": 0.757966, + "contextPrecision": 0.320755, + "contextRecall": 0.853774, "noResultRate": 0, - "meanContextChars": 1963.342105263158, - "unanswerableQuestions": 6, + "meanContextChars": 2321.264150943396, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 10, "unanswerableContextPassages": 3, - "chunkCount": 19, - "latencyP95Ms": 15.74 + "chunkCount": 53, + "latencyP95Ms": 14.42 }, { "strategy": "hybrid", "candidateK": 10, "contextK": 5, - "recallAt5": 0.921053, - "ndcgAt10": 0.857352, - "mapAt10": 0.80968, - "contextPrecision": 0.194737, - "contextRecall": 0.921053, + "recallAt5": 0.896226, + "ndcgAt10": 0.806807, + "mapAt10": 0.757966, + "contextPrecision": 0.203774, + "contextRecall": 0.896226, "noResultRate": 0, - "meanContextChars": 3345.0789473684213, - "unanswerableQuestions": 6, + "meanContextChars": 3994.509433962264, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 10, "unanswerableContextPassages": 5, - "chunkCount": 19, - "latencyP95Ms": 16.3 + "chunkCount": 53, + "latencyP95Ms": 13.8 }, { "strategy": "hybrid", "candidateK": 10, "contextK": 8, - "recallAt5": 0.921053, - "ndcgAt10": 0.857352, - "mapAt10": 0.80968, - "contextPrecision": 0.131579, - "contextRecall": 1, + "recallAt5": 0.896226, + "ndcgAt10": 0.806807, + "mapAt10": 0.757966, + "contextPrecision": 0.132075, + "contextRecall": 0.900943, "noResultRate": 0, - "meanContextChars": 5319.868421052632, - "unanswerableQuestions": 6, + "meanContextChars": 6547.641509433963, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 10, "unanswerableContextPassages": 8, - "chunkCount": 19, - "latencyP95Ms": 16.95 + "chunkCount": 53, + "latencyP95Ms": 13.04 }, { "strategy": "hybrid", "candidateK": 20, "contextK": 3, - "recallAt5": 0.921053, - "ndcgAt10": 0.857352, - "mapAt10": 0.80968, - "contextPrecision": 0.324561, - "contextRecall": 0.921053, + "recallAt5": 0.896226, + "ndcgAt10": 0.809819, + "mapAt10": 0.760469, + "contextPrecision": 0.320755, + "contextRecall": 0.853774, "noResultRate": 0, - "meanContextChars": 1958.7631578947369, - "unanswerableQuestions": 6, + "meanContextChars": 2322.830188679245, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, - "unanswerableCandidates": 19, + "unanswerableCandidates": 20, "unanswerableContextPassages": 3, - "chunkCount": 19, - "latencyP95Ms": 14.32 + "chunkCount": 53, + "latencyP95Ms": 15.55 }, { "strategy": "hybrid", "candidateK": 20, "contextK": 5, - "recallAt5": 0.921053, - "ndcgAt10": 0.857352, - "mapAt10": 0.80968, - "contextPrecision": 0.194737, - "contextRecall": 0.921053, + "recallAt5": 0.896226, + "ndcgAt10": 0.809819, + "mapAt10": 0.760469, + "contextPrecision": 0.203774, + "contextRecall": 0.896226, "noResultRate": 0, - "meanContextChars": 3282.1052631578946, - "unanswerableQuestions": 6, + "meanContextChars": 4011.377358490566, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, - "unanswerableCandidates": 19, + "unanswerableCandidates": 20, "unanswerableContextPassages": 5, - "chunkCount": 19, - "latencyP95Ms": 14.82 + "chunkCount": 53, + "latencyP95Ms": 16.36 }, { "strategy": "hybrid", "candidateK": 20, "contextK": 8, - "recallAt5": 0.921053, - "ndcgAt10": 0.857352, - "mapAt10": 0.80968, - "contextPrecision": 0.131579, - "contextRecall": 1, + "recallAt5": 0.896226, + "ndcgAt10": 0.809819, + "mapAt10": 0.760469, + "contextPrecision": 0.132075, + "contextRecall": 0.90566, "noResultRate": 0, - "meanContextChars": 5357.315789473684, - "unanswerableQuestions": 6, + "meanContextChars": 6660.264150943396, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, - "unanswerableCandidates": 19, + "unanswerableCandidates": 20, "unanswerableContextPassages": 8, - "chunkCount": 19, - "latencyP95Ms": 13.74 + "chunkCount": 53, + "latencyP95Ms": 14.66 }, { "strategy": "hybrid", "candidateK": 40, "contextK": 3, - "recallAt5": 0.921053, - "ndcgAt10": 0.857352, - "mapAt10": 0.80968, - "contextPrecision": 0.324561, - "contextRecall": 0.921053, + "recallAt5": 0.896226, + "ndcgAt10": 0.809819, + "mapAt10": 0.760469, + "contextPrecision": 0.320755, + "contextRecall": 0.853774, "noResultRate": 0, - "meanContextChars": 1958.7631578947369, - "unanswerableQuestions": 6, + "meanContextChars": 2322.830188679245, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, - "unanswerableCandidates": 19, + "unanswerableCandidates": 40, "unanswerableContextPassages": 3, - "chunkCount": 19, - "latencyP95Ms": 18.46 + "chunkCount": 53, + "latencyP95Ms": 14.55 }, { "strategy": "hybrid", "candidateK": 40, "contextK": 5, - "recallAt5": 0.921053, - "ndcgAt10": 0.857352, - "mapAt10": 0.80968, - "contextPrecision": 0.194737, - "contextRecall": 0.921053, + "recallAt5": 0.896226, + "ndcgAt10": 0.809819, + "mapAt10": 0.760469, + "contextPrecision": 0.203774, + "contextRecall": 0.896226, "noResultRate": 0, - "meanContextChars": 3282.1052631578946, - "unanswerableQuestions": 6, + "meanContextChars": 3989.735849056604, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, - "unanswerableCandidates": 19, + "unanswerableCandidates": 40, "unanswerableContextPassages": 5, - "chunkCount": 19, - "latencyP95Ms": 16.37 + "chunkCount": 53, + "latencyP95Ms": 14.41 }, { "strategy": "hybrid", "candidateK": 40, "contextK": 8, - "recallAt5": 0.921053, - "ndcgAt10": 0.857352, - "mapAt10": 0.80968, - "contextPrecision": 0.131579, - "contextRecall": 1, + "recallAt5": 0.896226, + "ndcgAt10": 0.809819, + "mapAt10": 0.760469, + "contextPrecision": 0.132075, + "contextRecall": 0.90566, "noResultRate": 0, - "meanContextChars": 5357.315789473684, - "unanswerableQuestions": 6, + "meanContextChars": 6660.792452830188, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, - "unanswerableCandidates": 19, + "unanswerableCandidates": 40, "unanswerableContextPassages": 8, - "chunkCount": 19, - "latencyP95Ms": 20.86 + "chunkCount": 53, + "latencyP95Ms": 14.47 } ] } diff --git a/docs/eval/sweep-v1.6.md b/docs/eval/sweep-v1.6.md index a3c6dd7..46aeb3e 100644 --- a/docs/eval/sweep-v1.6.md +++ b/docs/eval/sweep-v1.6.md @@ -12,28 +12,28 @@ Each row differs from its neighbour in one parameter. | Strategy | candidateK | contextK | Recall@5 | nDCG@10 | MAP@10 | Context P | Context R | No-result | Context chars | Unans. abstained | Unans. cands | Unans. ctx | Index | p95 | | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | -| dense | 5 | 3 | 0.9211 | 0.8212 | 0.7851 | 0.3246 | 0.9211 | 0.0000 | 1976 | 0.0000 | 5.0 | 3.0 | 19 | 14.04 ms | -| dense | 5 | 5 | 0.9211 | 0.8212 | 0.7851 | 0.1947 | 0.9211 | 0.0000 | 3188 | 0.0000 | 5.0 | 5.0 | 19 | 13.63 ms | -| dense | 10 | 3 | 0.9211 | 0.8476 | 0.7965 | 0.3246 | 0.9211 | 0.0000 | 1976 | 0.0000 | 10.0 | 3.0 | 19 | 13.84 ms | -| dense | 10 | 5 | 0.9211 | 0.8476 | 0.7965 | 0.1947 | 0.9211 | 0.0000 | 3188 | 0.0000 | 10.0 | 5.0 | 19 | 15.21 ms | -| dense | 10 | 8 | 0.9211 | 0.8476 | 0.7965 | 0.1316 | 1.0000 | 0.0000 | 5151 | 0.0000 | 10.0 | 8.0 | 19 | 16.35 ms | -| dense | 20 | 3 | 0.9211 | 0.8476 | 0.7965 | 0.3246 | 0.9211 | 0.0000 | 1976 | 0.0000 | 19.0 | 3.0 | 19 | 16.45 ms | -| dense | 20 | 5 | 0.9211 | 0.8476 | 0.7965 | 0.1947 | 0.9211 | 0.0000 | 3188 | 0.0000 | 19.0 | 5.0 | 19 | 17.91 ms | -| dense | 20 | 8 | 0.9211 | 0.8476 | 0.7965 | 0.1316 | 1.0000 | 0.0000 | 5151 | 0.0000 | 19.0 | 8.0 | 19 | 14.13 ms | -| dense | 40 | 3 | 0.9211 | 0.8476 | 0.7965 | 0.3246 | 0.9211 | 0.0000 | 1976 | 0.0000 | 19.0 | 3.0 | 19 | 13.35 ms | -| dense | 40 | 5 | 0.9211 | 0.8476 | 0.7965 | 0.1947 | 0.9211 | 0.0000 | 3188 | 0.0000 | 19.0 | 5.0 | 19 | 12.73 ms | -| dense | 40 | 8 | 0.9211 | 0.8476 | 0.7965 | 0.1316 | 1.0000 | 0.0000 | 5151 | 0.0000 | 19.0 | 8.0 | 19 | 15.12 ms | -| hybrid | 5 | 3 | 0.9211 | 0.8309 | 0.7982 | 0.3246 | 0.9211 | 0.0000 | 1949 | 0.0000 | 5.0 | 3.0 | 19 | 14.01 ms | -| hybrid | 5 | 5 | 0.9211 | 0.8309 | 0.7982 | 0.1947 | 0.9211 | 0.0000 | 3216 | 0.0000 | 5.0 | 5.0 | 19 | 15.81 ms | -| hybrid | 10 | 3 | 0.9211 | 0.8574 | 0.8097 | 0.3246 | 0.9211 | 0.0000 | 1963 | 0.0000 | 10.0 | 3.0 | 19 | 15.74 ms | -| hybrid | 10 | 5 | 0.9211 | 0.8574 | 0.8097 | 0.1947 | 0.9211 | 0.0000 | 3345 | 0.0000 | 10.0 | 5.0 | 19 | 16.30 ms | -| hybrid | 10 | 8 | 0.9211 | 0.8574 | 0.8097 | 0.1316 | 1.0000 | 0.0000 | 5320 | 0.0000 | 10.0 | 8.0 | 19 | 16.95 ms | -| hybrid | 20 | 3 | 0.9211 | 0.8574 | 0.8097 | 0.3246 | 0.9211 | 0.0000 | 1959 | 0.0000 | 19.0 | 3.0 | 19 | 14.32 ms | -| hybrid | 20 | 5 | 0.9211 | 0.8574 | 0.8097 | 0.1947 | 0.9211 | 0.0000 | 3282 | 0.0000 | 19.0 | 5.0 | 19 | 14.82 ms | -| hybrid | 20 | 8 | 0.9211 | 0.8574 | 0.8097 | 0.1316 | 1.0000 | 0.0000 | 5357 | 0.0000 | 19.0 | 8.0 | 19 | 13.74 ms | -| hybrid | 40 | 3 | 0.9211 | 0.8574 | 0.8097 | 0.3246 | 0.9211 | 0.0000 | 1959 | 0.0000 | 19.0 | 3.0 | 19 | 18.46 ms | -| hybrid | 40 | 5 | 0.9211 | 0.8574 | 0.8097 | 0.1947 | 0.9211 | 0.0000 | 3282 | 0.0000 | 19.0 | 5.0 | 19 | 16.37 ms | -| hybrid | 40 | 8 | 0.9211 | 0.8574 | 0.8097 | 0.1316 | 1.0000 | 0.0000 | 5357 | 0.0000 | 19.0 | 8.0 | 19 | 20.86 ms | +| dense | 5 | 3 | 0.8774 | 0.7672 | 0.7231 | 0.3082 | 0.8160 | 0.0000 | 2312 | 0.0000 | 5.0 | 3.0 | 53 | 12.30 ms | +| dense | 5 | 5 | 0.8774 | 0.7672 | 0.7231 | 0.2000 | 0.8774 | 0.0000 | 4007 | 0.0000 | 5.0 | 5.0 | 53 | 12.37 ms | +| dense | 10 | 3 | 0.8774 | 0.7804 | 0.7280 | 0.3082 | 0.8160 | 0.0000 | 2312 | 0.0000 | 10.0 | 3.0 | 53 | 12.66 ms | +| dense | 10 | 5 | 0.8774 | 0.7804 | 0.7280 | 0.2000 | 0.8774 | 0.0000 | 4007 | 0.0000 | 10.0 | 5.0 | 53 | 13.42 ms | +| dense | 10 | 8 | 0.8774 | 0.7804 | 0.7280 | 0.1274 | 0.8774 | 0.0000 | 6659 | 0.0000 | 10.0 | 8.0 | 53 | 12.80 ms | +| dense | 20 | 3 | 0.8774 | 0.7804 | 0.7280 | 0.3082 | 0.8160 | 0.0000 | 2312 | 0.0000 | 20.0 | 3.0 | 53 | 13.23 ms | +| dense | 20 | 5 | 0.8774 | 0.7804 | 0.7280 | 0.2000 | 0.8774 | 0.0000 | 4007 | 0.0000 | 20.0 | 5.0 | 53 | 13.55 ms | +| dense | 20 | 8 | 0.8774 | 0.7804 | 0.7280 | 0.1274 | 0.8774 | 0.0000 | 6659 | 0.0000 | 20.0 | 8.0 | 53 | 13.75 ms | +| dense | 40 | 3 | 0.8774 | 0.7804 | 0.7280 | 0.3082 | 0.8160 | 0.0000 | 2312 | 0.0000 | 40.0 | 3.0 | 53 | 15.44 ms | +| dense | 40 | 5 | 0.8774 | 0.7804 | 0.7280 | 0.2000 | 0.8774 | 0.0000 | 4007 | 0.0000 | 40.0 | 5.0 | 53 | 14.98 ms | +| dense | 40 | 8 | 0.8774 | 0.7804 | 0.7280 | 0.1274 | 0.8774 | 0.0000 | 6659 | 0.0000 | 40.0 | 8.0 | 53 | 14.68 ms | +| hybrid | 5 | 3 | 0.9104 | 0.8101 | 0.7682 | 0.3208 | 0.8538 | 0.0000 | 2307 | 0.0000 | 5.0 | 3.0 | 53 | 13.45 ms | +| hybrid | 5 | 5 | 0.9104 | 0.8101 | 0.7682 | 0.2113 | 0.9104 | 0.0000 | 3937 | 0.0000 | 5.0 | 5.0 | 53 | 77.74 ms | +| hybrid | 10 | 3 | 0.8962 | 0.8068 | 0.7580 | 0.3208 | 0.8538 | 0.0000 | 2321 | 0.0000 | 10.0 | 3.0 | 53 | 14.42 ms | +| hybrid | 10 | 5 | 0.8962 | 0.8068 | 0.7580 | 0.2038 | 0.8962 | 0.0000 | 3995 | 0.0000 | 10.0 | 5.0 | 53 | 13.80 ms | +| hybrid | 10 | 8 | 0.8962 | 0.8068 | 0.7580 | 0.1321 | 0.9009 | 0.0000 | 6548 | 0.0000 | 10.0 | 8.0 | 53 | 13.04 ms | +| hybrid | 20 | 3 | 0.8962 | 0.8098 | 0.7605 | 0.3208 | 0.8538 | 0.0000 | 2323 | 0.0000 | 20.0 | 3.0 | 53 | 15.55 ms | +| hybrid | 20 | 5 | 0.8962 | 0.8098 | 0.7605 | 0.2038 | 0.8962 | 0.0000 | 4011 | 0.0000 | 20.0 | 5.0 | 53 | 16.36 ms | +| hybrid | 20 | 8 | 0.8962 | 0.8098 | 0.7605 | 0.1321 | 0.9057 | 0.0000 | 6660 | 0.0000 | 20.0 | 8.0 | 53 | 14.66 ms | +| hybrid | 40 | 3 | 0.8962 | 0.8098 | 0.7605 | 0.3208 | 0.8538 | 0.0000 | 2323 | 0.0000 | 40.0 | 3.0 | 53 | 14.55 ms | +| hybrid | 40 | 5 | 0.8962 | 0.8098 | 0.7605 | 0.2038 | 0.8962 | 0.0000 | 3990 | 0.0000 | 40.0 | 5.0 | 53 | 14.41 ms | +| hybrid | 40 | 8 | 0.8962 | 0.8098 | 0.7605 | 0.1321 | 0.9057 | 0.0000 | 6661 | 0.0000 | 40.0 | 8.0 | 53 | 14.47 ms | ## Skipped cells @@ -69,10 +69,10 @@ The harness refuses the same combination at the flag level, so a typo fails loud These are the columns a threshold decision should move, and they stay out of every other column. -Best nDCG@10 in this grid: `hybrid` candidateK=10, -contextK=3 (0.8574). -Best context precision: `dense` candidateK=5, -contextK=3 (0.3246). +Best nDCG@10 in this grid: `hybrid` candidateK=5, +contextK=3 (0.8101). +Best context precision: `hybrid` candidateK=5, +contextK=3 (0.3208). These are **not** recommendations. Selecting the grid maximum on the same questions is how a benchmark becomes a lookup table; the adoption rule in `baseline-v1.6.md` diff --git a/docs/eval/threshold-v1.6.json b/docs/eval/threshold-v1.6.json index cb995ef..15d8562 100644 --- a/docs/eval/threshold-v1.6.json +++ b/docs/eval/threshold-v1.6.json @@ -7,265 +7,265 @@ { "threshold": 0, "validation": { - "questions": 16, - "answerableQuestions": 13, + "questions": 24, + "answerableQuestions": 19, "noResultCount": 0, "noResultRate": 0, - "meanRetrieved": 19, + "meanRetrieved": 20, "unanswerable": { - "questions": 3, + "questions": 5, "abstentionCount": 0, "retrievalAbstentionRate": 0, - "meanCandidatesRetrieved": 19, + "meanCandidatesRetrieved": 20, "meanContextPassages": 3 }, "metrics": { - "recallAt1": 0.576923, - "recallAt5": 0.923077, - "recallAt10": 1, - "mrr": 0.778846, - "ndcgAt10": 0.827608, - "hitRateAt5": 0.923077, - "mapAt10": 0.766026, - "contextPrecision": 0.333333, - "contextRecall": 0.923077 + "recallAt1": 0.513158, + "recallAt5": 0.868421, + "recallAt10": 0.921053, + "mrr": 0.720175, + "ndcgAt10": 0.755611, + "hitRateAt5": 0.894737, + "mapAt10": 0.69386, + "contextPrecision": 0.315789, + "contextRecall": 0.802632 } }, "test": { - "questions": 28, - "answerableQuestions": 25, + "questions": 39, + "answerableQuestions": 34, "noResultCount": 0, "noResultRate": 0, - "meanRetrieved": 19, + "meanRetrieved": 20, "unanswerable": { - "questions": 3, + "questions": 5, "abstentionCount": 0, "retrievalAbstentionRate": 0, - "meanCandidatesRetrieved": 19, + "meanCandidatesRetrieved": 20, "meanContextPassages": 3 }, "metrics": { - "recallAt1": 0.7, - "recallAt5": 0.92, - "recallAt10": 1, - "mrr": 0.812381, - "ndcgAt10": 0.858056, - "hitRateAt5": 0.92, - "mapAt10": 0.812381, - "contextPrecision": 0.32, - "contextRecall": 0.92 + "recallAt1": 0.639706, + "recallAt5": 0.882353, + "recallAt10": 0.919118, + "mrr": 0.774079, + "ndcgAt10": 0.794329, + "hitRateAt5": 0.911765, + "mapAt10": 0.747141, + "contextPrecision": 0.303922, + "contextRecall": 0.823529 } } }, { "threshold": 0.3, "validation": { - "questions": 16, - "answerableQuestions": 13, + "questions": 24, + "answerableQuestions": 19, "noResultCount": 0, "noResultRate": 0, - "meanRetrieved": 19, + "meanRetrieved": 20, "unanswerable": { - "questions": 3, + "questions": 5, "abstentionCount": 0, "retrievalAbstentionRate": 0, - "meanCandidatesRetrieved": 19, + "meanCandidatesRetrieved": 20, "meanContextPassages": 3 }, "metrics": { - "recallAt1": 0.576923, - "recallAt5": 0.923077, - "recallAt10": 1, - "mrr": 0.778846, - "ndcgAt10": 0.827608, - "hitRateAt5": 0.923077, - "mapAt10": 0.766026, - "contextPrecision": 0.333333, - "contextRecall": 0.923077 + "recallAt1": 0.513158, + "recallAt5": 0.868421, + "recallAt10": 0.921053, + "mrr": 0.720175, + "ndcgAt10": 0.755611, + "hitRateAt5": 0.894737, + "mapAt10": 0.69386, + "contextPrecision": 0.315789, + "contextRecall": 0.802632 } }, "test": { - "questions": 28, - "answerableQuestions": 25, + "questions": 39, + "answerableQuestions": 34, "noResultCount": 0, "noResultRate": 0, - "meanRetrieved": 19, + "meanRetrieved": 20, "unanswerable": { - "questions": 3, + "questions": 5, "abstentionCount": 0, "retrievalAbstentionRate": 0, - "meanCandidatesRetrieved": 19, + "meanCandidatesRetrieved": 20, "meanContextPassages": 3 }, "metrics": { - "recallAt1": 0.7, - "recallAt5": 0.92, - "recallAt10": 1, - "mrr": 0.812381, - "ndcgAt10": 0.858056, - "hitRateAt5": 0.92, - "mapAt10": 0.812381, - "contextPrecision": 0.32, - "contextRecall": 0.92 + "recallAt1": 0.639706, + "recallAt5": 0.882353, + "recallAt10": 0.919118, + "mrr": 0.774079, + "ndcgAt10": 0.794329, + "hitRateAt5": 0.911765, + "mapAt10": 0.747141, + "contextPrecision": 0.303922, + "contextRecall": 0.823529 } } }, { "threshold": 0.4, "validation": { - "questions": 16, - "answerableQuestions": 13, + "questions": 24, + "answerableQuestions": 19, "noResultCount": 0, "noResultRate": 0, - "meanRetrieved": 19, + "meanRetrieved": 20, "unanswerable": { - "questions": 3, + "questions": 5, "abstentionCount": 0, "retrievalAbstentionRate": 0, - "meanCandidatesRetrieved": 19, + "meanCandidatesRetrieved": 20, "meanContextPassages": 3 }, "metrics": { - "recallAt1": 0.576923, - "recallAt5": 0.923077, - "recallAt10": 1, - "mrr": 0.778846, - "ndcgAt10": 0.827608, - "hitRateAt5": 0.923077, - "mapAt10": 0.766026, - "contextPrecision": 0.333333, - "contextRecall": 0.923077 + "recallAt1": 0.513158, + "recallAt5": 0.868421, + "recallAt10": 0.921053, + "mrr": 0.720175, + "ndcgAt10": 0.755611, + "hitRateAt5": 0.894737, + "mapAt10": 0.69386, + "contextPrecision": 0.315789, + "contextRecall": 0.802632 } }, "test": { - "questions": 28, - "answerableQuestions": 25, + "questions": 39, + "answerableQuestions": 34, "noResultCount": 0, "noResultRate": 0, - "meanRetrieved": 19, + "meanRetrieved": 20, "unanswerable": { - "questions": 3, + "questions": 5, "abstentionCount": 0, "retrievalAbstentionRate": 0, - "meanCandidatesRetrieved": 19, + "meanCandidatesRetrieved": 20, "meanContextPassages": 3 }, "metrics": { - "recallAt1": 0.7, - "recallAt5": 0.92, - "recallAt10": 1, - "mrr": 0.812381, - "ndcgAt10": 0.858056, - "hitRateAt5": 0.92, - "mapAt10": 0.812381, - "contextPrecision": 0.32, - "contextRecall": 0.92 + "recallAt1": 0.639706, + "recallAt5": 0.882353, + "recallAt10": 0.919118, + "mrr": 0.774079, + "ndcgAt10": 0.794329, + "hitRateAt5": 0.911765, + "mapAt10": 0.747141, + "contextPrecision": 0.303922, + "contextRecall": 0.823529 } } }, { "threshold": 0.5, "validation": { - "questions": 16, - "answerableQuestions": 13, + "questions": 24, + "answerableQuestions": 19, "noResultCount": 0, "noResultRate": 0, - "meanRetrieved": 19, + "meanRetrieved": 20, "unanswerable": { - "questions": 3, + "questions": 5, "abstentionCount": 0, "retrievalAbstentionRate": 0, - "meanCandidatesRetrieved": 19, + "meanCandidatesRetrieved": 20, "meanContextPassages": 3 }, "metrics": { - "recallAt1": 0.576923, - "recallAt5": 0.923077, - "recallAt10": 1, - "mrr": 0.778846, - "ndcgAt10": 0.827608, - "hitRateAt5": 0.923077, - "mapAt10": 0.766026, - "contextPrecision": 0.333333, - "contextRecall": 0.923077 + "recallAt1": 0.513158, + "recallAt5": 0.868421, + "recallAt10": 0.921053, + "mrr": 0.720175, + "ndcgAt10": 0.755611, + "hitRateAt5": 0.894737, + "mapAt10": 0.69386, + "contextPrecision": 0.315789, + "contextRecall": 0.802632 } }, "test": { - "questions": 28, - "answerableQuestions": 25, + "questions": 39, + "answerableQuestions": 34, "noResultCount": 0, "noResultRate": 0, - "meanRetrieved": 19, + "meanRetrieved": 20, "unanswerable": { - "questions": 3, + "questions": 5, "abstentionCount": 0, "retrievalAbstentionRate": 0, - "meanCandidatesRetrieved": 19, + "meanCandidatesRetrieved": 20, "meanContextPassages": 3 }, "metrics": { - "recallAt1": 0.7, - "recallAt5": 0.92, - "recallAt10": 1, - "mrr": 0.812381, - "ndcgAt10": 0.858056, - "hitRateAt5": 0.92, - "mapAt10": 0.812381, - "contextPrecision": 0.32, - "contextRecall": 0.92 + "recallAt1": 0.639706, + "recallAt5": 0.882353, + "recallAt10": 0.919118, + "mrr": 0.774079, + "ndcgAt10": 0.794329, + "hitRateAt5": 0.911765, + "mapAt10": 0.747141, + "contextPrecision": 0.303922, + "contextRecall": 0.823529 } } }, { "threshold": 0.6, "validation": { - "questions": 16, - "answerableQuestions": 13, + "questions": 24, + "answerableQuestions": 19, "noResultCount": 0, "noResultRate": 0, - "meanRetrieved": 19, + "meanRetrieved": 20, "unanswerable": { - "questions": 3, + "questions": 5, "abstentionCount": 0, "retrievalAbstentionRate": 0, - "meanCandidatesRetrieved": 19, + "meanCandidatesRetrieved": 20, "meanContextPassages": 3 }, "metrics": { - "recallAt1": 0.576923, - "recallAt5": 0.923077, - "recallAt10": 1, - "mrr": 0.778846, - "ndcgAt10": 0.827608, - "hitRateAt5": 0.923077, - "mapAt10": 0.766026, - "contextPrecision": 0.333333, - "contextRecall": 0.923077 + "recallAt1": 0.513158, + "recallAt5": 0.868421, + "recallAt10": 0.921053, + "mrr": 0.720175, + "ndcgAt10": 0.755611, + "hitRateAt5": 0.894737, + "mapAt10": 0.69386, + "contextPrecision": 0.315789, + "contextRecall": 0.802632 } }, "test": { - "questions": 28, - "answerableQuestions": 25, + "questions": 39, + "answerableQuestions": 34, "noResultCount": 0, "noResultRate": 0, - "meanRetrieved": 19, + "meanRetrieved": 20, "unanswerable": { - "questions": 3, + "questions": 5, "abstentionCount": 0, "retrievalAbstentionRate": 0, - "meanCandidatesRetrieved": 19, + "meanCandidatesRetrieved": 20, "meanContextPassages": 3 }, "metrics": { - "recallAt1": 0.7, - "recallAt5": 0.92, - "recallAt10": 1, - "mrr": 0.812381, - "ndcgAt10": 0.858056, - "hitRateAt5": 0.92, - "mapAt10": 0.812381, - "contextPrecision": 0.32, - "contextRecall": 0.92 + "recallAt1": 0.639706, + "recallAt5": 0.882353, + "recallAt10": 0.919118, + "mrr": 0.774079, + "ndcgAt10": 0.794329, + "hitRateAt5": 0.911765, + "mapAt10": 0.747141, + "contextPrecision": 0.303922, + "contextRecall": 0.823529 } } } diff --git a/docs/eval/threshold-v1.6.md b/docs/eval/threshold-v1.6.md index 0b08be6..3997609 100644 --- a/docs/eval/threshold-v1.6.md +++ b/docs/eval/threshold-v1.6.md @@ -16,11 +16,11 @@ passed the threshold (up to `candidateK`), **ctx** is how many reach the context | Threshold | n (val) | Recall@5 (val) | nDCG@10 (val) | No-result (val) | Unans. abstained (val) | Unans. cands (val) | Unans. ctx (val) | nDCG@10 (test) | Unans. abstained (test) | | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | -| 0 | 13 | 0.9231 | 0.8276 | 0.0000 | 0.0000 | 19.0 | 3.0 | 0.8581 | 0.0000 | -| 0.3 | 13 | 0.9231 | 0.8276 | 0.0000 | 0.0000 | 19.0 | 3.0 | 0.8581 | 0.0000 | -| 0.4 | 13 | 0.9231 | 0.8276 | 0.0000 | 0.0000 | 19.0 | 3.0 | 0.8581 | 0.0000 | -| 0.5 | 13 | 0.9231 | 0.8276 | 0.0000 | 0.0000 | 19.0 | 3.0 | 0.8581 | 0.0000 | -| 0.6 | 13 | 0.9231 | 0.8276 | 0.0000 | 0.0000 | 19.0 | 3.0 | 0.8581 | 0.0000 | +| 0 | 19 | 0.8684 | 0.7556 | 0.0000 | 0.0000 | 20.0 | 3.0 | 0.7943 | 0.0000 | +| 0.3 | 19 | 0.8684 | 0.7556 | 0.0000 | 0.0000 | 20.0 | 3.0 | 0.7943 | 0.0000 | +| 0.4 | 19 | 0.8684 | 0.7556 | 0.0000 | 0.0000 | 20.0 | 3.0 | 0.7943 | 0.0000 | +| 0.5 | 19 | 0.8684 | 0.7556 | 0.0000 | 0.0000 | 20.0 | 3.0 | 0.7943 | 0.0000 | +| 0.6 | 19 | 0.8684 | 0.7556 | 0.0000 | 0.0000 | 20.0 | 3.0 | 0.7943 | 0.0000 | ## Selection rule @@ -34,13 +34,13 @@ retrieval loss. ## Outcome -The sweep is **flat**: every threshold from 0 to 0.6 produces the same validation nDCG@10 (0.8276), the same Recall@5 (0.9231) and the same retrieval abstention rate on unanswerable questions (0/3). No candidate is ever filtered out, so the threshold is **non-binding** on this corpus. +The sweep is **flat**: every threshold from 0 to 0.6 produces the same validation nDCG@10 (0.7556), the same Recall@5 (0.8684) and the same retrieval abstention rate on unanswerable questions (0/5). No candidate is ever filtered out, so the threshold is **non-binding** on this corpus. **No evidence to change `threshold = 0.5`.** All thresholds hold the line equally; picking one would be arbitrary. The current value can be neither validated nor falsified here, which is a property of the corpus rather than of the threshold. Note also what this does *not* establish: abstention is a retrieval-layer statement — whether the model then declines to answer needs a generator eval. ## Caveat on this corpus -The split removes the most obvious form of overfitting, but 13 +The split removes the most obvious form of overfitting, but 19 answerable questions on the validation side is a thin basis for a decision, and the corpus is still small. A threshold is a product decision with a **refusal-rate** cost attached, so a recommendation here is only as good as the corpus behind it. Re-run this after the diff --git a/eval/corpus/coastal-monitoring.md b/eval/corpus/coastal-monitoring.md new file mode 100644 index 0000000..837713d --- /dev/null +++ b/eval/corpus/coastal-monitoring.md @@ -0,0 +1,70 @@ +# Coastal Monitoring Programme + +## Overview + +Coastal monitoring tracks a tidally driven water body at the land margin. Its distinguishing +problem is that the water body moves twice a day: a station is never at the same depth for +two consecutive visits, and a tidal phase that happens to coincide with a sampling run can +dominate the reading. + +The programme covers four shore stations and two offshore buoys. Shore stations are visited; +buoys are instrumented and telemeter. The two kinds of station are reported separately, +because mixing a telemetered series with a visited series is the most common defect in a +coastal dataset. + +## Sampling Interval + +Routine sampling runs **hourly** at the telemetered buoys and **fortnightly** at the shore +stations. The hourly cadence exists to resolve the tidal cycle, which a fortnightly cadence +would alias into a meaningless slow oscillation; a shore station cannot support it because +each visit is a boat trip. + +Every telemetered reading is stamped with its tidal phase. A reading without a phase stamp +is retained but cannot be compared against the shore series, and the quality-control pass +excludes it from any cross-station comparison. + +## Replicate Samples + +Field teams collect **three replicate samples** at each shore station. Three is the standard +for the programme because the boat trip, not the analysis, dominates the cost, so an extra +replicate is nearly free once the team is on site — but only three are taken, because the +shore stations are well mixed and the fourth replicate has never changed a decision. + +Buoys do not take replicates in the sampling sense; they take a burst of 30 readings over +60 seconds and report the median. The median is used rather than the mean because a single +wave splash is a large positive outlier in exactly the quantities a buoy measures. + +## Sensor Depth + +The shore stations carry a sensor at **1 metre below the surface**, and the offshore buoys +carry a sensor at **2 metres below the surface**. The buoy depth is greater because a buoy +in the wave zone is repeatedly lifted and dropped by swell, and a sensor closer to the +surface is out of the water a meaningful fraction of the time. + +Both depths are recorded as depth below the *instantaneous* surface. Coastal sensors are the +only programme where the surface reference changes fast enough to matter within a single +reading, so the timestamp and the depth are recorded together and neither is meaningful +alone. + +## Parameters + +The core parameters are water temperature, salinity, turbidity and wave height. Coastal +adds wave height, which no other programme records, and salinity, which only the estuary +programme also records but for a different reason. + +Wave height is recorded as significant wave height, the mean of the highest third of waves +in the burst, not as the maximum. The maximum is recorded separately and is not used for +trend analysis because it is dominated by rare events. + +## Quality Control + +Telemetered readings are passed through a spike filter that rejects a value more than four +standard deviations from its 24-hour rolling mean. The filter has an override: a reading +that is also accompanied by a wave height above the 99th percentile is retained rather than +rejected, on the grounds that a storm is exactly when an unusual value is most likely to be +real. + +Shore stations use a field blank and a blind duplicate, as the river and reservoir +programmes do. A station whose duplicate differs by more than 25 percent is re-visited, +which is a looser threshold than the reservoir programme's 20 percent because a coastal +station is inherently noisier and a tighter threshold flagged almost every visit. diff --git a/eval/corpus/estuary-monitoring.md b/eval/corpus/estuary-monitoring.md new file mode 100644 index 0000000..696dd00 --- /dev/null +++ b/eval/corpus/estuary-monitoring.md @@ -0,0 +1,67 @@ +# Estuary Monitoring Programme + +## Overview + +Estuary monitoring tracks the mixing zone where a river meets the sea. Its distinguishing +problem is that the water body has a gradient in all three dimensions at once: salinity and +turbidity change sharply over a few kilometres, and the position of that gradient moves with +the tide and with river flow. + +The programme covers two estuaries, each with a transect of five stations running from the +freshwater end to the mouth. A transect is the unit of reporting, not a station, because a +single station in an estuary describes a position in a gradient that has moved by the time +the next station is sampled. + +## Sampling Interval + +Routine sampling runs **daily** at the two lowest stations and **fortnightly** along the +rest of the transect. Daily sampling at the lower stations is used to track the salt wedge, +whose position responds to the tide within hours; the upper transect responds to river flow +over days and does not need it. + +Transect sampling is run on the ebb tide, and every station is occupied within a single ebb +to keep the transect a snapshot rather than a sequence. A transect that overruns its ebb is +discarded and repeated, because a transect sampled across a tidal reversal is not a gradient. + +## Replicate Samples + +Field teams collect **five replicate samples** at each transect station, the highest of any +programme in the network. Five because the estuary gradient means two samples taken a metre +apart can differ more than two samples taken a kilometre apart at a well-mixed site; the +within-station variance is high enough that three replicates do not estimate a mean reliably. + +Replicates are taken as a spatial cross rather than as a sequence: one at the nominal +position and four at 25 metres on each axis. A sequential set of replicates would all sample +the same parcel of water and would understate the variance that matters. + +## Sensor Depth + +Transect stations carry sensors at **2 metres below the surface** and at 1 metre above the +bed, paired so that the vertical salinity difference can be computed directly. The surface +depth is fixed at 2 metres rather than at 1 metre to keep the sensor below the freshwater +lens that floats on the saline layer at the freshwater end. + +The paired depths are the programme's defining feature. A single sensor in an estuary cannot +distinguish a change in salinity from a change in where the halocline sits, and those are +different findings. + +## Parameters + +The core parameters are salinity, turbidity, dissolved oxygen and temperature. Estuary adds +the position of the turbidity maximum, which is the programme's headline product and is not +recorded by any other programme. + +The turbidity maximum is reported as a distance from the freshwater end, not as a turbidity +value. Its position moves several kilometres over a tidal cycle, and a turbidity value +without a position cannot distinguish a stationary maximum from a passing one. + +## Quality Control + +Every transect includes one blind duplicate at a randomly chosen station. Precision is +estimated per transect rather than per station, because the quantity the programme reports +is a gradient and the error that matters is the error in the gradient. + +A transect whose blind duplicate differs by more than 30 percent is repeated. The threshold +is looser than any other programme's, and deliberately so: an estuary's true variance is +genuinely larger, and a tighter threshold would cause every transect to be repeated, which +in practice means none of them are. diff --git a/eval/corpus/groundwater-monitoring.md b/eval/corpus/groundwater-monitoring.md new file mode 100644 index 0000000..ca89489 --- /dev/null +++ b/eval/corpus/groundwater-monitoring.md @@ -0,0 +1,68 @@ +# Groundwater Monitoring Programme + +## Overview + +Groundwater monitoring tracks water below the surface rather than water in a channel. Its +distinguishing problem is that the water body cannot be seen, so every measurement is a +sample from a borehole that may or may not be connected to the aquifer the programme intends +to describe. + +The programme covers nine boreholes across three catchments. Eight are monitoring boreholes +used only for measurement; one is a supply borehole that is also pumped, and its readings +are reported separately because a pumped borehole's level reflects recent abstraction as +much as the aquifer. + +## Sampling Interval + +Routine sampling runs **monthly**, the least frequent cadence in the network. Groundwater +responds to rainfall over weeks to months, and a more frequent cadence measures the borehole +rather than the aquifer: a fortnightly series is dominated by the borehole's own equilibration +after each visit, which takes several days. + +Supply boreholes are additionally sampled immediately before and after each pumping cycle. +Those samples are paired and reported as a drawdown and recovery pair, not as routine +readings. + +## Replicate Samples + +Field teams collect **two replicate samples** at each borehole, the lowest of any programme +in the network. Two rather than three because groundwater is well mixed by the time it +reaches a borehole, and the dominant error is not within-sample variance but the borehole's +connection to the aquifer — an error that an extra replicate does not reduce. + +Because two replicates give no way to identify an outlier, any borehole whose two replicates +disagree by more than 10 percent is re-sampled entirely rather than resolved statistically. +Ten percent is tighter than any other programme's threshold precisely because there is no +third replicate to arbitrate. + +## Sensor Depth + +Boreholes carry a pressure transducer at **15 metres below the water table**, measured on the +first visit and recorded as an absolute elevation so that a falling water table does not +silently change the depth being sampled. The depth is the largest in the network by an order +of magnitude, which is a property of boreholes rather than a choice. + +The transducer records water level continuously. Water level is the programme's primary +quantity; chemistry is sampled monthly and level is logged every 15 minutes, and the two are +reported on separate axes because a single chart of both conceals the pattern in either. + +## Parameters + +The core parameters are water level, temperature, specific conductance and nitrate. Groundwater +adds water level, which is the only parameter in the network that is logged continuously +rather than sampled. + +Nitrate is recorded as the primary indicator of agricultural loading, and it is the parameter +the programme was established to track. Specific conductance is recorded as a cheap proxy for +salinity intrusion in the two coastal boreholes. + +## Quality Control + +Every visit records the water level before and after purging. A borehole in which the level +does not recover within 24 hours of purging is flagged, because the purge has drawn from a +body of water that is not being recharged and the sample may not represent the aquifer. + +Blind duplicates are submitted quarterly rather than per visit, which is the sparsest +verification in the network. The programme's position is that its dominant uncertainty is the +borehole-to-aquifer connection, which a duplicate cannot measure, so spending budget on more +duplicates would buy precision the programme cannot use. diff --git a/eval/corpus/reservoir-monitoring.md b/eval/corpus/reservoir-monitoring.md new file mode 100644 index 0000000..389eb79 --- /dev/null +++ b/eval/corpus/reservoir-monitoring.md @@ -0,0 +1,69 @@ +# Reservoir Monitoring Programme + +## Overview + +Reservoir monitoring tracks a standing water body whose level is managed rather than +natural. The programme's distinguishing problem is that the water body has an operator: +every measurement has to be paired with the release schedule, or a change in a reading is +indistinguishable from a change in how the reservoir was run that week. + +The programme covers three reservoirs in the upland basin. Each has a fixed monitoring +station at the dam face and two floating stations whose position is recorded on every +visit, because a reservoir's surface area changes enough over a season to move a floating +station hundreds of metres without anyone touching it. + +## Sampling Interval + +Routine sampling runs **fortnightly**, on a fixed Tuesday, so that the interval is +consistent across sites and operators. A fortnightly cadence is a deliberate compromise: +weekly was found to double cost without changing any trend, and monthly aliased against +the operator's own drawdown cycle, which is also roughly monthly and made the series very +hard to interpret. + +Event sampling is triggered by any release exceeding 20 percent of live capacity in a +single day. Event samples are additional to the routine cadence and are labelled with the +release event rather than with the calendar. + +## Replicate Samples + +Field teams collect **four replicate samples** at each station to control for local +variability. Four rather than three because the reservoir stations sit in a drawdown zone +where wind-driven mixing produces an occasional outlier; with three replicates a single +outlier is a third of the mean, and with four it can be identified and excluded on a stated +rule rather than discarded by feel. + +Replicates are taken within a 15-minute window. A replicate that falls outside that window +is recorded but excluded from the mean, because the reservoir can stratify and destratify on +that timescale in summer. + +## Sensor Depth + +The fixed station carries a sensor string at **5 metres below the surface**, and the two +floating stations carry a single sensor at **1 metre below the surface**. The asymmetry is +intentional: the dam face is deep and well mixed, while the floating stations are in the +drawdown zone where the interesting gradient is in the top metre. + +Depth is recorded as depth below the *current* surface, not below full capacity. Because the +surface moves, a sensor on a fixed string is at a different absolute elevation at different +times, and the programme records both so that a reader can reconstruct which was meant. + +## Parameters + +The core parameters are water temperature, dissolved oxygen, turbidity and chlorophyll-a. +Reservoir-specific parameters are residence time and drawdown rate, neither of which the +river or lake programmes record because neither has an operator-controlled outlet. + +Turbidity is recorded as the primary indicator of sediment resuspension during drawdown, +which is the process the programme exists to quantify. Chlorophyll-a is recorded as the +primary indicator of the algal response to nutrient loading. + +## Quality Control + +Every routine visit includes one field blank and one duplicate submitted blind. The +duplicate is used to estimate within-station precision; the blank is used to detect +contamination introduced by the sampling kit rather than by the reservoir. + +A station whose blind duplicate differs by more than 20 percent is flagged and re-visited +within seven days. Two consecutive flags retire the station's sensor string, because the +most common cause of a persistent discrepancy is a drifting sensor rather than a genuinely +patchy water body. diff --git a/eval/questions.jsonl b/eval/questions.jsonl index 1f4bb57..10b031a 100644 --- a/eval/questions.jsonl +++ b/eval/questions.jsonl @@ -52,3 +52,12 @@ {"id":"q052","type":"multi-hop","question":"Compare the retry count and context window of Gateway API v2 and v3.","relevant":[{"document":"api-gateway-v2.md","page":null,"block":9,"quote":"recommended retry count is **5**"},{"document":"api-gateway-v2.md","page":null,"block":15,"quote":"maximum context window is **32768 tokens**"},{"document":"api-gateway-v3.md","page":null,"block":9,"quote":"recommended retry count is **3**"},{"document":"api-gateway-v3.md","page":null,"block":14,"quote":"maximum context window is **65536 tokens**"}]} {"id":"q053","type":"unanswerable","answerable":false,"question":"How much GPU memory does Gateway API v2 require to serve a request?","relevant":[]} {"id":"q054","type":"unanswerable","answerable":false,"question":"What is the monthly subscription price of Gateway API v3?","relevant":[]} +{"id":"q055","type":"exact","question":"How many replicate samples are collected at each reservoir monitoring station?","relevant":[{"document":"reservoir-monitoring.md","page":null,"block":8,"quote":"four replicate samples"}]} +{"id":"q056","type":"hard-negative","question":"How many replicate samples are collected at each estuary transect station?","relevant":[{"document":"estuary-monitoring.md","page":null,"block":8,"quote":"five replicate samples"}]} +{"id":"q057","type":"hard-negative","question":"At what depth below the surface does the coastal programme place its shore-station sensor?","relevant":[{"document":"coastal-monitoring.md","page":null,"block":11,"quote":"1 metre below the surface"}]} +{"id":"q058","type":"hard-negative","question":"At what depth below the water table do groundwater boreholes carry their pressure transducer?","relevant":[{"document":"groundwater-monitoring.md","page":null,"block":11,"quote":"15 metres below the water table"}]} +{"id":"q059","type":"semantic","question":"Which monitoring programme samples most frequently?","relevant":[{"document":"coastal-monitoring.md","page":null,"block":5,"quote":"Routine sampling runs **hourly**"}]} +{"id":"q060","type":"multi-hop","question":"Compare the sampling interval and replicate count of the reservoir and groundwater programmes.","relevant":[{"document":"reservoir-monitoring.md","page":null,"block":5,"quote":"fortnightly"},{"document":"reservoir-monitoring.md","page":null,"block":8,"quote":"four replicate samples"},{"document":"groundwater-monitoring.md","page":null,"block":5,"quote":"monthly"},{"document":"groundwater-monitoring.md","page":null,"block":8,"quote":"two replicate samples"}]} +{"id":"q061","type":"semantic","question":"Why does the estuary programme take more replicate samples than the other monitoring programmes?","relevant":[{"document":"estuary-monitoring.md","page":null,"block":8,"quote":"within-station variance"}]} +{"id":"q062","type":"unanswerable","answerable":false,"question":"What is the annual operating cost of the coastal monitoring buoys?","relevant":[]} +{"id":"q063","type":"unanswerable","answerable":false,"question":"How many litres per second does the reservoir release downstream on a typical day?","relevant":[]} diff --git a/eval/splits.json b/eval/splits.json index 5aa96d3..021e265 100644 --- a/eval/splits.json +++ b/eval/splits.json @@ -52,5 +52,14 @@ "q051": "test", "q052": "test", "q053": "validation", - "q054": "test" + "q054": "test", + "q055": "validation", + "q056": "test", + "q057": "validation", + "q058": "test", + "q059": "test", + "q060": "validation", + "q061": "test", + "q062": "validation", + "q063": "test" } From 08fa7c112dc024ce708b68a9c169b4be121c6996 Mon Sep 17 00:00:00 2001 From: MrSibe Date: Wed, 30 Sep 2026 18:10:23 +0800 Subject: [PATCH 15/18] feat(eval): dense score diagnostics, and the answer is not a threshold MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Adds `npm run eval:scores`, which measures whether a single dense similarity threshold is capable of separating "this passage answers the question" from "this one does not" — before any threshold grid is chosen. Child 2 of #192, and it changes what the next step should be. ## First, the score is not a cosine `SQLiteVectorStore` computes `score = 1 - distance / 2`, and sqlite-vec's cosine distance is `1 - cosine`, so: ``` score = (1 + cosine) / 2 => threshold 0.5 == cosine 0.0 ``` The shipped `threshold = 0.5` is therefore **cosine ≥ 0**, not "cosine ≥ 0.5". Every guard that mattered here — the sweep grid, the threshold report, the source comment — was describing an affine map, not the cosine. Fixed in the store's comment, in `eval/README.md`, and reported as two columns throughout the new diagnostics. ## The measurement Dense only (hybrid's `score` is an RRF value, `1 / (60 + rank)`, and not on the same scale), `validation` split only (the distribution is used to choose where to sweep, so it must not see `test`), `threshold = 0` and `candidateK = 500` (above the 53-chunk index, so every chunk is scored for every query). ``` best relevant p10 0.8906 (cosine 0.781) p50 0.9359 (0.872) best non-relevant p50 0.9243 (cosine 0.849) p90 0.9386 (0.877) margin p10 -0.0272 p50 +0.0026 unanswerable max p50 0.9259 (cosine 0.852) p90 0.9424 (0.885) ``` Three things fall out of that: 1. **The best non-relevant passage outranks the best relevant one about half the time.** The margin's p50 is +0.0026 and its p10 is −0.0272. Dense similarity is barely informative about relevance on this corpus — the hard-negative clusters are doing exactly what they were built for. 2. **The unanswerable max sits above the worst required relevant passage** (p90 0.9424 vs p10 0.8906), so the two distributions are not separable. 3. **`cross-lingual` is the lowest of every type** (best relevant p10 0.8808) against `semantic` at 0.9208 — a threshold tuned on the aggregate would cut cross-lingual first. ## The threshold curve is the answer | threshold | raw cosine | answerable hit | answerable full recall | unanswerable abstain | | --- | --- | --- | --- | --- | | 0.875 | 0.75 | 1.0000 | 1.0000 | 0.0000 | | 0.900 | 0.80 | 0.8947 | 0.8947 | 0.2000 | | 0.925 | 0.85 | 0.6316 | 0.6316 | 0.4000 | | 0.950 | 0.900 | 0.1579 | 0.1579 | 1.0000 | **Abstention never rises without full recall falling.** Every threshold that refuses an unanswerable question refuses required relevant passages at the same rate. There is no operating point that buys the first without paying the second. So the conclusion is not "sweep 0.7–0.8 instead of 0–0.6". It is: **a single dense similarity threshold cannot carry both recall and abstention on this corpus**, and the next mechanism to evaluate is a different signal — reranker score, top1−top2 margin, per-query thresholds, or claim-level answerability. That is a much more useful finding than a best threshold would have been, and it is why this PR was worth doing before more clusters. ## What this PR does not do - It does not choose a threshold. That is now blocked on a signal that can separate the distributions. - The `eval:threshold` sweep grid is left as it is. Re-gridding it would move a knob whose curve is flat where it matters and precipitous where it does not — the diagnostic is the replacement for re-gridding, not a preamble to it. ## Testing - `npm run typecheck` — clean - `npm test` — 504 pass - `npm run eval:scores` — writes `docs/eval/scores-v1.6.{json,md}` - The committed baseline is **unchanged**: `--eval-scores` is off by default, so the CI-diffed JSON does not grow a score series it has no use for Part of #192 (child 2: corpus diagnostics). --- .prettierignore | 4 + docs/eval/scores-v1.6.json | 342 +++++++++++++++++++ docs/eval/scores-v1.6.md | 119 +++++++ eval/README.md | 13 + package.json | 3 +- scripts/eval-scores.mjs | 392 ++++++++++++++++++++++ src/main/eval/harness.ts | 24 +- src/main/eval/run.ts | 3 + src/main/eval/types.ts | 9 + src/main/vectorstore/SQLiteVectorStore.ts | 6 +- 10 files changed, 910 insertions(+), 5 deletions(-) create mode 100644 docs/eval/scores-v1.6.json create mode 100644 docs/eval/scores-v1.6.md create mode 100644 scripts/eval-scores.mjs diff --git a/.prettierignore b/.prettierignore index f6f4ae5..da32ca2 100644 --- a/.prettierignore +++ b/.prettierignore @@ -29,6 +29,10 @@ docs/eval/threshold-*.md docs/eval/sweep-*.json docs/eval/sweep-*.md +# And by `scripts/eval-scores.mjs`. Regenerate with `npm run eval:scores`. +docs/eval/scores-*.json +docs/eval/scores-*.md + # And the same again for `scripts/eval-retrieval.mjs` (#77). docs/eval/retrieval-*.json docs/eval/retrieval-*.md diff --git a/docs/eval/scores-v1.6.json b/docs/eval/scores-v1.6.json new file mode 100644 index 0000000..163c779 --- /dev/null +++ b/docs/eval/scores-v1.6.json @@ -0,0 +1,342 @@ +{ + "baseline": "v1.6", + "split": "validation", + "candidateK": 500, + "threshold": 0, + "indexSize": 53, + "counts": { + "answerable": 19, + "unanswerable": 5 + }, + "distributions": { + "bestRelevant": { + "p0": 0.880777, + "p10": 0.89056, + "p25": 0.912001, + "p50": 0.935891, + "p75": 0.942776, + "p90": 0.953038, + "p100": 0.953664 + }, + "worstRelevant": { + "p0": 0.880777, + "p10": 0.89056, + "p25": 0.908589, + "p50": 0.935467, + "p75": 0.942776, + "p90": 0.953038, + "p100": 0.953664 + }, + "bestNonRelevant": { + "p0": 0.890443, + "p10": 0.908132, + "p25": 0.917792, + "p50": 0.924346, + "p75": 0.933555, + "p90": 0.9386, + "p100": 0.945918 + }, + "margin": { + "p0": -0.02942099999999992, + "p10": -0.027232000000000034, + "p25": -0.00922400000000001, + "p50": 0.0026120000000000587, + "p75": 0.00807100000000005, + "p90": 0.032869999999999955, + "p100": 0.06226200000000004 + }, + "unanswerableMax": { + "p0": 0.897848, + "p10": 0.897848, + "p25": 0.912751, + "p50": 0.925936, + "p75": 0.934276, + "p90": 0.942441, + "p100": 0.942441 + } + }, + "byType": { + "exact": { + "questions": 4, + "bestRelevant": { + "p0": 0.912001, + "p10": 0.912001, + "p25": 0.912001, + "p50": 0.929376, + "p75": 0.936747, + "p90": 0.953038, + "p100": 0.953038 + }, + "worstRelevant": { + "p0": 0.912001, + "p10": 0.912001, + "p25": 0.912001, + "p50": 0.929376, + "p75": 0.936747, + "p90": 0.953038, + "p100": 0.953038 + } + }, + "semantic": { + "questions": 6, + "bestRelevant": { + "p0": 0.920784, + "p10": 0.920784, + "p25": 0.935467, + "p50": 0.939466, + "p75": 0.945637, + "p90": 0.953664, + "p100": 0.953664 + }, + "worstRelevant": { + "p0": 0.920784, + "p10": 0.920784, + "p25": 0.935467, + "p50": 0.939466, + "p75": 0.945637, + "p90": 0.953664, + "p100": 0.953664 + } + }, + "multi-hop": { + "questions": 2, + "bestRelevant": { + "p0": 0.924784, + "p10": 0.924784, + "p25": 0.924784, + "p50": 0.924784, + "p75": 0.93756, + "p90": 0.93756, + "p100": 0.93756 + }, + "worstRelevant": { + "p0": 0.903304, + "p10": 0.903304, + "p25": 0.903304, + "p50": 0.903304, + "p75": 0.933129, + "p90": 0.933129, + "p100": 0.933129 + } + }, + "zh": { + "questions": 1, + "bestRelevant": { + "p0": 0.952705, + "p10": 0.952705, + "p25": 0.952705, + "p50": 0.952705, + "p75": 0.952705, + "p90": 0.952705, + "p100": 0.952705 + }, + "worstRelevant": { + "p0": 0.952705, + "p10": 0.952705, + "p25": 0.952705, + "p50": 0.952705, + "p75": 0.952705, + "p90": 0.952705, + "p100": 0.952705 + } + }, + "cross-lingual": { + "questions": 4, + "bestRelevant": { + "p0": 0.880777, + "p10": 0.880777, + "p25": 0.880777, + "p50": 0.89056, + "p75": 0.904196, + "p90": 0.908589, + "p100": 0.908589 + }, + "worstRelevant": { + "p0": 0.880777, + "p10": 0.880777, + "p25": 0.880777, + "p50": 0.89056, + "p75": 0.904196, + "p90": 0.908589, + "p100": 0.908589 + } + }, + "hard-negative": { + "questions": 2, + "bestRelevant": { + "p0": 0.935891, + "p10": 0.935891, + "p25": 0.935891, + "p50": 0.935891, + "p75": 0.942776, + "p90": 0.942776, + "p100": 0.942776 + }, + "worstRelevant": { + "p0": 0.935891, + "p10": 0.935891, + "p25": 0.935891, + "p50": 0.935891, + "p75": 0.942776, + "p90": 0.942776, + "p100": 0.942776 + } + } + }, + "overlap": { + "worstRelevantP10": 0.89056, + "bestNonRelevantP90": 0.9386, + "unanswerableMaxP50": 0.925936, + "unanswerableMaxP90": 0.942441 + }, + "separable": false, + "curve": [ + { + "threshold": 0.5, + "cosine": 0, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.525, + "cosine": 0.05, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.55, + "cosine": 0.1, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.575, + "cosine": 0.15, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.6, + "cosine": 0.2, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.625, + "cosine": 0.25, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.65, + "cosine": 0.3, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.675, + "cosine": 0.35, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.7, + "cosine": 0.4, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.725, + "cosine": 0.45, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.75, + "cosine": 0.5, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.775, + "cosine": 0.55, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.8, + "cosine": 0.6, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.825, + "cosine": 0.65, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.85, + "cosine": 0.7, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.875, + "cosine": 0.75, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.9, + "cosine": 0.8, + "hitRate": 0.8947368421052632, + "fullRecallRate": 0.8947368421052632, + "abstentionRate": 0.2 + }, + { + "threshold": 0.925, + "cosine": 0.85, + "hitRate": 0.631578947368421, + "fullRecallRate": 0.631578947368421, + "abstentionRate": 0.4 + }, + { + "threshold": 0.95, + "cosine": 0.9, + "hitRate": 0.15789473684210525, + "fullRecallRate": 0.15789473684210525, + "abstentionRate": 1 + }, + { + "threshold": 0.975, + "cosine": 0.95, + "hitRate": 0, + "fullRecallRate": 0, + "abstentionRate": 1 + }, + { + "threshold": 1, + "cosine": 1, + "hitRate": 0, + "fullRecallRate": 0, + "abstentionRate": 1 + } + ] +} diff --git a/docs/eval/scores-v1.6.md b/docs/eval/scores-v1.6.md new file mode 100644 index 0000000..2ad38b7 --- /dev/null +++ b/docs/eval/scores-v1.6.md @@ -0,0 +1,119 @@ +# Dense score diagnostics — v1.6 (#192) + +Generated by `node scripts/eval-scores.mjs`. Numbers are harness output; do not edit them by hand. + +## Read this first: the score is not a cosine + +`SQLiteVectorStore` computes + +```text +score = 1 - distance / 2 +distance = 1 - cosine (sqlite-vec, distance_metric=cosine) +=> score = (1 + cosine) / 2 +``` + +so the configured `threshold` is an **affine map of the cosine**, not the cosine: + +| threshold | raw cosine | +| --- | --- | +| 0.3 | -0.4 | +| 0.4 | -0.2 | +| **0.5 (shipped)** | **0.0** | +| 0.6 | 0.2 | +| 0.8 | 0.6 | +| 1.0 | 1.0 | + +The shipped `threshold = 0.5` means **cosine ≥ 0**, which is very permissive. Every +threshold row below carries both columns so the two never get confused again, and every +mention of "the sweep looked too low" has to be read through this table. + +## What was measured + +Dense only, `validation` split only, `threshold = 0`, `candidateK = 500` (above the +index size, so every chunk is scored for every query), `contextK = 3`. + +- answerable questions: 19 +- unanswerable questions: 5 +- index size: 53 chunks + +Hybrid is deliberately excluded: its `score` is an RRF value (`1 / (60 + rank)`) and is not +on the same scale as a normalised cosine. + +## Distributions (score, with raw cosine in brackets) + +`best relevant` is the highest-scoring passage that covers ground truth; `worst relevant` +is the lowest one that still has to survive for the question to be fully answered; +`best non-relevant` is the highest-scoring passage that covers nothing; `margin` is the +first minus the third. + +| Distribution | n | min | p10 | p25 | p50 | p75 | p90 | max | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | +| best relevant | 19 | 0.8808 (0.762) | 0.8906 (0.781) | 0.9120 (0.824) | 0.9359 (0.872) | 0.9428 (0.886) | 0.9530 (0.906) | 0.9537 (0.907) | +| worst relevant | 19 | 0.8808 (0.762) | 0.8906 (0.781) | 0.9086 (0.817) | 0.9355 (0.871) | 0.9428 (0.886) | 0.9530 (0.906) | 0.9537 (0.907) | +| best non-relevant | 19 | 0.8904 (0.781) | 0.9081 (0.816) | 0.9178 (0.836) | 0.9243 (0.849) | 0.9336 (0.867) | 0.9386 (0.877) | 0.9459 (0.892) | +| margin (best rel − best non-rel) | 19 | -0.0294 (-1.059) | -0.0272 (-1.054) | -0.0092 (-1.018) | 0.0026 (-0.995) | 0.0081 (-0.984) | 0.0329 (-0.934) | 0.0623 (-0.875) | +| unanswerable max candidate | 5 | 0.8978 (0.796) | 0.8978 (0.796) | 0.9128 (0.826) | 0.9259 (0.852) | 0.9343 (0.869) | 0.9424 (0.885) | 0.9424 (0.885) | + +## By query type + +`best relevant` p50 and p10, and `worst relevant` p10 — the last is the one that decides +whether a cross-lingual question survives a threshold that a semantic question tolerates. + +| Type | n | best rel p50 | best rel p10 | worst rel p10 | +| --- | --- | --- | --- | --- | +| cross-lingual | 4 | 0.8906 | 0.8808 | 0.8808 | +| exact | 4 | 0.9294 | 0.9120 | 0.9120 | +| hard-negative | 2 | 0.9359 | 0.9359 | 0.9359 | +| multi-hop | 2 | 0.9248 | 0.9248 | 0.9033 | +| semantic | 6 | 0.9395 | 0.9208 | 0.9208 | +| zh | 1 | 0.9527 | 0.9527 | 0.9527 | + +## Threshold curve + +For each candidate threshold: **hit** = share of answerable questions that still have some +relevant passage; **full recall** = share whose every ground-truth block is still covered; +**abstain** = share of unanswerable questions that now return nothing. + +| threshold (score) | raw cosine | answerable hit | answerable full recall | unanswerable abstain | +| --- | --- | --- | --- | --- | +| 0.500 | 0.000 | 1.0000 | 1.0000 | 0.0000 | +| 0.525 | 0.050 | 1.0000 | 1.0000 | 0.0000 | +| 0.550 | 0.100 | 1.0000 | 1.0000 | 0.0000 | +| 0.575 | 0.150 | 1.0000 | 1.0000 | 0.0000 | +| 0.600 | 0.200 | 1.0000 | 1.0000 | 0.0000 | +| 0.625 | 0.250 | 1.0000 | 1.0000 | 0.0000 | +| 0.650 | 0.300 | 1.0000 | 1.0000 | 0.0000 | +| 0.675 | 0.350 | 1.0000 | 1.0000 | 0.0000 | +| 0.700 | 0.400 | 1.0000 | 1.0000 | 0.0000 | +| 0.725 | 0.450 | 1.0000 | 1.0000 | 0.0000 | +| 0.750 | 0.500 | 1.0000 | 1.0000 | 0.0000 | +| 0.775 | 0.550 | 1.0000 | 1.0000 | 0.0000 | +| 0.800 | 0.600 | 1.0000 | 1.0000 | 0.0000 | +| 0.825 | 0.650 | 1.0000 | 1.0000 | 0.0000 | +| 0.850 | 0.700 | 1.0000 | 1.0000 | 0.0000 | +| 0.875 | 0.750 | 1.0000 | 1.0000 | 0.0000 | +| 0.900 | 0.800 | 0.8947 | 0.8947 | 0.2000 | +| 0.925 | 0.850 | 0.6316 | 0.6316 | 0.4000 | +| 0.950 | 0.900 | 0.1579 | 0.1579 | 1.0000 | +| 0.975 | 0.950 | 0.0000 | 0.0000 | 1.0000 | +| 1.000 | 1.000 | 0.0000 | 0.0000 | 1.0000 | + +## The separation question + +The threshold decision turns on whether these overlap: + +| Landmark | score | raw cosine | +| --- | --- | --- | +| worst relevant, p10 | 0.8906 | 0.781 | +| best non-relevant, p90 | 0.9386 | 0.877 | +| unanswerable max, p50 | 0.9259 | 0.852 | +| unanswerable max, p90 | 0.9424 | 0.885 | + +**The distributions **overlap**, so a higher threshold buys abstention by giving up required relevant passages. If the curve above shows abstention rising only as full recall falls, then the honest conclusion is that **a single dense similarity threshold cannot carry both recall and abstention** — and the next mechanism to evaluate is not a finer threshold grid but a different signal (reranker score, top1−top2 margin, per-query thresholds, or claim-level answerability). + +## Reproduce + +```bash +npm run eval:prepare # one-time, networked model bootstrap +npm run eval:scores # offline; rewrites this file +``` diff --git a/eval/README.md b/eval/README.md index db93d5d..299eca7 100644 --- a/eval/README.md +++ b/eval/README.md @@ -12,9 +12,22 @@ npm run eval # offline and deterministic: run the harness, rewrite th npm run eval:retrieval # strategy comparison (#77); validation selects, test reports npm run eval:threshold # derive the similarity threshold on validation, report on test npm run eval:sweep # bounded grid over strategy × candidateK × contextK, one dashboard +npm run eval:scores # dense score distribution: can one threshold separate relevant from not? npm run eval:blocks eval/corpus/foo.md # print the block ordinals ground truth must use ``` +### `threshold` is not a cosine + +The vector store returns \`score = 1 - distance / 2\` and sqlite-vec's cosine distance is +\`1 - cosine\`, so: + +```text +score = (1 + cosine) / 2 score 0.5 == cosine 0.0 +``` + +The shipped \`threshold = 0.5\` therefore means **cosine ≥ 0**, which is very permissive. +\`npm run eval:scores\` reports both columns side by side so the two never get conflated. + ### The harness runs the production configuration From v1.6 the harness defaults to the parameters the app ships, so its numbers describe diff --git a/package.json b/package.json index 77a6406..418c579 100644 --- a/package.json +++ b/package.json @@ -44,7 +44,8 @@ "eval:retrieval": "npm run build && node --experimental-transform-types --disable-warning=ExperimentalWarning --disable-warning=MODULE_TYPELESS_PACKAGE_JSON scripts/eval-retrieval.mjs", "eval:threshold": "npm run build && node scripts/eval-threshold.mjs", "eval:sweep": "npm run build && node scripts/eval-sweep.mjs", - "eval:blocks": "node --experimental-transform-types --import ./test/ts-resolve.mjs --disable-warning=ExperimentalWarning --disable-warning=MODULE_TYPELESS_PACKAGE_JSON scripts/eval-blocks.mjs" + "eval:blocks": "node --experimental-transform-types --import ./test/ts-resolve.mjs --disable-warning=ExperimentalWarning --disable-warning=MODULE_TYPELESS_PACKAGE_JSON scripts/eval-blocks.mjs", + "eval:scores": "npm run build && node scripts/eval-scores.mjs" }, "//test": [ "`node --test` strips TypeScript types rather than compiling them, and strip-only", diff --git a/scripts/eval-scores.mjs b/scripts/eval-scores.mjs new file mode 100644 index 0000000..1356f2a --- /dev/null +++ b/scripts/eval-scores.mjs @@ -0,0 +1,392 @@ +#!/usr/bin/env node +/** + * Dense score diagnostics for #192. + * + * Answers one question before any threshold is chosen: **is a single dense similarity + * threshold even capable of separating "this answers the question" from "this does not"?** + * + * Three constraints, all deliberate: + * + * - **dense only.** A hybrid `score` is an RRF value (`1 / (60 + rank)`) and is not on the + * same scale as a normalised cosine. Mixing them would manufacture a new misleading + * number, which is what the threshold discussion is trying to avoid. + * - **validation only.** The distribution is used to choose where to sweep, so looking at + * `test` first would be tuning on the reporting side. + * - **the whole corpus per query** (`candidateK` ≥ index size) and `threshold = 0`, so the + * distribution is not already truncated by the parameter being investigated. + * + * Usage: + * node scripts/eval-scores.mjs + */ + +import { spawn } from 'node:child_process' +import { existsSync, mkdtempSync, mkdirSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join, resolve } from 'node:path' + +const OUT_MD = resolve(readArg('--out=', 'docs/eval/scores-v1.6.md')) +const OUT_JSON = OUT_MD.replace(/\.md$/, '.json') + +/** The shipped dense pipeline. Only the split and the candidate depth are changed. */ +const SPLIT = 'validation' +const CONTEXT_K = 3 +/** Above the index size, so every chunk is scored for every query. */ +const CANDIDATE_K = 500 + +function readArg(prefix, fallback) { + const arg = process.argv.find((value) => value.startsWith(prefix)) + return arg ? arg.slice(prefix.length) : fallback +} + +// Node 24 refuses to spawn a `.cmd`/`.bat` without `shell: true` (EINVAL), and the +// `.bin` entry is exactly that on Windows. Use the real binary the wrapper runs. +const { default: electronBinary } = await import('electron') +const executable = resolve(electronBinary) + +if (!existsSync(executable)) { + console.error('[scores] could not find the electron binary. Run `npm install` first.') + process.exit(1) +} + +function runHarness(outDir) { + return new Promise((resolvePromise, reject) => { + const args = [ + '.', + '--eval-harness', + '--eval-baseline=scores', + `--eval-out=${outDir}`, + `--eval-split=${SPLIT}`, + '--eval-retrieval=dense', + `--eval-candidate-k=${CANDIDATE_K}`, + `--eval-context-k=${CONTEXT_K}`, + '--eval-threshold=0', + '--eval-scores' + ] + + const isRoot = typeof process.getuid === 'function' && process.getuid() === 0 + if (isRoot || process.env.CI) args.push('--no-sandbox') + + const child = spawn(executable, args, { + stdio: ['ignore', 'pipe', 'pipe'], + env: { ...process.env, ELECTRON_DISABLE_SECURITY_WARNINGS: '1' } + }) + + child.on('error', reject) + child.on('exit', (code) => { + if (code !== 0) { + reject(new Error(`diagnostics harness exited with code ${code}`)) + return + } + const reportPath = join(outDir, 'baseline-scores.json') + if (!existsSync(reportPath)) { + reject(new Error('diagnostics harness wrote no report')) + return + } + resolvePromise(JSON.parse(readFileSync(reportPath, 'utf8'))) + }) + }) +} + +/** Nearest-rank quantile, matching `src/main/eval/metrics.ts`. */ +function quantile(values, p) { + if (values.length === 0) return null + const sorted = [...values].sort((a, b) => a - b) + const rank = Math.ceil((p / 100) * sorted.length) + return sorted[Math.min(Math.max(rank - 1, 0), sorted.length - 1)] +} + +/** + * `score = (1 + cosine) / 2`, from `SQLiteVectorStore`. Reported everywhere alongside the + * score because `threshold: 0.5` has been read as "cosine ≥ 0.5" and is really "cosine ≥ 0". + */ +const toCosine = (score) => (score === null ? null : 2 * score - 1) + +const QUANTILES = [0, 10, 25, 50, 75, 90, 100] + +function describe(values) { + if (values.length === 0) return null + const out = {} + for (const p of QUANTILES) out[`p${p}`] = quantile(values, p) + return out +} + +const workDir = mkdtempSync(join(tmpdir(), 'knownote-scores-')) +let report +try { + console.log('[scores] running the dense harness on validation, threshold 0') + report = await runHarness(workDir) +} finally { + rmSync(workDir, { recursive: true, force: true }) +} + +const perQuestion = report.perQuestion ?? [] +const answerable = perQuestion.filter((q) => q.answerable) +const unanswerable = perQuestion.filter((q) => !q.answerable) + +if (perQuestion.some((q) => !q.retrievedScores)) { + console.error('[scores] the report carries no scores; did --eval-scores reach the harness?') + process.exit(1) +} + +/** Per-question score landmarks. */ +function landmarks(question) { + const scores = question.retrievedScores + const relevant = [] + const nonRelevant = [] + scores.forEach((score, index) => { + if ((question.matchesByRank[index] ?? []).length > 0) relevant.push(score) + else nonRelevant.push(score) + }) + if (relevant.length === 0) return null + const bestRelevant = Math.max(...relevant) + const worstRelevant = Math.min(...relevant) + const bestNonRelevant = nonRelevant.length > 0 ? Math.max(...nonRelevant) : null + return { + id: question.id, + type: question.type, + bestRelevant, + worstRelevant, + bestNonRelevant, + margin: bestNonRelevant === null ? null : bestRelevant - bestNonRelevant + } +} + +const answerableLandmarks = answerable.map(landmarks).filter((entry) => entry !== null) +const unanswerableMax = unanswerable.map((q) => Math.max(...q.retrievedScores)) + +const byType = {} +for (const entry of answerableLandmarks) { + byType[entry.type] = byType[entry.type] ?? [] + byType[entry.type].push(entry) +} + +/** + * The curve that decides whether one threshold can do both jobs. For each candidate `t`: + * + * - **hit** — share of answerable questions that still have *some* relevant passage; + * - **fullRecall** — share whose *every* ground-truth block is still covered; + * - **abstain** — share of unanswerable questions that now return nothing. + * + * A useful threshold needs `abstain` to rise while `fullRecall` holds. If every `t` that + * raises abstention also drops full recall, no single threshold can carry the job. + */ +const thresholdGrid = [] +for (let t = 0.5; t <= 1.0001; t += 0.025) thresholdGrid.push(Number(t.toFixed(3))) + +const curve = thresholdGrid.map((threshold) => { + let hit = 0 + let fullRecall = 0 + for (const question of answerable) { + const covered = new Set() + let coveredAbove = 0 + question.matchesByRank.forEach((matches, index) => { + if (question.retrievedScores[index] < threshold) return + for (const match of matches) { + if (!covered.has(match)) { + covered.add(match) + coveredAbove += 1 + } + } + }) + if (coveredAbove > 0) hit += 1 + if (question.relevantCount > 0 && coveredAbove === question.relevantCount) fullRecall += 1 + } + + const abstained = unanswerable.filter((q) => Math.max(...q.retrievedScores) < threshold).length + return { + threshold, + cosine: Number(toCosine(threshold).toFixed(3)), + hitRate: answerable.length === 0 ? 0 : hit / answerable.length, + fullRecallRate: answerable.length === 0 ? 0 : fullRecall / answerable.length, + abstentionRate: unanswerable.length === 0 ? 0 : abstained / unanswerable.length + } +}) + +const format4 = (value) => (value === null ? '—' : value.toFixed(4)) +const quantileRow = (label, values) => { + const d = describe(values) + if (!d) return `| ${label} | 0 | ${QUANTILES.map(() => '—').join(' | ')} |` + const cells = QUANTILES.map((p) => `${d[`p${p}`].toFixed(4)} (${toCosine(d[`p${p}`]).toFixed(3)})`) + return `| ${label} | ${values.length} | ${cells.join(' | ')} |` +} + +const relevantBest = answerableLandmarks.map((entry) => entry.bestRelevant) +const relevantWorst = answerableLandmarks.map((entry) => entry.worstRelevant) +const nonRelevantBest = answerableLandmarks + .map((entry) => entry.bestNonRelevant) + .filter((value) => value !== null) +const margins = answerableLandmarks.map((entry) => entry.margin).filter((value) => value !== null) + +const typeRows = Object.entries(byType) + .sort(([a], [b]) => a.localeCompare(b)) + .map(([type, entries]) => { + const best = describe(entries.map((entry) => entry.bestRelevant)) + const worst = describe(entries.map((entry) => entry.worstRelevant)) + return `| ${type} | ${entries.length} | ${format4(best.p50)} | ${format4(best.p10)} | ${format4(worst.p10)} |` + }) + .join('\n') + +const curveRows = curve + .map( + (row) => + `| ${row.threshold.toFixed(3)} | ${row.cosine.toFixed(3)} | ${format4(row.hitRate)} | ` + + `${format4(row.fullRecallRate)} | ${format4(row.abstentionRate)} |` + ) + .join('\n') + +/** Where the three distributions overlap, which is what the threshold decision turns on. */ +const overlap = { + worstRelevantP10: quantile(relevantWorst, 10), + bestNonRelevantP90: quantile(nonRelevantBest, 90), + unanswerableMaxP50: quantile(unanswerableMax, 50), + unanswerableMaxP90: quantile(unanswerableMax, 90) +} +const separable = + overlap.worstRelevantP10 !== null && + overlap.unanswerableMaxP90 !== null && + overlap.worstRelevantP10 > overlap.unanswerableMaxP90 + +const markdown = `# Dense score diagnostics — v1.6 (#192) + +Generated by \`node scripts/eval-scores.mjs\`. Numbers are harness output; do not edit them by hand. + +## Read this first: the score is not a cosine + +\`SQLiteVectorStore\` computes + +\`\`\`text +score = 1 - distance / 2 +distance = 1 - cosine (sqlite-vec, distance_metric=cosine) +=> score = (1 + cosine) / 2 +\`\`\` + +so the configured \`threshold\` is an **affine map of the cosine**, not the cosine: + +| threshold | raw cosine | +| --- | --- | +| 0.3 | -0.4 | +| 0.4 | -0.2 | +| **0.5 (shipped)** | **0.0** | +| 0.6 | 0.2 | +| 0.8 | 0.6 | +| 1.0 | 1.0 | + +The shipped \`threshold = 0.5\` means **cosine ≥ 0**, which is very permissive. Every +threshold row below carries both columns so the two never get confused again, and every +mention of "the sweep looked too low" has to be read through this table. + +## What was measured + +Dense only, \`${SPLIT}\` split only, \`threshold = 0\`, \`candidateK = ${CANDIDATE_K}\` (above the +index size, so every chunk is scored for every query), \`contextK = ${CONTEXT_K}\`. + +- answerable questions: ${answerable.length} +- unanswerable questions: ${unanswerable.length} +- index size: ${report.config.chunkCount} chunks + +Hybrid is deliberately excluded: its \`score\` is an RRF value (\`1 / (60 + rank)\`) and is not +on the same scale as a normalised cosine. + +## Distributions (score, with raw cosine in brackets) + +\`best relevant\` is the highest-scoring passage that covers ground truth; \`worst relevant\` +is the lowest one that still has to survive for the question to be fully answered; +\`best non-relevant\` is the highest-scoring passage that covers nothing; \`margin\` is the +first minus the third. + +| Distribution | n | min | p10 | p25 | p50 | p75 | p90 | max | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | +${quantileRow('best relevant', relevantBest)} +${quantileRow('worst relevant', relevantWorst)} +${quantileRow('best non-relevant', nonRelevantBest)} +${quantileRow('margin (best rel − best non-rel)', margins)} +${quantileRow('unanswerable max candidate', unanswerableMax)} + +## By query type + +\`best relevant\` p50 and p10, and \`worst relevant\` p10 — the last is the one that decides +whether a cross-lingual question survives a threshold that a semantic question tolerates. + +| Type | n | best rel p50 | best rel p10 | worst rel p10 | +| --- | --- | --- | --- | --- | +${typeRows} + +## Threshold curve + +For each candidate threshold: **hit** = share of answerable questions that still have some +relevant passage; **full recall** = share whose every ground-truth block is still covered; +**abstain** = share of unanswerable questions that now return nothing. + +| threshold (score) | raw cosine | answerable hit | answerable full recall | unanswerable abstain | +| --- | --- | --- | --- | --- | +${curveRows} + +## The separation question + +The threshold decision turns on whether these overlap: + +| Landmark | score | raw cosine | +| --- | --- | --- | +| worst relevant, p10 | ${format4(overlap.worstRelevantP10)} | ${overlap.worstRelevantP10 === null ? '—' : toCosine(overlap.worstRelevantP10).toFixed(3)} | +| best non-relevant, p90 | ${format4(overlap.bestNonRelevantP90)} | ${overlap.bestNonRelevantP90 === null ? '—' : toCosine(overlap.bestNonRelevantP90).toFixed(3)} | +| unanswerable max, p50 | ${format4(overlap.unanswerableMaxP50)} | ${overlap.unanswerableMaxP50 === null ? '—' : toCosine(overlap.unanswerableMaxP50).toFixed(3)} | +| unanswerable max, p90 | ${format4(overlap.unanswerableMaxP90)} | ${overlap.unanswerableMaxP90 === null ? '—' : toCosine(overlap.unanswerableMaxP90).toFixed(3)} | + +**${ + separable + ? 'The distributions separate at the p10/p90 landmarks, so a single threshold is a plausible mechanism on this corpus. Read the grid above for where to sweep.' + : 'The distributions **overlap**, so a higher threshold buys abstention by giving up required relevant passages. If the curve above shows abstention rising only as full recall falls, then the honest conclusion is that **a single dense similarity threshold cannot carry both recall and abstention** — and the next mechanism to evaluate is not a finer threshold grid but a different signal (reranker score, top1−top2 margin, per-query thresholds, or claim-level answerability).' +} + +## Reproduce + +\`\`\`bash +npm run eval:prepare # one-time, networked model bootstrap +npm run eval:scores # offline; rewrites this file +\`\`\` +` + +mkdirSync(resolve(OUT_MD, '..'), { recursive: true }) +writeFileSync( + OUT_JSON, + `${JSON.stringify( + { + baseline: 'v1.6', + split: SPLIT, + candidateK: CANDIDATE_K, + threshold: 0, + indexSize: report.config.chunkCount, + counts: { answerable: answerable.length, unanswerable: unanswerable.length }, + distributions: { + bestRelevant: describe(relevantBest), + worstRelevant: describe(relevantWorst), + bestNonRelevant: describe(nonRelevantBest), + margin: describe(margins), + unanswerableMax: describe(unanswerableMax) + }, + byType: Object.fromEntries( + Object.entries(byType).map(([type, entries]) => [ + type, + { + questions: entries.length, + bestRelevant: describe(entries.map((entry) => entry.bestRelevant)), + worstRelevant: describe(entries.map((entry) => entry.worstRelevant)) + } + ]) + ), + overlap, + separable, + curve + }, + null, + 2 + )}\n` +) +writeFileSync(OUT_MD, markdown) + +console.log(`[scores] answerable ${answerable.length}, unanswerable ${unanswerable.length}, index ${report.config.chunkCount}`) +console.log( + `[scores] worst relevant p10 = ${format4(overlap.worstRelevantP10)} (cosine ${overlap.worstRelevantP10 === null ? '—' : toCosine(overlap.worstRelevantP10).toFixed(3)}), unanswerable max p90 = ${format4(overlap.unanswerableMaxP90)}` +) +console.log(`[scores] separable = ${separable}`) +console.log(`[scores] wrote ${OUT_JSON} and ${OUT_MD}`) diff --git a/src/main/eval/harness.ts b/src/main/eval/harness.ts index 44427de..f34ee5b 100644 --- a/src/main/eval/harness.ts +++ b/src/main/eval/harness.ts @@ -72,6 +72,13 @@ export interface EvalHarnessOptions { threshold: number /** 分块配置(#78)。实验变体通过它选择策略;缺省时用生产默认值。 */ chunkOptions: ChunkOptions + /** + * 把每个 rank 的检索分数也写进报告(`--eval-scores`)。 + * + * 缺省关闭:分数序列会让基线膨胀一倍,而基线是 CI 逐字节 diff 的文件。score + * diagnostics(#192)需要它,生产基线不需要。 + */ + includeScores?: boolean /** 检索策略(#77):dense / sparse(BM25) / hybrid(RRF)。 */ strategy: RetrievalStrategy } @@ -298,7 +305,7 @@ export async function runEvalHarness( .filter((index) => index >= 0) }) - perQuestion.push({ + const questionReport: QuestionReport = { id: question.id, question: question.question, type: question.type ?? UNTAGGED, @@ -310,7 +317,10 @@ export async function runEvalHarness( .slice(0, options.contextK) .reduce((total, result) => total + result.content.length, 0), matchesByRank - }) + } + // Only when asked: the score series doubles the size of the committed baseline. + if (options.includeScores) questionReport.retrievedScores = results.map((r) => r.score) + perQuestion.push(questionReport) } // 不可答的问题不进排名指标:它们没有 ground truth,`recallAtK` 对它们返回的是 0/0 @@ -396,6 +406,14 @@ function roundMetrics(metrics: EvalMetrics): EvalMetrics { } export function stabilize(report: EvalReport): EvalReport { + // Scores are extra data, not metrics; they are rounded the same way so two runs of the + // same diagnostic diff cleanly. + const perQuestion = report.perQuestion.map((question) => + question.retrievedScores + ? { ...question, retrievedScores: question.retrievedScores.map(roundMetric) } + : question + ) + return { ...report, metrics: roundMetrics(report.metrics), @@ -407,7 +425,7 @@ export function stabilize(report: EvalReport): EvalReport { // full report keeps it for the #78 comparison. indexingMs: Math.round(report.timing.indexingMs) }, - perQuestion: report.perQuestion + perQuestion } } diff --git a/src/main/eval/run.ts b/src/main/eval/run.ts index 20c13fa..ad45091 100644 --- a/src/main/eval/run.ts +++ b/src/main/eval/run.ts @@ -200,6 +200,9 @@ export async function runEvalCli(argv: readonly string[] = process.argv): Promis contextK: readNumberOption(argv, '--eval-context-k=', 3), threshold: readNumberOption(argv, '--eval-threshold=', 0.5), chunkOptions: readChunkOptions(argv), + // `--eval-scores` puts the per-rank retrieval scores in the report. Off by default: + // the committed baseline is a CI-diffed file and the series doubles it. + includeScores: readBoolOption(argv, '--eval-scores=', argv.includes('--eval-scores')), strategy: readRetrievalStrategy(argv) } diff --git a/src/main/eval/types.ts b/src/main/eval/types.ts index 715396a..6a33005 100644 --- a/src/main/eval/types.ts +++ b/src/main/eval/types.ts @@ -197,6 +197,15 @@ export interface QuestionReport { contextChars: number /** Ground-truth indices matched by each retrieved rank, in rank order. */ matchesByRank: number[][] + /** + * 每个 rank 的检索分数,与 `matchesByRank` 同序。 + * + * 只在 `--eval-scores` 时出现:它是 score diagnostics(#192)需要的数据,而基线 + * JSON 默认不带它——分数序列会让基线膨胀一倍,而基线是 CI 要逐字节 diff 的文件。 + * + * 注意它不是 cosine:见 `SQLiteVectorStore`,`score = (1 + cosine) / 2`。 + */ + retrievedScores?: number[] } export interface EvalReport { diff --git a/src/main/vectorstore/SQLiteVectorStore.ts b/src/main/vectorstore/SQLiteVectorStore.ts index 8fc300b..8f696f1 100644 --- a/src/main/vectorstore/SQLiteVectorStore.ts +++ b/src/main/vectorstore/SQLiteVectorStore.ts @@ -171,7 +171,11 @@ export class SQLiteVectorStore implements VectorStore { // 转换结果 const queryResults: QueryResult[] = results.map((row) => { - // cosine 距离转相似度(0-1) + // `score = 1 - distance / 2 = (1 + cosine) / 2`。 + // + // **这不是 cosine 本身**,而是仿射映射:score 0.5 对应 cosine 0,score 0.6 对应 + // cosine 0.2,score 1.0 才是 cosine 1.0。配置里的 `threshold` 就是这个 score, + // 把 0.5 读成“cosine ≥ 0.5”会把门槛高估很多(#192 评审)。 const score = 1 - row.distance / 2 return { From 036a95ae71f554a9f427a8cb43ca0129e1ce7f63 Mon Sep 17 00:00:00 2001 From: MrSibe Date: Wed, 30 Sep 2026 18:13:54 +0800 Subject: [PATCH 16/18] =?UTF-8?q?fix(eval):=20convert=20a=20score=20margin?= =?UTF-8?q?=20with=20=C3=972,=20not=20with=20the=20absolute=20affine=20tra?= =?UTF-8?q?nsform?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `score = (1 + cosine) / 2` is an affine map, so an **absolute** score converts as `cosine = 2·score − 1`. A **difference** of scores does not: the `+1` cancels and `Δcosine = 2·Δscore`. The margin row applied the absolute transform, which reported `2Δscore − 1` — for a margin near zero that is a cosine margin near **−1**, a sign flip on top of a scale error: ``` before after margin p10 -0.0272 (-1.054) -0.0272 (-0.054) margin p50 +0.0026 (-0.995) +0.0026 (+0.005) ``` `quantileRow` now takes the cosine transform as an argument and the margin row passes `toCosineMargin`, so the two cannot be confused at the call site again. The report also says which mapping applies to which row. This does not change the separability conclusion — the p90 for the best non-relevant passage is 0.9386 against a worst-required-relevant p10 of 0.8906, and that comparison uses absolute scores, which were always converted correctly. The margin itself is also now labelled in the report as an **oracle** quantity: at runtime nothing knows which result is relevant, so it measures how much the score separates the two, and it is not a signal the product could use. Part of #192 (correction to the score diagnostics). --- docs/eval/scores-v1.6.md | 11 ++++++++++- scripts/eval-scores.mjs | 30 +++++++++++++++++++++++++++--- 2 files changed, 37 insertions(+), 4 deletions(-) diff --git a/docs/eval/scores-v1.6.md b/docs/eval/scores-v1.6.md index 2ad38b7..18d8928 100644 --- a/docs/eval/scores-v1.6.md +++ b/docs/eval/scores-v1.6.md @@ -46,12 +46,21 @@ is the lowest one that still has to survive for the question to be fully answere `best non-relevant` is the highest-scoring passage that covers nothing; `margin` is the first minus the third. +The margin is an **oracle** quantity: at runtime nothing knows which result is relevant, so +it describes how much the score separates the two — it is not a signal a product could use. +Reading it as a candidate mechanism is the mistake the runtime-signal evaluation exists to +avoid. + +Where a row shows a raw cosine in brackets: an **absolute** score maps as +`cosine = 2·score − 1`, while a **margin** maps as `Δcosine = 2·Δscore` because the +`+1` cancels. The `margin` row uses the latter, the others the former. + | Distribution | n | min | p10 | p25 | p50 | p75 | p90 | max | | --- | --- | --- | --- | --- | --- | --- | --- | --- | | best relevant | 19 | 0.8808 (0.762) | 0.8906 (0.781) | 0.9120 (0.824) | 0.9359 (0.872) | 0.9428 (0.886) | 0.9530 (0.906) | 0.9537 (0.907) | | worst relevant | 19 | 0.8808 (0.762) | 0.8906 (0.781) | 0.9086 (0.817) | 0.9355 (0.871) | 0.9428 (0.886) | 0.9530 (0.906) | 0.9537 (0.907) | | best non-relevant | 19 | 0.8904 (0.781) | 0.9081 (0.816) | 0.9178 (0.836) | 0.9243 (0.849) | 0.9336 (0.867) | 0.9386 (0.877) | 0.9459 (0.892) | -| margin (best rel − best non-rel) | 19 | -0.0294 (-1.059) | -0.0272 (-1.054) | -0.0092 (-1.018) | 0.0026 (-0.995) | 0.0081 (-0.984) | 0.0329 (-0.934) | 0.0623 (-0.875) | +| margin (best rel − best non-rel) | 19 | -0.0294 (-0.059) | -0.0272 (-0.054) | -0.0092 (-0.018) | 0.0026 (0.005) | 0.0081 (0.016) | 0.0329 (0.066) | 0.0623 (0.125) | | unanswerable max candidate | 5 | 0.8978 (0.796) | 0.8978 (0.796) | 0.9128 (0.826) | 0.9259 (0.852) | 0.9343 (0.869) | 0.9424 (0.885) | 0.9424 (0.885) | ## By query type diff --git a/scripts/eval-scores.mjs b/scripts/eval-scores.mjs index 1356f2a..43e5bbf 100644 --- a/scripts/eval-scores.mjs +++ b/scripts/eval-scores.mjs @@ -101,6 +101,14 @@ function quantile(values, p) { */ const toCosine = (score) => (score === null ? null : 2 * score - 1) +/** + * A difference of two scores is not an affine map of a difference of two cosines: the + * `+1` cancels, so `Δcosine = 2 * Δscore`. Applying the absolute transform to a margin + * would have produced `2Δscore - 1`, which for a near-zero margin reports a cosine margin + * near −1 — a sign flip on top of a scale error. + */ +const toCosineMargin = (margin) => (margin === null ? null : 2 * margin) + const QUANTILES = [0, 10, 25, 50, 75, 90, 100] function describe(values) { @@ -203,10 +211,17 @@ const curve = thresholdGrid.map((threshold) => { }) const format4 = (value) => (value === null ? '—' : value.toFixed(4)) -const quantileRow = (label, values) => { +/** + * `transform` maps a value to its cosine counterpart. It defaults to the absolute-score + * transform and is overridden for margins, because the two do not share one. + */ +const quantileRow = (label, values, transform = toCosine) => { const d = describe(values) if (!d) return `| ${label} | 0 | ${QUANTILES.map(() => '—').join(' | ')} |` - const cells = QUANTILES.map((p) => `${d[`p${p}`].toFixed(4)} (${toCosine(d[`p${p}`]).toFixed(3)})`) + const cells = QUANTILES.map((p) => { + const value = d[`p${p}`] + return `${value.toFixed(4)} (${transform(value).toFixed(3)})` + }) return `| ${label} | ${values.length} | ${cells.join(' | ')} |` } @@ -294,12 +309,21 @@ is the lowest one that still has to survive for the question to be fully answere \`best non-relevant\` is the highest-scoring passage that covers nothing; \`margin\` is the first minus the third. +The margin is an **oracle** quantity: at runtime nothing knows which result is relevant, so +it describes how much the score separates the two — it is not a signal a product could use. +Reading it as a candidate mechanism is the mistake the runtime-signal evaluation exists to +avoid. + +Where a row shows a raw cosine in brackets: an **absolute** score maps as +\`cosine = 2·score − 1\`, while a **margin** maps as \`Δcosine = 2·Δscore\` because the +\`+1\` cancels. The \`margin\` row uses the latter, the others the former. + | Distribution | n | min | p10 | p25 | p50 | p75 | p90 | max | | --- | --- | --- | --- | --- | --- | --- | --- | --- | ${quantileRow('best relevant', relevantBest)} ${quantileRow('worst relevant', relevantWorst)} ${quantileRow('best non-relevant', nonRelevantBest)} -${quantileRow('margin (best rel − best non-rel)', margins)} +${quantileRow('margin (best rel − best non-rel)', margins, toCosineMargin)} ${quantileRow('unanswerable max candidate', unanswerableMax)} ## By query type From 8a99107bf408ec1f83126bbf6ee3b42f46f2dd4e Mon Sep 17 00:00:00 2001 From: MrSibe Date: Wed, 30 Sep 2026 18:16:49 +0800 Subject: [PATCH 17/18] feat(eval): per-question paired deltas, and hybrid's gain is five questions MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `npm run eval:paired` reports the dense ↔ hybrid pair per question and the classification per query type. It answers the one thing the aggregate comparison could not. The open question was: hybrid is ahead on the full set and level on `validation`, and an average cannot say whether that is a broad small gain or a handful of rescued cases. Those two readings imply different next steps. Child 2 of #192. ## What it found ``` validation 19 questions improved 2 / tied 16 / regressed 1 mean ΔnDCG@10 +0.0047 test 34 questions improved 5 / tied 29 / regressed 0 mean ΔnDCG@10 +0.0432 ``` **The gain is not broad — it is five questions on `test` and two on `validation`.** On the reporting side, 29 of 34 questions are untouched. The mean ΔnDCG@10 of +0.0432 is carried by `q013` (rank 11 → 4, ΔnDCG +0.43), `q059` (5 → 2), `q026` (2 → 1), `q049` (3 → 2) and `q052` (2 → 1). ## And it is not the story we had started telling ourselves The plausible story was "hybrid helps the hard negatives". The breakdown says otherwise: | type | n (test) | improved | tied | regressed | | --- | --- | --- | --- | --- | | cross-lingual | 5 | **0** | 5 | 0 | | hard-negative | 5 | **1** | 4 | 0 | | multi-hop | 3 | 1 | 2 | 0 | | semantic | 15 | **3** | 12 | 0 | | exact | 4 | 0 | 4 | 0 | Three of the five wins are `semantic`, one is `hard-negative` and one `multi-hop`. With n=1 of 5, "hybrid helps hard-negative questions" is **not** supported at this size — which is exactly the pattern that would have been asserted from the aggregate number alone. **`cross-lingual` is untouched on both splits: 0 improved, 0 regressed.** Adding a sparse channel does nothing for the category that is measurably weakest, which is itself a useful negative result and a reason not to reach for hybrid as the answer to it. No question was found by one strategy and missed by the other, on either split, so no regression hides behind the `0 = not found` sentinel. ## How Metrics come from `src/main/eval/metrics.ts` — imported, not reimplemented — so a delta cannot disagree with the metric it is a delta of. `firstRelevantRank` is `0` for "never found", so the miss cases are classified explicitly rather than folded into an arithmetic delta that cannot express them. `validation` is the selecting side; the `test` breakdown **explains** the observed difference and is labelled as not-for-selection in the report itself. Nothing here changes the shipped strategy. ## Testing - `npm run typecheck` — clean - `npm run eval:paired` — 4 harness runs, writes `docs/eval/paired-v1.6.{json,md}` - The committed baseline is unchanged Part of #192 (child 2: corpus diagnostics). --- .prettierignore | 4 + docs/eval/paired-v1.6.json | 1254 ++++++++++++++++++++++++++++++++++++ docs/eval/paired-v1.6.md | 97 +++ eval/README.md | 1 + package.json | 3 +- scripts/eval-paired.mjs | 368 +++++++++++ 6 files changed, 1726 insertions(+), 1 deletion(-) create mode 100644 docs/eval/paired-v1.6.json create mode 100644 docs/eval/paired-v1.6.md create mode 100644 scripts/eval-paired.mjs diff --git a/.prettierignore b/.prettierignore index da32ca2..f418484 100644 --- a/.prettierignore +++ b/.prettierignore @@ -33,6 +33,10 @@ docs/eval/sweep-*.md docs/eval/scores-*.json docs/eval/scores-*.md +# And by `scripts/eval-paired.mjs`. Regenerate with `npm run eval:paired`. +docs/eval/paired-*.json +docs/eval/paired-*.md + # And the same again for `scripts/eval-retrieval.mjs` (#77). docs/eval/retrieval-*.json docs/eval/retrieval-*.md diff --git a/docs/eval/paired-v1.6.json b/docs/eval/paired-v1.6.json new file mode 100644 index 0000000..dd1bfc2 --- /dev/null +++ b/docs/eval/paired-v1.6.json @@ -0,0 +1,1254 @@ +{ + "baseline": "v1.6", + "strategies": [ + "dense", + "hybrid" + ], + "splits": { + "validation": { + "questions": 19, + "meanDelta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0.004654087156011809 + }, + "byType": [ + { + "type": "cross-lingual", + "questions": 4, + "improved": 0, + "tied": 4, + "regressed": 0, + "meanDeltaNdcg10": 0, + "meanDeltaRank": 0 + }, + { + "type": "exact", + "questions": 4, + "improved": 2, + "tied": 2, + "regressed": 0, + "meanDeltaNdcg10": 0.05006329887451612, + "meanDeltaRank": 0.5 + }, + { + "type": "hard-negative", + "questions": 2, + "improved": 0, + "tied": 2, + "regressed": 0, + "meanDeltaNdcg10": 0, + "meanDeltaRank": 0 + }, + { + "type": "multi-hop", + "questions": 2, + "improved": 0, + "tied": 2, + "regressed": 0, + "meanDeltaNdcg10": 0.08861964339202388, + "meanDeltaRank": 0 + }, + { + "type": "semantic", + "questions": 6, + "improved": 0, + "tied": 5, + "regressed": 1, + "meanDeltaNdcg10": -0.04817747105298131, + "meanDeltaRank": -0.3333333333333333 + }, + { + "type": "zh", + "questions": 1, + "improved": 0, + "tied": 1, + "regressed": 0, + "meanDeltaNdcg10": 0, + "meanDeltaRank": 0 + } + ], + "pairs": [ + { + "id": "q002", + "type": "exact", + "question": "How many replicate samples are collected at each river station?", + "dense": { + "firstRelevantRank": 3, + "recallAt5": 1, + "ndcgAt10": 0.5 + }, + "hybrid": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "delta": { + "rank": 1, + "recallAt5": 0, + "ndcgAt10": 0.13092975357145753 + }, + "classification": "improved" + }, + { + "id": "q004", + "type": "semantic", + "question": "Why do nickel-rich battery packs need more aggressive thermal management?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q009", + "type": "semantic", + "question": "Why are trees with aggressive surface roots unsuitable for narrow verges?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q011", + "type": "exact", + "question": "Is the salt percentage in fermentation based on vegetable weight or water weight?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q015", + "type": "semantic", + "question": "Why does deep-water oxygen fall while a lake remains stratified?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q018", + "type": "semantic", + "question": "Why is one continuous planted roof layer better than several isolated planted beds?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q021", + "type": "semantic", + "question": "Why is wave energy harder to schedule ahead than tidal energy?", + "dense": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "hybrid": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q023", + "type": "multi-hop", + "question": "How does the river sampling protocol differ from the lake sampling protocol?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 0.5, + "ndcgAt10": 0.6131471927654584 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 0.5, + "ndcgAt10": 0.7903864795495061 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0.17723928678404777 + }, + "classification": "tied" + }, + { + "id": "q028", + "type": "zh", + "question": "茶叶储存的相对湿度上限是多少?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q030", + "type": "cross-lingual", + "question": "为什么潮汐能比风能和太阳能更容易提前安排发电?", + "dense": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "hybrid": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q037", + "type": "cross-lingual", + "question": "推移质为什么比悬移质更难测?", + "dense": { + "firstRelevantRank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "hybrid": { + "firstRelevantRank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "delta": { + "rank": null, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q040", + "type": "cross-lingual", + "question": "富镍电池为什么对散热要求更高?", + "dense": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "hybrid": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q043", + "type": "cross-lingual", + "question": "花期遭遇晚霜,主要受损的是什么?", + "dense": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "hybrid": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q045", + "type": "exact", + "question": "What is the default retry count in Gateway API v2?", + "dense": { + "firstRelevantRank": 4, + "recallAt5": 1, + "ndcgAt10": 0.43067655807339306 + }, + "hybrid": { + "firstRelevantRank": 3, + "recallAt5": 1, + "ndcgAt10": 0.5 + }, + "delta": { + "rank": 1, + "recallAt5": 0, + "ndcgAt10": 0.06932344192660694 + }, + "classification": "improved" + }, + { + "id": "q047", + "type": "hard-negative", + "question": "What is the context window of Gateway API v3, in tokens?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q050", + "type": "semantic", + "question": "Which Gateway API version waits longest between retries after a failed request?", + "dense": { + "firstRelevantRank": 10, + "recallAt5": 0, + "ndcgAt10": 0.2890648263178879 + }, + "hybrid": { + "firstRelevantRank": 12, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "delta": { + "rank": -2, + "recallAt5": 0, + "ndcgAt10": -0.2890648263178879 + }, + "classification": "regressed" + }, + { + "id": "q055", + "type": "exact", + "question": "How many replicate samples are collected at each reservoir monitoring station?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q057", + "type": "hard-negative", + "question": "At what depth below the surface does the coastal programme place its shore-station sensor?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q060", + "type": "multi-hop", + "question": "Compare the sampling interval and replicate count of the reservoir and groundwater programmes.", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + } + ] + }, + "test": { + "questions": 34, + "meanDelta": { + "rank": 0.3939393939393939, + "recallAt5": 0.029411764705882353, + "ndcgAt10": 0.04318215877040841 + }, + "byType": [ + { + "type": "cross-lingual", + "questions": 5, + "improved": 0, + "tied": 5, + "regressed": 0, + "meanDeltaNdcg10": 0, + "meanDeltaRank": 0 + }, + { + "type": "exact", + "questions": 4, + "improved": 0, + "tied": 4, + "regressed": 0, + "meanDeltaNdcg10": 0, + "meanDeltaRank": 0 + }, + { + "type": "hard-negative", + "questions": 5, + "improved": 1, + "tied": 4, + "regressed": 0, + "meanDeltaNdcg10": 0.026185950714291507, + "meanDeltaRank": 0.2 + }, + { + "type": "multi-hop", + "questions": 3, + "improved": 1, + "tied": 2, + "regressed": 0, + "meanDeltaNdcg10": 0.09781329792785898, + "meanDeltaRank": 0.3333333333333333 + }, + { + "type": "semantic", + "questions": 15, + "improved": 3, + "tied": 12, + "regressed": 0, + "meanDeltaNdcg10": 0.06958825005592344, + "meanDeltaRank": 0.7333333333333333 + }, + { + "type": "zh", + "questions": 2, + "improved": 0, + "tied": 2, + "regressed": 0, + "meanDeltaNdcg10": 0, + "meanDeltaRank": 0 + } + ], + "pairs": [ + { + "id": "q001", + "type": "semantic", + "question": "Why is bedload harder to measure than suspended sediment?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q003", + "type": "semantic", + "question": "What is the central trade-off in lithium-ion cell design?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q005", + "type": "semantic", + "question": "What happens once the separator in a battery cell melts?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q006", + "type": "exact", + "question": "At what temperature do honeybees begin to forage?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q007", + "type": "semantic", + "question": "What does a late frost damage during full bloom?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q008", + "type": "semantic", + "question": "Why is a continuous tree canopy more effective at cooling than isolated trees?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q010", + "type": "exact", + "question": "At what temperature is lactic acid fermentation fastest?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q012", + "type": "semantic", + "question": "Why must tidal turbines be sited in places with very fast currents?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q013", + "type": "semantic", + "question": "What is the main environmental concern for tidal energy installations?", + "dense": { + "firstRelevantRank": 11, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "hybrid": { + "firstRelevantRank": 4, + "recallAt5": 1, + "ndcgAt10": 0.43067655807339306 + }, + "delta": { + "rank": 7, + "recallAt5": 1, + "ndcgAt10": 0.43067655807339306 + }, + "classification": "improved" + }, + { + "id": "q014", + "type": "exact", + "question": "In lake monitoring, how is the sampling depth actually recorded?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q016", + "type": "semantic", + "question": "How do supercapacitors hold their charge?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q017", + "type": "semantic", + "question": "Why can solitary bees pollinate a bloom week that is too cold for honeybee hives?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q019", + "type": "semantic", + "question": "Where do acetic acid bacteria sit in a vinegar culture?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q020", + "type": "semantic", + "question": "What happens if a vinegar culture is sealed airtight?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q022", + "type": "exact", + "question": "Where does siting for wave energy devices concentrate, and where does it not?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q024", + "type": "multi-hop", + "question": "A street canopy and a green roof are both said to cool; what surface does each one shade?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q025", + "type": "semantic", + "question": "Which preservation method depends on keeping air away from the food?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q026", + "type": "semantic", + "question": "Why can one cold morning cost a grower the whole crop even when colonies are brought in?", + "dense": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 1, + "recallAt5": 0, + "ndcgAt10": 0.36907024642854247 + }, + "classification": "improved" + }, + { + "id": "q027", + "type": "zh", + "question": "绿茶应该怎样保存才能减缓氧化?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q029", + "type": "zh", + "question": "为什么冷冻保存的茶叶取出后不能立刻打开包装?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q038", + "type": "cross-lingual", + "question": "河流监测中,每个站点要采几份平行样?", + "dense": { + "firstRelevantRank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "hybrid": { + "firstRelevantRank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "delta": { + "rank": null, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q039", + "type": "cross-lingual", + "question": "设计锂离子电芯时最核心的取舍是什么?", + "dense": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "hybrid": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q041", + "type": "cross-lingual", + "question": "电芯隔膜一旦熔化会导致什么后果?", + "dense": { + "firstRelevantRank": 4, + "recallAt5": 1, + "ndcgAt10": 0.43067655807339306 + }, + "hybrid": { + "firstRelevantRank": 4, + "recallAt5": 1, + "ndcgAt10": 0.43067655807339306 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q042", + "type": "cross-lingual", + "question": "蜜蜂大致从什么温度开始出巢觅食?", + "dense": { + "firstRelevantRank": 3, + "recallAt5": 1, + "ndcgAt10": 0.5 + }, + "hybrid": { + "firstRelevantRank": 3, + "recallAt5": 1, + "ndcgAt10": 0.5 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q044", + "type": "cross-lingual", + "question": "为什么成片的树冠比孤立的树降温效果更好?", + "dense": { + "firstRelevantRank": 9, + "recallAt5": 0, + "ndcgAt10": 0.3010299956639812 + }, + "hybrid": { + "firstRelevantRank": 9, + "recallAt5": 0, + "ndcgAt10": 0.3010299956639812 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q046", + "type": "hard-negative", + "question": "What is the default request timeout of Gateway API v1?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q048", + "type": "hard-negative", + "question": "What is the default rate limit of Gateway API v2?", + "dense": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "hybrid": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q049", + "type": "hard-negative", + "question": "What is the default request timeout of Gateway API v3?", + "dense": { + "firstRelevantRank": 3, + "recallAt5": 1, + "ndcgAt10": 0.5 + }, + "hybrid": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "delta": { + "rank": 1, + "recallAt5": 0, + "ndcgAt10": 0.13092975357145753 + }, + "classification": "improved" + }, + { + "id": "q051", + "type": "multi-hop", + "question": "Compare the default timeouts and rate limits of Gateway API v1 and v3.", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 0.75, + "ndcgAt10": 0.6366824387328317 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 0.75, + "ndcgAt10": 0.6366824387328317 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q052", + "type": "multi-hop", + "question": "Compare the retry count and context window of Gateway API v2 and v3.", + "dense": { + "firstRelevantRank": 2, + "recallAt5": 0.25, + "ndcgAt10": 0.35914753008966527 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 0.25, + "ndcgAt10": 0.6525874238732422 + }, + "delta": { + "rank": 1, + "recallAt5": 0, + "ndcgAt10": 0.29343989378357693 + }, + "classification": "improved" + }, + { + "id": "q056", + "type": "hard-negative", + "question": "How many replicate samples are collected at each estuary transect station?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q058", + "type": "hard-negative", + "question": "At what depth below the water table do groundwater boreholes carry their pressure transducer?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q059", + "type": "semantic", + "question": "Which monitoring programme samples most frequently?", + "dense": { + "firstRelevantRank": 5, + "recallAt5": 1, + "ndcgAt10": 0.38685280723454163 + }, + "hybrid": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "delta": { + "rank": 3, + "recallAt5": 0, + "ndcgAt10": 0.2440769463369159 + }, + "classification": "improved" + }, + { + "id": "q061", + "type": "semantic", + "question": "Why does the estuary programme take more replicate samples than the other monitoring programmes?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + } + ] + } + } +} diff --git a/docs/eval/paired-v1.6.md b/docs/eval/paired-v1.6.md new file mode 100644 index 0000000..dc185dd --- /dev/null +++ b/docs/eval/paired-v1.6.md @@ -0,0 +1,97 @@ +# Paired dense ↔ hybrid deltas — v1.6 (#192) + +Generated by `node scripts/eval-paired.mjs`. Numbers are harness output; do not edit them by hand. + +## What this answers + +The aggregate comparison says hybrid is ahead on the full set and level on `validation`. +An average cannot say whether that is a broad small gain or a few rescued cases, and those +two readings imply different next steps. This is the per-question pair. + +Metrics come from `src/main/eval/metrics.ts`, so a delta cannot disagree with the metric it +is a delta of. Unanswerable questions are excluded — they have no rank to compare, and the +abstention diagnostics cover them. + +`validation` is the side that **selects**; the `test` breakdown below **explains** the +difference that was observed and must not be used to choose. Nothing here changes the +shipped strategy. + +## Validation (this side selects) + +Chosen on this side, so a win here is eligible to inform a decision. + +Questions compared: 19. Mean ΔnDCG@10 +0.0047, +mean Δrank 0.00. + +### By query type + +| Type | n | improved | tied | regressed | mean ΔnDCG@10 | mean Δrank | +| --- | --- | --- | --- | --- | --- | --- | +| cross-lingual | 4 | 0 | 4 | 0 | 0.0000 | 0.00 | +| exact | 4 | 2 | 2 | 0 | +0.0501 | +0.50 | +| hard-negative | 2 | 0 | 2 | 0 | 0.0000 | 0.00 | +| multi-hop | 2 | 0 | 2 | 0 | +0.0886 | 0.00 | +| semantic | 6 | 0 | 5 | 1 | -0.0482 | -0.33 | +| zh | 1 | 0 | 1 | 0 | 0.0000 | 0.00 | + +### Biggest rank wins + +| id | type | dense rank | hybrid rank | Δrank | ΔnDCG@10 | +| --- | --- | --- | --- | --- | --- | +| q002 | exact | 3 | 2 | +1 | +0.1309 | +| q045 | exact | 4 | 3 | +1 | +0.0693 | + +### Biggest rank regressions + +| id | type | dense rank | hybrid rank | Δrank | ΔnDCG@10 | +| --- | --- | --- | --- | --- | --- | +| q050 | semantic | 10 | 12 | -2 | -0.2891 | + +**Found by hybrid, missed by dense**: none. + +**Lost by hybrid, found by dense**: none. + +## Test (explanatory only) + +Not used to choose anything. Present because the aggregate difference this is meant to explain was measured here. + +Questions compared: 34. Mean ΔnDCG@10 +0.0432, +mean Δrank +0.39. + +### By query type + +| Type | n | improved | tied | regressed | mean ΔnDCG@10 | mean Δrank | +| --- | --- | --- | --- | --- | --- | --- | +| cross-lingual | 5 | 0 | 5 | 0 | 0.0000 | 0.00 | +| exact | 4 | 0 | 4 | 0 | 0.0000 | 0.00 | +| hard-negative | 5 | 1 | 4 | 0 | +0.0262 | +0.20 | +| multi-hop | 3 | 1 | 2 | 0 | +0.0978 | +0.33 | +| semantic | 15 | 3 | 12 | 0 | +0.0696 | +0.73 | +| zh | 2 | 0 | 2 | 0 | 0.0000 | 0.00 | + +### Biggest rank wins + +| id | type | dense rank | hybrid rank | Δrank | ΔnDCG@10 | +| --- | --- | --- | --- | --- | --- | +| q013 | semantic | 11 | 4 | +7 | +0.4307 | +| q059 | semantic | 5 | 2 | +3 | +0.2441 | +| q026 | semantic | 2 | 1 | +1 | +0.3691 | +| q049 | hard-negative | 3 | 2 | +1 | +0.1309 | +| q052 | multi-hop | 2 | 1 | +1 | +0.2934 | + +### Biggest rank regressions + +| id | type | dense rank | hybrid rank | Δrank | ΔnDCG@10 | +| --- | --- | --- | --- | --- | --- | +| — | — | — | — | + +**Found by hybrid, missed by dense**: none. + +**Lost by hybrid, found by dense**: none. + +## Reproduce + +```bash +npm run eval:prepare # one-time, networked model bootstrap +npm run eval:paired # offline; rewrites this file +``` diff --git a/eval/README.md b/eval/README.md index 299eca7..22afea3 100644 --- a/eval/README.md +++ b/eval/README.md @@ -13,6 +13,7 @@ npm run eval:retrieval # strategy comparison (#77); validation selects, test re npm run eval:threshold # derive the similarity threshold on validation, report on test npm run eval:sweep # bounded grid over strategy × candidateK × contextK, one dashboard npm run eval:scores # dense score distribution: can one threshold separate relevant from not? +npm run eval:paired # per-question dense ↔ hybrid deltas, by query type npm run eval:blocks eval/corpus/foo.md # print the block ordinals ground truth must use ``` diff --git a/package.json b/package.json index 418c579..8f04662 100644 --- a/package.json +++ b/package.json @@ -45,7 +45,8 @@ "eval:threshold": "npm run build && node scripts/eval-threshold.mjs", "eval:sweep": "npm run build && node scripts/eval-sweep.mjs", "eval:blocks": "node --experimental-transform-types --import ./test/ts-resolve.mjs --disable-warning=ExperimentalWarning --disable-warning=MODULE_TYPELESS_PACKAGE_JSON scripts/eval-blocks.mjs", - "eval:scores": "npm run build && node scripts/eval-scores.mjs" + "eval:scores": "npm run build && node scripts/eval-scores.mjs", + "eval:paired": "npm run build && node --experimental-transform-types --import ./test/ts-resolve.mjs --disable-warning=ExperimentalWarning --disable-warning=MODULE_TYPELESS_PACKAGE_JSON scripts/eval-paired.mjs" }, "//test": [ "`node --test` strips TypeScript types rather than compiling them, and strip-only", diff --git a/scripts/eval-paired.mjs b/scripts/eval-paired.mjs new file mode 100644 index 0000000..b097947 --- /dev/null +++ b/scripts/eval-paired.mjs @@ -0,0 +1,368 @@ +#!/usr/bin/env node +/** + * Per-query paired deltas for dense ↔ hybrid (#192). + * + * The aggregate comparison left a question it cannot answer: hybrid is ahead on the full + * set and level on `validation`, and an average cannot say whether that is a broad small + * gain or a handful of cases it rescued. A mean over forty questions cannot distinguish + * "helps a little everywhere" from "helps a lot twice". + * + * So this reports the pair per question and the classification per query type: + * + * improved / tied / regressed, with the rank move and the nDCG@10 delta + * + * `validation` is the side that selects; the `test` breakdown **explains** the observed + * difference and must not be used to choose. Nothing here changes the shipped strategy. + * + * Metrics come from `src/main/eval/metrics.ts` rather than being reimplemented, so a delta + * cannot disagree with the metric it is a delta of. + * + * Usage: + * node scripts/eval-paired.mjs + */ + +import { spawn } from 'node:child_process' +import { existsSync, mkdtempSync, mkdirSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join, resolve } from 'node:path' + +const OUT_MD = resolve(readArg('--out=', 'docs/eval/paired-v1.6.md')) +const OUT_JSON = OUT_MD.replace(/\.md$/, '.json') + +const SPLITS = ['validation', 'test'] +const STRATEGIES = [ + { id: 'dense', label: 'dense (vector)' }, + { id: 'hybrid', label: 'hybrid (RRF of dense + BM25)' } +] + +/** How many rows to show in each mover table. */ +const MOVER_LIMIT = 8 + +function readArg(prefix, fallback) { + const arg = process.argv.find((value) => value.startsWith(prefix)) + return arg ? arg.slice(prefix.length) : fallback +} + +// Node 24 refuses to spawn a `.cmd`/`.bat` without `shell: true` (EINVAL), and the +// `.bin` entry is exactly that on Windows. Use the real binary the wrapper runs. +const { default: electronBinary } = await import('electron') +const executable = resolve(electronBinary) + +if (!existsSync(executable)) { + console.error('[paired] could not find the electron binary. Run `npm install` first.') + process.exit(1) +} + +const { ndcgAtK, recallAtK } = await import('../src/main/eval/metrics.ts') + +function runStrategy(strategy, split, outDir) { + return new Promise((resolvePromise, reject) => { + const args = [ + '.', + '--eval-harness', + '--eval-baseline=paired', + `--eval-out=${outDir}`, + `--eval-split=${split}`, + `--eval-retrieval=${strategy.id}` + ] + + const isRoot = typeof process.getuid === 'function' && process.getuid() === 0 + if (isRoot || process.env.CI) args.push('--no-sandbox') + + const child = spawn(executable, args, { + stdio: ['ignore', 'pipe', 'pipe'], + env: { ...process.env, ELECTRON_DISABLE_SECURITY_WARNINGS: '1' } + }) + + child.on('error', reject) + child.on('exit', (code) => { + if (code !== 0) { + reject(new Error(`${strategy.id} (${split}) exited with code ${code}`)) + return + } + const reportPath = join(outDir, 'baseline-paired.json') + if (!existsSync(reportPath)) { + reject(new Error(`${strategy.id} (${split}) wrote no report`)) + return + } + resolvePromise(JSON.parse(readFileSync(reportPath, 'utf8'))) + }) + }) +} + +/** One side's per-question view, with the metrics the comparison is made of. */ +function sideOf(question) { + return { + firstRelevantRank: question.firstRelevantRank, + recallAt5: recallAtK(question.matchesByRank, question.relevantCount, 5), + ndcgAt10: ndcgAtK(question.matchesByRank, question.relevantCount, 10) + } +} + +/** + * Classify the pair. `firstRelevantRank` is 0 for "never found", so the miss cases cannot be + * folded into an arithmetic delta: going from found to not-found is a regression no rank + * number expresses. + */ +function classify(dense, hybrid) { + const denseFound = dense.firstRelevantRank > 0 + const hybridFound = hybrid.firstRelevantRank > 0 + + if (denseFound && hybridFound) { + const rankDelta = dense.firstRelevantRank - hybrid.firstRelevantRank + return { + classification: rankDelta > 0 ? 'improved' : rankDelta < 0 ? 'regressed' : 'tied', + rankDelta + } + } + if (!denseFound && hybridFound) return { classification: 'improved', rankDelta: null } + if (denseFound && !hybridFound) return { classification: 'regressed', rankDelta: null } + return { classification: 'tied', rankDelta: null } +} + +/** Human-readable rank move that survives the 0 = "not found" sentinel. */ +const rankText = (rank) => (rank > 0 ? String(rank) : 'not found') + +const format4 = (value) => (value === null ? '—' : value.toFixed(4)) +const signed = (value, digits = 4) => + value === null ? '—' : `${value > 0 ? '+' : ''}${value.toFixed(digits)}` + +const workDir = mkdtempSync(join(tmpdir(), 'knownote-paired-')) +const bySplit = {} + +try { + for (const split of SPLITS) { + const reports = {} + for (const strategy of STRATEGIES) { + console.log(`[paired] ${split}: ${strategy.label}`) + const outDir = join(workDir, `${split}-${strategy.id}`) + mkdirSync(outDir, { recursive: true }) + reports[strategy.id] = await runStrategy(strategy, split, outDir) + } + + const denseById = new Map(reports.dense.perQuestion.map((q) => [q.id, q])) + const pairs = [] + for (const hybridQuestion of reports.hybrid.perQuestion) { + const denseQuestion = denseById.get(hybridQuestion.id) + if (!denseQuestion) continue + // Unanswerable questions have no rank to compare; they are measured by the + // abstention diagnostics, not here. + if (!denseQuestion.answerable) continue + + const dense = sideOf(denseQuestion) + const hybrid = sideOf(hybridQuestion) + const { classification, rankDelta } = classify(dense, hybrid) + pairs.push({ + id: hybridQuestion.id, + type: hybridQuestion.type, + question: hybridQuestion.question, + dense, + hybrid, + delta: { + rank: rankDelta, + recallAt5: hybrid.recallAt5 - dense.recallAt5, + ndcgAt10: hybrid.ndcgAt10 - dense.ndcgAt10 + }, + classification + }) + } + + bySplit[split] = { + questions: pairs.length, + pairs, + byType: aggregateByType(pairs), + meanDelta: { + rank: mean(pairs.map((p) => p.delta.rank).filter((value) => value !== null)), + recallAt5: mean(pairs.map((p) => p.delta.recallAt5)), + ndcgAt10: mean(pairs.map((p) => p.delta.ndcgAt10)) + } + } + } +} finally { + rmSync(workDir, { recursive: true, force: true }) +} + +function mean(values) { + if (values.length === 0) return null + return values.reduce((a, b) => a + b, 0) / values.length +} + +function aggregateByType(pairs) { + const types = [...new Set(pairs.map((p) => p.type))].sort() + return types.map((type) => { + const group = pairs.filter((p) => p.type === type) + const count = (label) => group.filter((p) => p.classification === label).length + return { + type, + questions: group.length, + improved: count('improved'), + tied: count('tied'), + regressed: count('regressed'), + meanDeltaNdcg10: mean(group.map((p) => p.delta.ndcgAt10)), + meanDeltaRank: mean(group.map((p) => p.delta.rank).filter((value) => value !== null)) + } + }) +} + +/** Wins and regressions, by rank move. Misses are listed separately and always first. */ +function movers(pairs) { + const withRank = pairs.filter((p) => p.delta.rank !== null && p.delta.rank !== 0) + const wins = withRank + .filter((p) => p.delta.rank > 0) + .sort((a, b) => b.delta.rank - a.delta.rank) + .slice(0, MOVER_LIMIT) + const losses = withRank + .filter((p) => p.delta.rank < 0) + .sort((a, b) => a.delta.rank - b.delta.rank) + .slice(0, MOVER_LIMIT) + const foundByHybrid = pairs.filter( + (p) => p.dense.firstRelevantRank === 0 && p.hybrid.firstRelevantRank > 0 + ) + const lostByHybrid = pairs.filter( + (p) => p.dense.firstRelevantRank > 0 && p.hybrid.firstRelevantRank === 0 + ) + return { wins, losses, foundByHybrid, lostByHybrid } +} + +const typeTable = (rows) => + rows + .map( + (row) => + `| ${row.type} | ${row.questions} | ${row.improved} | ${row.tied} | ${row.regressed} | ` + + `${signed(row.meanDeltaNdcg10)} | ${signed(row.meanDeltaRank, 2)} |` + ) + .join('\n') + +const moverTable = (rows) => + rows.length === 0 + ? '| — | — | — | — |' + : rows + .map( + (p) => + `| ${p.id} | ${p.type} | ${rankText(p.dense.firstRelevantRank)} | ` + + `${rankText(p.hybrid.firstRelevantRank)} | ${signed(p.delta.rank, 0)} | ` + + `${signed(p.delta.ndcgAt10)} |` + ) + .join('\n') + +const splitSection = (name, data, note) => { + const { wins, losses, foundByHybrid, lostByHybrid } = movers(data.pairs) + return `## ${name} + +${note} + +Questions compared: ${data.questions}. Mean ΔnDCG@10 ${signed(data.meanDelta.ndcgAt10)}, +mean Δrank ${signed(data.meanDelta.rank, 2)}. + +### By query type + +| Type | n | improved | tied | regressed | mean ΔnDCG@10 | mean Δrank | +| --- | --- | --- | --- | --- | --- | --- | +${typeTable(data.byType)} + +### Biggest rank wins + +| id | type | dense rank | hybrid rank | Δrank | ΔnDCG@10 | +| --- | --- | --- | --- | --- | --- | +${moverTable(wins)} + +### Biggest rank regressions + +| id | type | dense rank | hybrid rank | Δrank | ΔnDCG@10 | +| --- | --- | --- | --- | --- | --- | +${moverTable(losses)} + +${ + foundByHybrid.length > 0 + ? `**Found by hybrid, missed by dense** (${foundByHybrid.length}): ${foundByHybrid + .map((p) => `${p.id} (${p.type})`) + .join(', ')}` + : '**Found by hybrid, missed by dense**: none.' +} + +${ + lostByHybrid.length > 0 + ? `**Lost by hybrid, found by dense** (${lostByHybrid.length}): ${lostByHybrid + .map((p) => `${p.id} (${p.type})`) + .join(', ')}` + : '**Lost by hybrid, found by dense**: none.' +} +` +} + +const markdown = `# Paired dense ↔ hybrid deltas — v1.6 (#192) + +Generated by \`node scripts/eval-paired.mjs\`. Numbers are harness output; do not edit them by hand. + +## What this answers + +The aggregate comparison says hybrid is ahead on the full set and level on \`validation\`. +An average cannot say whether that is a broad small gain or a few rescued cases, and those +two readings imply different next steps. This is the per-question pair. + +Metrics come from \`src/main/eval/metrics.ts\`, so a delta cannot disagree with the metric it +is a delta of. Unanswerable questions are excluded — they have no rank to compare, and the +abstention diagnostics cover them. + +\`validation\` is the side that **selects**; the \`test\` breakdown below **explains** the +difference that was observed and must not be used to choose. Nothing here changes the +shipped strategy. + +${splitSection( + 'Validation (this side selects)', + bySplit.validation, + 'Chosen on this side, so a win here is eligible to inform a decision.' +)} +${splitSection( + 'Test (explanatory only)', + bySplit.test, + 'Not used to choose anything. Present because the aggregate difference this is meant to explain was measured here.' +)} +## Reproduce + +\`\`\`bash +npm run eval:prepare # one-time, networked model bootstrap +npm run eval:paired # offline; rewrites this file +\`\`\` +` + +mkdirSync(resolve(OUT_MD, '..'), { recursive: true }) +writeFileSync( + OUT_JSON, + `${JSON.stringify( + { + baseline: 'v1.6', + strategies: ['dense', 'hybrid'], + splits: Object.fromEntries( + SPLITS.map((split) => [ + split, + { + questions: bySplit[split].questions, + meanDelta: bySplit[split].meanDelta, + byType: bySplit[split].byType, + pairs: bySplit[split].pairs + } + ]) + ) + }, + null, + 2 + )}\n` +) +writeFileSync(OUT_MD, markdown) + +for (const split of SPLITS) { + const data = bySplit[split] + const counts = data.byType.reduce( + (acc, row) => ({ + improved: acc.improved + row.improved, + tied: acc.tied + row.tied, + regressed: acc.regressed + row.regressed + }), + { improved: 0, tied: 0, regressed: 0 } + ) + console.log( + `[paired] ${split}: ${data.questions} questions, improved ${counts.improved} / tied ${counts.tied} / regressed ${counts.regressed}, mean ΔnDCG@10 ${signed(data.meanDelta.ndcgAt10)}` + ) +} +console.log(`[paired] wrote ${OUT_JSON} and ${OUT_MD}`) From d36883243cb4e424ac39040443c4eac008d56079 Mon Sep 17 00:00:00 2001 From: MrSibe Date: Wed, 30 Sep 2026 18:24:33 +0800 Subject: [PATCH 18/18] feat(eval): fifteen near-miss unanswerable questions, and the last eval PR MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Closes RAG Eval v2 Phase 1. This is the final expansion of the eval system — the conclusions below are the point of the phase, not more infrastructure. ## What was added Fifteen unanswerable questions whose **subject is discussed at length in the corpus and whose answer is not there** — "How much GPU memory does Gateway API v3 need?", "Which laboratory is accredited to analyse the estuary transect samples?", "How many staff work on the reservoir monitoring programme?". Every specific term was checked absent before the question was written (`gpu`, `vram`, `sla`, `uptime`, `expiry`, `sdk`, `self-hosted`, `accredit`, `vendor`, `supplier`, `funding`, `grant`, `staff`, `warranty`, `retention`, `encrypt`, `pricing`, `invoice`, `audit` all return nothing). That check is what makes them unanswerable rather than believed to be, and it is the part that cannot be automated away. `validation` unanswerable goes from **5 to 15**, so abstention resolution goes from 20 % per question to 6.7 %. The earlier "Who won the 2018 FIFA World Cup?" questions were a sanity check for a query that is obviously out of scope; these are the realistic shape — the user's actual hallucination risk is asking a question about a document that *is* in the library. ## The conclusion survives the larger sample ``` unanswerable max candidate min 0.8978 (cosine 0.796) p50 0.9279 (0.856) max 0.9435 (0.887) worst required relevant p10 0.8906 (cosine 0.781) ``` Even the **lowest** unanswerable top score is above the p10 of the worst required relevant passage. With n=15 instead of n=5 the distributions still do not separate, and the threshold curve is unchanged in shape: | threshold | raw cosine | answerable full recall | unanswerable abstain | | --- | --- | --- | --- | | 0.875 | 0.750 | 1.0000 | 0.0000 | | 0.900 | 0.800 | 0.8947 | 0.0667 | | 0.925 | 0.850 | 0.6316 | 0.4000 | | 0.950 | 0.900 | 0.1579 | 1.0000 | Abstention still never rises without full recall falling. The finer resolution makes the finding *more* solid, not different. Regenerated to stay consistent: baseline, scores, threshold and sweep. `paired-v1.6` is unchanged, because adding unanswerable questions does not touch any answerable one. ## Where Phase 1 leaves the product 1. The original 19-chunk benchmark could not guide a product decision; the current one (53 chunks, 78 questions, per-type metrics, held-out split) can. 2. **Cross-lingual retrieval is the largest measured gap**: Recall@5 0.6667 and nDCG@10 0.4173 against 0.93–1.00 for same-language questions. Chinese over a Chinese source scores 1.0000, so it is the cross-language matching, not Chinese. 3. **Hybrid has ranking value and no held-out mandate.** Its full-set advantage is five questions out of 34 (`paired-v1.6`), mostly `semantic`, and it does nothing for cross-lingual. Dense stays the default. 4. `contextK = 3` is justified against 5 and 8 on this corpus; `candidateK` above 10 is not distinguishable. 5. `threshold = 0.5` means **cosine ≥ 0**, not cosine ≥ 0.5. 6. **A single embedding similarity threshold cannot decide answerability** — the relevant and non-relevant and unanswerable score distributions overlap, and every threshold that abstains also drops required evidence. (6) is the phase's real result: it says stop tuning this knob and reach for a different mechanism. ## Stop line Phase 1 ends here. Not done, and deliberately not started: public benchmarks, reranker eval, generator faithfulness, citation entailment, growing the corpus to 300 chunks, an `Abstention Signal Evaluation`. Those build a more complete benchmark; they do not unblock KnowNote development, and (6) already says the next useful work is a different retrieval mechanism rather than a better measurement of this one. --- docs/eval/baseline-v1.6.json | 484 +++++++++++++++++++++++++++++++++- docs/eval/baseline-v1.6.md | 10 +- docs/eval/scores-v1.6.json | 16 +- docs/eval/scores-v1.6.md | 8 +- docs/eval/sweep-v1.6.json | 90 +++---- docs/eval/sweep-v1.6.md | 44 ++-- docs/eval/threshold-v1.6.json | 40 +-- docs/eval/threshold-v1.6.md | 2 +- eval/questions.jsonl | 15 ++ eval/splits.json | 17 +- 10 files changed, 618 insertions(+), 108 deletions(-) diff --git a/docs/eval/baseline-v1.6.json b/docs/eval/baseline-v1.6.json index d8171a1..2bddcea 100644 --- a/docs/eval/baseline-v1.6.json +++ b/docs/eval/baseline-v1.6.json @@ -17,7 +17,7 @@ "threshold": 0.5, "corpus": "eval/corpus", "documents": 21, - "questions": 63, + "questions": 78, "chunkCount": 53 }, "metrics": { @@ -124,7 +124,7 @@ } ], "unanswerable": { - "questions": 10, + "questions": 25, "abstentionCount": 0, "retrievalAbstentionRate": 0, "meanCandidatesRetrieved": 20, @@ -2277,6 +2277,486 @@ [], [] ] + }, + { + "id": "q064", + "question": "How much GPU memory does Gateway API v3 need to serve a request?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2908, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q065", + "question": "What is the uptime SLA for Gateway API v2?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2875, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q066", + "question": "How long does an authentication token stay valid before it expires?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2649, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q067", + "question": "Which SDK version should a client use with Gateway API v3?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2908, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q068", + "question": "Can Gateway API v2 be self-hosted on-premise?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2875, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q069", + "question": "How is usage invoiced for Gateway API v2?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2875, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q070", + "question": "Which laboratory is accredited to analyse the estuary transect samples?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2902, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q071", + "question": "Which vendor supplies the coastal monitoring buoys?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2826, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q072", + "question": "What is the annual funding for the groundwater monitoring programme?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2933, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q073", + "question": "How many staff work on the reservoir monitoring programme?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2096, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q074", + "question": "What is the warranty period on the estuary monitoring sensors?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2876, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q075", + "question": "How long are the coastal monitoring readings retained before deletion?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2749, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q076", + "question": "What encryption standard is used to transmit the river monitoring readings?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2746, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q077", + "question": "When was the last external audit of the lake monitoring programme?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2941, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q078", + "question": "What is the unit price of a river monitoring sediment sampler?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2789, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] } ] } diff --git a/docs/eval/baseline-v1.6.md b/docs/eval/baseline-v1.6.md index 65b9366..ffa7f59 100644 --- a/docs/eval/baseline-v1.6.md +++ b/docs/eval/baseline-v1.6.md @@ -12,7 +12,7 @@ Generated by `npm run eval`. The numbers below are harness output — do not edi | Ranks | `candidateK=20, threshold=0.5` | | Context width | `contextK=3` | | Corpus | `eval/corpus` (21 documents) | -| Split | `all` (63 questions, 53 answerable) | +| Split | `all` (78 questions, 53 answerable) | | Index size | 53 chunks | ## Metrics @@ -63,13 +63,13 @@ with a small second one means the threshold filters nothing and the window is al | Metric | Value | | --- | --- | -| Unanswerable questions | 10 | -| Retrieval abstained | 0.0000 (0/10) | +| Unanswerable questions | 25 | +| Retrieval abstained | 0.0000 (0/25) | | Mean candidates passing the threshold | 20.00 | | Mean passages in the context window | 3.00 | -Timing is informational only and is **not** frozen: indexing 2796 ms, query -p50 13.67 ms, p95 16.34 ms on the +Timing is informational only and is **not** frozen: indexing 5921 ms, query +p50 53.92 ms, p95 85.84 ms on the machine that produced this file. Timing and index size depend on hardware and on the corpus, so they must never be the reason two runs differ. diff --git a/docs/eval/scores-v1.6.json b/docs/eval/scores-v1.6.json index 163c779..4cfd04e 100644 --- a/docs/eval/scores-v1.6.json +++ b/docs/eval/scores-v1.6.json @@ -6,7 +6,7 @@ "indexSize": 53, "counts": { "answerable": 19, - "unanswerable": 5 + "unanswerable": 15 }, "distributions": { "bestRelevant": { @@ -47,12 +47,12 @@ }, "unanswerableMax": { "p0": 0.897848, - "p10": 0.897848, - "p25": 0.912751, - "p50": 0.925936, - "p75": 0.934276, + "p10": 0.908616, + "p25": 0.914179, + "p50": 0.927896, + "p75": 0.936526, "p90": 0.942441, - "p100": 0.942441 + "p100": 0.943454 } }, "byType": { @@ -186,7 +186,7 @@ "overlap": { "worstRelevantP10": 0.89056, "bestNonRelevantP90": 0.9386, - "unanswerableMaxP50": 0.925936, + "unanswerableMaxP50": 0.927896, "unanswerableMaxP90": 0.942441 }, "separable": false, @@ -308,7 +308,7 @@ "cosine": 0.8, "hitRate": 0.8947368421052632, "fullRecallRate": 0.8947368421052632, - "abstentionRate": 0.2 + "abstentionRate": 0.06666666666666667 }, { "threshold": 0.925, diff --git a/docs/eval/scores-v1.6.md b/docs/eval/scores-v1.6.md index 18d8928..6bff4b4 100644 --- a/docs/eval/scores-v1.6.md +++ b/docs/eval/scores-v1.6.md @@ -33,7 +33,7 @@ Dense only, `validation` split only, `threshold = 0`, `candidateK = 500` (above index size, so every chunk is scored for every query), `contextK = 3`. - answerable questions: 19 -- unanswerable questions: 5 +- unanswerable questions: 15 - index size: 53 chunks Hybrid is deliberately excluded: its `score` is an RRF value (`1 / (60 + rank)`) and is not @@ -61,7 +61,7 @@ Where a row shows a raw cosine in brackets: an **absolute** score maps as | worst relevant | 19 | 0.8808 (0.762) | 0.8906 (0.781) | 0.9086 (0.817) | 0.9355 (0.871) | 0.9428 (0.886) | 0.9530 (0.906) | 0.9537 (0.907) | | best non-relevant | 19 | 0.8904 (0.781) | 0.9081 (0.816) | 0.9178 (0.836) | 0.9243 (0.849) | 0.9336 (0.867) | 0.9386 (0.877) | 0.9459 (0.892) | | margin (best rel − best non-rel) | 19 | -0.0294 (-0.059) | -0.0272 (-0.054) | -0.0092 (-0.018) | 0.0026 (0.005) | 0.0081 (0.016) | 0.0329 (0.066) | 0.0623 (0.125) | -| unanswerable max candidate | 5 | 0.8978 (0.796) | 0.8978 (0.796) | 0.9128 (0.826) | 0.9259 (0.852) | 0.9343 (0.869) | 0.9424 (0.885) | 0.9424 (0.885) | +| unanswerable max candidate | 15 | 0.8978 (0.796) | 0.9086 (0.817) | 0.9142 (0.828) | 0.9279 (0.856) | 0.9365 (0.873) | 0.9424 (0.885) | 0.9435 (0.887) | ## By query type @@ -101,7 +101,7 @@ relevant passage; **full recall** = share whose every ground-truth block is stil | 0.825 | 0.650 | 1.0000 | 1.0000 | 0.0000 | | 0.850 | 0.700 | 1.0000 | 1.0000 | 0.0000 | | 0.875 | 0.750 | 1.0000 | 1.0000 | 0.0000 | -| 0.900 | 0.800 | 0.8947 | 0.8947 | 0.2000 | +| 0.900 | 0.800 | 0.8947 | 0.8947 | 0.0667 | | 0.925 | 0.850 | 0.6316 | 0.6316 | 0.4000 | | 0.950 | 0.900 | 0.1579 | 0.1579 | 1.0000 | | 0.975 | 0.950 | 0.0000 | 0.0000 | 1.0000 | @@ -115,7 +115,7 @@ The threshold decision turns on whether these overlap: | --- | --- | --- | | worst relevant, p10 | 0.8906 | 0.781 | | best non-relevant, p90 | 0.9386 | 0.877 | -| unanswerable max, p50 | 0.9259 | 0.852 | +| unanswerable max, p50 | 0.9279 | 0.856 | | unanswerable max, p90 | 0.9424 | 0.885 | **The distributions **overlap**, so a higher threshold buys abstention by giving up required relevant passages. If the curve above shows abstention rising only as full recall falls, then the honest conclusion is that **a single dense similarity threshold cannot carry both recall and abstention** — and the next mechanism to evaluate is not a finer threshold grid but a different signal (reranker score, top1−top2 margin, per-query thresholds, or claim-level answerability). diff --git a/docs/eval/sweep-v1.6.json b/docs/eval/sweep-v1.6.json index 0a47cf1..ba157c1 100644 --- a/docs/eval/sweep-v1.6.json +++ b/docs/eval/sweep-v1.6.json @@ -12,12 +12,12 @@ "contextRecall": 0.816038, "noResultRate": 0, "meanContextChars": 2311.811320754717, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 5, "unanswerableContextPassages": 3, "chunkCount": 53, - "latencyP95Ms": 12.3 + "latencyP95Ms": 16.18 }, { "strategy": "dense", @@ -30,12 +30,12 @@ "contextRecall": 0.877358, "noResultRate": 0, "meanContextChars": 4007.0377358490564, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 5, "unanswerableContextPassages": 5, "chunkCount": 53, - "latencyP95Ms": 12.37 + "latencyP95Ms": 86.91 }, { "strategy": "dense", @@ -48,12 +48,12 @@ "contextRecall": 0.816038, "noResultRate": 0, "meanContextChars": 2311.811320754717, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 10, "unanswerableContextPassages": 3, "chunkCount": 53, - "latencyP95Ms": 12.66 + "latencyP95Ms": 100.17 }, { "strategy": "dense", @@ -66,12 +66,12 @@ "contextRecall": 0.877358, "noResultRate": 0, "meanContextChars": 4007.0377358490564, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 10, "unanswerableContextPassages": 5, "chunkCount": 53, - "latencyP95Ms": 13.42 + "latencyP95Ms": 81.38 }, { "strategy": "dense", @@ -84,12 +84,12 @@ "contextRecall": 0.877358, "noResultRate": 0, "meanContextChars": 6658.698113207547, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 10, "unanswerableContextPassages": 8, "chunkCount": 53, - "latencyP95Ms": 12.8 + "latencyP95Ms": 18.17 }, { "strategy": "dense", @@ -102,12 +102,12 @@ "contextRecall": 0.816038, "noResultRate": 0, "meanContextChars": 2311.811320754717, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 20, "unanswerableContextPassages": 3, "chunkCount": 53, - "latencyP95Ms": 13.23 + "latencyP95Ms": 17.01 }, { "strategy": "dense", @@ -120,12 +120,12 @@ "contextRecall": 0.877358, "noResultRate": 0, "meanContextChars": 4007.0377358490564, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 20, "unanswerableContextPassages": 5, "chunkCount": 53, - "latencyP95Ms": 13.55 + "latencyP95Ms": 97.74 }, { "strategy": "dense", @@ -138,12 +138,12 @@ "contextRecall": 0.877358, "noResultRate": 0, "meanContextChars": 6658.698113207547, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 20, "unanswerableContextPassages": 8, "chunkCount": 53, - "latencyP95Ms": 13.75 + "latencyP95Ms": 85.3 }, { "strategy": "dense", @@ -156,12 +156,12 @@ "contextRecall": 0.816038, "noResultRate": 0, "meanContextChars": 2311.811320754717, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 40, "unanswerableContextPassages": 3, "chunkCount": 53, - "latencyP95Ms": 15.44 + "latencyP95Ms": 57.28 }, { "strategy": "dense", @@ -174,12 +174,12 @@ "contextRecall": 0.877358, "noResultRate": 0, "meanContextChars": 4007.0377358490564, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 40, "unanswerableContextPassages": 5, "chunkCount": 53, - "latencyP95Ms": 14.98 + "latencyP95Ms": 16.64 }, { "strategy": "dense", @@ -192,12 +192,12 @@ "contextRecall": 0.877358, "noResultRate": 0, "meanContextChars": 6658.698113207547, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 40, "unanswerableContextPassages": 8, "chunkCount": 53, - "latencyP95Ms": 14.68 + "latencyP95Ms": 99.32 }, { "strategy": "hybrid", @@ -210,12 +210,12 @@ "contextRecall": 0.853774, "noResultRate": 0, "meanContextChars": 2307.3207547169814, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 5, "unanswerableContextPassages": 3, "chunkCount": 53, - "latencyP95Ms": 13.45 + "latencyP95Ms": 90.4 }, { "strategy": "hybrid", @@ -228,12 +228,12 @@ "contextRecall": 0.910377, "noResultRate": 0, "meanContextChars": 3936.566037735849, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 5, "unanswerableContextPassages": 5, "chunkCount": 53, - "latencyP95Ms": 77.74 + "latencyP95Ms": 89.63 }, { "strategy": "hybrid", @@ -246,12 +246,12 @@ "contextRecall": 0.853774, "noResultRate": 0, "meanContextChars": 2321.264150943396, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 10, "unanswerableContextPassages": 3, "chunkCount": 53, - "latencyP95Ms": 14.42 + "latencyP95Ms": 83.59 }, { "strategy": "hybrid", @@ -263,13 +263,13 @@ "contextPrecision": 0.203774, "contextRecall": 0.896226, "noResultRate": 0, - "meanContextChars": 3994.509433962264, - "unanswerableQuestions": 10, + "meanContextChars": 3985.811320754717, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 10, "unanswerableContextPassages": 5, "chunkCount": 53, - "latencyP95Ms": 13.8 + "latencyP95Ms": 89.79 }, { "strategy": "hybrid", @@ -282,12 +282,12 @@ "contextRecall": 0.900943, "noResultRate": 0, "meanContextChars": 6547.641509433963, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 10, "unanswerableContextPassages": 8, "chunkCount": 53, - "latencyP95Ms": 13.04 + "latencyP95Ms": 14.96 }, { "strategy": "hybrid", @@ -300,12 +300,12 @@ "contextRecall": 0.853774, "noResultRate": 0, "meanContextChars": 2322.830188679245, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 20, "unanswerableContextPassages": 3, "chunkCount": 53, - "latencyP95Ms": 15.55 + "latencyP95Ms": 82.31 }, { "strategy": "hybrid", @@ -318,12 +318,12 @@ "contextRecall": 0.896226, "noResultRate": 0, "meanContextChars": 4011.377358490566, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 20, "unanswerableContextPassages": 5, "chunkCount": 53, - "latencyP95Ms": 16.36 + "latencyP95Ms": 18.62 }, { "strategy": "hybrid", @@ -336,12 +336,12 @@ "contextRecall": 0.90566, "noResultRate": 0, "meanContextChars": 6660.264150943396, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 20, "unanswerableContextPassages": 8, "chunkCount": 53, - "latencyP95Ms": 14.66 + "latencyP95Ms": 16.13 }, { "strategy": "hybrid", @@ -354,12 +354,12 @@ "contextRecall": 0.853774, "noResultRate": 0, "meanContextChars": 2322.830188679245, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 40, "unanswerableContextPassages": 3, "chunkCount": 53, - "latencyP95Ms": 14.55 + "latencyP95Ms": 97.73 }, { "strategy": "hybrid", @@ -372,12 +372,12 @@ "contextRecall": 0.896226, "noResultRate": 0, "meanContextChars": 3989.735849056604, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 40, "unanswerableContextPassages": 5, "chunkCount": 53, - "latencyP95Ms": 14.41 + "latencyP95Ms": 100.05 }, { "strategy": "hybrid", @@ -390,12 +390,12 @@ "contextRecall": 0.90566, "noResultRate": 0, "meanContextChars": 6660.792452830188, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 40, "unanswerableContextPassages": 8, "chunkCount": 53, - "latencyP95Ms": 14.47 + "latencyP95Ms": 16.94 } ] } diff --git a/docs/eval/sweep-v1.6.md b/docs/eval/sweep-v1.6.md index 46aeb3e..af8cb92 100644 --- a/docs/eval/sweep-v1.6.md +++ b/docs/eval/sweep-v1.6.md @@ -12,28 +12,28 @@ Each row differs from its neighbour in one parameter. | Strategy | candidateK | contextK | Recall@5 | nDCG@10 | MAP@10 | Context P | Context R | No-result | Context chars | Unans. abstained | Unans. cands | Unans. ctx | Index | p95 | | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | -| dense | 5 | 3 | 0.8774 | 0.7672 | 0.7231 | 0.3082 | 0.8160 | 0.0000 | 2312 | 0.0000 | 5.0 | 3.0 | 53 | 12.30 ms | -| dense | 5 | 5 | 0.8774 | 0.7672 | 0.7231 | 0.2000 | 0.8774 | 0.0000 | 4007 | 0.0000 | 5.0 | 5.0 | 53 | 12.37 ms | -| dense | 10 | 3 | 0.8774 | 0.7804 | 0.7280 | 0.3082 | 0.8160 | 0.0000 | 2312 | 0.0000 | 10.0 | 3.0 | 53 | 12.66 ms | -| dense | 10 | 5 | 0.8774 | 0.7804 | 0.7280 | 0.2000 | 0.8774 | 0.0000 | 4007 | 0.0000 | 10.0 | 5.0 | 53 | 13.42 ms | -| dense | 10 | 8 | 0.8774 | 0.7804 | 0.7280 | 0.1274 | 0.8774 | 0.0000 | 6659 | 0.0000 | 10.0 | 8.0 | 53 | 12.80 ms | -| dense | 20 | 3 | 0.8774 | 0.7804 | 0.7280 | 0.3082 | 0.8160 | 0.0000 | 2312 | 0.0000 | 20.0 | 3.0 | 53 | 13.23 ms | -| dense | 20 | 5 | 0.8774 | 0.7804 | 0.7280 | 0.2000 | 0.8774 | 0.0000 | 4007 | 0.0000 | 20.0 | 5.0 | 53 | 13.55 ms | -| dense | 20 | 8 | 0.8774 | 0.7804 | 0.7280 | 0.1274 | 0.8774 | 0.0000 | 6659 | 0.0000 | 20.0 | 8.0 | 53 | 13.75 ms | -| dense | 40 | 3 | 0.8774 | 0.7804 | 0.7280 | 0.3082 | 0.8160 | 0.0000 | 2312 | 0.0000 | 40.0 | 3.0 | 53 | 15.44 ms | -| dense | 40 | 5 | 0.8774 | 0.7804 | 0.7280 | 0.2000 | 0.8774 | 0.0000 | 4007 | 0.0000 | 40.0 | 5.0 | 53 | 14.98 ms | -| dense | 40 | 8 | 0.8774 | 0.7804 | 0.7280 | 0.1274 | 0.8774 | 0.0000 | 6659 | 0.0000 | 40.0 | 8.0 | 53 | 14.68 ms | -| hybrid | 5 | 3 | 0.9104 | 0.8101 | 0.7682 | 0.3208 | 0.8538 | 0.0000 | 2307 | 0.0000 | 5.0 | 3.0 | 53 | 13.45 ms | -| hybrid | 5 | 5 | 0.9104 | 0.8101 | 0.7682 | 0.2113 | 0.9104 | 0.0000 | 3937 | 0.0000 | 5.0 | 5.0 | 53 | 77.74 ms | -| hybrid | 10 | 3 | 0.8962 | 0.8068 | 0.7580 | 0.3208 | 0.8538 | 0.0000 | 2321 | 0.0000 | 10.0 | 3.0 | 53 | 14.42 ms | -| hybrid | 10 | 5 | 0.8962 | 0.8068 | 0.7580 | 0.2038 | 0.8962 | 0.0000 | 3995 | 0.0000 | 10.0 | 5.0 | 53 | 13.80 ms | -| hybrid | 10 | 8 | 0.8962 | 0.8068 | 0.7580 | 0.1321 | 0.9009 | 0.0000 | 6548 | 0.0000 | 10.0 | 8.0 | 53 | 13.04 ms | -| hybrid | 20 | 3 | 0.8962 | 0.8098 | 0.7605 | 0.3208 | 0.8538 | 0.0000 | 2323 | 0.0000 | 20.0 | 3.0 | 53 | 15.55 ms | -| hybrid | 20 | 5 | 0.8962 | 0.8098 | 0.7605 | 0.2038 | 0.8962 | 0.0000 | 4011 | 0.0000 | 20.0 | 5.0 | 53 | 16.36 ms | -| hybrid | 20 | 8 | 0.8962 | 0.8098 | 0.7605 | 0.1321 | 0.9057 | 0.0000 | 6660 | 0.0000 | 20.0 | 8.0 | 53 | 14.66 ms | -| hybrid | 40 | 3 | 0.8962 | 0.8098 | 0.7605 | 0.3208 | 0.8538 | 0.0000 | 2323 | 0.0000 | 40.0 | 3.0 | 53 | 14.55 ms | -| hybrid | 40 | 5 | 0.8962 | 0.8098 | 0.7605 | 0.2038 | 0.8962 | 0.0000 | 3990 | 0.0000 | 40.0 | 5.0 | 53 | 14.41 ms | -| hybrid | 40 | 8 | 0.8962 | 0.8098 | 0.7605 | 0.1321 | 0.9057 | 0.0000 | 6661 | 0.0000 | 40.0 | 8.0 | 53 | 14.47 ms | +| dense | 5 | 3 | 0.8774 | 0.7672 | 0.7231 | 0.3082 | 0.8160 | 0.0000 | 2312 | 0.0000 | 5.0 | 3.0 | 53 | 16.18 ms | +| dense | 5 | 5 | 0.8774 | 0.7672 | 0.7231 | 0.2000 | 0.8774 | 0.0000 | 4007 | 0.0000 | 5.0 | 5.0 | 53 | 86.91 ms | +| dense | 10 | 3 | 0.8774 | 0.7804 | 0.7280 | 0.3082 | 0.8160 | 0.0000 | 2312 | 0.0000 | 10.0 | 3.0 | 53 | 100.17 ms | +| dense | 10 | 5 | 0.8774 | 0.7804 | 0.7280 | 0.2000 | 0.8774 | 0.0000 | 4007 | 0.0000 | 10.0 | 5.0 | 53 | 81.38 ms | +| dense | 10 | 8 | 0.8774 | 0.7804 | 0.7280 | 0.1274 | 0.8774 | 0.0000 | 6659 | 0.0000 | 10.0 | 8.0 | 53 | 18.17 ms | +| dense | 20 | 3 | 0.8774 | 0.7804 | 0.7280 | 0.3082 | 0.8160 | 0.0000 | 2312 | 0.0000 | 20.0 | 3.0 | 53 | 17.01 ms | +| dense | 20 | 5 | 0.8774 | 0.7804 | 0.7280 | 0.2000 | 0.8774 | 0.0000 | 4007 | 0.0000 | 20.0 | 5.0 | 53 | 97.74 ms | +| dense | 20 | 8 | 0.8774 | 0.7804 | 0.7280 | 0.1274 | 0.8774 | 0.0000 | 6659 | 0.0000 | 20.0 | 8.0 | 53 | 85.30 ms | +| dense | 40 | 3 | 0.8774 | 0.7804 | 0.7280 | 0.3082 | 0.8160 | 0.0000 | 2312 | 0.0000 | 40.0 | 3.0 | 53 | 57.28 ms | +| dense | 40 | 5 | 0.8774 | 0.7804 | 0.7280 | 0.2000 | 0.8774 | 0.0000 | 4007 | 0.0000 | 40.0 | 5.0 | 53 | 16.64 ms | +| dense | 40 | 8 | 0.8774 | 0.7804 | 0.7280 | 0.1274 | 0.8774 | 0.0000 | 6659 | 0.0000 | 40.0 | 8.0 | 53 | 99.32 ms | +| hybrid | 5 | 3 | 0.9104 | 0.8101 | 0.7682 | 0.3208 | 0.8538 | 0.0000 | 2307 | 0.0000 | 5.0 | 3.0 | 53 | 90.40 ms | +| hybrid | 5 | 5 | 0.9104 | 0.8101 | 0.7682 | 0.2113 | 0.9104 | 0.0000 | 3937 | 0.0000 | 5.0 | 5.0 | 53 | 89.63 ms | +| hybrid | 10 | 3 | 0.8962 | 0.8068 | 0.7580 | 0.3208 | 0.8538 | 0.0000 | 2321 | 0.0000 | 10.0 | 3.0 | 53 | 83.59 ms | +| hybrid | 10 | 5 | 0.8962 | 0.8068 | 0.7580 | 0.2038 | 0.8962 | 0.0000 | 3986 | 0.0000 | 10.0 | 5.0 | 53 | 89.79 ms | +| hybrid | 10 | 8 | 0.8962 | 0.8068 | 0.7580 | 0.1321 | 0.9009 | 0.0000 | 6548 | 0.0000 | 10.0 | 8.0 | 53 | 14.96 ms | +| hybrid | 20 | 3 | 0.8962 | 0.8098 | 0.7605 | 0.3208 | 0.8538 | 0.0000 | 2323 | 0.0000 | 20.0 | 3.0 | 53 | 82.31 ms | +| hybrid | 20 | 5 | 0.8962 | 0.8098 | 0.7605 | 0.2038 | 0.8962 | 0.0000 | 4011 | 0.0000 | 20.0 | 5.0 | 53 | 18.62 ms | +| hybrid | 20 | 8 | 0.8962 | 0.8098 | 0.7605 | 0.1321 | 0.9057 | 0.0000 | 6660 | 0.0000 | 20.0 | 8.0 | 53 | 16.13 ms | +| hybrid | 40 | 3 | 0.8962 | 0.8098 | 0.7605 | 0.3208 | 0.8538 | 0.0000 | 2323 | 0.0000 | 40.0 | 3.0 | 53 | 97.73 ms | +| hybrid | 40 | 5 | 0.8962 | 0.8098 | 0.7605 | 0.2038 | 0.8962 | 0.0000 | 3990 | 0.0000 | 40.0 | 5.0 | 53 | 100.05 ms | +| hybrid | 40 | 8 | 0.8962 | 0.8098 | 0.7605 | 0.1321 | 0.9057 | 0.0000 | 6661 | 0.0000 | 40.0 | 8.0 | 53 | 16.94 ms | ## Skipped cells diff --git a/docs/eval/threshold-v1.6.json b/docs/eval/threshold-v1.6.json index 15d8562..fbf9e89 100644 --- a/docs/eval/threshold-v1.6.json +++ b/docs/eval/threshold-v1.6.json @@ -7,13 +7,13 @@ { "threshold": 0, "validation": { - "questions": 24, + "questions": 34, "answerableQuestions": 19, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 20, "unanswerable": { - "questions": 5, + "questions": 15, "abstentionCount": 0, "retrievalAbstentionRate": 0, "meanCandidatesRetrieved": 20, @@ -32,13 +32,13 @@ } }, "test": { - "questions": 39, + "questions": 44, "answerableQuestions": 34, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 20, "unanswerable": { - "questions": 5, + "questions": 10, "abstentionCount": 0, "retrievalAbstentionRate": 0, "meanCandidatesRetrieved": 20, @@ -60,13 +60,13 @@ { "threshold": 0.3, "validation": { - "questions": 24, + "questions": 34, "answerableQuestions": 19, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 20, "unanswerable": { - "questions": 5, + "questions": 15, "abstentionCount": 0, "retrievalAbstentionRate": 0, "meanCandidatesRetrieved": 20, @@ -85,13 +85,13 @@ } }, "test": { - "questions": 39, + "questions": 44, "answerableQuestions": 34, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 20, "unanswerable": { - "questions": 5, + "questions": 10, "abstentionCount": 0, "retrievalAbstentionRate": 0, "meanCandidatesRetrieved": 20, @@ -113,13 +113,13 @@ { "threshold": 0.4, "validation": { - "questions": 24, + "questions": 34, "answerableQuestions": 19, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 20, "unanswerable": { - "questions": 5, + "questions": 15, "abstentionCount": 0, "retrievalAbstentionRate": 0, "meanCandidatesRetrieved": 20, @@ -138,13 +138,13 @@ } }, "test": { - "questions": 39, + "questions": 44, "answerableQuestions": 34, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 20, "unanswerable": { - "questions": 5, + "questions": 10, "abstentionCount": 0, "retrievalAbstentionRate": 0, "meanCandidatesRetrieved": 20, @@ -166,13 +166,13 @@ { "threshold": 0.5, "validation": { - "questions": 24, + "questions": 34, "answerableQuestions": 19, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 20, "unanswerable": { - "questions": 5, + "questions": 15, "abstentionCount": 0, "retrievalAbstentionRate": 0, "meanCandidatesRetrieved": 20, @@ -191,13 +191,13 @@ } }, "test": { - "questions": 39, + "questions": 44, "answerableQuestions": 34, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 20, "unanswerable": { - "questions": 5, + "questions": 10, "abstentionCount": 0, "retrievalAbstentionRate": 0, "meanCandidatesRetrieved": 20, @@ -219,13 +219,13 @@ { "threshold": 0.6, "validation": { - "questions": 24, + "questions": 34, "answerableQuestions": 19, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 20, "unanswerable": { - "questions": 5, + "questions": 15, "abstentionCount": 0, "retrievalAbstentionRate": 0, "meanCandidatesRetrieved": 20, @@ -244,13 +244,13 @@ } }, "test": { - "questions": 39, + "questions": 44, "answerableQuestions": 34, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 20, "unanswerable": { - "questions": 5, + "questions": 10, "abstentionCount": 0, "retrievalAbstentionRate": 0, "meanCandidatesRetrieved": 20, diff --git a/docs/eval/threshold-v1.6.md b/docs/eval/threshold-v1.6.md index 3997609..6dcc74e 100644 --- a/docs/eval/threshold-v1.6.md +++ b/docs/eval/threshold-v1.6.md @@ -34,7 +34,7 @@ retrieval loss. ## Outcome -The sweep is **flat**: every threshold from 0 to 0.6 produces the same validation nDCG@10 (0.7556), the same Recall@5 (0.8684) and the same retrieval abstention rate on unanswerable questions (0/5). No candidate is ever filtered out, so the threshold is **non-binding** on this corpus. +The sweep is **flat**: every threshold from 0 to 0.6 produces the same validation nDCG@10 (0.7556), the same Recall@5 (0.8684) and the same retrieval abstention rate on unanswerable questions (0/15). No candidate is ever filtered out, so the threshold is **non-binding** on this corpus. **No evidence to change `threshold = 0.5`.** All thresholds hold the line equally; picking one would be arbitrary. The current value can be neither validated nor falsified here, which is a property of the corpus rather than of the threshold. Note also what this does *not* establish: abstention is a retrieval-layer statement — whether the model then declines to answer needs a generator eval. diff --git a/eval/questions.jsonl b/eval/questions.jsonl index 10b031a..44fa762 100644 --- a/eval/questions.jsonl +++ b/eval/questions.jsonl @@ -61,3 +61,18 @@ {"id":"q061","type":"semantic","question":"Why does the estuary programme take more replicate samples than the other monitoring programmes?","relevant":[{"document":"estuary-monitoring.md","page":null,"block":8,"quote":"within-station variance"}]} {"id":"q062","type":"unanswerable","answerable":false,"question":"What is the annual operating cost of the coastal monitoring buoys?","relevant":[]} {"id":"q063","type":"unanswerable","answerable":false,"question":"How many litres per second does the reservoir release downstream on a typical day?","relevant":[]} +{"id":"q064","question":"How much GPU memory does Gateway API v3 need to serve a request?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q065","question":"What is the uptime SLA for Gateway API v2?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q066","question":"How long does an authentication token stay valid before it expires?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q067","question":"Which SDK version should a client use with Gateway API v3?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q068","question":"Can Gateway API v2 be self-hosted on-premise?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q069","question":"How is usage invoiced for Gateway API v2?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q070","question":"Which laboratory is accredited to analyse the estuary transect samples?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q071","question":"Which vendor supplies the coastal monitoring buoys?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q072","question":"What is the annual funding for the groundwater monitoring programme?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q073","question":"How many staff work on the reservoir monitoring programme?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q074","question":"What is the warranty period on the estuary monitoring sensors?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q075","question":"How long are the coastal monitoring readings retained before deletion?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q076","question":"What encryption standard is used to transmit the river monitoring readings?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q077","question":"When was the last external audit of the lake monitoring programme?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q078","question":"What is the unit price of a river monitoring sediment sampler?","type":"unanswerable","answerable":false,"relevant":[]} diff --git a/eval/splits.json b/eval/splits.json index 021e265..6e13b15 100644 --- a/eval/splits.json +++ b/eval/splits.json @@ -61,5 +61,20 @@ "q060": "validation", "q061": "test", "q062": "validation", - "q063": "test" + "q063": "test", + "q064": "validation", + "q065": "validation", + "q066": "validation", + "q067": "test", + "q068": "validation", + "q069": "test", + "q070": "validation", + "q071": "validation", + "q072": "test", + "q073": "validation", + "q074": "test", + "q075": "validation", + "q076": "validation", + "q077": "test", + "q078": "validation" }