diff --git a/.github/workflows/verify.yml b/.github/workflows/verify.yml index 031edae..70c1f6a 100644 --- a/.github/workflows/verify.yml +++ b/.github/workflows/verify.yml @@ -110,6 +110,6 @@ jobs: run: | xvfb-run -a npm run eval:prepare xvfb-run -a node scripts/eval.mjs - cp docs/eval/baseline-v1.4.json /tmp/eval-a.json + cp docs/eval/baseline-v1.5.json /tmp/eval-a.json xvfb-run -a node scripts/eval.mjs - diff /tmp/eval-a.json docs/eval/baseline-v1.4.json + diff /tmp/eval-a.json docs/eval/baseline-v1.5.json diff --git a/.prettierignore b/.prettierignore index bb7d473..06feaef 100644 --- a/.prettierignore +++ b/.prettierignore @@ -13,3 +13,8 @@ src/main/db/migrations/meta # `npm run eval`; the baseline's readability is not the formatter's business. docs/eval/baseline-*.json docs/eval/baseline-*.md + +# Same reason as the baseline: generated by `scripts/eval-chunking.mjs`. Regenerate +# with `npm run eval:chunking`; the formatter must not be a second writer. +docs/eval/chunking-*.json +docs/eval/chunking-*.md diff --git a/docs/eval/baseline-v1.5.json b/docs/eval/baseline-v1.5.json new file mode 100644 index 0000000..b214bfb --- /dev/null +++ b/docs/eval/baseline-v1.5.json @@ -0,0 +1,666 @@ +{ + "baseline": "v1.5", + "generatedBy": "npm run eval", + "config": { + "embedding": "Xenova/multilingual-e5-small@761b726dd34fb83930e26aab4e9ac3899aa1fa78 q8 (384d, local)", + "chunking": { + "chunkSize": 1000, + "chunkOverlap": 100, + "minChunkSize": 100, + "allowSpanPages": false, + "respectHeadings": false + }, + "retrieval": "dense", + "topK": 10, + "threshold": 0, + "evidenceK": 5, + "corpus": "eval/corpus", + "documents": 13, + "questions": 30, + "chunkCount": 19 + }, + "metrics": { + "recallAt1": 0.833333, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 0.927778, + "ndcgAt10": 0.94375, + "evidencePrecisionAt5": 0.213333 + }, + "perQuestion": [ + { + "id": "q001", + "question": "Why is bedload harder to measure than suspended sediment?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q002", + "question": "How many replicate samples are collected at each river station?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q003", + "question": "What is the central trade-off in lithium-ion cell design?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q004", + "question": "Why do nickel-rich battery packs need more aggressive thermal management?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q005", + "question": "What happens once the separator in a battery cell melts?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q006", + "question": "At what temperature do honeybees begin to forage?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q007", + "question": "What does a late frost damage during full bloom?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q008", + "question": "Why is a continuous tree canopy more effective at cooling than isolated trees?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q009", + "question": "Why are trees with aggressive surface roots unsuitable for narrow verges?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q010", + "question": "At what temperature is lactic acid fermentation fastest?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q011", + "question": "Is the salt percentage in fermentation based on vegetable weight or water weight?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q012", + "question": "Why must tidal turbines be sited in places with very fast currents?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q013", + "question": "What is the main environmental concern for tidal energy installations?", + "firstRelevantRank": 3, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [], + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q014", + "question": "In lake monitoring, how is the sampling depth actually recorded?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q015", + "question": "Why does deep-water oxygen fall while a lake remains stratified?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q016", + "question": "How do supercapacitors hold their charge?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q017", + "question": "Why can solitary bees pollinate a bloom week that is too cold for honeybee hives?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q018", + "question": "Why is one continuous planted roof layer better than several isolated planted beds?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q019", + "question": "Where do acetic acid bacteria sit in a vinegar culture?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q020", + "question": "What happens if a vinegar culture is sealed airtight?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q021", + "question": "Why is wave energy harder to schedule ahead than tidal energy?", + "firstRelevantRank": 2, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q022", + "question": "Where does siting for wave energy devices concentrate, and where does it not?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q023", + "question": "How does the river sampling protocol differ from the lake sampling protocol?", + "firstRelevantRank": 1, + "relevantCount": 2, + "retrievedCount": 10, + "matchesByRank": [ + [ + 1 + ], + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q024", + "question": "A street canopy and a green roof are both said to cool; what surface does each one shade?", + "firstRelevantRank": 1, + "relevantCount": 2, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [ + 1 + ], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q025", + "question": "Which preservation method depends on keeping air away from the food?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q026", + "question": "Why can one cold morning cost a grower the whole crop even when colonies are brought in?", + "firstRelevantRank": 2, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q027", + "question": "绿茶应该怎样保存才能减缓氧化?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q028", + "question": "茶叶储存的相对湿度上限是多少?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q029", + "question": "为什么冷冻保存的茶叶取出后不能立刻打开包装?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q030", + "question": "为什么潮汐能比风能和太阳能更容易提前安排发电?", + "firstRelevantRank": 2, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [] + ] + } + ] +} diff --git a/docs/eval/baseline-v1.5.md b/docs/eval/baseline-v1.5.md new file mode 100644 index 0000000..797a698 --- /dev/null +++ b/docs/eval/baseline-v1.5.md @@ -0,0 +1,61 @@ +# RAG eval baseline — v1.5 + +Generated by `npm run eval`. The numbers below are harness output — do not edit them by hand. + +## Configuration + +| Setting | Value | +| --- | --- | +| Embedding | `Xenova/multilingual-e5-small@761b726dd34fb83930e26aab4e9ac3899aa1fa78 q8 (384d, local)` | +| Chunking | `chunkSize=1000, chunkOverlap=100, minChunkSize=100, allowSpanPages=false, respectHeadings=false` | +| Retrieval | `dense` | +| Ranks | `topK=10, threshold=0` | +| Evidence per query | `evidenceK=5` | +| Corpus | `eval/corpus` (13 documents, 30 questions) | +| Index size | 19 chunks | + +## Metrics + +| Metric | Value | +| --- | --- | +| Recall@1 | 0.8333 | +| Recall@5 | 1.0000 | +| Recall@10 | 1.0000 | +| MRR | 0.9278 | +| nDCG@10 | 0.9437 | +| Evidence precision@5 | 0.2133 | + +Timing is informational only and is **not** frozen: indexing 1941 ms, query +p50 14.39 ms, p95 27.71 ms on the +machine that produced this file. Timing and index size depend on hardware and on the +corpus, so they must never be the reason two runs differ. + +## Definitions + +- A retrieved passage is relevant when its provenance covers a ground-truth block. +- **Recall@k** is the share of ground-truth blocks covered by the first `k` passages. +- **Evidence precision@5** is the share of the first `5` + retrieved passages that cover a ground-truth block. This is **retrieval precision**, not + answer citation recall: the harness runs no model and produces no answer. Answer-level + citation correctness is covered by the resolver (#70); a model-driven answer eval is a + separate deliverable. +- Ground truth is expressed in corpus identity (`document` relative path + `block` + ordinal + optional `quote`), never a runtime `documentId`/`blockId`. + +## Comparison protocol + +v1.5 experiments (#77, #78) are reported as a **delta against this file**. The +adopted-change rule is: + +> Adopt a change only if Recall@5 improves and nDCG@10 does not regress. A change +> that trades a large latency increase for a marginal recall gain is a product +> decision, not an automatic win, and must be stated as such. + +A changed result must be reproducible with: + +```bash +npm run eval:prepare # one-time, networked model bootstrap +npm run eval # offline and deterministic +``` + +The raw report is committed next to this summary as `baseline-v1.5.json`. diff --git a/docs/eval/chunking-v1.5.json b/docs/eval/chunking-v1.5.json new file mode 100644 index 0000000..5c37a66 --- /dev/null +++ b/docs/eval/chunking-v1.5.json @@ -0,0 +1,107 @@ +{ + "baseline": "v1.4", + "variants": [ + { + "id": "baseline", + "label": "baseline v1.4 (500/50)", + "chunkSize": 500, + "chunkOverlap": 50, + "allowSpanPages": false, + "respectHeadings": false, + "chunkCount": 36, + "recallAt1": 0.766667, + "recallAt5": 0.933333, + "recallAt10": 1, + "mrr": 0.849206, + "ndcgAt10": 0.889891, + "evidencePrecisionAt5": 0.2, + "indexingMs": 2041, + "latencyP95Ms": 22.45 + }, + { + "id": "small", + "label": "small (250/25)", + "chunkSize": 250, + "chunkOverlap": 25, + "allowSpanPages": false, + "respectHeadings": false, + "chunkCount": 70, + "recallAt1": 0.733333, + "recallAt5": 0.983333, + "recallAt10": 1, + "mrr": 0.837222, + "ndcgAt10": 0.878708, + "evidencePrecisionAt5": 0.306667, + "indexingMs": 2056, + "latencyP95Ms": 22.56 + }, + { + "id": "large", + "label": "large (1000/100)", + "chunkSize": 1000, + "chunkOverlap": 100, + "allowSpanPages": false, + "respectHeadings": false, + "chunkCount": 19, + "recallAt1": 0.833333, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 0.927778, + "ndcgAt10": 0.94375, + "evidencePrecisionAt5": 0.213333, + "indexingMs": 1955, + "latencyP95Ms": 27.09 + }, + { + "id": "span-pages", + "label": "large + page spanning (1000/100)", + "chunkSize": 1000, + "chunkOverlap": 100, + "allowSpanPages": true, + "respectHeadings": false, + "chunkCount": 19, + "recallAt1": 0.833333, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 0.927778, + "ndcgAt10": 0.94375, + "evidencePrecisionAt5": 0.213333, + "indexingMs": 1993, + "latencyP95Ms": 29.67 + }, + { + "id": "headings", + "label": "heading-aware (500/50)", + "chunkSize": 500, + "chunkOverlap": 50, + "allowSpanPages": false, + "respectHeadings": true, + "chunkCount": 52, + "recallAt1": 0.766667, + "recallAt5": 0.9, + "recallAt10": 1, + "mrr": 0.831746, + "ndcgAt10": 0.875607, + "evidencePrecisionAt5": 0.193333, + "indexingMs": 1999, + "latencyP95Ms": 21.64 + }, + { + "id": "headings-large", + "label": "heading-aware large (1000/100)", + "chunkSize": 1000, + "chunkOverlap": 100, + "allowSpanPages": false, + "respectHeadings": true, + "chunkCount": 52, + "recallAt1": 0.766667, + "recallAt5": 0.9, + "recallAt10": 1, + "mrr": 0.831746, + "ndcgAt10": 0.875607, + "evidencePrecisionAt5": 0.193333, + "indexingMs": 2018, + "latencyP95Ms": 22.57 + } + ] +} diff --git a/docs/eval/chunking-v1.5.md b/docs/eval/chunking-v1.5.md new file mode 100644 index 0000000..84830b4 --- /dev/null +++ b/docs/eval/chunking-v1.5.md @@ -0,0 +1,40 @@ +# Chunking experiments — v1.5 (#78) + +Generated by `node scripts/eval-chunking.mjs`. Numbers are harness output; do not edit them by hand. + +## What was measured + +Every variant runs the real RAG eval harness against the same corpus and the same 30 +questions as `baseline-v1.4.json`, with dense retrieval held fixed. Only the chunking +configuration changes, so a difference in the metrics is a difference in the input +distribution retrieval is measured on. + +| Variant | size/overlap | Recall@1 | Recall@5 | MRR | nDCG@10 | Evidence P@5 | Index size (Δ) | Indexing | Query p95 | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | +| baseline v1.4 (500/50) | `500/50` | 0.7667 | 0.9333 | 0.8492 | 0.8899 | 0.2000 | 36 (+0.0%) | 2041 ms | 22.45 ms | +| small (250/25) | `250/25` | 0.7333 | 0.9833 | 0.8372 | 0.8787 | 0.3067 | 70 (+94.4%) | 2056 ms | 22.56 ms | +| large (1000/100) | `1000/100` | 0.8333 | 1.0000 | 0.9278 | 0.9437 | 0.2133 | 19 (-47.2%) | 1955 ms | 27.09 ms | +| large + page spanning (1000/100) | `1000/100 + span` | 0.8333 | 1.0000 | 0.9278 | 0.9437 | 0.2133 | 19 (-47.2%) | 1993 ms | 29.67 ms | +| heading-aware (500/50) | `500/50 + headings` | 0.7667 | 0.9000 | 0.8317 | 0.8756 | 0.1933 | 52 (+44.4%) | 1999 ms | 21.64 ms | +| heading-aware large (1000/100) | `1000/100 + headings` | 0.7667 | 0.9000 | 0.8317 | 0.8756 | 0.1933 | 52 (+44.4%) | 2018 ms | 22.57 ms | + +Timing depends on hardware and is informational, exactly as in the frozen baseline. + +## Adoption rule + +> Adopt a change only if Recall@5 improves and nDCG@10 does not regress. A change +> that trades a large latency or index-size increase for a marginal recall gain is a +> product decision, not an automatic win. + +## Outcome + +**Adopted: `large (1000/100)`.** It clears the rule (Recall@5 1.0000 vs baseline 0.9333, nDCG@10 0.9437 vs 0.8899), improves Recall@1 and MRR as well, and *shrinks* the index (19 vs 36 chunks). `DEFAULT_CHUNK_OPTIONS` is now `chunkSize=1000, chunkOverlap=100`, and the frozen successor baseline is `docs/eval/baseline-v1.5.json`. + +**Limitation to state plainly:** the corpus is small (13 short documents, 19–36 chunks), so Recall@5 saturates near 1.0 and is the least discriminating metric here; Recall@1 and MRR carry the result. The adoption should be re-checked on a larger, multi-format corpus before it is treated as settled. + +## Reproduce + +```bash +npm run eval:prepare # one-time, networked model bootstrap +npm run eval:chunking # offline; runs every variant and rewrites this file +``` diff --git a/package.json b/package.json index 64c9f9f..d5ebbca 100644 --- a/package.json +++ b/package.json @@ -31,6 +31,7 @@ "smoke:packaged": "node scripts/smoke-packaged.mjs", "eval:prepare": "npm run build && node scripts/eval.mjs --prepare", "eval": "npm run build && node scripts/eval.mjs", + "eval:chunking": "npm run build && node scripts/eval-chunking.mjs", "build:win": "npm run build && electron-builder --win", "build:mac": "npm run build && electron-builder --mac", "build:linux": "npm run build && electron-builder --linux", diff --git a/scripts/eval-chunking.mjs b/scripts/eval-chunking.mjs new file mode 100644 index 0000000..ecc0d4a --- /dev/null +++ b/scripts/eval-chunking.mjs @@ -0,0 +1,257 @@ +#!/usr/bin/env node +/** + * Chunking experiments for #78. + * + * Runs the real RAG eval harness once per chunking variant, against the same + * corpus and questions as the frozen v1.4 baseline, and writes the comparison the + * issue asks for: retrieval metrics plus index size and indexing latency, as a + * delta against the baseline variant. + * + * The harness itself does the measuring; this script only orchestrates and + * tabulates. It deliberately does not invent metrics. + * + * Usage: + * node scripts/eval-chunking.mjs + * node scripts/eval-chunking.mjs --out=docs/eval/chunking-v1.5.md + * + * The embedding model must already be prepared (`npm run eval:prepare`). + */ + +import { spawn } from 'node:child_process' +import { existsSync, mkdtempSync, mkdirSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join, resolve } from 'node:path' + +const OUT_MD = resolve(readArg('--out=', 'docs/eval/chunking-v1.5.md')) +const OUT_JSON = OUT_MD.replace(/\.md$/, '.json') + +/** + * The variants under test. + * + * `baseline` is the shipped default; every delta is computed against it. The set is + * deliberately small: size down, size up, page-spanning, heading-aware, and the + * two heading-aware/large combinations that the first two suggest. + */ +const VARIANTS = [ + // The frozen v1.4 configuration, stated explicitly so it stays 500/50 even after + // the default changed (#78 adoption). + { id: 'baseline', label: 'baseline v1.4 (500/50)', args: { size: 500, overlap: 50 } }, + { id: 'small', label: 'small (250/25)', args: { size: 250, overlap: 25 } }, + { id: 'large', label: 'large (1000/100)', args: { size: 1000, overlap: 100 } }, + { + id: 'span-pages', + label: 'large + page spanning (1000/100)', + args: { size: 1000, overlap: 100, spanPages: true } + }, + { + id: 'headings', + label: 'heading-aware (500/50)', + args: { size: 500, overlap: 50, headings: true } + }, + { + id: 'headings-large', + label: 'heading-aware large (1000/100)', + args: { size: 1000, overlap: 100, headings: true } + } +] + +/** The variant this spike adopted as the shipped default, or null if none. */ +const ADOPTED_ID = 'large' + +function readArg(prefix, fallback) { + const arg = process.argv.find((value) => value.startsWith(prefix)) + return arg ? arg.slice(prefix.length) : fallback +} + +const executable = resolve( + 'node_modules/.bin', + process.platform === 'win32' ? 'electron.cmd' : 'electron' +) + +if (!existsSync(executable)) { + console.error('[chunking] could not find the electron binary. Run `npm install` first.') + process.exit(1) +} + +function chunkFlags(args) { + const flags = [] + if (args.size !== undefined) flags.push(`--eval-chunk-size=${args.size}`) + if (args.overlap !== undefined) flags.push(`--eval-chunk-overlap=${args.overlap}`) + if (args.min !== undefined) flags.push(`--eval-chunk-min=${args.min}`) + if (args.spanPages !== undefined) flags.push(`--eval-allow-span-pages=${args.spanPages}`) + if (args.headings !== undefined) flags.push(`--eval-respect-headings=${args.headings}`) + return flags +} + +/** Run one variant into its own output dir; resolves when the process exits. */ +function runVariant(variant, outDir) { + return new Promise((resolvePromise, reject) => { + const args = [ + '.', + '--eval-harness', + '--eval-baseline=v1.4', + `--eval-out=${outDir}`, + ...chunkFlags(variant.args) + ] + + // Containers and CI runners lack the Chromium sandbox helpers; the harness + // never renders, so running unsandboxed is safe there. + const isRoot = typeof process.getuid === 'function' && process.getuid() === 0 + if (isRoot || process.env.CI) args.push('--no-sandbox') + + const child = spawn(executable, args, { + stdio: ['ignore', 'pipe', 'pipe'], + env: { ...process.env, ELECTRON_DISABLE_SECURITY_WARNINGS: '1' } + }) + + let stdout = '' + child.stdout.on('data', (data) => { + stdout += data.toString() + }) + child.stderr.on('data', () => {}) + + child.on('error', reject) + child.on('exit', (code) => { + if (code !== 0) { + reject(new Error(`variant ${variant.id} exited with code ${code}`)) + return + } + const metricsLine = /\[eval\] metrics (\{.*\})/.exec(stdout) + if (!metricsLine) { + reject(new Error(`variant ${variant.id} printed no metrics line`)) + return + } + try { + resolvePromise(JSON.parse(metricsLine[1])) + } catch (error) { + reject( + new Error(`variant ${variant.id} printed an unreadable metrics line: ${error.message}`) + ) + } + }) + }) +} + +/** + * Read the indexing time and query p95 out of the rendered report. + * + * Timing is deliberately excluded from the committed deterministic JSON (a PR must + * be able to diff it), so the markdown is the only place the harness reports it. + */ +function readTiming(mdPath) { + if (!existsSync(mdPath)) return { indexingMs: null, latencyP95Ms: null } + const text = readFileSync(mdPath, 'utf8') + const indexing = /indexing (\d+) ms/.exec(text) + const p95 = /p95 ([\d.]+) ms/.exec(text) + return { + indexingMs: indexing ? Number(indexing[1]) : null, + latencyP95Ms: p95 ? Number(p95[1]) : null + } +} + +const workDir = mkdtempSync(join(tmpdir(), 'knownote-chunking-')) +const results = [] + +try { + for (const variant of VARIANTS) { + const outDir = join(workDir, variant.id) + mkdirSync(outDir, { recursive: true }) + console.log(`[chunking] running ${variant.label}`) + const metrics = await runVariant(variant, outDir) + + const report = JSON.parse(readFileSync(join(outDir, 'baseline-v1.4.json'), 'utf8')) + const timing = readTiming(join(outDir, 'baseline-v1.4.md')) + + results.push({ + id: variant.id, + label: variant.label, + chunkSize: report.config.chunking.chunkSize, + chunkOverlap: report.config.chunking.chunkOverlap, + allowSpanPages: report.config.chunking.allowSpanPages, + respectHeadings: report.config.chunking.respectHeadings, + chunkCount: report.config.chunkCount, + ...metrics, + ...timing + }) + } +} finally { + rmSync(workDir, { recursive: true, force: true }) +} + +const baseline = results.find((result) => result.id === 'baseline') +if (!baseline) throw new Error('the baseline variant did not run') + +const delta = (value, base) => (base === 0 ? value - base : value / base - 1) +const formatDelta = (value) => `${value >= 0 ? '+' : ''}${(value * 100).toFixed(1)}%` +const format4 = (value) => value.toFixed(4) + +const rows = results.map((result) => { + const args = `${result.chunkSize}/${result.chunkOverlap}${result.respectHeadings ? ' + headings' : ''}${result.allowSpanPages ? ' + span' : ''}` + return `| ${result.label} | \`${args}\` | ${format4(result.recallAt1)} | ${format4(result.recallAt5)} | ${format4(result.mrr)} | ${format4(result.ndcgAt10)} | ${format4(result.evidencePrecisionAt5)} | ${result.chunkCount} (${formatDelta(delta(result.chunkCount, baseline.chunkCount))}) | ${result.indexingMs} ms | ${result.latencyP95Ms?.toFixed(2)} ms |` +}) + +/** + * Adoption follows the rule frozen in the baseline report: Recall@5 must improve + * and nDCG@10 must not regress. Index size and latency are reported so a win that + * costs 3x latency or 2x index is stated as a trade-off, not hidden. + */ +const adopted = results.filter( + (result) => + result.id !== 'baseline' && + result.recallAt5 > baseline.recallAt5 && + result.ndcgAt10 >= baseline.ndcgAt10 +) +const winner = + adopted.sort((a, b) => b.recallAt5 - a.recallAt5 || b.ndcgAt10 - a.ndcgAt10)[0] ?? null + +const markdown = `# Chunking experiments — v1.5 (#78) + +Generated by \`node scripts/eval-chunking.mjs\`. Numbers are harness output; do not edit them by hand. + +## What was measured + +Every variant runs the real RAG eval harness against the same corpus and the same 30 +questions as \`baseline-v1.4.json\`, with dense retrieval held fixed. Only the chunking +configuration changes, so a difference in the metrics is a difference in the input +distribution retrieval is measured on. + +| Variant | size/overlap | Recall@1 | Recall@5 | MRR | nDCG@10 | Evidence P@5 | Index size (Δ) | Indexing | Query p95 | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | +${rows.join('\n')} + +Timing depends on hardware and is informational, exactly as in the frozen baseline. + +## Adoption rule + +> Adopt a change only if Recall@5 improves and nDCG@10 does not regress. A change +> that trades a large latency or index-size increase for a marginal recall gain is a +> product decision, not an automatic win. + +## Outcome + +${ + winner && winner.id === ADOPTED_ID + ? `**Adopted: \`${winner.label}\`.** It clears the rule (Recall@5 ${format4(winner.recallAt5)} vs baseline ${format4(baseline.recallAt5)}, nDCG@10 ${format4(winner.ndcgAt10)} vs ${format4(baseline.ndcgAt10)}), improves Recall@1 and MRR as well, and *shrinks* the index (${winner.chunkCount} vs ${baseline.chunkCount} chunks). \`DEFAULT_CHUNK_OPTIONS\` is now \`chunkSize=${winner.chunkSize}, chunkOverlap=${winner.chunkOverlap}\`, and the frozen successor baseline is \`docs/eval/baseline-v1.5.json\`. + +**Limitation to state plainly:** the corpus is small (13 short documents, ${winner.chunkCount}–${baseline.chunkCount} chunks), so Recall@5 saturates near 1.0 and is the least discriminating metric here; Recall@1 and MRR carry the result. The adoption should be re-checked on a larger, multi-format corpus before it is treated as settled.` + : winner + ? `\`${winner.label}\` clears the rule (Recall@5 ${format4(winner.recallAt5)} vs baseline ${format4(baseline.recallAt5)}, nDCG@10 ${format4(winner.ndcgAt10)} vs ${format4(baseline.ndcgAt10)}), but was **not** adopted. See the decision recorded in the repository.` + : `No variant cleared the rule. **The shipped default is kept.** A negative result is the point of the experiment: it is the measurement that says the change is not worth making, not a failure to deliver.` +} + +## Reproduce + +\`\`\`bash +npm run eval:prepare # one-time, networked model bootstrap +npm run eval:chunking # offline; runs every variant and rewrites this file +\`\`\` +` + +mkdirSync(resolve(OUT_MD, '..'), { recursive: true }) +writeFileSync(OUT_JSON, `${JSON.stringify({ baseline: 'v1.4', variants: results }, null, 2)}\n`) +writeFileSync(OUT_MD, markdown) + +console.log(`[chunking] wrote ${OUT_JSON} and ${OUT_MD}`) +console.log( + `[chunking] ${winner ? `best clearing variant: ${winner.label}` : 'no variant cleared the rule; keep the baseline'}` +) diff --git a/src/main/eval/harness.ts b/src/main/eval/harness.ts index 0678a45..713a4ed 100644 --- a/src/main/eval/harness.ts +++ b/src/main/eval/harness.ts @@ -14,10 +14,10 @@ import { readdir, readFile } from 'fs/promises' import { join, posix } from 'path' import { and, eq } from 'drizzle-orm' -import { documentBlocks, notebooks } from '../db/schema' +import { documentBlocks, notebooks, chunks } from '../db/schema' import type { getDatabase } from '../db' import type { KnowledgeService } from '../services/KnowledgeService' -import { DEFAULT_CHUNK_OPTIONS } from '../services/ChunkingService' +import { DEFAULT_CHUNK_OPTIONS, type ChunkOptions } from '../services/ChunkingService' import { LOCAL_EMBEDDING_MODEL } from '../embedding/localModel' import { evidencePrecisionAtK, @@ -52,6 +52,8 @@ export interface EvalHarnessOptions { threshold: number /** How many retrieved passages the evidence-precision metric looks at. */ evidenceK: number + /** 分块配置(#78)。实验变体通过它选择策略;缺省时用生产默认值。 */ + chunkOptions: ChunkOptions } const NOTEBOOK_ID = 'eval-notebook' @@ -130,8 +132,10 @@ function resolveGroundTruth( async function indexCorpus( db: Db, knowledgeService: KnowledgeService, - corpusDir: string -): Promise> { + corpusDir: string, + chunkOptions: ChunkOptions +): Promise<{ documentIds: Map; chunkCount: number; indexingMs: number }> { + const indexingStarted = performance.now() const now = new Date() db.insert(notebooks) .values({ id: NOTEBOOK_ID, title: 'Eval corpus', createdAt: now, updatedAt: now }) @@ -143,12 +147,23 @@ async function indexCorpus( for (const file of files) { const documentId = await knowledgeService.addDocumentFromFile( NOTEBOOK_ID, - join(corpusDir, file) + join(corpusDir, file), + undefined, + chunkOptions ) documentIds.set(posix.normalize(file), documentId) } - return documentIds + // Index size is a first-class result of a chunking change: more chunks cost more + // to store and to scan, so a recall win that doubles the index is a trade-off, + // not a free win. + const chunkCount = db + .select({ id: chunks.id }) + .from(chunks) + .where(eq(chunks.notebookId, NOTEBOOK_ID)) + .all().length + + return { documentIds, chunkCount, indexingMs: performance.now() - indexingStarted } } export async function runEvalHarness( @@ -156,7 +171,12 @@ export async function runEvalHarness( knowledgeService: KnowledgeService, options: EvalHarnessOptions ): Promise { - const documentIds = await indexCorpus(db, knowledgeService, options.corpusDir) + const { documentIds, chunkCount, indexingMs } = await indexCorpus( + db, + knowledgeService, + options.corpusDir, + options.chunkOptions + ) const questions = parseQuestions(await readFile(options.questionsPath, 'utf-8')) const perQuestion: QuestionReport[] = [] @@ -202,9 +222,9 @@ export async function runEvalHarness( ) } - const chunking = DEFAULT_CHUNK_OPTIONS + const chunking = { ...DEFAULT_CHUNK_OPTIONS, ...options.chunkOptions } return { - baseline: 'v1.4', + baseline: options.baseline, generatedBy: 'npm run eval', config: { embedding: `${LOCAL_EMBEDDING_MODEL.id}@${LOCAL_EMBEDDING_MODEL.revision} ${LOCAL_EMBEDDING_MODEL.dtype} (${LOCAL_EMBEDDING_MODEL.dimensions}d, local)`, @@ -212,7 +232,8 @@ export async function runEvalHarness( chunkSize: chunking.chunkSize, chunkOverlap: chunking.chunkOverlap, minChunkSize: chunking.minChunkSize, - allowSpanPages: chunking.allowSpanPages + allowSpanPages: chunking.allowSpanPages, + respectHeadings: chunking.respectHeadings }, retrieval: 'dense', topK: options.topK, @@ -220,12 +241,14 @@ export async function runEvalHarness( evidenceK: options.evidenceK, corpus: options.corpusLabel, documents: documentIds.size, - questions: questions.length + questions: questions.length, + chunkCount }, metrics, timing: { latencyP50Ms: percentile(latencies, 50), - latencyP95Ms: percentile(latencies, 95) + latencyP95Ms: percentile(latencies, 95), + indexingMs }, perQuestion } @@ -246,7 +269,10 @@ export function stabilize(report: EvalReport): EvalReport { }, timing: { latencyP50Ms: round(report.timing.latencyP50Ms), - latencyP95Ms: round(report.timing.latencyP95Ms) + latencyP95Ms: round(report.timing.latencyP95Ms), + // Throughput is informational and excluded from the deterministic report; the + // full report keeps it for the #78 comparison. + indexingMs: Math.round(report.timing.indexingMs) }, perQuestion: report.perQuestion } diff --git a/src/main/eval/metrics.ts b/src/main/eval/metrics.ts index c7dd764..41f0d23 100644 --- a/src/main/eval/metrics.ts +++ b/src/main/eval/metrics.ts @@ -35,14 +35,31 @@ export function reciprocalRank(matchesByRank: MatchMatrix): number { /** * nDCG@k with binary gains. The ideal ranking puts every ground-truth location * first, so the discount is a plain log base 2. + * + * A rank only gains when it covers a ground-truth location that no earlier rank + * already covered. Without that, several passages recovering the *same* block each + * scored a gain while the ideal ranking only counted the block once, and nDCG went + * above 1 — which is not a valid ranking metric. Smaller chunks overlap more and + * recover the same block from several ranks, so the bug only became visible under + * #78. (Found by the #78 chunking experiment.) */ export function ndcgAtK(matchesByRank: MatchMatrix, groundTruthCount: number, k: number): number { if (groundTruthCount === 0) return 0 let dcg = 0 + const covered = new Set() const limit = Math.min(matchesByRank.length, k) for (let i = 0; i < limit; i++) { - if (matchesByRank[i].length > 0) dcg += 1 / Math.log2(i + 2) + let fresh = false + for (const match of matchesByRank[i]) { + // An index at or beyond `groundTruthCount` is not part of the declared ground + // truth and cannot be a gain; ignoring it keeps DCG <= IDCG by construction. + if (match < groundTruthCount && !covered.has(match)) { + covered.add(match) + fresh = true + } + } + if (fresh) dcg += 1 / Math.log2(i + 2) } let idcg = 0 diff --git a/src/main/eval/report.ts b/src/main/eval/report.ts index bcec7c9..1dfbd04 100644 --- a/src/main/eval/report.ts +++ b/src/main/eval/report.ts @@ -22,11 +22,12 @@ Generated by \`${report.generatedBy}\`. The numbers below are harness output — | Setting | Value | | --- | --- | | Embedding | \`${config.embedding}\` | -| Chunking | \`chunkSize=${chunking.chunkSize}, chunkOverlap=${chunking.chunkOverlap}, minChunkSize=${chunking.minChunkSize}, allowSpanPages=${chunking.allowSpanPages}\` | +| Chunking | \`chunkSize=${chunking.chunkSize}, chunkOverlap=${chunking.chunkOverlap}, minChunkSize=${chunking.minChunkSize}, allowSpanPages=${chunking.allowSpanPages}, respectHeadings=${chunking.respectHeadings}\` | | Retrieval | \`${config.retrieval}\` | | Ranks | \`topK=${config.topK}, threshold=${config.threshold}\` | | Evidence per query | \`evidenceK=${config.evidenceK}\` | | Corpus | \`${config.corpus}\` (${config.documents} documents, ${config.questions} questions) | +| Index size | ${config.chunkCount} chunks | ## Metrics @@ -39,9 +40,10 @@ Generated by \`${report.generatedBy}\`. The numbers below are harness output — | nDCG@10 | ${format(metrics.ndcgAt10)} | | Evidence precision@${config.evidenceK} | ${format(metrics.evidencePrecisionAt5)} | -Timing is informational only and is **not** frozen: p50 ${timing.latencyP50Ms.toFixed(2)} ms, -p95 ${timing.latencyP95Ms.toFixed(2)} ms on the machine that produced this file. Query -latency depends on hardware and load, so it must never be the reason two runs differ. +Timing is informational only and is **not** frozen: indexing ${timing.indexingMs} ms, query +p50 ${timing.latencyP50Ms.toFixed(2)} ms, p95 ${timing.latencyP95Ms.toFixed(2)} ms on the +machine that produced this file. Timing and index size depend on hardware and on the +corpus, so they must never be the reason two runs differ. ## Definitions diff --git a/src/main/eval/run.ts b/src/main/eval/run.ts index d999722..1548608 100644 --- a/src/main/eval/run.ts +++ b/src/main/eval/run.ts @@ -20,6 +20,7 @@ import { ConnectionManager } from '../models/ConnectionManager' import { EmbeddingService } from '../services/EmbeddingService' import { KnowledgeService } from '../services/KnowledgeService' import { isModelInstalled } from '../embedding/ModelRegistry' +import { DEFAULT_CHUNK_OPTIONS, type ChunkOptions } from '../services/ChunkingService' import { runEvalHarness, stabilize, @@ -40,6 +41,52 @@ function readOption(argv: readonly string[], prefix: string, fallback: string): return arg ? arg.slice(prefix.length) : fallback } +function readNumberOption(argv: readonly string[], prefix: string, fallback: number): number { + const raw = argv.find((value) => value.startsWith(prefix)) + if (!raw) return fallback + const parsed = Number(raw.slice(prefix.length)) + if (!Number.isFinite(parsed)) { + throw new Error(`${prefix} expects a number, got ${JSON.stringify(raw.slice(prefix.length))}`) + } + return parsed +} + +function readBoolOption(argv: readonly string[], prefix: string, fallback: boolean): boolean { + const raw = argv.find((value) => value.startsWith(prefix)) + if (!raw) return fallback + const value = raw.slice(prefix.length) + if (value !== 'true' && value !== 'false') { + throw new Error(`${prefix} expects true or false, got ${JSON.stringify(value)}`) + } + return value === 'true' +} + +/** + * Chunking config for one run (#78). The defaults are the production defaults, so + * `npm run eval` with no flags still measures what ships. + */ +function readChunkOptions(argv: readonly string[]): ChunkOptions { + return { + chunkSize: readNumberOption(argv, '--eval-chunk-size=', DEFAULT_CHUNK_OPTIONS.chunkSize), + chunkOverlap: readNumberOption( + argv, + '--eval-chunk-overlap=', + DEFAULT_CHUNK_OPTIONS.chunkOverlap + ), + minChunkSize: readNumberOption(argv, '--eval-chunk-min=', DEFAULT_CHUNK_OPTIONS.minChunkSize), + allowSpanPages: readBoolOption( + argv, + '--eval-allow-span-pages=', + DEFAULT_CHUNK_OPTIONS.allowSpanPages + ), + respectHeadings: readBoolOption( + argv, + '--eval-respect-headings=', + DEFAULT_CHUNK_OPTIONS.respectHeadings + ) + } +} + /** * A ConnectionManager that can never produce a remote backend, so the harness * always measures the built-in local model regardless of the developer's own @@ -109,10 +156,11 @@ export async function runEvalCli(argv: readonly string[] = process.argv): Promis // identical on every machine and checkout. corpusLabel: relative(process.cwd(), corpusDir) || 'eval/corpus', questionsPath, - baseline: 'v1.4', + baseline: readOption(argv, '--eval-baseline=', 'v1.5'), topK: 10, threshold: 0, - evidenceK: 5 + evidenceK: 5, + chunkOptions: readChunkOptions(argv) } const report = stabilize(await runEvalHarness(getDatabase(), knowledgeService, options)) diff --git a/src/main/eval/types.ts b/src/main/eval/types.ts index cdb31e0..9d3e4e4 100644 --- a/src/main/eval/types.ts +++ b/src/main/eval/types.ts @@ -71,6 +71,8 @@ export interface EvalReport { chunkOverlap: number minChunkSize: number allowSpanPages: boolean + /** 标题处强制断节(#78)。 */ + respectHeadings: boolean } retrieval: string topK: number @@ -80,9 +82,12 @@ export interface EvalReport { corpus: string documents: number questions: number + /** 索引出的 chunk 总数(#78 的 index size)。 */ + chunkCount: number } metrics: EvalMetrics - timing: { latencyP50Ms: number; latencyP95Ms: number } + /** `indexingMs` 只用于 #78 的吞吐比较;它不在确定报告里,也不该成为差异原因。 */ + timing: { latencyP50Ms: number; latencyP95Ms: number; indexingMs: number } perQuestion: QuestionReport[] } diff --git a/src/main/services/ChunkingService.ts b/src/main/services/ChunkingService.ts index 5696f0f..d16579e 100644 --- a/src/main/services/ChunkingService.ts +++ b/src/main/services/ChunkingService.ts @@ -21,17 +21,30 @@ export interface ChunkOptions { separators?: string[] // 回退窗口在何处断开的优先级列表 minChunkSize?: number // 整篇文本不超过它时直接作为一块 allowSpanPages?: boolean // 允许一个 chunk 跨页,默认 false(分页文档在页边界断开) + /** + * 在标题处强制开始新 chunk(#78),默认 false。 + * + * false 时标题只是原子单元,仍会和后面的段落装进同一个 chunk;true 时一个 chunk + * 不会横跨章节,代价是章节短时会产生更多、更小的 chunk。 + */ + respectHeadings?: boolean } /** * 默认分块参数。导出的原因只有一个:eval baseline 报告必须引用真实值,而不是 * 把它手抄一遍。改了默认值而没有重新跑 baseline,差异会从报告里直接暴露出来。 + * + * `chunkSize=1000 / chunkOverlap=100` 是 #78 的实验结论:在 v1.4 语料上,相比 + * 500/50,它让 Recall@1 0.7667 → 0.8333、MRR 0.8492 → 0.9278、nDCG@10 + * 0.8899 → 0.9437,同时索引从 36 个 chunk 降到 19 个。完整对比见 + * `docs/eval/chunking-v1.5.md`,冻结后的基线是 `docs/eval/baseline-v1.5.json`。 */ export const DEFAULT_CHUNK_OPTIONS: Required = { - chunkSize: 500, - chunkOverlap: 50, + chunkSize: 1000, + chunkOverlap: 100, minChunkSize: 100, allowSpanPages: false, + respectHeadings: false, separators: [ '\n\n\n', // 多个空行(章节分隔) '\n\n', // 段落分隔 @@ -257,6 +270,16 @@ export class ChunkingService { } for (const unit of units) { + // 标题处强制断节(#78):一个 chunk 不横跨章节。无重叠 —— 跨章节重叠会把 + // 上一节的尾巴带进下一节,正是这个策略要避免的。 + if ( + opts.respectHeadings && + current.length > 0 && + blocks[unit.blockIndex].kind === 'heading' + ) { + flush() + } + if (current.length === 0) { current.push(unit) continue diff --git a/src/main/services/KnowledgeService.ts b/src/main/services/KnowledgeService.ts index 5dd3711..8f4e085 100644 --- a/src/main/services/KnowledgeService.ts +++ b/src/main/services/KnowledgeService.ts @@ -284,7 +284,8 @@ export class KnowledgeService { async addDocumentFromFile( notebookId: string, filePath: string, - onProgress?: IndexProgressCallback + onProgress?: IndexProgressCallback, + chunkOptions?: ChunkOptions ): Promise { const db = getDatabase() const documentId = `doc_${Date.now()}_${Math.random().toString(36).slice(2, 9)}` @@ -307,7 +308,13 @@ export class KnowledgeService { .run() // 先拷贝,再解析原文件:本地副本是重新索引/结构恢复时读取的东西。 - await this.ingestFile(documentId, filePath, { copyFrom: filePath }, 'import', onProgress) + await this.ingestFile( + documentId, + filePath, + { copyFrom: filePath, chunkOptions }, + 'import', + onProgress + ) return documentId } @@ -321,7 +328,7 @@ export class KnowledgeService { private async ingestFile( documentId: string, filePath: string, - options: { copyFrom?: string }, + options: { copyFrom?: string; chunkOptions?: ChunkOptions }, kind: IngestionRunKind, onProgress?: IndexProgressCallback ): Promise { @@ -372,7 +379,7 @@ export class KnowledgeService { runId, parseResult.content, parseResult.structure ?? undefined, - {}, + { chunkOptions: options.chunkOptions }, onProgress ) diff --git a/test/evalMetrics.test.ts b/test/evalMetrics.test.ts index 10d7e03..1520dc6 100644 --- a/test/evalMetrics.test.ts +++ b/test/evalMetrics.test.ts @@ -4,6 +4,7 @@ import { evidencePrecisionAtK, firstRelevantRank, mean, + type MatchMatrix, ndcgAtK, percentile, recallAtK, @@ -53,6 +54,45 @@ test('nDCG@k discounts a later hit and is 1 when the hit is first', () => { assert.equal(ndcgAtK([[], []], 1, 10), 0) }) +test('nDCG@k counts a ground-truth location once, however many ranks recover it', () => { + // The bug #78 exposed: three passages all covering the same single ground truth + // used to score three gains against an ideal that only has one, so nDCG was 3 + // times the valid maximum. Only the first rank is a fresh gain now. + assert.equal(ndcgAtK([[0], [0], [0]], 1, 10), 1) + // Two ground truths recovered from the first rank is one binary gain against an + // ideal that would place them at ranks 1 and 2. + assert.ok( + Math.abs( + ndcgAtK( + [ + [0, 1], + [0, 1] + ], + 2, + 10 + ) - + 1 / (1 + 1 / Math.log2(3)) + ) < 1e-12 + ) +}) + +test('nDCG@k never exceeds 1', () => { + const cases: Array<{ matrix: MatchMatrix; count: number }> = [ + { matrix: [[0], [0], [0]], count: 1 }, + { matrix: [[0], [0], [0]], count: 3 }, + { matrix: [[0], [1], [0, 1]], count: 2 }, + { matrix: [[0, 1, 2]], count: 3 }, + { matrix: [[], [0], [], [1], [], [2]], count: 3 }, + { matrix: [[]], count: 0 }, + { matrix: [], count: 0 } + ] + + for (const { matrix, count } of cases) { + const value = ndcgAtK(matrix, count, 10) + assert.ok(value >= 0 && value <= 1, `nDCG out of range: ${value} for count ${count}`) + } +}) + test('evidence precision counts grounded passages over retrieved passages', () => { // 2 of 3 retrieved passages cover a ground-truth block assert.equal(evidencePrecisionAtK([[0], [], [1]], 3), 2 / 3)