diff --git a/.prettierignore b/.prettierignore index 1a115ff..f418484 100644 --- a/.prettierignore +++ b/.prettierignore @@ -19,6 +19,24 @@ docs/eval/baseline-*.md docs/eval/chunking-*.json docs/eval/chunking-*.md +# Same reason again: generated by `scripts/eval-threshold.mjs`. Regenerate with +# `npm run eval:threshold`. +docs/eval/threshold-*.json +docs/eval/threshold-*.md + +# Same reason again: generated by `scripts/eval-sweep.mjs`. Regenerate with +# `npm run eval:sweep`. +docs/eval/sweep-*.json +docs/eval/sweep-*.md + +# And by `scripts/eval-scores.mjs`. Regenerate with `npm run eval:scores`. +docs/eval/scores-*.json +docs/eval/scores-*.md + +# And by `scripts/eval-paired.mjs`. Regenerate with `npm run eval:paired`. +docs/eval/paired-*.json +docs/eval/paired-*.md + # And the same again for `scripts/eval-retrieval.mjs` (#77). docs/eval/retrieval-*.json docs/eval/retrieval-*.md diff --git a/docs/eval/README.md b/docs/eval/README.md new file mode 100644 index 0000000..17cb1af --- /dev/null +++ b/docs/eval/README.md @@ -0,0 +1,61 @@ +# Eval closeout — v1.6 + +This directory holds the retrieval eval reports. The versioned report `*.json` / `*.md` files here are +harness output — regenerate it with the script named in its header, never edit it by +hand. `eval/README.md` documents the harness itself. + +## What this closeout lands + +The eval work reaches v1.6 and stops at the retrieval layer: + +- **Two Ks.** `candidateK` is the first-stage width per channel, `contextK` is how many + passages reach the prompt. `baseline-v1.6` records both. +- **An explicit split.** `eval/splits.json` assigns every question to `validation` or + `test`. A parameter is selected on `validation` and reported on `test`. +- **Shipped metrics.** Hit rate@5, MAP@10, context precision/recall, per-query-type + breakdown, and a separate unanswerable group. +- **Experiments, not wins.** `retrieval-v1.6`, `threshold-v1.6`, `sweep-v1.6`, + `scores-v1.6` and `paired-v1.6` compare strategies and parameters against the + baseline. + +## What did **not** change + +Production retrieval is unchanged: **dense**, `candidateK = 20`, `contextK = 3`, +`threshold = 0.5` (`src/main/ipc/chatHandlers.ts`). No strategy cleared the adoption +rule on `validation`, so `retrieval-v1.6` reports the shipped dense strategy and no +winner. The threshold sweep is **flat** — every threshold from 0 to 0.6 behaves +identically on this corpus — so there is no evidence to move `threshold` either. These +are negative results: validation does not authorize changing the defaults. They do +not establish that other strategies have no value. + +## Corpus state + +21 documents, 53 chunks, 78 questions (53 answerable, 25 unanswerable). The unanswerable +group is what a threshold decision should move; on this corpus it does not, because no +candidate is ever filtered out. + +## Phase 1 boundary + +Retrieval Eval v2 Phase 1 (#192) closes with this integration. No further K, +threshold, hybrid or retrieval-metric tuning is planned under that epic. + +Generator evaluation is tracked in #213 and citation evaluation in #214. Public +QASPER/MIRACL-zh subsets, a pinned offline reranker (historical PR #170), further +corpus scaling and token accounting beyond the character proxy are deferred, not +claimed complete and not prerequisites for this phase's closure. + +The original ~300-chunk target is unmet. At 53 chunks, only 19 answerable questions +are in `validation`: this is a small diagnostic benchmark, not evidence of broad +real-world performance. Expanding it is explicitly outside this closeout. + +## Regenerate + +```bash +npm run eval:prepare # one-time, networked: pin the embedding model +npm run eval # baseline-v1.6.{json,md} +npm run eval:retrieval # retrieval-v1.6.{json,md} +npm run eval:threshold # threshold-v1.6.{json,md} +npm run eval:sweep # sweep-v1.6.{json,md} +npm run eval:scores # scores-v1.6.{json,md} +npm run eval:paired # paired-v1.6.{json,md} +``` diff --git a/docs/eval/baseline-v1.6.json b/docs/eval/baseline-v1.6.json index eb15f14..2bddcea 100644 --- a/docs/eval/baseline-v1.6.json +++ b/docs/eval/baseline-v1.6.json @@ -11,30 +11,135 @@ "respectHeadings": false }, "retrieval": "dense", + "split": "all", "candidateK": 20, "contextK": 3, "threshold": 0.5, - "evidenceK": 5, "corpus": "eval/corpus", - "documents": 13, - "questions": 30, - "chunkCount": 19 + "documents": 21, + "questions": 78, + "chunkCount": 53 }, "metrics": { - "recallAt1": 0.833333, - "recallAt5": 1, - "recallAt10": 1, - "mrr": 0.927778, - "ndcgAt10": 0.94375, - "evidencePrecisionAt5": 0.213333 + "recallAt1": 0.59434, + "recallAt5": 0.877358, + "recallAt10": 0.919811, + "mrr": 0.754755, + "ndcgAt10": 0.780449, + "hitRateAt5": 0.90566, + "mapAt10": 0.72804, + "contextPrecision": 0.308176, + "contextRecall": 0.816038 + }, + "byType": [ + { + "type": "cross-lingual", + "questions": 9, + "metrics": { + "recallAt1": 0, + "recallAt5": 0.666667, + "recallAt10": 0.777778, + "mrr": 0.299383, + "ndcgAt10": 0.41727, + "hitRateAt5": 0.666667, + "mapAt10": 0.299383, + "contextPrecision": 0.185185, + "contextRecall": 0.555556 + } + }, + { + "type": "exact", + "questions": 8, + "metrics": { + "recallAt1": 0.75, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 0.822917, + "ndcgAt10": 0.866335, + "hitRateAt5": 1, + "mapAt10": 0.822917, + "contextPrecision": 0.291667, + "contextRecall": 0.875 + } + }, + { + "type": "hard-negative", + "questions": 7, + "metrics": { + "recallAt1": 0.714286, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 0.833333, + "ndcgAt10": 0.875847, + "hitRateAt5": 1, + "mapAt10": 0.833333, + "contextPrecision": 0.333333, + "contextRecall": 1 + } + }, + { + "type": "multi-hop", + "questions": 5, + "metrics": { + "recallAt1": 0.3, + "recallAt5": 0.7, + "recallAt10": 0.75, + "mrr": 0.9, + "ndcgAt10": 0.721795, + "hitRateAt5": 1, + "mapAt10": 0.635, + "contextPrecision": 0.6, + "contextRecall": 0.65 + } + }, + { + "type": "semantic", + "questions": 21, + "metrics": { + "recallAt1": 0.761905, + "recallAt5": 0.904762, + "recallAt10": 0.952381, + "mrr": 0.828139, + "ndcgAt10": 0.85418, + "hitRateAt5": 0.904762, + "mapAt10": 0.82381, + "contextPrecision": 0.285714, + "contextRecall": 0.857143 + } + }, + { + "type": "zh", + "questions": 3, + "metrics": { + "recallAt1": 1, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 1, + "ndcgAt10": 1, + "hitRateAt5": 1, + "mapAt10": 1, + "contextPrecision": 0.333333, + "contextRecall": 1 + } + } + ], + "unanswerable": { + "questions": 25, + "abstentionCount": 0, + "retrievalAbstentionRate": 0, + "meanCandidatesRetrieved": 20, + "meanContextPassages": 3 }, "perQuestion": [ { "id": "q001", "question": "Why is bedload harder to measure than suspended sediment?", + "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, + "contextChars": 2198, "matchesByRank": [ [ 0 @@ -56,16 +161,22 @@ [], [], [], + [], [] ] }, { "id": "q002", "question": "How many replicate samples are collected at each river station?", - "firstRelevantRank": 1, + "type": "exact", + "answerable": true, + "firstRelevantRank": 3, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, + "contextChars": 2297, "matchesByRank": [ + [], + [], [ 0 ], @@ -85,16 +196,18 @@ [], [], [], - [], [] ] }, { "id": "q003", "question": "What is the central trade-off in lithium-ion cell design?", + "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, + "contextChars": 2679, "matchesByRank": [ [ 0 @@ -116,15 +229,19 @@ [], [], [], + [], [] ] }, { "id": "q004", "question": "Why do nickel-rich battery packs need more aggressive thermal management?", + "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, + "contextChars": 2113, "matchesByRank": [ [ 0 @@ -146,15 +263,19 @@ [], [], [], + [], [] ] }, { "id": "q005", "question": "What happens once the separator in a battery cell melts?", + "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, + "contextChars": 2157, "matchesByRank": [ [ 0 @@ -176,15 +297,19 @@ [], [], [], + [], [] ] }, { "id": "q006", "question": "At what temperature do honeybees begin to forage?", + "type": "exact", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, + "contextChars": 2364, "matchesByRank": [ [ 0 @@ -206,15 +331,19 @@ [], [], [], + [], [] ] }, { "id": "q007", "question": "What does a late frost damage during full bloom?", + "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, + "contextChars": 2047, "matchesByRank": [ [ 0 @@ -236,15 +365,19 @@ [], [], [], + [], [] ] }, { "id": "q008", "question": "Why is a continuous tree canopy more effective at cooling than isolated trees?", + "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, + "contextChars": 2057, "matchesByRank": [ [ 0 @@ -266,15 +399,19 @@ [], [], [], + [], [] ] }, { "id": "q009", "question": "Why are trees with aggressive surface roots unsuitable for narrow verges?", + "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, + "contextChars": 2136, "matchesByRank": [ [ 0 @@ -296,15 +433,19 @@ [], [], [], + [], [] ] }, { "id": "q010", "question": "At what temperature is lactic acid fermentation fastest?", + "type": "exact", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, + "contextChars": 2682, "matchesByRank": [ [ 0 @@ -326,15 +467,19 @@ [], [], [], + [], [] ] }, { "id": "q011", "question": "Is the salt percentage in fermentation based on vegetable weight or water weight?", + "type": "exact", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, + "contextChars": 2051, "matchesByRank": [ [ 0 @@ -356,15 +501,19 @@ [], [], [], + [], [] ] }, { "id": "q012", "question": "Why must tidal turbines be sited in places with very fast currents?", + "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, + "contextChars": 2687, "matchesByRank": [ [ 0 @@ -386,21 +535,22 @@ [], [], [], + [], [] ] }, { "id": "q013", "question": "What is the main environmental concern for tidal energy installations?", - "firstRelevantRank": 3, + "type": "semantic", + "answerable": true, + "firstRelevantRank": 11, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, + "contextChars": 2579, "matchesByRank": [ [], [], - [ - 0 - ], [], [], [], @@ -409,6 +559,10 @@ [], [], [], + [ + 0 + ], + [], [], [], [], @@ -422,9 +576,12 @@ { "id": "q014", "question": "In lake monitoring, how is the sampling depth actually recorded?", + "type": "exact", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, + "contextChars": 2910, "matchesByRank": [ [ 0 @@ -446,15 +603,19 @@ [], [], [], + [], [] ] }, { "id": "q015", "question": "Why does deep-water oxygen fall while a lake remains stratified?", + "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, + "contextChars": 2884, "matchesByRank": [ [ 0 @@ -476,15 +637,19 @@ [], [], [], + [], [] ] }, { "id": "q016", "question": "How do supercapacitors hold their charge?", + "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, + "contextChars": 2702, "matchesByRank": [ [ 0 @@ -506,15 +671,19 @@ [], [], [], + [], [] ] }, { "id": "q017", "question": "Why can solitary bees pollinate a bloom week that is too cold for honeybee hives?", + "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, + "contextChars": 1773, "matchesByRank": [ [ 0 @@ -536,15 +705,19 @@ [], [], [], + [], [] ] }, { "id": "q018", "question": "Why is one continuous planted roof layer better than several isolated planted beds?", + "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, + "contextChars": 2561, "matchesByRank": [ [ 0 @@ -566,15 +739,19 @@ [], [], [], + [], [] ] }, { "id": "q019", "question": "Where do acetic acid bacteria sit in a vinegar culture?", + "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, + "contextChars": 2636, "matchesByRank": [ [ 0 @@ -596,15 +773,19 @@ [], [], [], + [], [] ] }, { "id": "q020", "question": "What happens if a vinegar culture is sealed airtight?", + "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, + "contextChars": 2269, "matchesByRank": [ [ 0 @@ -626,15 +807,19 @@ [], [], [], + [], [] ] }, { "id": "q021", "question": "Why is wave energy harder to schedule ahead than tidal energy?", + "type": "semantic", + "answerable": true, "firstRelevantRank": 2, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, + "contextChars": 2580, "matchesByRank": [ [], [ @@ -656,15 +841,19 @@ [], [], [], + [], [] ] }, { "id": "q022", "question": "Where does siting for wave energy devices concentrate, and where does it not?", + "type": "exact", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, + "contextChars": 2580, "matchesByRank": [ [ 0 @@ -686,23 +875,24 @@ [], [], [], + [], [] ] }, { "id": "q023", "question": "How does the river sampling protocol differ from the lake sampling protocol?", + "type": "multi-hop", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 2, - "retrievedCount": 19, + "retrievedCount": 20, + "contextChars": 2912, "matchesByRank": [ [ 1 ], [], - [ - 0 - ], [], [], [], @@ -718,15 +908,22 @@ [], [], [], + [], + [ + 0 + ], [] ] }, { "id": "q024", "question": "A street canopy and a green roof are both said to cool; what surface does each one shade?", + "type": "multi-hop", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 2, - "retrievedCount": 19, + "retrievedCount": 20, + "contextChars": 1970, "matchesByRank": [ [ 0 @@ -750,15 +947,19 @@ [], [], [], + [], [] ] }, { "id": "q025", "question": "Which preservation method depends on keeping air away from the food?", + "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, + "contextChars": 1373, "matchesByRank": [ [ 0 @@ -780,15 +981,19 @@ [], [], [], + [], [] ] }, { "id": "q026", "question": "Why can one cold morning cost a grower the whole crop even when colonies are brought in?", + "type": "semantic", + "answerable": true, "firstRelevantRank": 2, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, + "contextChars": 1867, "matchesByRank": [ [], [ @@ -810,15 +1015,19 @@ [], [], [], + [], [] ] }, { "id": "q027", "question": "绿茶应该怎样保存才能减缓氧化?", + "type": "zh", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, + "contextChars": 1453, "matchesByRank": [ [ 0 @@ -840,15 +1049,19 @@ [], [], [], + [], [] ] }, { "id": "q028", "question": "茶叶储存的相对湿度上限是多少?", + "type": "zh", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, + "contextChars": 1376, "matchesByRank": [ [ 0 @@ -870,15 +1083,19 @@ [], [], [], + [], [] ] }, { "id": "q029", "question": "为什么冷冻保存的茶叶取出后不能立刻打开包装?", + "type": "zh", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, + "contextChars": 1294, "matchesByRank": [ [ 0 @@ -900,15 +1117,19 @@ [], [], [], + [], [] ] }, { "id": "q030", "question": "为什么潮汐能比风能和太阳能更容易提前安排发电?", + "type": "cross-lingual", + "answerable": true, "firstRelevantRank": 2, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, + "contextChars": 1955, "matchesByRank": [ [], [ @@ -930,6 +1151,1610 @@ [], [], [], + [], + [] + ] + }, + { + "id": "q031", + "question": "What is the installed capacity of the tidal energy installation described?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2545, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q032", + "question": "Which laboratory published the river monitoring protocol?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2789, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q033", + "question": "What is the boiling point of mercury?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2755, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q034", + "question": "What is the retail price of the lithium-ion cells discussed?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2679, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q035", + "question": "How many megawatts does the tidal array generate?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2687, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q036", + "question": "Who won the 2018 FIFA World Cup?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2819, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q037", + "question": "推移质为什么比悬移质更难测?", + "type": "cross-lingual", + "answerable": true, + "firstRelevantRank": 0, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 1986, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q038", + "question": "河流监测中,每个站点要采几份平行样?", + "type": "cross-lingual", + "answerable": true, + "firstRelevantRank": 0, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 1337, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q039", + "question": "设计锂离子电芯时最核心的取舍是什么?", + "type": "cross-lingual", + "answerable": true, + "firstRelevantRank": 2, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 1994, + "matchesByRank": [ + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q040", + "question": "富镍电池为什么对散热要求更高?", + "type": "cross-lingual", + "answerable": true, + "firstRelevantRank": 2, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 1994, + "matchesByRank": [ + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q041", + "question": "电芯隔膜一旦熔化会导致什么后果?", + "type": "cross-lingual", + "answerable": true, + "firstRelevantRank": 4, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 1294, + "matchesByRank": [ + [], + [], + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q042", + "question": "蜜蜂大致从什么温度开始出巢觅食?", + "type": "cross-lingual", + "answerable": true, + "firstRelevantRank": 3, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 1321, + "matchesByRank": [ + [], + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q043", + "question": "花期遭遇晚霜,主要受损的是什么?", + "type": "cross-lingual", + "answerable": true, + "firstRelevantRank": 2, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 801, + "matchesByRank": [ + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q044", + "question": "为什么成片的树冠比孤立的树降温效果更好?", + "type": "cross-lingual", + "answerable": true, + "firstRelevantRank": 9, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 1447, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q045", + "question": "What is the default retry count in Gateway API v2?", + "type": "exact", + "answerable": true, + "firstRelevantRank": 4, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 2898, + "matchesByRank": [ + [], + [], + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q046", + "question": "What is the default request timeout of Gateway API v1?", + "type": "hard-negative", + "answerable": true, + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 2909, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q047", + "question": "What is the context window of Gateway API v3, in tokens?", + "type": "hard-negative", + "answerable": true, + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 2899, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q048", + "question": "What is the default rate limit of Gateway API v2?", + "type": "hard-negative", + "answerable": true, + "firstRelevantRank": 2, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 2854, + "matchesByRank": [ + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q049", + "question": "What is the default request timeout of Gateway API v3?", + "type": "hard-negative", + "answerable": true, + "firstRelevantRank": 3, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 2921, + "matchesByRank": [ + [], + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q050", + "question": "Which Gateway API version waits longest between retries after a failed request?", + "type": "semantic", + "answerable": true, + "firstRelevantRank": 10, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 2910, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q051", + "question": "Compare the default timeouts and rate limits of Gateway API v1 and v3.", + "type": "multi-hop", + "answerable": true, + "firstRelevantRank": 1, + "relevantCount": 4, + "retrievedCount": 20, + "contextChars": 2888, + "matchesByRank": [ + [ + 0 + ], + [ + 2, + 3 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [ + 1 + ], + [ + 3 + ], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q052", + "question": "Compare the retry count and context window of Gateway API v2 and v3.", + "type": "multi-hop", + "answerable": true, + "firstRelevantRank": 2, + "relevantCount": 4, + "retrievedCount": 20, + "contextChars": 2899, + "matchesByRank": [ + [], + [ + 3 + ], + [], + [], + [], + [], + [], + [], + [], + [ + 1 + ], + [ + 2 + ], + [ + 0 + ], + [], + [ + 1 + ], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q053", + "question": "How much GPU memory does Gateway API v2 require to serve a request?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2875, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q054", + "question": "What is the monthly subscription price of Gateway API v3?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2908, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q055", + "question": "How many replicate samples are collected at each reservoir monitoring station?", + "type": "exact", + "answerable": true, + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 2819, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q056", + "question": "How many replicate samples are collected at each estuary transect station?", + "type": "hard-negative", + "answerable": true, + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 2835, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q057", + "question": "At what depth below the surface does the coastal programme place its shore-station sensor?", + "type": "hard-negative", + "answerable": true, + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 2859, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q058", + "question": "At what depth below the water table do groundwater boreholes carry their pressure transducer?", + "type": "hard-negative", + "answerable": true, + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 2876, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q059", + "question": "Which monitoring programme samples most frequently?", + "type": "semantic", + "answerable": true, + "firstRelevantRank": 5, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 2933, + "matchesByRank": [ + [], + [], + [], + [], + [ + 0 + ], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q060", + "question": "Compare the sampling interval and replicate count of the reservoir and groundwater programmes.", + "type": "multi-hop", + "answerable": true, + "firstRelevantRank": 1, + "relevantCount": 4, + "retrievedCount": 20, + "contextChars": 2879, + "matchesByRank": [ + [ + 2 + ], + [ + 2, + 3 + ], + [ + 0 + ], + [ + 0, + 1 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q061", + "question": "Why does the estuary programme take more replicate samples than the other monitoring programmes?", + "type": "semantic", + "answerable": true, + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 2831, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q062", + "question": "What is the annual operating cost of the coastal monitoring buoys?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2826, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q063", + "question": "How many litres per second does the reservoir release downstream on a typical day?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2809, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q064", + "question": "How much GPU memory does Gateway API v3 need to serve a request?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2908, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q065", + "question": "What is the uptime SLA for Gateway API v2?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2875, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q066", + "question": "How long does an authentication token stay valid before it expires?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2649, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q067", + "question": "Which SDK version should a client use with Gateway API v3?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2908, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q068", + "question": "Can Gateway API v2 be self-hosted on-premise?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2875, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q069", + "question": "How is usage invoiced for Gateway API v2?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2875, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q070", + "question": "Which laboratory is accredited to analyse the estuary transect samples?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2902, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q071", + "question": "Which vendor supplies the coastal monitoring buoys?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2826, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q072", + "question": "What is the annual funding for the groundwater monitoring programme?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2933, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q073", + "question": "How many staff work on the reservoir monitoring programme?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2096, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q074", + "question": "What is the warranty period on the estuary monitoring sensors?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2876, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q075", + "question": "How long are the coastal monitoring readings retained before deletion?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2749, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q076", + "question": "What encryption standard is used to transmit the river monitoring readings?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2746, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q077", + "question": "When was the last external audit of the lake monitoring programme?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2941, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q078", + "question": "What is the unit price of a river monitoring sediment sampler?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2789, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], [] ] } diff --git a/docs/eval/baseline-v1.6.md b/docs/eval/baseline-v1.6.md index 5418dd6..ffa7f59 100644 --- a/docs/eval/baseline-v1.6.md +++ b/docs/eval/baseline-v1.6.md @@ -11,23 +11,65 @@ Generated by `npm run eval`. The numbers below are harness output — do not edi | Retrieval | `dense` | | Ranks | `candidateK=20, threshold=0.5` | | Context width | `contextK=3` | -| Evidence per query | `evidenceK=5` | -| Corpus | `eval/corpus` (13 documents, 30 questions) | -| Index size | 19 chunks | +| Corpus | `eval/corpus` (21 documents) | +| Split | `all` (78 questions, 53 answerable) | +| Index size | 53 chunks | ## Metrics | Metric | Value | | --- | --- | -| Recall@1 | 0.8333 | -| Recall@5 | 1.0000 | -| Recall@10 | 1.0000 | -| MRR | 0.9278 | -| nDCG@10 | 0.9437 | -| Evidence precision@5 | 0.2133 | - -Timing is informational only and is **not** frozen: indexing 1582 ms, query -p50 11.35 ms, p95 18.82 ms on the +| Recall@1 | 0.5943 | +| Recall@5 | 0.8774 | +| Recall@10 | 0.9198 | +| MRR | 0.7548 | +| nDCG@10 | 0.7804 | +| Hit rate@5 | 0.9057 | +| MAP@10 | 0.7280 | +| Context precision@3 | 0.3082 | +| Context recall@3 | 0.8160 | + +### By query type + +A single average hides a change that helps one kind of question and hurts another. +The type comes from `type` in `questions.jsonl`; untagged questions report as +`untagged` rather than disappearing. + +| Type | Questions | Recall@5 | nDCG@10 | Hit rate@5 | MAP@10 | +| --- | --- | --- | --- | --- | --- | +| cross-lingual | 9 | 0.6667 | 0.4173 | 0.6667 | 0.2994 | +| exact | 8 | 1.0000 | 0.8663 | 1.0000 | 0.8229 | +| hard-negative | 7 | 1.0000 | 0.8758 | 1.0000 | 0.8333 | +| multi-hop | 5 | 0.7000 | 0.7218 | 1.0000 | 0.6350 | +| semantic | 21 | 0.9048 | 0.8542 | 0.9048 | 0.8238 | +| zh | 3 | 1.0000 | 1.0000 | 1.0000 | 1.0000 | + +### Unanswerable questions + +These carry no ground truth, so the correct outcome is that retrieval returns nothing. They +are excluded from every metric above — a missing ground truth is not a miss — and reported +here instead. + +**This measures retrieval-level abstention, not the model refusing.** No generator runs in +this harness, so it can show that no candidate passed the threshold; it cannot show that the +final answer would say "not in your sources". A true system refusal rate needs a +generator eval. + +A higher abstention rate is better on this row, the opposite of how every other row reads. +The two sizes are kept apart on purpose: **candidates passing the threshold** can be as high +as `candidateK` (the harness fetches that many to compute `Recall@10`), while **passages in +the context window** is what a user's prompt would actually receive. A large first number +with a small second one means the threshold filters nothing and the window is all noise. + +| Metric | Value | +| --- | --- | +| Unanswerable questions | 25 | +| Retrieval abstained | 0.0000 (0/25) | +| Mean candidates passing the threshold | 20.00 | +| Mean passages in the context window | 3.00 | + +Timing is informational only and is **not** frozen: indexing 5921 ms, query +p50 53.92 ms, p95 85.84 ms on the machine that produced this file. Timing and index size depend on hardware and on the corpus, so they must never be the reason two runs differ. @@ -35,11 +77,17 @@ corpus, so they must never be the reason two runs differ. - A retrieved passage is relevant when its provenance covers a ground-truth block. - **Recall@k** is the share of ground-truth blocks covered by the first `k` passages. -- **Evidence precision@5** is the share of the first `5` - retrieved passages that cover a ground-truth block. This is **retrieval precision**, not - answer citation recall: the harness runs no model and produces no answer. Answer-level - citation correctness is covered by the resolver (#70); a model-driven answer eval is a - separate deliverable. +- **Context precision@3** is the share of the first `3` + retrieved passages that cover a ground-truth block. **Context recall@3** is + the share of the needed ground-truth blocks that made it into that same window. Both are + deterministic: the dataset says which blocks answer the question, so no model is needed to + score the window. Together they are the trade-off a `contextK` decision actually makes — + a wider window finds more and carries more noise. +- This is **retrieval precision/recall, not answer citation recall**: the harness runs no + model and produces no answer. Answer-level citation correctness is covered by the + resolver (#70). Faithfulness, completeness and answer correctness need a generative model + and are **not evaluated here** — the harness runs offline with only the pinned embedding + model, the same constraint that keeps the reranker unmeasured (#170). - Ground truth is expressed in corpus identity (`document` relative path + `block` ordinal + optional `quote`), never a runtime `documentId`/`blockId`. @@ -51,9 +99,13 @@ first-stage width per channel, `contextK` is how many passages the chat prompt t and `threshold` is the similarity floor the app ships. A benchmark that does not mirror those parameters measures a retriever nobody runs. The adopted-change rule is: -> Adopt a change only if Recall@5 improves and nDCG@10 does not regress. A change -> that trades a large latency increase for a marginal recall gain is a product -> decision, not an automatic win, and must be stated as such. +> Adopt a strategy when it improves the **first metric with headroom** — in the order +> Recall@5, nDCG@10, MRR, MAP@10 — and regresses none of the others. A metric already +> at its maximum has no headroom and cannot decide anything; a rule that depends on +> one is unsatisfiable, not strict (#192 child 10). +> +> A change that trades a large latency increase for a marginal quality gain is a +> product decision, not an automatic win, and must be stated as such. A changed result must be reproducible with: diff --git a/docs/eval/paired-v1.6.json b/docs/eval/paired-v1.6.json new file mode 100644 index 0000000..dd1bfc2 --- /dev/null +++ b/docs/eval/paired-v1.6.json @@ -0,0 +1,1254 @@ +{ + "baseline": "v1.6", + "strategies": [ + "dense", + "hybrid" + ], + "splits": { + "validation": { + "questions": 19, + "meanDelta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0.004654087156011809 + }, + "byType": [ + { + "type": "cross-lingual", + "questions": 4, + "improved": 0, + "tied": 4, + "regressed": 0, + "meanDeltaNdcg10": 0, + "meanDeltaRank": 0 + }, + { + "type": "exact", + "questions": 4, + "improved": 2, + "tied": 2, + "regressed": 0, + "meanDeltaNdcg10": 0.05006329887451612, + "meanDeltaRank": 0.5 + }, + { + "type": "hard-negative", + "questions": 2, + "improved": 0, + "tied": 2, + "regressed": 0, + "meanDeltaNdcg10": 0, + "meanDeltaRank": 0 + }, + { + "type": "multi-hop", + "questions": 2, + "improved": 0, + "tied": 2, + "regressed": 0, + "meanDeltaNdcg10": 0.08861964339202388, + "meanDeltaRank": 0 + }, + { + "type": "semantic", + "questions": 6, + "improved": 0, + "tied": 5, + "regressed": 1, + "meanDeltaNdcg10": -0.04817747105298131, + "meanDeltaRank": -0.3333333333333333 + }, + { + "type": "zh", + "questions": 1, + "improved": 0, + "tied": 1, + "regressed": 0, + "meanDeltaNdcg10": 0, + "meanDeltaRank": 0 + } + ], + "pairs": [ + { + "id": "q002", + "type": "exact", + "question": "How many replicate samples are collected at each river station?", + "dense": { + "firstRelevantRank": 3, + "recallAt5": 1, + "ndcgAt10": 0.5 + }, + "hybrid": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "delta": { + "rank": 1, + "recallAt5": 0, + "ndcgAt10": 0.13092975357145753 + }, + "classification": "improved" + }, + { + "id": "q004", + "type": "semantic", + "question": "Why do nickel-rich battery packs need more aggressive thermal management?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q009", + "type": "semantic", + "question": "Why are trees with aggressive surface roots unsuitable for narrow verges?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q011", + "type": "exact", + "question": "Is the salt percentage in fermentation based on vegetable weight or water weight?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q015", + "type": "semantic", + "question": "Why does deep-water oxygen fall while a lake remains stratified?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q018", + "type": "semantic", + "question": "Why is one continuous planted roof layer better than several isolated planted beds?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q021", + "type": "semantic", + "question": "Why is wave energy harder to schedule ahead than tidal energy?", + "dense": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "hybrid": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q023", + "type": "multi-hop", + "question": "How does the river sampling protocol differ from the lake sampling protocol?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 0.5, + "ndcgAt10": 0.6131471927654584 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 0.5, + "ndcgAt10": 0.7903864795495061 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0.17723928678404777 + }, + "classification": "tied" + }, + { + "id": "q028", + "type": "zh", + "question": "茶叶储存的相对湿度上限是多少?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q030", + "type": "cross-lingual", + "question": "为什么潮汐能比风能和太阳能更容易提前安排发电?", + "dense": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "hybrid": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q037", + "type": "cross-lingual", + "question": "推移质为什么比悬移质更难测?", + "dense": { + "firstRelevantRank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "hybrid": { + "firstRelevantRank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "delta": { + "rank": null, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q040", + "type": "cross-lingual", + "question": "富镍电池为什么对散热要求更高?", + "dense": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "hybrid": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q043", + "type": "cross-lingual", + "question": "花期遭遇晚霜,主要受损的是什么?", + "dense": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "hybrid": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q045", + "type": "exact", + "question": "What is the default retry count in Gateway API v2?", + "dense": { + "firstRelevantRank": 4, + "recallAt5": 1, + "ndcgAt10": 0.43067655807339306 + }, + "hybrid": { + "firstRelevantRank": 3, + "recallAt5": 1, + "ndcgAt10": 0.5 + }, + "delta": { + "rank": 1, + "recallAt5": 0, + "ndcgAt10": 0.06932344192660694 + }, + "classification": "improved" + }, + { + "id": "q047", + "type": "hard-negative", + "question": "What is the context window of Gateway API v3, in tokens?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q050", + "type": "semantic", + "question": "Which Gateway API version waits longest between retries after a failed request?", + "dense": { + "firstRelevantRank": 10, + "recallAt5": 0, + "ndcgAt10": 0.2890648263178879 + }, + "hybrid": { + "firstRelevantRank": 12, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "delta": { + "rank": -2, + "recallAt5": 0, + "ndcgAt10": -0.2890648263178879 + }, + "classification": "regressed" + }, + { + "id": "q055", + "type": "exact", + "question": "How many replicate samples are collected at each reservoir monitoring station?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q057", + "type": "hard-negative", + "question": "At what depth below the surface does the coastal programme place its shore-station sensor?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q060", + "type": "multi-hop", + "question": "Compare the sampling interval and replicate count of the reservoir and groundwater programmes.", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + } + ] + }, + "test": { + "questions": 34, + "meanDelta": { + "rank": 0.3939393939393939, + "recallAt5": 0.029411764705882353, + "ndcgAt10": 0.04318215877040841 + }, + "byType": [ + { + "type": "cross-lingual", + "questions": 5, + "improved": 0, + "tied": 5, + "regressed": 0, + "meanDeltaNdcg10": 0, + "meanDeltaRank": 0 + }, + { + "type": "exact", + "questions": 4, + "improved": 0, + "tied": 4, + "regressed": 0, + "meanDeltaNdcg10": 0, + "meanDeltaRank": 0 + }, + { + "type": "hard-negative", + "questions": 5, + "improved": 1, + "tied": 4, + "regressed": 0, + "meanDeltaNdcg10": 0.026185950714291507, + "meanDeltaRank": 0.2 + }, + { + "type": "multi-hop", + "questions": 3, + "improved": 1, + "tied": 2, + "regressed": 0, + "meanDeltaNdcg10": 0.09781329792785898, + "meanDeltaRank": 0.3333333333333333 + }, + { + "type": "semantic", + "questions": 15, + "improved": 3, + "tied": 12, + "regressed": 0, + "meanDeltaNdcg10": 0.06958825005592344, + "meanDeltaRank": 0.7333333333333333 + }, + { + "type": "zh", + "questions": 2, + "improved": 0, + "tied": 2, + "regressed": 0, + "meanDeltaNdcg10": 0, + "meanDeltaRank": 0 + } + ], + "pairs": [ + { + "id": "q001", + "type": "semantic", + "question": "Why is bedload harder to measure than suspended sediment?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q003", + "type": "semantic", + "question": "What is the central trade-off in lithium-ion cell design?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q005", + "type": "semantic", + "question": "What happens once the separator in a battery cell melts?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q006", + "type": "exact", + "question": "At what temperature do honeybees begin to forage?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q007", + "type": "semantic", + "question": "What does a late frost damage during full bloom?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q008", + "type": "semantic", + "question": "Why is a continuous tree canopy more effective at cooling than isolated trees?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q010", + "type": "exact", + "question": "At what temperature is lactic acid fermentation fastest?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q012", + "type": "semantic", + "question": "Why must tidal turbines be sited in places with very fast currents?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q013", + "type": "semantic", + "question": "What is the main environmental concern for tidal energy installations?", + "dense": { + "firstRelevantRank": 11, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "hybrid": { + "firstRelevantRank": 4, + "recallAt5": 1, + "ndcgAt10": 0.43067655807339306 + }, + "delta": { + "rank": 7, + "recallAt5": 1, + "ndcgAt10": 0.43067655807339306 + }, + "classification": "improved" + }, + { + "id": "q014", + "type": "exact", + "question": "In lake monitoring, how is the sampling depth actually recorded?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q016", + "type": "semantic", + "question": "How do supercapacitors hold their charge?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q017", + "type": "semantic", + "question": "Why can solitary bees pollinate a bloom week that is too cold for honeybee hives?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q019", + "type": "semantic", + "question": "Where do acetic acid bacteria sit in a vinegar culture?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q020", + "type": "semantic", + "question": "What happens if a vinegar culture is sealed airtight?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q022", + "type": "exact", + "question": "Where does siting for wave energy devices concentrate, and where does it not?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q024", + "type": "multi-hop", + "question": "A street canopy and a green roof are both said to cool; what surface does each one shade?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q025", + "type": "semantic", + "question": "Which preservation method depends on keeping air away from the food?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q026", + "type": "semantic", + "question": "Why can one cold morning cost a grower the whole crop even when colonies are brought in?", + "dense": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 1, + "recallAt5": 0, + "ndcgAt10": 0.36907024642854247 + }, + "classification": "improved" + }, + { + "id": "q027", + "type": "zh", + "question": "绿茶应该怎样保存才能减缓氧化?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q029", + "type": "zh", + "question": "为什么冷冻保存的茶叶取出后不能立刻打开包装?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q038", + "type": "cross-lingual", + "question": "河流监测中,每个站点要采几份平行样?", + "dense": { + "firstRelevantRank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "hybrid": { + "firstRelevantRank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "delta": { + "rank": null, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q039", + "type": "cross-lingual", + "question": "设计锂离子电芯时最核心的取舍是什么?", + "dense": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "hybrid": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q041", + "type": "cross-lingual", + "question": "电芯隔膜一旦熔化会导致什么后果?", + "dense": { + "firstRelevantRank": 4, + "recallAt5": 1, + "ndcgAt10": 0.43067655807339306 + }, + "hybrid": { + "firstRelevantRank": 4, + "recallAt5": 1, + "ndcgAt10": 0.43067655807339306 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q042", + "type": "cross-lingual", + "question": "蜜蜂大致从什么温度开始出巢觅食?", + "dense": { + "firstRelevantRank": 3, + "recallAt5": 1, + "ndcgAt10": 0.5 + }, + "hybrid": { + "firstRelevantRank": 3, + "recallAt5": 1, + "ndcgAt10": 0.5 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q044", + "type": "cross-lingual", + "question": "为什么成片的树冠比孤立的树降温效果更好?", + "dense": { + "firstRelevantRank": 9, + "recallAt5": 0, + "ndcgAt10": 0.3010299956639812 + }, + "hybrid": { + "firstRelevantRank": 9, + "recallAt5": 0, + "ndcgAt10": 0.3010299956639812 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q046", + "type": "hard-negative", + "question": "What is the default request timeout of Gateway API v1?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q048", + "type": "hard-negative", + "question": "What is the default rate limit of Gateway API v2?", + "dense": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "hybrid": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q049", + "type": "hard-negative", + "question": "What is the default request timeout of Gateway API v3?", + "dense": { + "firstRelevantRank": 3, + "recallAt5": 1, + "ndcgAt10": 0.5 + }, + "hybrid": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "delta": { + "rank": 1, + "recallAt5": 0, + "ndcgAt10": 0.13092975357145753 + }, + "classification": "improved" + }, + { + "id": "q051", + "type": "multi-hop", + "question": "Compare the default timeouts and rate limits of Gateway API v1 and v3.", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 0.75, + "ndcgAt10": 0.6366824387328317 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 0.75, + "ndcgAt10": 0.6366824387328317 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q052", + "type": "multi-hop", + "question": "Compare the retry count and context window of Gateway API v2 and v3.", + "dense": { + "firstRelevantRank": 2, + "recallAt5": 0.25, + "ndcgAt10": 0.35914753008966527 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 0.25, + "ndcgAt10": 0.6525874238732422 + }, + "delta": { + "rank": 1, + "recallAt5": 0, + "ndcgAt10": 0.29343989378357693 + }, + "classification": "improved" + }, + { + "id": "q056", + "type": "hard-negative", + "question": "How many replicate samples are collected at each estuary transect station?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q058", + "type": "hard-negative", + "question": "At what depth below the water table do groundwater boreholes carry their pressure transducer?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q059", + "type": "semantic", + "question": "Which monitoring programme samples most frequently?", + "dense": { + "firstRelevantRank": 5, + "recallAt5": 1, + "ndcgAt10": 0.38685280723454163 + }, + "hybrid": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "delta": { + "rank": 3, + "recallAt5": 0, + "ndcgAt10": 0.2440769463369159 + }, + "classification": "improved" + }, + { + "id": "q061", + "type": "semantic", + "question": "Why does the estuary programme take more replicate samples than the other monitoring programmes?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + } + ] + } + } +} diff --git a/docs/eval/paired-v1.6.md b/docs/eval/paired-v1.6.md new file mode 100644 index 0000000..dc185dd --- /dev/null +++ b/docs/eval/paired-v1.6.md @@ -0,0 +1,97 @@ +# Paired dense ↔ hybrid deltas — v1.6 (#192) + +Generated by `node scripts/eval-paired.mjs`. Numbers are harness output; do not edit them by hand. + +## What this answers + +The aggregate comparison says hybrid is ahead on the full set and level on `validation`. +An average cannot say whether that is a broad small gain or a few rescued cases, and those +two readings imply different next steps. This is the per-question pair. + +Metrics come from `src/main/eval/metrics.ts`, so a delta cannot disagree with the metric it +is a delta of. Unanswerable questions are excluded — they have no rank to compare, and the +abstention diagnostics cover them. + +`validation` is the side that **selects**; the `test` breakdown below **explains** the +difference that was observed and must not be used to choose. Nothing here changes the +shipped strategy. + +## Validation (this side selects) + +Chosen on this side, so a win here is eligible to inform a decision. + +Questions compared: 19. Mean ΔnDCG@10 +0.0047, +mean Δrank 0.00. + +### By query type + +| Type | n | improved | tied | regressed | mean ΔnDCG@10 | mean Δrank | +| --- | --- | --- | --- | --- | --- | --- | +| cross-lingual | 4 | 0 | 4 | 0 | 0.0000 | 0.00 | +| exact | 4 | 2 | 2 | 0 | +0.0501 | +0.50 | +| hard-negative | 2 | 0 | 2 | 0 | 0.0000 | 0.00 | +| multi-hop | 2 | 0 | 2 | 0 | +0.0886 | 0.00 | +| semantic | 6 | 0 | 5 | 1 | -0.0482 | -0.33 | +| zh | 1 | 0 | 1 | 0 | 0.0000 | 0.00 | + +### Biggest rank wins + +| id | type | dense rank | hybrid rank | Δrank | ΔnDCG@10 | +| --- | --- | --- | --- | --- | --- | +| q002 | exact | 3 | 2 | +1 | +0.1309 | +| q045 | exact | 4 | 3 | +1 | +0.0693 | + +### Biggest rank regressions + +| id | type | dense rank | hybrid rank | Δrank | ΔnDCG@10 | +| --- | --- | --- | --- | --- | --- | +| q050 | semantic | 10 | 12 | -2 | -0.2891 | + +**Found by hybrid, missed by dense**: none. + +**Lost by hybrid, found by dense**: none. + +## Test (explanatory only) + +Not used to choose anything. Present because the aggregate difference this is meant to explain was measured here. + +Questions compared: 34. Mean ΔnDCG@10 +0.0432, +mean Δrank +0.39. + +### By query type + +| Type | n | improved | tied | regressed | mean ΔnDCG@10 | mean Δrank | +| --- | --- | --- | --- | --- | --- | --- | +| cross-lingual | 5 | 0 | 5 | 0 | 0.0000 | 0.00 | +| exact | 4 | 0 | 4 | 0 | 0.0000 | 0.00 | +| hard-negative | 5 | 1 | 4 | 0 | +0.0262 | +0.20 | +| multi-hop | 3 | 1 | 2 | 0 | +0.0978 | +0.33 | +| semantic | 15 | 3 | 12 | 0 | +0.0696 | +0.73 | +| zh | 2 | 0 | 2 | 0 | 0.0000 | 0.00 | + +### Biggest rank wins + +| id | type | dense rank | hybrid rank | Δrank | ΔnDCG@10 | +| --- | --- | --- | --- | --- | --- | +| q013 | semantic | 11 | 4 | +7 | +0.4307 | +| q059 | semantic | 5 | 2 | +3 | +0.2441 | +| q026 | semantic | 2 | 1 | +1 | +0.3691 | +| q049 | hard-negative | 3 | 2 | +1 | +0.1309 | +| q052 | multi-hop | 2 | 1 | +1 | +0.2934 | + +### Biggest rank regressions + +| id | type | dense rank | hybrid rank | Δrank | ΔnDCG@10 | +| --- | --- | --- | --- | --- | --- | +| — | — | — | — | + +**Found by hybrid, missed by dense**: none. + +**Lost by hybrid, found by dense**: none. + +## Reproduce + +```bash +npm run eval:prepare # one-time, networked model bootstrap +npm run eval:paired # offline; rewrites this file +``` diff --git a/docs/eval/retrieval-v1.6.json b/docs/eval/retrieval-v1.6.json new file mode 100644 index 0000000..4bac003 --- /dev/null +++ b/docs/eval/retrieval-v1.6.json @@ -0,0 +1,93 @@ +{ + "baseline": "v1.6", + "split": { + "selects": "validation", + "reports": "test", + "manifest": "eval/splits.json" + }, + "decision": { + "primary": "recallAt5", + "saturated": [], + "winner": null + }, + "validation": [ + { + "id": "dense", + "label": "dense (vector)", + "split": "validation", + "questions": 34, + "answerableCount": 19, + "chunking": "1000/100", + "chunkCount": 53, + "recallAt1": 0.513158, + "recallAt5": 0.868421, + "recallAt10": 0.921053, + "mrr": 0.720175, + "ndcgAt10": 0.755611, + "hitRateAt5": 0.894737, + "mapAt10": 0.69386, + "contextPrecision": 0.315789, + "contextRecall": 0.802632, + "latencyP95Ms": 13.56 + }, + { + "id": "sparse", + "label": "sparse (BM25)", + "split": "validation", + "questions": 34, + "answerableCount": 19, + "chunking": "1000/100", + "chunkCount": 53, + "recallAt1": 0.460526, + "recallAt5": 0.671053, + "recallAt10": 0.684211, + "mrr": 0.600877, + "ndcgAt10": 0.610495, + "hitRateAt5": 0.684211, + "mapAt10": 0.576817, + "contextPrecision": 0.263158, + "contextRecall": 0.657895, + "latencyP95Ms": 4.25 + }, + { + "id": "hybrid", + "label": "hybrid (RRF of dense + BM25)", + "split": "validation", + "questions": 34, + "answerableCount": 19, + "chunking": "1000/100", + "chunkCount": 53, + "recallAt1": 0.513158, + "recallAt5": 0.868421, + "recallAt10": 0.894737, + "mrr": 0.732456, + "ndcgAt10": 0.760265, + "hitRateAt5": 0.894737, + "mapAt10": 0.707018, + "contextPrecision": 0.333333, + "contextRecall": 0.855263, + "latencyP95Ms": 15.87 + } + ], + "test": [ + { + "id": "dense", + "label": "dense (vector)", + "split": "test", + "questions": 44, + "answerableCount": 34, + "chunking": "1000/100", + "chunkCount": 53, + "recallAt1": 0.639706, + "recallAt5": 0.882353, + "recallAt10": 0.919118, + "mrr": 0.774079, + "ndcgAt10": 0.794329, + "hitRateAt5": 0.911765, + "mapAt10": 0.747141, + "contextPrecision": 0.303922, + "contextRecall": 0.823529, + "latencyP95Ms": 14.44 + } + ] +} diff --git a/docs/eval/retrieval-v1.6.md b/docs/eval/retrieval-v1.6.md new file mode 100644 index 0000000..3b823fd --- /dev/null +++ b/docs/eval/retrieval-v1.6.md @@ -0,0 +1,59 @@ +# Retrieval experiments — v1.6 (#77, #192) + +Generated by `node scripts/eval-retrieval.mjs`. Numbers are harness output; do not edit them by hand. + +## What was measured + +Each strategy runs the real harness on the **validation** split of +`eval/splits.json`, with chunking held fixed at 1000/100. + +The strategy is then chosen **there**, and only the chosen one (plus the shipped default) +is re-run on **test**. The choice never sees `test`; `test` only reports. The previous +version decided on `split = all`, which scored the choice on the questions it was fitted +to. + +## Validation — this is where the choice happens + +| Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Context P | n | p95 | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | +| dense (vector) | 0.5132 | 0.8684 | 0.7202 | 0.7556 | 0.6939 | 0.3158 | 34 | 13.56 ms | +| sparse (BM25) | 0.4605 | 0.6711 | 0.6009 | 0.6105 | 0.5768 | 0.2632 | 34 | 4.25 ms | +| hybrid (RRF of dense + BM25) | 0.5132 | 0.8684 | 0.7325 | 0.7603 | 0.7070 | 0.3333 | 34 | 15.87 ms | + +## Test — reported, not selected + +| Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Context P | n | p95 | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | +| dense (vector) | 0.6397 | 0.8824 | 0.7741 | 0.7943 | 0.7471 | 0.3039 | 44 | 14.44 ms | + +## Not evaluated + +**Reranking.** The issue lists "hybrid + reranker" as a step, but a cross-encoder +model is not available offline and inventing its numbers would defeat the point of +the harness. It stays open until a model can be pinned the way the embedding model +is. + +## Adoption rule + +> Adopt a strategy when it improves the **first metric with headroom** — in the order +> `recallAt5`, `nDCG@10`, `MRR`, `MAP@10` — and regresses none of the others. A metric +> already at its maximum has no headroom and cannot decide anything; a rule that depends +> on one is unsatisfiable, not strict (#192 child 10). +> +> The rule is applied on `validation`. A change that trades a large latency increase for a +> marginal quality gain is a product decision, not an automatic win. + +## Outcome + +**No strategy cleared the rule on validation**, so there is no adoption candidate and +`test` reports the shipped strategy only. No metric in the rule is saturated on the validation split. + +A negative result is the point of the experiment: it is the measurement that says the extra +machinery is not worth its cost on this corpus, not a failure to deliver. + +## Reproduce + +```bash +npm run eval:prepare # one-time, networked model bootstrap +npm run eval:retrieval # offline; runs every strategy and rewrites this file +``` diff --git a/docs/eval/scores-v1.6.json b/docs/eval/scores-v1.6.json new file mode 100644 index 0000000..4cfd04e --- /dev/null +++ b/docs/eval/scores-v1.6.json @@ -0,0 +1,342 @@ +{ + "baseline": "v1.6", + "split": "validation", + "candidateK": 500, + "threshold": 0, + "indexSize": 53, + "counts": { + "answerable": 19, + "unanswerable": 15 + }, + "distributions": { + "bestRelevant": { + "p0": 0.880777, + "p10": 0.89056, + "p25": 0.912001, + "p50": 0.935891, + "p75": 0.942776, + "p90": 0.953038, + "p100": 0.953664 + }, + "worstRelevant": { + "p0": 0.880777, + "p10": 0.89056, + "p25": 0.908589, + "p50": 0.935467, + "p75": 0.942776, + "p90": 0.953038, + "p100": 0.953664 + }, + "bestNonRelevant": { + "p0": 0.890443, + "p10": 0.908132, + "p25": 0.917792, + "p50": 0.924346, + "p75": 0.933555, + "p90": 0.9386, + "p100": 0.945918 + }, + "margin": { + "p0": -0.02942099999999992, + "p10": -0.027232000000000034, + "p25": -0.00922400000000001, + "p50": 0.0026120000000000587, + "p75": 0.00807100000000005, + "p90": 0.032869999999999955, + "p100": 0.06226200000000004 + }, + "unanswerableMax": { + "p0": 0.897848, + "p10": 0.908616, + "p25": 0.914179, + "p50": 0.927896, + "p75": 0.936526, + "p90": 0.942441, + "p100": 0.943454 + } + }, + "byType": { + "exact": { + "questions": 4, + "bestRelevant": { + "p0": 0.912001, + "p10": 0.912001, + "p25": 0.912001, + "p50": 0.929376, + "p75": 0.936747, + "p90": 0.953038, + "p100": 0.953038 + }, + "worstRelevant": { + "p0": 0.912001, + "p10": 0.912001, + "p25": 0.912001, + "p50": 0.929376, + "p75": 0.936747, + "p90": 0.953038, + "p100": 0.953038 + } + }, + "semantic": { + "questions": 6, + "bestRelevant": { + "p0": 0.920784, + "p10": 0.920784, + "p25": 0.935467, + "p50": 0.939466, + "p75": 0.945637, + "p90": 0.953664, + "p100": 0.953664 + }, + "worstRelevant": { + "p0": 0.920784, + "p10": 0.920784, + "p25": 0.935467, + "p50": 0.939466, + "p75": 0.945637, + "p90": 0.953664, + "p100": 0.953664 + } + }, + "multi-hop": { + "questions": 2, + "bestRelevant": { + "p0": 0.924784, + "p10": 0.924784, + "p25": 0.924784, + "p50": 0.924784, + "p75": 0.93756, + "p90": 0.93756, + "p100": 0.93756 + }, + "worstRelevant": { + "p0": 0.903304, + "p10": 0.903304, + "p25": 0.903304, + "p50": 0.903304, + "p75": 0.933129, + "p90": 0.933129, + "p100": 0.933129 + } + }, + "zh": { + "questions": 1, + "bestRelevant": { + "p0": 0.952705, + "p10": 0.952705, + "p25": 0.952705, + "p50": 0.952705, + "p75": 0.952705, + "p90": 0.952705, + "p100": 0.952705 + }, + "worstRelevant": { + "p0": 0.952705, + "p10": 0.952705, + "p25": 0.952705, + "p50": 0.952705, + "p75": 0.952705, + "p90": 0.952705, + "p100": 0.952705 + } + }, + "cross-lingual": { + "questions": 4, + "bestRelevant": { + "p0": 0.880777, + "p10": 0.880777, + "p25": 0.880777, + "p50": 0.89056, + "p75": 0.904196, + "p90": 0.908589, + "p100": 0.908589 + }, + "worstRelevant": { + "p0": 0.880777, + "p10": 0.880777, + "p25": 0.880777, + "p50": 0.89056, + "p75": 0.904196, + "p90": 0.908589, + "p100": 0.908589 + } + }, + "hard-negative": { + "questions": 2, + "bestRelevant": { + "p0": 0.935891, + "p10": 0.935891, + "p25": 0.935891, + "p50": 0.935891, + "p75": 0.942776, + "p90": 0.942776, + "p100": 0.942776 + }, + "worstRelevant": { + "p0": 0.935891, + "p10": 0.935891, + "p25": 0.935891, + "p50": 0.935891, + "p75": 0.942776, + "p90": 0.942776, + "p100": 0.942776 + } + } + }, + "overlap": { + "worstRelevantP10": 0.89056, + "bestNonRelevantP90": 0.9386, + "unanswerableMaxP50": 0.927896, + "unanswerableMaxP90": 0.942441 + }, + "separable": false, + "curve": [ + { + "threshold": 0.5, + "cosine": 0, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.525, + "cosine": 0.05, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.55, + "cosine": 0.1, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.575, + "cosine": 0.15, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.6, + "cosine": 0.2, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.625, + "cosine": 0.25, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.65, + "cosine": 0.3, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.675, + "cosine": 0.35, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.7, + "cosine": 0.4, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.725, + "cosine": 0.45, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.75, + "cosine": 0.5, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.775, + "cosine": 0.55, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.8, + "cosine": 0.6, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.825, + "cosine": 0.65, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.85, + "cosine": 0.7, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.875, + "cosine": 0.75, + "hitRate": 1, + "fullRecallRate": 1, + "abstentionRate": 0 + }, + { + "threshold": 0.9, + "cosine": 0.8, + "hitRate": 0.8947368421052632, + "fullRecallRate": 0.8947368421052632, + "abstentionRate": 0.06666666666666667 + }, + { + "threshold": 0.925, + "cosine": 0.85, + "hitRate": 0.631578947368421, + "fullRecallRate": 0.631578947368421, + "abstentionRate": 0.4 + }, + { + "threshold": 0.95, + "cosine": 0.9, + "hitRate": 0.15789473684210525, + "fullRecallRate": 0.15789473684210525, + "abstentionRate": 1 + }, + { + "threshold": 0.975, + "cosine": 0.95, + "hitRate": 0, + "fullRecallRate": 0, + "abstentionRate": 1 + }, + { + "threshold": 1, + "cosine": 1, + "hitRate": 0, + "fullRecallRate": 0, + "abstentionRate": 1 + } + ] +} diff --git a/docs/eval/scores-v1.6.md b/docs/eval/scores-v1.6.md new file mode 100644 index 0000000..6bff4b4 --- /dev/null +++ b/docs/eval/scores-v1.6.md @@ -0,0 +1,128 @@ +# Dense score diagnostics — v1.6 (#192) + +Generated by `node scripts/eval-scores.mjs`. Numbers are harness output; do not edit them by hand. + +## Read this first: the score is not a cosine + +`SQLiteVectorStore` computes + +```text +score = 1 - distance / 2 +distance = 1 - cosine (sqlite-vec, distance_metric=cosine) +=> score = (1 + cosine) / 2 +``` + +so the configured `threshold` is an **affine map of the cosine**, not the cosine: + +| threshold | raw cosine | +| --- | --- | +| 0.3 | -0.4 | +| 0.4 | -0.2 | +| **0.5 (shipped)** | **0.0** | +| 0.6 | 0.2 | +| 0.8 | 0.6 | +| 1.0 | 1.0 | + +The shipped `threshold = 0.5` means **cosine ≥ 0**, which is very permissive. Every +threshold row below carries both columns so the two never get confused again, and every +mention of "the sweep looked too low" has to be read through this table. + +## What was measured + +Dense only, `validation` split only, `threshold = 0`, `candidateK = 500` (above the +index size, so every chunk is scored for every query), `contextK = 3`. + +- answerable questions: 19 +- unanswerable questions: 15 +- index size: 53 chunks + +Hybrid is deliberately excluded: its `score` is an RRF value (`1 / (60 + rank)`) and is not +on the same scale as a normalised cosine. + +## Distributions (score, with raw cosine in brackets) + +`best relevant` is the highest-scoring passage that covers ground truth; `worst relevant` +is the lowest one that still has to survive for the question to be fully answered; +`best non-relevant` is the highest-scoring passage that covers nothing; `margin` is the +first minus the third. + +The margin is an **oracle** quantity: at runtime nothing knows which result is relevant, so +it describes how much the score separates the two — it is not a signal a product could use. +Reading it as a candidate mechanism is the mistake the runtime-signal evaluation exists to +avoid. + +Where a row shows a raw cosine in brackets: an **absolute** score maps as +`cosine = 2·score − 1`, while a **margin** maps as `Δcosine = 2·Δscore` because the +`+1` cancels. The `margin` row uses the latter, the others the former. + +| Distribution | n | min | p10 | p25 | p50 | p75 | p90 | max | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | +| best relevant | 19 | 0.8808 (0.762) | 0.8906 (0.781) | 0.9120 (0.824) | 0.9359 (0.872) | 0.9428 (0.886) | 0.9530 (0.906) | 0.9537 (0.907) | +| worst relevant | 19 | 0.8808 (0.762) | 0.8906 (0.781) | 0.9086 (0.817) | 0.9355 (0.871) | 0.9428 (0.886) | 0.9530 (0.906) | 0.9537 (0.907) | +| best non-relevant | 19 | 0.8904 (0.781) | 0.9081 (0.816) | 0.9178 (0.836) | 0.9243 (0.849) | 0.9336 (0.867) | 0.9386 (0.877) | 0.9459 (0.892) | +| margin (best rel − best non-rel) | 19 | -0.0294 (-0.059) | -0.0272 (-0.054) | -0.0092 (-0.018) | 0.0026 (0.005) | 0.0081 (0.016) | 0.0329 (0.066) | 0.0623 (0.125) | +| unanswerable max candidate | 15 | 0.8978 (0.796) | 0.9086 (0.817) | 0.9142 (0.828) | 0.9279 (0.856) | 0.9365 (0.873) | 0.9424 (0.885) | 0.9435 (0.887) | + +## By query type + +`best relevant` p50 and p10, and `worst relevant` p10 — the last is the one that decides +whether a cross-lingual question survives a threshold that a semantic question tolerates. + +| Type | n | best rel p50 | best rel p10 | worst rel p10 | +| --- | --- | --- | --- | --- | +| cross-lingual | 4 | 0.8906 | 0.8808 | 0.8808 | +| exact | 4 | 0.9294 | 0.9120 | 0.9120 | +| hard-negative | 2 | 0.9359 | 0.9359 | 0.9359 | +| multi-hop | 2 | 0.9248 | 0.9248 | 0.9033 | +| semantic | 6 | 0.9395 | 0.9208 | 0.9208 | +| zh | 1 | 0.9527 | 0.9527 | 0.9527 | + +## Threshold curve + +For each candidate threshold: **hit** = share of answerable questions that still have some +relevant passage; **full recall** = share whose every ground-truth block is still covered; +**abstain** = share of unanswerable questions that now return nothing. + +| threshold (score) | raw cosine | answerable hit | answerable full recall | unanswerable abstain | +| --- | --- | --- | --- | --- | +| 0.500 | 0.000 | 1.0000 | 1.0000 | 0.0000 | +| 0.525 | 0.050 | 1.0000 | 1.0000 | 0.0000 | +| 0.550 | 0.100 | 1.0000 | 1.0000 | 0.0000 | +| 0.575 | 0.150 | 1.0000 | 1.0000 | 0.0000 | +| 0.600 | 0.200 | 1.0000 | 1.0000 | 0.0000 | +| 0.625 | 0.250 | 1.0000 | 1.0000 | 0.0000 | +| 0.650 | 0.300 | 1.0000 | 1.0000 | 0.0000 | +| 0.675 | 0.350 | 1.0000 | 1.0000 | 0.0000 | +| 0.700 | 0.400 | 1.0000 | 1.0000 | 0.0000 | +| 0.725 | 0.450 | 1.0000 | 1.0000 | 0.0000 | +| 0.750 | 0.500 | 1.0000 | 1.0000 | 0.0000 | +| 0.775 | 0.550 | 1.0000 | 1.0000 | 0.0000 | +| 0.800 | 0.600 | 1.0000 | 1.0000 | 0.0000 | +| 0.825 | 0.650 | 1.0000 | 1.0000 | 0.0000 | +| 0.850 | 0.700 | 1.0000 | 1.0000 | 0.0000 | +| 0.875 | 0.750 | 1.0000 | 1.0000 | 0.0000 | +| 0.900 | 0.800 | 0.8947 | 0.8947 | 0.0667 | +| 0.925 | 0.850 | 0.6316 | 0.6316 | 0.4000 | +| 0.950 | 0.900 | 0.1579 | 0.1579 | 1.0000 | +| 0.975 | 0.950 | 0.0000 | 0.0000 | 1.0000 | +| 1.000 | 1.000 | 0.0000 | 0.0000 | 1.0000 | + +## The separation question + +The threshold decision turns on whether these overlap: + +| Landmark | score | raw cosine | +| --- | --- | --- | +| worst relevant, p10 | 0.8906 | 0.781 | +| best non-relevant, p90 | 0.9386 | 0.877 | +| unanswerable max, p50 | 0.9279 | 0.856 | +| unanswerable max, p90 | 0.9424 | 0.885 | + +**The distributions **overlap**, so a higher threshold buys abstention by giving up required relevant passages. If the curve above shows abstention rising only as full recall falls, then the honest conclusion is that **a single dense similarity threshold cannot carry both recall and abstention** — and the next mechanism to evaluate is not a finer threshold grid but a different signal (reranker score, top1−top2 margin, per-query thresholds, or claim-level answerability). + +## Reproduce + +```bash +npm run eval:prepare # one-time, networked model bootstrap +npm run eval:scores # offline; rewrites this file +``` diff --git a/docs/eval/sweep-v1.6.json b/docs/eval/sweep-v1.6.json new file mode 100644 index 0000000..ba157c1 --- /dev/null +++ b/docs/eval/sweep-v1.6.json @@ -0,0 +1,401 @@ +{ + "baseline": "v1.6", + "rows": [ + { + "strategy": "dense", + "candidateK": 5, + "contextK": 3, + "recallAt5": 0.877358, + "ndcgAt10": 0.767186, + "mapAt10": 0.723113, + "contextPrecision": 0.308176, + "contextRecall": 0.816038, + "noResultRate": 0, + "meanContextChars": 2311.811320754717, + "unanswerableQuestions": 25, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 5, + "unanswerableContextPassages": 3, + "chunkCount": 53, + "latencyP95Ms": 16.18 + }, + { + "strategy": "dense", + "candidateK": 5, + "contextK": 5, + "recallAt5": 0.877358, + "ndcgAt10": 0.767186, + "mapAt10": 0.723113, + "contextPrecision": 0.2, + "contextRecall": 0.877358, + "noResultRate": 0, + "meanContextChars": 4007.0377358490564, + "unanswerableQuestions": 25, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 5, + "unanswerableContextPassages": 5, + "chunkCount": 53, + "latencyP95Ms": 86.91 + }, + { + "strategy": "dense", + "candidateK": 10, + "contextK": 3, + "recallAt5": 0.877358, + "ndcgAt10": 0.780449, + "mapAt10": 0.72804, + "contextPrecision": 0.308176, + "contextRecall": 0.816038, + "noResultRate": 0, + "meanContextChars": 2311.811320754717, + "unanswerableQuestions": 25, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 10, + "unanswerableContextPassages": 3, + "chunkCount": 53, + "latencyP95Ms": 100.17 + }, + { + "strategy": "dense", + "candidateK": 10, + "contextK": 5, + "recallAt5": 0.877358, + "ndcgAt10": 0.780449, + "mapAt10": 0.72804, + "contextPrecision": 0.2, + "contextRecall": 0.877358, + "noResultRate": 0, + "meanContextChars": 4007.0377358490564, + "unanswerableQuestions": 25, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 10, + "unanswerableContextPassages": 5, + "chunkCount": 53, + "latencyP95Ms": 81.38 + }, + { + "strategy": "dense", + "candidateK": 10, + "contextK": 8, + "recallAt5": 0.877358, + "ndcgAt10": 0.780449, + "mapAt10": 0.72804, + "contextPrecision": 0.127358, + "contextRecall": 0.877358, + "noResultRate": 0, + "meanContextChars": 6658.698113207547, + "unanswerableQuestions": 25, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 10, + "unanswerableContextPassages": 8, + "chunkCount": 53, + "latencyP95Ms": 18.17 + }, + { + "strategy": "dense", + "candidateK": 20, + "contextK": 3, + "recallAt5": 0.877358, + "ndcgAt10": 0.780449, + "mapAt10": 0.72804, + "contextPrecision": 0.308176, + "contextRecall": 0.816038, + "noResultRate": 0, + "meanContextChars": 2311.811320754717, + "unanswerableQuestions": 25, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 20, + "unanswerableContextPassages": 3, + "chunkCount": 53, + "latencyP95Ms": 17.01 + }, + { + "strategy": "dense", + "candidateK": 20, + "contextK": 5, + "recallAt5": 0.877358, + "ndcgAt10": 0.780449, + "mapAt10": 0.72804, + "contextPrecision": 0.2, + "contextRecall": 0.877358, + "noResultRate": 0, + "meanContextChars": 4007.0377358490564, + "unanswerableQuestions": 25, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 20, + "unanswerableContextPassages": 5, + "chunkCount": 53, + "latencyP95Ms": 97.74 + }, + { + "strategy": "dense", + "candidateK": 20, + "contextK": 8, + "recallAt5": 0.877358, + "ndcgAt10": 0.780449, + "mapAt10": 0.72804, + "contextPrecision": 0.127358, + "contextRecall": 0.877358, + "noResultRate": 0, + "meanContextChars": 6658.698113207547, + "unanswerableQuestions": 25, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 20, + "unanswerableContextPassages": 8, + "chunkCount": 53, + "latencyP95Ms": 85.3 + }, + { + "strategy": "dense", + "candidateK": 40, + "contextK": 3, + "recallAt5": 0.877358, + "ndcgAt10": 0.780449, + "mapAt10": 0.72804, + "contextPrecision": 0.308176, + "contextRecall": 0.816038, + "noResultRate": 0, + "meanContextChars": 2311.811320754717, + "unanswerableQuestions": 25, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 40, + "unanswerableContextPassages": 3, + "chunkCount": 53, + "latencyP95Ms": 57.28 + }, + { + "strategy": "dense", + "candidateK": 40, + "contextK": 5, + "recallAt5": 0.877358, + "ndcgAt10": 0.780449, + "mapAt10": 0.72804, + "contextPrecision": 0.2, + "contextRecall": 0.877358, + "noResultRate": 0, + "meanContextChars": 4007.0377358490564, + "unanswerableQuestions": 25, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 40, + "unanswerableContextPassages": 5, + "chunkCount": 53, + "latencyP95Ms": 16.64 + }, + { + "strategy": "dense", + "candidateK": 40, + "contextK": 8, + "recallAt5": 0.877358, + "ndcgAt10": 0.780449, + "mapAt10": 0.72804, + "contextPrecision": 0.127358, + "contextRecall": 0.877358, + "noResultRate": 0, + "meanContextChars": 6658.698113207547, + "unanswerableQuestions": 25, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 40, + "unanswerableContextPassages": 8, + "chunkCount": 53, + "latencyP95Ms": 99.32 + }, + { + "strategy": "hybrid", + "candidateK": 5, + "contextK": 3, + "recallAt5": 0.910377, + "ndcgAt10": 0.810136, + "mapAt10": 0.768239, + "contextPrecision": 0.320755, + "contextRecall": 0.853774, + "noResultRate": 0, + "meanContextChars": 2307.3207547169814, + "unanswerableQuestions": 25, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 5, + "unanswerableContextPassages": 3, + "chunkCount": 53, + "latencyP95Ms": 90.4 + }, + { + "strategy": "hybrid", + "candidateK": 5, + "contextK": 5, + "recallAt5": 0.910377, + "ndcgAt10": 0.810136, + "mapAt10": 0.768239, + "contextPrecision": 0.211321, + "contextRecall": 0.910377, + "noResultRate": 0, + "meanContextChars": 3936.566037735849, + "unanswerableQuestions": 25, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 5, + "unanswerableContextPassages": 5, + "chunkCount": 53, + "latencyP95Ms": 89.63 + }, + { + "strategy": "hybrid", + "candidateK": 10, + "contextK": 3, + "recallAt5": 0.896226, + "ndcgAt10": 0.806807, + "mapAt10": 0.757966, + "contextPrecision": 0.320755, + "contextRecall": 0.853774, + "noResultRate": 0, + "meanContextChars": 2321.264150943396, + "unanswerableQuestions": 25, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 10, + "unanswerableContextPassages": 3, + "chunkCount": 53, + "latencyP95Ms": 83.59 + }, + { + "strategy": "hybrid", + "candidateK": 10, + "contextK": 5, + "recallAt5": 0.896226, + "ndcgAt10": 0.806807, + "mapAt10": 0.757966, + "contextPrecision": 0.203774, + "contextRecall": 0.896226, + "noResultRate": 0, + "meanContextChars": 3985.811320754717, + "unanswerableQuestions": 25, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 10, + "unanswerableContextPassages": 5, + "chunkCount": 53, + "latencyP95Ms": 89.79 + }, + { + "strategy": "hybrid", + "candidateK": 10, + "contextK": 8, + "recallAt5": 0.896226, + "ndcgAt10": 0.806807, + "mapAt10": 0.757966, + "contextPrecision": 0.132075, + "contextRecall": 0.900943, + "noResultRate": 0, + "meanContextChars": 6547.641509433963, + "unanswerableQuestions": 25, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 10, + "unanswerableContextPassages": 8, + "chunkCount": 53, + "latencyP95Ms": 14.96 + }, + { + "strategy": "hybrid", + "candidateK": 20, + "contextK": 3, + "recallAt5": 0.896226, + "ndcgAt10": 0.809819, + "mapAt10": 0.760469, + "contextPrecision": 0.320755, + "contextRecall": 0.853774, + "noResultRate": 0, + "meanContextChars": 2322.830188679245, + "unanswerableQuestions": 25, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 20, + "unanswerableContextPassages": 3, + "chunkCount": 53, + "latencyP95Ms": 82.31 + }, + { + "strategy": "hybrid", + "candidateK": 20, + "contextK": 5, + "recallAt5": 0.896226, + "ndcgAt10": 0.809819, + "mapAt10": 0.760469, + "contextPrecision": 0.203774, + "contextRecall": 0.896226, + "noResultRate": 0, + "meanContextChars": 4011.377358490566, + "unanswerableQuestions": 25, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 20, + "unanswerableContextPassages": 5, + "chunkCount": 53, + "latencyP95Ms": 18.62 + }, + { + "strategy": "hybrid", + "candidateK": 20, + "contextK": 8, + "recallAt5": 0.896226, + "ndcgAt10": 0.809819, + "mapAt10": 0.760469, + "contextPrecision": 0.132075, + "contextRecall": 0.90566, + "noResultRate": 0, + "meanContextChars": 6660.264150943396, + "unanswerableQuestions": 25, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 20, + "unanswerableContextPassages": 8, + "chunkCount": 53, + "latencyP95Ms": 16.13 + }, + { + "strategy": "hybrid", + "candidateK": 40, + "contextK": 3, + "recallAt5": 0.896226, + "ndcgAt10": 0.809819, + "mapAt10": 0.760469, + "contextPrecision": 0.320755, + "contextRecall": 0.853774, + "noResultRate": 0, + "meanContextChars": 2322.830188679245, + "unanswerableQuestions": 25, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 40, + "unanswerableContextPassages": 3, + "chunkCount": 53, + "latencyP95Ms": 97.73 + }, + { + "strategy": "hybrid", + "candidateK": 40, + "contextK": 5, + "recallAt5": 0.896226, + "ndcgAt10": 0.809819, + "mapAt10": 0.760469, + "contextPrecision": 0.203774, + "contextRecall": 0.896226, + "noResultRate": 0, + "meanContextChars": 3989.735849056604, + "unanswerableQuestions": 25, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 40, + "unanswerableContextPassages": 5, + "chunkCount": 53, + "latencyP95Ms": 100.05 + }, + { + "strategy": "hybrid", + "candidateK": 40, + "contextK": 8, + "recallAt5": 0.896226, + "ndcgAt10": 0.809819, + "mapAt10": 0.760469, + "contextPrecision": 0.132075, + "contextRecall": 0.90566, + "noResultRate": 0, + "meanContextChars": 6660.792452830188, + "unanswerableQuestions": 25, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 40, + "unanswerableContextPassages": 8, + "chunkCount": 53, + "latencyP95Ms": 16.94 + } + ] +} diff --git a/docs/eval/sweep-v1.6.md b/docs/eval/sweep-v1.6.md new file mode 100644 index 0000000..af8cb92 --- /dev/null +++ b/docs/eval/sweep-v1.6.md @@ -0,0 +1,86 @@ +# Parameter sweep — v1.6 (#192) + +Generated by `node scripts/eval-sweep.mjs`. Numbers are harness output; do not edit them by hand. + +## What was measured + +The real harness, the same corpus, chunking held fixed, over +dense / hybrid × candidateK {5, 10, 20, 40} × contextK {3, 5, 8} — 22 runs. +Each row differs from its neighbour in one parameter. + +2 further cell(s) were **skipped** because `contextK > candidateK`; see below. + +| Strategy | candidateK | contextK | Recall@5 | nDCG@10 | MAP@10 | Context P | Context R | No-result | Context chars | Unans. abstained | Unans. cands | Unans. ctx | Index | p95 | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | +| dense | 5 | 3 | 0.8774 | 0.7672 | 0.7231 | 0.3082 | 0.8160 | 0.0000 | 2312 | 0.0000 | 5.0 | 3.0 | 53 | 16.18 ms | +| dense | 5 | 5 | 0.8774 | 0.7672 | 0.7231 | 0.2000 | 0.8774 | 0.0000 | 4007 | 0.0000 | 5.0 | 5.0 | 53 | 86.91 ms | +| dense | 10 | 3 | 0.8774 | 0.7804 | 0.7280 | 0.3082 | 0.8160 | 0.0000 | 2312 | 0.0000 | 10.0 | 3.0 | 53 | 100.17 ms | +| dense | 10 | 5 | 0.8774 | 0.7804 | 0.7280 | 0.2000 | 0.8774 | 0.0000 | 4007 | 0.0000 | 10.0 | 5.0 | 53 | 81.38 ms | +| dense | 10 | 8 | 0.8774 | 0.7804 | 0.7280 | 0.1274 | 0.8774 | 0.0000 | 6659 | 0.0000 | 10.0 | 8.0 | 53 | 18.17 ms | +| dense | 20 | 3 | 0.8774 | 0.7804 | 0.7280 | 0.3082 | 0.8160 | 0.0000 | 2312 | 0.0000 | 20.0 | 3.0 | 53 | 17.01 ms | +| dense | 20 | 5 | 0.8774 | 0.7804 | 0.7280 | 0.2000 | 0.8774 | 0.0000 | 4007 | 0.0000 | 20.0 | 5.0 | 53 | 97.74 ms | +| dense | 20 | 8 | 0.8774 | 0.7804 | 0.7280 | 0.1274 | 0.8774 | 0.0000 | 6659 | 0.0000 | 20.0 | 8.0 | 53 | 85.30 ms | +| dense | 40 | 3 | 0.8774 | 0.7804 | 0.7280 | 0.3082 | 0.8160 | 0.0000 | 2312 | 0.0000 | 40.0 | 3.0 | 53 | 57.28 ms | +| dense | 40 | 5 | 0.8774 | 0.7804 | 0.7280 | 0.2000 | 0.8774 | 0.0000 | 4007 | 0.0000 | 40.0 | 5.0 | 53 | 16.64 ms | +| dense | 40 | 8 | 0.8774 | 0.7804 | 0.7280 | 0.1274 | 0.8774 | 0.0000 | 6659 | 0.0000 | 40.0 | 8.0 | 53 | 99.32 ms | +| hybrid | 5 | 3 | 0.9104 | 0.8101 | 0.7682 | 0.3208 | 0.8538 | 0.0000 | 2307 | 0.0000 | 5.0 | 3.0 | 53 | 90.40 ms | +| hybrid | 5 | 5 | 0.9104 | 0.8101 | 0.7682 | 0.2113 | 0.9104 | 0.0000 | 3937 | 0.0000 | 5.0 | 5.0 | 53 | 89.63 ms | +| hybrid | 10 | 3 | 0.8962 | 0.8068 | 0.7580 | 0.3208 | 0.8538 | 0.0000 | 2321 | 0.0000 | 10.0 | 3.0 | 53 | 83.59 ms | +| hybrid | 10 | 5 | 0.8962 | 0.8068 | 0.7580 | 0.2038 | 0.8962 | 0.0000 | 3986 | 0.0000 | 10.0 | 5.0 | 53 | 89.79 ms | +| hybrid | 10 | 8 | 0.8962 | 0.8068 | 0.7580 | 0.1321 | 0.9009 | 0.0000 | 6548 | 0.0000 | 10.0 | 8.0 | 53 | 14.96 ms | +| hybrid | 20 | 3 | 0.8962 | 0.8098 | 0.7605 | 0.3208 | 0.8538 | 0.0000 | 2323 | 0.0000 | 20.0 | 3.0 | 53 | 82.31 ms | +| hybrid | 20 | 5 | 0.8962 | 0.8098 | 0.7605 | 0.2038 | 0.8962 | 0.0000 | 4011 | 0.0000 | 20.0 | 5.0 | 53 | 18.62 ms | +| hybrid | 20 | 8 | 0.8962 | 0.8098 | 0.7605 | 0.1321 | 0.9057 | 0.0000 | 6660 | 0.0000 | 20.0 | 8.0 | 53 | 16.13 ms | +| hybrid | 40 | 3 | 0.8962 | 0.8098 | 0.7605 | 0.3208 | 0.8538 | 0.0000 | 2323 | 0.0000 | 40.0 | 3.0 | 53 | 97.73 ms | +| hybrid | 40 | 5 | 0.8962 | 0.8098 | 0.7605 | 0.2038 | 0.8962 | 0.0000 | 3990 | 0.0000 | 40.0 | 5.0 | 53 | 100.05 ms | +| hybrid | 40 | 8 | 0.8962 | 0.8098 | 0.7605 | 0.1321 | 0.9057 | 0.0000 | 6661 | 0.0000 | 40.0 | 8.0 | 53 | 16.94 ms | + + +## Skipped cells + +`contextK > candidateK` cannot be filled: the harness fetches `candidateK` +passages, so a wider window would contain fewer passages than it claims. These +2 cell(s) are excluded rather than reported as equal to a narrower one: + +- `dense` candidateK=5, contextK=8 +- `hybrid` candidateK=5, contextK=8 + +The harness refuses the same combination at the flag level, so a typo fails loudly. + + +## How to read it + +- **`candidateK`** moves the ranking metrics and latency: it is how wide the first + stage searches. It cannot change `Context P`/`Context R`, because those look at the + first `contextK` of the fused list and a prefix is unaffected by how deep the list was. +- **`contextK`** moves `Context P` and `Context R` and the context size, not the + ranking metrics. Wider recall rises and precision falls; that is the trade, and both + columns are here so it is visible rather than argued about. +- **Context chars** is a proxy for prompt size, not a token count: the harness pins the + embedding model, not any generation model's tokenizer. +- **No-result** is the share of *answerable* questions whose retrieval returned nothing — + a miss, and the lower the better. +- **Unans. abstained / cands / ctx** describe the *unanswerable* questions, where the + direction flips: there is no ground truth, so abstaining is correct. `abstained` is the + share where nothing passed the threshold; `cands` is how many candidates did (up to + `candidateK`, since the harness fetches that many for `Recall@10`); `ctx` is how many + actually reach the context window, i.e. `min(candidates, contextK)`. A high `cands` with + the usual `ctx` means the threshold is filtering nothing and the window is all noise. + These are the columns a threshold decision should move, and they stay out of every other + column. + +Best nDCG@10 in this grid: `hybrid` candidateK=5, +contextK=3 (0.8101). +Best context precision: `hybrid` candidateK=5, +contextK=3 (0.3208). + +These are **not** recommendations. Selecting the grid maximum on the same questions is +how a benchmark becomes a lookup table; the adoption rule in `baseline-v1.6.md` +decides, and the `validation`/`test` split is what keeps that honest. + +## Reproduce + +```bash +npm run eval:prepare # one-time, networked model bootstrap +npm run eval:sweep # offline; rewrites this file +``` diff --git a/docs/eval/threshold-v1.6.json b/docs/eval/threshold-v1.6.json new file mode 100644 index 0000000..fbf9e89 --- /dev/null +++ b/docs/eval/threshold-v1.6.json @@ -0,0 +1,273 @@ +{ + "baseline": "v1.6", + "productionThreshold": 0.5, + "flat": true, + "recommended": 0.5, + "rows": [ + { + "threshold": 0, + "validation": { + "questions": 34, + "answerableQuestions": 19, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 20, + "unanswerable": { + "questions": 15, + "abstentionCount": 0, + "retrievalAbstentionRate": 0, + "meanCandidatesRetrieved": 20, + "meanContextPassages": 3 + }, + "metrics": { + "recallAt1": 0.513158, + "recallAt5": 0.868421, + "recallAt10": 0.921053, + "mrr": 0.720175, + "ndcgAt10": 0.755611, + "hitRateAt5": 0.894737, + "mapAt10": 0.69386, + "contextPrecision": 0.315789, + "contextRecall": 0.802632 + } + }, + "test": { + "questions": 44, + "answerableQuestions": 34, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 20, + "unanswerable": { + "questions": 10, + "abstentionCount": 0, + "retrievalAbstentionRate": 0, + "meanCandidatesRetrieved": 20, + "meanContextPassages": 3 + }, + "metrics": { + "recallAt1": 0.639706, + "recallAt5": 0.882353, + "recallAt10": 0.919118, + "mrr": 0.774079, + "ndcgAt10": 0.794329, + "hitRateAt5": 0.911765, + "mapAt10": 0.747141, + "contextPrecision": 0.303922, + "contextRecall": 0.823529 + } + } + }, + { + "threshold": 0.3, + "validation": { + "questions": 34, + "answerableQuestions": 19, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 20, + "unanswerable": { + "questions": 15, + "abstentionCount": 0, + "retrievalAbstentionRate": 0, + "meanCandidatesRetrieved": 20, + "meanContextPassages": 3 + }, + "metrics": { + "recallAt1": 0.513158, + "recallAt5": 0.868421, + "recallAt10": 0.921053, + "mrr": 0.720175, + "ndcgAt10": 0.755611, + "hitRateAt5": 0.894737, + "mapAt10": 0.69386, + "contextPrecision": 0.315789, + "contextRecall": 0.802632 + } + }, + "test": { + "questions": 44, + "answerableQuestions": 34, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 20, + "unanswerable": { + "questions": 10, + "abstentionCount": 0, + "retrievalAbstentionRate": 0, + "meanCandidatesRetrieved": 20, + "meanContextPassages": 3 + }, + "metrics": { + "recallAt1": 0.639706, + "recallAt5": 0.882353, + "recallAt10": 0.919118, + "mrr": 0.774079, + "ndcgAt10": 0.794329, + "hitRateAt5": 0.911765, + "mapAt10": 0.747141, + "contextPrecision": 0.303922, + "contextRecall": 0.823529 + } + } + }, + { + "threshold": 0.4, + "validation": { + "questions": 34, + "answerableQuestions": 19, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 20, + "unanswerable": { + "questions": 15, + "abstentionCount": 0, + "retrievalAbstentionRate": 0, + "meanCandidatesRetrieved": 20, + "meanContextPassages": 3 + }, + "metrics": { + "recallAt1": 0.513158, + "recallAt5": 0.868421, + "recallAt10": 0.921053, + "mrr": 0.720175, + "ndcgAt10": 0.755611, + "hitRateAt5": 0.894737, + "mapAt10": 0.69386, + "contextPrecision": 0.315789, + "contextRecall": 0.802632 + } + }, + "test": { + "questions": 44, + "answerableQuestions": 34, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 20, + "unanswerable": { + "questions": 10, + "abstentionCount": 0, + "retrievalAbstentionRate": 0, + "meanCandidatesRetrieved": 20, + "meanContextPassages": 3 + }, + "metrics": { + "recallAt1": 0.639706, + "recallAt5": 0.882353, + "recallAt10": 0.919118, + "mrr": 0.774079, + "ndcgAt10": 0.794329, + "hitRateAt5": 0.911765, + "mapAt10": 0.747141, + "contextPrecision": 0.303922, + "contextRecall": 0.823529 + } + } + }, + { + "threshold": 0.5, + "validation": { + "questions": 34, + "answerableQuestions": 19, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 20, + "unanswerable": { + "questions": 15, + "abstentionCount": 0, + "retrievalAbstentionRate": 0, + "meanCandidatesRetrieved": 20, + "meanContextPassages": 3 + }, + "metrics": { + "recallAt1": 0.513158, + "recallAt5": 0.868421, + "recallAt10": 0.921053, + "mrr": 0.720175, + "ndcgAt10": 0.755611, + "hitRateAt5": 0.894737, + "mapAt10": 0.69386, + "contextPrecision": 0.315789, + "contextRecall": 0.802632 + } + }, + "test": { + "questions": 44, + "answerableQuestions": 34, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 20, + "unanswerable": { + "questions": 10, + "abstentionCount": 0, + "retrievalAbstentionRate": 0, + "meanCandidatesRetrieved": 20, + "meanContextPassages": 3 + }, + "metrics": { + "recallAt1": 0.639706, + "recallAt5": 0.882353, + "recallAt10": 0.919118, + "mrr": 0.774079, + "ndcgAt10": 0.794329, + "hitRateAt5": 0.911765, + "mapAt10": 0.747141, + "contextPrecision": 0.303922, + "contextRecall": 0.823529 + } + } + }, + { + "threshold": 0.6, + "validation": { + "questions": 34, + "answerableQuestions": 19, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 20, + "unanswerable": { + "questions": 15, + "abstentionCount": 0, + "retrievalAbstentionRate": 0, + "meanCandidatesRetrieved": 20, + "meanContextPassages": 3 + }, + "metrics": { + "recallAt1": 0.513158, + "recallAt5": 0.868421, + "recallAt10": 0.921053, + "mrr": 0.720175, + "ndcgAt10": 0.755611, + "hitRateAt5": 0.894737, + "mapAt10": 0.69386, + "contextPrecision": 0.315789, + "contextRecall": 0.802632 + } + }, + "test": { + "questions": 44, + "answerableQuestions": 34, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 20, + "unanswerable": { + "questions": 10, + "abstentionCount": 0, + "retrievalAbstentionRate": 0, + "meanCandidatesRetrieved": 20, + "meanContextPassages": 3 + }, + "metrics": { + "recallAt1": 0.639706, + "recallAt5": 0.882353, + "recallAt10": 0.919118, + "mrr": 0.774079, + "ndcgAt10": 0.794329, + "hitRateAt5": 0.911765, + "mapAt10": 0.747141, + "contextPrecision": 0.303922, + "contextRecall": 0.823529 + } + } + } + ] +} diff --git a/docs/eval/threshold-v1.6.md b/docs/eval/threshold-v1.6.md new file mode 100644 index 0000000..6dcc74e --- /dev/null +++ b/docs/eval/threshold-v1.6.md @@ -0,0 +1,54 @@ +# Threshold derivation — v1.6 (#192) + +Generated by `node scripts/eval-threshold.mjs`. Numbers are harness output; do not edit them by hand. + +## What was measured + +The real harness, the same corpus and the production retrieval config +(`candidateK=20, contextK=3`), once per candidate threshold. The **validation** +split selects; the **test** split reports. The split is the committed manifest +`eval/splits.json`, so the same questions are on the same side on every machine. + +Quality columns cover the answerable questions only; **Unans.** columns cover the +unanswerable ones, where returning nothing is the desired outcome and so a *higher* +abstention rate is better. Two sizes are kept apart: **cands** is how many candidates +passed the threshold (up to `candidateK`), **ctx** is how many reach the context window. + +| Threshold | n (val) | Recall@5 (val) | nDCG@10 (val) | No-result (val) | Unans. abstained (val) | Unans. cands (val) | Unans. ctx (val) | nDCG@10 (test) | Unans. abstained (test) | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | +| 0 | 19 | 0.8684 | 0.7556 | 0.0000 | 0.0000 | 20.0 | 3.0 | 0.7943 | 0.0000 | +| 0.3 | 19 | 0.8684 | 0.7556 | 0.0000 | 0.0000 | 20.0 | 3.0 | 0.7943 | 0.0000 | +| 0.4 | 19 | 0.8684 | 0.7556 | 0.0000 | 0.0000 | 20.0 | 3.0 | 0.7943 | 0.0000 | +| 0.5 | 19 | 0.8684 | 0.7556 | 0.0000 | 0.0000 | 20.0 | 3.0 | 0.7943 | 0.0000 | +| 0.6 | 19 | 0.8684 | 0.7556 | 0.0000 | 0.0000 | 20.0 | 3.0 | 0.7943 | 0.0000 | + +## Selection rule + +Hold the answerable quality line — validation nDCG@10 and context recall must not +regress versus `threshold = 0` — then take the threshold that abstains on the most +unanswerable questions. Tie-break on the lowest threshold. + +Raising a threshold is only worth anything if it stops unsupported context before the +prompt; the quality gate is there so an abstention gain can never be bought with a +retrieval loss. + +## Outcome + +The sweep is **flat**: every threshold from 0 to 0.6 produces the same validation nDCG@10 (0.7556), the same Recall@5 (0.8684) and the same retrieval abstention rate on unanswerable questions (0/15). No candidate is ever filtered out, so the threshold is **non-binding** on this corpus. + +**No evidence to change `threshold = 0.5`.** All thresholds hold the line equally; picking one would be arbitrary. The current value can be neither validated nor falsified here, which is a property of the corpus rather than of the threshold. Note also what this does *not* establish: abstention is a retrieval-layer statement — whether the model then declines to answer needs a generator eval. + +## Caveat on this corpus + +The split removes the most obvious form of overfitting, but 19 +answerable questions on the validation side is a thin basis for a decision, and the corpus +is still small. A threshold is a product decision with a **refusal-rate** cost attached, so +a recommendation here is only as good as the corpus behind it. Re-run this after the +corpus grows (#192 child 2). + +## Reproduce + +```bash +npm run eval:prepare # one-time, networked model bootstrap +npm run eval:threshold # offline; rewrites this file +``` diff --git a/eval/README.md b/eval/README.md index 191ac61..22afea3 100644 --- a/eval/README.md +++ b/eval/README.md @@ -7,10 +7,28 @@ every experiment (#77, #78) is reported as a delta against that file. ## Commands ```bash -npm run eval:prepare # one-time, networked: download the pinned embedding model -npm run eval # offline and deterministic: run the harness, rewrite the baseline +npm run eval:prepare # one-time, networked: download the pinned embedding model +npm run eval # offline and deterministic: run the harness, rewrite the baseline +npm run eval:retrieval # strategy comparison (#77); validation selects, test reports +npm run eval:threshold # derive the similarity threshold on validation, report on test +npm run eval:sweep # bounded grid over strategy × candidateK × contextK, one dashboard +npm run eval:scores # dense score distribution: can one threshold separate relevant from not? +npm run eval:paired # per-question dense ↔ hybrid deltas, by query type +npm run eval:blocks eval/corpus/foo.md # print the block ordinals ground truth must use ``` +### `threshold` is not a cosine + +The vector store returns \`score = 1 - distance / 2\` and sqlite-vec's cosine distance is +\`1 - cosine\`, so: + +```text +score = (1 + cosine) / 2 score 0.5 == cosine 0.0 +``` + +The shipped \`threshold = 0.5\` therefore means **cosine ≥ 0**, which is very permissive. +\`npm run eval:scores\` reports both columns side by side so the two never get conflated. + ### The harness runs the production configuration From v1.6 the harness defaults to the parameters the app ships, so its numbers describe @@ -22,6 +40,7 @@ the product rather than a research setup. Two Ks, because they answer different | `--eval-context-k=` | `3` | passages the chat prompt actually takes (`chatHandlers.ts`) | | `--eval-threshold=` | `0.5` | the similarity floor the app ships | | `--eval-retrieval=` | `dense` | `dense`, `sparse`, or `hybrid` | +| `--eval-split=` | `all` | `all`, `validation`, or `test` — a deterministic id-based split | | `--eval-baseline=` | `v1.6` | name written into `docs/eval/baseline-.{json,md}` | Ranking metrics are computed at `candidateK` depth, not at `contextK`: `Recall@10` needs @@ -29,6 +48,36 @@ at least ten results, and truncation only takes a prefix of the candidate list, truncation cannot change the ranking it is measured on. `contextK` is recorded so the report describes the whole online path. +### Swept parameters are chosen on `validation`, reported on `test` + +A parameter picked on the same questions it is scored on is a fitted number, not a +result. `eval/splits.json` is the committed assignment; `--eval-split=validation` selects +from it and `test` is the rest. + +It is an explicit manifest rather than a hash of the question id. A hash +(`hash(id) % 3`) is actually *stable* — it is computed per id, so adding a question does +not move the existing ones. What it cannot do is express the experimental design: + +- it does not stratify a small corpus, so a rare type (`multi-hop`, `cross-lingual`) can + end up entirely on one side without anyone choosing that — which is what happened; and +- a newly added question is assigned silently instead of deliberately, and `test` is the + side a choice must not be fitted to. + +With a manifest, a question with no entry is **refused** rather than defaulted, so every +new question is assigned on purpose. + +`npm run eval:sweep` runs a bounded grid (`strategy × candidateK × contextK`) and +writes one dashboard with quality, context precision/recall, prompt size, index size +and latency side by side. Its grid maximum is labelled as **not** a recommendation: +selecting on the same questions is how a benchmark becomes a lookup table. + +`contextK > candidateK` is not a cell in that grid. The harness fetches `candidateK` +passages, so a wider window can never be filled; the sweep skips those combinations +and names them in the report, and the harness refuses the same combination from the +command line. Before this was enforced, `contextK=8` at `candidateK=5` was reported as +identical to `contextK=5` — not because 8 assessed the same as 5, but because +passages 6–8 did not exist. + `eval:prepare` downloads the pinned `multilingual-e5-small` revision into the app's model cache and verifies it. `eval` never touches the network: if the model is missing it stops with @@ -47,6 +96,7 @@ does not mean "no download". The 134 MB of weights are not committed. eval/ corpus/ first-party documents (markdown today) questions.jsonl one question per line; committed with the corpus + splits.json the committed validation/test assignment ``` The corpus is authored for this repository and carries the repository's GPL-3.0 licence, @@ -59,6 +109,8 @@ dataset needs a new question. { "id": "q001", "question": "Why is bedload harder to measure than suspended sediment?", + "type": "semantic", + "answerable": true, "relevant": [ { "document": "river-monitoring.md", @@ -71,13 +123,25 @@ dataset needs a new question. } ``` +An **unanswerable** question is the same shape with `"answerable": false` and `"relevant": []`. +The harness refuses the reverse combination in either direction — an answerable question +with no ground truth, or an unanswerable one carrying some — because both are silent: the +first reads as a permanent miss, the second as a normal hit. + Ground truth uses **corpus identity, never database identity**: - `document` is the corpus-relative path. -- `block` is the block ordinal inside the document (`document_blocks.order`). +- `block` is the **`document_blocks.order` the ingestion pipeline produced**, not a line + number and not a paragraph index a human counted. Use `npm run eval:blocks ` to + print the real ordinals through the same loader the harness uses — guessing them is how a + dataset drifts. - `page` is `null` for unpaginated sources. - `quote` is an optional excerpt. The runner fails if the referenced block no longer contains it, so a parser change cannot silently move the ground truth. +- `type` is an optional query class; the report groups every metric by it. Values in use: + `exact` (number/name/detail), `semantic` (why/how), `multi-hop` (two or more blocks), + `cross-lingual` (question language differs from the source), `zh` (Chinese over a + Chinese source). Untagged questions report as `untagged`. Runtime `documentId`s are random and `blockId`s embed them, so neither may appear here. This is what lets #78 change chunking without invalidating the dataset: the ground truth @@ -115,11 +179,32 @@ from unanswerable queries. - **Recall@1/5/10** — share of ground-truth blocks covered by the first k passages. - **MRR** — reciprocal rank of the first relevant passage. - **nDCG@10** — binary-gain discounted cumulative gain. -- **Evidence precision@5** — of the first 5 retrieved passages, the share that cover a - ground-truth block. This is **retrieval precision, not answer citation recall**: the - harness runs no model and produces no answer. Answer-level citation correctness is - covered by the resolver (#70); a model-driven answer eval would be a separate - deliverable. +- **Hit rate@5** — share of questions with at least one relevant passage in the first 5. + Deliberately blunt: it says the answer was *reachable*, where Recall@5 says the material + was *complete*. A two-passage question that finds one scores 1.0 and 0.5 respectively. +- **MAP@10** — mean average precision. The one metric here that combines ranking position + with coverage, so pulling a second relevant passage from rank 9 to rank 2 moves it. +- **Context precision@`contextK`** — of the first `contextK` retrieved passages, the share + that cover a ground-truth block. **Context recall@`contextK`** — the share of the needed + ground-truth blocks that made it into that same window. Both are deterministic: the + dataset says which blocks answer the question, so no model is needed to score the window. + Together they are the trade-off a `contextK` decision actually makes — a wider window + finds more and carries more noise. +- This is **retrieval precision/recall, not answer citation recall**: the harness runs no + model and produces no answer. Answer-level citation correctness is covered by the + resolver (#70). Faithfulness, completeness and answer correctness need a generative model + and are **not evaluated here** — the harness runs offline with only the pinned embedding + model, the same constraint that keeps the reranker unmeasured (#170). +- **By query type** — the same metrics per `type` in `questions.jsonl` (`exact`, + `semantic`, `multi-hop`, `cross-lingual`, `zh`). A single average hides a change that + helps one kind of question and hurts another; the current baseline already shows this, + with `cross-lingual` at nDCG 0.63 against 0.93–1.00 elsewhere. +- **Unanswerable questions** — a separate group, never averaged in. They have no ground + truth, so `Recall` on them is 0/0 rather than 0, and the correct outcome is that + retrieval finds nothing. The reported **no-results** rate is the opposite of a miss: +higher is better, and `mean passages retrieved` is how much irrelevant context was pulled + in anyway. This is the only metric a similarity-threshold decision should move, which is + why the threshold sweep reports it separately. - **Latency p50/p95** — informational only. Timing is **not** frozen, and the committed JSON excludes it so two runs diff cleanly. diff --git a/eval/corpus/api-gateway-migration.md b/eval/corpus/api-gateway-migration.md new file mode 100644 index 0000000..9bb7dbe --- /dev/null +++ b/eval/corpus/api-gateway-migration.md @@ -0,0 +1,79 @@ +# Gateway API Migration Guide + +## Overview + +This guide covers moving a client between Gateway API versions. It deliberately restates +the numbers that differ between versions, because the most common migration defect is a +client that keeps a v1 constant while pointing at a v2 or v3 route. + +The three live surfaces are `/v1/complete`, `/v2/generate` and `/v3/chat`. Any of the three +will reject a body shaped for a different one with `400`, so a silently wrong version is +usually a routing mistake rather than a schema mistake. + +## Choosing a Target + +New integrations should target v3. Existing v2 integrations should move to v3 only when +they need structured output or the larger window, because v3 changes the rate-limit +accounting in a way that can halve effective throughput for structured-output workloads. + +v1 integrations should migrate to v2 at minimum. v1 has no streaming mode, and every +long-answer workload written against v1 pays for it in perceived latency. + +## Endpoint Changes + +| | v1 | v2 | v3 | +| --- | --- | --- | --- | +| Path | `/v1/complete` | `/v2/generate` | `/v3/chat` | +| Body | `prompt` | `messages` | `messages` | +| Streaming | none | server-sent events | server-sent events with `phase` | + +The version lives in the path on all three. A client that versioned the host instead will +not be routed by the gateway at all, and the failure looks like a DNS failure rather than a +version mismatch. + +## Timeout and Retry Changes + +| | v1 | v2 | v3 | +| --- | --- | --- | --- | +| Default timeout | 30000 ms | 60000 ms | 45000 ms | +| Recommended retries | 2 | 5 | 3 | +| Backoff base | 500 ms | 2000 ms | 1000 ms | + +A client migrating from v1 to v2 that keeps the v1 backoff of 500 ms will retry far more +aggressively than the version it is talking to expects, which is a common cause of +self-inflicted `429`s during a cutover. + +Moving from v2 to v3 in the other direction is the riskier one: v3 has fewer recommended +retries and a lower default timeout, so a client tuned for v2's patience will give up +earlier than it did before. + +## Rate Limit Changes + +| | v1 | v2 | v3 | +| --- | --- | --- | --- | +| Default limit | 600 rpm | 3000 rpm | 1200 rpm | +| Burst | 60 | 300 | 120 | + +The v3 structured-output path counts as two requests. A v2 workload that produced 1000 +structured responses per minute was comfortably inside v2's 3000 rpm budget and is almost +exactly at v3's effective 600-per-minute structured ceiling, so the migration is a +throughput change even though the headline number only fell from 3000 to 1200. + +## Context Window Changes + +| | v1 | v2 | v3 | +| --- | --- | --- | --- | +| Window | 8192 tokens | 32768 tokens | 65536 tokens | +| Auto-truncation | no | no | no | + +None of the three truncates automatically; all three reject an oversized request with `400`. +Clients that relied on an upstream provider's truncation find the migration fails loudly +rather than quietly, which is intentional. + +## Checklist + +Before cutting over, confirm the path, the body shape, the timeout, the retry count, the +backoff base and the rate-limit budget. Of those six, the two that are missed most often in +practice are the backoff base and the rate-limit budget, because neither produces an error — +they produce a client that is slower or noisier than it was, which is easy to attribute to +the model rather than to the migration. diff --git a/eval/corpus/api-gateway-v1.md b/eval/corpus/api-gateway-v1.md new file mode 100644 index 0000000..8fc2fe0 --- /dev/null +++ b/eval/corpus/api-gateway-v1.md @@ -0,0 +1,58 @@ +# Gateway API v1 Reference + +## Overview + +The v1 Gateway API is the first generally available surface for text completion. It +accepts a prompt and returns a completion, with no notion of roles, tools or streaming +frames beyond newline-delimited chunks. Clients are expected to be long-lived processes +that hold a single connection open and issue many requests over it. + +v1 is closed to new features. It receives security fixes only, and the deprecation notice +on the v1 endpoint names v2 as the supported successor. + +## Endpoint + +Requests go to the `/v1/complete` path. The path is versioned rather than the host, so a +client that hardcodes the host will silently keep talking to v1 after an upgrade. The +request body carries `prompt`, `max_tokens` and an optional `stop` array; there is no +`system` field, and a system instruction has to be concatenated into the prompt. + +Responses are returned as a single JSON object. v1 has no streaming mode, which is the +change most often cited in migration discussions. + +## Timeouts and Retry + +The default request timeout is **30000 milliseconds**. A request that has not produced any +output within that window is cancelled by the gateway, not by the client, and the client +sees a `504` with the body `{"error":"upstream_timeout"}`. + +The recommended retry count is **2**. The gateway does not retry on the client's behalf, so +this is a client-side contract rather than an enforced limit. The recommended backoff base +is **500 milliseconds**, doubled on each subsequent attempt, which produces waits of 500 ms +then 1000 ms for the two allowed retries. Jitter is not required by v1 but is recommended. + +Retrying a timed-out request is safe because v1 has no server-side session state. Retrying +a request that failed with `429` is not useful unless the `Retry-After` header is honoured +first. + +## Rate Limits + +The default rate limit is **600 requests per minute**, counted per API key rather than per +connection. Bursts of up to 60 requests may be issued within any one-second window before +the limiter engages, so a short burst is allowed even when the per-minute budget is nearly +spent. + +Exceeding the limit returns `429` with a `Retry-After` header in seconds. The limiter +counts a request when the body has been fully received, not when the response is produced, +which means a slow upstream does not consume budget twice. + +## Context Window + +The maximum context window is **8192 tokens**, counting the prompt and the completion +together. A request whose prompt alone exceeds the window is rejected with `400` rather +than truncated, because silent truncation was found to produce worse answers than a +visible failure. + +Token counting uses the same tokenizer as the model, so an approximation by character +count will disagree near the boundary. The gateway exposes a `/v1/tokenize` helper for +clients that want an exact count before sending. diff --git a/eval/corpus/api-gateway-v2.md b/eval/corpus/api-gateway-v2.md new file mode 100644 index 0000000..d39613a --- /dev/null +++ b/eval/corpus/api-gateway-v2.md @@ -0,0 +1,65 @@ +# Gateway API v2 Reference + +## Overview + +The v2 Gateway API replaces the single-prompt completion surface with a message list. It +introduces roles, a real streaming mode and server-side sessions, and it is the surface the +deprecation notice on v1 points at. v2 is feature-frozen: it receives correctness and +security fixes, and v3 is the current recommended target for new integrations. + +The message list is the change that forces most migrations. A v1 prompt with a concatenated +system instruction has to be split into a `system` message and a `user` message, and clients +that relied on concatenation usually find their prompts measurably worse until they split +them. + +## Endpoint + +Requests go to the `/v2/generate` path. Unlike v1, the version is part of the route and +the gateway rejects a v1-shaped body with `400` and a pointer at the migration guide. The +request body carries `messages`, `max_tokens`, `stream` and an optional `tools` array. + +Streaming is enabled per request with `stream: true`, and frames are server-sent events +rather than newline-delimited JSON. Frames carry a monotonically increasing `index`; a +client that reconnects mid-stream must resume from the last index it acknowledged. + +## Timeouts and Retry + +The default request timeout is **60000 milliseconds**, doubled from v1 because v2 sessions +are allowed to think for longer before the first token. The timeout covers the whole turn, +not the gap between frames. + +The recommended retry count is **5**, with a recommended backoff base of **2000 +milliseconds** and full jitter. The higher retry count exists because v2 introduced +server-side sessions, and a retried request may attach to the same session rather than +starting a new one — retrying is therefore usually cheaper than it was in v1. + +A timeout is reported as `504` with `{"error":"upstream_timeout"}` exactly as in v1, so a +client that only inspects that field cannot tell which version produced it. + +## Rate Limits + +The default rate limit is **3000 requests per minute**, counted per API key. Sessions are +counted separately: opening a session costs one request, and each subsequent turn on that +session costs one request, so a long conversation consumes budget linearly. + +Bursts of up to 300 requests may be issued within one second. When a session is already +open, the limiter applies the request to the session's own budget first, which means a +bursty client with many open sessions can exhaust the per-minute budget much faster than +the raw request count suggests. + +## Context Window + +The maximum context window is **32768 tokens**. v2 grew the window partly to make room for +tool definitions, which are counted in the same budget as messages. A request that exceeds +the window is rejected with `400`; v2 does not offer automatic truncation either. + +Because tools are counted, a request with a large tool schema can exceed the window even +when the conversation itself is short. The gateway reports the split between message tokens +and tool tokens in the `usage` block of every response so a client can see which side grew. + +## Sessions + +A session is created implicitly by the first request that omits `session_id`. Sessions +expire after 30 minutes of inactivity. A request that names an expired session is not an +error: the gateway starts a new one and reports the new id, a behaviour that has surprised +several integrators into thinking their retry had lost context. diff --git a/eval/corpus/api-gateway-v3.md b/eval/corpus/api-gateway-v3.md new file mode 100644 index 0000000..e298021 --- /dev/null +++ b/eval/corpus/api-gateway-v3.md @@ -0,0 +1,61 @@ +# Gateway API v3 Reference + +## Overview + +The v3 Gateway API is the current recommended surface. It keeps the message list and +streaming model introduced in v2 and adds structured output, explicit reasoning budgets and +a per-request deadline. v1 and v2 remain available but receive fixes only. + +v3 is not wire-compatible with v2. A v2 body sent to the v3 route is rejected with `400` +and a link to the migration guide, exactly as a v1 body is rejected by v2. + +## Endpoint + +Requests go to the `/v3/chat` path. The body carries `messages`, `max_tokens`, `stream`, +an optional `response_format` describing structured output, and an optional `deadline_ms` +that overrides the default timeout for one request. + +Streaming frames are server-sent events and now carry both an `index` and a `phase`, so a +client can distinguish reasoning frames from answer frames without inspecting the text. +Structured output is delivered as a single final frame; partial structured output is not +emitted, because a half-parsed object was found to be worse than no object. + +## Timeouts and Retry + +The default request timeout is **45000 milliseconds**, between v1's 30 s and v2's 60 s, +chosen after measuring that the median v3 turn finishes in about 11 s while the long tail +benefits from more room than v1 gave. + +The recommended retry count is **3**, with a backoff base of **1000 milliseconds** and full +jitter. v3 adds `deadline_ms`, and a request that carries it uses that value instead of the +default: a client that sets `deadline_ms` to 10000 is not retried by the gateway past the +client's own deadline, which makes the two settings interact in a way v1 and v2 had no +equivalent for. + +## Rate Limits + +The default rate limit is **1200 requests per minute**. Structured-output requests are +counted as two requests, because the gateway runs a validation pass over the produced object +before returning it; a client that migrates a high-volume v2 workload to structured output +can therefore exhaust its budget at half the expected request count. + +Bursts of up to 120 requests may be issued within one second. The limiter applies reasoning +tokens against a separate budget from requests, so a client with long reasoning turns can +hit the token budget before the request budget. + +## Context Window + +The maximum context window is **65536 tokens**, the largest of the three versions, and the +only one where tool schemas, reasoning tokens and messages are reported as three separate +line items in `usage` rather than folded together. + +A request that exceeds the window is rejected with `400`. v3 does not truncate +automatically, but it does report how many tokens the request would need, which makes a +programmatic retry at a smaller size possible without re-tokenising the input. + +## Structured Output + +`response_format` accepts a JSON schema and a strictness flag. Strict mode is slower but +guarantees the object validates against the schema. Non-strict mode is the default and can +return an object that parses but does not match, which the gateway flags in a +`validation_errors` array rather than failing the request. diff --git a/eval/corpus/coastal-monitoring.md b/eval/corpus/coastal-monitoring.md new file mode 100644 index 0000000..837713d --- /dev/null +++ b/eval/corpus/coastal-monitoring.md @@ -0,0 +1,70 @@ +# Coastal Monitoring Programme + +## Overview + +Coastal monitoring tracks a tidally driven water body at the land margin. Its distinguishing +problem is that the water body moves twice a day: a station is never at the same depth for +two consecutive visits, and a tidal phase that happens to coincide with a sampling run can +dominate the reading. + +The programme covers four shore stations and two offshore buoys. Shore stations are visited; +buoys are instrumented and telemeter. The two kinds of station are reported separately, +because mixing a telemetered series with a visited series is the most common defect in a +coastal dataset. + +## Sampling Interval + +Routine sampling runs **hourly** at the telemetered buoys and **fortnightly** at the shore +stations. The hourly cadence exists to resolve the tidal cycle, which a fortnightly cadence +would alias into a meaningless slow oscillation; a shore station cannot support it because +each visit is a boat trip. + +Every telemetered reading is stamped with its tidal phase. A reading without a phase stamp +is retained but cannot be compared against the shore series, and the quality-control pass +excludes it from any cross-station comparison. + +## Replicate Samples + +Field teams collect **three replicate samples** at each shore station. Three is the standard +for the programme because the boat trip, not the analysis, dominates the cost, so an extra +replicate is nearly free once the team is on site — but only three are taken, because the +shore stations are well mixed and the fourth replicate has never changed a decision. + +Buoys do not take replicates in the sampling sense; they take a burst of 30 readings over +60 seconds and report the median. The median is used rather than the mean because a single +wave splash is a large positive outlier in exactly the quantities a buoy measures. + +## Sensor Depth + +The shore stations carry a sensor at **1 metre below the surface**, and the offshore buoys +carry a sensor at **2 metres below the surface**. The buoy depth is greater because a buoy +in the wave zone is repeatedly lifted and dropped by swell, and a sensor closer to the +surface is out of the water a meaningful fraction of the time. + +Both depths are recorded as depth below the *instantaneous* surface. Coastal sensors are the +only programme where the surface reference changes fast enough to matter within a single +reading, so the timestamp and the depth are recorded together and neither is meaningful +alone. + +## Parameters + +The core parameters are water temperature, salinity, turbidity and wave height. Coastal +adds wave height, which no other programme records, and salinity, which only the estuary +programme also records but for a different reason. + +Wave height is recorded as significant wave height, the mean of the highest third of waves +in the burst, not as the maximum. The maximum is recorded separately and is not used for +trend analysis because it is dominated by rare events. + +## Quality Control + +Telemetered readings are passed through a spike filter that rejects a value more than four +standard deviations from its 24-hour rolling mean. The filter has an override: a reading +that is also accompanied by a wave height above the 99th percentile is retained rather than +rejected, on the grounds that a storm is exactly when an unusual value is most likely to be +real. + +Shore stations use a field blank and a blind duplicate, as the river and reservoir +programmes do. A station whose duplicate differs by more than 25 percent is re-visited, +which is a looser threshold than the reservoir programme's 20 percent because a coastal +station is inherently noisier and a tighter threshold flagged almost every visit. diff --git a/eval/corpus/estuary-monitoring.md b/eval/corpus/estuary-monitoring.md new file mode 100644 index 0000000..696dd00 --- /dev/null +++ b/eval/corpus/estuary-monitoring.md @@ -0,0 +1,67 @@ +# Estuary Monitoring Programme + +## Overview + +Estuary monitoring tracks the mixing zone where a river meets the sea. Its distinguishing +problem is that the water body has a gradient in all three dimensions at once: salinity and +turbidity change sharply over a few kilometres, and the position of that gradient moves with +the tide and with river flow. + +The programme covers two estuaries, each with a transect of five stations running from the +freshwater end to the mouth. A transect is the unit of reporting, not a station, because a +single station in an estuary describes a position in a gradient that has moved by the time +the next station is sampled. + +## Sampling Interval + +Routine sampling runs **daily** at the two lowest stations and **fortnightly** along the +rest of the transect. Daily sampling at the lower stations is used to track the salt wedge, +whose position responds to the tide within hours; the upper transect responds to river flow +over days and does not need it. + +Transect sampling is run on the ebb tide, and every station is occupied within a single ebb +to keep the transect a snapshot rather than a sequence. A transect that overruns its ebb is +discarded and repeated, because a transect sampled across a tidal reversal is not a gradient. + +## Replicate Samples + +Field teams collect **five replicate samples** at each transect station, the highest of any +programme in the network. Five because the estuary gradient means two samples taken a metre +apart can differ more than two samples taken a kilometre apart at a well-mixed site; the +within-station variance is high enough that three replicates do not estimate a mean reliably. + +Replicates are taken as a spatial cross rather than as a sequence: one at the nominal +position and four at 25 metres on each axis. A sequential set of replicates would all sample +the same parcel of water and would understate the variance that matters. + +## Sensor Depth + +Transect stations carry sensors at **2 metres below the surface** and at 1 metre above the +bed, paired so that the vertical salinity difference can be computed directly. The surface +depth is fixed at 2 metres rather than at 1 metre to keep the sensor below the freshwater +lens that floats on the saline layer at the freshwater end. + +The paired depths are the programme's defining feature. A single sensor in an estuary cannot +distinguish a change in salinity from a change in where the halocline sits, and those are +different findings. + +## Parameters + +The core parameters are salinity, turbidity, dissolved oxygen and temperature. Estuary adds +the position of the turbidity maximum, which is the programme's headline product and is not +recorded by any other programme. + +The turbidity maximum is reported as a distance from the freshwater end, not as a turbidity +value. Its position moves several kilometres over a tidal cycle, and a turbidity value +without a position cannot distinguish a stationary maximum from a passing one. + +## Quality Control + +Every transect includes one blind duplicate at a randomly chosen station. Precision is +estimated per transect rather than per station, because the quantity the programme reports +is a gradient and the error that matters is the error in the gradient. + +A transect whose blind duplicate differs by more than 30 percent is repeated. The threshold +is looser than any other programme's, and deliberately so: an estuary's true variance is +genuinely larger, and a tighter threshold would cause every transect to be repeated, which +in practice means none of them are. diff --git a/eval/corpus/groundwater-monitoring.md b/eval/corpus/groundwater-monitoring.md new file mode 100644 index 0000000..ca89489 --- /dev/null +++ b/eval/corpus/groundwater-monitoring.md @@ -0,0 +1,68 @@ +# Groundwater Monitoring Programme + +## Overview + +Groundwater monitoring tracks water below the surface rather than water in a channel. Its +distinguishing problem is that the water body cannot be seen, so every measurement is a +sample from a borehole that may or may not be connected to the aquifer the programme intends +to describe. + +The programme covers nine boreholes across three catchments. Eight are monitoring boreholes +used only for measurement; one is a supply borehole that is also pumped, and its readings +are reported separately because a pumped borehole's level reflects recent abstraction as +much as the aquifer. + +## Sampling Interval + +Routine sampling runs **monthly**, the least frequent cadence in the network. Groundwater +responds to rainfall over weeks to months, and a more frequent cadence measures the borehole +rather than the aquifer: a fortnightly series is dominated by the borehole's own equilibration +after each visit, which takes several days. + +Supply boreholes are additionally sampled immediately before and after each pumping cycle. +Those samples are paired and reported as a drawdown and recovery pair, not as routine +readings. + +## Replicate Samples + +Field teams collect **two replicate samples** at each borehole, the lowest of any programme +in the network. Two rather than three because groundwater is well mixed by the time it +reaches a borehole, and the dominant error is not within-sample variance but the borehole's +connection to the aquifer — an error that an extra replicate does not reduce. + +Because two replicates give no way to identify an outlier, any borehole whose two replicates +disagree by more than 10 percent is re-sampled entirely rather than resolved statistically. +Ten percent is tighter than any other programme's threshold precisely because there is no +third replicate to arbitrate. + +## Sensor Depth + +Boreholes carry a pressure transducer at **15 metres below the water table**, measured on the +first visit and recorded as an absolute elevation so that a falling water table does not +silently change the depth being sampled. The depth is the largest in the network by an order +of magnitude, which is a property of boreholes rather than a choice. + +The transducer records water level continuously. Water level is the programme's primary +quantity; chemistry is sampled monthly and level is logged every 15 minutes, and the two are +reported on separate axes because a single chart of both conceals the pattern in either. + +## Parameters + +The core parameters are water level, temperature, specific conductance and nitrate. Groundwater +adds water level, which is the only parameter in the network that is logged continuously +rather than sampled. + +Nitrate is recorded as the primary indicator of agricultural loading, and it is the parameter +the programme was established to track. Specific conductance is recorded as a cheap proxy for +salinity intrusion in the two coastal boreholes. + +## Quality Control + +Every visit records the water level before and after purging. A borehole in which the level +does not recover within 24 hours of purging is flagged, because the purge has drawn from a +body of water that is not being recharged and the sample may not represent the aquifer. + +Blind duplicates are submitted quarterly rather than per visit, which is the sparsest +verification in the network. The programme's position is that its dominant uncertainty is the +borehole-to-aquifer connection, which a duplicate cannot measure, so spending budget on more +duplicates would buy precision the programme cannot use. diff --git a/eval/corpus/reservoir-monitoring.md b/eval/corpus/reservoir-monitoring.md new file mode 100644 index 0000000..389eb79 --- /dev/null +++ b/eval/corpus/reservoir-monitoring.md @@ -0,0 +1,69 @@ +# Reservoir Monitoring Programme + +## Overview + +Reservoir monitoring tracks a standing water body whose level is managed rather than +natural. The programme's distinguishing problem is that the water body has an operator: +every measurement has to be paired with the release schedule, or a change in a reading is +indistinguishable from a change in how the reservoir was run that week. + +The programme covers three reservoirs in the upland basin. Each has a fixed monitoring +station at the dam face and two floating stations whose position is recorded on every +visit, because a reservoir's surface area changes enough over a season to move a floating +station hundreds of metres without anyone touching it. + +## Sampling Interval + +Routine sampling runs **fortnightly**, on a fixed Tuesday, so that the interval is +consistent across sites and operators. A fortnightly cadence is a deliberate compromise: +weekly was found to double cost without changing any trend, and monthly aliased against +the operator's own drawdown cycle, which is also roughly monthly and made the series very +hard to interpret. + +Event sampling is triggered by any release exceeding 20 percent of live capacity in a +single day. Event samples are additional to the routine cadence and are labelled with the +release event rather than with the calendar. + +## Replicate Samples + +Field teams collect **four replicate samples** at each station to control for local +variability. Four rather than three because the reservoir stations sit in a drawdown zone +where wind-driven mixing produces an occasional outlier; with three replicates a single +outlier is a third of the mean, and with four it can be identified and excluded on a stated +rule rather than discarded by feel. + +Replicates are taken within a 15-minute window. A replicate that falls outside that window +is recorded but excluded from the mean, because the reservoir can stratify and destratify on +that timescale in summer. + +## Sensor Depth + +The fixed station carries a sensor string at **5 metres below the surface**, and the two +floating stations carry a single sensor at **1 metre below the surface**. The asymmetry is +intentional: the dam face is deep and well mixed, while the floating stations are in the +drawdown zone where the interesting gradient is in the top metre. + +Depth is recorded as depth below the *current* surface, not below full capacity. Because the +surface moves, a sensor on a fixed string is at a different absolute elevation at different +times, and the programme records both so that a reader can reconstruct which was meant. + +## Parameters + +The core parameters are water temperature, dissolved oxygen, turbidity and chlorophyll-a. +Reservoir-specific parameters are residence time and drawdown rate, neither of which the +river or lake programmes record because neither has an operator-controlled outlet. + +Turbidity is recorded as the primary indicator of sediment resuspension during drawdown, +which is the process the programme exists to quantify. Chlorophyll-a is recorded as the +primary indicator of the algal response to nutrient loading. + +## Quality Control + +Every routine visit includes one field blank and one duplicate submitted blind. The +duplicate is used to estimate within-station precision; the blank is used to detect +contamination introduced by the sampling kit rather than by the reservoir. + +A station whose blind duplicate differs by more than 20 percent is flagged and re-visited +within seven days. Two consecutive flags retire the station's sensor string, because the +most common cause of a persistent discrepancy is a drifting sensor rather than a genuinely +patchy water body. diff --git a/eval/questions.jsonl b/eval/questions.jsonl index e125d9c..44fa762 100644 --- a/eval/questions.jsonl +++ b/eval/questions.jsonl @@ -1,30 +1,78 @@ -{"id":"q001","question":"Why is bedload harder to measure than suspended sediment?","relevant":[{"document":"river-monitoring.md","page":null,"block":4,"quote":"Bedload is the harder fraction to measure"}]} -{"id":"q002","question":"How many replicate samples are collected at each river station?","relevant":[{"document":"river-monitoring.md","page":null,"block":6,"quote":"three replicate samples at each station"}]} -{"id":"q003","question":"What is the central trade-off in lithium-ion cell design?","relevant":[{"document":"battery-chemistry.md","page":null,"block":2,"quote":"trade energy density against thermal stability"}]} -{"id":"q004","question":"Why do nickel-rich battery packs need more aggressive thermal management?","relevant":[{"document":"battery-chemistry.md","page":null,"block":4,"quote":"release oxygen at lower temperatures than iron phosphate"}]} -{"id":"q005","question":"What happens once the separator in a battery cell melts?","relevant":[{"document":"battery-chemistry.md","page":null,"block":6,"quote":"Once the separator melts, the cell shorts internally"}]} -{"id":"q006","question":"At what temperature do honeybees begin to forage?","relevant":[{"document":"orchard-pollination.md","page":null,"block":4,"quote":"above roughly twelve degrees Celsius"}]} -{"id":"q007","question":"What does a late frost damage during full bloom?","relevant":[{"document":"orchard-pollination.md","page":null,"block":6,"quote":"destroys the flower's ovary rather than the petals"}]} -{"id":"q008","question":"Why is a continuous tree canopy more effective at cooling than isolated trees?","relevant":[{"document":"urban-canopy.md","page":null,"block":4,"quote":"asphalt stores far more heat during the day"}]} -{"id":"q009","question":"Why are trees with aggressive surface roots unsuitable for narrow verges?","relevant":[{"document":"urban-canopy.md","page":null,"block":6,"quote":"aggressive surface roots lift pavements"}]} -{"id":"q010","question":"At what temperature is lactic acid fermentation fastest?","relevant":[{"document":"fermentation.md","page":null,"block":4,"quote":"fastest between twenty and twenty-four degrees Celsius"}]} -{"id":"q011","question":"Is the salt percentage in fermentation based on vegetable weight or water weight?","relevant":[{"document":"fermentation.md","page":null,"block":6,"quote":"percentage of the vegetable weight, not the water weight"}]} -{"id":"q012","question":"Why must tidal turbines be sited in places with very fast currents?","relevant":[{"document":"tidal-energy.md","page":null,"block":4,"quote":"power scales with the cube of velocity"}]} -{"id":"q013","question":"What is the main environmental concern for tidal energy installations?","relevant":[{"document":"tidal-energy.md","page":null,"block":6,"quote":"change in sediment transport"}]} -{"id":"q014","question":"In lake monitoring, how is the sampling depth actually recorded?","relevant":[{"document":"lake-monitoring.md","page":null,"block":6,"quote":"Depth is recorded from the profiling sonde"}]} -{"id":"q015","question":"Why does deep-water oxygen fall while a lake remains stratified?","relevant":[{"document":"lake-monitoring.md","page":null,"block":4,"quote":"Oxygen in the hypolimnion is not replenished"}]} -{"id":"q016","question":"How do supercapacitors hold their charge?","relevant":[{"document":"supercapacitors.md","page":null,"block":4,"quote":"electric double layer formed at the surface of a porous carbon electrode"}]} -{"id":"q017","question":"Why can solitary bees pollinate a bloom week that is too cold for honeybee hives?","relevant":[{"document":"wild-pollinators.md","page":null,"block":4,"quote":"forage at lower temperatures than honeybees"}]} -{"id":"q018","question":"Why is one continuous planted roof layer better than several isolated planted beds?","relevant":[{"document":"green-roofs.md","page":null,"block":4,"quote":"an exposed membrane absorbs far more heat during the day"}]} -{"id":"q019","question":"Where do acetic acid bacteria sit in a vinegar culture?","relevant":[{"document":"vinegar-production.md","page":null,"block":4,"quote":"surface of the liquid where oxygen is available"}]} -{"id":"q020","question":"What happens if a vinegar culture is sealed airtight?","relevant":[{"document":"vinegar-production.md","page":null,"block":6,"quote":"Sealing a vinegar culture airtight stops the conversion entirely"}]} -{"id":"q021","question":"Why is wave energy harder to schedule ahead than tidal energy?","relevant":[{"document":"wave-energy.md","page":null,"block":2,"quote":"driven by wind rather than by the moon"}]} -{"id":"q022","question":"Where does siting for wave energy devices concentrate, and where does it not?","relevant":[{"document":"wave-energy.md","page":null,"block":4,"quote":"exposed headlands rather than on sheltered channels"}]} -{"id":"q023","question":"How does the river sampling protocol differ from the lake sampling protocol?","relevant":[{"document":"river-monitoring.md","page":null,"block":6,"quote":"three replicate samples at each station"},{"document":"lake-monitoring.md","page":null,"block":6,"quote":"Depth is recorded from the profiling sonde"}]} -{"id":"q024","question":"A street canopy and a green roof are both said to cool; what surface does each one shade?","relevant":[{"document":"urban-canopy.md","page":null,"block":4,"quote":"asphalt stores far more heat during the day"},{"document":"green-roofs.md","page":null,"block":4,"quote":"an exposed membrane absorbs far more heat during the day"}]} -{"id":"q025","question":"Which preservation method depends on keeping air away from the food?","relevant":[{"document":"fermentation.md","page":null,"block":2,"quote":"an anaerobic environment"}]} -{"id":"q026","question":"Why can one cold morning cost a grower the whole crop even when colonies are brought in?","relevant":[{"document":"orchard-pollination.md","page":null,"block":4,"quote":"above roughly twelve degrees Celsius"}]} -{"id":"q027","question":"绿茶应该怎样保存才能减缓氧化?","relevant":[{"document":"chinese-tea-storage.md","page":null,"block":2,"quote":"绿茶最容易氧化变质,需要低温密封保存"}]} -{"id":"q028","question":"茶叶储存的相对湿度上限是多少?","relevant":[{"document":"chinese-tea-storage.md","page":null,"block":4,"quote":"相对湿度应保持在百分之五十以下"}]} -{"id":"q029","question":"为什么冷冻保存的茶叶取出后不能立刻打开包装?","relevant":[{"document":"chinese-tea-storage.md","page":null,"block":6,"quote":"冷凝水会直接落在茶叶上"}]} -{"id":"q030","question":"为什么潮汐能比风能和太阳能更容易提前安排发电?","relevant":[{"document":"tidal-energy.md","page":null,"block":2,"quote":"predictable decades ahead, unlike wind or solar"}]} +{"id":"q001","question":"Why is bedload harder to measure than suspended sediment?","relevant":[{"document":"river-monitoring.md","page":null,"block":4,"quote":"Bedload is the harder fraction to measure"}],"type":"semantic"} +{"id":"q002","question":"How many replicate samples are collected at each river station?","relevant":[{"document":"river-monitoring.md","page":null,"block":6,"quote":"three replicate samples at each station"}],"type":"exact"} +{"id":"q003","question":"What is the central trade-off in lithium-ion cell design?","relevant":[{"document":"battery-chemistry.md","page":null,"block":2,"quote":"trade energy density against thermal stability"}],"type":"semantic"} +{"id":"q004","question":"Why do nickel-rich battery packs need more aggressive thermal management?","relevant":[{"document":"battery-chemistry.md","page":null,"block":4,"quote":"release oxygen at lower temperatures than iron phosphate"}],"type":"semantic"} +{"id":"q005","question":"What happens once the separator in a battery cell melts?","relevant":[{"document":"battery-chemistry.md","page":null,"block":6,"quote":"Once the separator melts, the cell shorts internally"}],"type":"semantic"} +{"id":"q006","question":"At what temperature do honeybees begin to forage?","relevant":[{"document":"orchard-pollination.md","page":null,"block":4,"quote":"above roughly twelve degrees Celsius"}],"type":"exact"} +{"id":"q007","question":"What does a late frost damage during full bloom?","relevant":[{"document":"orchard-pollination.md","page":null,"block":6,"quote":"destroys the flower's ovary rather than the petals"}],"type":"semantic"} +{"id":"q008","question":"Why is a continuous tree canopy more effective at cooling than isolated trees?","relevant":[{"document":"urban-canopy.md","page":null,"block":4,"quote":"asphalt stores far more heat during the day"}],"type":"semantic"} +{"id":"q009","question":"Why are trees with aggressive surface roots unsuitable for narrow verges?","relevant":[{"document":"urban-canopy.md","page":null,"block":6,"quote":"aggressive surface roots lift pavements"}],"type":"semantic"} +{"id":"q010","question":"At what temperature is lactic acid fermentation fastest?","relevant":[{"document":"fermentation.md","page":null,"block":4,"quote":"fastest between twenty and twenty-four degrees Celsius"}],"type":"exact"} +{"id":"q011","question":"Is the salt percentage in fermentation based on vegetable weight or water weight?","relevant":[{"document":"fermentation.md","page":null,"block":6,"quote":"percentage of the vegetable weight, not the water weight"}],"type":"exact"} +{"id":"q012","question":"Why must tidal turbines be sited in places with very fast currents?","relevant":[{"document":"tidal-energy.md","page":null,"block":4,"quote":"power scales with the cube of velocity"}],"type":"semantic"} +{"id":"q013","question":"What is the main environmental concern for tidal energy installations?","relevant":[{"document":"tidal-energy.md","page":null,"block":6,"quote":"change in sediment transport"}],"type":"semantic"} +{"id":"q014","question":"In lake monitoring, how is the sampling depth actually recorded?","relevant":[{"document":"lake-monitoring.md","page":null,"block":6,"quote":"Depth is recorded from the profiling sonde"}],"type":"exact"} +{"id":"q015","question":"Why does deep-water oxygen fall while a lake remains stratified?","relevant":[{"document":"lake-monitoring.md","page":null,"block":4,"quote":"Oxygen in the hypolimnion is not replenished"}],"type":"semantic"} +{"id":"q016","question":"How do supercapacitors hold their charge?","relevant":[{"document":"supercapacitors.md","page":null,"block":4,"quote":"electric double layer formed at the surface of a porous carbon electrode"}],"type":"semantic"} +{"id":"q017","question":"Why can solitary bees pollinate a bloom week that is too cold for honeybee hives?","relevant":[{"document":"wild-pollinators.md","page":null,"block":4,"quote":"forage at lower temperatures than honeybees"}],"type":"semantic"} +{"id":"q018","question":"Why is one continuous planted roof layer better than several isolated planted beds?","relevant":[{"document":"green-roofs.md","page":null,"block":4,"quote":"an exposed membrane absorbs far more heat during the day"}],"type":"semantic"} +{"id":"q019","question":"Where do acetic acid bacteria sit in a vinegar culture?","relevant":[{"document":"vinegar-production.md","page":null,"block":4,"quote":"surface of the liquid where oxygen is available"}],"type":"semantic"} +{"id":"q020","question":"What happens if a vinegar culture is sealed airtight?","relevant":[{"document":"vinegar-production.md","page":null,"block":6,"quote":"Sealing a vinegar culture airtight stops the conversion entirely"}],"type":"semantic"} +{"id":"q021","question":"Why is wave energy harder to schedule ahead than tidal energy?","relevant":[{"document":"wave-energy.md","page":null,"block":2,"quote":"driven by wind rather than by the moon"}],"type":"semantic"} +{"id":"q022","question":"Where does siting for wave energy devices concentrate, and where does it not?","relevant":[{"document":"wave-energy.md","page":null,"block":4,"quote":"exposed headlands rather than on sheltered channels"}],"type":"exact"} +{"id":"q023","question":"How does the river sampling protocol differ from the lake sampling protocol?","relevant":[{"document":"river-monitoring.md","page":null,"block":6,"quote":"three replicate samples at each station"},{"document":"lake-monitoring.md","page":null,"block":6,"quote":"Depth is recorded from the profiling sonde"}],"type":"multi-hop"} +{"id":"q024","question":"A street canopy and a green roof are both said to cool; what surface does each one shade?","relevant":[{"document":"urban-canopy.md","page":null,"block":4,"quote":"asphalt stores far more heat during the day"},{"document":"green-roofs.md","page":null,"block":4,"quote":"an exposed membrane absorbs far more heat during the day"}],"type":"multi-hop"} +{"id":"q025","question":"Which preservation method depends on keeping air away from the food?","relevant":[{"document":"fermentation.md","page":null,"block":2,"quote":"an anaerobic environment"}],"type":"semantic"} +{"id":"q026","question":"Why can one cold morning cost a grower the whole crop even when colonies are brought in?","relevant":[{"document":"orchard-pollination.md","page":null,"block":4,"quote":"above roughly twelve degrees Celsius"}],"type":"semantic"} +{"id":"q027","question":"绿茶应该怎样保存才能减缓氧化?","relevant":[{"document":"chinese-tea-storage.md","page":null,"block":2,"quote":"绿茶最容易氧化变质,需要低温密封保存"}],"type":"zh"} +{"id":"q028","question":"茶叶储存的相对湿度上限是多少?","relevant":[{"document":"chinese-tea-storage.md","page":null,"block":4,"quote":"相对湿度应保持在百分之五十以下"}],"type":"zh"} +{"id":"q029","question":"为什么冷冻保存的茶叶取出后不能立刻打开包装?","relevant":[{"document":"chinese-tea-storage.md","page":null,"block":6,"quote":"冷凝水会直接落在茶叶上"}],"type":"zh"} +{"id":"q030","question":"为什么潮汐能比风能和太阳能更容易提前安排发电?","relevant":[{"document":"tidal-energy.md","page":null,"block":2,"quote":"predictable decades ahead, unlike wind or solar"}],"type":"cross-lingual"} +{"id":"q031","question":"What is the installed capacity of the tidal energy installation described?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q032","question":"Which laboratory published the river monitoring protocol?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q033","question":"What is the boiling point of mercury?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q034","question":"What is the retail price of the lithium-ion cells discussed?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q035","question":"How many megawatts does the tidal array generate?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q036","question":"Who won the 2018 FIFA World Cup?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q037","question":"推移质为什么比悬移质更难测?","type":"cross-lingual","relevant":[{"document":"river-monitoring.md","page":null,"block":4,"quote":"Bedload is the harder fraction to measure"}]} +{"id":"q038","question":"河流监测中,每个站点要采几份平行样?","type":"cross-lingual","relevant":[{"document":"river-monitoring.md","page":null,"block":6,"quote":"three replicate samples at each station"}]} +{"id":"q039","question":"设计锂离子电芯时最核心的取舍是什么?","type":"cross-lingual","relevant":[{"document":"battery-chemistry.md","page":null,"block":2,"quote":"trade energy density against thermal stability"}]} +{"id":"q040","question":"富镍电池为什么对散热要求更高?","type":"cross-lingual","relevant":[{"document":"battery-chemistry.md","page":null,"block":4,"quote":"release oxygen at lower temperatures than iron phosphate"}]} +{"id":"q041","question":"电芯隔膜一旦熔化会导致什么后果?","type":"cross-lingual","relevant":[{"document":"battery-chemistry.md","page":null,"block":6,"quote":"Once the separator melts, the cell shorts internally"}]} +{"id":"q042","question":"蜜蜂大致从什么温度开始出巢觅食?","type":"cross-lingual","relevant":[{"document":"orchard-pollination.md","page":null,"block":4,"quote":"above roughly twelve degrees Celsius"}]} +{"id":"q043","question":"花期遭遇晚霜,主要受损的是什么?","type":"cross-lingual","relevant":[{"document":"orchard-pollination.md","page":null,"block":6,"quote":"destroys the flower's ovary rather than the petals"}]} +{"id":"q044","question":"为什么成片的树冠比孤立的树降温效果更好?","type":"cross-lingual","relevant":[{"document":"urban-canopy.md","page":null,"block":4,"quote":"asphalt stores far more heat during the day"}]} +{"id":"q045","type":"exact","question":"What is the default retry count in Gateway API v2?","relevant":[{"document":"api-gateway-v2.md","page":null,"block":9,"quote":"recommended retry count is **5**"}]} +{"id":"q046","type":"hard-negative","question":"What is the default request timeout of Gateway API v1?","relevant":[{"document":"api-gateway-v1.md","page":null,"block":8,"quote":"default request timeout is **30000 milliseconds**"}]} +{"id":"q047","type":"hard-negative","question":"What is the context window of Gateway API v3, in tokens?","relevant":[{"document":"api-gateway-v3.md","page":null,"block":14,"quote":"maximum context window is **65536 tokens**"}]} +{"id":"q048","type":"hard-negative","question":"What is the default rate limit of Gateway API v2?","relevant":[{"document":"api-gateway-v2.md","page":null,"block":12,"quote":"default rate limit is **3000 requests per minute**"}]} +{"id":"q049","type":"hard-negative","question":"What is the default request timeout of Gateway API v3?","relevant":[{"document":"api-gateway-v3.md","page":null,"block":8,"quote":"default request timeout is **45000 milliseconds**"}]} +{"id":"q050","type":"semantic","question":"Which Gateway API version waits longest between retries after a failed request?","relevant":[{"document":"api-gateway-v2.md","page":null,"block":9,"quote":"backoff base of **2000 milliseconds**"}]} +{"id":"q051","type":"multi-hop","question":"Compare the default timeouts and rate limits of Gateway API v1 and v3.","relevant":[{"document":"api-gateway-v1.md","page":null,"block":8,"quote":"default request timeout is **30000 milliseconds**"},{"document":"api-gateway-v1.md","page":null,"block":12,"quote":"default rate limit is **600 requests per minute**"},{"document":"api-gateway-v3.md","page":null,"block":8,"quote":"default request timeout is **45000 milliseconds**"},{"document":"api-gateway-v3.md","page":null,"block":11,"quote":"default rate limit is **1200 requests per minute**"}]} +{"id":"q052","type":"multi-hop","question":"Compare the retry count and context window of Gateway API v2 and v3.","relevant":[{"document":"api-gateway-v2.md","page":null,"block":9,"quote":"recommended retry count is **5**"},{"document":"api-gateway-v2.md","page":null,"block":15,"quote":"maximum context window is **32768 tokens**"},{"document":"api-gateway-v3.md","page":null,"block":9,"quote":"recommended retry count is **3**"},{"document":"api-gateway-v3.md","page":null,"block":14,"quote":"maximum context window is **65536 tokens**"}]} +{"id":"q053","type":"unanswerable","answerable":false,"question":"How much GPU memory does Gateway API v2 require to serve a request?","relevant":[]} +{"id":"q054","type":"unanswerable","answerable":false,"question":"What is the monthly subscription price of Gateway API v3?","relevant":[]} +{"id":"q055","type":"exact","question":"How many replicate samples are collected at each reservoir monitoring station?","relevant":[{"document":"reservoir-monitoring.md","page":null,"block":8,"quote":"four replicate samples"}]} +{"id":"q056","type":"hard-negative","question":"How many replicate samples are collected at each estuary transect station?","relevant":[{"document":"estuary-monitoring.md","page":null,"block":8,"quote":"five replicate samples"}]} +{"id":"q057","type":"hard-negative","question":"At what depth below the surface does the coastal programme place its shore-station sensor?","relevant":[{"document":"coastal-monitoring.md","page":null,"block":11,"quote":"1 metre below the surface"}]} +{"id":"q058","type":"hard-negative","question":"At what depth below the water table do groundwater boreholes carry their pressure transducer?","relevant":[{"document":"groundwater-monitoring.md","page":null,"block":11,"quote":"15 metres below the water table"}]} +{"id":"q059","type":"semantic","question":"Which monitoring programme samples most frequently?","relevant":[{"document":"coastal-monitoring.md","page":null,"block":5,"quote":"Routine sampling runs **hourly**"}]} +{"id":"q060","type":"multi-hop","question":"Compare the sampling interval and replicate count of the reservoir and groundwater programmes.","relevant":[{"document":"reservoir-monitoring.md","page":null,"block":5,"quote":"fortnightly"},{"document":"reservoir-monitoring.md","page":null,"block":8,"quote":"four replicate samples"},{"document":"groundwater-monitoring.md","page":null,"block":5,"quote":"monthly"},{"document":"groundwater-monitoring.md","page":null,"block":8,"quote":"two replicate samples"}]} +{"id":"q061","type":"semantic","question":"Why does the estuary programme take more replicate samples than the other monitoring programmes?","relevant":[{"document":"estuary-monitoring.md","page":null,"block":8,"quote":"within-station variance"}]} +{"id":"q062","type":"unanswerable","answerable":false,"question":"What is the annual operating cost of the coastal monitoring buoys?","relevant":[]} +{"id":"q063","type":"unanswerable","answerable":false,"question":"How many litres per second does the reservoir release downstream on a typical day?","relevant":[]} +{"id":"q064","question":"How much GPU memory does Gateway API v3 need to serve a request?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q065","question":"What is the uptime SLA for Gateway API v2?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q066","question":"How long does an authentication token stay valid before it expires?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q067","question":"Which SDK version should a client use with Gateway API v3?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q068","question":"Can Gateway API v2 be self-hosted on-premise?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q069","question":"How is usage invoiced for Gateway API v2?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q070","question":"Which laboratory is accredited to analyse the estuary transect samples?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q071","question":"Which vendor supplies the coastal monitoring buoys?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q072","question":"What is the annual funding for the groundwater monitoring programme?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q073","question":"How many staff work on the reservoir monitoring programme?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q074","question":"What is the warranty period on the estuary monitoring sensors?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q075","question":"How long are the coastal monitoring readings retained before deletion?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q076","question":"What encryption standard is used to transmit the river monitoring readings?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q077","question":"When was the last external audit of the lake monitoring programme?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q078","question":"What is the unit price of a river monitoring sediment sampler?","type":"unanswerable","answerable":false,"relevant":[]} diff --git a/eval/splits.json b/eval/splits.json new file mode 100644 index 0000000..6e13b15 --- /dev/null +++ b/eval/splits.json @@ -0,0 +1,80 @@ +{ + "q001": "test", + "q002": "validation", + "q003": "test", + "q004": "validation", + "q005": "test", + "q006": "test", + "q007": "test", + "q008": "test", + "q009": "validation", + "q010": "test", + "q011": "validation", + "q012": "test", + "q013": "test", + "q014": "test", + "q015": "validation", + "q016": "test", + "q017": "test", + "q018": "validation", + "q019": "test", + "q020": "test", + "q021": "validation", + "q022": "test", + "q023": "validation", + "q024": "test", + "q025": "test", + "q026": "test", + "q027": "test", + "q028": "validation", + "q029": "test", + "q030": "validation", + "q031": "validation", + "q032": "test", + "q033": "validation", + "q034": "test", + "q035": "validation", + "q036": "test", + "q037": "validation", + "q038": "test", + "q039": "test", + "q040": "validation", + "q041": "test", + "q042": "test", + "q043": "validation", + "q044": "test", + "q045": "validation", + "q046": "test", + "q047": "validation", + "q048": "test", + "q049": "test", + "q050": "validation", + "q051": "test", + "q052": "test", + "q053": "validation", + "q054": "test", + "q055": "validation", + "q056": "test", + "q057": "validation", + "q058": "test", + "q059": "test", + "q060": "validation", + "q061": "test", + "q062": "validation", + "q063": "test", + "q064": "validation", + "q065": "validation", + "q066": "validation", + "q067": "test", + "q068": "validation", + "q069": "test", + "q070": "validation", + "q071": "validation", + "q072": "test", + "q073": "validation", + "q074": "test", + "q075": "validation", + "q076": "validation", + "q077": "test", + "q078": "validation" +} diff --git a/package.json b/package.json index 7bcf86c..8f04662 100644 --- a/package.json +++ b/package.json @@ -41,7 +41,12 @@ "db:migrate": "drizzle-kit migrate", "db:push": "drizzle-kit push", "db:studio": "drizzle-kit studio", - "eval:retrieval": "npm run build && node scripts/eval-retrieval.mjs" + "eval:retrieval": "npm run build && node --experimental-transform-types --disable-warning=ExperimentalWarning --disable-warning=MODULE_TYPELESS_PACKAGE_JSON scripts/eval-retrieval.mjs", + "eval:threshold": "npm run build && node scripts/eval-threshold.mjs", + "eval:sweep": "npm run build && node scripts/eval-sweep.mjs", + "eval:blocks": "node --experimental-transform-types --import ./test/ts-resolve.mjs --disable-warning=ExperimentalWarning --disable-warning=MODULE_TYPELESS_PACKAGE_JSON scripts/eval-blocks.mjs", + "eval:scores": "npm run build && node scripts/eval-scores.mjs", + "eval:paired": "npm run build && node --experimental-transform-types --import ./test/ts-resolve.mjs --disable-warning=ExperimentalWarning --disable-warning=MODULE_TYPELESS_PACKAGE_JSON scripts/eval-paired.mjs" }, "//test": [ "`node --test` strips TypeScript types rather than compiling them, and strip-only", diff --git a/scripts/eval-blocks.mjs b/scripts/eval-blocks.mjs new file mode 100644 index 0000000..ad861f9 --- /dev/null +++ b/scripts/eval-blocks.mjs @@ -0,0 +1,54 @@ +#!/usr/bin/env node +/** + * Print the block ordinals the harness will resolve ground truth against (#192). + * + * Ground truth in `eval/questions.jsonl` is expressed as `document` + `block` + `quote`, + * and `block` is the `document_blocks.order` the ingestion pipeline produced — not a line + * number and not a paragraph index a human counted. Authoring a corpus by hand and then + * guessing those ordinals is how a dataset quietly drifts, so this prints the real ones + * through the **same loader and block builder the harness uses**. + * + * Usage: + * node --experimental-transform-types scripts/eval-blocks.mjs eval/corpus/river-monitoring.md + * node --experimental-transform-types scripts/eval-blocks.mjs eval/corpus/*.md + */ + +import { readFile, readdir } from 'node:fs/promises' +import { extname, join, resolve } from 'node:path' +import { MarkdownLoader } from '../src/main/services/loaders/MarkdownLoader.ts' +import { buildDocumentBlocks } from '../src/main/services/blocks/documentBlocks.ts' + +const targets = process.argv.slice(2) +if (targets.length === 0) { + console.error('usage: node --experimental-transform-types scripts/eval-blocks.mjs [...]') + process.exit(1) +} + +const loader = new MarkdownLoader() + +async function printFile(path) { + const buffer = await readFile(path) + const result = await loader.loadFromBuffer(buffer) + const blocks = buildDocumentBlocks({ content: result.content, structure: result.structure }) + + console.log(`\n${path} — ${blocks.length} blocks`) + for (const block of blocks) { + const text = block.text.replace(/\s+/g, ' ').trim() + const shown = text.length > 96 ? `${text.slice(0, 93)}...` : text + console.log(` ${String(block.order).padStart(3)} ${block.kind.padEnd(9)} ${shown}`) + } + return blocks.length +} + +let total = 0 +for (const target of targets) { + const path = resolve(target) + if (extname(path) === '.md') { + total += await printFile(path) + } else { + // A directory: every markdown file in it, which is the whole corpus case. + const entries = (await readdir(path)).filter((name) => name.endsWith('.md')).sort() + for (const entry of entries) total += await printFile(join(path, entry)) + } +} +console.log(`\ntotal blocks: ${total}`) diff --git a/scripts/eval-chunking.mjs b/scripts/eval-chunking.mjs index 70f4e45..52bebbf 100644 --- a/scripts/eval-chunking.mjs +++ b/scripts/eval-chunking.mjs @@ -187,7 +187,7 @@ const format4 = (value) => value.toFixed(4) const rows = results.map((result) => { const args = `${result.chunkSize}/${result.chunkOverlap}${result.respectHeadings ? ' + headings' : ''}${result.allowSpanPages ? ' + span' : ''}` - return `| ${result.label} | \`${args}\` | ${format4(result.recallAt1)} | ${format4(result.recallAt5)} | ${format4(result.mrr)} | ${format4(result.ndcgAt10)} | ${format4(result.evidencePrecisionAt5)} | ${result.chunkCount} (${formatDelta(delta(result.chunkCount, baseline.chunkCount))}) | ${result.indexingMs} ms | ${result.latencyP95Ms?.toFixed(2)} ms |` + return `| ${result.label} | \`${args}\` | ${format4(result.recallAt1)} | ${format4(result.recallAt5)} | ${format4(result.mrr)} | ${format4(result.ndcgAt10)} | ${format4(result.contextPrecision)} | ${result.chunkCount} (${formatDelta(delta(result.chunkCount, baseline.chunkCount))}) | ${result.indexingMs} ms | ${result.latencyP95Ms?.toFixed(2)} ms |` }) /** @@ -215,7 +215,7 @@ questions as \`baseline-v1.4.json\`, with dense retrieval held fixed. Only the c configuration changes, so a difference in the metrics is a difference in the input distribution retrieval is measured on. -| Variant | size/overlap | Recall@1 | Recall@5 | MRR | nDCG@10 | Evidence P@5 | Index size (Δ) | Indexing | Query p95 | +| Variant | size/overlap | Recall@1 | Recall@5 | MRR | nDCG@10 | Context P | Index size (Δ) | Indexing | Query p95 | | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | ${rows.join('\n')} diff --git a/scripts/eval-paired.mjs b/scripts/eval-paired.mjs new file mode 100644 index 0000000..d02172d --- /dev/null +++ b/scripts/eval-paired.mjs @@ -0,0 +1,367 @@ +#!/usr/bin/env node +/** + * Per-query paired deltas for dense ↔ hybrid (#192). + * + * The aggregate comparison left a question it cannot answer: hybrid is ahead on the full + * set and level on `validation`, and an average cannot say whether that is a broad small + * gain or a handful of cases it rescued. A mean over forty questions cannot distinguish + * "helps a little everywhere" from "helps a lot twice". + * + * So this reports the pair per question and the classification per query type: + * + * improved / tied / regressed, with the rank move and the nDCG@10 delta + * + * `validation` is the side that selects; the `test` breakdown **explains** the observed + * difference and must not be used to choose. Nothing here changes the shipped strategy. + * + * Metrics come from `src/main/eval/metrics.ts` rather than being reimplemented, so a delta + * cannot disagree with the metric it is a delta of. + * + * Usage: + * node scripts/eval-paired.mjs + */ + +import { spawn } from 'node:child_process' +import { existsSync, mkdtempSync, mkdirSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join, resolve } from 'node:path' + +const OUT_MD = resolve(readArg('--out=', 'docs/eval/paired-v1.6.md')) +const OUT_JSON = OUT_MD.replace(/\.md$/, '.json') + +const SPLITS = ['validation', 'test'] +const STRATEGIES = [ + { id: 'dense', label: 'dense (vector)' }, + { id: 'hybrid', label: 'hybrid (RRF of dense + BM25)' } +] + +/** How many rows to show in each mover table. */ +const MOVER_LIMIT = 8 + +function readArg(prefix, fallback) { + const arg = process.argv.find((value) => value.startsWith(prefix)) + return arg ? arg.slice(prefix.length) : fallback +} + +// Node 24 refuses to spawn a `.cmd`/`.bat` without `shell: true` (EINVAL), and the +// `.bin` entry is exactly that on Windows. Use the real binary the wrapper runs. +const { default: electronBinary } = await import('electron') +const executable = resolve(electronBinary) + +if (!existsSync(executable)) { + console.error('[paired] could not find the electron binary. Run `npm install` first.') + process.exit(1) +} + +const { ndcgAtK, recallAtK } = await import('../src/main/eval/metrics.ts') + +function runStrategy(strategy, split, outDir) { + return new Promise((resolvePromise, reject) => { + const args = [ + '.', + '--eval-harness', + '--eval-baseline=paired', + `--eval-out=${outDir}`, + `--eval-split=${split}`, + `--eval-retrieval=${strategy.id}` + ] + + const isRoot = typeof process.getuid === 'function' && process.getuid() === 0 + if (isRoot || process.env.CI) args.push('--no-sandbox') + + const child = spawn(executable, args, { + stdio: ['ignore', 'pipe', 'pipe'], + env: { ...process.env, ELECTRON_DISABLE_SECURITY_WARNINGS: '1' } + }) + + child.on('error', reject) + child.on('exit', (code) => { + if (code !== 0) { + reject(new Error(`${strategy.id} (${split}) exited with code ${code}`)) + return + } + const reportPath = join(outDir, 'baseline-paired.json') + if (!existsSync(reportPath)) { + reject(new Error(`${strategy.id} (${split}) wrote no report`)) + return + } + resolvePromise(JSON.parse(readFileSync(reportPath, 'utf8'))) + }) + }) +} + +/** One side's per-question view, with the metrics the comparison is made of. */ +function sideOf(question) { + return { + firstRelevantRank: question.firstRelevantRank, + recallAt5: recallAtK(question.matchesByRank, question.relevantCount, 5), + ndcgAt10: ndcgAtK(question.matchesByRank, question.relevantCount, 10) + } +} + +/** + * Classify the pair. `firstRelevantRank` is 0 for "never found", so the miss cases cannot be + * folded into an arithmetic delta: going from found to not-found is a regression no rank + * number expresses. + */ +function classify(dense, hybrid) { + const denseFound = dense.firstRelevantRank > 0 + const hybridFound = hybrid.firstRelevantRank > 0 + + if (denseFound && hybridFound) { + const rankDelta = dense.firstRelevantRank - hybrid.firstRelevantRank + return { + classification: rankDelta > 0 ? 'improved' : rankDelta < 0 ? 'regressed' : 'tied', + rankDelta + } + } + if (!denseFound && hybridFound) return { classification: 'improved', rankDelta: null } + if (denseFound && !hybridFound) return { classification: 'regressed', rankDelta: null } + return { classification: 'tied', rankDelta: null } +} + +/** Human-readable rank move that survives the 0 = "not found" sentinel. */ +const rankText = (rank) => (rank > 0 ? String(rank) : 'not found') + +const signed = (value, digits = 4) => + value === null ? '—' : `${value > 0 ? '+' : ''}${value.toFixed(digits)}` + +const workDir = mkdtempSync(join(tmpdir(), 'knownote-paired-')) +const bySplit = {} + +try { + for (const split of SPLITS) { + const reports = {} + for (const strategy of STRATEGIES) { + console.log(`[paired] ${split}: ${strategy.label}`) + const outDir = join(workDir, `${split}-${strategy.id}`) + mkdirSync(outDir, { recursive: true }) + reports[strategy.id] = await runStrategy(strategy, split, outDir) + } + + const denseById = new Map(reports.dense.perQuestion.map((q) => [q.id, q])) + const pairs = [] + for (const hybridQuestion of reports.hybrid.perQuestion) { + const denseQuestion = denseById.get(hybridQuestion.id) + if (!denseQuestion) continue + // Unanswerable questions have no rank to compare; they are measured by the + // abstention diagnostics, not here. + if (!denseQuestion.answerable) continue + + const dense = sideOf(denseQuestion) + const hybrid = sideOf(hybridQuestion) + const { classification, rankDelta } = classify(dense, hybrid) + pairs.push({ + id: hybridQuestion.id, + type: hybridQuestion.type, + question: hybridQuestion.question, + dense, + hybrid, + delta: { + rank: rankDelta, + recallAt5: hybrid.recallAt5 - dense.recallAt5, + ndcgAt10: hybrid.ndcgAt10 - dense.ndcgAt10 + }, + classification + }) + } + + bySplit[split] = { + questions: pairs.length, + pairs, + byType: aggregateByType(pairs), + meanDelta: { + rank: mean(pairs.map((p) => p.delta.rank).filter((value) => value !== null)), + recallAt5: mean(pairs.map((p) => p.delta.recallAt5)), + ndcgAt10: mean(pairs.map((p) => p.delta.ndcgAt10)) + } + } + } +} finally { + rmSync(workDir, { recursive: true, force: true }) +} + +function mean(values) { + if (values.length === 0) return null + return values.reduce((a, b) => a + b, 0) / values.length +} + +function aggregateByType(pairs) { + const types = [...new Set(pairs.map((p) => p.type))].sort() + return types.map((type) => { + const group = pairs.filter((p) => p.type === type) + const count = (label) => group.filter((p) => p.classification === label).length + return { + type, + questions: group.length, + improved: count('improved'), + tied: count('tied'), + regressed: count('regressed'), + meanDeltaNdcg10: mean(group.map((p) => p.delta.ndcgAt10)), + meanDeltaRank: mean(group.map((p) => p.delta.rank).filter((value) => value !== null)) + } + }) +} + +/** Wins and regressions, by rank move. Misses are listed separately and always first. */ +function movers(pairs) { + const withRank = pairs.filter((p) => p.delta.rank !== null && p.delta.rank !== 0) + const wins = withRank + .filter((p) => p.delta.rank > 0) + .sort((a, b) => b.delta.rank - a.delta.rank) + .slice(0, MOVER_LIMIT) + const losses = withRank + .filter((p) => p.delta.rank < 0) + .sort((a, b) => a.delta.rank - b.delta.rank) + .slice(0, MOVER_LIMIT) + const foundByHybrid = pairs.filter( + (p) => p.dense.firstRelevantRank === 0 && p.hybrid.firstRelevantRank > 0 + ) + const lostByHybrid = pairs.filter( + (p) => p.dense.firstRelevantRank > 0 && p.hybrid.firstRelevantRank === 0 + ) + return { wins, losses, foundByHybrid, lostByHybrid } +} + +const typeTable = (rows) => + rows + .map( + (row) => + `| ${row.type} | ${row.questions} | ${row.improved} | ${row.tied} | ${row.regressed} | ` + + `${signed(row.meanDeltaNdcg10)} | ${signed(row.meanDeltaRank, 2)} |` + ) + .join('\n') + +const moverTable = (rows) => + rows.length === 0 + ? '| — | — | — | — |' + : rows + .map( + (p) => + `| ${p.id} | ${p.type} | ${rankText(p.dense.firstRelevantRank)} | ` + + `${rankText(p.hybrid.firstRelevantRank)} | ${signed(p.delta.rank, 0)} | ` + + `${signed(p.delta.ndcgAt10)} |` + ) + .join('\n') + +const splitSection = (name, data, note) => { + const { wins, losses, foundByHybrid, lostByHybrid } = movers(data.pairs) + return `## ${name} + +${note} + +Questions compared: ${data.questions}. Mean ΔnDCG@10 ${signed(data.meanDelta.ndcgAt10)}, +mean Δrank ${signed(data.meanDelta.rank, 2)}. + +### By query type + +| Type | n | improved | tied | regressed | mean ΔnDCG@10 | mean Δrank | +| --- | --- | --- | --- | --- | --- | --- | +${typeTable(data.byType)} + +### Biggest rank wins + +| id | type | dense rank | hybrid rank | Δrank | ΔnDCG@10 | +| --- | --- | --- | --- | --- | --- | +${moverTable(wins)} + +### Biggest rank regressions + +| id | type | dense rank | hybrid rank | Δrank | ΔnDCG@10 | +| --- | --- | --- | --- | --- | --- | +${moverTable(losses)} + +${ + foundByHybrid.length > 0 + ? `**Found by hybrid, missed by dense** (${foundByHybrid.length}): ${foundByHybrid + .map((p) => `${p.id} (${p.type})`) + .join(', ')}` + : '**Found by hybrid, missed by dense**: none.' +} + +${ + lostByHybrid.length > 0 + ? `**Lost by hybrid, found by dense** (${lostByHybrid.length}): ${lostByHybrid + .map((p) => `${p.id} (${p.type})`) + .join(', ')}` + : '**Lost by hybrid, found by dense**: none.' +} +` +} + +const markdown = `# Paired dense ↔ hybrid deltas — v1.6 (#192) + +Generated by \`node scripts/eval-paired.mjs\`. Numbers are harness output; do not edit them by hand. + +## What this answers + +The aggregate comparison says hybrid is ahead on the full set and level on \`validation\`. +An average cannot say whether that is a broad small gain or a few rescued cases, and those +two readings imply different next steps. This is the per-question pair. + +Metrics come from \`src/main/eval/metrics.ts\`, so a delta cannot disagree with the metric it +is a delta of. Unanswerable questions are excluded — they have no rank to compare, and the +abstention diagnostics cover them. + +\`validation\` is the side that **selects**; the \`test\` breakdown below **explains** the +difference that was observed and must not be used to choose. Nothing here changes the +shipped strategy. + +${splitSection( + 'Validation (this side selects)', + bySplit.validation, + 'Chosen on this side, so a win here is eligible to inform a decision.' +)} +${splitSection( + 'Test (explanatory only)', + bySplit.test, + 'Not used to choose anything. Present because the aggregate difference this is meant to explain was measured here.' +)} +## Reproduce + +\`\`\`bash +npm run eval:prepare # one-time, networked model bootstrap +npm run eval:paired # offline; rewrites this file +\`\`\` +` + +mkdirSync(resolve(OUT_MD, '..'), { recursive: true }) +writeFileSync( + OUT_JSON, + `${JSON.stringify( + { + baseline: 'v1.6', + strategies: ['dense', 'hybrid'], + splits: Object.fromEntries( + SPLITS.map((split) => [ + split, + { + questions: bySplit[split].questions, + meanDelta: bySplit[split].meanDelta, + byType: bySplit[split].byType, + pairs: bySplit[split].pairs + } + ]) + ) + }, + null, + 2 + )}\n` +) +writeFileSync(OUT_MD, markdown) + +for (const split of SPLITS) { + const data = bySplit[split] + const counts = data.byType.reduce( + (acc, row) => ({ + improved: acc.improved + row.improved, + tied: acc.tied + row.tied, + regressed: acc.regressed + row.regressed + }), + { improved: 0, tied: 0, regressed: 0 } + ) + console.log( + `[paired] ${split}: ${data.questions} questions, improved ${counts.improved} / tied ${counts.tied} / regressed ${counts.regressed}, mean ΔnDCG@10 ${signed(data.meanDelta.ndcgAt10)}` + ) +} +console.log(`[paired] wrote ${OUT_JSON} and ${OUT_MD}`) diff --git a/scripts/eval-retrieval.mjs b/scripts/eval-retrieval.mjs index bccc605..d06d5ad 100644 --- a/scripts/eval-retrieval.mjs +++ b/scripts/eval-retrieval.mjs @@ -1,10 +1,15 @@ #!/usr/bin/env node /** - * Retrieval experiments for #77. + * Retrieval experiments for #77, with held-out strategy adoption (#192). * - * Runs the real RAG eval harness once per retrieval strategy against the frozen - * chunk baseline (`baseline-v1.5.json`, 1000/100), holding chunking fixed, and - * writes the comparison the issue asks for as a delta against dense. + * Runs the real RAG eval harness once per retrieval strategy on the **validation** split, + * decides there with the amended adoption rule (#192 child 10), and then re-runs the + * shipped strategy and the selected one on the **test** split. Selection never sees + * `test`; `test` only reports. + * + * That split is the whole point. The previous version decided on `split = all`, which + * meant the strategy was chosen and scored on the same questions — the same mistake the + * threshold experiment had already been fixed for. * * The harness does the measuring; this script only orchestrates and tabulates. * @@ -19,7 +24,7 @@ import { existsSync, mkdtempSync, mkdirSync, readFileSync, rmSync, writeFileSync import { tmpdir } from 'node:os' import { join, resolve } from 'node:path' -const OUT_MD = resolve(readArg('--out=', 'docs/eval/retrieval-v1.5.md')) +const OUT_MD = resolve(readArg('--out=', 'docs/eval/retrieval-v1.6.md')) const OUT_JSON = OUT_MD.replace(/\.md$/, '.json') /** @@ -34,6 +39,9 @@ const STRATEGIES = [ { id: 'hybrid', label: 'hybrid (RRF of dense + BM25)' } ] +/** The shipped strategy. Everything is reported as a delta against it. */ +const BASELINE_ID = 'dense' + function readArg(prefix, fallback) { const arg = process.argv.find((value) => value.startsWith(prefix)) return arg ? arg.slice(prefix.length) : fallback @@ -49,13 +57,18 @@ if (!existsSync(executable)) { process.exit(1) } -function runStrategy(strategy, outDir) { +// The rule lives in `src/main/eval/adoption.ts` so it can be unit tested; it decides +// whether a shipped default moves. +const { ADOPTION_METRICS, decideAdoption } = await import('../src/main/eval/adoption.ts') + +function runStrategy(strategy, split, outDir) { return new Promise((resolvePromise, reject) => { const args = [ '.', '--eval-harness', - '--eval-baseline=v1.5', + '--eval-baseline=v1.6', `--eval-out=${outDir}`, + `--eval-split=${split}`, `--eval-retrieval=${strategy.id}` ] @@ -67,115 +80,154 @@ function runStrategy(strategy, outDir) { env: { ...process.env, ELECTRON_DISABLE_SECURITY_WARNINGS: '1' } }) - let stdout = '' - child.stdout.on('data', (data) => { - stdout += data.toString() - }) - child.stderr.on('data', () => {}) - child.on('error', reject) child.on('exit', (code) => { if (code !== 0) { - reject(new Error(`strategy ${strategy.id} exited with code ${code}`)) + reject(new Error(`${strategy.id} (${split}) exited with code ${code}`)) return } - const metricsLine = /\[eval\] metrics (\{.*\})/.exec(stdout) - if (!metricsLine) { - reject(new Error(`strategy ${strategy.id} printed no metrics line`)) + const reportPath = join(outDir, 'baseline-v1.6.json') + if (!existsSync(reportPath)) { + reject(new Error(`${strategy.id} (${split}) wrote no report`)) return } - try { - resolvePromise(JSON.parse(metricsLine[1])) - } catch (error) { - reject( - new Error(`strategy ${strategy.id} printed an unreadable metrics line: ${error.message}`) - ) - } + const report = JSON.parse(readFileSync(reportPath, 'utf8')) + resolvePromise({ + id: strategy.id, + label: strategy.label, + split, + questions: report.config.questions, + answerableCount: report.byType.reduce((total, entry) => total + entry.questions, 0), + chunking: `${report.config.chunking.chunkSize}/${report.config.chunking.chunkOverlap}`, + chunkCount: report.config.chunkCount, + ...report.metrics, + ...readTiming(join(outDir, 'baseline-v1.6.md')) + }) }) }) } -/** Throughput and p95 are informational and excluded from the deterministic JSON. */ +/** p95 is informational and excluded from the deterministic JSON. */ function readTiming(mdPath) { - if (!existsSync(mdPath)) return { indexingMs: null, latencyP95Ms: null } + if (!existsSync(mdPath)) return { latencyP95Ms: null } const text = readFileSync(mdPath, 'utf8') - const indexing = /indexing (\d+) ms/.exec(text) const p95 = /p95 ([\d.]+) ms/.exec(text) - return { - indexingMs: indexing ? Number(indexing[1]) : null, - latencyP95Ms: p95 ? Number(p95[1]) : null - } + return { latencyP95Ms: p95 ? Number(p95[1]) : null } +} + +function runInto(workDir, strategy, split) { + const outDir = join(workDir, `${split}-${strategy.id}`) + mkdirSync(outDir, { recursive: true }) + return runStrategy(strategy, split, outDir) } +const format4 = (value) => value.toFixed(4) + const workDir = mkdtempSync(join(tmpdir(), 'knownote-retrieval-')) -const results = [] +let validation = [] +let test = [] +let decision = { primary: null, saturated: [], winner: null } try { + // ── Validation: this is the only phase that may choose ──────────────────────── for (const strategy of STRATEGIES) { - const outDir = join(workDir, strategy.id) - mkdirSync(outDir, { recursive: true }) - console.log(`[retrieval] running ${strategy.label}`) - const metrics = await runStrategy(strategy, outDir) - - const report = JSON.parse(readFileSync(join(outDir, 'baseline-v1.5.json'), 'utf8')) - results.push({ - id: strategy.id, - label: strategy.label, - chunking: `${report.config.chunking.chunkSize}/${report.config.chunking.chunkOverlap}`, - chunkCount: report.config.chunkCount, - ...metrics, - ...readTiming(join(outDir, 'baseline-v1.5.md')) - }) + console.log(`[retrieval] validation: ${strategy.label}`) + validation.push(await runInto(workDir, strategy, 'validation')) } -} finally { - rmSync(workDir, { recursive: true, force: true }) -} -const baseline = results.find((result) => result.id === 'dense') -if (!baseline) throw new Error('the dense strategy did not run') + const baselineRow = validation.find((row) => row.id === BASELINE_ID) + if (!baselineRow) throw new Error('the dense strategy did not run on validation') -const format4 = (value) => value.toFixed(4) + decision = decideAdoption(baselineRow, validation) -const rows = results.map( - (result) => - `| ${result.label} | ${format4(result.recallAt1)} | ${format4(result.recallAt5)} | ${format4(result.mrr)} | ${format4(result.ndcgAt10)} | ${format4(result.evidencePrecisionAt5)} | ${result.latencyP95Ms?.toFixed(2)} ms |` -) + // ── Test: reports only. Never lets the choice see these questions. ──────────── + const testIds = [BASELINE_ID] + if (decision.winner && decision.winner.id !== BASELINE_ID) testIds.push(decision.winner.id) -/** - * The rule frozen in the baseline report: Recall@5 must improve and nDCG@10 must - * not regress. Latency is reported so a win that costs 5x latency is stated as a - * trade-off, not hidden. - */ -const adopted = results.filter( - (result) => - result.id !== 'dense' && - result.recallAt5 > baseline.recallAt5 && - result.ndcgAt10 >= baseline.ndcgAt10 -) -const winner = - adopted.sort((a, b) => b.recallAt5 - a.recallAt5 || b.ndcgAt10 - a.ndcgAt10)[0] ?? null - -let outcome -if (winner) { - outcome = `\`${winner.label}\` clears the rule (Recall@5 ${format4(winner.recallAt5)} vs dense ${format4(baseline.recallAt5)}, nDCG@10 ${format4(winner.ndcgAt10)} vs ${format4(baseline.ndcgAt10)}).` -} else if (baseline.recallAt5 === 1) { - outcome = `Recall@5 is saturated at 1.0000, so the rule's first condition cannot be met by any strategy. **Dense stays the default**, and the non-dense strategies are reported as inconclusive rather than adopted or rejected on a metric that cannot move.` -} else { - outcome = `No strategy cleared the rule. **Dense stays the default.** A negative result is the point of the experiment: it is the measurement that says the extra machinery is not worth its cost on this corpus, not a failure to deliver.` + for (const id of testIds) { + const strategy = STRATEGIES.find((entry) => entry.id === id) + console.log(`[retrieval] test: ${strategy.label}`) + test.push(await runInto(workDir, strategy, 'test')) + } +} finally { + rmSync(workDir, { recursive: true, force: true }) } -const markdown = `# Retrieval experiments — v1.5 (#77) +const winner = decision.winner +const saturationNote = decision.saturated.length + ? `Saturated on the validation split (no headroom, so they cannot decide): ${decision.saturated + .map((key) => `\`${key}\``) + .join(', ')}.` + : 'No metric in the rule is saturated on the validation split.' + +const metricColumns = ['recallAt1', 'recallAt5', 'mrr', 'ndcgAt10', 'mapAt10', 'contextPrecision'] +const tableRow = (row) => + `| ${row.label} | ${metricColumns.map((key) => format4(row[key])).join(' | ')} | ${row.questions} | ${ + row.latencyP95Ms?.toFixed(2) ?? '—' + } ms |` + +const validationTable = validation.map(tableRow).join('\n') +const testTable = test.map(tableRow).join('\n') + +/** Per-metric delta of the selected strategy against dense, on the test split. */ +const testDense = test.find((row) => row.id === BASELINE_ID) +const testWinner = winner ? test.find((row) => row.id === winner.id) : null + +const deltaRows = + testWinner && testDense + ? ADOPTION_METRICS.map((key) => { + const delta = testWinner[key] - testDense[key] + const sign = delta > 0 ? '+' : '' + return `| ${key} | ${format4(testDense[key])} | ${format4(testWinner[key])} | ${sign}${format4(delta)} |` + }).join('\n') + : '' + +const outcome = winner + ? `**\`${winner.label}\` clears the rule on validation.** The deciding metric was \`${decision.primary}\` (${format4( + winner[decision.primary] + )} vs dense ${format4( + validation.find((row) => row.id === BASELINE_ID)[decision.primary] + )}), and it regressed none of ${ADOPTION_METRICS.map((key) => `\`${key}\``).join(', ')}. ${saturationNote} + +That decision was made on questions in \`test\` **not** seeing. What follows is the held-out +result, and it is the only number that should inform shipping it: + +| Metric | dense (test) | selected (test) | delta | +| --- | --- | --- | --- | +${deltaRows} + +Shipping a new default is a product decision this script does not make. It measures.` + : `**No strategy cleared the rule on validation**, so there is no adoption candidate and +\`test\` reports the shipped strategy only. ${saturationNote} + +A negative result is the point of the experiment: it is the measurement that says the extra +machinery is not worth its cost on this corpus, not a failure to deliver.` + +const markdown = `# Retrieval experiments — v1.6 (#77, #192) Generated by \`node scripts/eval-retrieval.mjs\`. Numbers are harness output; do not edit them by hand. ## What was measured -Every strategy runs the real RAG eval harness against the same corpus and the same 30 -questions as \`baseline-v1.5.json\`, with chunking held fixed at ${baseline.chunking}. Only the retrieval strategy changes. +Each strategy runs the real harness on the **validation** split of +\`eval/splits.json\`, with chunking held fixed at ${validation[0]?.chunking ?? '1000/100'}. + +The strategy is then chosen **there**, and only the chosen one (plus the shipped default) +is re-run on **test**. The choice never sees \`test\`; \`test\` only reports. The previous +version decided on \`split = all\`, which scored the choice on the questions it was fitted +to. -| Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | Evidence P@5 | Query p95 | -| --- | --- | --- | --- | --- | --- | --- | -${rows.join('\n')} +## Validation — this is where the choice happens + +| Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Context P | n | p95 | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | +${validationTable} + +## Test — reported, not selected + +| Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Context P | n | p95 | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | +${testTable} ## Not evaluated @@ -184,20 +236,15 @@ model is not available offline and inventing its numbers would defeat the point the harness. It stays open until a model can be pinned the way the embedding model is. -## Corpus limitation - -The rule's Recall@5 condition is **saturated** on this corpus: dense already scores -1.0000, so no strategy can improve it and the rule can therefore never be met here. -The metrics that still discriminate are Recall@1, MRR and nDCG@10. A hybrid result -that is better on all three but equal on Recall@5 is therefore *inconclusive*, not a -negative result, and the default is left unchanged until the comparison can run on a -corpus where Recall@5 is not already perfect. - ## Adoption rule -> Adopt a change only if Recall@5 improves and nDCG@10 does not regress. A change -> that trades a large latency increase for a marginal recall gain is a product -> decision, not an automatic win. +> Adopt a strategy when it improves the **first metric with headroom** — in the order +> \`recallAt5\`, \`nDCG@10\`, \`MRR\`, \`MAP@10\` — and regresses none of the others. A metric +> already at its maximum has no headroom and cannot decide anything; a rule that depends +> on one is unsatisfiable, not strict (#192 child 10). +> +> The rule is applied on \`validation\`. A change that trades a large latency increase for a +> marginal quality gain is a product decision, not an automatic win. ## Outcome @@ -214,11 +261,21 @@ npm run eval:retrieval # offline; runs every strategy and rewrites this file mkdirSync(resolve(OUT_MD, '..'), { recursive: true }) writeFileSync( OUT_JSON, - `${JSON.stringify({ baseline: baseline.id, chunking: baseline.chunking, strategies: results }, null, 2)}\n` + `${JSON.stringify( + { + baseline: 'v1.6', + split: { selects: 'validation', reports: 'test', manifest: 'eval/splits.json' }, + decision: { primary: decision.primary, saturated: decision.saturated, winner: winner?.id ?? null }, + validation, + test + }, + null, + 2 + )}\n` ) writeFileSync(OUT_MD, markdown) -console.log(`[retrieval] wrote ${OUT_JSON} and ${OUT_MD}`) console.log( - `[retrieval] ${winner ? `best clearing strategy: ${winner.label}` : 'no strategy cleared the rule; keep dense'}` + `[retrieval] ${winner ? `validation selected: ${winner.label}` : 'validation selected nothing; test reports dense'}` ) +console.log(`[retrieval] wrote ${OUT_JSON} and ${OUT_MD}`) diff --git a/scripts/eval-scores.mjs b/scripts/eval-scores.mjs new file mode 100644 index 0000000..43e5bbf --- /dev/null +++ b/scripts/eval-scores.mjs @@ -0,0 +1,416 @@ +#!/usr/bin/env node +/** + * Dense score diagnostics for #192. + * + * Answers one question before any threshold is chosen: **is a single dense similarity + * threshold even capable of separating "this answers the question" from "this does not"?** + * + * Three constraints, all deliberate: + * + * - **dense only.** A hybrid `score` is an RRF value (`1 / (60 + rank)`) and is not on the + * same scale as a normalised cosine. Mixing them would manufacture a new misleading + * number, which is what the threshold discussion is trying to avoid. + * - **validation only.** The distribution is used to choose where to sweep, so looking at + * `test` first would be tuning on the reporting side. + * - **the whole corpus per query** (`candidateK` ≥ index size) and `threshold = 0`, so the + * distribution is not already truncated by the parameter being investigated. + * + * Usage: + * node scripts/eval-scores.mjs + */ + +import { spawn } from 'node:child_process' +import { existsSync, mkdtempSync, mkdirSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join, resolve } from 'node:path' + +const OUT_MD = resolve(readArg('--out=', 'docs/eval/scores-v1.6.md')) +const OUT_JSON = OUT_MD.replace(/\.md$/, '.json') + +/** The shipped dense pipeline. Only the split and the candidate depth are changed. */ +const SPLIT = 'validation' +const CONTEXT_K = 3 +/** Above the index size, so every chunk is scored for every query. */ +const CANDIDATE_K = 500 + +function readArg(prefix, fallback) { + const arg = process.argv.find((value) => value.startsWith(prefix)) + return arg ? arg.slice(prefix.length) : fallback +} + +// Node 24 refuses to spawn a `.cmd`/`.bat` without `shell: true` (EINVAL), and the +// `.bin` entry is exactly that on Windows. Use the real binary the wrapper runs. +const { default: electronBinary } = await import('electron') +const executable = resolve(electronBinary) + +if (!existsSync(executable)) { + console.error('[scores] could not find the electron binary. Run `npm install` first.') + process.exit(1) +} + +function runHarness(outDir) { + return new Promise((resolvePromise, reject) => { + const args = [ + '.', + '--eval-harness', + '--eval-baseline=scores', + `--eval-out=${outDir}`, + `--eval-split=${SPLIT}`, + '--eval-retrieval=dense', + `--eval-candidate-k=${CANDIDATE_K}`, + `--eval-context-k=${CONTEXT_K}`, + '--eval-threshold=0', + '--eval-scores' + ] + + const isRoot = typeof process.getuid === 'function' && process.getuid() === 0 + if (isRoot || process.env.CI) args.push('--no-sandbox') + + const child = spawn(executable, args, { + stdio: ['ignore', 'pipe', 'pipe'], + env: { ...process.env, ELECTRON_DISABLE_SECURITY_WARNINGS: '1' } + }) + + child.on('error', reject) + child.on('exit', (code) => { + if (code !== 0) { + reject(new Error(`diagnostics harness exited with code ${code}`)) + return + } + const reportPath = join(outDir, 'baseline-scores.json') + if (!existsSync(reportPath)) { + reject(new Error('diagnostics harness wrote no report')) + return + } + resolvePromise(JSON.parse(readFileSync(reportPath, 'utf8'))) + }) + }) +} + +/** Nearest-rank quantile, matching `src/main/eval/metrics.ts`. */ +function quantile(values, p) { + if (values.length === 0) return null + const sorted = [...values].sort((a, b) => a - b) + const rank = Math.ceil((p / 100) * sorted.length) + return sorted[Math.min(Math.max(rank - 1, 0), sorted.length - 1)] +} + +/** + * `score = (1 + cosine) / 2`, from `SQLiteVectorStore`. Reported everywhere alongside the + * score because `threshold: 0.5` has been read as "cosine ≥ 0.5" and is really "cosine ≥ 0". + */ +const toCosine = (score) => (score === null ? null : 2 * score - 1) + +/** + * A difference of two scores is not an affine map of a difference of two cosines: the + * `+1` cancels, so `Δcosine = 2 * Δscore`. Applying the absolute transform to a margin + * would have produced `2Δscore - 1`, which for a near-zero margin reports a cosine margin + * near −1 — a sign flip on top of a scale error. + */ +const toCosineMargin = (margin) => (margin === null ? null : 2 * margin) + +const QUANTILES = [0, 10, 25, 50, 75, 90, 100] + +function describe(values) { + if (values.length === 0) return null + const out = {} + for (const p of QUANTILES) out[`p${p}`] = quantile(values, p) + return out +} + +const workDir = mkdtempSync(join(tmpdir(), 'knownote-scores-')) +let report +try { + console.log('[scores] running the dense harness on validation, threshold 0') + report = await runHarness(workDir) +} finally { + rmSync(workDir, { recursive: true, force: true }) +} + +const perQuestion = report.perQuestion ?? [] +const answerable = perQuestion.filter((q) => q.answerable) +const unanswerable = perQuestion.filter((q) => !q.answerable) + +if (perQuestion.some((q) => !q.retrievedScores)) { + console.error('[scores] the report carries no scores; did --eval-scores reach the harness?') + process.exit(1) +} + +/** Per-question score landmarks. */ +function landmarks(question) { + const scores = question.retrievedScores + const relevant = [] + const nonRelevant = [] + scores.forEach((score, index) => { + if ((question.matchesByRank[index] ?? []).length > 0) relevant.push(score) + else nonRelevant.push(score) + }) + if (relevant.length === 0) return null + const bestRelevant = Math.max(...relevant) + const worstRelevant = Math.min(...relevant) + const bestNonRelevant = nonRelevant.length > 0 ? Math.max(...nonRelevant) : null + return { + id: question.id, + type: question.type, + bestRelevant, + worstRelevant, + bestNonRelevant, + margin: bestNonRelevant === null ? null : bestRelevant - bestNonRelevant + } +} + +const answerableLandmarks = answerable.map(landmarks).filter((entry) => entry !== null) +const unanswerableMax = unanswerable.map((q) => Math.max(...q.retrievedScores)) + +const byType = {} +for (const entry of answerableLandmarks) { + byType[entry.type] = byType[entry.type] ?? [] + byType[entry.type].push(entry) +} + +/** + * The curve that decides whether one threshold can do both jobs. For each candidate `t`: + * + * - **hit** — share of answerable questions that still have *some* relevant passage; + * - **fullRecall** — share whose *every* ground-truth block is still covered; + * - **abstain** — share of unanswerable questions that now return nothing. + * + * A useful threshold needs `abstain` to rise while `fullRecall` holds. If every `t` that + * raises abstention also drops full recall, no single threshold can carry the job. + */ +const thresholdGrid = [] +for (let t = 0.5; t <= 1.0001; t += 0.025) thresholdGrid.push(Number(t.toFixed(3))) + +const curve = thresholdGrid.map((threshold) => { + let hit = 0 + let fullRecall = 0 + for (const question of answerable) { + const covered = new Set() + let coveredAbove = 0 + question.matchesByRank.forEach((matches, index) => { + if (question.retrievedScores[index] < threshold) return + for (const match of matches) { + if (!covered.has(match)) { + covered.add(match) + coveredAbove += 1 + } + } + }) + if (coveredAbove > 0) hit += 1 + if (question.relevantCount > 0 && coveredAbove === question.relevantCount) fullRecall += 1 + } + + const abstained = unanswerable.filter((q) => Math.max(...q.retrievedScores) < threshold).length + return { + threshold, + cosine: Number(toCosine(threshold).toFixed(3)), + hitRate: answerable.length === 0 ? 0 : hit / answerable.length, + fullRecallRate: answerable.length === 0 ? 0 : fullRecall / answerable.length, + abstentionRate: unanswerable.length === 0 ? 0 : abstained / unanswerable.length + } +}) + +const format4 = (value) => (value === null ? '—' : value.toFixed(4)) +/** + * `transform` maps a value to its cosine counterpart. It defaults to the absolute-score + * transform and is overridden for margins, because the two do not share one. + */ +const quantileRow = (label, values, transform = toCosine) => { + const d = describe(values) + if (!d) return `| ${label} | 0 | ${QUANTILES.map(() => '—').join(' | ')} |` + const cells = QUANTILES.map((p) => { + const value = d[`p${p}`] + return `${value.toFixed(4)} (${transform(value).toFixed(3)})` + }) + return `| ${label} | ${values.length} | ${cells.join(' | ')} |` +} + +const relevantBest = answerableLandmarks.map((entry) => entry.bestRelevant) +const relevantWorst = answerableLandmarks.map((entry) => entry.worstRelevant) +const nonRelevantBest = answerableLandmarks + .map((entry) => entry.bestNonRelevant) + .filter((value) => value !== null) +const margins = answerableLandmarks.map((entry) => entry.margin).filter((value) => value !== null) + +const typeRows = Object.entries(byType) + .sort(([a], [b]) => a.localeCompare(b)) + .map(([type, entries]) => { + const best = describe(entries.map((entry) => entry.bestRelevant)) + const worst = describe(entries.map((entry) => entry.worstRelevant)) + return `| ${type} | ${entries.length} | ${format4(best.p50)} | ${format4(best.p10)} | ${format4(worst.p10)} |` + }) + .join('\n') + +const curveRows = curve + .map( + (row) => + `| ${row.threshold.toFixed(3)} | ${row.cosine.toFixed(3)} | ${format4(row.hitRate)} | ` + + `${format4(row.fullRecallRate)} | ${format4(row.abstentionRate)} |` + ) + .join('\n') + +/** Where the three distributions overlap, which is what the threshold decision turns on. */ +const overlap = { + worstRelevantP10: quantile(relevantWorst, 10), + bestNonRelevantP90: quantile(nonRelevantBest, 90), + unanswerableMaxP50: quantile(unanswerableMax, 50), + unanswerableMaxP90: quantile(unanswerableMax, 90) +} +const separable = + overlap.worstRelevantP10 !== null && + overlap.unanswerableMaxP90 !== null && + overlap.worstRelevantP10 > overlap.unanswerableMaxP90 + +const markdown = `# Dense score diagnostics — v1.6 (#192) + +Generated by \`node scripts/eval-scores.mjs\`. Numbers are harness output; do not edit them by hand. + +## Read this first: the score is not a cosine + +\`SQLiteVectorStore\` computes + +\`\`\`text +score = 1 - distance / 2 +distance = 1 - cosine (sqlite-vec, distance_metric=cosine) +=> score = (1 + cosine) / 2 +\`\`\` + +so the configured \`threshold\` is an **affine map of the cosine**, not the cosine: + +| threshold | raw cosine | +| --- | --- | +| 0.3 | -0.4 | +| 0.4 | -0.2 | +| **0.5 (shipped)** | **0.0** | +| 0.6 | 0.2 | +| 0.8 | 0.6 | +| 1.0 | 1.0 | + +The shipped \`threshold = 0.5\` means **cosine ≥ 0**, which is very permissive. Every +threshold row below carries both columns so the two never get confused again, and every +mention of "the sweep looked too low" has to be read through this table. + +## What was measured + +Dense only, \`${SPLIT}\` split only, \`threshold = 0\`, \`candidateK = ${CANDIDATE_K}\` (above the +index size, so every chunk is scored for every query), \`contextK = ${CONTEXT_K}\`. + +- answerable questions: ${answerable.length} +- unanswerable questions: ${unanswerable.length} +- index size: ${report.config.chunkCount} chunks + +Hybrid is deliberately excluded: its \`score\` is an RRF value (\`1 / (60 + rank)\`) and is not +on the same scale as a normalised cosine. + +## Distributions (score, with raw cosine in brackets) + +\`best relevant\` is the highest-scoring passage that covers ground truth; \`worst relevant\` +is the lowest one that still has to survive for the question to be fully answered; +\`best non-relevant\` is the highest-scoring passage that covers nothing; \`margin\` is the +first minus the third. + +The margin is an **oracle** quantity: at runtime nothing knows which result is relevant, so +it describes how much the score separates the two — it is not a signal a product could use. +Reading it as a candidate mechanism is the mistake the runtime-signal evaluation exists to +avoid. + +Where a row shows a raw cosine in brackets: an **absolute** score maps as +\`cosine = 2·score − 1\`, while a **margin** maps as \`Δcosine = 2·Δscore\` because the +\`+1\` cancels. The \`margin\` row uses the latter, the others the former. + +| Distribution | n | min | p10 | p25 | p50 | p75 | p90 | max | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | +${quantileRow('best relevant', relevantBest)} +${quantileRow('worst relevant', relevantWorst)} +${quantileRow('best non-relevant', nonRelevantBest)} +${quantileRow('margin (best rel − best non-rel)', margins, toCosineMargin)} +${quantileRow('unanswerable max candidate', unanswerableMax)} + +## By query type + +\`best relevant\` p50 and p10, and \`worst relevant\` p10 — the last is the one that decides +whether a cross-lingual question survives a threshold that a semantic question tolerates. + +| Type | n | best rel p50 | best rel p10 | worst rel p10 | +| --- | --- | --- | --- | --- | +${typeRows} + +## Threshold curve + +For each candidate threshold: **hit** = share of answerable questions that still have some +relevant passage; **full recall** = share whose every ground-truth block is still covered; +**abstain** = share of unanswerable questions that now return nothing. + +| threshold (score) | raw cosine | answerable hit | answerable full recall | unanswerable abstain | +| --- | --- | --- | --- | --- | +${curveRows} + +## The separation question + +The threshold decision turns on whether these overlap: + +| Landmark | score | raw cosine | +| --- | --- | --- | +| worst relevant, p10 | ${format4(overlap.worstRelevantP10)} | ${overlap.worstRelevantP10 === null ? '—' : toCosine(overlap.worstRelevantP10).toFixed(3)} | +| best non-relevant, p90 | ${format4(overlap.bestNonRelevantP90)} | ${overlap.bestNonRelevantP90 === null ? '—' : toCosine(overlap.bestNonRelevantP90).toFixed(3)} | +| unanswerable max, p50 | ${format4(overlap.unanswerableMaxP50)} | ${overlap.unanswerableMaxP50 === null ? '—' : toCosine(overlap.unanswerableMaxP50).toFixed(3)} | +| unanswerable max, p90 | ${format4(overlap.unanswerableMaxP90)} | ${overlap.unanswerableMaxP90 === null ? '—' : toCosine(overlap.unanswerableMaxP90).toFixed(3)} | + +**${ + separable + ? 'The distributions separate at the p10/p90 landmarks, so a single threshold is a plausible mechanism on this corpus. Read the grid above for where to sweep.' + : 'The distributions **overlap**, so a higher threshold buys abstention by giving up required relevant passages. If the curve above shows abstention rising only as full recall falls, then the honest conclusion is that **a single dense similarity threshold cannot carry both recall and abstention** — and the next mechanism to evaluate is not a finer threshold grid but a different signal (reranker score, top1−top2 margin, per-query thresholds, or claim-level answerability).' +} + +## Reproduce + +\`\`\`bash +npm run eval:prepare # one-time, networked model bootstrap +npm run eval:scores # offline; rewrites this file +\`\`\` +` + +mkdirSync(resolve(OUT_MD, '..'), { recursive: true }) +writeFileSync( + OUT_JSON, + `${JSON.stringify( + { + baseline: 'v1.6', + split: SPLIT, + candidateK: CANDIDATE_K, + threshold: 0, + indexSize: report.config.chunkCount, + counts: { answerable: answerable.length, unanswerable: unanswerable.length }, + distributions: { + bestRelevant: describe(relevantBest), + worstRelevant: describe(relevantWorst), + bestNonRelevant: describe(nonRelevantBest), + margin: describe(margins), + unanswerableMax: describe(unanswerableMax) + }, + byType: Object.fromEntries( + Object.entries(byType).map(([type, entries]) => [ + type, + { + questions: entries.length, + bestRelevant: describe(entries.map((entry) => entry.bestRelevant)), + worstRelevant: describe(entries.map((entry) => entry.worstRelevant)) + } + ]) + ), + overlap, + separable, + curve + }, + null, + 2 + )}\n` +) +writeFileSync(OUT_MD, markdown) + +console.log(`[scores] answerable ${answerable.length}, unanswerable ${unanswerable.length}, index ${report.config.chunkCount}`) +console.log( + `[scores] worst relevant p10 = ${format4(overlap.worstRelevantP10)} (cosine ${overlap.worstRelevantP10 === null ? '—' : toCosine(overlap.worstRelevantP10).toFixed(3)}), unanswerable max p90 = ${format4(overlap.unanswerableMaxP90)}` +) +console.log(`[scores] separable = ${separable}`) +console.log(`[scores] wrote ${OUT_JSON} and ${OUT_MD}`) diff --git a/scripts/eval-sweep.mjs b/scripts/eval-sweep.mjs new file mode 100644 index 0000000..207f9d9 --- /dev/null +++ b/scripts/eval-sweep.mjs @@ -0,0 +1,263 @@ +#!/usr/bin/env node +/** + * Parameter sweep + dashboard for #192 (child 7). + * + * Runs the real harness over a bounded grid of `strategy × candidateK × contextK` and + * writes one table. The point is not to find a single number — #192 is explicit that a + * single aggregate "RAG score" says almost nothing — but to make the trade-offs + * visible side by side: quality, the context window's precision/recall, prompt size, + * index size and latency. + * + * Chunking is held fixed so a row differs from its neighbour in one parameter. + * + * Usage: + * node scripts/eval-sweep.mjs + * node scripts/eval-sweep.mjs --strategies=dense --candidate-k=10,20 --context-k=3,5 + * + * The embedding model must already be prepared (`npm run eval:prepare`). + */ + +import { spawn } from 'node:child_process' +import { existsSync, mkdtempSync, mkdirSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join, resolve } from 'node:path' + +const OUT_MD = resolve(readArg('--out=', 'docs/eval/sweep-v1.6.md')) +const OUT_JSON = OUT_MD.replace(/\.md$/, '.json') + +const STRATEGIES = readList('--strategies=', ['dense', 'hybrid']) +const CANDIDATE_KS = readNumberList('--candidate-k=', [5, 10, 20, 40]) +const CONTEXT_KS = readNumberList('--context-k=', [3, 5, 8]) + +function readArg(prefix, fallback) { + const arg = process.argv.find((value) => value.startsWith(prefix)) + return arg ? arg.slice(prefix.length) : fallback +} + +function readList(prefix, fallback) { + const raw = readArg(prefix, null) + return raw ? raw.split(',').filter(Boolean) : fallback +} + +function readNumberList(prefix, fallback) { + const raw = readArg(prefix, null) + if (!raw) return fallback + return raw.split(',').map((value) => { + const parsed = Number(value) + if (!Number.isFinite(parsed)) throw new Error(`${prefix} expects numbers, got ${value}`) + return parsed + }) +} + +// Node 24 refuses to spawn a `.cmd`/`.bat` without `shell: true` (EINVAL), and the +// `.bin` entry is exactly that on Windows. Use the real binary the wrapper runs. +const { default: electronBinary } = await import('electron') +const executable = resolve(electronBinary) + +if (!existsSync(executable)) { + console.error('[sweep] could not find the electron binary. Run `npm install` first.') + process.exit(1) +} + +function runOne(strategy, candidateK, contextK, outDir) { + return new Promise((resolvePromise, reject) => { + const args = [ + '.', + '--eval-harness', + '--eval-baseline=v1.6', + `--eval-out=${outDir}`, + `--eval-retrieval=${strategy}`, + `--eval-candidate-k=${candidateK}`, + `--eval-context-k=${contextK}` + ] + + const isRoot = typeof process.getuid === 'function' && process.getuid() === 0 + if (isRoot || process.env.CI) args.push('--no-sandbox') + + const child = spawn(executable, args, { + stdio: ['ignore', 'pipe', 'pipe'], + env: { ...process.env, ELECTRON_DISABLE_SECURITY_WARNINGS: '1' } + }) + + child.on('error', reject) + child.on('exit', (code) => { + if (code !== 0) { + reject(new Error(`${strategy} candidateK=${candidateK} contextK=${contextK} exited ${code}`)) + return + } + const reportPath = join(outDir, 'baseline-v1.6.json') + if (!existsSync(reportPath)) { + reject(new Error(`${strategy} candidateK=${candidateK} contextK=${contextK} wrote no report`)) + return + } + resolvePromise(JSON.parse(readFileSync(reportPath, 'utf8'))) + }) + }) +} + +/** p95 and index size are informational; the deterministic JSON excludes timing. */ +function readTiming(mdPath) { + if (!existsSync(mdPath)) return { latencyP95Ms: null } + const text = readFileSync(mdPath, 'utf8') + const p95 = /p95 ([\d.]+) ms/.exec(text) + return { latencyP95Ms: p95 ? Number(p95[1]) : null } +} + +const mean = (values) => (values.length === 0 ? 0 : values.reduce((a, b) => a + b, 0) / values.length) + +/** + * `contextK > candidateK` is not a cell, it is an arithmetic mistake: the harness only + * fetches `candidateK` passages, so the window can never be filled. Skipping them is why + * the default grid no longer contains rows that looked like "8 is as good as 5" when + * passages 6-8 were simply never retrieved (#192 review). The harness refuses the same + * combination, so a typo on the command line fails loudly instead of silently. + */ +const grid = [] +const skipped = [] +for (const strategy of STRATEGIES) { + for (const candidateK of CANDIDATE_KS) { + for (const contextK of CONTEXT_KS) { + if (contextK > candidateK) skipped.push({ strategy, candidateK, contextK }) + else grid.push({ strategy, candidateK, contextK }) + } + } +} + +if (skipped.length > 0) { + console.log( + `[sweep] skipped ${skipped.length} cell(s) with contextK > candidateK: ` + + skipped.map((c) => `${c.strategy} ${c.candidateK}/${c.contextK}`).join(', ') + ) +} + +const workDir = mkdtempSync(join(tmpdir(), 'knownote-sweep-')) +const rows = [] + +try { + for (const cell of grid) { + const { strategy, candidateK, contextK } = cell + console.log(`[sweep] ${strategy} candidateK=${candidateK} contextK=${contextK}`) + const outDir = join(workDir, `${strategy}-${candidateK}-${contextK}`) + mkdirSync(outDir, { recursive: true }) + const report = await runOne(strategy, candidateK, contextK, outDir) + const perQuestion = report.perQuestion ?? [] + // 质量列只看可答的问题,拒答列只看不可答的问题。把两者平均到一起,会给出一个 + // 看起来很干净的检索分数,即使同一份语料把“谁赢了 2018 世界杯”用 19 条上下文 + // 回答了(#192 评审)。 + const answerable = perQuestion.filter((q) => q.answerable) + const noResult = answerable.filter((q) => q.retrievedCount === 0).length + + rows.push({ + strategy, + candidateK, + contextK, + recallAt5: report.metrics.recallAt5, + ndcgAt10: report.metrics.ndcgAt10, + mapAt10: report.metrics.mapAt10, + contextPrecision: report.metrics.contextPrecision, + contextRecall: report.metrics.contextRecall, + noResultRate: answerable.length === 0 ? 0 : noResult / answerable.length, + meanContextChars: mean(answerable.map((q) => q.contextChars)), + unanswerableQuestions: report.unanswerable.questions, + unanswerableAbstentionRate: report.unanswerable.retrievalAbstentionRate, + unanswerableCandidates: report.unanswerable.meanCandidatesRetrieved, + unanswerableContextPassages: report.unanswerable.meanContextPassages, + chunkCount: report.config.chunkCount, + ...readTiming(join(outDir, 'baseline-v1.6.md')) + }) + } +} finally { + rmSync(workDir, { recursive: true, force: true }) +} + +const format4 = (value) => value.toFixed(4) + +const tableRows = rows + .map( + (row) => + `| ${row.strategy} | ${row.candidateK} | ${row.contextK} | ${format4(row.recallAt5)} | ` + + `${format4(row.ndcgAt10)} | ${format4(row.mapAt10)} | ${format4(row.contextPrecision)} | ` + + `${format4(row.contextRecall)} | ${format4(row.noResultRate)} | ` + + `${Math.round(row.meanContextChars)} | ${format4(row.unanswerableAbstentionRate)} | ` + + `${row.unanswerableCandidates.toFixed(1)} | ${row.unanswerableContextPassages.toFixed(1)} | ` + + `${row.chunkCount} | ${row.latencyP95Ms?.toFixed(2) ?? '—'} ms |` + ) + .join('\n') + +const best = (key, filter = () => true) => + rows.filter(filter).reduce((a, b) => (a === null || b[key] > a[key] ? b : a), null) + +/** + * A skipped cell is reported, not silently dropped: "we did not measure this" and "this + * measured the same as its neighbour" are different statements, and the earlier version + * of this file showed the second when it meant the first. + */ +const skippedNote = + skipped.length === 0 + ? '' + : `\n\n## Skipped cells\n\n\`contextK > candidateK\` cannot be filled: the harness fetches \`candidateK\`\npassages, so a wider window would contain fewer passages than it claims. These +${skipped.length} cell(s) are excluded rather than reported as equal to a narrower one:\n\n${skipped + .map((c) => `- \`${c.strategy}\` candidateK=${c.candidateK}, contextK=${c.contextK}`) + .join('\n')}\n\nThe harness refuses the same combination at the flag level, so a typo fails loudly.\n` + +const bestNdcg = best('ndcgAt10') +const bestContextPrecision = best('contextPrecision') + +const markdown = `# Parameter sweep — v1.6 (#192) + +Generated by \`node scripts/eval-sweep.mjs\`. Numbers are harness output; do not edit them by hand. + +## What was measured + +The real harness, the same corpus, chunking held fixed, over +${STRATEGIES.join(' / ')} × candidateK {${CANDIDATE_KS.join(', ')}} × contextK {${CONTEXT_KS.join(', ')}} — ${rows.length} runs. +Each row differs from its neighbour in one parameter.${skipped.length > 0 ? `\n\n${skipped.length} further cell(s) were **skipped** because \`contextK > candidateK\`; see below.` : ''} + +| Strategy | candidateK | contextK | Recall@5 | nDCG@10 | MAP@10 | Context P | Context R | No-result | Context chars | Unans. abstained | Unans. cands | Unans. ctx | Index | p95 | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | +${tableRows} +${skippedNote} + +## How to read it + +- **\`candidateK\`** moves the ranking metrics and latency: it is how wide the first + stage searches. It cannot change \`Context P\`/\`Context R\`, because those look at the + first \`contextK\` of the fused list and a prefix is unaffected by how deep the list was. +- **\`contextK\`** moves \`Context P\` and \`Context R\` and the context size, not the + ranking metrics. Wider recall rises and precision falls; that is the trade, and both + columns are here so it is visible rather than argued about. +- **Context chars** is a proxy for prompt size, not a token count: the harness pins the + embedding model, not any generation model's tokenizer. +- **No-result** is the share of *answerable* questions whose retrieval returned nothing — + a miss, and the lower the better. +- **Unans. abstained / cands / ctx** describe the *unanswerable* questions, where the + direction flips: there is no ground truth, so abstaining is correct. \`abstained\` is the + share where nothing passed the threshold; \`cands\` is how many candidates did (up to + \`candidateK\`, since the harness fetches that many for \`Recall@10\`); \`ctx\` is how many + actually reach the context window, i.e. \`min(candidates, contextK)\`. A high \`cands\` with + the usual \`ctx\` means the threshold is filtering nothing and the window is all noise. + These are the columns a threshold decision should move, and they stay out of every other + column. + +Best nDCG@10 in this grid: \`${bestNdcg.strategy}\` candidateK=${bestNdcg.candidateK}, +contextK=${bestNdcg.contextK} (${format4(bestNdcg.ndcgAt10)}). +Best context precision: \`${bestContextPrecision.strategy}\` candidateK=${bestContextPrecision.candidateK}, +contextK=${bestContextPrecision.contextK} (${format4(bestContextPrecision.contextPrecision)}). + +These are **not** recommendations. Selecting the grid maximum on the same questions is +how a benchmark becomes a lookup table; the adoption rule in \`baseline-v1.6.md\` +decides, and the \`validation\`/\`test\` split is what keeps that honest. + +## Reproduce + +\`\`\`bash +npm run eval:prepare # one-time, networked model bootstrap +npm run eval:sweep # offline; rewrites this file +\`\`\` +` + +mkdirSync(resolve(OUT_MD, '..'), { recursive: true }) +writeFileSync(OUT_JSON, `${JSON.stringify({ baseline: 'v1.6', rows }, null, 2)}\n`) +writeFileSync(OUT_MD, markdown) + +console.log(`[sweep] wrote ${OUT_JSON} and ${OUT_MD}`) diff --git a/scripts/eval-threshold.mjs b/scripts/eval-threshold.mjs new file mode 100644 index 0000000..74fe7b4 --- /dev/null +++ b/scripts/eval-threshold.mjs @@ -0,0 +1,273 @@ +#!/usr/bin/env node +/** + * Threshold derivation for #192 (child 4). + * + * `threshold: 0.5` was hand-picked, and a cosine score has no universal meaning: + * the distribution depends on the embedding model, the language, the query type and + * the chunk length. This runs the real harness once per candidate threshold on a + * **validation** split, picks a winner there, and then reports that winner on the + * **test** split — so the number that justifies the choice is not the number the + * choice was fitted to. + * + * The harness owns the split (`--eval-split=`), so both sides are measured by the + * same code path that produces the frozen baseline. + * + * Usage: + * node scripts/eval-threshold.mjs + * + * The embedding model must already be prepared (`npm run eval:prepare`). + */ + +import { spawn } from 'node:child_process' +import { existsSync, mkdtempSync, mkdirSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join, resolve } from 'node:path' + +const OUT_MD = resolve(readArg('--out=', 'docs/eval/threshold-v1.6.md')) +const OUT_JSON = OUT_MD.replace(/\.md$/, '.json') + +/** + * 0 is the "no floor" arm: it keeps the ranking intact and lets a downstream stage + * filter. It is here because #192 says the first stage may legitimately run with no + * threshold at all. + */ +const THRESHOLDS = [0, 0.3, 0.4, 0.5, 0.6] + +/** The threshold the app currently ships, so the report can say whether it holds up. */ +const PRODUCTION_THRESHOLD = 0.5 + +function readArg(prefix, fallback) { + const arg = process.argv.find((value) => value.startsWith(prefix)) + return arg ? arg.slice(prefix.length) : fallback +} + +// Node 24 refuses to spawn a `.cmd`/`.bat` without `shell: true` (EINVAL), and the +// `.bin` entry is exactly that on Windows. Use the real binary the wrapper runs. +const { default: electronBinary } = await import('electron') +const executable = resolve(electronBinary) + +if (!existsSync(executable)) { + console.error('[threshold] could not find the electron binary. Run `npm install` first.') + process.exit(1) +} + +function runOne(threshold, split, outDir) { + return new Promise((resolvePromise, reject) => { + const args = [ + '.', + '--eval-harness', + '--eval-baseline=v1.6', + `--eval-out=${outDir}`, + `--eval-split=${split}`, + `--eval-threshold=${threshold}` + ] + + const isRoot = typeof process.getuid === 'function' && process.getuid() === 0 + if (isRoot || process.env.CI) args.push('--no-sandbox') + + const child = spawn(executable, args, { + stdio: ['ignore', 'pipe', 'pipe'], + env: { ...process.env, ELECTRON_DISABLE_SECURITY_WARNINGS: '1' } + }) + + child.on('error', reject) + child.on('exit', (code) => { + if (code !== 0) { + reject(new Error(`threshold ${threshold} (${split}) exited with code ${code}`)) + return + } + const reportPath = join(outDir, 'baseline-v1.6.json') + if (!existsSync(reportPath)) { + reject(new Error(`threshold ${threshold} (${split}) wrote no report`)) + return + } + const report = JSON.parse(readFileSync(reportPath, 'utf8')) + resolvePromise(summarize(report)) + }) + }) +} + +/** + * The frozen metrics plus the two the experiment exists for: how often the threshold + * turns an answerable question into "no results at all", and how often it makes an + * unanswerable one return nothing (which is the desired outcome for those). + * + * A threshold that scores well on ranking metrics but keeps returning the whole corpus + * for a question the sources do not answer is invisible unless the second number is + * counted, and the two have to be split: one is a miss, the other is a correct refusal. + */ +function summarize(report) { + const perQuestion = report.perQuestion ?? [] + const answerable = perQuestion.filter((q) => q.answerable) + const noResult = answerable.filter((q) => q.retrievedCount === 0).length + return { + questions: perQuestion.length, + answerableQuestions: answerable.length, + noResultCount: noResult, + noResultRate: answerable.length === 0 ? 0 : noResult / answerable.length, + meanRetrieved: + answerable.length === 0 + ? 0 + : answerable.reduce((a, q) => a + q.retrievedCount, 0) / answerable.length, + /** 不可答问题时希望返回空,所以这里的“高”是好事。 */ + unanswerable: report.unanswerable, + metrics: report.metrics + } +} + +const format4 = (value) => value.toFixed(4) + +const workDir = mkdtempSync(join(tmpdir(), 'knownote-threshold-')) +const rows = [] + +try { + for (const threshold of THRESHOLDS) { + const validationDir = join(workDir, `${threshold}-validation`) + const testDir = join(workDir, `${threshold}-test`) + mkdirSync(validationDir, { recursive: true }) + mkdirSync(testDir, { recursive: true }) + + console.log(`[threshold] threshold ${threshold}: validation`) + const validation = await runOne(threshold, 'validation', validationDir) + console.log(`[threshold] threshold ${threshold}: test`) + const test = await runOne(threshold, 'test', testDir) + + rows.push({ threshold, validation, test }) + } +} finally { + rmSync(workDir, { recursive: true, force: true }) +} + +/** + * Selection rule, stated so it can be argued with (#192). + * + * Raising the threshold is only worth it if it refuses more of what the sources do not + * answer. So: take the widest threshold that does not regress the answerable quality + * metrics (nDCG@10 and context recall on validation) as the reference, then among the + * thresholds that hold that line, pick the one that refuses the most unanswerable + * questions; break ties on the lowest threshold. + * + * `threshold = 0` is the reference, because it is the arm that keeps the ranking intact. + */ +const reference = rows.find((row) => row.threshold === 0) ?? rows[0] +const EPSILON = 1e-9 +const holdsTheLine = (row) => + row.validation.metrics.ndcgAt10 >= reference.validation.metrics.ndcgAt10 - EPSILON && + row.validation.metrics.contextRecall >= reference.validation.metrics.contextRecall - EPSILON + +const eligible = rows.filter(holdsTheLine) +const ranked = [...eligible].sort( + (a, b) => + b.validation.unanswerable.retrievalAbstentionRate - + a.validation.unanswerable.retrievalAbstentionRate || + a.threshold - b.threshold +) +const winner = ranked[0] +const production = rows.find((row) => row.threshold === PRODUCTION_THRESHOLD) +if (!winner || !production) { + throw new Error('Threshold sweep requires an eligible candidate and the production baseline') +} +/** + * A flat sweep is not a weak recommendation, it is no recommendation: if no threshold + * changes either the answerable quality or the unanswerable refusal, the corpus cannot + * tell them apart and moving a product parameter on that evidence would be noise dressed + * as a result. + */ +const flat = rows.every( + (row) => + row.validation.metrics.ndcgAt10 === reference.validation.metrics.ndcgAt10 && + row.validation.unanswerable.retrievalAbstentionRate === + reference.validation.unanswerable.retrievalAbstentionRate +) + +/** + * Abstention is the point of the second number: `abstentionCount` of the unanswerable + * questions returned nothing, which is the correct outcome *at the retrieval layer*. The + * rest returned candidates the sources cannot support. + */ +const abstained = (row) => `${row.unanswerable.abstentionCount}/${row.unanswerable.questions}` +const describe = (row) => + `nDCG@10 ${format4(row.metrics.ndcgAt10)}, Recall@5 ${format4(row.metrics.recallAt5)}, ` + + `answerable no-result ${format4(row.noResultRate)}, ` + + `retrieval abstained on unanswerable ${abstained(row)}, ` + + `context passages ${row.unanswerable.meanContextPassages.toFixed(1)}` + +const outcome = flat + ? `The sweep is **flat**: every threshold from ${THRESHOLDS[0]} to ${THRESHOLDS[THRESHOLDS.length - 1]} produces the same validation nDCG@10 (${format4(reference.validation.metrics.ndcgAt10)}), the same Recall@5 (${format4(reference.validation.metrics.recallAt5)}) and the same retrieval abstention rate on unanswerable questions (${abstained(reference.validation)}). No candidate is ever filtered out, so the threshold is **non-binding** on this corpus.\n\n**No evidence to change \`threshold = ${PRODUCTION_THRESHOLD}\`.** All thresholds hold the line equally; picking one would be arbitrary. The current value can be neither validated nor falsified here, which is a property of the corpus rather than of the threshold. Note also what this does *not* establish: abstention is a retrieval-layer statement — whether the model then declines to answer needs a generator eval.` + : `**Recommended: \`threshold = ${winner.threshold}\`.**\n\n- **Validation**: ${describe(winner.validation)}\n- **Test**: ${describe(winner.test)}\n- Production ships \`${PRODUCTION_THRESHOLD}\`: validation ${describe(production.validation)}.\n\nThe rule held answerable quality at the \`threshold = 0\` level (nDCG@10 and context recall must not regress, on the validation split) and then took the threshold that abstains on the most unanswerable questions. So this is an abstention gain, not a quality gain — if answerable quality had fallen, the threshold would have been ineligible regardless of how much it abstained.` + +const tableRows = rows + .map( + (row) => + `| ${row.threshold} | ${row.validation.answerableQuestions} | ${format4(row.validation.metrics.recallAt5)} | ` + + `${format4(row.validation.metrics.ndcgAt10)} | ${format4(row.validation.noResultRate)} | ` + + `${format4(row.validation.unanswerable.retrievalAbstentionRate)} | ` + + `${row.validation.unanswerable.meanCandidatesRetrieved.toFixed(1)} | ` + + `${row.validation.unanswerable.meanContextPassages.toFixed(1)} | ${format4(row.test.metrics.ndcgAt10)} | ` + + `${format4(row.test.unanswerable.retrievalAbstentionRate)} |` + ) + .join('\n') + +const markdown = `# Threshold derivation — v1.6 (#192) + +Generated by \`node scripts/eval-threshold.mjs\`. Numbers are harness output; do not edit them by hand. + +## What was measured + +The real harness, the same corpus and the production retrieval config +(\`candidateK=20, contextK=3\`), once per candidate threshold. The **validation** +split selects; the **test** split reports. The split is the committed manifest +\`eval/splits.json\`, so the same questions are on the same side on every machine. + +Quality columns cover the answerable questions only; **Unans.** columns cover the +unanswerable ones, where returning nothing is the desired outcome and so a *higher* +abstention rate is better. Two sizes are kept apart: **cands** is how many candidates +passed the threshold (up to \`candidateK\`), **ctx** is how many reach the context window. + +| Threshold | n (val) | Recall@5 (val) | nDCG@10 (val) | No-result (val) | Unans. abstained (val) | Unans. cands (val) | Unans. ctx (val) | nDCG@10 (test) | Unans. abstained (test) | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | +${tableRows} + +## Selection rule + +Hold the answerable quality line — validation nDCG@10 and context recall must not +regress versus \`threshold = 0\` — then take the threshold that abstains on the most +unanswerable questions. Tie-break on the lowest threshold. + +Raising a threshold is only worth anything if it stops unsupported context before the +prompt; the quality gate is there so an abstention gain can never be bought with a +retrieval loss. + +## Outcome + +${outcome} + +## Caveat on this corpus + +The split removes the most obvious form of overfitting, but ${reference.validation.answerableQuestions} +answerable questions on the validation side is a thin basis for a decision, and the corpus +is still small. A threshold is a product decision with a **refusal-rate** cost attached, so +a recommendation here is only as good as the corpus behind it. Re-run this after the +corpus grows (#192 child 2). + +## Reproduce + +\`\`\`bash +npm run eval:prepare # one-time, networked model bootstrap +npm run eval:threshold # offline; rewrites this file +\`\`\` +` + +mkdirSync(resolve(OUT_MD, '..'), { recursive: true }) +writeFileSync( + OUT_JSON, + `${JSON.stringify({ baseline: 'v1.6', productionThreshold: PRODUCTION_THRESHOLD, flat, recommended: flat ? PRODUCTION_THRESHOLD : winner.threshold, rows }, null, 2)}\n` +) +writeFileSync(OUT_MD, markdown) + +console.log( + flat + ? `[threshold] flat sweep; no evidence to move off ${PRODUCTION_THRESHOLD}` + : `[threshold] recommended ${winner.threshold} (validation nDCG@10 ${format4(winner.validation.metrics.ndcgAt10)})` +) +console.log(`[threshold] wrote ${OUT_JSON} and ${OUT_MD}`) diff --git a/src/main/eval/adoption.ts b/src/main/eval/adoption.ts new file mode 100644 index 0000000..e63c519 --- /dev/null +++ b/src/main/eval/adoption.ts @@ -0,0 +1,87 @@ +/** + * The adoption rule for retrieval experiments (#192 child 10). + * + * The v1.5 rule was "Recall@5 must improve and nDCG@10 must not regress". On a corpus + * where dense already scores Recall@5 = 1.0000 that condition can never be met, so the + * rule was not strict, it was **unsatisfiable**, and every strategy comparison came + * back "inconclusive" — including a hybrid that was better on every other metric. + * + * The amendment: a metric at its maximum has no headroom and is not allowed to decide. + * The deciding metric is the first one with headroom, and a strategy is adopted when + * it improves that metric and regresses none of the others. + * + * Kept here, not inside the experiment script, so the rule can be unit tested: it + * decides whether a production default moves, and "the script printed a different + * sentence" is not a test. + */ + +/** Priority order. Recall first because a RAG miss cannot be repaired downstream. */ +export const ADOPTION_METRICS = ['recallAt5', 'ndcgAt10', 'mrr', 'mapAt10'] as const + +export type AdoptionMetric = (typeof ADOPTION_METRICS)[number] + +export type MetricBag = Record + +/** Float slack: metrics are rounded to 6 decimals before this runs. */ +export const ADOPTION_EPSILON = 1e-9 + +/** Metrics already at their maximum, which therefore cannot decide a comparison. */ +export function saturatedMetrics( + metrics: MetricBag, + order: readonly string[] = ADOPTION_METRICS +): string[] { + return order.filter((key) => metrics[key] >= 1 - ADOPTION_EPSILON) +} + +/** The first metric in priority order with room to improve, or `null` if none has. */ +export function decidingMetric( + metrics: MetricBag, + order: readonly string[] = ADOPTION_METRICS +): string | null { + return order.find((key) => metrics[key] < 1 - ADOPTION_EPSILON) ?? null +} + +/** Metrics where `candidate` is worse than `baseline` beyond the epsilon. */ +export function regressedMetrics( + candidate: MetricBag, + baseline: MetricBag, + order: readonly string[] = ADOPTION_METRICS +): string[] { + return order.filter((key) => candidate[key] < baseline[key] - ADOPTION_EPSILON) +} + +export interface AdoptionDecision { + /** The metric that decides, or `null` when every metric is already maxed out. */ + primary: string | null + saturated: string[] + /** The best strategy that clears the rule, or `null`. */ + winner: MetricBag | null +} + +/** + * Pick the strategy to recommend. `baseline` is the shipped one and must be part of + * `candidates`; a candidate that improves the deciding metric and regresses nothing + * else clears the rule, and the best such candidate by the deciding metric wins. + */ +export function decideAdoption( + baseline: MetricBag, + candidates: readonly MetricBag[], + order: readonly string[] = ADOPTION_METRICS +): AdoptionDecision { + const saturated = saturatedMetrics(baseline, order) + const primary = decidingMetric(baseline, order) + + if (primary === null) return { primary: null, saturated, winner: null } + + const cleared = candidates.filter( + (candidate) => + candidate !== baseline && + candidate[primary] > baseline[primary] + ADOPTION_EPSILON && + regressedMetrics(candidate, baseline, order).length === 0 + ) + + const winner = + [...cleared].sort((a, b) => b[primary] - a[primary])[0] ?? null + + return { primary, saturated, winner } +} diff --git a/src/main/eval/harness.ts b/src/main/eval/harness.ts index b23775c..f34ee5b 100644 --- a/src/main/eval/harness.ts +++ b/src/main/eval/harness.ts @@ -12,7 +12,7 @@ */ import { readdir, readFile } from 'fs/promises' -import { join, posix } from 'path' +import { join, posix, relative } from 'path' import { and, eq } from 'drizzle-orm' import { documentBlocks, notebooks, chunks } from '../db/schema' import type { getDatabase } from '../db' @@ -21,8 +21,10 @@ import { DEFAULT_CHUNK_OPTIONS, type ChunkOptions } from '../services/ChunkingSe import type { RetrievalStrategy } from '../services/retrieval' import { LOCAL_EMBEDDING_MODEL } from '../embedding/localModel' import { + averagePrecisionAtK, evidencePrecisionAtK, firstRelevantRank, + hitRateAtK, mean, ndcgAtK, percentile, @@ -31,12 +33,17 @@ import { } from './metrics' import type { EvalDeterministicReport, + EvalMetrics, EvalQuestion, EvalReport, EvalRelevantLocation, + EvalSplit, + EvalTypeBreakdown, QuestionReport, - ResolvedGroundTruth + ResolvedGroundTruth, + SplitAssignment } from './types' +import { assertQuestionShape, parseSplitAssignment, selectSplit } from './types' type Db = ReturnType @@ -46,7 +53,11 @@ export interface EvalHarnessOptions { /** Repo-relative label recorded in the report, so the JSON is machine-independent. */ corpusLabel: string questionsPath: string + /** 切分清单(`eval/splits.json`)。`split: 'all'` 时不读。 */ + splitsPath: string baseline: string + /** 本次只评这一份切分(#192);缺省 `all`。 */ + split: EvalSplit /** * 第一阶段每个通道的宽度,也是排名指标的评估深度(#77)。 * @@ -59,16 +70,50 @@ export interface EvalHarnessOptions { contextK: number /** Similarity floor; 0 keeps the ranking intact for ranking metrics. */ threshold: number - /** How many retrieved passages the evidence-precision metric looks at. */ - evidenceK: number /** 分块配置(#78)。实验变体通过它选择策略;缺省时用生产默认值。 */ chunkOptions: ChunkOptions + /** + * 把每个 rank 的检索分数也写进报告(`--eval-scores`)。 + * + * 缺省关闭:分数序列会让基线膨胀一倍,而基线是 CI 逐字节 diff 的文件。score + * diagnostics(#192)需要它,生产基线不需要。 + */ + includeScores?: boolean /** 检索策略(#77):dense / sparse(BM25) / hybrid(RRF)。 */ strategy: RetrievalStrategy } const NOTEBOOK_ID = 'eval-notebook' +/** A question with no `type` is grouped here rather than dropped from the report. */ +const UNTAGGED = 'untagged' + +/** + * 同一套指标既算总平均,也算每个查询类别(#192)。用一个函数是因为分组平均必须与 + * 总平均是同一个定义,否则两个数就不可比。 + */ +function summarize(perQuestion: readonly QuestionReport[], contextK: number): EvalMetrics { + return { + recallAt1: mean(perQuestion.map((q) => recallAtK(q.matchesByRank, q.relevantCount, 1))), + recallAt5: mean(perQuestion.map((q) => recallAtK(q.matchesByRank, q.relevantCount, 5))), + recallAt10: mean(perQuestion.map((q) => recallAtK(q.matchesByRank, q.relevantCount, 10))), + mrr: mean(perQuestion.map((q) => reciprocalRank(q.matchesByRank))), + ndcgAt10: mean(perQuestion.map((q) => ndcgAtK(q.matchesByRank, q.relevantCount, 10))), + hitRateAt5: mean(perQuestion.map((q) => hitRateAtK(q.matchesByRank, 5))), + mapAt10: mean( + perQuestion.map((q) => averagePrecisionAtK(q.matchesByRank, q.relevantCount, 10)) + ), + // 两个 context 指标共用同一个窗口,因为它们回答的是同一个问题的两面:送进 prompt + // 的那几条里有多少是相关的,以及需要的东西有多少真的进去了。 + contextPrecision: mean( + perQuestion.map((q) => evidencePrecisionAtK(q.matchesByRank, contextK)) + ), + contextRecall: mean( + perQuestion.map((q) => recallAtK(q.matchesByRank, q.relevantCount, contextK)) + ) + } +} + /** Normalised comparison for the optional quote drift check. */ const normalize = (text: string): string => text.toLowerCase().replace(/\s+/g, ' ').trim() @@ -177,18 +222,59 @@ async function indexCorpus( return { documentIds, chunkCount, indexingMs: performance.now() - indexingStarted } } +/** + * 读切分清单。`all` 不需要清单,所以 `all` 的运行不会因为缺清单而失败。 + * + * JSON 解析错误会把文件路径带上:清单是提交在仓库里的,一份写坏的清单应该指向它自己。 + */ +async function loadSplitAssignment(options: EvalHarnessOptions): Promise { + if (options.split === 'all') return {} + + const label = relative(process.cwd(), options.splitsPath).split(/[\\/]/).join('/') + let parsed: unknown + try { + parsed = JSON.parse(await readFile(options.splitsPath, 'utf-8')) + } catch (error) { + throw new Error(`${label} could not be read as JSON: ${String(error)}`) + } + return parseSplitAssignment(parsed, label) +} + export async function runEvalHarness( db: Db, knowledgeService: KnowledgeService, options: EvalHarnessOptions ): Promise { + // A context window wider than the retrieval depth can never be filled: the harness + // fetches `candidateK` passages and the context metrics look at `contextK` of them. + // + // Without this check the sweep silently produced rows where `contextK=5` and + // `contextK=8` at `candidateK=5` were **identical**, not because 8 assessed the same + // as 5 but because passages 6-8 did not exist (#192 review). A wrong number that + // looks like a measurement is worse than a failure. + if (options.contextK > options.candidateK) { + throw new Error( + `contextK (${options.contextK}) cannot exceed candidateK (${options.candidateK}): ` + + 'the harness retrieves candidateK passages, so a wider context window can never be filled.' + ) + } + const { documentIds, chunkCount, indexingMs } = await indexCorpus( db, knowledgeService, options.corpusDir, options.chunkOptions ) - const questions = parseQuestions(await readFile(options.questionsPath, 'utf-8')) + const allQuestions = parseQuestions(await readFile(options.questionsPath, 'utf-8')) + // 数据集自身的契约先校验:可答必须有 ground truth,不可答必须没有。搞反时指标不会 + // 报错,只会静静地失去意义。 + for (const question of allQuestions) assertQuestionShape(question) + + const assignment = await loadSplitAssignment(options) + const questions = selectSplit(allQuestions, options.split, assignment) + if (questions.length === 0) { + throw new Error(`eval split "${options.split}" selected no questions from ${options.questionsPath}`) + } const perQuestion: QuestionReport[] = [] const latencies: number[] = [] @@ -219,24 +305,51 @@ export async function runEvalHarness( .filter((index) => index >= 0) }) - perQuestion.push({ + const questionReport: QuestionReport = { id: question.id, question: question.question, + type: question.type ?? UNTAGGED, + answerable: question.answerable ?? true, firstRelevantRank: firstRelevantRank(matchesByRank), relevantCount: groundTruth.length, retrievedCount: results.length, + contextChars: results + .slice(0, options.contextK) + .reduce((total, result) => total + result.content.length, 0), matchesByRank - }) + } + // Only when asked: the score series doubles the size of the committed baseline. + if (options.includeScores) questionReport.retrievedScores = results.map((r) => r.score) + perQuestion.push(questionReport) } - const metrics = { - recallAt1: mean(perQuestion.map((q) => recallAtK(q.matchesByRank, q.relevantCount, 1))), - recallAt5: mean(perQuestion.map((q) => recallAtK(q.matchesByRank, q.relevantCount, 5))), - recallAt10: mean(perQuestion.map((q) => recallAtK(q.matchesByRank, q.relevantCount, 10))), - mrr: mean(perQuestion.map((q) => reciprocalRank(q.matchesByRank))), - ndcgAt10: mean(perQuestion.map((q) => ndcgAtK(q.matchesByRank, q.relevantCount, 10))), - evidencePrecisionAt5: mean( - perQuestion.map((q) => evidencePrecisionAtK(q.matchesByRank, options.evidenceK)) + // 不可答的问题不进排名指标:它们没有 ground truth,`recallAtK` 对它们返回的是 0/0 + // 而不是 0,把“该拒答”算成“漏报”会让整张表失真。它们自成一组。 + const answerable = perQuestion.filter((q) => q.answerable) + const unanswerableQuestions = perQuestion.filter((q) => !q.answerable) + + const metrics = summarize(answerable, options.contextK) + + // 每个类别一行,按类别名排序,所以同一个 JSON 在两次运行之间可 diff。 + const byType: EvalTypeBreakdown[] = [...new Set(answerable.map((q) => q.type))] + .sort() + .map((type) => { + const group = answerable.filter((q) => q.type === type) + return { type, questions: group.length, metrics: summarize(group, options.contextK) } + }) + + const abstentionCount = unanswerableQuestions.filter((q) => q.retrievedCount === 0).length + const unanswerable = { + questions: unanswerableQuestions.length, + abstentionCount, + // 方向与其它指标相反:没有相关资料时,检索层返回空才是对的。 + retrievalAbstentionRate: + unanswerableQuestions.length === 0 ? 0 : abstentionCount / unanswerableQuestions.length, + // 通过 threshold 的候选数,不是送进 prompt 的条数:harness 为了算 Recall@10 取满了 + // candidateK,把这个数当成 prompt 宽度会把问题说大。 + meanCandidatesRetrieved: mean(unanswerableQuestions.map((q) => q.retrievedCount)), + meanContextPassages: mean( + unanswerableQuestions.map((q) => Math.min(q.retrievedCount, options.contextK)) ) } @@ -254,16 +367,18 @@ export async function runEvalHarness( respectHeadings: chunking.respectHeadings }, retrieval: options.strategy, + split: options.split, candidateK: options.candidateK, contextK: options.contextK, threshold: options.threshold, - evidenceK: options.evidenceK, corpus: options.corpusLabel, documents: documentIds.size, questions: questions.length, chunkCount }, metrics, + byType, + unanswerable, timing: { latencyP50Ms: percentile(latencies, 50), latencyP95Ms: percentile(latencies, 95), @@ -274,26 +389,43 @@ export async function runEvalHarness( } /** Round metrics to a stable number of decimals so the JSON diffs cleanly. */ +const roundMetric = (value: number): number => Number(value.toFixed(6)) + +function roundMetrics(metrics: EvalMetrics): EvalMetrics { + return { + recallAt1: roundMetric(metrics.recallAt1), + recallAt5: roundMetric(metrics.recallAt5), + recallAt10: roundMetric(metrics.recallAt10), + mrr: roundMetric(metrics.mrr), + ndcgAt10: roundMetric(metrics.ndcgAt10), + hitRateAt5: roundMetric(metrics.hitRateAt5), + mapAt10: roundMetric(metrics.mapAt10), + contextPrecision: roundMetric(metrics.contextPrecision), + contextRecall: roundMetric(metrics.contextRecall) + } +} + export function stabilize(report: EvalReport): EvalReport { - const round = (value: number): number => Number(value.toFixed(6)) + // Scores are extra data, not metrics; they are rounded the same way so two runs of the + // same diagnostic diff cleanly. + const perQuestion = report.perQuestion.map((question) => + question.retrievedScores + ? { ...question, retrievedScores: question.retrievedScores.map(roundMetric) } + : question + ) + return { ...report, - metrics: { - recallAt1: round(report.metrics.recallAt1), - recallAt5: round(report.metrics.recallAt5), - recallAt10: round(report.metrics.recallAt10), - mrr: round(report.metrics.mrr), - ndcgAt10: round(report.metrics.ndcgAt10), - evidencePrecisionAt5: round(report.metrics.evidencePrecisionAt5) - }, + metrics: roundMetrics(report.metrics), + byType: report.byType.map((entry) => ({ ...entry, metrics: roundMetrics(entry.metrics) })), timing: { - latencyP50Ms: round(report.timing.latencyP50Ms), - latencyP95Ms: round(report.timing.latencyP95Ms), + latencyP50Ms: roundMetric(report.timing.latencyP50Ms), + latencyP95Ms: roundMetric(report.timing.latencyP95Ms), // Throughput is informational and excluded from the deterministic report; the // full report keeps it for the #78 comparison. indexingMs: Math.round(report.timing.indexingMs) }, - perQuestion: report.perQuestion + perQuestion } } @@ -304,6 +436,8 @@ export function toDeterministicReport(report: EvalReport): EvalDeterministicRepo generatedBy: report.generatedBy, config: report.config, metrics: report.metrics, + byType: report.byType, + unanswerable: report.unanswerable, perQuestion: report.perQuestion } } diff --git a/src/main/eval/metrics.ts b/src/main/eval/metrics.ts index 41f0d23..2dfda96 100644 --- a/src/main/eval/metrics.ts +++ b/src/main/eval/metrics.ts @@ -32,6 +32,56 @@ export function reciprocalRank(matchesByRank: MatchMatrix): number { return rank === 0 ? 0 : 1 / rank } +/** + * Hit rate@k: 1 when any of the first `k` ranks covers ground truth, else 0. + * + * Deliberately the bluntest metric here. Recall@k says *how much* of the ground + * truth was found; hit rate says only whether the answer was findable at all. When + * a question needs two passages, a run that finds one scores 0.5 recall and 1.0 hit + * rate — and the second number is the one that says "the model had a chance". + */ +export function hitRateAtK(matchesByRank: MatchMatrix, k: number): number { + return matchesByRank.slice(0, k).some((matches) => matches.length > 0) ? 1 : 0 +} + +/** + * Average precision@k, averaged over the ground-truth locations. + * + * Precision is measured at each rank that recovers a *fresh* ground-truth location + * (the same rule nDCG uses), then normalised by the total number of ground-truth + * locations. That makes AP the one metric here that combines ranking position with + * coverage: pulling a second relevant passage from rank 9 to rank 2 moves it, while + * recall@5 sits still. + */ +export function averagePrecisionAtK( + matchesByRank: MatchMatrix, + groundTruthCount: number, + k: number +): number { + if (groundTruthCount === 0) return 0 + + const covered = new Set() + let found = 0 + let sum = 0 + const limit = Math.min(matchesByRank.length, k) + + for (let i = 0; i < limit; i++) { + let fresh = false + for (const match of matchesByRank[i]) { + if (match < groundTruthCount && !covered.has(match)) { + covered.add(match) + fresh = true + } + } + if (fresh) { + found += 1 + sum += found / (i + 1) + } + } + + return sum / groundTruthCount +} + /** * nDCG@k with binary gains. The ideal ranking puts every ground-truth location * first, so the discount is a plain log base 2. diff --git a/src/main/eval/report.ts b/src/main/eval/report.ts index b596704..bc3290e 100644 --- a/src/main/eval/report.ts +++ b/src/main/eval/report.ts @@ -13,6 +13,30 @@ export function renderMarkdown(report: EvalReport): string { const { config, metrics, timing } = report const chunking = config.chunking + // Built before the template because a nested template literal would terminate the + // outer one; the map is a statement here, not an interpolation. + const typeRows = report.byType + .map( + (entry) => + `| ${entry.type} | ${entry.questions} | ${format(entry.metrics.recallAt5)} | ` + + `${format(entry.metrics.ndcgAt10)} | ${format(entry.metrics.hitRateAt5)} | ` + + `${format(entry.metrics.mapAt10)} |` + ) + .join('\n') + + const unanswerable = report.unanswerable + // byType only ever covers answerable questions, so its sizes add up to that count. + const answerableCount = report.byType.reduce((total, entry) => total + entry.questions, 0) + const unanswerableNote = + unanswerable.questions === 0 + ? 'This corpus carries **no** unanswerable question yet, so retrieval abstention is not\n' + + 'measured. The threshold cannot be tuned against it either: every question is answerable,\n' + + 'so every threshold returns something.' + : `| Unanswerable questions | ${unanswerable.questions} |\n` + + `| Retrieval abstained | ${format(unanswerable.retrievalAbstentionRate)} (${unanswerable.abstentionCount}/${unanswerable.questions}) |\n` + + `| Mean candidates passing the threshold | ${unanswerable.meanCandidatesRetrieved.toFixed(2)} |\n` + + `| Mean passages in the context window | ${unanswerable.meanContextPassages.toFixed(2)} |` + return `# RAG eval baseline — ${report.baseline} Generated by \`${report.generatedBy}\`. The numbers below are harness output — do not edit them by hand. @@ -26,8 +50,8 @@ Generated by \`${report.generatedBy}\`. The numbers below are harness output — | Retrieval | \`${config.retrieval}\` | | Ranks | \`candidateK=${config.candidateK}, threshold=${config.threshold}\` | | Context width | \`contextK=${config.contextK}\` | -| Evidence per query | \`evidenceK=${config.evidenceK}\` | -| Corpus | \`${config.corpus}\` (${config.documents} documents, ${config.questions} questions) | +| Corpus | \`${config.corpus}\` (${config.documents} documents) | +| Split | \`${config.split}\` (${config.questions} questions, ${answerableCount} answerable) | | Index size | ${config.chunkCount} chunks | ## Metrics @@ -39,7 +63,41 @@ Generated by \`${report.generatedBy}\`. The numbers below are harness output — | Recall@10 | ${format(metrics.recallAt10)} | | MRR | ${format(metrics.mrr)} | | nDCG@10 | ${format(metrics.ndcgAt10)} | -| Evidence precision@${config.evidenceK} | ${format(metrics.evidencePrecisionAt5)} | +| Hit rate@5 | ${format(metrics.hitRateAt5)} | +| MAP@10 | ${format(metrics.mapAt10)} | +| Context precision@${config.contextK} | ${format(metrics.contextPrecision)} | +| Context recall@${config.contextK} | ${format(metrics.contextRecall)} | + +### By query type + +A single average hides a change that helps one kind of question and hurts another. +The type comes from \`type\` in \`questions.jsonl\`; untagged questions report as +\`untagged\` rather than disappearing. + +| Type | Questions | Recall@5 | nDCG@10 | Hit rate@5 | MAP@10 | +| --- | --- | --- | --- | --- | --- | +${typeRows} + +### Unanswerable questions + +These carry no ground truth, so the correct outcome is that retrieval returns nothing. They +are excluded from every metric above — a missing ground truth is not a miss — and reported +here instead. + +**This measures retrieval-level abstention, not the model refusing.** No generator runs in +this harness, so it can show that no candidate passed the threshold; it cannot show that the +final answer would say "not in your sources". A true system refusal rate needs a +generator eval. + +A higher abstention rate is better on this row, the opposite of how every other row reads. +The two sizes are kept apart on purpose: **candidates passing the threshold** can be as high +as \`candidateK\` (the harness fetches that many to compute \`Recall@10\`), while **passages in +the context window** is what a user's prompt would actually receive. A large first number +with a small second one means the threshold filters nothing and the window is all noise. + +| Metric | Value | +| --- | --- | +${unanswerableNote} Timing is informational only and is **not** frozen: indexing ${timing.indexingMs} ms, query p50 ${timing.latencyP50Ms.toFixed(2)} ms, p95 ${timing.latencyP95Ms.toFixed(2)} ms on the @@ -50,11 +108,17 @@ corpus, so they must never be the reason two runs differ. - A retrieved passage is relevant when its provenance covers a ground-truth block. - **Recall@k** is the share of ground-truth blocks covered by the first \`k\` passages. -- **Evidence precision@${config.evidenceK}** is the share of the first \`${config.evidenceK}\` - retrieved passages that cover a ground-truth block. This is **retrieval precision**, not - answer citation recall: the harness runs no model and produces no answer. Answer-level - citation correctness is covered by the resolver (#70); a model-driven answer eval is a - separate deliverable. +- **Context precision@${config.contextK}** is the share of the first \`${config.contextK}\` + retrieved passages that cover a ground-truth block. **Context recall@${config.contextK}** is + the share of the needed ground-truth blocks that made it into that same window. Both are + deterministic: the dataset says which blocks answer the question, so no model is needed to + score the window. Together they are the trade-off a \`contextK\` decision actually makes — + a wider window finds more and carries more noise. +- This is **retrieval precision/recall, not answer citation recall**: the harness runs no + model and produces no answer. Answer-level citation correctness is covered by the + resolver (#70). Faithfulness, completeness and answer correctness need a generative model + and are **not evaluated here** — the harness runs offline with only the pinned embedding + model, the same constraint that keeps the reranker unmeasured (#170). - Ground truth is expressed in corpus identity (\`document\` relative path + \`block\` ordinal + optional \`quote\`), never a runtime \`documentId\`/\`blockId\`. @@ -66,9 +130,13 @@ first-stage width per channel, \`contextK\` is how many passages the chat prompt and \`threshold\` is the similarity floor the app ships. A benchmark that does not mirror those parameters measures a retriever nobody runs. The adopted-change rule is: -> Adopt a change only if Recall@5 improves and nDCG@10 does not regress. A change -> that trades a large latency increase for a marginal recall gain is a product -> decision, not an automatic win, and must be stated as such. +> Adopt a strategy when it improves the **first metric with headroom** — in the order +> Recall@5, nDCG@10, MRR, MAP@10 — and regresses none of the others. A metric already +> at its maximum has no headroom and cannot decide anything; a rule that depends on +> one is unsatisfiable, not strict (#192 child 10). +> +> A change that trades a large latency increase for a marginal quality gain is a +> product decision, not an automatic win, and must be stated as such. A changed result must be reproducible with: diff --git a/src/main/eval/run.ts b/src/main/eval/run.ts index 29085ad..ad45091 100644 --- a/src/main/eval/run.ts +++ b/src/main/eval/run.ts @@ -22,6 +22,7 @@ import { KnowledgeService } from '../services/KnowledgeService' import { isModelInstalled } from '../embedding/ModelRegistry' import { DEFAULT_CHUNK_OPTIONS, type ChunkOptions } from '../services/ChunkingService' import type { RetrievalStrategy } from '../services/retrieval' +import type { EvalSplit } from './types' import { runEvalHarness, stabilize, @@ -83,6 +84,17 @@ function readRetrievalStrategy(argv: readonly string[]): RetrievalStrategy { return raw } +/** + * 本次评估的切分(#192)。默认 `all`;阈值这类扫参要用 `validation` 选、`test` 报。 + */ +function readSplit(argv: readonly string[]): EvalSplit { + const raw = readOption(argv, '--eval-split=', 'all') + if (raw !== 'all' && raw !== 'validation' && raw !== 'test') { + throw new Error(`--eval-split expects all, validation or test, got ${JSON.stringify(raw)}`) + } + return raw +} + /** * Chunking config for one run (#78). The defaults are the production defaults, so * `npm run eval` with no flags still measures what ships. @@ -129,6 +141,7 @@ export async function runEvalCli(argv: readonly string[] = process.argv): Promis const prepare = argv.includes(EVAL_PREPARE_FLAG) const corpusDir = resolve(readOption(argv, '--eval-corpus=', 'eval/corpus')) const questionsPath = resolve(readOption(argv, '--eval-questions=', 'eval/questions.jsonl')) + const splitsPath = resolve(readOption(argv, '--eval-splits=', 'eval/splits.json')) const outDir = resolve(readOption(argv, '--eval-out=', 'docs/eval')) // The real profile is captured before redirecting: the model cache lives under @@ -178,14 +191,18 @@ export async function runEvalCli(argv: readonly string[] = process.argv): Promis // committed JSON is identical on every machine and checkout. corpusLabel: repoRelative(corpusDir) || 'eval/corpus', questionsPath, + splitsPath, baseline: readOption(argv, '--eval-baseline=', 'v1.6'), + split: readSplit(argv), // 默认就是生产配置(#77):先取宽,融合,再把 contextK 条送进 prompt。一个不镜像 // 线上参数的 benchmark 量的是用户永远不会跑的检索器。 candidateK: readNumberOption(argv, '--eval-candidate-k=', 20), contextK: readNumberOption(argv, '--eval-context-k=', 3), threshold: readNumberOption(argv, '--eval-threshold=', 0.5), - evidenceK: 5, chunkOptions: readChunkOptions(argv), + // `--eval-scores` puts the per-rank retrieval scores in the report. Off by default: + // the committed baseline is a CI-diffed file and the series doubles it. + includeScores: readBoolOption(argv, '--eval-scores=', argv.includes('--eval-scores')), strategy: readRetrievalStrategy(argv) } diff --git a/src/main/eval/types.ts b/src/main/eval/types.ts index c77fb90..6a33005 100644 --- a/src/main/eval/types.ts +++ b/src/main/eval/types.ts @@ -24,6 +24,111 @@ export interface EvalQuestion { question: string relevant: EvalRelevantLocation[] goldAnswer?: string + /** + * 这份语料里能不能回答(#192)。缺省 `true`。 + * + * `false` 的问题必须 `relevant: []`:它的正确答案是“资料里没有”,所以既不能拿 + * Recall 去惩罚它,也不能让它的“命中”看起来像成功。它评的是另一件事:该拒答的 + * 时候,检索有没有硬找出一堆相似但无关的上下文。 + */ + answerable?: boolean + /** + * 查询类别(#192)。自由字符串,因为语料还会长出新类别;报告按出现过的值分组, + * 缺省归入 `untagged`。 + * + * 分类的意义是:一个提升实体查询、却弄坏释义查询的策略,不该在总平均上显示成 + * 「没变化」。 + */ + type?: string +} + +/** + * 评估切分(#192)。 + * + * 阈值这类参数必须在**没参与选择**的问题上报数,否则扫参的结果只是把测试集背下来 + * 了。切分按 question id 确定性计算,所以同一份 `questions.jsonl` 在任何机器上切出 + * 同一份 validation / test。 + */ +export type EvalSplit = 'all' | 'validation' | 'test' + +/** 问题 id → 它属于切分的哪一边。 */ +export type SplitAssignment = Record + +/** + * 解析提交在仓库里的切分清单(`eval/splits.json`)。 + * + * 用显式清单而不是按 id 哈希(#192 评审)。先说清楚哈希**不是**哪里坏: + * `hash(id) % 3` 是逐 id 独立计算的,所以它是**稳定**的 —— 新增一道题不会挪动已有的题。 + * + * 它真正不能做的是表达实验设计意图: + * + * - 它无法保证小样本类别分层,于是 multi-hop / cross-lingual 这类稀有类别可能在无 + * 人选择的情况下整体落到某一侧; + * - 新增的题会被默默分到一侧,而不是被决定 —— 而 test 正是“选择”不该被拟合的那一侧。 + * + * 清单让“哪道题在哪一侧”成为一个被 review 的声明,而不是一个被算出来的结果。 + */ +export function parseSplitAssignment(raw: unknown, source: string): SplitAssignment { + if (!raw || typeof raw !== 'object' || Array.isArray(raw)) { + throw new Error( + `${source} must be a JSON object mapping a question id to "validation" or "test"` + ) + } + + const assignment: SplitAssignment = {} + for (const [id, side] of Object.entries(raw as Record)) { + if (side !== 'validation' && side !== 'test') { + throw new Error( + `${source}: question ${id} is assigned ${JSON.stringify(side)}; ` + + 'expected "validation" or "test"' + ) + } + assignment[id] = side + } + return assignment +} + +/** + * 选出一份切分。`all` 不需要清单;`validation`/`test` **必须**在清单里有条目。 + * + * 缺条目就报错,而不是默认归入某一边:一个刚加进来的问题应当先被人工决定属于哪一边, + * 而不是静静泄漏进 test。 + */ +export function selectSplit( + questions: readonly EvalQuestion[], + split: EvalSplit, + assignment: SplitAssignment +): EvalQuestion[] { + if (split === 'all') return [...questions] + return questions.filter((question) => { + const side = assignment[question.id] + if (side === undefined) { + throw new Error( + `question ${question.id} has no entry in the split manifest; assign it to ` + + '"validation" or "test" deliberately rather than letting it default into a side' + ) + } + return side === split + }) +} + +/** + * 数据集自身的契约(#192):可答的必须有 ground truth,不可答的必须没有。 + * + * 两者搞反时指标不会报错,只会静静地失去意义 —— 一个可答但没有 ground truth 的问题 + * 会被当成永远漏报,一个不可答却带着 ground truth 的问题会被当成正常命中。 + */ +export function assertQuestionShape(question: EvalQuestion): void { + const answerable = question.answerable ?? true + if (answerable && question.relevant.length === 0) { + throw new Error(`question ${question.id} is answerable but has no ground truth`) + } + if (!answerable && question.relevant.length > 0) { + throw new Error( + `question ${question.id} is unanswerable but carries ${question.relevant.length} ` + + 'ground-truth location(s)' + ) + } } /** One resolved ground-truth location, after runtime id mapping. */ @@ -42,23 +147,65 @@ export interface EvalMetrics { mrr: number ndcgAt10: number /** - * Share of the first `evidenceK` retrieved passages that cover ground truth. + * 前 5 名至少命中一个 ground-truth 块的问题占比。 + * + * 与 Recall@5 并列而不是替代:多块问题只要命中一块,hit rate 就是 1, + * 而 Recall@5 只有 0.5。「模型有没有机会」和「材料齐不齐」是两件事。 + */ + hitRateAt5: number + /** AP@10:把「排序位置」和「覆盖面」合成一个数的那个指标。 */ + mapAt10: number + /** + * 前 `contextK` 条证据里真的命中 ground-truth 的比例(context 条的精确率)。 * - * This is **retrieval precision**, not answer citation recall: no model runs in - * this harness and no answer is produced. Answer-level citation correctness is - * the resolver's job (#70) and would need a separate, model-driven eval. + * 这是**检索精度**,不是回答的引用召回:harness 不跑模型、不产生回答。回答层的引用 + * 正确性是 resolver 的事(#70),需要一个真正跑模型的 eval。 */ - evidencePrecisionAt5: number + contextPrecision: number + /** + * 答案需要的 ground-truth 块有多少进了 `contextK` 宽的窗口。 + * + * 与 Recall@10 的区别在于它量的是**窗口**:证据排在第 4、而 contextK=3 时,模型 + * 看不到它。这是一个产品指标,不只是检索指标。 + */ + contextRecall: number +} + +/** 按查询类别聚合的同一套指标(#192)。总平均会掩盖方向相反的两个变化。 */ +export interface EvalTypeBreakdown { + type: string + questions: number + metrics: EvalMetrics } export interface QuestionReport { id: string question: string + /** 查询类别,与 `EvalQuestion.type` 一致;缺省为 `untagged`。 */ + type: string + /** 与 `EvalQuestion.answerable` 一致(缺省 true)。 */ + answerable: boolean firstRelevantRank: number relevantCount: number retrievedCount: number + /** + * 送进 context 窗口的前 `contextK` 条证据的字符数(#192 child 7)。 + * + * 是字符数而不是 token 数:不同模型的分词器不同,而 harness 只固定了 embedding + * 模型。把它当 prompt 预算的代理看,不要当成某个模型的 token 数。 + */ + contextChars: number /** Ground-truth indices matched by each retrieved rank, in rank order. */ matchesByRank: number[][] + /** + * 每个 rank 的检索分数,与 `matchesByRank` 同序。 + * + * 只在 `--eval-scores` 时出现:它是 score diagnostics(#192)需要的数据,而基线 + * JSON 默认不带它——分数序列会让基线膨胀一倍,而基线是 CI 要逐字节 diff 的文件。 + * + * 注意它不是 cosine:见 `SQLiteVectorStore`,`score = (1 + cosine) / 2`。 + */ + retrievedScores?: number[] } export interface EvalReport { @@ -75,6 +222,8 @@ export interface EvalReport { respectHeadings: boolean } retrieval: string + /** 本次评估用了哪一份切分(#192):`all` / `validation` / `test`。 */ + split: string /** * 第一阶段每个通道的宽度(#77)。排名指标(Recall@K / MRR / nDCG@K)在这个深度上 * 计算,所以它必须 ≥ 指标里最大的 K。 @@ -83,13 +232,11 @@ export interface EvalReport { /** * 生产 prompt 实际取用的证据条数(#77)。 * - * 快照里记它是为了让 benchmark 描述整条线上链路,而不只是检索器;它不影响排名 - * 指标 —— 截断只是取候选列表的前缀,前缀的排序不变。 + * 快照里记它是为了让 benchmark 描述整条线上链路,而不只是检索器;它也是两个 + * context 指标的窗口宽度。 */ contextK: number threshold: number - /** How many retrieved passages the evidence-precision metric looks at. */ - evidenceK: number corpus: string documents: number questions: number @@ -97,6 +244,33 @@ export interface EvalReport { chunkCount: number } metrics: EvalMetrics + /** 每个查询类别一行;类别来自 `questions.jsonl` 的 `type`。只含可答的问题。 */ + byType: EvalTypeBreakdown[] + /** + * 不可答问题的单独一组(#192)。 + * + * 它们不进 `metrics`/`byType`:没有 ground truth,Recall 对它们是 0/0 而不是 0。 + * + * 这一组量的是**检索层弃权**,不是模型拒答:harness 不跑生成模型,所以它只能证明 + * “没有候选通过 threshold”,不能证明最终回答会说“资料里没有”。真正的 system refusal + * 要等 generator eval。 + */ + unanswerable: { + questions: number + /** 没有任何候选通过 threshold 的问题数。 */ + abstentionCount: number + /** `abstentionCount / questions`,在检索层含义下越高越好。 */ + retrievalAbstentionRate: number + /** + * 通过 threshold 的候选数,**不是**送进 prompt 的条数。 + * + * 上限是 `candidateK`(harness 为了算 Recall@10 故意取满),所以这个数接近 + * `candidateK` 时说明 threshold 基本没挡掉任何东西。 + */ + meanCandidatesRetrieved: number + /** 真正进入 context 窗口的条数:`min(retrievedCount, contextK)`。 */ + meanContextPassages: number + } /** `indexingMs` 只用于 #78 的吞吐比较;它不在确定报告里,也不该成为差异原因。 */ timing: { latencyP50Ms: number; latencyP95Ms: number; indexingMs: number } perQuestion: QuestionReport[] @@ -112,5 +286,7 @@ export interface EvalDeterministicReport { generatedBy: string config: EvalReport['config'] metrics: EvalMetrics + byType: EvalTypeBreakdown[] + unanswerable: EvalReport['unanswerable'] perQuestion: QuestionReport[] } diff --git a/src/main/services/retrieval/DenseRetriever.ts b/src/main/services/retrieval/DenseRetriever.ts index 95fece4..dc6e715 100644 --- a/src/main/services/retrieval/DenseRetriever.ts +++ b/src/main/services/retrieval/DenseRetriever.ts @@ -12,7 +12,7 @@ import type { EmbeddingService } from '../EmbeddingService' import type { CandidateHit } from './candidates' import { hydrateEvidence } from './evidence' import { buildRetrievalTrace } from './trace' -import { effectiveCandidateK, DEFAULT_TOP_K } from './types' +import { denseChannelThreshold, effectiveCandidateK, DEFAULT_TOP_K } from './types' import type { RetrievalRequest, RetrievalResult, Retriever } from './types' const STRATEGY = 'dense' @@ -31,7 +31,7 @@ export class DenseRetriever implements Retriever { // 第一阶段按 `candidateK` 取宽;`topK` 的截断由调用方决定,因为 hybrid 需要的是 // 比最终交付更宽的一池子候选。 const candidateK = effectiveCandidateK(request) - const threshold = request.threshold ?? 0.5 + const threshold = denseChannelThreshold('dense', request.threshold) // E5 要求 query 前缀,与索引时的 document 前缀区分 await this.embeddingService.ensureReady() @@ -51,7 +51,7 @@ export class DenseRetriever implements Retriever { async search(request: RetrievalRequest): Promise { const topK = request.topK ?? DEFAULT_TOP_K const candidateK = effectiveCandidateK(request) - const threshold = request.threshold ?? 0.5 + const threshold = denseChannelThreshold('dense', request.threshold) const startedAt = performance.now() // 单策略没有可精排的下游,取宽再截到 `topK` 与直接按 `topK` 查 KNN 等价; @@ -66,7 +66,7 @@ export class DenseRetriever implements Retriever { filter: request.filter, candidateK, topK, - threshold, + denseThreshold: threshold, durationMs: performance.now() - startedAt }) } diff --git a/src/main/services/retrieval/HybridRetriever.ts b/src/main/services/retrieval/HybridRetriever.ts index d0a78b9..7f946bf 100644 --- a/src/main/services/retrieval/HybridRetriever.ts +++ b/src/main/services/retrieval/HybridRetriever.ts @@ -6,6 +6,7 @@ import { DenseRetriever } from './DenseRetriever' import { hydrateEvidence } from './evidence' import { buildRetrievalTrace } from './trace' import { + denseChannelThreshold, effectiveCandidateK, DEFAULT_TOP_K, type RetrievalRequest, @@ -44,9 +45,12 @@ export class HybridRetriever implements Retriever { const startedAt = performance.now() let hits: CandidateHit[] - // BM25 没有「相似度阈值」这个概念,所以 sparse 的 trace 里 threshold 保持缺省, - // 而不是拿 dense 的 0.5 冒充。 - let threshold: number | undefined + // 传给 dense 通道的值和写进 trace 的值是 **同一个** 变量,来自同一个函数。 + // `hybrid` 也跑 dense 所以也有阈值;`sparse` 没有,于是它是 undefined。 + // + // 分开算两次就是 #192 评审发现的 bug:hybrid 的 dense 腿用着 0.5,而 trace 写 + // `undefined`,快照于是声称那次 hybrid 没有阈值。 + const denseThreshold = denseChannelThreshold(strategy, request.threshold) if (strategy === 'sparse') { hits = searchChunksFts(request.notebookId, request.query, { @@ -54,15 +58,17 @@ export class HybridRetriever implements Retriever { documentIds: request.filter?.documentIds }).slice(0, topK) } else if (strategy === 'hybrid') { - const denseHits = await this.dense.candidateHits(request) + const denseHits = await this.dense.candidateHits({ ...request, threshold: denseThreshold }) const sparseHits = searchChunksFts(request.notebookId, request.query, { limit: candidateK, documentIds: request.filter?.documentIds }) hits = rrfFuse([denseHits, sparseHits]).slice(0, topK) } else { - threshold = request.threshold ?? 0.5 - hits = (await this.dense.candidateHits({ ...request, threshold })).slice(0, topK) + hits = (await this.dense.candidateHits({ ...request, threshold: denseThreshold })).slice( + 0, + topK + ) } const evidence = hits.length === 0 ? [] : hydrateEvidence(getDatabase(), hits) @@ -74,7 +80,7 @@ export class HybridRetriever implements Retriever { filter: request.filter, candidateK, topK, - threshold, + denseThreshold, durationMs: performance.now() - startedAt }) } diff --git a/src/main/services/retrieval/trace.ts b/src/main/services/retrieval/trace.ts index ae788cc..70a7c80 100644 --- a/src/main/services/retrieval/trace.ts +++ b/src/main/services/retrieval/trace.ts @@ -5,7 +5,7 @@ export interface RetrievalTraceInput { filter?: RetrievalFilter candidateK: number topK: number - threshold?: number + denseThreshold?: number durationMs: number } @@ -30,7 +30,7 @@ export function buildRetrievalTrace(input: RetrievalTraceInput): RetrievalTrace durationMs: input.durationMs } - if (input.threshold !== undefined) trace.threshold = input.threshold + if (input.denseThreshold !== undefined) trace.denseThreshold = input.denseThreshold return trace } diff --git a/src/main/services/retrieval/types.ts b/src/main/services/retrieval/types.ts index 921a345..fc04fae 100644 --- a/src/main/services/retrieval/types.ts +++ b/src/main/services/retrieval/types.ts @@ -110,6 +110,27 @@ export function effectiveCandidateK(request: { return Math.max(request.candidateK ?? DEFAULT_CANDIDATE_K, topK) } +/** 没有指定时 dense 通道的相似度下限,与引入双 K 之前一致。 */ +export const DEFAULT_DENSE_THRESHOLD = 0.5 + +/** + * 某个策略真正作用在 **dense 通道** 上的相似度下限。 + * + * `hybrid` 也跑 dense,所以它同样有阈值;只有 `sparse` 没有,因为 BM25 没有「相似 + * 度阈值」这个概念。 + * + * 用 **一个** 函数产出这个值,是为了让「传给 dense 通道的值」和「写进 trace 的值」无法 + * 再分开:#192 评审发现的 bug 就是它们各自有一个 `?? 0.5` —— hybrid 的 dense 腿用了 + * 0.5,而 trace 写的 `threshold: undefined`,于是快照无法复现那次检索。 + */ +export function denseChannelThreshold( + strategy: RetrievalStrategy, + requested?: number +): number | undefined { + if (strategy === 'sparse') return undefined + return requested ?? DEFAULT_DENSE_THRESHOLD +} + /** * 一次检索实际生效的参数。 * @@ -127,7 +148,16 @@ export interface RetrievalTrace { candidateK: number /** 最终交付的证据条数。 */ topK: number - threshold?: number + /** + * 真正作用在 **dense 通道** 上的相似度下限。 + * + * 字段名不是 `threshold` 而是 `denseThreshold`,因为 `hybrid` 也在跑 dense:它不 + * 是「没有阈值」,而是 dense 那一路有 0.5。叫 `threshold` 会让快照看上去说 hybrid + * 没有阈值,于是“这次检索是怎么发生的”就复现不出来了(#192 评审)。 + * + * 缺省只表示 **dense 通道没跑**(`sparse`),不是「阈值等于 0」。 + */ + denseThreshold?: number durationMs: number } diff --git a/src/main/vectorstore/SQLiteVectorStore.ts b/src/main/vectorstore/SQLiteVectorStore.ts index 8fc300b..8f696f1 100644 --- a/src/main/vectorstore/SQLiteVectorStore.ts +++ b/src/main/vectorstore/SQLiteVectorStore.ts @@ -171,7 +171,11 @@ export class SQLiteVectorStore implements VectorStore { // 转换结果 const queryResults: QueryResult[] = results.map((row) => { - // cosine 距离转相似度(0-1) + // `score = 1 - distance / 2 = (1 + cosine) / 2`。 + // + // **这不是 cosine 本身**,而是仿射映射:score 0.5 对应 cosine 0,score 0.6 对应 + // cosine 0.2,score 1.0 才是 cosine 1.0。配置里的 `threshold` 就是这个 score, + // 把 0.5 读成“cosine ≥ 0.5”会把门槛高估很多(#192 评审)。 const score = 1 - row.distance / 2 return { diff --git a/src/shared/types/chat.ts b/src/shared/types/chat.ts index 55e7170..fdd237b 100644 --- a/src/shared/types/chat.ts +++ b/src/shared/types/chat.ts @@ -162,8 +162,16 @@ export interface RetrievalSnapshot { */ candidateK: number topK: number - /** 缺省表示该策略没有阈值,不是「阈值等于 0」。 */ - threshold?: number + /** + * 作用在 **dense 通道** 上的相似度下限。 + * + * `hybrid` 也有这个值(它的 dense 那一路),所以缺省只表示 dense 通道没跑 + * (`sparse`),不是「阈值等于 0」。 + * + * #192 评审之前这个字段叫 `threshold`:那时 hybrid 的快照写的是 undefined,而它 + * 实际跑了带 0.5 的 dense。旧记录里的值本来就是 dense 阈值,解析时按 dense 阈值读。 + */ + denseThreshold?: number durationMs: number } diff --git a/src/shared/utils/answerSources.ts b/src/shared/utils/answerSources.ts index ffb7271..321b478 100644 --- a/src/shared/utils/answerSources.ts +++ b/src/shared/utils/answerSources.ts @@ -115,8 +115,12 @@ export const parseRetrievalSnapshot = (metadata: unknown): RetrievalSnapshot | n } } - const threshold = toFiniteNumber(candidate.threshold) - if (threshold !== undefined) snapshot.threshold = threshold + // Written as `threshold` before the #192 review established that `hybrid` also runs a + // dense leg. A legacy value is a dense threshold and is read as one; the backfill says + // what that retrieval actually did. + const denseThreshold = + toFiniteNumber(candidate.denseThreshold) ?? toFiniteNumber(candidate.threshold) + if (denseThreshold !== undefined) snapshot.denseThreshold = denseThreshold return snapshot } diff --git a/test/answerSources.test.ts b/test/answerSources.test.ts index 5de2726..3996f39 100644 --- a/test/answerSources.test.ts +++ b/test/answerSources.test.ts @@ -153,7 +153,7 @@ test('a retrieval snapshot round-trips', () => { scope: { documentIds: ['doc_1', 'doc_2'] }, candidateK: 20, topK: 8, - threshold: 0.5, + denseThreshold: 0.5, durationMs: 42.4 } }) @@ -163,7 +163,7 @@ test('a retrieval snapshot round-trips', () => { scope: { documentIds: ['doc_1', 'doc_2'] }, candidateK: 20, topK: 8, - threshold: 0.5, + denseThreshold: 0.5, durationMs: 42.4 }) }) @@ -174,7 +174,26 @@ test('a snapshot without a scope means the whole notebook', () => { }) assert.deepEqual(snapshot?.scope, {}) - assert.equal(snapshot?.threshold, undefined) + assert.equal(snapshot?.denseThreshold, undefined) +}) + +/** + * Snapshots written before the #192 review called the field `threshold`. The value was + * always the dense leg's floor, so it is read as one rather than dropped. + */ +test('a snapshot with the legacy `threshold` field reads it as the dense threshold', () => { + const snapshot = parseRetrievalSnapshot({ + retrievalSnapshot: { + strategy: 'hybrid', + scope: {}, + topK: 3, + threshold: 0.5, + durationMs: 7 + } + }) + + assert.equal(snapshot?.denseThreshold, 0.5) + assert.equal(snapshot?.candidateK, 3) }) /** diff --git a/test/evalAdoption.test.ts b/test/evalAdoption.test.ts new file mode 100644 index 0000000..ad61e05 --- /dev/null +++ b/test/evalAdoption.test.ts @@ -0,0 +1,86 @@ +import { test } from 'node:test' +import assert from 'node:assert/strict' +import { + ADOPTION_METRICS, + decideAdoption, + decidingMetric, + regressedMetrics, + saturatedMetrics +} from '../src/main/eval/adoption.ts' + +/** + * The adoption rule (#192 child 10) decides whether a shipped default moves, so it is + * pinned here rather than trusted to the sentence the experiment script prints. + * + * The bug it exists for: the v1.5 rule made Recall@5 — already 1.0000 on the corpus — + * its deciding condition, so a hybrid that was better on every other metric came back + * "inconclusive" forever. + */ + +const bag = (overrides: Record = {}): Record => ({ + recallAt5: 0.9, + ndcgAt10: 0.9, + mrr: 0.9, + mapAt10: 0.9, + ...overrides +}) + +test('a metric at its maximum is saturated and cannot decide', () => { + assert.deepEqual(saturatedMetrics(bag({ recallAt5: 1 })), ['recallAt5']) + assert.deepEqual(saturatedMetrics(bag()), []) +}) + +test('the deciding metric is the first one with headroom', () => { + assert.equal(decidingMetric(bag({ recallAt5: 1 })), 'ndcgAt10') + assert.equal(decidingMetric(bag({ recallAt5: 1, ndcgAt10: 1 })), 'mrr') + assert.equal(decidingMetric(bag()), 'recallAt5') + assert.equal(decidingMetric(bag({ recallAt5: 1, ndcgAt10: 1, mrr: 1, mapAt10: 1 })), null) +}) + +test('a regression is measured against the baseline, beyond the float slack', () => { + const baseline = bag() + assert.deepEqual(regressedMetrics(bag({ ndcgAt10: 0.8 }), baseline), ['ndcgAt10']) + // Rounding to 6 decimals must not read as a regression. + assert.deepEqual(regressedMetrics(bag({ ndcgAt10: 0.9 - 1e-12 }), baseline), []) +}) + +test('the v1.5 stalemate is resolved: improving the deciding metric is enough', () => { + const dense = bag({ recallAt5: 1, ndcgAt10: 0.9437, mrr: 0.9278, mapAt10: 0.9222 }) + const hybrid = bag({ recallAt5: 1, ndcgAt10: 0.9561, mrr: 0.9444, mapAt10: 0.9389 }) + + const decision = decideAdoption(dense, [dense, hybrid]) + + assert.deepEqual(decision.saturated, ['recallAt5']) + assert.equal(decision.primary, 'ndcgAt10') + assert.equal(decision.winner, hybrid) +}) + +test('improving one metric while regressing another does not clear the rule', () => { + const dense = bag({ recallAt5: 1, ndcgAt10: 0.9437, mrr: 0.9278 }) + // Better nDCG, worse MRR: exactly the trade the rule refuses to make silently. + const trade = bag({ recallAt5: 1, ndcgAt10: 0.99, mrr: 0.9 }) + + assert.equal(decideAdoption(dense, [dense, trade]).winner, null) +}) + +test('the baseline never clears the rule against itself', () => { + const dense = bag({ recallAt5: 1 }) + assert.equal(decideAdoption(dense, [dense]).winner, null) +}) + +test('when every metric is saturated nothing can be adopted', () => { + const all = bag({ recallAt5: 1, ndcgAt10: 1, mrr: 1, mapAt10: 1 }) + const decision = decideAdoption(all, [all, bag({ ndcgAt10: 1 })]) + + assert.equal(decision.primary, null) + assert.equal(decision.winner, null) + assert.deepEqual(decision.saturated, [...ADOPTION_METRICS]) +}) + +test('the best clearing candidate by the deciding metric wins', () => { + const dense = bag({ recallAt5: 1, ndcgAt10: 0.9 }) + const good = bag({ recallAt5: 1, ndcgAt10: 0.95 }) + const better = bag({ recallAt5: 1, ndcgAt10: 0.98 }) + + assert.equal(decideAdoption(dense, [dense, good, better]).winner, better) +}) diff --git a/test/evalHarness.test.ts b/test/evalHarness.test.ts new file mode 100644 index 0000000..65a51e1 --- /dev/null +++ b/test/evalHarness.test.ts @@ -0,0 +1,64 @@ +import { test } from 'node:test' +import assert from 'node:assert/strict' +import { runEvalHarness } from '../src/main/eval/harness.ts' + +/** + * The harness's own invariants (#192). These run before any database or retrieval work, + * so they can be pinned without a store: the check is the first statement of + * `runEvalHarness`, and a violation must fail loudly rather than produce a number. + */ + +const options = (overrides: Record = {}): Record => ({ + corpusDir: 'eval/corpus', + corpusLabel: 'eval/corpus', + questionsPath: 'eval/questions.jsonl', + splitsPath: 'eval/splits.json', + baseline: 'test', + split: 'all', + candidateK: 20, + contextK: 3, + threshold: 0.5, + chunkOptions: { + chunkSize: 1000, + chunkOverlap: 100, + minChunkSize: 100, + allowSpanPages: false, + respectHeadings: false + }, + strategy: 'dense', + ...overrides +}) + +/** + * A context window wider than the retrieval depth can never be filled. The sweep used + * to contain `candidateK=5, contextK=8` and report it as identical to `contextK=5` — + * not because 8 assessed the same as 5, but because passages 6-8 did not exist (#192 + * review). A wrong number that looks like a measurement is worse than a failure. + */ +test('a context window wider than the retrieval depth is refused', async () => { + await assert.rejects( + () => runEvalHarness({} as never, {} as never, options({ candidateK: 5, contextK: 8 }) as never), + /contextK \(8\) cannot exceed candidateK \(5\)/ + ) +}) + +test('the refusal names both numbers, so the fix is obvious', async () => { + await assert.rejects( + () => runEvalHarness({} as never, {} as never, options({ candidateK: 3, contextK: 10 }) as never), + (error: Error) => { + assert.match(error.message, /candidateK \(3\)/) + assert.match(error.message, /contextK \(10\)/) + assert.match(error.message, /can never be filled/) + return true + } + ) +}) + +test('contextK equal to candidateK is allowed and reaches further invariants', async () => { + // No database is provided, so the call must fail *after* the window check — proving the + // boundary value is accepted rather than rejected. + await assert.rejects( + () => runEvalHarness({} as never, {} as never, options({ candidateK: 5, contextK: 5 }) as never), + (error: Error) => !/cannot exceed/.test(error.message) + ) +}) diff --git a/test/evalMetrics.test.ts b/test/evalMetrics.test.ts index 1520dc6..3cd8d0e 100644 --- a/test/evalMetrics.test.ts +++ b/test/evalMetrics.test.ts @@ -1,8 +1,10 @@ import { test } from 'node:test' import assert from 'node:assert/strict' import { + averagePrecisionAtK, evidencePrecisionAtK, firstRelevantRank, + hitRateAtK, mean, type MatchMatrix, ndcgAtK, @@ -101,6 +103,42 @@ test('evidence precision counts grounded passages over retrieved passages', () = assert.equal(evidencePrecisionAtK([[0], [0], [0]], 3), 1) }) +test('hit rate@k says whether the answer was reachable at all', () => { + assert.equal(hitRateAtK([[0], [], []], 5), 1) + assert.equal(hitRateAtK([[], [], [1]], 1), 0) + assert.equal(hitRateAtK([[], [], [1]], 3), 1) + assert.equal(hitRateAtK([[], []], 5), 0) + assert.equal(hitRateAtK([], 5), 0) +}) + +/** + * The distinction the metric exists for: a two-passage question that finds only + * one has 0.5 recall but full hit rate — the model had a chance, and recall is what + * says the material was incomplete. + */ +test('hit rate can be 1 while recall@k is only half', () => { + const matches: MatchMatrix = [[0], []] + assert.equal(hitRateAtK(matches, 5), 1) + assert.equal(recallAtK(matches, 2, 5), 0.5) +}) + +test('average precision rewards finding the same ground truth earlier', () => { + // One relevant at rank 1: AP = 1. + assert.equal(averagePrecisionAtK([[0]], 1, 10), 1) + // One relevant at rank 2: P@1 = 0, P@2 = 1/2, so AP = 1/2. + assert.equal(averagePrecisionAtK([[], [0]], 1, 10), 0.5) + // Two relevant at ranks 1 and 2: (1 + 2/2) / 2 = 1. + assert.equal(averagePrecisionAtK([[0], [1]], 2, 10), 1) + // Two relevant, but the second is only found at rank 4: (1 + 2/4) / 2 = 0.75. + assert.equal(averagePrecisionAtK([[0], [], [], [1]], 2, 10), 0.75) +}) + +test('average precision counts a repeated match once and never exceeds 1', () => { + assert.equal(averagePrecisionAtK([[0], [0], [0]], 1, 10), 1) + assert.equal(averagePrecisionAtK([[], []], 0, 10), 0) + assert.equal(averagePrecisionAtK([], 3, 10), 0) +}) + test('mean and percentile handle the empty and single cases', () => { assert.equal(mean([]), 0) assert.equal(mean([1, 2, 3]), 2) diff --git a/test/evalSplit.test.ts b/test/evalSplit.test.ts new file mode 100644 index 0000000..a57e7dc --- /dev/null +++ b/test/evalSplit.test.ts @@ -0,0 +1,106 @@ +import { test } from 'node:test' +import assert from 'node:assert/strict' +import { + assertQuestionShape, + parseSplitAssignment, + selectSplit, + type EvalQuestion +} from '../src/main/eval/types.ts' + +/** + * The eval split (#192). Parameters that get swept have to be chosen on questions that + * did not take part in the choice. + * + * It used to be a hash of the question id. That is reproducible and, because it is computed + * per id, *stable* when a question is added — but it cannot express the experimental + * design: it does not stratify a small corpus, so a rare query type can end up entirely on + * one side without anyone choosing that, and a new question is assigned silently rather + * than deliberately. The split is now an explicit committed manifest. + */ + +const question = (id: string): EvalQuestion => ({ + id, + question: id, + relevant: [{ document: 'a.md', page: null, block: 0 }] +}) + +const assignment = { q1: 'validation', q2: 'test', q3: 'test' } as const + +test('the manifest maps ids to one of the two sides', () => { + assert.deepEqual(parseSplitAssignment({ q1: 'validation', q2: 'test' }, 'splits.json'), { + q1: 'validation', + q2: 'test' + }) +}) + +test('a manifest that is not an object is rejected, naming the file', () => { + for (const bad of [null, undefined, [], 'validation', 3]) { + assert.throws( + () => parseSplitAssignment(bad, 'eval/splits.json'), + /eval\/splits\.json must be a JSON object/ + ) + } +}) + +test('an unknown side is rejected and names the question', () => { + assert.throws( + () => parseSplitAssignment({ q7: 'train' }, 'eval/splits.json'), + /question q7 is assigned "train"/ + ) +}) + +test('`all` is the whole set and needs no manifest', () => { + const questions = [question('q1'), question('q2')] + assert.deepEqual( + selectSplit(questions, 'all', {}).map((q) => q.id), + ['q1', 'q2'] + ) +}) + +test('the two sides partition the questions and keep their order', () => { + const questions = [question('q1'), question('q2'), question('q3')] + const validation = selectSplit(questions, 'validation', assignment) + const testSide = selectSplit(questions, 'test', assignment) + + assert.deepEqual(validation.map((q) => q.id), ['q1']) + assert.deepEqual(testSide.map((q) => q.id), ['q2', 'q3']) + assert.equal(validation.length + testSide.length, questions.length) +}) + +/** + * A new question must be assigned deliberately. Defaulting it into `test` would leak it + * into the reporting side, which is the side a choice must not be fitted to. + */ +test('a question with no manifest entry is refused, not defaulted', () => { + assert.throws( + () => selectSplit([question('q9')], 'validation', assignment), + /question q9 has no entry in the split manifest/ + ) +}) + +/** + * Reversing these two is silent: an answerable question with no ground truth reads as a + * permanent miss, and an unanswerable one with ground truth reads as a normal hit. + */ +test('the dataset contract ties answerability to having ground truth', () => { + const answerable = question('q1') + const unanswerable: EvalQuestion = { id: 'q2', question: 'q2', answerable: false, relevant: [] } + + assert.doesNotThrow(() => assertQuestionShape(answerable)) + assert.doesNotThrow(() => assertQuestionShape(unanswerable)) + + assert.throws( + () => assertQuestionShape({ ...answerable, relevant: [] }), + /question q1 is answerable but has no ground truth/ + ) + assert.throws( + () => assertQuestionShape({ ...unanswerable, relevant: [answerable.relevant[0]] }), + /question q2 is unanswerable but carries 1 ground-truth location/ + ) +}) + +test('omitting `answerable` means answerable', () => { + // The pre-#192 dataset has no `answerable` field at all, and every one of those + // questions is answerable; the default has to preserve that. + assert.doesNotThrow(() => assertQuestionShape(question('q1'))) +}) diff --git a/test/retrievalContract.test.ts b/test/retrievalContract.test.ts index 9d32292..a662346 100644 --- a/test/retrievalContract.test.ts +++ b/test/retrievalContract.test.ts @@ -2,7 +2,9 @@ import { test } from 'node:test' import assert from 'node:assert/strict' import { buildRetrievalTrace } from '../src/main/services/retrieval/trace.ts' import { + denseChannelThreshold, DEFAULT_CANDIDATE_K, + DEFAULT_DENSE_THRESHOLD, effectiveCandidateK } from '../src/main/services/retrieval/types.ts' @@ -20,7 +22,7 @@ test('a trace carries the effective search parameters and no empty fields', () = strategy: 'dense', candidateK: 20, topK: 5, - threshold: 0.5, + denseThreshold: 0.5, durationMs: 12.5 }) @@ -29,16 +31,40 @@ test('a trace carries the effective search parameters and no empty fields', () = scope: {}, candidateK: 20, topK: 5, - threshold: 0.5, + denseThreshold: 0.5, durationMs: 12.5 }) - // An unset threshold means "no threshold", not "threshold 0". + // An absent dense threshold means "no dense leg ran", not "threshold 0". assert.equal( - 'threshold' in buildRetrievalTrace({ strategy: 'dense', candidateK: 20, topK: 5, durationMs: 1 }), + 'denseThreshold' in + buildRetrievalTrace({ strategy: 'sparse', candidateK: 20, topK: 5, durationMs: 1 }), false ) }) +/** + * The bug the #192 review found: `hybrid` runs a dense leg, so it *has* a similarity + * floor. The retriever applied 0.5 to that leg while the trace recorded `undefined`, so + * the snapshot could not reproduce the retrieval it was describing. + */ +test('a strategy traces the threshold its dense leg actually applied', () => { + assert.equal(denseChannelThreshold('dense'), DEFAULT_DENSE_THRESHOLD) + assert.equal(denseChannelThreshold('hybrid'), DEFAULT_DENSE_THRESHOLD) + assert.equal(denseChannelThreshold('hybrid', 0.3), 0.3) + assert.equal(denseChannelThreshold('dense', 0), 0) + // Only a strategy with no dense leg has no dense threshold. + assert.equal(denseChannelThreshold('sparse', 0.3), undefined) + + const hybrid = buildRetrievalTrace({ + strategy: 'hybrid', + candidateK: 20, + topK: 3, + denseThreshold: denseChannelThreshold('hybrid'), + durationMs: 4 + }) + assert.equal(hybrid.denseThreshold, DEFAULT_DENSE_THRESHOLD) +}) + test('the trace keeps the first-stage width and the final count apart', () => { const trace = buildRetrievalTrace({ strategy: 'hybrid',