diff --git a/docs/eval/baseline-v1.6.json b/docs/eval/baseline-v1.6.json index 4c41432..e087046 100644 --- a/docs/eval/baseline-v1.6.json +++ b/docs/eval/baseline-v1.6.json @@ -110,9 +110,10 @@ ], "unanswerable": { "questions": 6, - "noResultCount": 0, - "noResultRate": 0, - "meanRetrieved": 19 + "abstentionCount": 0, + "retrievalAbstentionRate": 0, + "meanCandidatesRetrieved": 19, + "meanContextPassages": 3 }, "perQuestion": [ { diff --git a/docs/eval/baseline-v1.6.md b/docs/eval/baseline-v1.6.md index 2fcaf5d..1484d95 100644 --- a/docs/eval/baseline-v1.6.md +++ b/docs/eval/baseline-v1.6.md @@ -45,20 +45,30 @@ The type comes from `type` in `questions.jsonl`; untagged questions report as ### Unanswerable questions -These carry no ground truth, so the correct outcome is that retrieval finds nothing. They +These carry no ground truth, so the correct outcome is that retrieval returns nothing. They are excluded from every metric above — a missing ground truth is not a miss — and reported -here instead. A higher **no-results** rate is better on this row, which is the opposite of -how it reads everywhere else, and `mean passages retrieved` is how much irrelevant context -was pulled in anyway. This is the row a threshold decision should move. +here instead. + +**This measures retrieval-level abstention, not the model refusing.** No generator runs in +this harness, so it can show that no candidate passed the threshold; it cannot show that the +final answer would say "not in your sources". A true system refusal rate needs a +generator eval. + +A higher abstention rate is better on this row, the opposite of how every other row reads. +The two sizes are kept apart on purpose: **candidates passing the threshold** can be as high +as `candidateK` (the harness fetches that many to compute `Recall@10`), while **passages in +the context window** is what a user's prompt would actually receive. A large first number +with a small second one means the threshold filters nothing and the window is all noise. | Metric | Value | | --- | --- | | Unanswerable questions | 6 | -| Returned no results | 0.0000 (0/6) | -| Mean passages retrieved | 19.00 | +| Retrieval abstained | 0.0000 (0/6) | +| Mean candidates passing the threshold | 19.00 | +| Mean passages in the context window | 3.00 | -Timing is informational only and is **not** frozen: indexing 1552 ms, query -p50 11.82 ms, p95 14.45 ms on the +Timing is informational only and is **not** frozen: indexing 1583 ms, query +p50 13.38 ms, p95 16.66 ms on the machine that produced this file. Timing and index size depend on hardware and on the corpus, so they must never be the reason two runs differ. diff --git a/docs/eval/retrieval-v1.6.json b/docs/eval/retrieval-v1.6.json index cdfee1c..62b9139 100644 --- a/docs/eval/retrieval-v1.6.json +++ b/docs/eval/retrieval-v1.6.json @@ -1,66 +1,93 @@ { - "baseline": "dense", - "chunking": "1000/100", - "strategies": [ + "baseline": "v1.6", + "split": { + "selects": "validation", + "reports": "test", + "manifest": "eval/splits.json" + }, + "decision": { + "primary": "recallAt5", + "saturated": [], + "winner": null + }, + "validation": [ { "id": "dense", "label": "dense (vector)", + "split": "validation", + "questions": 16, + "answerableCount": 13, "chunking": "1000/100", "chunkCount": 19, - "split": "all", - "questions": 44, - "answerableCount": 38, - "recallAt1": 0.657895, - "recallAt5": 0.921053, + "recallAt1": 0.576923, + "recallAt5": 0.923077, "recallAt10": 1, - "mrr": 0.800909, - "ndcgAt10": 0.84764, - "hitRateAt5": 0.921053, - "mapAt10": 0.796523, - "contextPrecision": 0.324561, - "contextRecall": 0.921053, - "indexingMs": 1532, - "latencyP95Ms": 14.38 + "mrr": 0.778846, + "ndcgAt10": 0.827608, + "hitRateAt5": 0.923077, + "mapAt10": 0.766026, + "contextPrecision": 0.333333, + "contextRecall": 0.923077, + "latencyP95Ms": 16.02 }, { "id": "sparse", "label": "sparse (BM25)", + "split": "validation", + "questions": 16, + "answerableCount": 13, "chunking": "1000/100", "chunkCount": 19, - "split": "all", - "questions": 44, - "answerableCount": 38, - "recallAt1": 0.578947, - "recallAt5": 0.684211, - "recallAt10": 0.684211, - "mrr": 0.635965, - "ndcgAt10": 0.64607, - "hitRateAt5": 0.684211, - "mapAt10": 0.631579, - "contextPrecision": 0.245614, - "contextRecall": 0.684211, - "indexingMs": 1529, - "latencyP95Ms": 1.85 + "recallAt1": 0.576923, + "recallAt5": 0.615385, + "recallAt10": 0.615385, + "mrr": 0.615385, + "ndcgAt10": 0.609209, + "hitRateAt5": 0.615385, + "mapAt10": 0.602564, + "contextPrecision": 0.230769, + "contextRecall": 0.615385, + "latencyP95Ms": 2.96 }, { "id": "hybrid", "label": "hybrid (RRF of dense + BM25)", + "split": "validation", + "questions": 16, + "answerableCount": 13, "chunking": "1000/100", "chunkCount": 19, - "split": "all", - "questions": 44, - "answerableCount": 38, - "recallAt1": 0.684211, - "recallAt5": 0.921053, + "recallAt1": 0.576923, + "recallAt5": 0.923077, "recallAt10": 1, - "mrr": 0.814066, - "ndcgAt10": 0.857352, - "hitRateAt5": 0.921053, - "mapAt10": 0.80968, - "contextPrecision": 0.324561, - "contextRecall": 0.921053, - "indexingMs": 1579, - "latencyP95Ms": 17.04 + "mrr": 0.778846, + "ndcgAt10": 0.827608, + "hitRateAt5": 0.923077, + "mapAt10": 0.766026, + "contextPrecision": 0.333333, + "contextRecall": 0.923077, + "latencyP95Ms": 19.47 + } + ], + "test": [ + { + "id": "dense", + "label": "dense (vector)", + "split": "test", + "questions": 28, + "answerableCount": 25, + "chunking": "1000/100", + "chunkCount": 19, + "recallAt1": 0.7, + "recallAt5": 0.92, + "recallAt10": 1, + "mrr": 0.812381, + "ndcgAt10": 0.858056, + "hitRateAt5": 0.92, + "mapAt10": 0.812381, + "contextPrecision": 0.32, + "contextRecall": 0.92, + "latencyP95Ms": 14.49 } ] } diff --git a/docs/eval/retrieval-v1.6.md b/docs/eval/retrieval-v1.6.md index c00fe52..9d5009a 100644 --- a/docs/eval/retrieval-v1.6.md +++ b/docs/eval/retrieval-v1.6.md @@ -4,15 +4,27 @@ Generated by `node scripts/eval-retrieval.mjs`. Numbers are harness output; do n ## What was measured -Every strategy runs the real RAG eval harness against the same corpus and the same questions -as `baseline-v1.6.json` (split `all`, 44 questions of which -38 are answerable), with chunking held fixed at 1000/100. Only the retrieval strategy changes. +Each strategy runs the real harness on the **validation** split of +`eval/splits.json`, with chunking held fixed at 1000/100. -| Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Context P | Query p95 | -| --- | --- | --- | --- | --- | --- | --- | --- | -| dense (vector) | 0.6579 | 0.9211 | 0.8009 | 0.8476 | 0.7965 | 0.3246 | 14.38 ms | -| sparse (BM25) | 0.5789 | 0.6842 | 0.6360 | 0.6461 | 0.6316 | 0.2456 | 1.85 ms | -| hybrid (RRF of dense + BM25) | 0.6842 | 0.9211 | 0.8141 | 0.8574 | 0.8097 | 0.3246 | 17.04 ms | +The strategy is then chosen **there**, and only the chosen one (plus the shipped default) +is re-run on **test**. The choice never sees `test`; `test` only reports. The previous +version decided on `split = all`, which scored the choice on the questions it was fitted +to. + +## Validation — this is where the choice happens + +| Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Context P | n | p95 | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | +| dense (vector) | 0.5769 | 0.9231 | 0.7788 | 0.8276 | 0.7660 | 0.3333 | 16 | 16.02 ms | +| sparse (BM25) | 0.5769 | 0.6154 | 0.6154 | 0.6092 | 0.6026 | 0.2308 | 16 | 2.96 ms | +| hybrid (RRF of dense + BM25) | 0.5769 | 0.9231 | 0.7788 | 0.8276 | 0.7660 | 0.3333 | 16 | 19.47 ms | + +## Test — reported, not selected + +| Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Context P | n | p95 | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | +| dense (vector) | 0.7000 | 0.9200 | 0.8124 | 0.8581 | 0.8124 | 0.3200 | 28 | 14.49 ms | ## Not evaluated @@ -21,11 +33,6 @@ model is not available offline and inventing its numbers would defeat the point the harness. It stays open until a model can be pinned the way the embedding model is. -## Saturation - -No metric in the rule is saturated on this corpus. The deciding metric on this corpus is `recallAt5`. A saturated metric is still reported, because "this corpus cannot move it" is itself -information; it is just not allowed to decide the comparison. - ## Adoption rule > Adopt a strategy when it improves the **first metric with headroom** — in the order @@ -33,12 +40,16 @@ information; it is just not allowed to decide the comparison. > already at its maximum has no headroom and cannot decide anything; a rule that depends > on one is unsatisfiable, not strict (#192 child 10). > -> A change that trades a large latency increase for a marginal quality gain is a product -> decision, not an automatic win. +> The rule is applied on `validation`. A change that trades a large latency increase for a +> marginal quality gain is a product decision, not an automatic win. ## Outcome -No strategy cleared the rule. The deciding metric was `recallAt5` (dense 0.9211); the strategies either failed to improve it or regressed another metric. **Dense stays the default.** A negative result is the point of the experiment: it is the measurement that says the extra machinery is not worth its cost on this corpus, not a failure to deliver. No metric in the rule is saturated on this corpus. +**No strategy cleared the rule on validation**, so there is no adoption candidate and +`test` reports the shipped strategy only. No metric in the rule is saturated on the validation split. + +A negative result is the point of the experiment: it is the measurement that says the extra +machinery is not worth its cost on this corpus, not a failure to deliver. ## Reproduce diff --git a/docs/eval/sweep-v1.6.json b/docs/eval/sweep-v1.6.json index d027f58..985549e 100644 --- a/docs/eval/sweep-v1.6.json +++ b/docs/eval/sweep-v1.6.json @@ -13,10 +13,11 @@ "noResultRate": 0, "meanContextChars": 1976.1315789473683, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 5, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 5, + "unanswerableContextPassages": 3, "chunkCount": 19, - "latencyP95Ms": 13.57 + "latencyP95Ms": 14.04 }, { "strategy": "dense", @@ -30,10 +31,11 @@ "noResultRate": 0, "meanContextChars": 3188.1315789473683, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 5, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 5, + "unanswerableContextPassages": 5, "chunkCount": 19, - "latencyP95Ms": 14.14 + "latencyP95Ms": 13.63 }, { "strategy": "dense", @@ -47,10 +49,11 @@ "noResultRate": 0, "meanContextChars": 1976.1315789473683, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 10, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 10, + "unanswerableContextPassages": 3, "chunkCount": 19, - "latencyP95Ms": 13.3 + "latencyP95Ms": 13.84 }, { "strategy": "dense", @@ -64,10 +67,11 @@ "noResultRate": 0, "meanContextChars": 3188.1315789473683, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 10, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 10, + "unanswerableContextPassages": 5, "chunkCount": 19, - "latencyP95Ms": 12.61 + "latencyP95Ms": 15.21 }, { "strategy": "dense", @@ -81,10 +85,11 @@ "noResultRate": 0, "meanContextChars": 5151.289473684211, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 10, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 10, + "unanswerableContextPassages": 8, "chunkCount": 19, - "latencyP95Ms": 13.99 + "latencyP95Ms": 16.35 }, { "strategy": "dense", @@ -98,10 +103,11 @@ "noResultRate": 0, "meanContextChars": 1976.1315789473683, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 19, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 19, + "unanswerableContextPassages": 3, "chunkCount": 19, - "latencyP95Ms": 13.61 + "latencyP95Ms": 16.45 }, { "strategy": "dense", @@ -115,10 +121,11 @@ "noResultRate": 0, "meanContextChars": 3188.1315789473683, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 19, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 19, + "unanswerableContextPassages": 5, "chunkCount": 19, - "latencyP95Ms": 14.62 + "latencyP95Ms": 17.91 }, { "strategy": "dense", @@ -132,10 +139,11 @@ "noResultRate": 0, "meanContextChars": 5151.289473684211, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 19, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 19, + "unanswerableContextPassages": 8, "chunkCount": 19, - "latencyP95Ms": 15.2 + "latencyP95Ms": 14.13 }, { "strategy": "dense", @@ -149,10 +157,11 @@ "noResultRate": 0, "meanContextChars": 1976.1315789473683, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 19, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 19, + "unanswerableContextPassages": 3, "chunkCount": 19, - "latencyP95Ms": 13.66 + "latencyP95Ms": 13.35 }, { "strategy": "dense", @@ -166,10 +175,11 @@ "noResultRate": 0, "meanContextChars": 3188.1315789473683, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 19, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 19, + "unanswerableContextPassages": 5, "chunkCount": 19, - "latencyP95Ms": 14.4 + "latencyP95Ms": 12.73 }, { "strategy": "dense", @@ -183,10 +193,11 @@ "noResultRate": 0, "meanContextChars": 5151.289473684211, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 19, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 19, + "unanswerableContextPassages": 8, "chunkCount": 19, - "latencyP95Ms": 14.39 + "latencyP95Ms": 15.12 }, { "strategy": "hybrid", @@ -200,10 +211,11 @@ "noResultRate": 0, "meanContextChars": 1948.657894736842, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 5, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 5, + "unanswerableContextPassages": 3, "chunkCount": 19, - "latencyP95Ms": 16.98 + "latencyP95Ms": 14.01 }, { "strategy": "hybrid", @@ -217,10 +229,11 @@ "noResultRate": 0, "meanContextChars": 3215.5789473684213, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 5, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 5, + "unanswerableContextPassages": 5, "chunkCount": 19, - "latencyP95Ms": 16.11 + "latencyP95Ms": 15.81 }, { "strategy": "hybrid", @@ -234,10 +247,11 @@ "noResultRate": 0, "meanContextChars": 1963.342105263158, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 10, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 10, + "unanswerableContextPassages": 3, "chunkCount": 19, - "latencyP95Ms": 16.46 + "latencyP95Ms": 15.74 }, { "strategy": "hybrid", @@ -249,12 +263,13 @@ "contextPrecision": 0.194737, "contextRecall": 0.921053, "noResultRate": 0, - "meanContextChars": 3332.9473684210525, + "meanContextChars": 3345.0789473684213, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 10, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 10, + "unanswerableContextPassages": 5, "chunkCount": 19, - "latencyP95Ms": 14.83 + "latencyP95Ms": 16.3 }, { "strategy": "hybrid", @@ -268,10 +283,11 @@ "noResultRate": 0, "meanContextChars": 5319.868421052632, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 10, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 10, + "unanswerableContextPassages": 8, "chunkCount": 19, - "latencyP95Ms": 17.38 + "latencyP95Ms": 16.95 }, { "strategy": "hybrid", @@ -285,10 +301,11 @@ "noResultRate": 0, "meanContextChars": 1958.7631578947369, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 19, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 19, + "unanswerableContextPassages": 3, "chunkCount": 19, - "latencyP95Ms": 20.14 + "latencyP95Ms": 14.32 }, { "strategy": "hybrid", @@ -302,10 +319,11 @@ "noResultRate": 0, "meanContextChars": 3282.1052631578946, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 19, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 19, + "unanswerableContextPassages": 5, "chunkCount": 19, - "latencyP95Ms": 14.23 + "latencyP95Ms": 14.82 }, { "strategy": "hybrid", @@ -319,10 +337,11 @@ "noResultRate": 0, "meanContextChars": 5357.315789473684, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 19, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 19, + "unanswerableContextPassages": 8, "chunkCount": 19, - "latencyP95Ms": 14.06 + "latencyP95Ms": 13.74 }, { "strategy": "hybrid", @@ -336,10 +355,11 @@ "noResultRate": 0, "meanContextChars": 1958.7631578947369, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 19, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 19, + "unanswerableContextPassages": 3, "chunkCount": 19, - "latencyP95Ms": 15.39 + "latencyP95Ms": 18.46 }, { "strategy": "hybrid", @@ -353,10 +373,11 @@ "noResultRate": 0, "meanContextChars": 3282.1052631578946, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 19, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 19, + "unanswerableContextPassages": 5, "chunkCount": 19, - "latencyP95Ms": 18.4 + "latencyP95Ms": 16.37 }, { "strategy": "hybrid", @@ -370,10 +391,11 @@ "noResultRate": 0, "meanContextChars": 5357.315789473684, "unanswerableQuestions": 6, - "unanswerableNoResultRate": 0, - "unanswerableMeanRetrieved": 19, + "unanswerableAbstentionRate": 0, + "unanswerableCandidates": 19, + "unanswerableContextPassages": 8, "chunkCount": 19, - "latencyP95Ms": 18.41 + "latencyP95Ms": 20.86 } ] } diff --git a/docs/eval/sweep-v1.6.md b/docs/eval/sweep-v1.6.md index b7e76cc..a3c6dd7 100644 --- a/docs/eval/sweep-v1.6.md +++ b/docs/eval/sweep-v1.6.md @@ -10,30 +10,30 @@ Each row differs from its neighbour in one parameter. 2 further cell(s) were **skipped** because `contextK > candidateK`; see below. -| Strategy | candidateK | contextK | Recall@5 | nDCG@10 | MAP@10 | Context P | Context R | No-result | Context chars | Unans. no-result | Unans. retrieved | Index | p95 | -| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | -| dense | 5 | 3 | 0.9211 | 0.8212 | 0.7851 | 0.3246 | 0.9211 | 0.0000 | 1976 | 0.0000 | 5.0 | 19 | 13.57 ms | -| dense | 5 | 5 | 0.9211 | 0.8212 | 0.7851 | 0.1947 | 0.9211 | 0.0000 | 3188 | 0.0000 | 5.0 | 19 | 14.14 ms | -| dense | 10 | 3 | 0.9211 | 0.8476 | 0.7965 | 0.3246 | 0.9211 | 0.0000 | 1976 | 0.0000 | 10.0 | 19 | 13.30 ms | -| dense | 10 | 5 | 0.9211 | 0.8476 | 0.7965 | 0.1947 | 0.9211 | 0.0000 | 3188 | 0.0000 | 10.0 | 19 | 12.61 ms | -| dense | 10 | 8 | 0.9211 | 0.8476 | 0.7965 | 0.1316 | 1.0000 | 0.0000 | 5151 | 0.0000 | 10.0 | 19 | 13.99 ms | -| dense | 20 | 3 | 0.9211 | 0.8476 | 0.7965 | 0.3246 | 0.9211 | 0.0000 | 1976 | 0.0000 | 19.0 | 19 | 13.61 ms | -| dense | 20 | 5 | 0.9211 | 0.8476 | 0.7965 | 0.1947 | 0.9211 | 0.0000 | 3188 | 0.0000 | 19.0 | 19 | 14.62 ms | -| dense | 20 | 8 | 0.9211 | 0.8476 | 0.7965 | 0.1316 | 1.0000 | 0.0000 | 5151 | 0.0000 | 19.0 | 19 | 15.20 ms | -| dense | 40 | 3 | 0.9211 | 0.8476 | 0.7965 | 0.3246 | 0.9211 | 0.0000 | 1976 | 0.0000 | 19.0 | 19 | 13.66 ms | -| dense | 40 | 5 | 0.9211 | 0.8476 | 0.7965 | 0.1947 | 0.9211 | 0.0000 | 3188 | 0.0000 | 19.0 | 19 | 14.40 ms | -| dense | 40 | 8 | 0.9211 | 0.8476 | 0.7965 | 0.1316 | 1.0000 | 0.0000 | 5151 | 0.0000 | 19.0 | 19 | 14.39 ms | -| hybrid | 5 | 3 | 0.9211 | 0.8309 | 0.7982 | 0.3246 | 0.9211 | 0.0000 | 1949 | 0.0000 | 5.0 | 19 | 16.98 ms | -| hybrid | 5 | 5 | 0.9211 | 0.8309 | 0.7982 | 0.1947 | 0.9211 | 0.0000 | 3216 | 0.0000 | 5.0 | 19 | 16.11 ms | -| hybrid | 10 | 3 | 0.9211 | 0.8574 | 0.8097 | 0.3246 | 0.9211 | 0.0000 | 1963 | 0.0000 | 10.0 | 19 | 16.46 ms | -| hybrid | 10 | 5 | 0.9211 | 0.8574 | 0.8097 | 0.1947 | 0.9211 | 0.0000 | 3333 | 0.0000 | 10.0 | 19 | 14.83 ms | -| hybrid | 10 | 8 | 0.9211 | 0.8574 | 0.8097 | 0.1316 | 1.0000 | 0.0000 | 5320 | 0.0000 | 10.0 | 19 | 17.38 ms | -| hybrid | 20 | 3 | 0.9211 | 0.8574 | 0.8097 | 0.3246 | 0.9211 | 0.0000 | 1959 | 0.0000 | 19.0 | 19 | 20.14 ms | -| hybrid | 20 | 5 | 0.9211 | 0.8574 | 0.8097 | 0.1947 | 0.9211 | 0.0000 | 3282 | 0.0000 | 19.0 | 19 | 14.23 ms | -| hybrid | 20 | 8 | 0.9211 | 0.8574 | 0.8097 | 0.1316 | 1.0000 | 0.0000 | 5357 | 0.0000 | 19.0 | 19 | 14.06 ms | -| hybrid | 40 | 3 | 0.9211 | 0.8574 | 0.8097 | 0.3246 | 0.9211 | 0.0000 | 1959 | 0.0000 | 19.0 | 19 | 15.39 ms | -| hybrid | 40 | 5 | 0.9211 | 0.8574 | 0.8097 | 0.1947 | 0.9211 | 0.0000 | 3282 | 0.0000 | 19.0 | 19 | 18.40 ms | -| hybrid | 40 | 8 | 0.9211 | 0.8574 | 0.8097 | 0.1316 | 1.0000 | 0.0000 | 5357 | 0.0000 | 19.0 | 19 | 18.41 ms | +| Strategy | candidateK | contextK | Recall@5 | nDCG@10 | MAP@10 | Context P | Context R | No-result | Context chars | Unans. abstained | Unans. cands | Unans. ctx | Index | p95 | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | +| dense | 5 | 3 | 0.9211 | 0.8212 | 0.7851 | 0.3246 | 0.9211 | 0.0000 | 1976 | 0.0000 | 5.0 | 3.0 | 19 | 14.04 ms | +| dense | 5 | 5 | 0.9211 | 0.8212 | 0.7851 | 0.1947 | 0.9211 | 0.0000 | 3188 | 0.0000 | 5.0 | 5.0 | 19 | 13.63 ms | +| dense | 10 | 3 | 0.9211 | 0.8476 | 0.7965 | 0.3246 | 0.9211 | 0.0000 | 1976 | 0.0000 | 10.0 | 3.0 | 19 | 13.84 ms | +| dense | 10 | 5 | 0.9211 | 0.8476 | 0.7965 | 0.1947 | 0.9211 | 0.0000 | 3188 | 0.0000 | 10.0 | 5.0 | 19 | 15.21 ms | +| dense | 10 | 8 | 0.9211 | 0.8476 | 0.7965 | 0.1316 | 1.0000 | 0.0000 | 5151 | 0.0000 | 10.0 | 8.0 | 19 | 16.35 ms | +| dense | 20 | 3 | 0.9211 | 0.8476 | 0.7965 | 0.3246 | 0.9211 | 0.0000 | 1976 | 0.0000 | 19.0 | 3.0 | 19 | 16.45 ms | +| dense | 20 | 5 | 0.9211 | 0.8476 | 0.7965 | 0.1947 | 0.9211 | 0.0000 | 3188 | 0.0000 | 19.0 | 5.0 | 19 | 17.91 ms | +| dense | 20 | 8 | 0.9211 | 0.8476 | 0.7965 | 0.1316 | 1.0000 | 0.0000 | 5151 | 0.0000 | 19.0 | 8.0 | 19 | 14.13 ms | +| dense | 40 | 3 | 0.9211 | 0.8476 | 0.7965 | 0.3246 | 0.9211 | 0.0000 | 1976 | 0.0000 | 19.0 | 3.0 | 19 | 13.35 ms | +| dense | 40 | 5 | 0.9211 | 0.8476 | 0.7965 | 0.1947 | 0.9211 | 0.0000 | 3188 | 0.0000 | 19.0 | 5.0 | 19 | 12.73 ms | +| dense | 40 | 8 | 0.9211 | 0.8476 | 0.7965 | 0.1316 | 1.0000 | 0.0000 | 5151 | 0.0000 | 19.0 | 8.0 | 19 | 15.12 ms | +| hybrid | 5 | 3 | 0.9211 | 0.8309 | 0.7982 | 0.3246 | 0.9211 | 0.0000 | 1949 | 0.0000 | 5.0 | 3.0 | 19 | 14.01 ms | +| hybrid | 5 | 5 | 0.9211 | 0.8309 | 0.7982 | 0.1947 | 0.9211 | 0.0000 | 3216 | 0.0000 | 5.0 | 5.0 | 19 | 15.81 ms | +| hybrid | 10 | 3 | 0.9211 | 0.8574 | 0.8097 | 0.3246 | 0.9211 | 0.0000 | 1963 | 0.0000 | 10.0 | 3.0 | 19 | 15.74 ms | +| hybrid | 10 | 5 | 0.9211 | 0.8574 | 0.8097 | 0.1947 | 0.9211 | 0.0000 | 3345 | 0.0000 | 10.0 | 5.0 | 19 | 16.30 ms | +| hybrid | 10 | 8 | 0.9211 | 0.8574 | 0.8097 | 0.1316 | 1.0000 | 0.0000 | 5320 | 0.0000 | 10.0 | 8.0 | 19 | 16.95 ms | +| hybrid | 20 | 3 | 0.9211 | 0.8574 | 0.8097 | 0.3246 | 0.9211 | 0.0000 | 1959 | 0.0000 | 19.0 | 3.0 | 19 | 14.32 ms | +| hybrid | 20 | 5 | 0.9211 | 0.8574 | 0.8097 | 0.1947 | 0.9211 | 0.0000 | 3282 | 0.0000 | 19.0 | 5.0 | 19 | 14.82 ms | +| hybrid | 20 | 8 | 0.9211 | 0.8574 | 0.8097 | 0.1316 | 1.0000 | 0.0000 | 5357 | 0.0000 | 19.0 | 8.0 | 19 | 13.74 ms | +| hybrid | 40 | 3 | 0.9211 | 0.8574 | 0.8097 | 0.3246 | 0.9211 | 0.0000 | 1959 | 0.0000 | 19.0 | 3.0 | 19 | 18.46 ms | +| hybrid | 40 | 5 | 0.9211 | 0.8574 | 0.8097 | 0.1947 | 0.9211 | 0.0000 | 3282 | 0.0000 | 19.0 | 5.0 | 19 | 16.37 ms | +| hybrid | 40 | 8 | 0.9211 | 0.8574 | 0.8097 | 0.1316 | 1.0000 | 0.0000 | 5357 | 0.0000 | 19.0 | 8.0 | 19 | 20.86 ms | ## Skipped cells @@ -60,10 +60,14 @@ The harness refuses the same combination at the flag level, so a typo fails loud embedding model, not any generation model's tokenizer. - **No-result** is the share of *answerable* questions whose retrieval returned nothing — a miss, and the lower the better. -- **Unans. no-result / retrieved** are the same idea for the *unanswerable* questions, - where the direction flips: there is no ground truth, so returning nothing is correct and - `retrieved` is how much irrelevant context was pulled in anyway. These two are the - columns a threshold decision should move, and they are kept out of every other column. +- **Unans. abstained / cands / ctx** describe the *unanswerable* questions, where the + direction flips: there is no ground truth, so abstaining is correct. `abstained` is the + share where nothing passed the threshold; `cands` is how many candidates did (up to + `candidateK`, since the harness fetches that many for `Recall@10`); `ctx` is how many + actually reach the context window, i.e. `min(candidates, contextK)`. A high `cands` with + the usual `ctx` means the threshold is filtering nothing and the window is all noise. + These are the columns a threshold decision should move, and they stay out of every other + column. Best nDCG@10 in this grid: `hybrid` candidateK=10, contextK=3 (0.8574). diff --git a/docs/eval/threshold-v1.6.json b/docs/eval/threshold-v1.6.json index 468dc8d..cb995ef 100644 --- a/docs/eval/threshold-v1.6.json +++ b/docs/eval/threshold-v1.6.json @@ -14,9 +14,10 @@ "meanRetrieved": 19, "unanswerable": { "questions": 3, - "noResultCount": 0, - "noResultRate": 0, - "meanRetrieved": 19 + "abstentionCount": 0, + "retrievalAbstentionRate": 0, + "meanCandidatesRetrieved": 19, + "meanContextPassages": 3 }, "metrics": { "recallAt1": 0.576923, @@ -38,9 +39,10 @@ "meanRetrieved": 19, "unanswerable": { "questions": 3, - "noResultCount": 0, - "noResultRate": 0, - "meanRetrieved": 19 + "abstentionCount": 0, + "retrievalAbstentionRate": 0, + "meanCandidatesRetrieved": 19, + "meanContextPassages": 3 }, "metrics": { "recallAt1": 0.7, @@ -65,9 +67,10 @@ "meanRetrieved": 19, "unanswerable": { "questions": 3, - "noResultCount": 0, - "noResultRate": 0, - "meanRetrieved": 19 + "abstentionCount": 0, + "retrievalAbstentionRate": 0, + "meanCandidatesRetrieved": 19, + "meanContextPassages": 3 }, "metrics": { "recallAt1": 0.576923, @@ -89,9 +92,10 @@ "meanRetrieved": 19, "unanswerable": { "questions": 3, - "noResultCount": 0, - "noResultRate": 0, - "meanRetrieved": 19 + "abstentionCount": 0, + "retrievalAbstentionRate": 0, + "meanCandidatesRetrieved": 19, + "meanContextPassages": 3 }, "metrics": { "recallAt1": 0.7, @@ -116,9 +120,10 @@ "meanRetrieved": 19, "unanswerable": { "questions": 3, - "noResultCount": 0, - "noResultRate": 0, - "meanRetrieved": 19 + "abstentionCount": 0, + "retrievalAbstentionRate": 0, + "meanCandidatesRetrieved": 19, + "meanContextPassages": 3 }, "metrics": { "recallAt1": 0.576923, @@ -140,9 +145,10 @@ "meanRetrieved": 19, "unanswerable": { "questions": 3, - "noResultCount": 0, - "noResultRate": 0, - "meanRetrieved": 19 + "abstentionCount": 0, + "retrievalAbstentionRate": 0, + "meanCandidatesRetrieved": 19, + "meanContextPassages": 3 }, "metrics": { "recallAt1": 0.7, @@ -167,9 +173,10 @@ "meanRetrieved": 19, "unanswerable": { "questions": 3, - "noResultCount": 0, - "noResultRate": 0, - "meanRetrieved": 19 + "abstentionCount": 0, + "retrievalAbstentionRate": 0, + "meanCandidatesRetrieved": 19, + "meanContextPassages": 3 }, "metrics": { "recallAt1": 0.576923, @@ -191,9 +198,10 @@ "meanRetrieved": 19, "unanswerable": { "questions": 3, - "noResultCount": 0, - "noResultRate": 0, - "meanRetrieved": 19 + "abstentionCount": 0, + "retrievalAbstentionRate": 0, + "meanCandidatesRetrieved": 19, + "meanContextPassages": 3 }, "metrics": { "recallAt1": 0.7, @@ -218,9 +226,10 @@ "meanRetrieved": 19, "unanswerable": { "questions": 3, - "noResultCount": 0, - "noResultRate": 0, - "meanRetrieved": 19 + "abstentionCount": 0, + "retrievalAbstentionRate": 0, + "meanCandidatesRetrieved": 19, + "meanContextPassages": 3 }, "metrics": { "recallAt1": 0.576923, @@ -242,9 +251,10 @@ "meanRetrieved": 19, "unanswerable": { "questions": 3, - "noResultCount": 0, - "noResultRate": 0, - "meanRetrieved": 19 + "abstentionCount": 0, + "retrievalAbstentionRate": 0, + "meanCandidatesRetrieved": 19, + "meanContextPassages": 3 }, "metrics": { "recallAt1": 0.7, diff --git a/docs/eval/threshold-v1.6.md b/docs/eval/threshold-v1.6.md index 45e1f10..0b08be6 100644 --- a/docs/eval/threshold-v1.6.md +++ b/docs/eval/threshold-v1.6.md @@ -11,30 +11,32 @@ split selects; the **test** split reports. The split is the committed manifest Quality columns cover the answerable questions only; **Unans.** columns cover the unanswerable ones, where returning nothing is the desired outcome and so a *higher* -no-result rate is better. +abstention rate is better. Two sizes are kept apart: **cands** is how many candidates +passed the threshold (up to `candidateK`), **ctx** is how many reach the context window. -| Threshold | n (val) | Recall@5 (val) | nDCG@10 (val) | No-result (val) | Unans. no-result (val) | Unans. retrieved (val) | nDCG@10 (test) | Unans. no-result (test) | -| --- | --- | --- | --- | --- | --- | --- | --- | --- | -| 0 | 13 | 0.9231 | 0.8276 | 0.0000 | 0.0000 | 19.0 | 0.8581 | 0.0000 | -| 0.3 | 13 | 0.9231 | 0.8276 | 0.0000 | 0.0000 | 19.0 | 0.8581 | 0.0000 | -| 0.4 | 13 | 0.9231 | 0.8276 | 0.0000 | 0.0000 | 19.0 | 0.8581 | 0.0000 | -| 0.5 | 13 | 0.9231 | 0.8276 | 0.0000 | 0.0000 | 19.0 | 0.8581 | 0.0000 | -| 0.6 | 13 | 0.9231 | 0.8276 | 0.0000 | 0.0000 | 19.0 | 0.8581 | 0.0000 | +| Threshold | n (val) | Recall@5 (val) | nDCG@10 (val) | No-result (val) | Unans. abstained (val) | Unans. cands (val) | Unans. ctx (val) | nDCG@10 (test) | Unans. abstained (test) | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | +| 0 | 13 | 0.9231 | 0.8276 | 0.0000 | 0.0000 | 19.0 | 3.0 | 0.8581 | 0.0000 | +| 0.3 | 13 | 0.9231 | 0.8276 | 0.0000 | 0.0000 | 19.0 | 3.0 | 0.8581 | 0.0000 | +| 0.4 | 13 | 0.9231 | 0.8276 | 0.0000 | 0.0000 | 19.0 | 3.0 | 0.8581 | 0.0000 | +| 0.5 | 13 | 0.9231 | 0.8276 | 0.0000 | 0.0000 | 19.0 | 3.0 | 0.8581 | 0.0000 | +| 0.6 | 13 | 0.9231 | 0.8276 | 0.0000 | 0.0000 | 19.0 | 3.0 | 0.8581 | 0.0000 | ## Selection rule Hold the answerable quality line — validation nDCG@10 and context recall must not -regress versus `threshold = 0` — then take the threshold that refuses the most +regress versus `threshold = 0` — then take the threshold that abstains on the most unanswerable questions. Tie-break on the lowest threshold. -Raising a threshold is only worth anything if it refuses what the sources do not answer; -the quality gate is there so a refusal gain can never be bought with a retrieval loss. +Raising a threshold is only worth anything if it stops unsupported context before the +prompt; the quality gate is there so an abstention gain can never be bought with a +retrieval loss. ## Outcome -The sweep is **flat**: every threshold from 0 to 0.6 produces the same validation nDCG@10 (0.8276), the same Recall@5 (0.9231) and the same unanswerable refusal rate (0/3). No passage is ever filtered out, so the threshold is **non-binding** on this corpus — E5 does not score these query/chunk pairs below the top of the swept range. +The sweep is **flat**: every threshold from 0 to 0.6 produces the same validation nDCG@10 (0.8276), the same Recall@5 (0.9231) and the same retrieval abstention rate on unanswerable questions (0/3). No candidate is ever filtered out, so the threshold is **non-binding** on this corpus. -**No evidence to change `threshold = 0.5`.** All thresholds hold the line equally; picking one would be arbitrary. The current value can be neither validated nor falsified here, which is a property of the corpus rather than of the threshold. +**No evidence to change `threshold = 0.5`.** All thresholds hold the line equally; picking one would be arbitrary. The current value can be neither validated nor falsified here, which is a property of the corpus rather than of the threshold. Note also what this does *not* establish: abstention is a retrieval-layer statement — whether the model then declines to answer needs a generator eval. ## Caveat on this corpus diff --git a/eval/README.md b/eval/README.md index f01989c..24deb75 100644 --- a/eval/README.md +++ b/eval/README.md @@ -39,11 +39,17 @@ A parameter picked on the same questions it is scored on is a fitted number, not result. `eval/splits.json` is the committed assignment; `--eval-split=validation` selects from it and `test` is the rest. -It is an explicit manifest rather than a hash of the question id. A hash is reproducible -but not *stable*: adding a question moves others between the sides, and a rare query type -can end up entirely on one side without anyone choosing that. With a manifest, a question -with no entry is **refused** rather than defaulted, so a new question is assigned -deliberately instead of leaking into `test`. +It is an explicit manifest rather than a hash of the question id. A hash +(`hash(id) % 3`) is actually *stable* — it is computed per id, so adding a question does +not move the existing ones. What it cannot do is express the experimental design: + +- it does not stratify a small corpus, so a rare type (`multi-hop`, `cross-lingual`) can + end up entirely on one side without anyone choosing that — which is what happened; and +- a newly added question is assigned silently instead of deliberately, and `test` is the + side a choice must not be fitted to. + +With a manifest, a question with no entry is **refused** rather than defaulted, so every +new question is assigned on purpose. `npm run eval:sweep` runs a bounded grid (`strategy × candidateK × contextK`) and writes one dashboard with quality, context precision/recall, prompt size, index size diff --git a/scripts/eval-retrieval.mjs b/scripts/eval-retrieval.mjs index 0ed9a6c..d06d5ad 100644 --- a/scripts/eval-retrieval.mjs +++ b/scripts/eval-retrieval.mjs @@ -1,15 +1,17 @@ #!/usr/bin/env node /** - * Retrieval experiments for #77. + * Retrieval experiments for #77, with held-out strategy adoption (#192). * - * Runs the real RAG eval harness once per retrieval strategy against the frozen - * chunk baseline (`baseline-v1.6.json`, 1000/100), holding chunking fixed, and - * writes the comparison the issue asks for as a delta against dense. + * Runs the real RAG eval harness once per retrieval strategy on the **validation** split, + * decides there with the amended adoption rule (#192 child 10), and then re-runs the + * shipped strategy and the selected one on the **test** split. Selection never sees + * `test`; `test` only reports. * - * The harness does the measuring; this script only orchestrates and tabulates. + * That split is the whole point. The previous version decided on `split = all`, which + * meant the strategy was chosen and scored on the same questions — the same mistake the + * threshold experiment had already been fixed for. * - * The adoption rule is the amended one from #192 child 10: the deciding metric is the - * first metric with headroom, not a metric that the corpus has already maxed out. + * The harness does the measuring; this script only orchestrates and tabulates. * * Usage: * node scripts/eval-retrieval.mjs @@ -37,6 +39,9 @@ const STRATEGIES = [ { id: 'hybrid', label: 'hybrid (RRF of dense + BM25)' } ] +/** The shipped strategy. Everything is reported as a delta against it. */ +const BASELINE_ID = 'dense' + function readArg(prefix, fallback) { const arg = process.argv.find((value) => value.startsWith(prefix)) return arg ? arg.slice(prefix.length) : fallback @@ -52,13 +57,18 @@ if (!existsSync(executable)) { process.exit(1) } -function runStrategy(strategy, outDir) { +// The rule lives in `src/main/eval/adoption.ts` so it can be unit tested; it decides +// whether a shipped default moves. +const { ADOPTION_METRICS, decideAdoption } = await import('../src/main/eval/adoption.ts') + +function runStrategy(strategy, split, outDir) { return new Promise((resolvePromise, reject) => { const args = [ '.', '--eval-harness', '--eval-baseline=v1.6', `--eval-out=${outDir}`, + `--eval-split=${split}`, `--eval-retrieval=${strategy.id}` ] @@ -70,108 +80,128 @@ function runStrategy(strategy, outDir) { env: { ...process.env, ELECTRON_DISABLE_SECURITY_WARNINGS: '1' } }) - let stdout = '' - child.stdout.on('data', (data) => { - stdout += data.toString() - }) - child.stderr.on('data', () => {}) - child.on('error', reject) child.on('exit', (code) => { if (code !== 0) { - reject(new Error(`strategy ${strategy.id} exited with code ${code}`)) + reject(new Error(`${strategy.id} (${split}) exited with code ${code}`)) return } - const metricsLine = /\[eval\] metrics (\{.*\})/.exec(stdout) - if (!metricsLine) { - reject(new Error(`strategy ${strategy.id} printed no metrics line`)) + const reportPath = join(outDir, 'baseline-v1.6.json') + if (!existsSync(reportPath)) { + reject(new Error(`${strategy.id} (${split}) wrote no report`)) return } - try { - resolvePromise(JSON.parse(metricsLine[1])) - } catch (error) { - reject( - new Error(`strategy ${strategy.id} printed an unreadable metrics line: ${error.message}`) - ) - } + const report = JSON.parse(readFileSync(reportPath, 'utf8')) + resolvePromise({ + id: strategy.id, + label: strategy.label, + split, + questions: report.config.questions, + answerableCount: report.byType.reduce((total, entry) => total + entry.questions, 0), + chunking: `${report.config.chunking.chunkSize}/${report.config.chunking.chunkOverlap}`, + chunkCount: report.config.chunkCount, + ...report.metrics, + ...readTiming(join(outDir, 'baseline-v1.6.md')) + }) }) }) } -/** Throughput and p95 are informational and excluded from the deterministic JSON. */ +/** p95 is informational and excluded from the deterministic JSON. */ function readTiming(mdPath) { - if (!existsSync(mdPath)) return { indexingMs: null, latencyP95Ms: null } + if (!existsSync(mdPath)) return { latencyP95Ms: null } const text = readFileSync(mdPath, 'utf8') - const indexing = /indexing (\d+) ms/.exec(text) const p95 = /p95 ([\d.]+) ms/.exec(text) - return { - indexingMs: indexing ? Number(indexing[1]) : null, - latencyP95Ms: p95 ? Number(p95[1]) : null - } + return { latencyP95Ms: p95 ? Number(p95[1]) : null } } +function runInto(workDir, strategy, split) { + const outDir = join(workDir, `${split}-${strategy.id}`) + mkdirSync(outDir, { recursive: true }) + return runStrategy(strategy, split, outDir) +} + +const format4 = (value) => value.toFixed(4) + const workDir = mkdtempSync(join(tmpdir(), 'knownote-retrieval-')) -const results = [] +let validation = [] +let test = [] +let decision = { primary: null, saturated: [], winner: null } try { + // ── Validation: this is the only phase that may choose ──────────────────────── for (const strategy of STRATEGIES) { - const outDir = join(workDir, strategy.id) - mkdirSync(outDir, { recursive: true }) - console.log(`[retrieval] running ${strategy.label}`) - const metrics = await runStrategy(strategy, outDir) - - const report = JSON.parse(readFileSync(join(outDir, 'baseline-v1.6.json'), 'utf8')) - results.push({ - id: strategy.id, - label: strategy.label, - chunking: `${report.config.chunking.chunkSize}/${report.config.chunking.chunkOverlap}`, - chunkCount: report.config.chunkCount, - split: report.config.split, - questions: report.config.questions, - answerableCount: report.byType.reduce((total, entry) => total + entry.questions, 0), - ...metrics, - ...readTiming(join(outDir, 'baseline-v1.6.md')) - }) + console.log(`[retrieval] validation: ${strategy.label}`) + validation.push(await runInto(workDir, strategy, 'validation')) } -} finally { - rmSync(workDir, { recursive: true, force: true }) -} -const baseline = results.find((result) => result.id === 'dense') -if (!baseline) throw new Error('the dense strategy did not run') + const baselineRow = validation.find((row) => row.id === BASELINE_ID) + if (!baselineRow) throw new Error('the dense strategy did not run on validation') -const format4 = (value) => value.toFixed(4) + decision = decideAdoption(baselineRow, validation) -const rows = results.map( - (result) => - `| ${result.label} | ${format4(result.recallAt1)} | ${format4(result.recallAt5)} | ${format4(result.mrr)} | ${format4(result.ndcgAt10)} | ${format4(result.mapAt10)} | ${format4(result.contextPrecision)} | ${result.latencyP95Ms?.toFixed(2)} ms |` -) + // ── Test: reports only. Never lets the choice see these questions. ──────────── + const testIds = [BASELINE_ID] + if (decision.winner && decision.winner.id !== BASELINE_ID) testIds.push(decision.winner.id) -/** - * The rule lives in `src/main/eval/adoption.ts` so it can be unit tested: it decides - * whether a shipped default moves, and testing it by reading the sentence this script - * prints would be a test of the sentence. - * - * Latency stays in the table so a win that costs 5x latency is stated as a trade-off, - * not hidden. - */ -const { ADOPTION_METRICS, decideAdoption } = await import('../src/main/eval/adoption.ts') -const { primary, saturated, winner } = decideAdoption(baseline, results) + for (const id of testIds) { + const strategy = STRATEGIES.find((entry) => entry.id === id) + console.log(`[retrieval] test: ${strategy.label}`) + test.push(await runInto(workDir, strategy, 'test')) + } +} finally { + rmSync(workDir, { recursive: true, force: true }) +} -const saturationNote = saturated.length - ? `Saturated (no headroom, so they cannot decide anything): ${saturated +const winner = decision.winner +const saturationNote = decision.saturated.length + ? `Saturated on the validation split (no headroom, so they cannot decide): ${decision.saturated .map((key) => `\`${key}\``) .join(', ')}.` - : 'No metric in the rule is saturated on this corpus.' - -let outcome -if (winner) { - outcome = `\`${winner.label}\` **clears the rule**: it improves the deciding metric \`${primary}\` (${format4(winner[primary])} vs dense ${format4(baseline[primary])}) and regresses none of ${ADOPTION_METRICS.map((key) => `\`${key}\``).join(', ')}. ${saturationNote}\n\nChanging the shipped default is a separate decision, and this script does not make it — it reports the measurement.` -} else if (primary === null) { - outcome = `Every metric in the rule is already at its maximum on this corpus, so no strategy can clear any of them. **Dense stays the default**; the comparison is inconclusive by construction, not negative.` -} else { - outcome = `No strategy cleared the rule. The deciding metric was \`${primary}\` (dense ${format4(baseline[primary])}); the strategies either failed to improve it or regressed another metric. **Dense stays the default.** A negative result is the point of the experiment: it is the measurement that says the extra machinery is not worth its cost on this corpus, not a failure to deliver. ${saturationNote}` -} + : 'No metric in the rule is saturated on the validation split.' + +const metricColumns = ['recallAt1', 'recallAt5', 'mrr', 'ndcgAt10', 'mapAt10', 'contextPrecision'] +const tableRow = (row) => + `| ${row.label} | ${metricColumns.map((key) => format4(row[key])).join(' | ')} | ${row.questions} | ${ + row.latencyP95Ms?.toFixed(2) ?? '—' + } ms |` + +const validationTable = validation.map(tableRow).join('\n') +const testTable = test.map(tableRow).join('\n') + +/** Per-metric delta of the selected strategy against dense, on the test split. */ +const testDense = test.find((row) => row.id === BASELINE_ID) +const testWinner = winner ? test.find((row) => row.id === winner.id) : null + +const deltaRows = + testWinner && testDense + ? ADOPTION_METRICS.map((key) => { + const delta = testWinner[key] - testDense[key] + const sign = delta > 0 ? '+' : '' + return `| ${key} | ${format4(testDense[key])} | ${format4(testWinner[key])} | ${sign}${format4(delta)} |` + }).join('\n') + : '' + +const outcome = winner + ? `**\`${winner.label}\` clears the rule on validation.** The deciding metric was \`${decision.primary}\` (${format4( + winner[decision.primary] + )} vs dense ${format4( + validation.find((row) => row.id === BASELINE_ID)[decision.primary] + )}), and it regressed none of ${ADOPTION_METRICS.map((key) => `\`${key}\``).join(', ')}. ${saturationNote} + +That decision was made on questions in \`test\` **not** seeing. What follows is the held-out +result, and it is the only number that should inform shipping it: + +| Metric | dense (test) | selected (test) | delta | +| --- | --- | --- | --- | +${deltaRows} + +Shipping a new default is a product decision this script does not make. It measures.` + : `**No strategy cleared the rule on validation**, so there is no adoption candidate and +\`test\` reports the shipped strategy only. ${saturationNote} + +A negative result is the point of the experiment: it is the measurement that says the extra +machinery is not worth its cost on this corpus, not a failure to deliver.` const markdown = `# Retrieval experiments — v1.6 (#77, #192) @@ -179,13 +209,25 @@ Generated by \`node scripts/eval-retrieval.mjs\`. Numbers are harness output; do ## What was measured -Every strategy runs the real RAG eval harness against the same corpus and the same questions -as \`baseline-v1.6.json\` (split \`${baseline.split}\`, ${baseline.questions} questions of which -${baseline.answerableCount} are answerable), with chunking held fixed at ${baseline.chunking}. Only the retrieval strategy changes. +Each strategy runs the real harness on the **validation** split of +\`eval/splits.json\`, with chunking held fixed at ${validation[0]?.chunking ?? '1000/100'}. + +The strategy is then chosen **there**, and only the chosen one (plus the shipped default) +is re-run on **test**. The choice never sees \`test\`; \`test\` only reports. The previous +version decided on \`split = all\`, which scored the choice on the questions it was fitted +to. + +## Validation — this is where the choice happens + +| Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Context P | n | p95 | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | +${validationTable} + +## Test — reported, not selected -| Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Context P | Query p95 | -| --- | --- | --- | --- | --- | --- | --- | --- | -${rows.join('\n')} +| Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Context P | n | p95 | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | +${testTable} ## Not evaluated @@ -194,13 +236,6 @@ model is not available offline and inventing its numbers would defeat the point the harness. It stays open until a model can be pinned the way the embedding model is. -## Saturation - -${saturationNote} The deciding metric on this corpus is ${ - primary === null ? 'none — every metric in the rule is already maxed out' : `\`${primary}\`` -}. A saturated metric is still reported, because "this corpus cannot move it" is itself -information; it is just not allowed to decide the comparison. - ## Adoption rule > Adopt a strategy when it improves the **first metric with headroom** — in the order @@ -208,8 +243,8 @@ information; it is just not allowed to decide the comparison. > already at its maximum has no headroom and cannot decide anything; a rule that depends > on one is unsatisfiable, not strict (#192 child 10). > -> A change that trades a large latency increase for a marginal quality gain is a product -> decision, not an automatic win. +> The rule is applied on \`validation\`. A change that trades a large latency increase for a +> marginal quality gain is a product decision, not an automatic win. ## Outcome @@ -226,11 +261,21 @@ npm run eval:retrieval # offline; runs every strategy and rewrites this file mkdirSync(resolve(OUT_MD, '..'), { recursive: true }) writeFileSync( OUT_JSON, - `${JSON.stringify({ baseline: baseline.id, chunking: baseline.chunking, strategies: results }, null, 2)}\n` + `${JSON.stringify( + { + baseline: 'v1.6', + split: { selects: 'validation', reports: 'test', manifest: 'eval/splits.json' }, + decision: { primary: decision.primary, saturated: decision.saturated, winner: winner?.id ?? null }, + validation, + test + }, + null, + 2 + )}\n` ) writeFileSync(OUT_MD, markdown) -console.log(`[retrieval] wrote ${OUT_JSON} and ${OUT_MD}`) console.log( - `[retrieval] ${winner ? `best clearing strategy: ${winner.label}` : 'no strategy cleared the rule; keep dense'}` + `[retrieval] ${winner ? `validation selected: ${winner.label}` : 'validation selected nothing; test reports dense'}` ) +console.log(`[retrieval] wrote ${OUT_JSON} and ${OUT_MD}`) diff --git a/scripts/eval-sweep.mjs b/scripts/eval-sweep.mjs index 45a3eb0..207f9d9 100644 --- a/scripts/eval-sweep.mjs +++ b/scripts/eval-sweep.mjs @@ -159,8 +159,9 @@ try { noResultRate: answerable.length === 0 ? 0 : noResult / answerable.length, meanContextChars: mean(answerable.map((q) => q.contextChars)), unanswerableQuestions: report.unanswerable.questions, - unanswerableNoResultRate: report.unanswerable.noResultRate, - unanswerableMeanRetrieved: report.unanswerable.meanRetrieved, + unanswerableAbstentionRate: report.unanswerable.retrievalAbstentionRate, + unanswerableCandidates: report.unanswerable.meanCandidatesRetrieved, + unanswerableContextPassages: report.unanswerable.meanContextPassages, chunkCount: report.config.chunkCount, ...readTiming(join(outDir, 'baseline-v1.6.md')) }) @@ -177,8 +178,9 @@ const tableRows = rows `| ${row.strategy} | ${row.candidateK} | ${row.contextK} | ${format4(row.recallAt5)} | ` + `${format4(row.ndcgAt10)} | ${format4(row.mapAt10)} | ${format4(row.contextPrecision)} | ` + `${format4(row.contextRecall)} | ${format4(row.noResultRate)} | ` + - `${Math.round(row.meanContextChars)} | ${format4(row.unanswerableNoResultRate)} | ` + - `${row.unanswerableMeanRetrieved.toFixed(1)} | ${row.chunkCount} | ${row.latencyP95Ms?.toFixed(2) ?? '—'} ms |` + `${Math.round(row.meanContextChars)} | ${format4(row.unanswerableAbstentionRate)} | ` + + `${row.unanswerableCandidates.toFixed(1)} | ${row.unanswerableContextPassages.toFixed(1)} | ` + + `${row.chunkCount} | ${row.latencyP95Ms?.toFixed(2) ?? '—'} ms |` ) .join('\n') @@ -211,8 +213,8 @@ The real harness, the same corpus, chunking held fixed, over ${STRATEGIES.join(' / ')} × candidateK {${CANDIDATE_KS.join(', ')}} × contextK {${CONTEXT_KS.join(', ')}} — ${rows.length} runs. Each row differs from its neighbour in one parameter.${skipped.length > 0 ? `\n\n${skipped.length} further cell(s) were **skipped** because \`contextK > candidateK\`; see below.` : ''} -| Strategy | candidateK | contextK | Recall@5 | nDCG@10 | MAP@10 | Context P | Context R | No-result | Context chars | Unans. no-result | Unans. retrieved | Index | p95 | -| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | +| Strategy | candidateK | contextK | Recall@5 | nDCG@10 | MAP@10 | Context P | Context R | No-result | Context chars | Unans. abstained | Unans. cands | Unans. ctx | Index | p95 | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | ${tableRows} ${skippedNote} @@ -228,10 +230,14 @@ ${skippedNote} embedding model, not any generation model's tokenizer. - **No-result** is the share of *answerable* questions whose retrieval returned nothing — a miss, and the lower the better. -- **Unans. no-result / retrieved** are the same idea for the *unanswerable* questions, - where the direction flips: there is no ground truth, so returning nothing is correct and - \`retrieved\` is how much irrelevant context was pulled in anyway. These two are the - columns a threshold decision should move, and they are kept out of every other column. +- **Unans. abstained / cands / ctx** describe the *unanswerable* questions, where the + direction flips: there is no ground truth, so abstaining is correct. \`abstained\` is the + share where nothing passed the threshold; \`cands\` is how many candidates did (up to + \`candidateK\`, since the harness fetches that many for \`Recall@10\`); \`ctx\` is how many + actually reach the context window, i.e. \`min(candidates, contextK)\`. A high \`cands\` with + the usual \`ctx\` means the threshold is filtering nothing and the window is all noise. + These are the columns a threshold decision should move, and they stay out of every other + column. Best nDCG@10 in this grid: \`${bestNdcg.strategy}\` candidateK=${bestNdcg.candidateK}, contextK=${bestNdcg.contextK} (${format4(bestNdcg.ndcgAt10)}). diff --git a/scripts/eval-threshold.mjs b/scripts/eval-threshold.mjs index 8eb9697..1da038b 100644 --- a/scripts/eval-threshold.mjs +++ b/scripts/eval-threshold.mjs @@ -158,7 +158,8 @@ const holdsTheLine = (row) => const eligible = rows.filter(holdsTheLine) const ranked = [...eligible].sort( (a, b) => - b.validation.unanswerable.noResultRate - a.validation.unanswerable.noResultRate || + b.validation.unanswerable.retrievalAbstentionRate - + a.validation.unanswerable.retrievalAbstentionRate || a.threshold - b.threshold ) /** @@ -170,32 +171,35 @@ const ranked = [...eligible].sort( const flat = rows.every( (row) => row.validation.metrics.ndcgAt10 === reference.validation.metrics.ndcgAt10 && - row.validation.unanswerable.noResultRate === reference.validation.unanswerable.noResultRate + row.validation.unanswerable.retrievalAbstentionRate === + reference.validation.unanswerable.retrievalAbstentionRate ) /** - * Refusals are the point of the second number: `noResultCount` of the unanswerable - * questions came back empty, which is the correct outcome. The rest returned passages the - * sources cannot support. + * Abstention is the point of the second number: `abstentionCount` of the unanswerable + * questions returned nothing, which is the correct outcome *at the retrieval layer*. The + * rest returned candidates the sources cannot support. */ -const refusals = (row) => `${row.unanswerable.noResultCount}/${row.unanswerable.questions}` +const abstained = (row) => `${row.unanswerable.abstentionCount}/${row.unanswerable.questions}` const describe = (row) => `nDCG@10 ${format4(row.metrics.ndcgAt10)}, Recall@5 ${format4(row.metrics.recallAt5)}, ` + `answerable no-result ${format4(row.noResultRate)}, ` + - `unanswerable refused ${refusals(row)}` + `retrieval abstained on unanswerable ${abstained(row)}, ` + + `context passages ${row.unanswerable.meanContextPassages.toFixed(1)}` const outcome = flat - ? `The sweep is **flat**: every threshold from ${THRESHOLDS[0]} to ${THRESHOLDS[THRESHOLDS.length - 1]} produces the same validation nDCG@10 (${format4(reference.validation.metrics.ndcgAt10)}), the same Recall@5 (${format4(reference.validation.metrics.recallAt5)}) and the same unanswerable refusal rate (${refusals(reference.validation)}). No passage is ever filtered out, so the threshold is **non-binding** on this corpus — E5 does not score these query/chunk pairs below the top of the swept range.\n\n**No evidence to change \`threshold = ${PRODUCTION_THRESHOLD}\`.** All thresholds hold the line equally; picking one would be arbitrary. The current value can be neither validated nor falsified here, which is a property of the corpus rather than of the threshold.` - : `**Recommended: \`threshold = ${winner.threshold}\`.**\n\n- **Validation**: ${describe(winner.validation)}\n- **Test**: ${describe(winner.test)}\n- Production ships \`${PRODUCTION_THRESHOLD}\`: validation ${describe(production.validation)}.\n\nThe rule held answerable quality at the \`threshold = 0\` level (nDCG@10 and context recall must not regress, on the validation split) and then took the threshold that refuses the most unanswerable questions. So this is a refusal gain, not a quality gain — if answerable quality had fallen, the threshold would have been ineligible regardless of how much it refused.` + ? `The sweep is **flat**: every threshold from ${THRESHOLDS[0]} to ${THRESHOLDS[THRESHOLDS.length - 1]} produces the same validation nDCG@10 (${format4(reference.validation.metrics.ndcgAt10)}), the same Recall@5 (${format4(reference.validation.metrics.recallAt5)}) and the same retrieval abstention rate on unanswerable questions (${abstained(reference.validation)}). No candidate is ever filtered out, so the threshold is **non-binding** on this corpus.\n\n**No evidence to change \`threshold = ${PRODUCTION_THRESHOLD}\`.** All thresholds hold the line equally; picking one would be arbitrary. The current value can be neither validated nor falsified here, which is a property of the corpus rather than of the threshold. Note also what this does *not* establish: abstention is a retrieval-layer statement — whether the model then declines to answer needs a generator eval.` + : `**Recommended: \`threshold = ${winner.threshold}\`.**\n\n- **Validation**: ${describe(winner.validation)}\n- **Test**: ${describe(winner.test)}\n- Production ships \`${PRODUCTION_THRESHOLD}\`: validation ${describe(production.validation)}.\n\nThe rule held answerable quality at the \`threshold = 0\` level (nDCG@10 and context recall must not regress, on the validation split) and then took the threshold that abstains on the most unanswerable questions. So this is an abstention gain, not a quality gain — if answerable quality had fallen, the threshold would have been ineligible regardless of how much it abstained.` const tableRows = rows .map( (row) => `| ${row.threshold} | ${row.validation.answerableQuestions} | ${format4(row.validation.metrics.recallAt5)} | ` + `${format4(row.validation.metrics.ndcgAt10)} | ${format4(row.validation.noResultRate)} | ` + - `${format4(row.validation.unanswerable.noResultRate)} | ` + - `${row.validation.unanswerable.meanRetrieved.toFixed(1)} | ${format4(row.test.metrics.ndcgAt10)} | ` + - `${format4(row.test.unanswerable.noResultRate)} |` + `${format4(row.validation.unanswerable.retrievalAbstentionRate)} | ` + + `${row.validation.unanswerable.meanCandidatesRetrieved.toFixed(1)} | ` + + `${row.validation.unanswerable.meanContextPassages.toFixed(1)} | ${format4(row.test.metrics.ndcgAt10)} | ` + + `${format4(row.test.unanswerable.retrievalAbstentionRate)} |` ) .join('\n') @@ -212,20 +216,22 @@ split selects; the **test** split reports. The split is the committed manifest Quality columns cover the answerable questions only; **Unans.** columns cover the unanswerable ones, where returning nothing is the desired outcome and so a *higher* -no-result rate is better. +abstention rate is better. Two sizes are kept apart: **cands** is how many candidates +passed the threshold (up to \`candidateK\`), **ctx** is how many reach the context window. -| Threshold | n (val) | Recall@5 (val) | nDCG@10 (val) | No-result (val) | Unans. no-result (val) | Unans. retrieved (val) | nDCG@10 (test) | Unans. no-result (test) | -| --- | --- | --- | --- | --- | --- | --- | --- | --- | +| Threshold | n (val) | Recall@5 (val) | nDCG@10 (val) | No-result (val) | Unans. abstained (val) | Unans. cands (val) | Unans. ctx (val) | nDCG@10 (test) | Unans. abstained (test) | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | ${tableRows} ## Selection rule Hold the answerable quality line — validation nDCG@10 and context recall must not -regress versus \`threshold = 0\` — then take the threshold that refuses the most +regress versus \`threshold = 0\` — then take the threshold that abstains on the most unanswerable questions. Tie-break on the lowest threshold. -Raising a threshold is only worth anything if it refuses what the sources do not answer; -the quality gate is there so a refusal gain can never be bought with a retrieval loss. +Raising a threshold is only worth anything if it stops unsupported context before the +prompt; the quality gate is there so an abstention gain can never be bought with a +retrieval loss. ## Outcome diff --git a/src/main/eval/harness.ts b/src/main/eval/harness.ts index 3297cc0..44427de 100644 --- a/src/main/eval/harness.ts +++ b/src/main/eval/harness.ts @@ -328,14 +328,19 @@ export async function runEvalHarness( return { type, questions: group.length, metrics: summarize(group, options.contextK) } }) - const unanswerableNoResults = unanswerableQuestions.filter((q) => q.retrievedCount === 0).length + const abstentionCount = unanswerableQuestions.filter((q) => q.retrievedCount === 0).length const unanswerable = { questions: unanswerableQuestions.length, - noResultCount: unanswerableNoResults, - // 目标方向与其他指标相反:没有相关资料时,返回空才是对的。 - noResultRate: - unanswerableQuestions.length === 0 ? 0 : unanswerableNoResults / unanswerableQuestions.length, - meanRetrieved: mean(unanswerableQuestions.map((q) => q.retrievedCount)) + abstentionCount, + // 方向与其它指标相反:没有相关资料时,检索层返回空才是对的。 + retrievalAbstentionRate: + unanswerableQuestions.length === 0 ? 0 : abstentionCount / unanswerableQuestions.length, + // 通过 threshold 的候选数,不是送进 prompt 的条数:harness 为了算 Recall@10 取满了 + // candidateK,把这个数当成 prompt 宽度会把问题说大。 + meanCandidatesRetrieved: mean(unanswerableQuestions.map((q) => q.retrievedCount)), + meanContextPassages: mean( + unanswerableQuestions.map((q) => Math.min(q.retrievedCount, options.contextK)) + ) } const chunking = { ...DEFAULT_CHUNK_OPTIONS, ...options.chunkOptions } diff --git a/src/main/eval/report.ts b/src/main/eval/report.ts index 9c2b96f..bc3290e 100644 --- a/src/main/eval/report.ts +++ b/src/main/eval/report.ts @@ -29,12 +29,13 @@ export function renderMarkdown(report: EvalReport): string { const answerableCount = report.byType.reduce((total, entry) => total + entry.questions, 0) const unanswerableNote = unanswerable.questions === 0 - ? 'This corpus carries **no** unanswerable question yet, so refusal is not measured.\n' + - 'The threshold cannot be tuned against it either: every question is answerable, so\n' + - 'every threshold returns something.' + ? 'This corpus carries **no** unanswerable question yet, so retrieval abstention is not\n' + + 'measured. The threshold cannot be tuned against it either: every question is answerable,\n' + + 'so every threshold returns something.' : `| Unanswerable questions | ${unanswerable.questions} |\n` + - `| Returned no results | ${format(unanswerable.noResultRate)} (${unanswerable.noResultCount}/${unanswerable.questions}) |\n` + - `| Mean passages retrieved | ${unanswerable.meanRetrieved.toFixed(2)} |` + `| Retrieval abstained | ${format(unanswerable.retrievalAbstentionRate)} (${unanswerable.abstentionCount}/${unanswerable.questions}) |\n` + + `| Mean candidates passing the threshold | ${unanswerable.meanCandidatesRetrieved.toFixed(2)} |\n` + + `| Mean passages in the context window | ${unanswerable.meanContextPassages.toFixed(2)} |` return `# RAG eval baseline — ${report.baseline} @@ -79,11 +80,20 @@ ${typeRows} ### Unanswerable questions -These carry no ground truth, so the correct outcome is that retrieval finds nothing. They +These carry no ground truth, so the correct outcome is that retrieval returns nothing. They are excluded from every metric above — a missing ground truth is not a miss — and reported -here instead. A higher **no-results** rate is better on this row, which is the opposite of -how it reads everywhere else, and \`mean passages retrieved\` is how much irrelevant context -was pulled in anyway. This is the row a threshold decision should move. +here instead. + +**This measures retrieval-level abstention, not the model refusing.** No generator runs in +this harness, so it can show that no candidate passed the threshold; it cannot show that the +final answer would say "not in your sources". A true system refusal rate needs a +generator eval. + +A higher abstention rate is better on this row, the opposite of how every other row reads. +The two sizes are kept apart on purpose: **candidates passing the threshold** can be as high +as \`candidateK\` (the harness fetches that many to compute \`Recall@10\`), while **passages in +the context window** is what a user's prompt would actually receive. A large first number +with a small second one means the threshold filters nothing and the window is all noise. | Metric | Value | | --- | --- | diff --git a/src/main/eval/types.ts b/src/main/eval/types.ts index de162bf..715396a 100644 --- a/src/main/eval/types.ts +++ b/src/main/eval/types.ts @@ -57,10 +57,16 @@ export type SplitAssignment = Record /** * 解析提交在仓库里的切分清单(`eval/splits.json`)。 * - * 用显式清单而不是 id 哈希(#192 评审):哈希看着确定,但它的确定是“每次结果一样”, - * 不是“每次划分一样”——往 `questions.jsonl` 里加一道题,会把其它题在 validation / - * test 之间挪动,而一个稀有类别(multi-hop、cross-lingual)可以在无人选择的情况下整体 - * 落到某一边。清单让划分是被 review 的,不是被算出来的。 + * 用显式清单而不是按 id 哈希(#192 评审)。先说清楚哈希**不是**哪里坏: + * `hash(id) % 3` 是逐 id 独立计算的,所以它是**稳定**的 —— 新增一道题不会挪动已有的题。 + * + * 它真正不能做的是表达实验设计意图: + * + * - 它无法保证小样本类别分层,于是 multi-hop / cross-lingual 这类稀有类别可能在无 + * 人选择的情况下整体落到某一侧; + * - 新增的题会被默默分到一侧,而不是被决定 —— 而 test 正是“选择”不该被拟合的那一侧。 + * + * 清单让“哪道题在哪一侧”成为一个被 review 的声明,而不是一个被算出来的结果。 */ export function parseSplitAssignment(raw: unknown, source: string): SplitAssignment { if (!raw || typeof raw !== 'object' || Array.isArray(raw)) { @@ -235,14 +241,26 @@ export interface EvalReport { * 不可答问题的单独一组(#192)。 * * 它们不进 `metrics`/`byType`:没有 ground truth,Recall 对它们是 0/0 而不是 0。 - * 它们评的是“该拒答时有没有硬找”——`noResultRate` 越接近 1 越好(在真的没有相关 - * 资料时返回空),`meanRetrieved` 则是“硬找了多少条相似但无关的上下文”。 + * + * 这一组量的是**检索层弃权**,不是模型拒答:harness 不跑生成模型,所以它只能证明 + * “没有候选通过 threshold”,不能证明最终回答会说“资料里没有”。真正的 system refusal + * 要等 generator eval。 */ unanswerable: { questions: number - noResultCount: number - noResultRate: number - meanRetrieved: number + /** 没有任何候选通过 threshold 的问题数。 */ + abstentionCount: number + /** `abstentionCount / questions`,在检索层含义下越高越好。 */ + retrievalAbstentionRate: number + /** + * 通过 threshold 的候选数,**不是**送进 prompt 的条数。 + * + * 上限是 `candidateK`(harness 为了算 Recall@10 故意取满),所以这个数接近 + * `candidateK` 时说明 threshold 基本没挡掉任何东西。 + */ + meanCandidatesRetrieved: number + /** 真正进入 context 窗口的条数:`min(retrievedCount, contextK)`。 */ + meanContextPassages: number } /** `indexingMs` 只用于 #78 的吞吐比较;它不在确定报告里,也不该成为差异原因。 */ timing: { latencyP50Ms: number; latencyP95Ms: number; indexingMs: number } diff --git a/test/evalSplit.test.ts b/test/evalSplit.test.ts index c091051..a57e7dc 100644 --- a/test/evalSplit.test.ts +++ b/test/evalSplit.test.ts @@ -11,10 +11,11 @@ import { * The eval split (#192). Parameters that get swept have to be chosen on questions that * did not take part in the choice. * - * It used to be a hash of the question id. That was reproducible but not *stable*: - * adding a question moved others between the sides, and a rare query type could end up - * entirely on one side without anyone choosing that. The split is now an explicit - * committed manifest, and these pin the properties that make it trustworthy. + * It used to be a hash of the question id. That is reproducible and, because it is computed + * per id, *stable* when a question is added — but it cannot express the experimental + * design: it does not stratify a small corpus, so a rare query type can end up entirely on + * one side without anyone choosing that, and a new question is assigned silently rather + * than deliberately. The split is now an explicit committed manifest. */ const question = (id: string): EvalQuestion => ({