diff --git a/docs/eval/baseline-v1.6.json b/docs/eval/baseline-v1.6.json index cfb0455..e598a6c 100644 --- a/docs/eval/baseline-v1.6.json +++ b/docs/eval/baseline-v1.6.json @@ -15,7 +15,6 @@ "candidateK": 20, "contextK": 3, "threshold": 0.5, - "evidenceK": 5, "corpus": "eval/corpus", "documents": 13, "questions": 30, @@ -29,7 +28,8 @@ "ndcgAt10": 0.94375, "hitRateAt5": 1, "mapAt10": 0.922222, - "evidencePrecisionAt5": 0.213333 + "contextPrecision": 0.355556, + "contextRecall": 1 }, "byType": [ { @@ -43,7 +43,8 @@ "ndcgAt10": 0.63093, "hitRateAt5": 1, "mapAt10": 0.5, - "evidencePrecisionAt5": 0.2 + "contextPrecision": 0.333333, + "contextRecall": 1 } }, { @@ -57,7 +58,8 @@ "ndcgAt10": 1, "hitRateAt5": 1, "mapAt10": 1, - "evidencePrecisionAt5": 0.2 + "contextPrecision": 0.333333, + "contextRecall": 1 } }, { @@ -71,7 +73,8 @@ "ndcgAt10": 0.95986, "hitRateAt5": 1, "mapAt10": 0.916667, - "evidencePrecisionAt5": 0.4 + "contextPrecision": 0.666667, + "contextRecall": 1 } }, { @@ -85,7 +88,8 @@ "ndcgAt10": 0.931214, "hitRateAt5": 1, "mapAt10": 0.907407, - "evidencePrecisionAt5": 0.2 + "contextPrecision": 0.333333, + "contextRecall": 1 } }, { @@ -99,7 +103,8 @@ "ndcgAt10": 1, "hitRateAt5": 1, "mapAt10": 1, - "evidencePrecisionAt5": 0.2 + "contextPrecision": 0.333333, + "contextRecall": 1 } } ], diff --git a/docs/eval/baseline-v1.6.md b/docs/eval/baseline-v1.6.md index ce25b05..467fafb 100644 --- a/docs/eval/baseline-v1.6.md +++ b/docs/eval/baseline-v1.6.md @@ -11,7 +11,6 @@ Generated by `npm run eval`. The numbers below are harness output — do not edi | Retrieval | `dense` | | Ranks | `candidateK=20, threshold=0.5` | | Context width | `contextK=3` | -| Evidence per query | `evidenceK=5` | | Corpus | `eval/corpus` (13 documents) | | Split | `all` (30 questions) | | Index size | 19 chunks | @@ -27,7 +26,8 @@ Generated by `npm run eval`. The numbers below are harness output — do not edi | nDCG@10 | 0.9437 | | Hit rate@5 | 1.0000 | | MAP@10 | 0.9222 | -| Evidence precision@5 | 0.2133 | +| Context precision@3 | 0.3556 | +| Context recall@3 | 1.0000 | ### By query type @@ -43,8 +43,8 @@ The type comes from `type` in `questions.jsonl`; untagged questions report as | semantic | 18 | 1.0000 | 0.9312 | 1.0000 | 0.9074 | | zh | 3 | 1.0000 | 1.0000 | 1.0000 | 1.0000 | -Timing is informational only and is **not** frozen: indexing 1551 ms, query -p50 11.73 ms, p95 18.71 ms on the +Timing is informational only and is **not** frozen: indexing 1499 ms, query +p50 11.58 ms, p95 16.69 ms on the machine that produced this file. Timing and index size depend on hardware and on the corpus, so they must never be the reason two runs differ. @@ -52,11 +52,17 @@ corpus, so they must never be the reason two runs differ. - A retrieved passage is relevant when its provenance covers a ground-truth block. - **Recall@k** is the share of ground-truth blocks covered by the first `k` passages. -- **Evidence precision@5** is the share of the first `5` - retrieved passages that cover a ground-truth block. This is **retrieval precision**, not - answer citation recall: the harness runs no model and produces no answer. Answer-level - citation correctness is covered by the resolver (#70); a model-driven answer eval is a - separate deliverable. +- **Context precision@3** is the share of the first `3` + retrieved passages that cover a ground-truth block. **Context recall@3** is + the share of the needed ground-truth blocks that made it into that same window. Both are + deterministic: the dataset says which blocks answer the question, so no model is needed to + score the window. Together they are the trade-off a `contextK` decision actually makes — + a wider window finds more and carries more noise. +- This is **retrieval precision/recall, not answer citation recall**: the harness runs no + model and produces no answer. Answer-level citation correctness is covered by the + resolver (#70). Faithfulness, completeness and answer correctness need a generative model + and are **not evaluated here** — the harness runs offline with only the pinned embedding + model, the same constraint that keeps the reranker unmeasured (#170). - Ground truth is expressed in corpus identity (`document` relative path + `block` ordinal + optional `quote`), never a runtime `documentId`/`blockId`. diff --git a/docs/eval/retrieval-v1.6.json b/docs/eval/retrieval-v1.6.json index 8fed18f..6782582 100644 --- a/docs/eval/retrieval-v1.6.json +++ b/docs/eval/retrieval-v1.6.json @@ -14,9 +14,10 @@ "ndcgAt10": 0.94375, "hitRateAt5": 1, "mapAt10": 0.922222, - "evidencePrecisionAt5": 0.213333, - "indexingMs": 1531, - "latencyP95Ms": 14.81 + "contextPrecision": 0.355556, + "contextRecall": 1, + "indexingMs": 1548, + "latencyP95Ms": 20.51 }, { "id": "sparse", @@ -30,9 +31,10 @@ "ndcgAt10": 0.818355, "hitRateAt5": 0.866667, "mapAt10": 0.8, - "evidencePrecisionAt5": 0.186667, - "indexingMs": 1503, - "latencyP95Ms": 1.89 + "contextPrecision": 0.311111, + "contextRecall": 0.866667, + "indexingMs": 1490, + "latencyP95Ms": 1.96 }, { "id": "hybrid", @@ -46,9 +48,10 @@ "ndcgAt10": 0.956053, "hitRateAt5": 1, "mapAt10": 0.938889, - "evidencePrecisionAt5": 0.213333, - "indexingMs": 1548, - "latencyP95Ms": 19.15 + "contextPrecision": 0.355556, + "contextRecall": 1, + "indexingMs": 1507, + "latencyP95Ms": 20.66 } ] } diff --git a/docs/eval/retrieval-v1.6.md b/docs/eval/retrieval-v1.6.md index 9651e44..c13d4e4 100644 --- a/docs/eval/retrieval-v1.6.md +++ b/docs/eval/retrieval-v1.6.md @@ -7,11 +7,11 @@ Generated by `node scripts/eval-retrieval.mjs`. Numbers are harness output; do n Every strategy runs the real RAG eval harness against the same corpus and the same 30 questions as `baseline-v1.6.json`, with chunking held fixed at 1000/100. Only the retrieval strategy changes. -| Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Evidence P@5 | Query p95 | +| Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Context P | Query p95 | | --- | --- | --- | --- | --- | --- | --- | --- | -| dense (vector) | 0.8333 | 1.0000 | 0.9278 | 0.9437 | 0.9222 | 0.2133 | 14.81 ms | -| sparse (BM25) | 0.7333 | 0.8667 | 0.8056 | 0.8184 | 0.8000 | 0.1867 | 1.89 ms | -| hybrid (RRF of dense + BM25) | 0.8667 | 1.0000 | 0.9444 | 0.9561 | 0.9389 | 0.2133 | 19.15 ms | +| dense (vector) | 0.8333 | 1.0000 | 0.9278 | 0.9437 | 0.9222 | 0.3556 | 20.51 ms | +| sparse (BM25) | 0.7333 | 0.8667 | 0.8056 | 0.8184 | 0.8000 | 0.3111 | 1.96 ms | +| hybrid (RRF of dense + BM25) | 0.8667 | 1.0000 | 0.9444 | 0.9561 | 0.9389 | 0.3556 | 20.66 ms | ## Not evaluated diff --git a/eval/README.md b/eval/README.md index 2333a26..913073e 100644 --- a/eval/README.md +++ b/eval/README.md @@ -135,11 +135,17 @@ from unanswerable queries. was *complete*. A two-passage question that finds one scores 1.0 and 0.5 respectively. - **MAP@10** — mean average precision. The one metric here that combines ranking position with coverage, so pulling a second relevant passage from rank 9 to rank 2 moves it. -- **Evidence precision@5** — of the first 5 retrieved passages, the share that cover a - ground-truth block. This is **retrieval precision, not answer citation recall**: the - harness runs no model and produces no answer. Answer-level citation correctness is - covered by the resolver (#70); a model-driven answer eval would be a separate - deliverable. +- **Context precision@`contextK`** — of the first `contextK` retrieved passages, the share + that cover a ground-truth block. **Context recall@`contextK`** — the share of the needed + ground-truth blocks that made it into that same window. Both are deterministic: the + dataset says which blocks answer the question, so no model is needed to score the window. + Together they are the trade-off a `contextK` decision actually makes — a wider window + finds more and carries more noise. +- This is **retrieval precision/recall, not answer citation recall**: the harness runs no + model and produces no answer. Answer-level citation correctness is covered by the + resolver (#70). Faithfulness, completeness and answer correctness need a generative model + and are **not evaluated here** — the harness runs offline with only the pinned embedding + model, the same constraint that keeps the reranker unmeasured (#170). - **By query type** — the same metrics per `type` in `questions.jsonl` (`exact`, `semantic`, `multi-hop`, `cross-lingual`, `zh`). A single average hides a change that helps one kind of question and hurts another; the current baseline already shows this, diff --git a/scripts/eval-chunking.mjs b/scripts/eval-chunking.mjs index 70f4e45..52bebbf 100644 --- a/scripts/eval-chunking.mjs +++ b/scripts/eval-chunking.mjs @@ -187,7 +187,7 @@ const format4 = (value) => value.toFixed(4) const rows = results.map((result) => { const args = `${result.chunkSize}/${result.chunkOverlap}${result.respectHeadings ? ' + headings' : ''}${result.allowSpanPages ? ' + span' : ''}` - return `| ${result.label} | \`${args}\` | ${format4(result.recallAt1)} | ${format4(result.recallAt5)} | ${format4(result.mrr)} | ${format4(result.ndcgAt10)} | ${format4(result.evidencePrecisionAt5)} | ${result.chunkCount} (${formatDelta(delta(result.chunkCount, baseline.chunkCount))}) | ${result.indexingMs} ms | ${result.latencyP95Ms?.toFixed(2)} ms |` + return `| ${result.label} | \`${args}\` | ${format4(result.recallAt1)} | ${format4(result.recallAt5)} | ${format4(result.mrr)} | ${format4(result.ndcgAt10)} | ${format4(result.contextPrecision)} | ${result.chunkCount} (${formatDelta(delta(result.chunkCount, baseline.chunkCount))}) | ${result.indexingMs} ms | ${result.latencyP95Ms?.toFixed(2)} ms |` }) /** @@ -215,7 +215,7 @@ questions as \`baseline-v1.4.json\`, with dense retrieval held fixed. Only the c configuration changes, so a difference in the metrics is a difference in the input distribution retrieval is measured on. -| Variant | size/overlap | Recall@1 | Recall@5 | MRR | nDCG@10 | Evidence P@5 | Index size (Δ) | Indexing | Query p95 | +| Variant | size/overlap | Recall@1 | Recall@5 | MRR | nDCG@10 | Context P | Index size (Δ) | Indexing | Query p95 | | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | ${rows.join('\n')} diff --git a/scripts/eval-retrieval.mjs b/scripts/eval-retrieval.mjs index d52b5bd..5964a28 100644 --- a/scripts/eval-retrieval.mjs +++ b/scripts/eval-retrieval.mjs @@ -141,7 +141,7 @@ const format4 = (value) => value.toFixed(4) const rows = results.map( (result) => - `| ${result.label} | ${format4(result.recallAt1)} | ${format4(result.recallAt5)} | ${format4(result.mrr)} | ${format4(result.ndcgAt10)} | ${format4(result.mapAt10)} | ${format4(result.evidencePrecisionAt5)} | ${result.latencyP95Ms?.toFixed(2)} ms |` + `| ${result.label} | ${format4(result.recallAt1)} | ${format4(result.recallAt5)} | ${format4(result.mrr)} | ${format4(result.ndcgAt10)} | ${format4(result.mapAt10)} | ${format4(result.contextPrecision)} | ${result.latencyP95Ms?.toFixed(2)} ms |` ) /** @@ -179,7 +179,7 @@ Generated by \`node scripts/eval-retrieval.mjs\`. Numbers are harness output; do Every strategy runs the real RAG eval harness against the same corpus and the same 30 questions as \`baseline-v1.6.json\`, with chunking held fixed at ${baseline.chunking}. Only the retrieval strategy changes. -| Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Evidence P@5 | Query p95 | +| Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Context P | Query p95 | | --- | --- | --- | --- | --- | --- | --- | --- | ${rows.join('\n')} diff --git a/src/main/eval/harness.ts b/src/main/eval/harness.ts index 7ae5114..c33c43d 100644 --- a/src/main/eval/harness.ts +++ b/src/main/eval/harness.ts @@ -67,8 +67,6 @@ export interface EvalHarnessOptions { contextK: number /** Similarity floor; 0 keeps the ranking intact for ranking metrics. */ threshold: number - /** How many retrieved passages the evidence-precision metric looks at. */ - evidenceK: number /** 分块配置(#78)。实验变体通过它选择策略;缺省时用生产默认值。 */ chunkOptions: ChunkOptions /** 检索策略(#77):dense / sparse(BM25) / hybrid(RRF)。 */ @@ -84,7 +82,7 @@ const UNTAGGED = 'untagged' * 同一套指标既算总平均,也算每个查询类别(#192)。用一个函数是因为分组平均必须与 * 总平均是同一个定义,否则两个数就不可比。 */ -function summarize(perQuestion: readonly QuestionReport[], evidenceK: number): EvalMetrics { +function summarize(perQuestion: readonly QuestionReport[], contextK: number): EvalMetrics { return { recallAt1: mean(perQuestion.map((q) => recallAtK(q.matchesByRank, q.relevantCount, 1))), recallAt5: mean(perQuestion.map((q) => recallAtK(q.matchesByRank, q.relevantCount, 5))), @@ -95,8 +93,13 @@ function summarize(perQuestion: readonly QuestionReport[], evidenceK: number): E mapAt10: mean( perQuestion.map((q) => averagePrecisionAtK(q.matchesByRank, q.relevantCount, 10)) ), - evidencePrecisionAt5: mean( - perQuestion.map((q) => evidencePrecisionAtK(q.matchesByRank, evidenceK)) + // 两个 context 指标共用同一个窗口,因为它们回答的是同一个问题的两面:送进 prompt + // 的那几条里有多少是相关的,以及需要的东西有多少真的进去了。 + contextPrecision: mean( + perQuestion.map((q) => evidencePrecisionAtK(q.matchesByRank, contextK)) + ), + contextRecall: mean( + perQuestion.map((q) => recallAtK(q.matchesByRank, q.relevantCount, contextK)) ) } } @@ -266,14 +269,14 @@ export async function runEvalHarness( }) } - const metrics = summarize(perQuestion, options.evidenceK) + const metrics = summarize(perQuestion, options.contextK) // 每个类别一行,按类别名排序,所以同一个 JSON 在两次运行之间可 diff。 const byType: EvalTypeBreakdown[] = [...new Set(perQuestion.map((q) => q.type))] .sort() .map((type) => { const group = perQuestion.filter((q) => q.type === type) - return { type, questions: group.length, metrics: summarize(group, options.evidenceK) } + return { type, questions: group.length, metrics: summarize(group, options.contextK) } }) const chunking = { ...DEFAULT_CHUNK_OPTIONS, ...options.chunkOptions } @@ -294,7 +297,6 @@ export async function runEvalHarness( candidateK: options.candidateK, contextK: options.contextK, threshold: options.threshold, - evidenceK: options.evidenceK, corpus: options.corpusLabel, documents: documentIds.size, questions: questions.length, @@ -323,7 +325,8 @@ function roundMetrics(metrics: EvalMetrics): EvalMetrics { ndcgAt10: roundMetric(metrics.ndcgAt10), hitRateAt5: roundMetric(metrics.hitRateAt5), mapAt10: roundMetric(metrics.mapAt10), - evidencePrecisionAt5: roundMetric(metrics.evidencePrecisionAt5) + contextPrecision: roundMetric(metrics.contextPrecision), + contextRecall: roundMetric(metrics.contextRecall) } } diff --git a/src/main/eval/report.ts b/src/main/eval/report.ts index 7b883f8..8f6dfaa 100644 --- a/src/main/eval/report.ts +++ b/src/main/eval/report.ts @@ -37,7 +37,6 @@ Generated by \`${report.generatedBy}\`. The numbers below are harness output — | Retrieval | \`${config.retrieval}\` | | Ranks | \`candidateK=${config.candidateK}, threshold=${config.threshold}\` | | Context width | \`contextK=${config.contextK}\` | -| Evidence per query | \`evidenceK=${config.evidenceK}\` | | Corpus | \`${config.corpus}\` (${config.documents} documents) | | Split | \`${config.split}\` (${config.questions} questions) | | Index size | ${config.chunkCount} chunks | @@ -53,7 +52,8 @@ Generated by \`${report.generatedBy}\`. The numbers below are harness output — | nDCG@10 | ${format(metrics.ndcgAt10)} | | Hit rate@5 | ${format(metrics.hitRateAt5)} | | MAP@10 | ${format(metrics.mapAt10)} | -| Evidence precision@${config.evidenceK} | ${format(metrics.evidencePrecisionAt5)} | +| Context precision@${config.contextK} | ${format(metrics.contextPrecision)} | +| Context recall@${config.contextK} | ${format(metrics.contextRecall)} | ### By query type @@ -74,11 +74,17 @@ corpus, so they must never be the reason two runs differ. - A retrieved passage is relevant when its provenance covers a ground-truth block. - **Recall@k** is the share of ground-truth blocks covered by the first \`k\` passages. -- **Evidence precision@${config.evidenceK}** is the share of the first \`${config.evidenceK}\` - retrieved passages that cover a ground-truth block. This is **retrieval precision**, not - answer citation recall: the harness runs no model and produces no answer. Answer-level - citation correctness is covered by the resolver (#70); a model-driven answer eval is a - separate deliverable. +- **Context precision@${config.contextK}** is the share of the first \`${config.contextK}\` + retrieved passages that cover a ground-truth block. **Context recall@${config.contextK}** is + the share of the needed ground-truth blocks that made it into that same window. Both are + deterministic: the dataset says which blocks answer the question, so no model is needed to + score the window. Together they are the trade-off a \`contextK\` decision actually makes — + a wider window finds more and carries more noise. +- This is **retrieval precision/recall, not answer citation recall**: the harness runs no + model and produces no answer. Answer-level citation correctness is covered by the + resolver (#70). Faithfulness, completeness and answer correctness need a generative model + and are **not evaluated here** — the harness runs offline with only the pinned embedding + model, the same constraint that keeps the reranker unmeasured (#170). - Ground truth is expressed in corpus identity (\`document\` relative path + \`block\` ordinal + optional \`quote\`), never a runtime \`documentId\`/\`blockId\`. diff --git a/src/main/eval/run.ts b/src/main/eval/run.ts index 802cf5e..d927286 100644 --- a/src/main/eval/run.ts +++ b/src/main/eval/run.ts @@ -197,7 +197,6 @@ export async function runEvalCli(argv: readonly string[] = process.argv): Promis candidateK: readNumberOption(argv, '--eval-candidate-k=', 20), contextK: readNumberOption(argv, '--eval-context-k=', 3), threshold: readNumberOption(argv, '--eval-threshold=', 0.5), - evidenceK: 5, chunkOptions: readChunkOptions(argv), strategy: readRetrievalStrategy(argv) } diff --git a/src/main/eval/types.ts b/src/main/eval/types.ts index 3a075f4..8833a23 100644 --- a/src/main/eval/types.ts +++ b/src/main/eval/types.ts @@ -82,13 +82,19 @@ export interface EvalMetrics { /** AP@10:把「排序位置」和「覆盖面」合成一个数的那个指标。 */ mapAt10: number /** - * Share of the first `evidenceK` retrieved passages that cover ground truth. + * 前 `contextK` 条证据里真的命中 ground-truth 的比例(context 条的精确率)。 * - * This is **retrieval precision**, not answer citation recall: no model runs in - * this harness and no answer is produced. Answer-level citation correctness is - * the resolver's job (#70) and would need a separate, model-driven eval. + * 这是**检索精度**,不是回答的引用召回:harness 不跑模型、不产生回答。回答层的引用 + * 正确性是 resolver 的事(#70),需要一个真正跑模型的 eval。 */ - evidencePrecisionAt5: number + contextPrecision: number + /** + * 答案需要的 ground-truth 块有多少进了 `contextK` 宽的窗口。 + * + * 与 Recall@10 的区别在于它量的是**窗口**:证据排在第 4、而 contextK=3 时,模型 + * 看不到它。这是一个产品指标,不只是检索指标。 + */ + contextRecall: number } /** 按查询类别聚合的同一套指标(#192)。总平均会掩盖方向相反的两个变化。 */ @@ -134,13 +140,11 @@ export interface EvalReport { /** * 生产 prompt 实际取用的证据条数(#77)。 * - * 快照里记它是为了让 benchmark 描述整条线上链路,而不只是检索器;它不影响排名 - * 指标 —— 截断只是取候选列表的前缀,前缀的排序不变。 + * 快照里记它是为了让 benchmark 描述整条线上链路,而不只是检索器;它也是两个 + * context 指标的窗口宽度。 */ contextK: number threshold: number - /** How many retrieved passages the evidence-precision metric looks at. */ - evidenceK: number corpus: string documents: number questions: number