From b77257a5ddf249e01d32ab511bb3e0bbf15f36f5 Mon Sep 17 00:00:00 2001 From: MrSibe Date: Wed, 30 Sep 2026 17:02:01 +0800 Subject: [PATCH] feat(eval): measure the context window, and name what still needs a model MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The harness measured the retriever but never the window the prompt actually gets. `evidenceK: 5` was a separate constant from the `contextK: 3` production uses, so the one metric that looked at a window looked at a different one than the product does. Child 5 of #192. **`contextK` is now the window for both context metrics.** - `contextPrecision@contextK` — of the first `contextK` passages, the share covering ground truth. This is what `evidencePrecisionAt5` was, at the production width. - `contextRecall@contextK` — the share of needed ground-truth blocks that made it into that window. Distinct from `Recall@10`: a block found at rank 4 is invisible when `contextK = 3`, and that is a product fact, not a ranking fact. Both are **deterministic**: the dataset says which blocks answer the question, so no model is needed to score a window. Together they are the trade-off a `contextK` decision makes — a wider window finds more and carries more noise — which is what the sweep in the next child needs. Baseline: `contextPrecision@3 = 0.3556`, `contextRecall@3 = 1.0000` — the needed evidence is always inside the top 3 on this corpus, but only about a third of what is inside the window is relevant. That second number is the one that says the window is paying for passages that do not answer the question. **What is deliberately not here.** Faithfulness, completeness, answer correctness and noise sensitivity need a generative model. The harness runs offline with only the pinned embedding model — the same constraint that keeps the reranker unmeasured (#170) — so this PR does not add them and does not fake them: the report's Definitions section now says so, and the "Not evaluated" line in `eval-retrieval.mjs` points at the same constraint. Adding an LLM judge is a separate change that has to solve model pinning first, not a line of code. `evidenceK` is removed from the harness config, so there is exactly one context width. ## Testing - `npm run typecheck` — clean - `npm test` — 497 pass - `npm run eval` twice — byte-identical `docs/eval/baseline-v1.6.json` - `npm run eval:retrieval` — regenerated; hybrid still clears the amended rule Part of #192 (child 5, deterministic half). --- docs/eval/baseline-v1.6.json | 19 ++++++++++++------- docs/eval/baseline-v1.6.md | 24 +++++++++++++++--------- docs/eval/retrieval-v1.6.json | 21 ++++++++++++--------- docs/eval/retrieval-v1.6.md | 8 ++++---- eval/README.md | 16 +++++++++++----- scripts/eval-chunking.mjs | 4 ++-- scripts/eval-retrieval.mjs | 4 ++-- src/main/eval/harness.ts | 21 ++++++++++++--------- src/main/eval/report.ts | 20 +++++++++++++------- src/main/eval/run.ts | 1 - src/main/eval/types.ts | 22 +++++++++++++--------- 11 files changed, 96 insertions(+), 64 deletions(-) diff --git a/docs/eval/baseline-v1.6.json b/docs/eval/baseline-v1.6.json index cfb0455..e598a6c 100644 --- a/docs/eval/baseline-v1.6.json +++ b/docs/eval/baseline-v1.6.json @@ -15,7 +15,6 @@ "candidateK": 20, "contextK": 3, "threshold": 0.5, - "evidenceK": 5, "corpus": "eval/corpus", "documents": 13, "questions": 30, @@ -29,7 +28,8 @@ "ndcgAt10": 0.94375, "hitRateAt5": 1, "mapAt10": 0.922222, - "evidencePrecisionAt5": 0.213333 + "contextPrecision": 0.355556, + "contextRecall": 1 }, "byType": [ { @@ -43,7 +43,8 @@ "ndcgAt10": 0.63093, "hitRateAt5": 1, "mapAt10": 0.5, - "evidencePrecisionAt5": 0.2 + "contextPrecision": 0.333333, + "contextRecall": 1 } }, { @@ -57,7 +58,8 @@ "ndcgAt10": 1, "hitRateAt5": 1, "mapAt10": 1, - "evidencePrecisionAt5": 0.2 + "contextPrecision": 0.333333, + "contextRecall": 1 } }, { @@ -71,7 +73,8 @@ "ndcgAt10": 0.95986, "hitRateAt5": 1, "mapAt10": 0.916667, - "evidencePrecisionAt5": 0.4 + "contextPrecision": 0.666667, + "contextRecall": 1 } }, { @@ -85,7 +88,8 @@ "ndcgAt10": 0.931214, "hitRateAt5": 1, "mapAt10": 0.907407, - "evidencePrecisionAt5": 0.2 + "contextPrecision": 0.333333, + "contextRecall": 1 } }, { @@ -99,7 +103,8 @@ "ndcgAt10": 1, "hitRateAt5": 1, "mapAt10": 1, - "evidencePrecisionAt5": 0.2 + "contextPrecision": 0.333333, + "contextRecall": 1 } } ], diff --git a/docs/eval/baseline-v1.6.md b/docs/eval/baseline-v1.6.md index ce25b05..467fafb 100644 --- a/docs/eval/baseline-v1.6.md +++ b/docs/eval/baseline-v1.6.md @@ -11,7 +11,6 @@ Generated by `npm run eval`. The numbers below are harness output — do not edi | Retrieval | `dense` | | Ranks | `candidateK=20, threshold=0.5` | | Context width | `contextK=3` | -| Evidence per query | `evidenceK=5` | | Corpus | `eval/corpus` (13 documents) | | Split | `all` (30 questions) | | Index size | 19 chunks | @@ -27,7 +26,8 @@ Generated by `npm run eval`. The numbers below are harness output — do not edi | nDCG@10 | 0.9437 | | Hit rate@5 | 1.0000 | | MAP@10 | 0.9222 | -| Evidence precision@5 | 0.2133 | +| Context precision@3 | 0.3556 | +| Context recall@3 | 1.0000 | ### By query type @@ -43,8 +43,8 @@ The type comes from `type` in `questions.jsonl`; untagged questions report as | semantic | 18 | 1.0000 | 0.9312 | 1.0000 | 0.9074 | | zh | 3 | 1.0000 | 1.0000 | 1.0000 | 1.0000 | -Timing is informational only and is **not** frozen: indexing 1551 ms, query -p50 11.73 ms, p95 18.71 ms on the +Timing is informational only and is **not** frozen: indexing 1499 ms, query +p50 11.58 ms, p95 16.69 ms on the machine that produced this file. Timing and index size depend on hardware and on the corpus, so they must never be the reason two runs differ. @@ -52,11 +52,17 @@ corpus, so they must never be the reason two runs differ. - A retrieved passage is relevant when its provenance covers a ground-truth block. - **Recall@k** is the share of ground-truth blocks covered by the first `k` passages. -- **Evidence precision@5** is the share of the first `5` - retrieved passages that cover a ground-truth block. This is **retrieval precision**, not - answer citation recall: the harness runs no model and produces no answer. Answer-level - citation correctness is covered by the resolver (#70); a model-driven answer eval is a - separate deliverable. +- **Context precision@3** is the share of the first `3` + retrieved passages that cover a ground-truth block. **Context recall@3** is + the share of the needed ground-truth blocks that made it into that same window. Both are + deterministic: the dataset says which blocks answer the question, so no model is needed to + score the window. Together they are the trade-off a `contextK` decision actually makes — + a wider window finds more and carries more noise. +- This is **retrieval precision/recall, not answer citation recall**: the harness runs no + model and produces no answer. Answer-level citation correctness is covered by the + resolver (#70). Faithfulness, completeness and answer correctness need a generative model + and are **not evaluated here** — the harness runs offline with only the pinned embedding + model, the same constraint that keeps the reranker unmeasured (#170). - Ground truth is expressed in corpus identity (`document` relative path + `block` ordinal + optional `quote`), never a runtime `documentId`/`blockId`. diff --git a/docs/eval/retrieval-v1.6.json b/docs/eval/retrieval-v1.6.json index 8fed18f..6782582 100644 --- a/docs/eval/retrieval-v1.6.json +++ b/docs/eval/retrieval-v1.6.json @@ -14,9 +14,10 @@ "ndcgAt10": 0.94375, "hitRateAt5": 1, "mapAt10": 0.922222, - "evidencePrecisionAt5": 0.213333, - "indexingMs": 1531, - "latencyP95Ms": 14.81 + "contextPrecision": 0.355556, + "contextRecall": 1, + "indexingMs": 1548, + "latencyP95Ms": 20.51 }, { "id": "sparse", @@ -30,9 +31,10 @@ "ndcgAt10": 0.818355, "hitRateAt5": 0.866667, "mapAt10": 0.8, - "evidencePrecisionAt5": 0.186667, - "indexingMs": 1503, - "latencyP95Ms": 1.89 + "contextPrecision": 0.311111, + "contextRecall": 0.866667, + "indexingMs": 1490, + "latencyP95Ms": 1.96 }, { "id": "hybrid", @@ -46,9 +48,10 @@ "ndcgAt10": 0.956053, "hitRateAt5": 1, "mapAt10": 0.938889, - "evidencePrecisionAt5": 0.213333, - "indexingMs": 1548, - "latencyP95Ms": 19.15 + "contextPrecision": 0.355556, + "contextRecall": 1, + "indexingMs": 1507, + "latencyP95Ms": 20.66 } ] } diff --git a/docs/eval/retrieval-v1.6.md b/docs/eval/retrieval-v1.6.md index 9651e44..c13d4e4 100644 --- a/docs/eval/retrieval-v1.6.md +++ b/docs/eval/retrieval-v1.6.md @@ -7,11 +7,11 @@ Generated by `node scripts/eval-retrieval.mjs`. Numbers are harness output; do n Every strategy runs the real RAG eval harness against the same corpus and the same 30 questions as `baseline-v1.6.json`, with chunking held fixed at 1000/100. Only the retrieval strategy changes. -| Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Evidence P@5 | Query p95 | +| Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Context P | Query p95 | | --- | --- | --- | --- | --- | --- | --- | --- | -| dense (vector) | 0.8333 | 1.0000 | 0.9278 | 0.9437 | 0.9222 | 0.2133 | 14.81 ms | -| sparse (BM25) | 0.7333 | 0.8667 | 0.8056 | 0.8184 | 0.8000 | 0.1867 | 1.89 ms | -| hybrid (RRF of dense + BM25) | 0.8667 | 1.0000 | 0.9444 | 0.9561 | 0.9389 | 0.2133 | 19.15 ms | +| dense (vector) | 0.8333 | 1.0000 | 0.9278 | 0.9437 | 0.9222 | 0.3556 | 20.51 ms | +| sparse (BM25) | 0.7333 | 0.8667 | 0.8056 | 0.8184 | 0.8000 | 0.3111 | 1.96 ms | +| hybrid (RRF of dense + BM25) | 0.8667 | 1.0000 | 0.9444 | 0.9561 | 0.9389 | 0.3556 | 20.66 ms | ## Not evaluated diff --git a/eval/README.md b/eval/README.md index 2333a26..913073e 100644 --- a/eval/README.md +++ b/eval/README.md @@ -135,11 +135,17 @@ from unanswerable queries. was *complete*. A two-passage question that finds one scores 1.0 and 0.5 respectively. - **MAP@10** — mean average precision. The one metric here that combines ranking position with coverage, so pulling a second relevant passage from rank 9 to rank 2 moves it. -- **Evidence precision@5** — of the first 5 retrieved passages, the share that cover a - ground-truth block. This is **retrieval precision, not answer citation recall**: the - harness runs no model and produces no answer. Answer-level citation correctness is - covered by the resolver (#70); a model-driven answer eval would be a separate - deliverable. +- **Context precision@`contextK`** — of the first `contextK` retrieved passages, the share + that cover a ground-truth block. **Context recall@`contextK`** — the share of the needed + ground-truth blocks that made it into that same window. Both are deterministic: the + dataset says which blocks answer the question, so no model is needed to score the window. + Together they are the trade-off a `contextK` decision actually makes — a wider window + finds more and carries more noise. +- This is **retrieval precision/recall, not answer citation recall**: the harness runs no + model and produces no answer. Answer-level citation correctness is covered by the + resolver (#70). Faithfulness, completeness and answer correctness need a generative model + and are **not evaluated here** — the harness runs offline with only the pinned embedding + model, the same constraint that keeps the reranker unmeasured (#170). - **By query type** — the same metrics per `type` in `questions.jsonl` (`exact`, `semantic`, `multi-hop`, `cross-lingual`, `zh`). A single average hides a change that helps one kind of question and hurts another; the current baseline already shows this, diff --git a/scripts/eval-chunking.mjs b/scripts/eval-chunking.mjs index 70f4e45..52bebbf 100644 --- a/scripts/eval-chunking.mjs +++ b/scripts/eval-chunking.mjs @@ -187,7 +187,7 @@ const format4 = (value) => value.toFixed(4) const rows = results.map((result) => { const args = `${result.chunkSize}/${result.chunkOverlap}${result.respectHeadings ? ' + headings' : ''}${result.allowSpanPages ? ' + span' : ''}` - return `| ${result.label} | \`${args}\` | ${format4(result.recallAt1)} | ${format4(result.recallAt5)} | ${format4(result.mrr)} | ${format4(result.ndcgAt10)} | ${format4(result.evidencePrecisionAt5)} | ${result.chunkCount} (${formatDelta(delta(result.chunkCount, baseline.chunkCount))}) | ${result.indexingMs} ms | ${result.latencyP95Ms?.toFixed(2)} ms |` + return `| ${result.label} | \`${args}\` | ${format4(result.recallAt1)} | ${format4(result.recallAt5)} | ${format4(result.mrr)} | ${format4(result.ndcgAt10)} | ${format4(result.contextPrecision)} | ${result.chunkCount} (${formatDelta(delta(result.chunkCount, baseline.chunkCount))}) | ${result.indexingMs} ms | ${result.latencyP95Ms?.toFixed(2)} ms |` }) /** @@ -215,7 +215,7 @@ questions as \`baseline-v1.4.json\`, with dense retrieval held fixed. Only the c configuration changes, so a difference in the metrics is a difference in the input distribution retrieval is measured on. -| Variant | size/overlap | Recall@1 | Recall@5 | MRR | nDCG@10 | Evidence P@5 | Index size (Δ) | Indexing | Query p95 | +| Variant | size/overlap | Recall@1 | Recall@5 | MRR | nDCG@10 | Context P | Index size (Δ) | Indexing | Query p95 | | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | ${rows.join('\n')} diff --git a/scripts/eval-retrieval.mjs b/scripts/eval-retrieval.mjs index d52b5bd..5964a28 100644 --- a/scripts/eval-retrieval.mjs +++ b/scripts/eval-retrieval.mjs @@ -141,7 +141,7 @@ const format4 = (value) => value.toFixed(4) const rows = results.map( (result) => - `| ${result.label} | ${format4(result.recallAt1)} | ${format4(result.recallAt5)} | ${format4(result.mrr)} | ${format4(result.ndcgAt10)} | ${format4(result.mapAt10)} | ${format4(result.evidencePrecisionAt5)} | ${result.latencyP95Ms?.toFixed(2)} ms |` + `| ${result.label} | ${format4(result.recallAt1)} | ${format4(result.recallAt5)} | ${format4(result.mrr)} | ${format4(result.ndcgAt10)} | ${format4(result.mapAt10)} | ${format4(result.contextPrecision)} | ${result.latencyP95Ms?.toFixed(2)} ms |` ) /** @@ -179,7 +179,7 @@ Generated by \`node scripts/eval-retrieval.mjs\`. Numbers are harness output; do Every strategy runs the real RAG eval harness against the same corpus and the same 30 questions as \`baseline-v1.6.json\`, with chunking held fixed at ${baseline.chunking}. Only the retrieval strategy changes. -| Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Evidence P@5 | Query p95 | +| Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Context P | Query p95 | | --- | --- | --- | --- | --- | --- | --- | --- | ${rows.join('\n')} diff --git a/src/main/eval/harness.ts b/src/main/eval/harness.ts index 7ae5114..c33c43d 100644 --- a/src/main/eval/harness.ts +++ b/src/main/eval/harness.ts @@ -67,8 +67,6 @@ export interface EvalHarnessOptions { contextK: number /** Similarity floor; 0 keeps the ranking intact for ranking metrics. */ threshold: number - /** How many retrieved passages the evidence-precision metric looks at. */ - evidenceK: number /** 分块配置(#78)。实验变体通过它选择策略;缺省时用生产默认值。 */ chunkOptions: ChunkOptions /** 检索策略(#77):dense / sparse(BM25) / hybrid(RRF)。 */ @@ -84,7 +82,7 @@ const UNTAGGED = 'untagged' * 同一套指标既算总平均,也算每个查询类别(#192)。用一个函数是因为分组平均必须与 * 总平均是同一个定义,否则两个数就不可比。 */ -function summarize(perQuestion: readonly QuestionReport[], evidenceK: number): EvalMetrics { +function summarize(perQuestion: readonly QuestionReport[], contextK: number): EvalMetrics { return { recallAt1: mean(perQuestion.map((q) => recallAtK(q.matchesByRank, q.relevantCount, 1))), recallAt5: mean(perQuestion.map((q) => recallAtK(q.matchesByRank, q.relevantCount, 5))), @@ -95,8 +93,13 @@ function summarize(perQuestion: readonly QuestionReport[], evidenceK: number): E mapAt10: mean( perQuestion.map((q) => averagePrecisionAtK(q.matchesByRank, q.relevantCount, 10)) ), - evidencePrecisionAt5: mean( - perQuestion.map((q) => evidencePrecisionAtK(q.matchesByRank, evidenceK)) + // 两个 context 指标共用同一个窗口,因为它们回答的是同一个问题的两面:送进 prompt + // 的那几条里有多少是相关的,以及需要的东西有多少真的进去了。 + contextPrecision: mean( + perQuestion.map((q) => evidencePrecisionAtK(q.matchesByRank, contextK)) + ), + contextRecall: mean( + perQuestion.map((q) => recallAtK(q.matchesByRank, q.relevantCount, contextK)) ) } } @@ -266,14 +269,14 @@ export async function runEvalHarness( }) } - const metrics = summarize(perQuestion, options.evidenceK) + const metrics = summarize(perQuestion, options.contextK) // 每个类别一行,按类别名排序,所以同一个 JSON 在两次运行之间可 diff。 const byType: EvalTypeBreakdown[] = [...new Set(perQuestion.map((q) => q.type))] .sort() .map((type) => { const group = perQuestion.filter((q) => q.type === type) - return { type, questions: group.length, metrics: summarize(group, options.evidenceK) } + return { type, questions: group.length, metrics: summarize(group, options.contextK) } }) const chunking = { ...DEFAULT_CHUNK_OPTIONS, ...options.chunkOptions } @@ -294,7 +297,6 @@ export async function runEvalHarness( candidateK: options.candidateK, contextK: options.contextK, threshold: options.threshold, - evidenceK: options.evidenceK, corpus: options.corpusLabel, documents: documentIds.size, questions: questions.length, @@ -323,7 +325,8 @@ function roundMetrics(metrics: EvalMetrics): EvalMetrics { ndcgAt10: roundMetric(metrics.ndcgAt10), hitRateAt5: roundMetric(metrics.hitRateAt5), mapAt10: roundMetric(metrics.mapAt10), - evidencePrecisionAt5: roundMetric(metrics.evidencePrecisionAt5) + contextPrecision: roundMetric(metrics.contextPrecision), + contextRecall: roundMetric(metrics.contextRecall) } } diff --git a/src/main/eval/report.ts b/src/main/eval/report.ts index 7b883f8..8f6dfaa 100644 --- a/src/main/eval/report.ts +++ b/src/main/eval/report.ts @@ -37,7 +37,6 @@ Generated by \`${report.generatedBy}\`. The numbers below are harness output — | Retrieval | \`${config.retrieval}\` | | Ranks | \`candidateK=${config.candidateK}, threshold=${config.threshold}\` | | Context width | \`contextK=${config.contextK}\` | -| Evidence per query | \`evidenceK=${config.evidenceK}\` | | Corpus | \`${config.corpus}\` (${config.documents} documents) | | Split | \`${config.split}\` (${config.questions} questions) | | Index size | ${config.chunkCount} chunks | @@ -53,7 +52,8 @@ Generated by \`${report.generatedBy}\`. The numbers below are harness output — | nDCG@10 | ${format(metrics.ndcgAt10)} | | Hit rate@5 | ${format(metrics.hitRateAt5)} | | MAP@10 | ${format(metrics.mapAt10)} | -| Evidence precision@${config.evidenceK} | ${format(metrics.evidencePrecisionAt5)} | +| Context precision@${config.contextK} | ${format(metrics.contextPrecision)} | +| Context recall@${config.contextK} | ${format(metrics.contextRecall)} | ### By query type @@ -74,11 +74,17 @@ corpus, so they must never be the reason two runs differ. - A retrieved passage is relevant when its provenance covers a ground-truth block. - **Recall@k** is the share of ground-truth blocks covered by the first \`k\` passages. -- **Evidence precision@${config.evidenceK}** is the share of the first \`${config.evidenceK}\` - retrieved passages that cover a ground-truth block. This is **retrieval precision**, not - answer citation recall: the harness runs no model and produces no answer. Answer-level - citation correctness is covered by the resolver (#70); a model-driven answer eval is a - separate deliverable. +- **Context precision@${config.contextK}** is the share of the first \`${config.contextK}\` + retrieved passages that cover a ground-truth block. **Context recall@${config.contextK}** is + the share of the needed ground-truth blocks that made it into that same window. Both are + deterministic: the dataset says which blocks answer the question, so no model is needed to + score the window. Together they are the trade-off a \`contextK\` decision actually makes — + a wider window finds more and carries more noise. +- This is **retrieval precision/recall, not answer citation recall**: the harness runs no + model and produces no answer. Answer-level citation correctness is covered by the + resolver (#70). Faithfulness, completeness and answer correctness need a generative model + and are **not evaluated here** — the harness runs offline with only the pinned embedding + model, the same constraint that keeps the reranker unmeasured (#170). - Ground truth is expressed in corpus identity (\`document\` relative path + \`block\` ordinal + optional \`quote\`), never a runtime \`documentId\`/\`blockId\`. diff --git a/src/main/eval/run.ts b/src/main/eval/run.ts index 802cf5e..d927286 100644 --- a/src/main/eval/run.ts +++ b/src/main/eval/run.ts @@ -197,7 +197,6 @@ export async function runEvalCli(argv: readonly string[] = process.argv): Promis candidateK: readNumberOption(argv, '--eval-candidate-k=', 20), contextK: readNumberOption(argv, '--eval-context-k=', 3), threshold: readNumberOption(argv, '--eval-threshold=', 0.5), - evidenceK: 5, chunkOptions: readChunkOptions(argv), strategy: readRetrievalStrategy(argv) } diff --git a/src/main/eval/types.ts b/src/main/eval/types.ts index 3a075f4..8833a23 100644 --- a/src/main/eval/types.ts +++ b/src/main/eval/types.ts @@ -82,13 +82,19 @@ export interface EvalMetrics { /** AP@10:把「排序位置」和「覆盖面」合成一个数的那个指标。 */ mapAt10: number /** - * Share of the first `evidenceK` retrieved passages that cover ground truth. + * 前 `contextK` 条证据里真的命中 ground-truth 的比例(context 条的精确率)。 * - * This is **retrieval precision**, not answer citation recall: no model runs in - * this harness and no answer is produced. Answer-level citation correctness is - * the resolver's job (#70) and would need a separate, model-driven eval. + * 这是**检索精度**,不是回答的引用召回:harness 不跑模型、不产生回答。回答层的引用 + * 正确性是 resolver 的事(#70),需要一个真正跑模型的 eval。 */ - evidencePrecisionAt5: number + contextPrecision: number + /** + * 答案需要的 ground-truth 块有多少进了 `contextK` 宽的窗口。 + * + * 与 Recall@10 的区别在于它量的是**窗口**:证据排在第 4、而 contextK=3 时,模型 + * 看不到它。这是一个产品指标,不只是检索指标。 + */ + contextRecall: number } /** 按查询类别聚合的同一套指标(#192)。总平均会掩盖方向相反的两个变化。 */ @@ -134,13 +140,11 @@ export interface EvalReport { /** * 生产 prompt 实际取用的证据条数(#77)。 * - * 快照里记它是为了让 benchmark 描述整条线上链路,而不只是检索器;它不影响排名 - * 指标 —— 截断只是取候选列表的前缀,前缀的排序不变。 + * 快照里记它是为了让 benchmark 描述整条线上链路,而不只是检索器;它也是两个 + * context 指标的窗口宽度。 */ contextK: number threshold: number - /** How many retrieved passages the evidence-precision metric looks at. */ - evidenceK: number corpus: string documents: number questions: number