diff --git a/docs/eval/baseline-v1.6.json b/docs/eval/baseline-v1.6.json index e087046..d8171a1 100644 --- a/docs/eval/baseline-v1.6.json +++ b/docs/eval/baseline-v1.6.json @@ -16,20 +16,20 @@ "contextK": 3, "threshold": 0.5, "corpus": "eval/corpus", - "documents": 13, - "questions": 44, - "chunkCount": 19 + "documents": 21, + "questions": 63, + "chunkCount": 53 }, "metrics": { - "recallAt1": 0.657895, - "recallAt5": 0.921053, - "recallAt10": 1, - "mrr": 0.800909, - "ndcgAt10": 0.84764, - "hitRateAt5": 0.921053, - "mapAt10": 0.796523, - "contextPrecision": 0.324561, - "contextRecall": 0.921053 + "recallAt1": 0.59434, + "recallAt5": 0.877358, + "recallAt10": 0.919811, + "mrr": 0.754755, + "ndcgAt10": 0.780449, + "hitRateAt5": 0.90566, + "mapAt10": 0.72804, + "contextPrecision": 0.308176, + "contextRecall": 0.816038 }, "byType": [ { @@ -38,58 +38,73 @@ "metrics": { "recallAt1": 0, "recallAt5": 0.666667, - "recallAt10": 1, - "mrr": 0.344577, - "ndcgAt10": 0.503192, + "recallAt10": 0.777778, + "mrr": 0.299383, + "ndcgAt10": 0.41727, "hitRateAt5": 0.666667, - "mapAt10": 0.344577, - "contextPrecision": 0.222222, - "contextRecall": 0.666667 + "mapAt10": 0.299383, + "contextPrecision": 0.185185, + "contextRecall": 0.555556 } }, { "type": "exact", - "questions": 6, + "questions": 8, "metrics": { - "recallAt1": 1, + "recallAt1": 0.75, "recallAt5": 1, "recallAt10": 1, - "mrr": 1, - "ndcgAt10": 1, + "mrr": 0.822917, + "ndcgAt10": 0.866335, "hitRateAt5": 1, - "mapAt10": 1, - "contextPrecision": 0.333333, - "contextRecall": 1 + "mapAt10": 0.822917, + "contextPrecision": 0.291667, + "contextRecall": 0.875 } }, { - "type": "multi-hop", - "questions": 2, + "type": "hard-negative", + "questions": 7, "metrics": { - "recallAt1": 0.5, + "recallAt1": 0.714286, "recallAt5": 1, "recallAt10": 1, - "mrr": 1, - "ndcgAt10": 0.95986, + "mrr": 0.833333, + "ndcgAt10": 0.875847, "hitRateAt5": 1, - "mapAt10": 0.916667, - "contextPrecision": 0.666667, + "mapAt10": 0.833333, + "contextPrecision": 0.333333, "contextRecall": 1 } }, { - "type": "semantic", - "questions": 18, + "type": "multi-hop", + "questions": 5, "metrics": { - "recallAt1": 0.833333, - "recallAt5": 1, - "recallAt10": 1, - "mrr": 0.907407, - "ndcgAt10": 0.931214, + "recallAt1": 0.3, + "recallAt5": 0.7, + "recallAt10": 0.75, + "mrr": 0.9, + "ndcgAt10": 0.721795, "hitRateAt5": 1, - "mapAt10": 0.907407, - "contextPrecision": 0.333333, - "contextRecall": 1 + "mapAt10": 0.635, + "contextPrecision": 0.6, + "contextRecall": 0.65 + } + }, + { + "type": "semantic", + "questions": 21, + "metrics": { + "recallAt1": 0.761905, + "recallAt5": 0.904762, + "recallAt10": 0.952381, + "mrr": 0.828139, + "ndcgAt10": 0.85418, + "hitRateAt5": 0.904762, + "mapAt10": 0.82381, + "contextPrecision": 0.285714, + "contextRecall": 0.857143 } }, { @@ -109,10 +124,10 @@ } ], "unanswerable": { - "questions": 6, + "questions": 10, "abstentionCount": 0, "retrievalAbstentionRate": 0, - "meanCandidatesRetrieved": 19, + "meanCandidatesRetrieved": 20, "meanContextPassages": 3 }, "perQuestion": [ @@ -123,8 +138,8 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, - "contextChars": 2074, + "retrievedCount": 20, + "contextChars": 2198, "matchesByRank": [ [ 0 @@ -146,6 +161,7 @@ [], [], [], + [], [] ] }, @@ -154,11 +170,13 @@ "question": "How many replicate samples are collected at each river station?", "type": "exact", "answerable": true, - "firstRelevantRank": 1, + "firstRelevantRank": 3, "relevantCount": 1, - "retrievedCount": 19, - "contextChars": 2257, + "retrievedCount": 20, + "contextChars": 2297, "matchesByRank": [ + [], + [], [ 0 ], @@ -178,7 +196,6 @@ [], [], [], - [], [] ] }, @@ -189,8 +206,8 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, - "contextChars": 2558, + "retrievedCount": 20, + "contextChars": 2679, "matchesByRank": [ [ 0 @@ -212,6 +229,7 @@ [], [], [], + [], [] ] }, @@ -222,7 +240,7 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, "contextChars": 2113, "matchesByRank": [ [ @@ -245,6 +263,7 @@ [], [], [], + [], [] ] }, @@ -255,8 +274,8 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, - "contextChars": 2113, + "retrievedCount": 20, + "contextChars": 2157, "matchesByRank": [ [ 0 @@ -278,6 +297,7 @@ [], [], [], + [], [] ] }, @@ -288,8 +308,8 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, - "contextChars": 1775, + "retrievedCount": 20, + "contextChars": 2364, "matchesByRank": [ [ 0 @@ -311,6 +331,7 @@ [], [], [], + [], [] ] }, @@ -321,8 +342,8 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, - "contextChars": 2091, + "retrievedCount": 20, + "contextChars": 2047, "matchesByRank": [ [ 0 @@ -344,6 +365,7 @@ [], [], [], + [], [] ] }, @@ -354,8 +376,8 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, - "contextChars": 1970, + "retrievedCount": 20, + "contextChars": 2057, "matchesByRank": [ [ 0 @@ -377,6 +399,7 @@ [], [], [], + [], [] ] }, @@ -387,8 +410,8 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, - "contextChars": 1484, + "retrievedCount": 20, + "contextChars": 2136, "matchesByRank": [ [ 0 @@ -410,6 +433,7 @@ [], [], [], + [], [] ] }, @@ -420,7 +444,7 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, "contextChars": 2682, "matchesByRank": [ [ @@ -443,6 +467,7 @@ [], [], [], + [], [] ] }, @@ -453,7 +478,7 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, "contextChars": 2051, "matchesByRank": [ [ @@ -476,6 +501,7 @@ [], [], [], + [], [] ] }, @@ -486,8 +512,8 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, - "contextChars": 2007, + "retrievedCount": 20, + "contextChars": 2687, "matchesByRank": [ [ 0 @@ -509,6 +535,7 @@ [], [], [], + [], [] ] }, @@ -517,16 +544,13 @@ "question": "What is the main environmental concern for tidal energy installations?", "type": "semantic", "answerable": true, - "firstRelevantRank": 3, + "firstRelevantRank": 11, "relevantCount": 1, - "retrievedCount": 19, - "contextChars": 2007, + "retrievedCount": 20, + "contextChars": 2579, "matchesByRank": [ [], [], - [ - 0 - ], [], [], [], @@ -535,6 +559,10 @@ [], [], [], + [ + 0 + ], + [], [], [], [], @@ -552,8 +580,8 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, - "contextChars": 2257, + "retrievedCount": 20, + "contextChars": 2910, "matchesByRank": [ [ 0 @@ -575,6 +603,7 @@ [], [], [], + [], [] ] }, @@ -585,8 +614,8 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, - "contextChars": 2621, + "retrievedCount": 20, + "contextChars": 2884, "matchesByRank": [ [ 0 @@ -608,6 +637,7 @@ [], [], [], + [], [] ] }, @@ -618,8 +648,8 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, - "contextChars": 2558, + "retrievedCount": 20, + "contextChars": 2702, "matchesByRank": [ [ 0 @@ -641,6 +671,7 @@ [], [], [], + [], [] ] }, @@ -651,7 +682,7 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, "contextChars": 1773, "matchesByRank": [ [ @@ -674,6 +705,7 @@ [], [], [], + [], [] ] }, @@ -684,8 +716,8 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, - "contextChars": 1970, + "retrievedCount": 20, + "contextChars": 2561, "matchesByRank": [ [ 0 @@ -707,6 +739,7 @@ [], [], [], + [], [] ] }, @@ -717,8 +750,8 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, - "contextChars": 2556, + "retrievedCount": 20, + "contextChars": 2636, "matchesByRank": [ [ 0 @@ -740,6 +773,7 @@ [], [], [], + [], [] ] }, @@ -750,8 +784,8 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, - "contextChars": 2098, + "retrievedCount": 20, + "contextChars": 2269, "matchesByRank": [ [ 0 @@ -773,6 +807,7 @@ [], [], [], + [], [] ] }, @@ -783,8 +818,8 @@ "answerable": true, "firstRelevantRank": 2, "relevantCount": 1, - "retrievedCount": 19, - "contextChars": 2488, + "retrievedCount": 20, + "contextChars": 2580, "matchesByRank": [ [], [ @@ -806,6 +841,7 @@ [], [], [], + [], [] ] }, @@ -816,8 +852,8 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, - "contextChars": 2488, + "retrievedCount": 20, + "contextChars": 2580, "matchesByRank": [ [ 0 @@ -839,6 +875,7 @@ [], [], [], + [], [] ] }, @@ -849,16 +886,13 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 2, - "retrievedCount": 19, - "contextChars": 2257, + "retrievedCount": 20, + "contextChars": 2912, "matchesByRank": [ [ 1 ], [], - [ - 0 - ], [], [], [], @@ -874,6 +908,10 @@ [], [], [], + [], + [ + 0 + ], [] ] }, @@ -884,7 +922,7 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 2, - "retrievedCount": 19, + "retrievedCount": 20, "contextChars": 1970, "matchesByRank": [ [ @@ -909,6 +947,7 @@ [], [], [], + [], [] ] }, @@ -919,8 +958,8 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, - "contextChars": 2098, + "retrievedCount": 20, + "contextChars": 1373, "matchesByRank": [ [ 0 @@ -942,6 +981,7 @@ [], [], [], + [], [] ] }, @@ -952,7 +992,7 @@ "answerable": true, "firstRelevantRank": 2, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, "contextChars": 1867, "matchesByRank": [ [], @@ -975,6 +1015,7 @@ [], [], [], + [], [] ] }, @@ -985,7 +1026,7 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, "contextChars": 1453, "matchesByRank": [ [ @@ -1008,6 +1049,7 @@ [], [], [], + [], [] ] }, @@ -1018,8 +1060,8 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, - "contextChars": 1372, + "retrievedCount": 20, + "contextChars": 1376, "matchesByRank": [ [ 0 @@ -1041,6 +1083,7 @@ [], [], [], + [], [] ] }, @@ -1051,8 +1094,8 @@ "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, - "retrievedCount": 19, - "contextChars": 1500, + "retrievedCount": 20, + "contextChars": 1294, "matchesByRank": [ [ 0 @@ -1074,6 +1117,7 @@ [], [], [], + [], [] ] }, @@ -1084,8 +1128,8 @@ "answerable": true, "firstRelevantRank": 2, "relevantCount": 1, - "retrievedCount": 19, - "contextChars": 1860, + "retrievedCount": 20, + "contextChars": 1955, "matchesByRank": [ [], [ @@ -1107,6 +1151,7 @@ [], [], [], + [], [] ] }, @@ -1117,8 +1162,8 @@ "answerable": false, "firstRelevantRank": 0, "relevantCount": 0, - "retrievedCount": 19, - "contextChars": 2488, + "retrievedCount": 20, + "contextChars": 2545, "matchesByRank": [ [], [], @@ -1138,6 +1183,7 @@ [], [], [], + [], [] ] }, @@ -1148,8 +1194,8 @@ "answerable": false, "firstRelevantRank": 0, "relevantCount": 0, - "retrievedCount": 19, - "contextChars": 2257, + "retrievedCount": 20, + "contextChars": 2789, "matchesByRank": [ [], [], @@ -1169,6 +1215,7 @@ [], [], [], + [], [] ] }, @@ -1179,8 +1226,8 @@ "answerable": false, "firstRelevantRank": 0, "relevantCount": 0, - "retrievedCount": 19, - "contextChars": 2043, + "retrievedCount": 20, + "contextChars": 2755, "matchesByRank": [ [], [], @@ -1200,6 +1247,7 @@ [], [], [], + [], [] ] }, @@ -1210,8 +1258,8 @@ "answerable": false, "firstRelevantRank": 0, "relevantCount": 0, - "retrievedCount": 19, - "contextChars": 2588, + "retrievedCount": 20, + "contextChars": 2679, "matchesByRank": [ [], [], @@ -1231,6 +1279,7 @@ [], [], [], + [], [] ] }, @@ -1241,8 +1290,8 @@ "answerable": false, "firstRelevantRank": 0, "relevantCount": 0, - "retrievedCount": 19, - "contextChars": 2488, + "retrievedCount": 20, + "contextChars": 2687, "matchesByRank": [ [], [], @@ -1262,6 +1311,7 @@ [], [], [], + [], [] ] }, @@ -1272,8 +1322,8 @@ "answerable": false, "firstRelevantRank": 0, "relevantCount": 0, - "retrievedCount": 19, - "contextChars": 2071, + "retrievedCount": 20, + "contextChars": 2819, "matchesByRank": [ [], [], @@ -1293,6 +1343,7 @@ [], [], [], + [], [] ] }, @@ -1301,10 +1352,10 @@ "question": "推移质为什么比悬移质更难测?", "type": "cross-lingual", "answerable": true, - "firstRelevantRank": 8, + "firstRelevantRank": 0, "relevantCount": 1, - "retrievedCount": 19, - "contextChars": 2084, + "retrievedCount": 20, + "contextChars": 1986, "matchesByRank": [ [], [], @@ -1313,9 +1364,8 @@ [], [], [], - [ - 0 - ], + [], + [], [], [], [], @@ -1334,10 +1384,10 @@ "question": "河流监测中,每个站点要采几份平行样?", "type": "cross-lingual", "answerable": true, - "firstRelevantRank": 7, + "firstRelevantRank": 0, "relevantCount": 1, - "retrievedCount": 19, - "contextChars": 1480, + "retrievedCount": 20, + "contextChars": 1337, "matchesByRank": [ [], [], @@ -1345,9 +1395,8 @@ [], [], [], - [ - 0 - ], + [], + [], [], [], [], @@ -1369,7 +1418,7 @@ "answerable": true, "firstRelevantRank": 2, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, "contextChars": 1994, "matchesByRank": [ [], @@ -1392,6 +1441,7 @@ [], [], [], + [], [] ] }, @@ -1402,7 +1452,7 @@ "answerable": true, "firstRelevantRank": 2, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, "contextChars": 1994, "matchesByRank": [ [], @@ -1425,6 +1475,7 @@ [], [], [], + [], [] ] }, @@ -1433,11 +1484,12 @@ "question": "电芯隔膜一旦熔化会导致什么后果?", "type": "cross-lingual", "answerable": true, - "firstRelevantRank": 3, + "firstRelevantRank": 4, "relevantCount": 1, - "retrievedCount": 19, - "contextChars": 1449, + "retrievedCount": 20, + "contextChars": 1294, "matchesByRank": [ + [], [], [], [ @@ -1468,7 +1520,7 @@ "answerable": true, "firstRelevantRank": 3, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, "contextChars": 1321, "matchesByRank": [ [], @@ -1491,6 +1543,7 @@ [], [], [], + [], [] ] }, @@ -1501,8 +1554,8 @@ "answerable": true, "firstRelevantRank": 2, "relevantCount": 1, - "retrievedCount": 19, - "contextChars": 956, + "retrievedCount": 20, + "contextChars": 801, "matchesByRank": [ [], [ @@ -1524,6 +1577,7 @@ [], [], [], + [], [] ] }, @@ -1532,14 +1586,181 @@ "question": "为什么成片的树冠比孤立的树降温效果更好?", "type": "cross-lingual", "answerable": true, - "firstRelevantRank": 6, + "firstRelevantRank": 9, "relevantCount": 1, - "retrievedCount": 19, + "retrievedCount": 20, "contextChars": 1447, "matchesByRank": [ [], [], [], + [], + [], + [], + [], + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q045", + "question": "What is the default retry count in Gateway API v2?", + "type": "exact", + "answerable": true, + "firstRelevantRank": 4, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 2898, + "matchesByRank": [ + [], + [], + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q046", + "question": "What is the default request timeout of Gateway API v1?", + "type": "hard-negative", + "answerable": true, + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 2909, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q047", + "question": "What is the context window of Gateway API v3, in tokens?", + "type": "hard-negative", + "answerable": true, + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 2899, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q048", + "question": "What is the default rate limit of Gateway API v2?", + "type": "hard-negative", + "answerable": true, + "firstRelevantRank": 2, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 2854, + "matchesByRank": [ + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q049", + "question": "What is the default request timeout of Gateway API v3?", + "type": "hard-negative", + "answerable": true, + "firstRelevantRank": 3, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 2921, + "matchesByRank": [ [], [], [ @@ -1557,6 +1778,503 @@ [], [], [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q050", + "question": "Which Gateway API version waits longest between retries after a failed request?", + "type": "semantic", + "answerable": true, + "firstRelevantRank": 10, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 2910, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q051", + "question": "Compare the default timeouts and rate limits of Gateway API v1 and v3.", + "type": "multi-hop", + "answerable": true, + "firstRelevantRank": 1, + "relevantCount": 4, + "retrievedCount": 20, + "contextChars": 2888, + "matchesByRank": [ + [ + 0 + ], + [ + 2, + 3 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [ + 1 + ], + [ + 3 + ], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q052", + "question": "Compare the retry count and context window of Gateway API v2 and v3.", + "type": "multi-hop", + "answerable": true, + "firstRelevantRank": 2, + "relevantCount": 4, + "retrievedCount": 20, + "contextChars": 2899, + "matchesByRank": [ + [], + [ + 3 + ], + [], + [], + [], + [], + [], + [], + [], + [ + 1 + ], + [ + 2 + ], + [ + 0 + ], + [], + [ + 1 + ], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q053", + "question": "How much GPU memory does Gateway API v2 require to serve a request?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2875, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q054", + "question": "What is the monthly subscription price of Gateway API v3?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2908, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q055", + "question": "How many replicate samples are collected at each reservoir monitoring station?", + "type": "exact", + "answerable": true, + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 2819, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q056", + "question": "How many replicate samples are collected at each estuary transect station?", + "type": "hard-negative", + "answerable": true, + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 2835, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q057", + "question": "At what depth below the surface does the coastal programme place its shore-station sensor?", + "type": "hard-negative", + "answerable": true, + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 2859, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q058", + "question": "At what depth below the water table do groundwater boreholes carry their pressure transducer?", + "type": "hard-negative", + "answerable": true, + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 2876, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q059", + "question": "Which monitoring programme samples most frequently?", + "type": "semantic", + "answerable": true, + "firstRelevantRank": 5, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 2933, + "matchesByRank": [ + [], + [], + [], + [], + [ + 0 + ], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q060", + "question": "Compare the sampling interval and replicate count of the reservoir and groundwater programmes.", + "type": "multi-hop", + "answerable": true, + "firstRelevantRank": 1, + "relevantCount": 4, + "retrievedCount": 20, + "contextChars": 2879, + "matchesByRank": [ + [ + 2 + ], + [ + 2, + 3 + ], + [ + 0 + ], + [ + 0, + 1 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q061", + "question": "Why does the estuary programme take more replicate samples than the other monitoring programmes?", + "type": "semantic", + "answerable": true, + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 20, + "contextChars": 2831, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q062", + "question": "What is the annual operating cost of the coastal monitoring buoys?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2826, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q063", + "question": "How many litres per second does the reservoir release downstream on a typical day?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2809, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], [] ] } diff --git a/docs/eval/baseline-v1.6.md b/docs/eval/baseline-v1.6.md index 1484d95..65b9366 100644 --- a/docs/eval/baseline-v1.6.md +++ b/docs/eval/baseline-v1.6.md @@ -11,23 +11,23 @@ Generated by `npm run eval`. The numbers below are harness output — do not edi | Retrieval | `dense` | | Ranks | `candidateK=20, threshold=0.5` | | Context width | `contextK=3` | -| Corpus | `eval/corpus` (13 documents) | -| Split | `all` (44 questions, 38 answerable) | -| Index size | 19 chunks | +| Corpus | `eval/corpus` (21 documents) | +| Split | `all` (63 questions, 53 answerable) | +| Index size | 53 chunks | ## Metrics | Metric | Value | | --- | --- | -| Recall@1 | 0.6579 | -| Recall@5 | 0.9211 | -| Recall@10 | 1.0000 | -| MRR | 0.8009 | -| nDCG@10 | 0.8476 | -| Hit rate@5 | 0.9211 | -| MAP@10 | 0.7965 | -| Context precision@3 | 0.3246 | -| Context recall@3 | 0.9211 | +| Recall@1 | 0.5943 | +| Recall@5 | 0.8774 | +| Recall@10 | 0.9198 | +| MRR | 0.7548 | +| nDCG@10 | 0.7804 | +| Hit rate@5 | 0.9057 | +| MAP@10 | 0.7280 | +| Context precision@3 | 0.3082 | +| Context recall@3 | 0.8160 | ### By query type @@ -37,10 +37,11 @@ The type comes from `type` in `questions.jsonl`; untagged questions report as | Type | Questions | Recall@5 | nDCG@10 | Hit rate@5 | MAP@10 | | --- | --- | --- | --- | --- | --- | -| cross-lingual | 9 | 0.6667 | 0.5032 | 0.6667 | 0.3446 | -| exact | 6 | 1.0000 | 1.0000 | 1.0000 | 1.0000 | -| multi-hop | 2 | 1.0000 | 0.9599 | 1.0000 | 0.9167 | -| semantic | 18 | 1.0000 | 0.9312 | 1.0000 | 0.9074 | +| cross-lingual | 9 | 0.6667 | 0.4173 | 0.6667 | 0.2994 | +| exact | 8 | 1.0000 | 0.8663 | 1.0000 | 0.8229 | +| hard-negative | 7 | 1.0000 | 0.8758 | 1.0000 | 0.8333 | +| multi-hop | 5 | 0.7000 | 0.7218 | 1.0000 | 0.6350 | +| semantic | 21 | 0.9048 | 0.8542 | 0.9048 | 0.8238 | | zh | 3 | 1.0000 | 1.0000 | 1.0000 | 1.0000 | ### Unanswerable questions @@ -62,13 +63,13 @@ with a small second one means the threshold filters nothing and the window is al | Metric | Value | | --- | --- | -| Unanswerable questions | 6 | -| Retrieval abstained | 0.0000 (0/6) | -| Mean candidates passing the threshold | 19.00 | +| Unanswerable questions | 10 | +| Retrieval abstained | 0.0000 (0/10) | +| Mean candidates passing the threshold | 20.00 | | Mean passages in the context window | 3.00 | -Timing is informational only and is **not** frozen: indexing 1583 ms, query -p50 13.38 ms, p95 16.66 ms on the +Timing is informational only and is **not** frozen: indexing 2796 ms, query +p50 13.67 ms, p95 16.34 ms on the machine that produced this file. Timing and index size depend on hardware and on the corpus, so they must never be the reason two runs differ. diff --git a/docs/eval/retrieval-v1.6.json b/docs/eval/retrieval-v1.6.json index 62b9139..d94d41f 100644 --- a/docs/eval/retrieval-v1.6.json +++ b/docs/eval/retrieval-v1.6.json @@ -15,58 +15,58 @@ "id": "dense", "label": "dense (vector)", "split": "validation", - "questions": 16, - "answerableCount": 13, + "questions": 24, + "answerableCount": 19, "chunking": "1000/100", - "chunkCount": 19, - "recallAt1": 0.576923, - "recallAt5": 0.923077, - "recallAt10": 1, - "mrr": 0.778846, - "ndcgAt10": 0.827608, - "hitRateAt5": 0.923077, - "mapAt10": 0.766026, - "contextPrecision": 0.333333, - "contextRecall": 0.923077, - "latencyP95Ms": 16.02 + "chunkCount": 53, + "recallAt1": 0.513158, + "recallAt5": 0.868421, + "recallAt10": 0.921053, + "mrr": 0.720175, + "ndcgAt10": 0.755611, + "hitRateAt5": 0.894737, + "mapAt10": 0.69386, + "contextPrecision": 0.315789, + "contextRecall": 0.802632, + "latencyP95Ms": 13.27 }, { "id": "sparse", "label": "sparse (BM25)", "split": "validation", - "questions": 16, - "answerableCount": 13, + "questions": 24, + "answerableCount": 19, "chunking": "1000/100", - "chunkCount": 19, - "recallAt1": 0.576923, - "recallAt5": 0.615385, - "recallAt10": 0.615385, - "mrr": 0.615385, - "ndcgAt10": 0.609209, - "hitRateAt5": 0.615385, - "mapAt10": 0.602564, - "contextPrecision": 0.230769, - "contextRecall": 0.615385, - "latencyP95Ms": 2.96 + "chunkCount": 53, + "recallAt1": 0.460526, + "recallAt5": 0.671053, + "recallAt10": 0.684211, + "mrr": 0.600877, + "ndcgAt10": 0.610495, + "hitRateAt5": 0.684211, + "mapAt10": 0.576817, + "contextPrecision": 0.263158, + "contextRecall": 0.657895, + "latencyP95Ms": 2.26 }, { "id": "hybrid", "label": "hybrid (RRF of dense + BM25)", "split": "validation", - "questions": 16, - "answerableCount": 13, + "questions": 24, + "answerableCount": 19, "chunking": "1000/100", - "chunkCount": 19, - "recallAt1": 0.576923, - "recallAt5": 0.923077, - "recallAt10": 1, - "mrr": 0.778846, - "ndcgAt10": 0.827608, - "hitRateAt5": 0.923077, - "mapAt10": 0.766026, + "chunkCount": 53, + "recallAt1": 0.513158, + "recallAt5": 0.868421, + "recallAt10": 0.894737, + "mrr": 0.732456, + "ndcgAt10": 0.760265, + "hitRateAt5": 0.894737, + "mapAt10": 0.707018, "contextPrecision": 0.333333, - "contextRecall": 0.923077, - "latencyP95Ms": 19.47 + "contextRecall": 0.855263, + "latencyP95Ms": 13.61 } ], "test": [ @@ -74,20 +74,20 @@ "id": "dense", "label": "dense (vector)", "split": "test", - "questions": 28, - "answerableCount": 25, + "questions": 39, + "answerableCount": 34, "chunking": "1000/100", - "chunkCount": 19, - "recallAt1": 0.7, - "recallAt5": 0.92, - "recallAt10": 1, - "mrr": 0.812381, - "ndcgAt10": 0.858056, - "hitRateAt5": 0.92, - "mapAt10": 0.812381, - "contextPrecision": 0.32, - "contextRecall": 0.92, - "latencyP95Ms": 14.49 + "chunkCount": 53, + "recallAt1": 0.639706, + "recallAt5": 0.882353, + "recallAt10": 0.919118, + "mrr": 0.774079, + "ndcgAt10": 0.794329, + "hitRateAt5": 0.911765, + "mapAt10": 0.747141, + "contextPrecision": 0.303922, + "contextRecall": 0.823529, + "latencyP95Ms": 12.89 } ] } diff --git a/docs/eval/retrieval-v1.6.md b/docs/eval/retrieval-v1.6.md index 9d5009a..b34bdf6 100644 --- a/docs/eval/retrieval-v1.6.md +++ b/docs/eval/retrieval-v1.6.md @@ -16,15 +16,15 @@ to. | Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Context P | n | p95 | | --- | --- | --- | --- | --- | --- | --- | --- | --- | -| dense (vector) | 0.5769 | 0.9231 | 0.7788 | 0.8276 | 0.7660 | 0.3333 | 16 | 16.02 ms | -| sparse (BM25) | 0.5769 | 0.6154 | 0.6154 | 0.6092 | 0.6026 | 0.2308 | 16 | 2.96 ms | -| hybrid (RRF of dense + BM25) | 0.5769 | 0.9231 | 0.7788 | 0.8276 | 0.7660 | 0.3333 | 16 | 19.47 ms | +| dense (vector) | 0.5132 | 0.8684 | 0.7202 | 0.7556 | 0.6939 | 0.3158 | 24 | 13.27 ms | +| sparse (BM25) | 0.4605 | 0.6711 | 0.6009 | 0.6105 | 0.5768 | 0.2632 | 24 | 2.26 ms | +| hybrid (RRF of dense + BM25) | 0.5132 | 0.8684 | 0.7325 | 0.7603 | 0.7070 | 0.3333 | 24 | 13.61 ms | ## Test — reported, not selected | Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Context P | n | p95 | | --- | --- | --- | --- | --- | --- | --- | --- | --- | -| dense (vector) | 0.7000 | 0.9200 | 0.8124 | 0.8581 | 0.8124 | 0.3200 | 28 | 14.49 ms | +| dense (vector) | 0.6397 | 0.8824 | 0.7741 | 0.7943 | 0.7471 | 0.3039 | 39 | 12.89 ms | ## Not evaluated diff --git a/docs/eval/sweep-v1.6.json b/docs/eval/sweep-v1.6.json index 985549e..0a47cf1 100644 --- a/docs/eval/sweep-v1.6.json +++ b/docs/eval/sweep-v1.6.json @@ -5,397 +5,397 @@ "strategy": "dense", "candidateK": 5, "contextK": 3, - "recallAt5": 0.921053, - "ndcgAt10": 0.821192, - "mapAt10": 0.785088, - "contextPrecision": 0.324561, - "contextRecall": 0.921053, + "recallAt5": 0.877358, + "ndcgAt10": 0.767186, + "mapAt10": 0.723113, + "contextPrecision": 0.308176, + "contextRecall": 0.816038, "noResultRate": 0, - "meanContextChars": 1976.1315789473683, - "unanswerableQuestions": 6, + "meanContextChars": 2311.811320754717, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 5, "unanswerableContextPassages": 3, - "chunkCount": 19, - "latencyP95Ms": 14.04 + "chunkCount": 53, + "latencyP95Ms": 12.3 }, { "strategy": "dense", "candidateK": 5, "contextK": 5, - "recallAt5": 0.921053, - "ndcgAt10": 0.821192, - "mapAt10": 0.785088, - "contextPrecision": 0.194737, - "contextRecall": 0.921053, + "recallAt5": 0.877358, + "ndcgAt10": 0.767186, + "mapAt10": 0.723113, + "contextPrecision": 0.2, + "contextRecall": 0.877358, "noResultRate": 0, - "meanContextChars": 3188.1315789473683, - "unanswerableQuestions": 6, + "meanContextChars": 4007.0377358490564, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 5, "unanswerableContextPassages": 5, - "chunkCount": 19, - "latencyP95Ms": 13.63 + "chunkCount": 53, + "latencyP95Ms": 12.37 }, { "strategy": "dense", "candidateK": 10, "contextK": 3, - "recallAt5": 0.921053, - "ndcgAt10": 0.84764, - "mapAt10": 0.796523, - "contextPrecision": 0.324561, - "contextRecall": 0.921053, + "recallAt5": 0.877358, + "ndcgAt10": 0.780449, + "mapAt10": 0.72804, + "contextPrecision": 0.308176, + "contextRecall": 0.816038, "noResultRate": 0, - "meanContextChars": 1976.1315789473683, - "unanswerableQuestions": 6, + "meanContextChars": 2311.811320754717, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 10, "unanswerableContextPassages": 3, - "chunkCount": 19, - "latencyP95Ms": 13.84 + "chunkCount": 53, + "latencyP95Ms": 12.66 }, { "strategy": "dense", "candidateK": 10, "contextK": 5, - "recallAt5": 0.921053, - "ndcgAt10": 0.84764, - "mapAt10": 0.796523, - "contextPrecision": 0.194737, - "contextRecall": 0.921053, + "recallAt5": 0.877358, + "ndcgAt10": 0.780449, + "mapAt10": 0.72804, + "contextPrecision": 0.2, + "contextRecall": 0.877358, "noResultRate": 0, - "meanContextChars": 3188.1315789473683, - "unanswerableQuestions": 6, + "meanContextChars": 4007.0377358490564, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 10, "unanswerableContextPassages": 5, - "chunkCount": 19, - "latencyP95Ms": 15.21 + "chunkCount": 53, + "latencyP95Ms": 13.42 }, { "strategy": "dense", "candidateK": 10, "contextK": 8, - "recallAt5": 0.921053, - "ndcgAt10": 0.84764, - "mapAt10": 0.796523, - "contextPrecision": 0.131579, - "contextRecall": 1, + "recallAt5": 0.877358, + "ndcgAt10": 0.780449, + "mapAt10": 0.72804, + "contextPrecision": 0.127358, + "contextRecall": 0.877358, "noResultRate": 0, - "meanContextChars": 5151.289473684211, - "unanswerableQuestions": 6, + "meanContextChars": 6658.698113207547, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 10, "unanswerableContextPassages": 8, - "chunkCount": 19, - "latencyP95Ms": 16.35 + "chunkCount": 53, + "latencyP95Ms": 12.8 }, { "strategy": "dense", "candidateK": 20, "contextK": 3, - "recallAt5": 0.921053, - "ndcgAt10": 0.84764, - "mapAt10": 0.796523, - "contextPrecision": 0.324561, - "contextRecall": 0.921053, + "recallAt5": 0.877358, + "ndcgAt10": 0.780449, + "mapAt10": 0.72804, + "contextPrecision": 0.308176, + "contextRecall": 0.816038, "noResultRate": 0, - "meanContextChars": 1976.1315789473683, - "unanswerableQuestions": 6, + "meanContextChars": 2311.811320754717, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, - "unanswerableCandidates": 19, + "unanswerableCandidates": 20, "unanswerableContextPassages": 3, - "chunkCount": 19, - "latencyP95Ms": 16.45 + "chunkCount": 53, + "latencyP95Ms": 13.23 }, { "strategy": "dense", "candidateK": 20, "contextK": 5, - "recallAt5": 0.921053, - "ndcgAt10": 0.84764, - "mapAt10": 0.796523, - "contextPrecision": 0.194737, - "contextRecall": 0.921053, + "recallAt5": 0.877358, + "ndcgAt10": 0.780449, + "mapAt10": 0.72804, + "contextPrecision": 0.2, + "contextRecall": 0.877358, "noResultRate": 0, - "meanContextChars": 3188.1315789473683, - "unanswerableQuestions": 6, + "meanContextChars": 4007.0377358490564, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, - "unanswerableCandidates": 19, + "unanswerableCandidates": 20, "unanswerableContextPassages": 5, - "chunkCount": 19, - "latencyP95Ms": 17.91 + "chunkCount": 53, + "latencyP95Ms": 13.55 }, { "strategy": "dense", "candidateK": 20, "contextK": 8, - "recallAt5": 0.921053, - "ndcgAt10": 0.84764, - "mapAt10": 0.796523, - "contextPrecision": 0.131579, - "contextRecall": 1, + "recallAt5": 0.877358, + "ndcgAt10": 0.780449, + "mapAt10": 0.72804, + "contextPrecision": 0.127358, + "contextRecall": 0.877358, "noResultRate": 0, - "meanContextChars": 5151.289473684211, - "unanswerableQuestions": 6, + "meanContextChars": 6658.698113207547, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, - "unanswerableCandidates": 19, + "unanswerableCandidates": 20, "unanswerableContextPassages": 8, - "chunkCount": 19, - "latencyP95Ms": 14.13 + "chunkCount": 53, + "latencyP95Ms": 13.75 }, { "strategy": "dense", "candidateK": 40, "contextK": 3, - "recallAt5": 0.921053, - "ndcgAt10": 0.84764, - "mapAt10": 0.796523, - "contextPrecision": 0.324561, - "contextRecall": 0.921053, + "recallAt5": 0.877358, + "ndcgAt10": 0.780449, + "mapAt10": 0.72804, + "contextPrecision": 0.308176, + "contextRecall": 0.816038, "noResultRate": 0, - "meanContextChars": 1976.1315789473683, - "unanswerableQuestions": 6, + "meanContextChars": 2311.811320754717, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, - "unanswerableCandidates": 19, + "unanswerableCandidates": 40, "unanswerableContextPassages": 3, - "chunkCount": 19, - "latencyP95Ms": 13.35 + "chunkCount": 53, + "latencyP95Ms": 15.44 }, { "strategy": "dense", "candidateK": 40, "contextK": 5, - "recallAt5": 0.921053, - "ndcgAt10": 0.84764, - "mapAt10": 0.796523, - "contextPrecision": 0.194737, - "contextRecall": 0.921053, + "recallAt5": 0.877358, + "ndcgAt10": 0.780449, + "mapAt10": 0.72804, + "contextPrecision": 0.2, + "contextRecall": 0.877358, "noResultRate": 0, - "meanContextChars": 3188.1315789473683, - "unanswerableQuestions": 6, + "meanContextChars": 4007.0377358490564, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, - "unanswerableCandidates": 19, + "unanswerableCandidates": 40, "unanswerableContextPassages": 5, - "chunkCount": 19, - "latencyP95Ms": 12.73 + "chunkCount": 53, + "latencyP95Ms": 14.98 }, { "strategy": "dense", "candidateK": 40, "contextK": 8, - "recallAt5": 0.921053, - "ndcgAt10": 0.84764, - "mapAt10": 0.796523, - "contextPrecision": 0.131579, - "contextRecall": 1, + "recallAt5": 0.877358, + "ndcgAt10": 0.780449, + "mapAt10": 0.72804, + "contextPrecision": 0.127358, + "contextRecall": 0.877358, "noResultRate": 0, - "meanContextChars": 5151.289473684211, - "unanswerableQuestions": 6, + "meanContextChars": 6658.698113207547, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, - "unanswerableCandidates": 19, + "unanswerableCandidates": 40, "unanswerableContextPassages": 8, - "chunkCount": 19, - "latencyP95Ms": 15.12 + "chunkCount": 53, + "latencyP95Ms": 14.68 }, { "strategy": "hybrid", "candidateK": 5, "contextK": 3, - "recallAt5": 0.921053, - "ndcgAt10": 0.830904, - "mapAt10": 0.798246, - "contextPrecision": 0.324561, - "contextRecall": 0.921053, + "recallAt5": 0.910377, + "ndcgAt10": 0.810136, + "mapAt10": 0.768239, + "contextPrecision": 0.320755, + "contextRecall": 0.853774, "noResultRate": 0, - "meanContextChars": 1948.657894736842, - "unanswerableQuestions": 6, + "meanContextChars": 2307.3207547169814, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 5, "unanswerableContextPassages": 3, - "chunkCount": 19, - "latencyP95Ms": 14.01 + "chunkCount": 53, + "latencyP95Ms": 13.45 }, { "strategy": "hybrid", "candidateK": 5, "contextK": 5, - "recallAt5": 0.921053, - "ndcgAt10": 0.830904, - "mapAt10": 0.798246, - "contextPrecision": 0.194737, - "contextRecall": 0.921053, + "recallAt5": 0.910377, + "ndcgAt10": 0.810136, + "mapAt10": 0.768239, + "contextPrecision": 0.211321, + "contextRecall": 0.910377, "noResultRate": 0, - "meanContextChars": 3215.5789473684213, - "unanswerableQuestions": 6, + "meanContextChars": 3936.566037735849, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 5, "unanswerableContextPassages": 5, - "chunkCount": 19, - "latencyP95Ms": 15.81 + "chunkCount": 53, + "latencyP95Ms": 77.74 }, { "strategy": "hybrid", "candidateK": 10, "contextK": 3, - "recallAt5": 0.921053, - "ndcgAt10": 0.857352, - "mapAt10": 0.80968, - "contextPrecision": 0.324561, - "contextRecall": 0.921053, + "recallAt5": 0.896226, + "ndcgAt10": 0.806807, + "mapAt10": 0.757966, + "contextPrecision": 0.320755, + "contextRecall": 0.853774, "noResultRate": 0, - "meanContextChars": 1963.342105263158, - "unanswerableQuestions": 6, + "meanContextChars": 2321.264150943396, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 10, "unanswerableContextPassages": 3, - "chunkCount": 19, - "latencyP95Ms": 15.74 + "chunkCount": 53, + "latencyP95Ms": 14.42 }, { "strategy": "hybrid", "candidateK": 10, "contextK": 5, - "recallAt5": 0.921053, - "ndcgAt10": 0.857352, - "mapAt10": 0.80968, - "contextPrecision": 0.194737, - "contextRecall": 0.921053, + "recallAt5": 0.896226, + "ndcgAt10": 0.806807, + "mapAt10": 0.757966, + "contextPrecision": 0.203774, + "contextRecall": 0.896226, "noResultRate": 0, - "meanContextChars": 3345.0789473684213, - "unanswerableQuestions": 6, + "meanContextChars": 3994.509433962264, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 10, "unanswerableContextPassages": 5, - "chunkCount": 19, - "latencyP95Ms": 16.3 + "chunkCount": 53, + "latencyP95Ms": 13.8 }, { "strategy": "hybrid", "candidateK": 10, "contextK": 8, - "recallAt5": 0.921053, - "ndcgAt10": 0.857352, - "mapAt10": 0.80968, - "contextPrecision": 0.131579, - "contextRecall": 1, + "recallAt5": 0.896226, + "ndcgAt10": 0.806807, + "mapAt10": 0.757966, + "contextPrecision": 0.132075, + "contextRecall": 0.900943, "noResultRate": 0, - "meanContextChars": 5319.868421052632, - "unanswerableQuestions": 6, + "meanContextChars": 6547.641509433963, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 10, "unanswerableContextPassages": 8, - "chunkCount": 19, - "latencyP95Ms": 16.95 + "chunkCount": 53, + "latencyP95Ms": 13.04 }, { "strategy": "hybrid", "candidateK": 20, "contextK": 3, - "recallAt5": 0.921053, - "ndcgAt10": 0.857352, - "mapAt10": 0.80968, - "contextPrecision": 0.324561, - "contextRecall": 0.921053, + "recallAt5": 0.896226, + "ndcgAt10": 0.809819, + "mapAt10": 0.760469, + "contextPrecision": 0.320755, + "contextRecall": 0.853774, "noResultRate": 0, - "meanContextChars": 1958.7631578947369, - "unanswerableQuestions": 6, + "meanContextChars": 2322.830188679245, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, - "unanswerableCandidates": 19, + "unanswerableCandidates": 20, "unanswerableContextPassages": 3, - "chunkCount": 19, - "latencyP95Ms": 14.32 + "chunkCount": 53, + "latencyP95Ms": 15.55 }, { "strategy": "hybrid", "candidateK": 20, "contextK": 5, - "recallAt5": 0.921053, - "ndcgAt10": 0.857352, - "mapAt10": 0.80968, - "contextPrecision": 0.194737, - "contextRecall": 0.921053, + "recallAt5": 0.896226, + "ndcgAt10": 0.809819, + "mapAt10": 0.760469, + "contextPrecision": 0.203774, + "contextRecall": 0.896226, "noResultRate": 0, - "meanContextChars": 3282.1052631578946, - "unanswerableQuestions": 6, + "meanContextChars": 4011.377358490566, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, - "unanswerableCandidates": 19, + "unanswerableCandidates": 20, "unanswerableContextPassages": 5, - "chunkCount": 19, - "latencyP95Ms": 14.82 + "chunkCount": 53, + "latencyP95Ms": 16.36 }, { "strategy": "hybrid", "candidateK": 20, "contextK": 8, - "recallAt5": 0.921053, - "ndcgAt10": 0.857352, - "mapAt10": 0.80968, - "contextPrecision": 0.131579, - "contextRecall": 1, + "recallAt5": 0.896226, + "ndcgAt10": 0.809819, + "mapAt10": 0.760469, + "contextPrecision": 0.132075, + "contextRecall": 0.90566, "noResultRate": 0, - "meanContextChars": 5357.315789473684, - "unanswerableQuestions": 6, + "meanContextChars": 6660.264150943396, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, - "unanswerableCandidates": 19, + "unanswerableCandidates": 20, "unanswerableContextPassages": 8, - "chunkCount": 19, - "latencyP95Ms": 13.74 + "chunkCount": 53, + "latencyP95Ms": 14.66 }, { "strategy": "hybrid", "candidateK": 40, "contextK": 3, - "recallAt5": 0.921053, - "ndcgAt10": 0.857352, - "mapAt10": 0.80968, - "contextPrecision": 0.324561, - "contextRecall": 0.921053, + "recallAt5": 0.896226, + "ndcgAt10": 0.809819, + "mapAt10": 0.760469, + "contextPrecision": 0.320755, + "contextRecall": 0.853774, "noResultRate": 0, - "meanContextChars": 1958.7631578947369, - "unanswerableQuestions": 6, + "meanContextChars": 2322.830188679245, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, - "unanswerableCandidates": 19, + "unanswerableCandidates": 40, "unanswerableContextPassages": 3, - "chunkCount": 19, - "latencyP95Ms": 18.46 + "chunkCount": 53, + "latencyP95Ms": 14.55 }, { "strategy": "hybrid", "candidateK": 40, "contextK": 5, - "recallAt5": 0.921053, - "ndcgAt10": 0.857352, - "mapAt10": 0.80968, - "contextPrecision": 0.194737, - "contextRecall": 0.921053, + "recallAt5": 0.896226, + "ndcgAt10": 0.809819, + "mapAt10": 0.760469, + "contextPrecision": 0.203774, + "contextRecall": 0.896226, "noResultRate": 0, - "meanContextChars": 3282.1052631578946, - "unanswerableQuestions": 6, + "meanContextChars": 3989.735849056604, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, - "unanswerableCandidates": 19, + "unanswerableCandidates": 40, "unanswerableContextPassages": 5, - "chunkCount": 19, - "latencyP95Ms": 16.37 + "chunkCount": 53, + "latencyP95Ms": 14.41 }, { "strategy": "hybrid", "candidateK": 40, "contextK": 8, - "recallAt5": 0.921053, - "ndcgAt10": 0.857352, - "mapAt10": 0.80968, - "contextPrecision": 0.131579, - "contextRecall": 1, + "recallAt5": 0.896226, + "ndcgAt10": 0.809819, + "mapAt10": 0.760469, + "contextPrecision": 0.132075, + "contextRecall": 0.90566, "noResultRate": 0, - "meanContextChars": 5357.315789473684, - "unanswerableQuestions": 6, + "meanContextChars": 6660.792452830188, + "unanswerableQuestions": 10, "unanswerableAbstentionRate": 0, - "unanswerableCandidates": 19, + "unanswerableCandidates": 40, "unanswerableContextPassages": 8, - "chunkCount": 19, - "latencyP95Ms": 20.86 + "chunkCount": 53, + "latencyP95Ms": 14.47 } ] } diff --git a/docs/eval/sweep-v1.6.md b/docs/eval/sweep-v1.6.md index a3c6dd7..46aeb3e 100644 --- a/docs/eval/sweep-v1.6.md +++ b/docs/eval/sweep-v1.6.md @@ -12,28 +12,28 @@ Each row differs from its neighbour in one parameter. | Strategy | candidateK | contextK | Recall@5 | nDCG@10 | MAP@10 | Context P | Context R | No-result | Context chars | Unans. abstained | Unans. cands | Unans. ctx | Index | p95 | | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | -| dense | 5 | 3 | 0.9211 | 0.8212 | 0.7851 | 0.3246 | 0.9211 | 0.0000 | 1976 | 0.0000 | 5.0 | 3.0 | 19 | 14.04 ms | -| dense | 5 | 5 | 0.9211 | 0.8212 | 0.7851 | 0.1947 | 0.9211 | 0.0000 | 3188 | 0.0000 | 5.0 | 5.0 | 19 | 13.63 ms | -| dense | 10 | 3 | 0.9211 | 0.8476 | 0.7965 | 0.3246 | 0.9211 | 0.0000 | 1976 | 0.0000 | 10.0 | 3.0 | 19 | 13.84 ms | -| dense | 10 | 5 | 0.9211 | 0.8476 | 0.7965 | 0.1947 | 0.9211 | 0.0000 | 3188 | 0.0000 | 10.0 | 5.0 | 19 | 15.21 ms | -| dense | 10 | 8 | 0.9211 | 0.8476 | 0.7965 | 0.1316 | 1.0000 | 0.0000 | 5151 | 0.0000 | 10.0 | 8.0 | 19 | 16.35 ms | -| dense | 20 | 3 | 0.9211 | 0.8476 | 0.7965 | 0.3246 | 0.9211 | 0.0000 | 1976 | 0.0000 | 19.0 | 3.0 | 19 | 16.45 ms | -| dense | 20 | 5 | 0.9211 | 0.8476 | 0.7965 | 0.1947 | 0.9211 | 0.0000 | 3188 | 0.0000 | 19.0 | 5.0 | 19 | 17.91 ms | -| dense | 20 | 8 | 0.9211 | 0.8476 | 0.7965 | 0.1316 | 1.0000 | 0.0000 | 5151 | 0.0000 | 19.0 | 8.0 | 19 | 14.13 ms | -| dense | 40 | 3 | 0.9211 | 0.8476 | 0.7965 | 0.3246 | 0.9211 | 0.0000 | 1976 | 0.0000 | 19.0 | 3.0 | 19 | 13.35 ms | -| dense | 40 | 5 | 0.9211 | 0.8476 | 0.7965 | 0.1947 | 0.9211 | 0.0000 | 3188 | 0.0000 | 19.0 | 5.0 | 19 | 12.73 ms | -| dense | 40 | 8 | 0.9211 | 0.8476 | 0.7965 | 0.1316 | 1.0000 | 0.0000 | 5151 | 0.0000 | 19.0 | 8.0 | 19 | 15.12 ms | -| hybrid | 5 | 3 | 0.9211 | 0.8309 | 0.7982 | 0.3246 | 0.9211 | 0.0000 | 1949 | 0.0000 | 5.0 | 3.0 | 19 | 14.01 ms | -| hybrid | 5 | 5 | 0.9211 | 0.8309 | 0.7982 | 0.1947 | 0.9211 | 0.0000 | 3216 | 0.0000 | 5.0 | 5.0 | 19 | 15.81 ms | -| hybrid | 10 | 3 | 0.9211 | 0.8574 | 0.8097 | 0.3246 | 0.9211 | 0.0000 | 1963 | 0.0000 | 10.0 | 3.0 | 19 | 15.74 ms | -| hybrid | 10 | 5 | 0.9211 | 0.8574 | 0.8097 | 0.1947 | 0.9211 | 0.0000 | 3345 | 0.0000 | 10.0 | 5.0 | 19 | 16.30 ms | -| hybrid | 10 | 8 | 0.9211 | 0.8574 | 0.8097 | 0.1316 | 1.0000 | 0.0000 | 5320 | 0.0000 | 10.0 | 8.0 | 19 | 16.95 ms | -| hybrid | 20 | 3 | 0.9211 | 0.8574 | 0.8097 | 0.3246 | 0.9211 | 0.0000 | 1959 | 0.0000 | 19.0 | 3.0 | 19 | 14.32 ms | -| hybrid | 20 | 5 | 0.9211 | 0.8574 | 0.8097 | 0.1947 | 0.9211 | 0.0000 | 3282 | 0.0000 | 19.0 | 5.0 | 19 | 14.82 ms | -| hybrid | 20 | 8 | 0.9211 | 0.8574 | 0.8097 | 0.1316 | 1.0000 | 0.0000 | 5357 | 0.0000 | 19.0 | 8.0 | 19 | 13.74 ms | -| hybrid | 40 | 3 | 0.9211 | 0.8574 | 0.8097 | 0.3246 | 0.9211 | 0.0000 | 1959 | 0.0000 | 19.0 | 3.0 | 19 | 18.46 ms | -| hybrid | 40 | 5 | 0.9211 | 0.8574 | 0.8097 | 0.1947 | 0.9211 | 0.0000 | 3282 | 0.0000 | 19.0 | 5.0 | 19 | 16.37 ms | -| hybrid | 40 | 8 | 0.9211 | 0.8574 | 0.8097 | 0.1316 | 1.0000 | 0.0000 | 5357 | 0.0000 | 19.0 | 8.0 | 19 | 20.86 ms | +| dense | 5 | 3 | 0.8774 | 0.7672 | 0.7231 | 0.3082 | 0.8160 | 0.0000 | 2312 | 0.0000 | 5.0 | 3.0 | 53 | 12.30 ms | +| dense | 5 | 5 | 0.8774 | 0.7672 | 0.7231 | 0.2000 | 0.8774 | 0.0000 | 4007 | 0.0000 | 5.0 | 5.0 | 53 | 12.37 ms | +| dense | 10 | 3 | 0.8774 | 0.7804 | 0.7280 | 0.3082 | 0.8160 | 0.0000 | 2312 | 0.0000 | 10.0 | 3.0 | 53 | 12.66 ms | +| dense | 10 | 5 | 0.8774 | 0.7804 | 0.7280 | 0.2000 | 0.8774 | 0.0000 | 4007 | 0.0000 | 10.0 | 5.0 | 53 | 13.42 ms | +| dense | 10 | 8 | 0.8774 | 0.7804 | 0.7280 | 0.1274 | 0.8774 | 0.0000 | 6659 | 0.0000 | 10.0 | 8.0 | 53 | 12.80 ms | +| dense | 20 | 3 | 0.8774 | 0.7804 | 0.7280 | 0.3082 | 0.8160 | 0.0000 | 2312 | 0.0000 | 20.0 | 3.0 | 53 | 13.23 ms | +| dense | 20 | 5 | 0.8774 | 0.7804 | 0.7280 | 0.2000 | 0.8774 | 0.0000 | 4007 | 0.0000 | 20.0 | 5.0 | 53 | 13.55 ms | +| dense | 20 | 8 | 0.8774 | 0.7804 | 0.7280 | 0.1274 | 0.8774 | 0.0000 | 6659 | 0.0000 | 20.0 | 8.0 | 53 | 13.75 ms | +| dense | 40 | 3 | 0.8774 | 0.7804 | 0.7280 | 0.3082 | 0.8160 | 0.0000 | 2312 | 0.0000 | 40.0 | 3.0 | 53 | 15.44 ms | +| dense | 40 | 5 | 0.8774 | 0.7804 | 0.7280 | 0.2000 | 0.8774 | 0.0000 | 4007 | 0.0000 | 40.0 | 5.0 | 53 | 14.98 ms | +| dense | 40 | 8 | 0.8774 | 0.7804 | 0.7280 | 0.1274 | 0.8774 | 0.0000 | 6659 | 0.0000 | 40.0 | 8.0 | 53 | 14.68 ms | +| hybrid | 5 | 3 | 0.9104 | 0.8101 | 0.7682 | 0.3208 | 0.8538 | 0.0000 | 2307 | 0.0000 | 5.0 | 3.0 | 53 | 13.45 ms | +| hybrid | 5 | 5 | 0.9104 | 0.8101 | 0.7682 | 0.2113 | 0.9104 | 0.0000 | 3937 | 0.0000 | 5.0 | 5.0 | 53 | 77.74 ms | +| hybrid | 10 | 3 | 0.8962 | 0.8068 | 0.7580 | 0.3208 | 0.8538 | 0.0000 | 2321 | 0.0000 | 10.0 | 3.0 | 53 | 14.42 ms | +| hybrid | 10 | 5 | 0.8962 | 0.8068 | 0.7580 | 0.2038 | 0.8962 | 0.0000 | 3995 | 0.0000 | 10.0 | 5.0 | 53 | 13.80 ms | +| hybrid | 10 | 8 | 0.8962 | 0.8068 | 0.7580 | 0.1321 | 0.9009 | 0.0000 | 6548 | 0.0000 | 10.0 | 8.0 | 53 | 13.04 ms | +| hybrid | 20 | 3 | 0.8962 | 0.8098 | 0.7605 | 0.3208 | 0.8538 | 0.0000 | 2323 | 0.0000 | 20.0 | 3.0 | 53 | 15.55 ms | +| hybrid | 20 | 5 | 0.8962 | 0.8098 | 0.7605 | 0.2038 | 0.8962 | 0.0000 | 4011 | 0.0000 | 20.0 | 5.0 | 53 | 16.36 ms | +| hybrid | 20 | 8 | 0.8962 | 0.8098 | 0.7605 | 0.1321 | 0.9057 | 0.0000 | 6660 | 0.0000 | 20.0 | 8.0 | 53 | 14.66 ms | +| hybrid | 40 | 3 | 0.8962 | 0.8098 | 0.7605 | 0.3208 | 0.8538 | 0.0000 | 2323 | 0.0000 | 40.0 | 3.0 | 53 | 14.55 ms | +| hybrid | 40 | 5 | 0.8962 | 0.8098 | 0.7605 | 0.2038 | 0.8962 | 0.0000 | 3990 | 0.0000 | 40.0 | 5.0 | 53 | 14.41 ms | +| hybrid | 40 | 8 | 0.8962 | 0.8098 | 0.7605 | 0.1321 | 0.9057 | 0.0000 | 6661 | 0.0000 | 40.0 | 8.0 | 53 | 14.47 ms | ## Skipped cells @@ -69,10 +69,10 @@ The harness refuses the same combination at the flag level, so a typo fails loud These are the columns a threshold decision should move, and they stay out of every other column. -Best nDCG@10 in this grid: `hybrid` candidateK=10, -contextK=3 (0.8574). -Best context precision: `dense` candidateK=5, -contextK=3 (0.3246). +Best nDCG@10 in this grid: `hybrid` candidateK=5, +contextK=3 (0.8101). +Best context precision: `hybrid` candidateK=5, +contextK=3 (0.3208). These are **not** recommendations. Selecting the grid maximum on the same questions is how a benchmark becomes a lookup table; the adoption rule in `baseline-v1.6.md` diff --git a/docs/eval/threshold-v1.6.json b/docs/eval/threshold-v1.6.json index cb995ef..15d8562 100644 --- a/docs/eval/threshold-v1.6.json +++ b/docs/eval/threshold-v1.6.json @@ -7,265 +7,265 @@ { "threshold": 0, "validation": { - "questions": 16, - "answerableQuestions": 13, + "questions": 24, + "answerableQuestions": 19, "noResultCount": 0, "noResultRate": 0, - "meanRetrieved": 19, + "meanRetrieved": 20, "unanswerable": { - "questions": 3, + "questions": 5, "abstentionCount": 0, "retrievalAbstentionRate": 0, - "meanCandidatesRetrieved": 19, + "meanCandidatesRetrieved": 20, "meanContextPassages": 3 }, "metrics": { - "recallAt1": 0.576923, - "recallAt5": 0.923077, - "recallAt10": 1, - "mrr": 0.778846, - "ndcgAt10": 0.827608, - "hitRateAt5": 0.923077, - "mapAt10": 0.766026, - "contextPrecision": 0.333333, - "contextRecall": 0.923077 + "recallAt1": 0.513158, + "recallAt5": 0.868421, + "recallAt10": 0.921053, + "mrr": 0.720175, + "ndcgAt10": 0.755611, + "hitRateAt5": 0.894737, + "mapAt10": 0.69386, + "contextPrecision": 0.315789, + "contextRecall": 0.802632 } }, "test": { - "questions": 28, - "answerableQuestions": 25, + "questions": 39, + "answerableQuestions": 34, "noResultCount": 0, "noResultRate": 0, - "meanRetrieved": 19, + "meanRetrieved": 20, "unanswerable": { - "questions": 3, + "questions": 5, "abstentionCount": 0, "retrievalAbstentionRate": 0, - "meanCandidatesRetrieved": 19, + "meanCandidatesRetrieved": 20, "meanContextPassages": 3 }, "metrics": { - "recallAt1": 0.7, - "recallAt5": 0.92, - "recallAt10": 1, - "mrr": 0.812381, - "ndcgAt10": 0.858056, - "hitRateAt5": 0.92, - "mapAt10": 0.812381, - "contextPrecision": 0.32, - "contextRecall": 0.92 + "recallAt1": 0.639706, + "recallAt5": 0.882353, + "recallAt10": 0.919118, + "mrr": 0.774079, + "ndcgAt10": 0.794329, + "hitRateAt5": 0.911765, + "mapAt10": 0.747141, + "contextPrecision": 0.303922, + "contextRecall": 0.823529 } } }, { "threshold": 0.3, "validation": { - "questions": 16, - "answerableQuestions": 13, + "questions": 24, + "answerableQuestions": 19, "noResultCount": 0, "noResultRate": 0, - "meanRetrieved": 19, + "meanRetrieved": 20, "unanswerable": { - "questions": 3, + "questions": 5, "abstentionCount": 0, "retrievalAbstentionRate": 0, - "meanCandidatesRetrieved": 19, + "meanCandidatesRetrieved": 20, "meanContextPassages": 3 }, "metrics": { - "recallAt1": 0.576923, - "recallAt5": 0.923077, - "recallAt10": 1, - "mrr": 0.778846, - "ndcgAt10": 0.827608, - "hitRateAt5": 0.923077, - "mapAt10": 0.766026, - "contextPrecision": 0.333333, - "contextRecall": 0.923077 + "recallAt1": 0.513158, + "recallAt5": 0.868421, + "recallAt10": 0.921053, + "mrr": 0.720175, + "ndcgAt10": 0.755611, + "hitRateAt5": 0.894737, + "mapAt10": 0.69386, + "contextPrecision": 0.315789, + "contextRecall": 0.802632 } }, "test": { - "questions": 28, - "answerableQuestions": 25, + "questions": 39, + "answerableQuestions": 34, "noResultCount": 0, "noResultRate": 0, - "meanRetrieved": 19, + "meanRetrieved": 20, "unanswerable": { - "questions": 3, + "questions": 5, "abstentionCount": 0, "retrievalAbstentionRate": 0, - "meanCandidatesRetrieved": 19, + "meanCandidatesRetrieved": 20, "meanContextPassages": 3 }, "metrics": { - "recallAt1": 0.7, - "recallAt5": 0.92, - "recallAt10": 1, - "mrr": 0.812381, - "ndcgAt10": 0.858056, - "hitRateAt5": 0.92, - "mapAt10": 0.812381, - "contextPrecision": 0.32, - "contextRecall": 0.92 + "recallAt1": 0.639706, + "recallAt5": 0.882353, + "recallAt10": 0.919118, + "mrr": 0.774079, + "ndcgAt10": 0.794329, + "hitRateAt5": 0.911765, + "mapAt10": 0.747141, + "contextPrecision": 0.303922, + "contextRecall": 0.823529 } } }, { "threshold": 0.4, "validation": { - "questions": 16, - "answerableQuestions": 13, + "questions": 24, + "answerableQuestions": 19, "noResultCount": 0, "noResultRate": 0, - "meanRetrieved": 19, + "meanRetrieved": 20, "unanswerable": { - "questions": 3, + "questions": 5, "abstentionCount": 0, "retrievalAbstentionRate": 0, - "meanCandidatesRetrieved": 19, + "meanCandidatesRetrieved": 20, "meanContextPassages": 3 }, "metrics": { - "recallAt1": 0.576923, - "recallAt5": 0.923077, - "recallAt10": 1, - "mrr": 0.778846, - "ndcgAt10": 0.827608, - "hitRateAt5": 0.923077, - "mapAt10": 0.766026, - "contextPrecision": 0.333333, - "contextRecall": 0.923077 + "recallAt1": 0.513158, + "recallAt5": 0.868421, + "recallAt10": 0.921053, + "mrr": 0.720175, + "ndcgAt10": 0.755611, + "hitRateAt5": 0.894737, + "mapAt10": 0.69386, + "contextPrecision": 0.315789, + "contextRecall": 0.802632 } }, "test": { - "questions": 28, - "answerableQuestions": 25, + "questions": 39, + "answerableQuestions": 34, "noResultCount": 0, "noResultRate": 0, - "meanRetrieved": 19, + "meanRetrieved": 20, "unanswerable": { - "questions": 3, + "questions": 5, "abstentionCount": 0, "retrievalAbstentionRate": 0, - "meanCandidatesRetrieved": 19, + "meanCandidatesRetrieved": 20, "meanContextPassages": 3 }, "metrics": { - "recallAt1": 0.7, - "recallAt5": 0.92, - "recallAt10": 1, - "mrr": 0.812381, - "ndcgAt10": 0.858056, - "hitRateAt5": 0.92, - "mapAt10": 0.812381, - "contextPrecision": 0.32, - "contextRecall": 0.92 + "recallAt1": 0.639706, + "recallAt5": 0.882353, + "recallAt10": 0.919118, + "mrr": 0.774079, + "ndcgAt10": 0.794329, + "hitRateAt5": 0.911765, + "mapAt10": 0.747141, + "contextPrecision": 0.303922, + "contextRecall": 0.823529 } } }, { "threshold": 0.5, "validation": { - "questions": 16, - "answerableQuestions": 13, + "questions": 24, + "answerableQuestions": 19, "noResultCount": 0, "noResultRate": 0, - "meanRetrieved": 19, + "meanRetrieved": 20, "unanswerable": { - "questions": 3, + "questions": 5, "abstentionCount": 0, "retrievalAbstentionRate": 0, - "meanCandidatesRetrieved": 19, + "meanCandidatesRetrieved": 20, "meanContextPassages": 3 }, "metrics": { - "recallAt1": 0.576923, - "recallAt5": 0.923077, - "recallAt10": 1, - "mrr": 0.778846, - "ndcgAt10": 0.827608, - "hitRateAt5": 0.923077, - "mapAt10": 0.766026, - "contextPrecision": 0.333333, - "contextRecall": 0.923077 + "recallAt1": 0.513158, + "recallAt5": 0.868421, + "recallAt10": 0.921053, + "mrr": 0.720175, + "ndcgAt10": 0.755611, + "hitRateAt5": 0.894737, + "mapAt10": 0.69386, + "contextPrecision": 0.315789, + "contextRecall": 0.802632 } }, "test": { - "questions": 28, - "answerableQuestions": 25, + "questions": 39, + "answerableQuestions": 34, "noResultCount": 0, "noResultRate": 0, - "meanRetrieved": 19, + "meanRetrieved": 20, "unanswerable": { - "questions": 3, + "questions": 5, "abstentionCount": 0, "retrievalAbstentionRate": 0, - "meanCandidatesRetrieved": 19, + "meanCandidatesRetrieved": 20, "meanContextPassages": 3 }, "metrics": { - "recallAt1": 0.7, - "recallAt5": 0.92, - "recallAt10": 1, - "mrr": 0.812381, - "ndcgAt10": 0.858056, - "hitRateAt5": 0.92, - "mapAt10": 0.812381, - "contextPrecision": 0.32, - "contextRecall": 0.92 + "recallAt1": 0.639706, + "recallAt5": 0.882353, + "recallAt10": 0.919118, + "mrr": 0.774079, + "ndcgAt10": 0.794329, + "hitRateAt5": 0.911765, + "mapAt10": 0.747141, + "contextPrecision": 0.303922, + "contextRecall": 0.823529 } } }, { "threshold": 0.6, "validation": { - "questions": 16, - "answerableQuestions": 13, + "questions": 24, + "answerableQuestions": 19, "noResultCount": 0, "noResultRate": 0, - "meanRetrieved": 19, + "meanRetrieved": 20, "unanswerable": { - "questions": 3, + "questions": 5, "abstentionCount": 0, "retrievalAbstentionRate": 0, - "meanCandidatesRetrieved": 19, + "meanCandidatesRetrieved": 20, "meanContextPassages": 3 }, "metrics": { - "recallAt1": 0.576923, - "recallAt5": 0.923077, - "recallAt10": 1, - "mrr": 0.778846, - "ndcgAt10": 0.827608, - "hitRateAt5": 0.923077, - "mapAt10": 0.766026, - "contextPrecision": 0.333333, - "contextRecall": 0.923077 + "recallAt1": 0.513158, + "recallAt5": 0.868421, + "recallAt10": 0.921053, + "mrr": 0.720175, + "ndcgAt10": 0.755611, + "hitRateAt5": 0.894737, + "mapAt10": 0.69386, + "contextPrecision": 0.315789, + "contextRecall": 0.802632 } }, "test": { - "questions": 28, - "answerableQuestions": 25, + "questions": 39, + "answerableQuestions": 34, "noResultCount": 0, "noResultRate": 0, - "meanRetrieved": 19, + "meanRetrieved": 20, "unanswerable": { - "questions": 3, + "questions": 5, "abstentionCount": 0, "retrievalAbstentionRate": 0, - "meanCandidatesRetrieved": 19, + "meanCandidatesRetrieved": 20, "meanContextPassages": 3 }, "metrics": { - "recallAt1": 0.7, - "recallAt5": 0.92, - "recallAt10": 1, - "mrr": 0.812381, - "ndcgAt10": 0.858056, - "hitRateAt5": 0.92, - "mapAt10": 0.812381, - "contextPrecision": 0.32, - "contextRecall": 0.92 + "recallAt1": 0.639706, + "recallAt5": 0.882353, + "recallAt10": 0.919118, + "mrr": 0.774079, + "ndcgAt10": 0.794329, + "hitRateAt5": 0.911765, + "mapAt10": 0.747141, + "contextPrecision": 0.303922, + "contextRecall": 0.823529 } } } diff --git a/docs/eval/threshold-v1.6.md b/docs/eval/threshold-v1.6.md index 0b08be6..3997609 100644 --- a/docs/eval/threshold-v1.6.md +++ b/docs/eval/threshold-v1.6.md @@ -16,11 +16,11 @@ passed the threshold (up to `candidateK`), **ctx** is how many reach the context | Threshold | n (val) | Recall@5 (val) | nDCG@10 (val) | No-result (val) | Unans. abstained (val) | Unans. cands (val) | Unans. ctx (val) | nDCG@10 (test) | Unans. abstained (test) | | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | -| 0 | 13 | 0.9231 | 0.8276 | 0.0000 | 0.0000 | 19.0 | 3.0 | 0.8581 | 0.0000 | -| 0.3 | 13 | 0.9231 | 0.8276 | 0.0000 | 0.0000 | 19.0 | 3.0 | 0.8581 | 0.0000 | -| 0.4 | 13 | 0.9231 | 0.8276 | 0.0000 | 0.0000 | 19.0 | 3.0 | 0.8581 | 0.0000 | -| 0.5 | 13 | 0.9231 | 0.8276 | 0.0000 | 0.0000 | 19.0 | 3.0 | 0.8581 | 0.0000 | -| 0.6 | 13 | 0.9231 | 0.8276 | 0.0000 | 0.0000 | 19.0 | 3.0 | 0.8581 | 0.0000 | +| 0 | 19 | 0.8684 | 0.7556 | 0.0000 | 0.0000 | 20.0 | 3.0 | 0.7943 | 0.0000 | +| 0.3 | 19 | 0.8684 | 0.7556 | 0.0000 | 0.0000 | 20.0 | 3.0 | 0.7943 | 0.0000 | +| 0.4 | 19 | 0.8684 | 0.7556 | 0.0000 | 0.0000 | 20.0 | 3.0 | 0.7943 | 0.0000 | +| 0.5 | 19 | 0.8684 | 0.7556 | 0.0000 | 0.0000 | 20.0 | 3.0 | 0.7943 | 0.0000 | +| 0.6 | 19 | 0.8684 | 0.7556 | 0.0000 | 0.0000 | 20.0 | 3.0 | 0.7943 | 0.0000 | ## Selection rule @@ -34,13 +34,13 @@ retrieval loss. ## Outcome -The sweep is **flat**: every threshold from 0 to 0.6 produces the same validation nDCG@10 (0.8276), the same Recall@5 (0.9231) and the same retrieval abstention rate on unanswerable questions (0/3). No candidate is ever filtered out, so the threshold is **non-binding** on this corpus. +The sweep is **flat**: every threshold from 0 to 0.6 produces the same validation nDCG@10 (0.7556), the same Recall@5 (0.8684) and the same retrieval abstention rate on unanswerable questions (0/5). No candidate is ever filtered out, so the threshold is **non-binding** on this corpus. **No evidence to change `threshold = 0.5`.** All thresholds hold the line equally; picking one would be arbitrary. The current value can be neither validated nor falsified here, which is a property of the corpus rather than of the threshold. Note also what this does *not* establish: abstention is a retrieval-layer statement — whether the model then declines to answer needs a generator eval. ## Caveat on this corpus -The split removes the most obvious form of overfitting, but 13 +The split removes the most obvious form of overfitting, but 19 answerable questions on the validation side is a thin basis for a decision, and the corpus is still small. A threshold is a product decision with a **refusal-rate** cost attached, so a recommendation here is only as good as the corpus behind it. Re-run this after the diff --git a/eval/README.md b/eval/README.md index 24deb75..db93d5d 100644 --- a/eval/README.md +++ b/eval/README.md @@ -9,9 +9,10 @@ every experiment (#77, #78) is reported as a delta against that file. ```bash npm run eval:prepare # one-time, networked: download the pinned embedding model npm run eval # offline and deterministic: run the harness, rewrite the baseline -npm run eval:retrieval # strategy comparison (#77) +npm run eval:retrieval # strategy comparison (#77); validation selects, test reports npm run eval:threshold # derive the similarity threshold on validation, report on test npm run eval:sweep # bounded grid over strategy × candidateK × contextK, one dashboard +npm run eval:blocks eval/corpus/foo.md # print the block ordinals ground truth must use ``` ### The harness runs the production configuration @@ -116,7 +117,10 @@ first reads as a permanent miss, the second as a normal hit. Ground truth uses **corpus identity, never database identity**: - `document` is the corpus-relative path. -- `block` is the block ordinal inside the document (`document_blocks.order`). +- `block` is the **`document_blocks.order` the ingestion pipeline produced**, not a line + number and not a paragraph index a human counted. Use `npm run eval:blocks ` to + print the real ordinals through the same loader the harness uses — guessing them is how a + dataset drifts. - `page` is `null` for unpaginated sources. - `quote` is an optional excerpt. The runner fails if the referenced block no longer contains it, so a parser change cannot silently move the ground truth. diff --git a/eval/corpus/api-gateway-migration.md b/eval/corpus/api-gateway-migration.md new file mode 100644 index 0000000..9bb7dbe --- /dev/null +++ b/eval/corpus/api-gateway-migration.md @@ -0,0 +1,79 @@ +# Gateway API Migration Guide + +## Overview + +This guide covers moving a client between Gateway API versions. It deliberately restates +the numbers that differ between versions, because the most common migration defect is a +client that keeps a v1 constant while pointing at a v2 or v3 route. + +The three live surfaces are `/v1/complete`, `/v2/generate` and `/v3/chat`. Any of the three +will reject a body shaped for a different one with `400`, so a silently wrong version is +usually a routing mistake rather than a schema mistake. + +## Choosing a Target + +New integrations should target v3. Existing v2 integrations should move to v3 only when +they need structured output or the larger window, because v3 changes the rate-limit +accounting in a way that can halve effective throughput for structured-output workloads. + +v1 integrations should migrate to v2 at minimum. v1 has no streaming mode, and every +long-answer workload written against v1 pays for it in perceived latency. + +## Endpoint Changes + +| | v1 | v2 | v3 | +| --- | --- | --- | --- | +| Path | `/v1/complete` | `/v2/generate` | `/v3/chat` | +| Body | `prompt` | `messages` | `messages` | +| Streaming | none | server-sent events | server-sent events with `phase` | + +The version lives in the path on all three. A client that versioned the host instead will +not be routed by the gateway at all, and the failure looks like a DNS failure rather than a +version mismatch. + +## Timeout and Retry Changes + +| | v1 | v2 | v3 | +| --- | --- | --- | --- | +| Default timeout | 30000 ms | 60000 ms | 45000 ms | +| Recommended retries | 2 | 5 | 3 | +| Backoff base | 500 ms | 2000 ms | 1000 ms | + +A client migrating from v1 to v2 that keeps the v1 backoff of 500 ms will retry far more +aggressively than the version it is talking to expects, which is a common cause of +self-inflicted `429`s during a cutover. + +Moving from v2 to v3 in the other direction is the riskier one: v3 has fewer recommended +retries and a lower default timeout, so a client tuned for v2's patience will give up +earlier than it did before. + +## Rate Limit Changes + +| | v1 | v2 | v3 | +| --- | --- | --- | --- | +| Default limit | 600 rpm | 3000 rpm | 1200 rpm | +| Burst | 60 | 300 | 120 | + +The v3 structured-output path counts as two requests. A v2 workload that produced 1000 +structured responses per minute was comfortably inside v2's 3000 rpm budget and is almost +exactly at v3's effective 600-per-minute structured ceiling, so the migration is a +throughput change even though the headline number only fell from 3000 to 1200. + +## Context Window Changes + +| | v1 | v2 | v3 | +| --- | --- | --- | --- | +| Window | 8192 tokens | 32768 tokens | 65536 tokens | +| Auto-truncation | no | no | no | + +None of the three truncates automatically; all three reject an oversized request with `400`. +Clients that relied on an upstream provider's truncation find the migration fails loudly +rather than quietly, which is intentional. + +## Checklist + +Before cutting over, confirm the path, the body shape, the timeout, the retry count, the +backoff base and the rate-limit budget. Of those six, the two that are missed most often in +practice are the backoff base and the rate-limit budget, because neither produces an error — +they produce a client that is slower or noisier than it was, which is easy to attribute to +the model rather than to the migration. diff --git a/eval/corpus/api-gateway-v1.md b/eval/corpus/api-gateway-v1.md new file mode 100644 index 0000000..8fc2fe0 --- /dev/null +++ b/eval/corpus/api-gateway-v1.md @@ -0,0 +1,58 @@ +# Gateway API v1 Reference + +## Overview + +The v1 Gateway API is the first generally available surface for text completion. It +accepts a prompt and returns a completion, with no notion of roles, tools or streaming +frames beyond newline-delimited chunks. Clients are expected to be long-lived processes +that hold a single connection open and issue many requests over it. + +v1 is closed to new features. It receives security fixes only, and the deprecation notice +on the v1 endpoint names v2 as the supported successor. + +## Endpoint + +Requests go to the `/v1/complete` path. The path is versioned rather than the host, so a +client that hardcodes the host will silently keep talking to v1 after an upgrade. The +request body carries `prompt`, `max_tokens` and an optional `stop` array; there is no +`system` field, and a system instruction has to be concatenated into the prompt. + +Responses are returned as a single JSON object. v1 has no streaming mode, which is the +change most often cited in migration discussions. + +## Timeouts and Retry + +The default request timeout is **30000 milliseconds**. A request that has not produced any +output within that window is cancelled by the gateway, not by the client, and the client +sees a `504` with the body `{"error":"upstream_timeout"}`. + +The recommended retry count is **2**. The gateway does not retry on the client's behalf, so +this is a client-side contract rather than an enforced limit. The recommended backoff base +is **500 milliseconds**, doubled on each subsequent attempt, which produces waits of 500 ms +then 1000 ms for the two allowed retries. Jitter is not required by v1 but is recommended. + +Retrying a timed-out request is safe because v1 has no server-side session state. Retrying +a request that failed with `429` is not useful unless the `Retry-After` header is honoured +first. + +## Rate Limits + +The default rate limit is **600 requests per minute**, counted per API key rather than per +connection. Bursts of up to 60 requests may be issued within any one-second window before +the limiter engages, so a short burst is allowed even when the per-minute budget is nearly +spent. + +Exceeding the limit returns `429` with a `Retry-After` header in seconds. The limiter +counts a request when the body has been fully received, not when the response is produced, +which means a slow upstream does not consume budget twice. + +## Context Window + +The maximum context window is **8192 tokens**, counting the prompt and the completion +together. A request whose prompt alone exceeds the window is rejected with `400` rather +than truncated, because silent truncation was found to produce worse answers than a +visible failure. + +Token counting uses the same tokenizer as the model, so an approximation by character +count will disagree near the boundary. The gateway exposes a `/v1/tokenize` helper for +clients that want an exact count before sending. diff --git a/eval/corpus/api-gateway-v2.md b/eval/corpus/api-gateway-v2.md new file mode 100644 index 0000000..d39613a --- /dev/null +++ b/eval/corpus/api-gateway-v2.md @@ -0,0 +1,65 @@ +# Gateway API v2 Reference + +## Overview + +The v2 Gateway API replaces the single-prompt completion surface with a message list. It +introduces roles, a real streaming mode and server-side sessions, and it is the surface the +deprecation notice on v1 points at. v2 is feature-frozen: it receives correctness and +security fixes, and v3 is the current recommended target for new integrations. + +The message list is the change that forces most migrations. A v1 prompt with a concatenated +system instruction has to be split into a `system` message and a `user` message, and clients +that relied on concatenation usually find their prompts measurably worse until they split +them. + +## Endpoint + +Requests go to the `/v2/generate` path. Unlike v1, the version is part of the route and +the gateway rejects a v1-shaped body with `400` and a pointer at the migration guide. The +request body carries `messages`, `max_tokens`, `stream` and an optional `tools` array. + +Streaming is enabled per request with `stream: true`, and frames are server-sent events +rather than newline-delimited JSON. Frames carry a monotonically increasing `index`; a +client that reconnects mid-stream must resume from the last index it acknowledged. + +## Timeouts and Retry + +The default request timeout is **60000 milliseconds**, doubled from v1 because v2 sessions +are allowed to think for longer before the first token. The timeout covers the whole turn, +not the gap between frames. + +The recommended retry count is **5**, with a recommended backoff base of **2000 +milliseconds** and full jitter. The higher retry count exists because v2 introduced +server-side sessions, and a retried request may attach to the same session rather than +starting a new one — retrying is therefore usually cheaper than it was in v1. + +A timeout is reported as `504` with `{"error":"upstream_timeout"}` exactly as in v1, so a +client that only inspects that field cannot tell which version produced it. + +## Rate Limits + +The default rate limit is **3000 requests per minute**, counted per API key. Sessions are +counted separately: opening a session costs one request, and each subsequent turn on that +session costs one request, so a long conversation consumes budget linearly. + +Bursts of up to 300 requests may be issued within one second. When a session is already +open, the limiter applies the request to the session's own budget first, which means a +bursty client with many open sessions can exhaust the per-minute budget much faster than +the raw request count suggests. + +## Context Window + +The maximum context window is **32768 tokens**. v2 grew the window partly to make room for +tool definitions, which are counted in the same budget as messages. A request that exceeds +the window is rejected with `400`; v2 does not offer automatic truncation either. + +Because tools are counted, a request with a large tool schema can exceed the window even +when the conversation itself is short. The gateway reports the split between message tokens +and tool tokens in the `usage` block of every response so a client can see which side grew. + +## Sessions + +A session is created implicitly by the first request that omits `session_id`. Sessions +expire after 30 minutes of inactivity. A request that names an expired session is not an +error: the gateway starts a new one and reports the new id, a behaviour that has surprised +several integrators into thinking their retry had lost context. diff --git a/eval/corpus/api-gateway-v3.md b/eval/corpus/api-gateway-v3.md new file mode 100644 index 0000000..e298021 --- /dev/null +++ b/eval/corpus/api-gateway-v3.md @@ -0,0 +1,61 @@ +# Gateway API v3 Reference + +## Overview + +The v3 Gateway API is the current recommended surface. It keeps the message list and +streaming model introduced in v2 and adds structured output, explicit reasoning budgets and +a per-request deadline. v1 and v2 remain available but receive fixes only. + +v3 is not wire-compatible with v2. A v2 body sent to the v3 route is rejected with `400` +and a link to the migration guide, exactly as a v1 body is rejected by v2. + +## Endpoint + +Requests go to the `/v3/chat` path. The body carries `messages`, `max_tokens`, `stream`, +an optional `response_format` describing structured output, and an optional `deadline_ms` +that overrides the default timeout for one request. + +Streaming frames are server-sent events and now carry both an `index` and a `phase`, so a +client can distinguish reasoning frames from answer frames without inspecting the text. +Structured output is delivered as a single final frame; partial structured output is not +emitted, because a half-parsed object was found to be worse than no object. + +## Timeouts and Retry + +The default request timeout is **45000 milliseconds**, between v1's 30 s and v2's 60 s, +chosen after measuring that the median v3 turn finishes in about 11 s while the long tail +benefits from more room than v1 gave. + +The recommended retry count is **3**, with a backoff base of **1000 milliseconds** and full +jitter. v3 adds `deadline_ms`, and a request that carries it uses that value instead of the +default: a client that sets `deadline_ms` to 10000 is not retried by the gateway past the +client's own deadline, which makes the two settings interact in a way v1 and v2 had no +equivalent for. + +## Rate Limits + +The default rate limit is **1200 requests per minute**. Structured-output requests are +counted as two requests, because the gateway runs a validation pass over the produced object +before returning it; a client that migrates a high-volume v2 workload to structured output +can therefore exhaust its budget at half the expected request count. + +Bursts of up to 120 requests may be issued within one second. The limiter applies reasoning +tokens against a separate budget from requests, so a client with long reasoning turns can +hit the token budget before the request budget. + +## Context Window + +The maximum context window is **65536 tokens**, the largest of the three versions, and the +only one where tool schemas, reasoning tokens and messages are reported as three separate +line items in `usage` rather than folded together. + +A request that exceeds the window is rejected with `400`. v3 does not truncate +automatically, but it does report how many tokens the request would need, which makes a +programmatic retry at a smaller size possible without re-tokenising the input. + +## Structured Output + +`response_format` accepts a JSON schema and a strictness flag. Strict mode is slower but +guarantees the object validates against the schema. Non-strict mode is the default and can +return an object that parses but does not match, which the gateway flags in a +`validation_errors` array rather than failing the request. diff --git a/eval/corpus/coastal-monitoring.md b/eval/corpus/coastal-monitoring.md new file mode 100644 index 0000000..837713d --- /dev/null +++ b/eval/corpus/coastal-monitoring.md @@ -0,0 +1,70 @@ +# Coastal Monitoring Programme + +## Overview + +Coastal monitoring tracks a tidally driven water body at the land margin. Its distinguishing +problem is that the water body moves twice a day: a station is never at the same depth for +two consecutive visits, and a tidal phase that happens to coincide with a sampling run can +dominate the reading. + +The programme covers four shore stations and two offshore buoys. Shore stations are visited; +buoys are instrumented and telemeter. The two kinds of station are reported separately, +because mixing a telemetered series with a visited series is the most common defect in a +coastal dataset. + +## Sampling Interval + +Routine sampling runs **hourly** at the telemetered buoys and **fortnightly** at the shore +stations. The hourly cadence exists to resolve the tidal cycle, which a fortnightly cadence +would alias into a meaningless slow oscillation; a shore station cannot support it because +each visit is a boat trip. + +Every telemetered reading is stamped with its tidal phase. A reading without a phase stamp +is retained but cannot be compared against the shore series, and the quality-control pass +excludes it from any cross-station comparison. + +## Replicate Samples + +Field teams collect **three replicate samples** at each shore station. Three is the standard +for the programme because the boat trip, not the analysis, dominates the cost, so an extra +replicate is nearly free once the team is on site — but only three are taken, because the +shore stations are well mixed and the fourth replicate has never changed a decision. + +Buoys do not take replicates in the sampling sense; they take a burst of 30 readings over +60 seconds and report the median. The median is used rather than the mean because a single +wave splash is a large positive outlier in exactly the quantities a buoy measures. + +## Sensor Depth + +The shore stations carry a sensor at **1 metre below the surface**, and the offshore buoys +carry a sensor at **2 metres below the surface**. The buoy depth is greater because a buoy +in the wave zone is repeatedly lifted and dropped by swell, and a sensor closer to the +surface is out of the water a meaningful fraction of the time. + +Both depths are recorded as depth below the *instantaneous* surface. Coastal sensors are the +only programme where the surface reference changes fast enough to matter within a single +reading, so the timestamp and the depth are recorded together and neither is meaningful +alone. + +## Parameters + +The core parameters are water temperature, salinity, turbidity and wave height. Coastal +adds wave height, which no other programme records, and salinity, which only the estuary +programme also records but for a different reason. + +Wave height is recorded as significant wave height, the mean of the highest third of waves +in the burst, not as the maximum. The maximum is recorded separately and is not used for +trend analysis because it is dominated by rare events. + +## Quality Control + +Telemetered readings are passed through a spike filter that rejects a value more than four +standard deviations from its 24-hour rolling mean. The filter has an override: a reading +that is also accompanied by a wave height above the 99th percentile is retained rather than +rejected, on the grounds that a storm is exactly when an unusual value is most likely to be +real. + +Shore stations use a field blank and a blind duplicate, as the river and reservoir +programmes do. A station whose duplicate differs by more than 25 percent is re-visited, +which is a looser threshold than the reservoir programme's 20 percent because a coastal +station is inherently noisier and a tighter threshold flagged almost every visit. diff --git a/eval/corpus/estuary-monitoring.md b/eval/corpus/estuary-monitoring.md new file mode 100644 index 0000000..696dd00 --- /dev/null +++ b/eval/corpus/estuary-monitoring.md @@ -0,0 +1,67 @@ +# Estuary Monitoring Programme + +## Overview + +Estuary monitoring tracks the mixing zone where a river meets the sea. Its distinguishing +problem is that the water body has a gradient in all three dimensions at once: salinity and +turbidity change sharply over a few kilometres, and the position of that gradient moves with +the tide and with river flow. + +The programme covers two estuaries, each with a transect of five stations running from the +freshwater end to the mouth. A transect is the unit of reporting, not a station, because a +single station in an estuary describes a position in a gradient that has moved by the time +the next station is sampled. + +## Sampling Interval + +Routine sampling runs **daily** at the two lowest stations and **fortnightly** along the +rest of the transect. Daily sampling at the lower stations is used to track the salt wedge, +whose position responds to the tide within hours; the upper transect responds to river flow +over days and does not need it. + +Transect sampling is run on the ebb tide, and every station is occupied within a single ebb +to keep the transect a snapshot rather than a sequence. A transect that overruns its ebb is +discarded and repeated, because a transect sampled across a tidal reversal is not a gradient. + +## Replicate Samples + +Field teams collect **five replicate samples** at each transect station, the highest of any +programme in the network. Five because the estuary gradient means two samples taken a metre +apart can differ more than two samples taken a kilometre apart at a well-mixed site; the +within-station variance is high enough that three replicates do not estimate a mean reliably. + +Replicates are taken as a spatial cross rather than as a sequence: one at the nominal +position and four at 25 metres on each axis. A sequential set of replicates would all sample +the same parcel of water and would understate the variance that matters. + +## Sensor Depth + +Transect stations carry sensors at **2 metres below the surface** and at 1 metre above the +bed, paired so that the vertical salinity difference can be computed directly. The surface +depth is fixed at 2 metres rather than at 1 metre to keep the sensor below the freshwater +lens that floats on the saline layer at the freshwater end. + +The paired depths are the programme's defining feature. A single sensor in an estuary cannot +distinguish a change in salinity from a change in where the halocline sits, and those are +different findings. + +## Parameters + +The core parameters are salinity, turbidity, dissolved oxygen and temperature. Estuary adds +the position of the turbidity maximum, which is the programme's headline product and is not +recorded by any other programme. + +The turbidity maximum is reported as a distance from the freshwater end, not as a turbidity +value. Its position moves several kilometres over a tidal cycle, and a turbidity value +without a position cannot distinguish a stationary maximum from a passing one. + +## Quality Control + +Every transect includes one blind duplicate at a randomly chosen station. Precision is +estimated per transect rather than per station, because the quantity the programme reports +is a gradient and the error that matters is the error in the gradient. + +A transect whose blind duplicate differs by more than 30 percent is repeated. The threshold +is looser than any other programme's, and deliberately so: an estuary's true variance is +genuinely larger, and a tighter threshold would cause every transect to be repeated, which +in practice means none of them are. diff --git a/eval/corpus/groundwater-monitoring.md b/eval/corpus/groundwater-monitoring.md new file mode 100644 index 0000000..ca89489 --- /dev/null +++ b/eval/corpus/groundwater-monitoring.md @@ -0,0 +1,68 @@ +# Groundwater Monitoring Programme + +## Overview + +Groundwater monitoring tracks water below the surface rather than water in a channel. Its +distinguishing problem is that the water body cannot be seen, so every measurement is a +sample from a borehole that may or may not be connected to the aquifer the programme intends +to describe. + +The programme covers nine boreholes across three catchments. Eight are monitoring boreholes +used only for measurement; one is a supply borehole that is also pumped, and its readings +are reported separately because a pumped borehole's level reflects recent abstraction as +much as the aquifer. + +## Sampling Interval + +Routine sampling runs **monthly**, the least frequent cadence in the network. Groundwater +responds to rainfall over weeks to months, and a more frequent cadence measures the borehole +rather than the aquifer: a fortnightly series is dominated by the borehole's own equilibration +after each visit, which takes several days. + +Supply boreholes are additionally sampled immediately before and after each pumping cycle. +Those samples are paired and reported as a drawdown and recovery pair, not as routine +readings. + +## Replicate Samples + +Field teams collect **two replicate samples** at each borehole, the lowest of any programme +in the network. Two rather than three because groundwater is well mixed by the time it +reaches a borehole, and the dominant error is not within-sample variance but the borehole's +connection to the aquifer — an error that an extra replicate does not reduce. + +Because two replicates give no way to identify an outlier, any borehole whose two replicates +disagree by more than 10 percent is re-sampled entirely rather than resolved statistically. +Ten percent is tighter than any other programme's threshold precisely because there is no +third replicate to arbitrate. + +## Sensor Depth + +Boreholes carry a pressure transducer at **15 metres below the water table**, measured on the +first visit and recorded as an absolute elevation so that a falling water table does not +silently change the depth being sampled. The depth is the largest in the network by an order +of magnitude, which is a property of boreholes rather than a choice. + +The transducer records water level continuously. Water level is the programme's primary +quantity; chemistry is sampled monthly and level is logged every 15 minutes, and the two are +reported on separate axes because a single chart of both conceals the pattern in either. + +## Parameters + +The core parameters are water level, temperature, specific conductance and nitrate. Groundwater +adds water level, which is the only parameter in the network that is logged continuously +rather than sampled. + +Nitrate is recorded as the primary indicator of agricultural loading, and it is the parameter +the programme was established to track. Specific conductance is recorded as a cheap proxy for +salinity intrusion in the two coastal boreholes. + +## Quality Control + +Every visit records the water level before and after purging. A borehole in which the level +does not recover within 24 hours of purging is flagged, because the purge has drawn from a +body of water that is not being recharged and the sample may not represent the aquifer. + +Blind duplicates are submitted quarterly rather than per visit, which is the sparsest +verification in the network. The programme's position is that its dominant uncertainty is the +borehole-to-aquifer connection, which a duplicate cannot measure, so spending budget on more +duplicates would buy precision the programme cannot use. diff --git a/eval/corpus/reservoir-monitoring.md b/eval/corpus/reservoir-monitoring.md new file mode 100644 index 0000000..389eb79 --- /dev/null +++ b/eval/corpus/reservoir-monitoring.md @@ -0,0 +1,69 @@ +# Reservoir Monitoring Programme + +## Overview + +Reservoir monitoring tracks a standing water body whose level is managed rather than +natural. The programme's distinguishing problem is that the water body has an operator: +every measurement has to be paired with the release schedule, or a change in a reading is +indistinguishable from a change in how the reservoir was run that week. + +The programme covers three reservoirs in the upland basin. Each has a fixed monitoring +station at the dam face and two floating stations whose position is recorded on every +visit, because a reservoir's surface area changes enough over a season to move a floating +station hundreds of metres without anyone touching it. + +## Sampling Interval + +Routine sampling runs **fortnightly**, on a fixed Tuesday, so that the interval is +consistent across sites and operators. A fortnightly cadence is a deliberate compromise: +weekly was found to double cost without changing any trend, and monthly aliased against +the operator's own drawdown cycle, which is also roughly monthly and made the series very +hard to interpret. + +Event sampling is triggered by any release exceeding 20 percent of live capacity in a +single day. Event samples are additional to the routine cadence and are labelled with the +release event rather than with the calendar. + +## Replicate Samples + +Field teams collect **four replicate samples** at each station to control for local +variability. Four rather than three because the reservoir stations sit in a drawdown zone +where wind-driven mixing produces an occasional outlier; with three replicates a single +outlier is a third of the mean, and with four it can be identified and excluded on a stated +rule rather than discarded by feel. + +Replicates are taken within a 15-minute window. A replicate that falls outside that window +is recorded but excluded from the mean, because the reservoir can stratify and destratify on +that timescale in summer. + +## Sensor Depth + +The fixed station carries a sensor string at **5 metres below the surface**, and the two +floating stations carry a single sensor at **1 metre below the surface**. The asymmetry is +intentional: the dam face is deep and well mixed, while the floating stations are in the +drawdown zone where the interesting gradient is in the top metre. + +Depth is recorded as depth below the *current* surface, not below full capacity. Because the +surface moves, a sensor on a fixed string is at a different absolute elevation at different +times, and the programme records both so that a reader can reconstruct which was meant. + +## Parameters + +The core parameters are water temperature, dissolved oxygen, turbidity and chlorophyll-a. +Reservoir-specific parameters are residence time and drawdown rate, neither of which the +river or lake programmes record because neither has an operator-controlled outlet. + +Turbidity is recorded as the primary indicator of sediment resuspension during drawdown, +which is the process the programme exists to quantify. Chlorophyll-a is recorded as the +primary indicator of the algal response to nutrient loading. + +## Quality Control + +Every routine visit includes one field blank and one duplicate submitted blind. The +duplicate is used to estimate within-station precision; the blank is used to detect +contamination introduced by the sampling kit rather than by the reservoir. + +A station whose blind duplicate differs by more than 20 percent is flagged and re-visited +within seven days. Two consecutive flags retire the station's sensor string, because the +most common cause of a persistent discrepancy is a drifting sensor rather than a genuinely +patchy water body. diff --git a/eval/questions.jsonl b/eval/questions.jsonl index c9a3d44..10b031a 100644 --- a/eval/questions.jsonl +++ b/eval/questions.jsonl @@ -42,3 +42,22 @@ {"id":"q042","question":"蜜蜂大致从什么温度开始出巢觅食?","type":"cross-lingual","relevant":[{"document":"orchard-pollination.md","page":null,"block":4,"quote":"above roughly twelve degrees Celsius"}]} {"id":"q043","question":"花期遭遇晚霜,主要受损的是什么?","type":"cross-lingual","relevant":[{"document":"orchard-pollination.md","page":null,"block":6,"quote":"destroys the flower's ovary rather than the petals"}]} {"id":"q044","question":"为什么成片的树冠比孤立的树降温效果更好?","type":"cross-lingual","relevant":[{"document":"urban-canopy.md","page":null,"block":4,"quote":"asphalt stores far more heat during the day"}]} +{"id":"q045","type":"exact","question":"What is the default retry count in Gateway API v2?","relevant":[{"document":"api-gateway-v2.md","page":null,"block":9,"quote":"recommended retry count is **5**"}]} +{"id":"q046","type":"hard-negative","question":"What is the default request timeout of Gateway API v1?","relevant":[{"document":"api-gateway-v1.md","page":null,"block":8,"quote":"default request timeout is **30000 milliseconds**"}]} +{"id":"q047","type":"hard-negative","question":"What is the context window of Gateway API v3, in tokens?","relevant":[{"document":"api-gateway-v3.md","page":null,"block":14,"quote":"maximum context window is **65536 tokens**"}]} +{"id":"q048","type":"hard-negative","question":"What is the default rate limit of Gateway API v2?","relevant":[{"document":"api-gateway-v2.md","page":null,"block":12,"quote":"default rate limit is **3000 requests per minute**"}]} +{"id":"q049","type":"hard-negative","question":"What is the default request timeout of Gateway API v3?","relevant":[{"document":"api-gateway-v3.md","page":null,"block":8,"quote":"default request timeout is **45000 milliseconds**"}]} +{"id":"q050","type":"semantic","question":"Which Gateway API version waits longest between retries after a failed request?","relevant":[{"document":"api-gateway-v2.md","page":null,"block":9,"quote":"backoff base of **2000 milliseconds**"}]} +{"id":"q051","type":"multi-hop","question":"Compare the default timeouts and rate limits of Gateway API v1 and v3.","relevant":[{"document":"api-gateway-v1.md","page":null,"block":8,"quote":"default request timeout is **30000 milliseconds**"},{"document":"api-gateway-v1.md","page":null,"block":12,"quote":"default rate limit is **600 requests per minute**"},{"document":"api-gateway-v3.md","page":null,"block":8,"quote":"default request timeout is **45000 milliseconds**"},{"document":"api-gateway-v3.md","page":null,"block":11,"quote":"default rate limit is **1200 requests per minute**"}]} +{"id":"q052","type":"multi-hop","question":"Compare the retry count and context window of Gateway API v2 and v3.","relevant":[{"document":"api-gateway-v2.md","page":null,"block":9,"quote":"recommended retry count is **5**"},{"document":"api-gateway-v2.md","page":null,"block":15,"quote":"maximum context window is **32768 tokens**"},{"document":"api-gateway-v3.md","page":null,"block":9,"quote":"recommended retry count is **3**"},{"document":"api-gateway-v3.md","page":null,"block":14,"quote":"maximum context window is **65536 tokens**"}]} +{"id":"q053","type":"unanswerable","answerable":false,"question":"How much GPU memory does Gateway API v2 require to serve a request?","relevant":[]} +{"id":"q054","type":"unanswerable","answerable":false,"question":"What is the monthly subscription price of Gateway API v3?","relevant":[]} +{"id":"q055","type":"exact","question":"How many replicate samples are collected at each reservoir monitoring station?","relevant":[{"document":"reservoir-monitoring.md","page":null,"block":8,"quote":"four replicate samples"}]} +{"id":"q056","type":"hard-negative","question":"How many replicate samples are collected at each estuary transect station?","relevant":[{"document":"estuary-monitoring.md","page":null,"block":8,"quote":"five replicate samples"}]} +{"id":"q057","type":"hard-negative","question":"At what depth below the surface does the coastal programme place its shore-station sensor?","relevant":[{"document":"coastal-monitoring.md","page":null,"block":11,"quote":"1 metre below the surface"}]} +{"id":"q058","type":"hard-negative","question":"At what depth below the water table do groundwater boreholes carry their pressure transducer?","relevant":[{"document":"groundwater-monitoring.md","page":null,"block":11,"quote":"15 metres below the water table"}]} +{"id":"q059","type":"semantic","question":"Which monitoring programme samples most frequently?","relevant":[{"document":"coastal-monitoring.md","page":null,"block":5,"quote":"Routine sampling runs **hourly**"}]} +{"id":"q060","type":"multi-hop","question":"Compare the sampling interval and replicate count of the reservoir and groundwater programmes.","relevant":[{"document":"reservoir-monitoring.md","page":null,"block":5,"quote":"fortnightly"},{"document":"reservoir-monitoring.md","page":null,"block":8,"quote":"four replicate samples"},{"document":"groundwater-monitoring.md","page":null,"block":5,"quote":"monthly"},{"document":"groundwater-monitoring.md","page":null,"block":8,"quote":"two replicate samples"}]} +{"id":"q061","type":"semantic","question":"Why does the estuary programme take more replicate samples than the other monitoring programmes?","relevant":[{"document":"estuary-monitoring.md","page":null,"block":8,"quote":"within-station variance"}]} +{"id":"q062","type":"unanswerable","answerable":false,"question":"What is the annual operating cost of the coastal monitoring buoys?","relevant":[]} +{"id":"q063","type":"unanswerable","answerable":false,"question":"How many litres per second does the reservoir release downstream on a typical day?","relevant":[]} diff --git a/eval/splits.json b/eval/splits.json index c903be1..021e265 100644 --- a/eval/splits.json +++ b/eval/splits.json @@ -42,5 +42,24 @@ "q041": "test", "q042": "test", "q043": "validation", - "q044": "test" + "q044": "test", + "q045": "validation", + "q046": "test", + "q047": "validation", + "q048": "test", + "q049": "test", + "q050": "validation", + "q051": "test", + "q052": "test", + "q053": "validation", + "q054": "test", + "q055": "validation", + "q056": "test", + "q057": "validation", + "q058": "test", + "q059": "test", + "q060": "validation", + "q061": "test", + "q062": "validation", + "q063": "test" } diff --git a/package.json b/package.json index 4d9134d..77a6406 100644 --- a/package.json +++ b/package.json @@ -43,7 +43,8 @@ "db:studio": "drizzle-kit studio", "eval:retrieval": "npm run build && node --experimental-transform-types --disable-warning=ExperimentalWarning --disable-warning=MODULE_TYPELESS_PACKAGE_JSON scripts/eval-retrieval.mjs", "eval:threshold": "npm run build && node scripts/eval-threshold.mjs", - "eval:sweep": "npm run build && node scripts/eval-sweep.mjs" + "eval:sweep": "npm run build && node scripts/eval-sweep.mjs", + "eval:blocks": "node --experimental-transform-types --import ./test/ts-resolve.mjs --disable-warning=ExperimentalWarning --disable-warning=MODULE_TYPELESS_PACKAGE_JSON scripts/eval-blocks.mjs" }, "//test": [ "`node --test` strips TypeScript types rather than compiling them, and strip-only", diff --git a/scripts/eval-blocks.mjs b/scripts/eval-blocks.mjs new file mode 100644 index 0000000..ad861f9 --- /dev/null +++ b/scripts/eval-blocks.mjs @@ -0,0 +1,54 @@ +#!/usr/bin/env node +/** + * Print the block ordinals the harness will resolve ground truth against (#192). + * + * Ground truth in `eval/questions.jsonl` is expressed as `document` + `block` + `quote`, + * and `block` is the `document_blocks.order` the ingestion pipeline produced — not a line + * number and not a paragraph index a human counted. Authoring a corpus by hand and then + * guessing those ordinals is how a dataset quietly drifts, so this prints the real ones + * through the **same loader and block builder the harness uses**. + * + * Usage: + * node --experimental-transform-types scripts/eval-blocks.mjs eval/corpus/river-monitoring.md + * node --experimental-transform-types scripts/eval-blocks.mjs eval/corpus/*.md + */ + +import { readFile, readdir } from 'node:fs/promises' +import { extname, join, resolve } from 'node:path' +import { MarkdownLoader } from '../src/main/services/loaders/MarkdownLoader.ts' +import { buildDocumentBlocks } from '../src/main/services/blocks/documentBlocks.ts' + +const targets = process.argv.slice(2) +if (targets.length === 0) { + console.error('usage: node --experimental-transform-types scripts/eval-blocks.mjs [...]') + process.exit(1) +} + +const loader = new MarkdownLoader() + +async function printFile(path) { + const buffer = await readFile(path) + const result = await loader.loadFromBuffer(buffer) + const blocks = buildDocumentBlocks({ content: result.content, structure: result.structure }) + + console.log(`\n${path} — ${blocks.length} blocks`) + for (const block of blocks) { + const text = block.text.replace(/\s+/g, ' ').trim() + const shown = text.length > 96 ? `${text.slice(0, 93)}...` : text + console.log(` ${String(block.order).padStart(3)} ${block.kind.padEnd(9)} ${shown}`) + } + return blocks.length +} + +let total = 0 +for (const target of targets) { + const path = resolve(target) + if (extname(path) === '.md') { + total += await printFile(path) + } else { + // A directory: every markdown file in it, which is the whole corpus case. + const entries = (await readdir(path)).filter((name) => name.endsWith('.md')).sort() + for (const entry of entries) total += await printFile(join(path, entry)) + } +} +console.log(`\ntotal blocks: ${total}`)