diff --git a/docs/eval/baseline-v1.6.json b/docs/eval/baseline-v1.6.json index 5f4b48a..4c41432 100644 --- a/docs/eval/baseline-v1.6.json +++ b/docs/eval/baseline-v1.6.json @@ -17,34 +17,34 @@ "threshold": 0.5, "corpus": "eval/corpus", "documents": 13, - "questions": 36, + "questions": 44, "chunkCount": 19 }, "metrics": { - "recallAt1": 0.833333, - "recallAt5": 1, + "recallAt1": 0.657895, + "recallAt5": 0.921053, "recallAt10": 1, - "mrr": 0.927778, - "ndcgAt10": 0.94375, - "hitRateAt5": 1, - "mapAt10": 0.922222, - "contextPrecision": 0.355556, - "contextRecall": 1 + "mrr": 0.800909, + "ndcgAt10": 0.84764, + "hitRateAt5": 0.921053, + "mapAt10": 0.796523, + "contextPrecision": 0.324561, + "contextRecall": 0.921053 }, "byType": [ { "type": "cross-lingual", - "questions": 1, + "questions": 9, "metrics": { "recallAt1": 0, - "recallAt5": 1, + "recallAt5": 0.666667, "recallAt10": 1, - "mrr": 0.5, - "ndcgAt10": 0.63093, - "hitRateAt5": 1, - "mapAt10": 0.5, - "contextPrecision": 0.333333, - "contextRecall": 1 + "mrr": 0.344577, + "ndcgAt10": 0.503192, + "hitRateAt5": 0.666667, + "mapAt10": 0.344577, + "contextPrecision": 0.222222, + "contextRecall": 0.666667 } }, { @@ -1294,6 +1294,270 @@ [], [] ] + }, + { + "id": "q037", + "question": "推移质为什么比悬移质更难测?", + "type": "cross-lingual", + "answerable": true, + "firstRelevantRank": 8, + "relevantCount": 1, + "retrievedCount": 19, + "contextChars": 2084, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q038", + "question": "河流监测中,每个站点要采几份平行样?", + "type": "cross-lingual", + "answerable": true, + "firstRelevantRank": 7, + "relevantCount": 1, + "retrievedCount": 19, + "contextChars": 1480, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q039", + "question": "设计锂离子电芯时最核心的取舍是什么?", + "type": "cross-lingual", + "answerable": true, + "firstRelevantRank": 2, + "relevantCount": 1, + "retrievedCount": 19, + "contextChars": 1994, + "matchesByRank": [ + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q040", + "question": "富镍电池为什么对散热要求更高?", + "type": "cross-lingual", + "answerable": true, + "firstRelevantRank": 2, + "relevantCount": 1, + "retrievedCount": 19, + "contextChars": 1994, + "matchesByRank": [ + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q041", + "question": "电芯隔膜一旦熔化会导致什么后果?", + "type": "cross-lingual", + "answerable": true, + "firstRelevantRank": 3, + "relevantCount": 1, + "retrievedCount": 19, + "contextChars": 1449, + "matchesByRank": [ + [], + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q042", + "question": "蜜蜂大致从什么温度开始出巢觅食?", + "type": "cross-lingual", + "answerable": true, + "firstRelevantRank": 3, + "relevantCount": 1, + "retrievedCount": 19, + "contextChars": 1321, + "matchesByRank": [ + [], + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q043", + "question": "花期遭遇晚霜,主要受损的是什么?", + "type": "cross-lingual", + "answerable": true, + "firstRelevantRank": 2, + "relevantCount": 1, + "retrievedCount": 19, + "contextChars": 956, + "matchesByRank": [ + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q044", + "question": "为什么成片的树冠比孤立的树降温效果更好?", + "type": "cross-lingual", + "answerable": true, + "firstRelevantRank": 6, + "relevantCount": 1, + "retrievedCount": 19, + "contextChars": 1447, + "matchesByRank": [ + [], + [], + [], + [], + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] } ] } diff --git a/docs/eval/baseline-v1.6.md b/docs/eval/baseline-v1.6.md index 23fffc2..2fcaf5d 100644 --- a/docs/eval/baseline-v1.6.md +++ b/docs/eval/baseline-v1.6.md @@ -12,22 +12,22 @@ Generated by `npm run eval`. The numbers below are harness output — do not edi | Ranks | `candidateK=20, threshold=0.5` | | Context width | `contextK=3` | | Corpus | `eval/corpus` (13 documents) | -| Split | `all` (36 questions, 30 answerable) | +| Split | `all` (44 questions, 38 answerable) | | Index size | 19 chunks | ## Metrics | Metric | Value | | --- | --- | -| Recall@1 | 0.8333 | -| Recall@5 | 1.0000 | +| Recall@1 | 0.6579 | +| Recall@5 | 0.9211 | | Recall@10 | 1.0000 | -| MRR | 0.9278 | -| nDCG@10 | 0.9437 | -| Hit rate@5 | 1.0000 | -| MAP@10 | 0.9222 | -| Context precision@3 | 0.3556 | -| Context recall@3 | 1.0000 | +| MRR | 0.8009 | +| nDCG@10 | 0.8476 | +| Hit rate@5 | 0.9211 | +| MAP@10 | 0.7965 | +| Context precision@3 | 0.3246 | +| Context recall@3 | 0.9211 | ### By query type @@ -37,7 +37,7 @@ The type comes from `type` in `questions.jsonl`; untagged questions report as | Type | Questions | Recall@5 | nDCG@10 | Hit rate@5 | MAP@10 | | --- | --- | --- | --- | --- | --- | -| cross-lingual | 1 | 1.0000 | 0.6309 | 1.0000 | 0.5000 | +| cross-lingual | 9 | 0.6667 | 0.5032 | 0.6667 | 0.3446 | | exact | 6 | 1.0000 | 1.0000 | 1.0000 | 1.0000 | | multi-hop | 2 | 1.0000 | 0.9599 | 1.0000 | 0.9167 | | semantic | 18 | 1.0000 | 0.9312 | 1.0000 | 0.9074 | @@ -57,8 +57,8 @@ was pulled in anyway. This is the row a threshold decision should move. | Returned no results | 0.0000 (0/6) | | Mean passages retrieved | 19.00 | -Timing is informational only and is **not** frozen: indexing 1496 ms, query -p50 11.42 ms, p95 17.31 ms on the +Timing is informational only and is **not** frozen: indexing 1552 ms, query +p50 11.82 ms, p95 14.45 ms on the machine that produced this file. Timing and index size depend on hardware and on the corpus, so they must never be the reason two runs differ. diff --git a/docs/eval/retrieval-v1.6.json b/docs/eval/retrieval-v1.6.json index 9bf0d74..cdfee1c 100644 --- a/docs/eval/retrieval-v1.6.json +++ b/docs/eval/retrieval-v1.6.json @@ -8,19 +8,19 @@ "chunking": "1000/100", "chunkCount": 19, "split": "all", - "questions": 36, - "answerableCount": 30, - "recallAt1": 0.833333, - "recallAt5": 1, + "questions": 44, + "answerableCount": 38, + "recallAt1": 0.657895, + "recallAt5": 0.921053, "recallAt10": 1, - "mrr": 0.927778, - "ndcgAt10": 0.94375, - "hitRateAt5": 1, - "mapAt10": 0.922222, - "contextPrecision": 0.355556, - "contextRecall": 1, - "indexingMs": 1518, - "latencyP95Ms": 14.2 + "mrr": 0.800909, + "ndcgAt10": 0.84764, + "hitRateAt5": 0.921053, + "mapAt10": 0.796523, + "contextPrecision": 0.324561, + "contextRecall": 0.921053, + "indexingMs": 1532, + "latencyP95Ms": 14.38 }, { "id": "sparse", @@ -28,19 +28,19 @@ "chunking": "1000/100", "chunkCount": 19, "split": "all", - "questions": 36, - "answerableCount": 30, - "recallAt1": 0.733333, - "recallAt5": 0.866667, - "recallAt10": 0.866667, - "mrr": 0.805556, - "ndcgAt10": 0.818355, - "hitRateAt5": 0.866667, - "mapAt10": 0.8, - "contextPrecision": 0.311111, - "contextRecall": 0.866667, - "indexingMs": 1513, - "latencyP95Ms": 2.46 + "questions": 44, + "answerableCount": 38, + "recallAt1": 0.578947, + "recallAt5": 0.684211, + "recallAt10": 0.684211, + "mrr": 0.635965, + "ndcgAt10": 0.64607, + "hitRateAt5": 0.684211, + "mapAt10": 0.631579, + "contextPrecision": 0.245614, + "contextRecall": 0.684211, + "indexingMs": 1529, + "latencyP95Ms": 1.85 }, { "id": "hybrid", @@ -48,19 +48,19 @@ "chunking": "1000/100", "chunkCount": 19, "split": "all", - "questions": 36, - "answerableCount": 30, - "recallAt1": 0.866667, - "recallAt5": 1, + "questions": 44, + "answerableCount": 38, + "recallAt1": 0.684211, + "recallAt5": 0.921053, "recallAt10": 1, - "mrr": 0.944444, - "ndcgAt10": 0.956053, - "hitRateAt5": 1, - "mapAt10": 0.938889, - "contextPrecision": 0.355556, - "contextRecall": 1, - "indexingMs": 1557, - "latencyP95Ms": 19.55 + "mrr": 0.814066, + "ndcgAt10": 0.857352, + "hitRateAt5": 0.921053, + "mapAt10": 0.80968, + "contextPrecision": 0.324561, + "contextRecall": 0.921053, + "indexingMs": 1579, + "latencyP95Ms": 17.04 } ] } diff --git a/docs/eval/retrieval-v1.6.md b/docs/eval/retrieval-v1.6.md index a091483..c00fe52 100644 --- a/docs/eval/retrieval-v1.6.md +++ b/docs/eval/retrieval-v1.6.md @@ -5,14 +5,14 @@ Generated by `node scripts/eval-retrieval.mjs`. Numbers are harness output; do n ## What was measured Every strategy runs the real RAG eval harness against the same corpus and the same questions -as `baseline-v1.6.json` (split `all`, 36 questions of which -30 are answerable), with chunking held fixed at 1000/100. Only the retrieval strategy changes. +as `baseline-v1.6.json` (split `all`, 44 questions of which +38 are answerable), with chunking held fixed at 1000/100. Only the retrieval strategy changes. | Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Context P | Query p95 | | --- | --- | --- | --- | --- | --- | --- | --- | -| dense (vector) | 0.8333 | 1.0000 | 0.9278 | 0.9437 | 0.9222 | 0.3556 | 14.20 ms | -| sparse (BM25) | 0.7333 | 0.8667 | 0.8056 | 0.8184 | 0.8000 | 0.3111 | 2.46 ms | -| hybrid (RRF of dense + BM25) | 0.8667 | 1.0000 | 0.9444 | 0.9561 | 0.9389 | 0.3556 | 19.55 ms | +| dense (vector) | 0.6579 | 0.9211 | 0.8009 | 0.8476 | 0.7965 | 0.3246 | 14.38 ms | +| sparse (BM25) | 0.5789 | 0.6842 | 0.6360 | 0.6461 | 0.6316 | 0.2456 | 1.85 ms | +| hybrid (RRF of dense + BM25) | 0.6842 | 0.9211 | 0.8141 | 0.8574 | 0.8097 | 0.3246 | 17.04 ms | ## Not evaluated @@ -23,7 +23,7 @@ is. ## Saturation -Saturated (no headroom, so they cannot decide anything): `recallAt5`. The deciding metric on this corpus is `ndcgAt10`. A saturated metric is still reported, because "this corpus cannot move it" is itself +No metric in the rule is saturated on this corpus. The deciding metric on this corpus is `recallAt5`. A saturated metric is still reported, because "this corpus cannot move it" is itself information; it is just not allowed to decide the comparison. ## Adoption rule @@ -38,9 +38,7 @@ information; it is just not allowed to decide the comparison. ## Outcome -`hybrid (RRF of dense + BM25)` **clears the rule**: it improves the deciding metric `ndcgAt10` (0.9561 vs dense 0.9437) and regresses none of `recallAt5`, `ndcgAt10`, `mrr`, `mapAt10`. Saturated (no headroom, so they cannot decide anything): `recallAt5`. - -Changing the shipped default is a separate decision, and this script does not make it — it reports the measurement. +No strategy cleared the rule. The deciding metric was `recallAt5` (dense 0.9211); the strategies either failed to improve it or regressed another metric. **Dense stays the default.** A negative result is the point of the experiment: it is the measurement that says the extra machinery is not worth its cost on this corpus, not a failure to deliver. No metric in the rule is saturated on this corpus. ## Reproduce diff --git a/docs/eval/sweep-v1.6.json b/docs/eval/sweep-v1.6.json index 67cdb5b..d027f58 100644 --- a/docs/eval/sweep-v1.6.json +++ b/docs/eval/sweep-v1.6.json @@ -5,375 +5,375 @@ "strategy": "dense", "candidateK": 5, "contextK": 3, - "recallAt5": 1, - "ndcgAt10": 0.94375, - "mapAt10": 0.922222, - "contextPrecision": 0.355556, - "contextRecall": 1, + "recallAt5": 0.921053, + "ndcgAt10": 0.821192, + "mapAt10": 0.785088, + "contextPrecision": 0.324561, + "contextRecall": 0.921053, "noResultRate": 0, - "meanContextChars": 2078.9333333333334, + "meanContextChars": 1976.1315789473683, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 5, "chunkCount": 19, - "latencyP95Ms": 98.32 + "latencyP95Ms": 13.57 }, { "strategy": "dense", "candidateK": 5, "contextK": 5, - "recallAt5": 1, - "ndcgAt10": 0.94375, - "mapAt10": 0.922222, - "contextPrecision": 0.213333, - "contextRecall": 1, + "recallAt5": 0.921053, + "ndcgAt10": 0.821192, + "mapAt10": 0.785088, + "contextPrecision": 0.194737, + "contextRecall": 0.921053, "noResultRate": 0, - "meanContextChars": 3326.8333333333335, + "meanContextChars": 3188.1315789473683, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 5, "chunkCount": 19, - "latencyP95Ms": 96.94 + "latencyP95Ms": 14.14 }, { "strategy": "dense", "candidateK": 10, "contextK": 3, - "recallAt5": 1, - "ndcgAt10": 0.94375, - "mapAt10": 0.922222, - "contextPrecision": 0.355556, - "contextRecall": 1, + "recallAt5": 0.921053, + "ndcgAt10": 0.84764, + "mapAt10": 0.796523, + "contextPrecision": 0.324561, + "contextRecall": 0.921053, "noResultRate": 0, - "meanContextChars": 2078.9333333333334, + "meanContextChars": 1976.1315789473683, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 10, "chunkCount": 19, - "latencyP95Ms": 84.19 + "latencyP95Ms": 13.3 }, { "strategy": "dense", "candidateK": 10, "contextK": 5, - "recallAt5": 1, - "ndcgAt10": 0.94375, - "mapAt10": 0.922222, - "contextPrecision": 0.213333, - "contextRecall": 1, + "recallAt5": 0.921053, + "ndcgAt10": 0.84764, + "mapAt10": 0.796523, + "contextPrecision": 0.194737, + "contextRecall": 0.921053, "noResultRate": 0, - "meanContextChars": 3326.8333333333335, + "meanContextChars": 3188.1315789473683, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 10, "chunkCount": 19, - "latencyP95Ms": 95.07 + "latencyP95Ms": 12.61 }, { "strategy": "dense", "candidateK": 10, "contextK": 8, - "recallAt5": 1, - "ndcgAt10": 0.94375, - "mapAt10": 0.922222, - "contextPrecision": 0.133333, + "recallAt5": 0.921053, + "ndcgAt10": 0.84764, + "mapAt10": 0.796523, + "contextPrecision": 0.131579, "contextRecall": 1, "noResultRate": 0, - "meanContextChars": 5311.633333333333, + "meanContextChars": 5151.289473684211, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 10, "chunkCount": 19, - "latencyP95Ms": 91.67 + "latencyP95Ms": 13.99 }, { "strategy": "dense", "candidateK": 20, "contextK": 3, - "recallAt5": 1, - "ndcgAt10": 0.94375, - "mapAt10": 0.922222, - "contextPrecision": 0.355556, - "contextRecall": 1, + "recallAt5": 0.921053, + "ndcgAt10": 0.84764, + "mapAt10": 0.796523, + "contextPrecision": 0.324561, + "contextRecall": 0.921053, "noResultRate": 0, - "meanContextChars": 2078.9333333333334, + "meanContextChars": 1976.1315789473683, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 16.2 + "latencyP95Ms": 13.61 }, { "strategy": "dense", "candidateK": 20, "contextK": 5, - "recallAt5": 1, - "ndcgAt10": 0.94375, - "mapAt10": 0.922222, - "contextPrecision": 0.213333, - "contextRecall": 1, + "recallAt5": 0.921053, + "ndcgAt10": 0.84764, + "mapAt10": 0.796523, + "contextPrecision": 0.194737, + "contextRecall": 0.921053, "noResultRate": 0, - "meanContextChars": 3326.8333333333335, + "meanContextChars": 3188.1315789473683, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 14.19 + "latencyP95Ms": 14.62 }, { "strategy": "dense", "candidateK": 20, "contextK": 8, - "recallAt5": 1, - "ndcgAt10": 0.94375, - "mapAt10": 0.922222, - "contextPrecision": 0.133333, + "recallAt5": 0.921053, + "ndcgAt10": 0.84764, + "mapAt10": 0.796523, + "contextPrecision": 0.131579, "contextRecall": 1, "noResultRate": 0, - "meanContextChars": 5311.633333333333, + "meanContextChars": 5151.289473684211, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 20.67 + "latencyP95Ms": 15.2 }, { "strategy": "dense", "candidateK": 40, "contextK": 3, - "recallAt5": 1, - "ndcgAt10": 0.94375, - "mapAt10": 0.922222, - "contextPrecision": 0.355556, - "contextRecall": 1, + "recallAt5": 0.921053, + "ndcgAt10": 0.84764, + "mapAt10": 0.796523, + "contextPrecision": 0.324561, + "contextRecall": 0.921053, "noResultRate": 0, - "meanContextChars": 2078.9333333333334, + "meanContextChars": 1976.1315789473683, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 19.91 + "latencyP95Ms": 13.66 }, { "strategy": "dense", "candidateK": 40, "contextK": 5, - "recallAt5": 1, - "ndcgAt10": 0.94375, - "mapAt10": 0.922222, - "contextPrecision": 0.213333, - "contextRecall": 1, + "recallAt5": 0.921053, + "ndcgAt10": 0.84764, + "mapAt10": 0.796523, + "contextPrecision": 0.194737, + "contextRecall": 0.921053, "noResultRate": 0, - "meanContextChars": 3326.8333333333335, + "meanContextChars": 3188.1315789473683, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 18.65 + "latencyP95Ms": 14.4 }, { "strategy": "dense", "candidateK": 40, "contextK": 8, - "recallAt5": 1, - "ndcgAt10": 0.94375, - "mapAt10": 0.922222, - "contextPrecision": 0.133333, + "recallAt5": 0.921053, + "ndcgAt10": 0.84764, + "mapAt10": 0.796523, + "contextPrecision": 0.131579, "contextRecall": 1, "noResultRate": 0, - "meanContextChars": 5311.633333333333, + "meanContextChars": 5151.289473684211, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 18.52 + "latencyP95Ms": 14.39 }, { "strategy": "hybrid", "candidateK": 5, "contextK": 3, - "recallAt5": 1, - "ndcgAt10": 0.956053, - "mapAt10": 0.938889, - "contextPrecision": 0.355556, - "contextRecall": 1, + "recallAt5": 0.921053, + "ndcgAt10": 0.830904, + "mapAt10": 0.798246, + "contextPrecision": 0.324561, + "contextRecall": 0.921053, "noResultRate": 0, - "meanContextChars": 2044.1333333333334, + "meanContextChars": 1948.657894736842, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 5, "chunkCount": 19, - "latencyP95Ms": 15 + "latencyP95Ms": 16.98 }, { "strategy": "hybrid", "candidateK": 5, "contextK": 5, - "recallAt5": 1, - "ndcgAt10": 0.956053, - "mapAt10": 0.938889, - "contextPrecision": 0.213333, - "contextRecall": 1, + "recallAt5": 0.921053, + "ndcgAt10": 0.830904, + "mapAt10": 0.798246, + "contextPrecision": 0.194737, + "contextRecall": 0.921053, "noResultRate": 0, - "meanContextChars": 3361.6, + "meanContextChars": 3215.5789473684213, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 5, "chunkCount": 19, - "latencyP95Ms": 14.4 + "latencyP95Ms": 16.11 }, { "strategy": "hybrid", "candidateK": 10, "contextK": 3, - "recallAt5": 1, - "ndcgAt10": 0.956053, - "mapAt10": 0.938889, - "contextPrecision": 0.355556, - "contextRecall": 1, + "recallAt5": 0.921053, + "ndcgAt10": 0.857352, + "mapAt10": 0.80968, + "contextPrecision": 0.324561, + "contextRecall": 0.921053, "noResultRate": 0, - "meanContextChars": 2062.733333333333, + "meanContextChars": 1963.342105263158, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 10, "chunkCount": 19, - "latencyP95Ms": 18.54 + "latencyP95Ms": 16.46 }, { "strategy": "hybrid", "candidateK": 10, "contextK": 5, - "recallAt5": 1, - "ndcgAt10": 0.956053, - "mapAt10": 0.938889, - "contextPrecision": 0.213333, - "contextRecall": 1, + "recallAt5": 0.921053, + "ndcgAt10": 0.857352, + "mapAt10": 0.80968, + "contextPrecision": 0.194737, + "contextRecall": 0.921053, "noResultRate": 0, - "meanContextChars": 3525.633333333333, + "meanContextChars": 3332.9473684210525, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 10, "chunkCount": 19, - "latencyP95Ms": 16.13 + "latencyP95Ms": 14.83 }, { "strategy": "hybrid", "candidateK": 10, "contextK": 8, - "recallAt5": 1, - "ndcgAt10": 0.956053, - "mapAt10": 0.938889, - "contextPrecision": 0.133333, + "recallAt5": 0.921053, + "ndcgAt10": 0.857352, + "mapAt10": 0.80968, + "contextPrecision": 0.131579, "contextRecall": 1, "noResultRate": 0, - "meanContextChars": 5525.166666666667, + "meanContextChars": 5319.868421052632, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 10, "chunkCount": 19, - "latencyP95Ms": 18.95 + "latencyP95Ms": 17.38 }, { "strategy": "hybrid", "candidateK": 20, "contextK": 3, - "recallAt5": 1, - "ndcgAt10": 0.956053, - "mapAt10": 0.938889, - "contextPrecision": 0.355556, - "contextRecall": 1, + "recallAt5": 0.921053, + "ndcgAt10": 0.857352, + "mapAt10": 0.80968, + "contextPrecision": 0.324561, + "contextRecall": 0.921053, "noResultRate": 0, - "meanContextChars": 2056.9333333333334, + "meanContextChars": 1958.7631578947369, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 21.22 + "latencyP95Ms": 20.14 }, { "strategy": "hybrid", "candidateK": 20, "contextK": 5, - "recallAt5": 1, - "ndcgAt10": 0.956053, - "mapAt10": 0.938889, - "contextPrecision": 0.213333, - "contextRecall": 1, + "recallAt5": 0.921053, + "ndcgAt10": 0.857352, + "mapAt10": 0.80968, + "contextPrecision": 0.194737, + "contextRecall": 0.921053, "noResultRate": 0, - "meanContextChars": 3445.866666666667, + "meanContextChars": 3282.1052631578946, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 23.23 + "latencyP95Ms": 14.23 }, { "strategy": "hybrid", "candidateK": 20, "contextK": 8, - "recallAt5": 1, - "ndcgAt10": 0.956053, - "mapAt10": 0.938889, - "contextPrecision": 0.133333, + "recallAt5": 0.921053, + "ndcgAt10": 0.857352, + "mapAt10": 0.80968, + "contextPrecision": 0.131579, "contextRecall": 1, "noResultRate": 0, - "meanContextChars": 5572.6, + "meanContextChars": 5357.315789473684, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 21.75 + "latencyP95Ms": 14.06 }, { "strategy": "hybrid", "candidateK": 40, "contextK": 3, - "recallAt5": 1, - "ndcgAt10": 0.956053, - "mapAt10": 0.938889, - "contextPrecision": 0.355556, - "contextRecall": 1, + "recallAt5": 0.921053, + "ndcgAt10": 0.857352, + "mapAt10": 0.80968, + "contextPrecision": 0.324561, + "contextRecall": 0.921053, "noResultRate": 0, - "meanContextChars": 2056.9333333333334, + "meanContextChars": 1958.7631578947369, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 20.35 + "latencyP95Ms": 15.39 }, { "strategy": "hybrid", "candidateK": 40, "contextK": 5, - "recallAt5": 1, - "ndcgAt10": 0.956053, - "mapAt10": 0.938889, - "contextPrecision": 0.213333, - "contextRecall": 1, + "recallAt5": 0.921053, + "ndcgAt10": 0.857352, + "mapAt10": 0.80968, + "contextPrecision": 0.194737, + "contextRecall": 0.921053, "noResultRate": 0, - "meanContextChars": 3445.866666666667, + "meanContextChars": 3282.1052631578946, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 20.35 + "latencyP95Ms": 18.4 }, { "strategy": "hybrid", "candidateK": 40, "contextK": 8, - "recallAt5": 1, - "ndcgAt10": 0.956053, - "mapAt10": 0.938889, - "contextPrecision": 0.133333, + "recallAt5": 0.921053, + "ndcgAt10": 0.857352, + "mapAt10": 0.80968, + "contextPrecision": 0.131579, "contextRecall": 1, "noResultRate": 0, - "meanContextChars": 5572.6, + "meanContextChars": 5357.315789473684, "unanswerableQuestions": 6, "unanswerableNoResultRate": 0, "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 20.91 + "latencyP95Ms": 18.41 } ] } diff --git a/docs/eval/sweep-v1.6.md b/docs/eval/sweep-v1.6.md index bfd5e48..b7e76cc 100644 --- a/docs/eval/sweep-v1.6.md +++ b/docs/eval/sweep-v1.6.md @@ -12,28 +12,28 @@ Each row differs from its neighbour in one parameter. | Strategy | candidateK | contextK | Recall@5 | nDCG@10 | MAP@10 | Context P | Context R | No-result | Context chars | Unans. no-result | Unans. retrieved | Index | p95 | | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | -| dense | 5 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 0.0000 | 5.0 | 19 | 98.32 ms | -| dense | 5 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 0.0000 | 5.0 | 19 | 96.94 ms | -| dense | 10 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 0.0000 | 10.0 | 19 | 84.19 ms | -| dense | 10 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 0.0000 | 10.0 | 19 | 95.07 ms | -| dense | 10 | 8 | 1.0000 | 0.9437 | 0.9222 | 0.1333 | 1.0000 | 0.0000 | 5312 | 0.0000 | 10.0 | 19 | 91.67 ms | -| dense | 20 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 0.0000 | 19.0 | 19 | 16.20 ms | -| dense | 20 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 0.0000 | 19.0 | 19 | 14.19 ms | -| dense | 20 | 8 | 1.0000 | 0.9437 | 0.9222 | 0.1333 | 1.0000 | 0.0000 | 5312 | 0.0000 | 19.0 | 19 | 20.67 ms | -| dense | 40 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 0.0000 | 19.0 | 19 | 19.91 ms | -| dense | 40 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 0.0000 | 19.0 | 19 | 18.65 ms | -| dense | 40 | 8 | 1.0000 | 0.9437 | 0.9222 | 0.1333 | 1.0000 | 0.0000 | 5312 | 0.0000 | 19.0 | 19 | 18.52 ms | -| hybrid | 5 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2044 | 0.0000 | 5.0 | 19 | 15.00 ms | -| hybrid | 5 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3362 | 0.0000 | 5.0 | 19 | 14.40 ms | -| hybrid | 10 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2063 | 0.0000 | 10.0 | 19 | 18.54 ms | -| hybrid | 10 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3526 | 0.0000 | 10.0 | 19 | 16.13 ms | -| hybrid | 10 | 8 | 1.0000 | 0.9561 | 0.9389 | 0.1333 | 1.0000 | 0.0000 | 5525 | 0.0000 | 10.0 | 19 | 18.95 ms | -| hybrid | 20 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2057 | 0.0000 | 19.0 | 19 | 21.22 ms | -| hybrid | 20 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3446 | 0.0000 | 19.0 | 19 | 23.23 ms | -| hybrid | 20 | 8 | 1.0000 | 0.9561 | 0.9389 | 0.1333 | 1.0000 | 0.0000 | 5573 | 0.0000 | 19.0 | 19 | 21.75 ms | -| hybrid | 40 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2057 | 0.0000 | 19.0 | 19 | 20.35 ms | -| hybrid | 40 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3446 | 0.0000 | 19.0 | 19 | 20.35 ms | -| hybrid | 40 | 8 | 1.0000 | 0.9561 | 0.9389 | 0.1333 | 1.0000 | 0.0000 | 5573 | 0.0000 | 19.0 | 19 | 20.91 ms | +| dense | 5 | 3 | 0.9211 | 0.8212 | 0.7851 | 0.3246 | 0.9211 | 0.0000 | 1976 | 0.0000 | 5.0 | 19 | 13.57 ms | +| dense | 5 | 5 | 0.9211 | 0.8212 | 0.7851 | 0.1947 | 0.9211 | 0.0000 | 3188 | 0.0000 | 5.0 | 19 | 14.14 ms | +| dense | 10 | 3 | 0.9211 | 0.8476 | 0.7965 | 0.3246 | 0.9211 | 0.0000 | 1976 | 0.0000 | 10.0 | 19 | 13.30 ms | +| dense | 10 | 5 | 0.9211 | 0.8476 | 0.7965 | 0.1947 | 0.9211 | 0.0000 | 3188 | 0.0000 | 10.0 | 19 | 12.61 ms | +| dense | 10 | 8 | 0.9211 | 0.8476 | 0.7965 | 0.1316 | 1.0000 | 0.0000 | 5151 | 0.0000 | 10.0 | 19 | 13.99 ms | +| dense | 20 | 3 | 0.9211 | 0.8476 | 0.7965 | 0.3246 | 0.9211 | 0.0000 | 1976 | 0.0000 | 19.0 | 19 | 13.61 ms | +| dense | 20 | 5 | 0.9211 | 0.8476 | 0.7965 | 0.1947 | 0.9211 | 0.0000 | 3188 | 0.0000 | 19.0 | 19 | 14.62 ms | +| dense | 20 | 8 | 0.9211 | 0.8476 | 0.7965 | 0.1316 | 1.0000 | 0.0000 | 5151 | 0.0000 | 19.0 | 19 | 15.20 ms | +| dense | 40 | 3 | 0.9211 | 0.8476 | 0.7965 | 0.3246 | 0.9211 | 0.0000 | 1976 | 0.0000 | 19.0 | 19 | 13.66 ms | +| dense | 40 | 5 | 0.9211 | 0.8476 | 0.7965 | 0.1947 | 0.9211 | 0.0000 | 3188 | 0.0000 | 19.0 | 19 | 14.40 ms | +| dense | 40 | 8 | 0.9211 | 0.8476 | 0.7965 | 0.1316 | 1.0000 | 0.0000 | 5151 | 0.0000 | 19.0 | 19 | 14.39 ms | +| hybrid | 5 | 3 | 0.9211 | 0.8309 | 0.7982 | 0.3246 | 0.9211 | 0.0000 | 1949 | 0.0000 | 5.0 | 19 | 16.98 ms | +| hybrid | 5 | 5 | 0.9211 | 0.8309 | 0.7982 | 0.1947 | 0.9211 | 0.0000 | 3216 | 0.0000 | 5.0 | 19 | 16.11 ms | +| hybrid | 10 | 3 | 0.9211 | 0.8574 | 0.8097 | 0.3246 | 0.9211 | 0.0000 | 1963 | 0.0000 | 10.0 | 19 | 16.46 ms | +| hybrid | 10 | 5 | 0.9211 | 0.8574 | 0.8097 | 0.1947 | 0.9211 | 0.0000 | 3333 | 0.0000 | 10.0 | 19 | 14.83 ms | +| hybrid | 10 | 8 | 0.9211 | 0.8574 | 0.8097 | 0.1316 | 1.0000 | 0.0000 | 5320 | 0.0000 | 10.0 | 19 | 17.38 ms | +| hybrid | 20 | 3 | 0.9211 | 0.8574 | 0.8097 | 0.3246 | 0.9211 | 0.0000 | 1959 | 0.0000 | 19.0 | 19 | 20.14 ms | +| hybrid | 20 | 5 | 0.9211 | 0.8574 | 0.8097 | 0.1947 | 0.9211 | 0.0000 | 3282 | 0.0000 | 19.0 | 19 | 14.23 ms | +| hybrid | 20 | 8 | 0.9211 | 0.8574 | 0.8097 | 0.1316 | 1.0000 | 0.0000 | 5357 | 0.0000 | 19.0 | 19 | 14.06 ms | +| hybrid | 40 | 3 | 0.9211 | 0.8574 | 0.8097 | 0.3246 | 0.9211 | 0.0000 | 1959 | 0.0000 | 19.0 | 19 | 15.39 ms | +| hybrid | 40 | 5 | 0.9211 | 0.8574 | 0.8097 | 0.1947 | 0.9211 | 0.0000 | 3282 | 0.0000 | 19.0 | 19 | 18.40 ms | +| hybrid | 40 | 8 | 0.9211 | 0.8574 | 0.8097 | 0.1316 | 1.0000 | 0.0000 | 5357 | 0.0000 | 19.0 | 19 | 18.41 ms | ## Skipped cells @@ -65,10 +65,10 @@ The harness refuses the same combination at the flag level, so a typo fails loud `retrieved` is how much irrelevant context was pulled in anyway. These two are the columns a threshold decision should move, and they are kept out of every other column. -Best nDCG@10 in this grid: `hybrid` candidateK=5, -contextK=3 (0.9561). +Best nDCG@10 in this grid: `hybrid` candidateK=10, +contextK=3 (0.8574). Best context precision: `dense` candidateK=5, -contextK=3 (0.3556). +contextK=3 (0.3246). These are **not** recommendations. Selecting the grid maximum on the same questions is how a benchmark becomes a lookup table; the adoption rule in `baseline-v1.6.md` diff --git a/docs/eval/threshold-v1.6.json b/docs/eval/threshold-v1.6.json index 7b4931f..468dc8d 100644 --- a/docs/eval/threshold-v1.6.json +++ b/docs/eval/threshold-v1.6.json @@ -7,8 +7,8 @@ { "threshold": 0, "validation": { - "questions": 13, - "answerableQuestions": 10, + "questions": 16, + "answerableQuestions": 13, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 19, @@ -19,20 +19,20 @@ "meanRetrieved": 19 }, "metrics": { - "recallAt1": 0.75, - "recallAt5": 1, + "recallAt1": 0.576923, + "recallAt5": 0.923077, "recallAt10": 1, - "mrr": 0.9, - "ndcgAt10": 0.918158, - "hitRateAt5": 1, - "mapAt10": 0.883333, - "contextPrecision": 0.366667, - "contextRecall": 1 + "mrr": 0.778846, + "ndcgAt10": 0.827608, + "hitRateAt5": 0.923077, + "mapAt10": 0.766026, + "contextPrecision": 0.333333, + "contextRecall": 0.923077 } }, "test": { - "questions": 23, - "answerableQuestions": 20, + "questions": 28, + "answerableQuestions": 25, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 19, @@ -43,23 +43,23 @@ "meanRetrieved": 19 }, "metrics": { - "recallAt1": 0.875, - "recallAt5": 1, + "recallAt1": 0.7, + "recallAt5": 0.92, "recallAt10": 1, - "mrr": 0.941667, - "ndcgAt10": 0.956546, - "hitRateAt5": 1, - "mapAt10": 0.941667, - "contextPrecision": 0.35, - "contextRecall": 1 + "mrr": 0.812381, + "ndcgAt10": 0.858056, + "hitRateAt5": 0.92, + "mapAt10": 0.812381, + "contextPrecision": 0.32, + "contextRecall": 0.92 } } }, { "threshold": 0.3, "validation": { - "questions": 13, - "answerableQuestions": 10, + "questions": 16, + "answerableQuestions": 13, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 19, @@ -70,20 +70,20 @@ "meanRetrieved": 19 }, "metrics": { - "recallAt1": 0.75, - "recallAt5": 1, + "recallAt1": 0.576923, + "recallAt5": 0.923077, "recallAt10": 1, - "mrr": 0.9, - "ndcgAt10": 0.918158, - "hitRateAt5": 1, - "mapAt10": 0.883333, - "contextPrecision": 0.366667, - "contextRecall": 1 + "mrr": 0.778846, + "ndcgAt10": 0.827608, + "hitRateAt5": 0.923077, + "mapAt10": 0.766026, + "contextPrecision": 0.333333, + "contextRecall": 0.923077 } }, "test": { - "questions": 23, - "answerableQuestions": 20, + "questions": 28, + "answerableQuestions": 25, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 19, @@ -94,23 +94,23 @@ "meanRetrieved": 19 }, "metrics": { - "recallAt1": 0.875, - "recallAt5": 1, + "recallAt1": 0.7, + "recallAt5": 0.92, "recallAt10": 1, - "mrr": 0.941667, - "ndcgAt10": 0.956546, - "hitRateAt5": 1, - "mapAt10": 0.941667, - "contextPrecision": 0.35, - "contextRecall": 1 + "mrr": 0.812381, + "ndcgAt10": 0.858056, + "hitRateAt5": 0.92, + "mapAt10": 0.812381, + "contextPrecision": 0.32, + "contextRecall": 0.92 } } }, { "threshold": 0.4, "validation": { - "questions": 13, - "answerableQuestions": 10, + "questions": 16, + "answerableQuestions": 13, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 19, @@ -121,20 +121,20 @@ "meanRetrieved": 19 }, "metrics": { - "recallAt1": 0.75, - "recallAt5": 1, + "recallAt1": 0.576923, + "recallAt5": 0.923077, "recallAt10": 1, - "mrr": 0.9, - "ndcgAt10": 0.918158, - "hitRateAt5": 1, - "mapAt10": 0.883333, - "contextPrecision": 0.366667, - "contextRecall": 1 + "mrr": 0.778846, + "ndcgAt10": 0.827608, + "hitRateAt5": 0.923077, + "mapAt10": 0.766026, + "contextPrecision": 0.333333, + "contextRecall": 0.923077 } }, "test": { - "questions": 23, - "answerableQuestions": 20, + "questions": 28, + "answerableQuestions": 25, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 19, @@ -145,23 +145,23 @@ "meanRetrieved": 19 }, "metrics": { - "recallAt1": 0.875, - "recallAt5": 1, + "recallAt1": 0.7, + "recallAt5": 0.92, "recallAt10": 1, - "mrr": 0.941667, - "ndcgAt10": 0.956546, - "hitRateAt5": 1, - "mapAt10": 0.941667, - "contextPrecision": 0.35, - "contextRecall": 1 + "mrr": 0.812381, + "ndcgAt10": 0.858056, + "hitRateAt5": 0.92, + "mapAt10": 0.812381, + "contextPrecision": 0.32, + "contextRecall": 0.92 } } }, { "threshold": 0.5, "validation": { - "questions": 13, - "answerableQuestions": 10, + "questions": 16, + "answerableQuestions": 13, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 19, @@ -172,20 +172,20 @@ "meanRetrieved": 19 }, "metrics": { - "recallAt1": 0.75, - "recallAt5": 1, + "recallAt1": 0.576923, + "recallAt5": 0.923077, "recallAt10": 1, - "mrr": 0.9, - "ndcgAt10": 0.918158, - "hitRateAt5": 1, - "mapAt10": 0.883333, - "contextPrecision": 0.366667, - "contextRecall": 1 + "mrr": 0.778846, + "ndcgAt10": 0.827608, + "hitRateAt5": 0.923077, + "mapAt10": 0.766026, + "contextPrecision": 0.333333, + "contextRecall": 0.923077 } }, "test": { - "questions": 23, - "answerableQuestions": 20, + "questions": 28, + "answerableQuestions": 25, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 19, @@ -196,23 +196,23 @@ "meanRetrieved": 19 }, "metrics": { - "recallAt1": 0.875, - "recallAt5": 1, + "recallAt1": 0.7, + "recallAt5": 0.92, "recallAt10": 1, - "mrr": 0.941667, - "ndcgAt10": 0.956546, - "hitRateAt5": 1, - "mapAt10": 0.941667, - "contextPrecision": 0.35, - "contextRecall": 1 + "mrr": 0.812381, + "ndcgAt10": 0.858056, + "hitRateAt5": 0.92, + "mapAt10": 0.812381, + "contextPrecision": 0.32, + "contextRecall": 0.92 } } }, { "threshold": 0.6, "validation": { - "questions": 13, - "answerableQuestions": 10, + "questions": 16, + "answerableQuestions": 13, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 19, @@ -223,20 +223,20 @@ "meanRetrieved": 19 }, "metrics": { - "recallAt1": 0.75, - "recallAt5": 1, + "recallAt1": 0.576923, + "recallAt5": 0.923077, "recallAt10": 1, - "mrr": 0.9, - "ndcgAt10": 0.918158, - "hitRateAt5": 1, - "mapAt10": 0.883333, - "contextPrecision": 0.366667, - "contextRecall": 1 + "mrr": 0.778846, + "ndcgAt10": 0.827608, + "hitRateAt5": 0.923077, + "mapAt10": 0.766026, + "contextPrecision": 0.333333, + "contextRecall": 0.923077 } }, "test": { - "questions": 23, - "answerableQuestions": 20, + "questions": 28, + "answerableQuestions": 25, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 19, @@ -247,15 +247,15 @@ "meanRetrieved": 19 }, "metrics": { - "recallAt1": 0.875, - "recallAt5": 1, + "recallAt1": 0.7, + "recallAt5": 0.92, "recallAt10": 1, - "mrr": 0.941667, - "ndcgAt10": 0.956546, - "hitRateAt5": 1, - "mapAt10": 0.941667, - "contextPrecision": 0.35, - "contextRecall": 1 + "mrr": 0.812381, + "ndcgAt10": 0.858056, + "hitRateAt5": 0.92, + "mapAt10": 0.812381, + "contextPrecision": 0.32, + "contextRecall": 0.92 } } } diff --git a/docs/eval/threshold-v1.6.md b/docs/eval/threshold-v1.6.md index 700c45c..45e1f10 100644 --- a/docs/eval/threshold-v1.6.md +++ b/docs/eval/threshold-v1.6.md @@ -15,11 +15,11 @@ no-result rate is better. | Threshold | n (val) | Recall@5 (val) | nDCG@10 (val) | No-result (val) | Unans. no-result (val) | Unans. retrieved (val) | nDCG@10 (test) | Unans. no-result (test) | | --- | --- | --- | --- | --- | --- | --- | --- | --- | -| 0 | 10 | 1.0000 | 0.9182 | 0.0000 | 0.0000 | 19.0 | 0.9565 | 0.0000 | -| 0.3 | 10 | 1.0000 | 0.9182 | 0.0000 | 0.0000 | 19.0 | 0.9565 | 0.0000 | -| 0.4 | 10 | 1.0000 | 0.9182 | 0.0000 | 0.0000 | 19.0 | 0.9565 | 0.0000 | -| 0.5 | 10 | 1.0000 | 0.9182 | 0.0000 | 0.0000 | 19.0 | 0.9565 | 0.0000 | -| 0.6 | 10 | 1.0000 | 0.9182 | 0.0000 | 0.0000 | 19.0 | 0.9565 | 0.0000 | +| 0 | 13 | 0.9231 | 0.8276 | 0.0000 | 0.0000 | 19.0 | 0.8581 | 0.0000 | +| 0.3 | 13 | 0.9231 | 0.8276 | 0.0000 | 0.0000 | 19.0 | 0.8581 | 0.0000 | +| 0.4 | 13 | 0.9231 | 0.8276 | 0.0000 | 0.0000 | 19.0 | 0.8581 | 0.0000 | +| 0.5 | 13 | 0.9231 | 0.8276 | 0.0000 | 0.0000 | 19.0 | 0.8581 | 0.0000 | +| 0.6 | 13 | 0.9231 | 0.8276 | 0.0000 | 0.0000 | 19.0 | 0.8581 | 0.0000 | ## Selection rule @@ -32,13 +32,13 @@ the quality gate is there so a refusal gain can never be bought with a retrieval ## Outcome -The sweep is **flat**: every threshold from 0 to 0.6 produces the same validation nDCG@10 (0.9182), the same Recall@5 (1.0000) and the same unanswerable refusal rate (0/3). No passage is ever filtered out, so the threshold is **non-binding** on this corpus — E5 does not score these query/chunk pairs below the top of the swept range. +The sweep is **flat**: every threshold from 0 to 0.6 produces the same validation nDCG@10 (0.8276), the same Recall@5 (0.9231) and the same unanswerable refusal rate (0/3). No passage is ever filtered out, so the threshold is **non-binding** on this corpus — E5 does not score these query/chunk pairs below the top of the swept range. **No evidence to change `threshold = 0.5`.** All thresholds hold the line equally; picking one would be arbitrary. The current value can be neither validated nor falsified here, which is a property of the corpus rather than of the threshold. ## Caveat on this corpus -The split removes the most obvious form of overfitting, but 10 +The split removes the most obvious form of overfitting, but 13 answerable questions on the validation side is a thin basis for a decision, and the corpus is still small. A threshold is a product decision with a **refusal-rate** cost attached, so a recommendation here is only as good as the corpus behind it. Re-run this after the diff --git a/eval/questions.jsonl b/eval/questions.jsonl index d762dd3..c9a3d44 100644 --- a/eval/questions.jsonl +++ b/eval/questions.jsonl @@ -34,3 +34,11 @@ {"id":"q034","question":"What is the retail price of the lithium-ion cells discussed?","type":"unanswerable","answerable":false,"relevant":[]} {"id":"q035","question":"How many megawatts does the tidal array generate?","type":"unanswerable","answerable":false,"relevant":[]} {"id":"q036","question":"Who won the 2018 FIFA World Cup?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q037","question":"推移质为什么比悬移质更难测?","type":"cross-lingual","relevant":[{"document":"river-monitoring.md","page":null,"block":4,"quote":"Bedload is the harder fraction to measure"}]} +{"id":"q038","question":"河流监测中,每个站点要采几份平行样?","type":"cross-lingual","relevant":[{"document":"river-monitoring.md","page":null,"block":6,"quote":"three replicate samples at each station"}]} +{"id":"q039","question":"设计锂离子电芯时最核心的取舍是什么?","type":"cross-lingual","relevant":[{"document":"battery-chemistry.md","page":null,"block":2,"quote":"trade energy density against thermal stability"}]} +{"id":"q040","question":"富镍电池为什么对散热要求更高?","type":"cross-lingual","relevant":[{"document":"battery-chemistry.md","page":null,"block":4,"quote":"release oxygen at lower temperatures than iron phosphate"}]} +{"id":"q041","question":"电芯隔膜一旦熔化会导致什么后果?","type":"cross-lingual","relevant":[{"document":"battery-chemistry.md","page":null,"block":6,"quote":"Once the separator melts, the cell shorts internally"}]} +{"id":"q042","question":"蜜蜂大致从什么温度开始出巢觅食?","type":"cross-lingual","relevant":[{"document":"orchard-pollination.md","page":null,"block":4,"quote":"above roughly twelve degrees Celsius"}]} +{"id":"q043","question":"花期遭遇晚霜,主要受损的是什么?","type":"cross-lingual","relevant":[{"document":"orchard-pollination.md","page":null,"block":6,"quote":"destroys the flower's ovary rather than the petals"}]} +{"id":"q044","question":"为什么成片的树冠比孤立的树降温效果更好?","type":"cross-lingual","relevant":[{"document":"urban-canopy.md","page":null,"block":4,"quote":"asphalt stores far more heat during the day"}]} diff --git a/eval/splits.json b/eval/splits.json index a608f76..c903be1 100644 --- a/eval/splits.json +++ b/eval/splits.json @@ -34,5 +34,13 @@ "q033": "validation", "q034": "test", "q035": "validation", - "q036": "test" + "q036": "test", + "q037": "validation", + "q038": "test", + "q039": "test", + "q040": "validation", + "q041": "test", + "q042": "test", + "q043": "validation", + "q044": "test" }