From d45246347c7e7423bc6ff5fb58b761073f082574 Mon Sep 17 00:00:00 2001 From: MrSibe Date: Wed, 30 Sep 2026 17:35:38 +0800 Subject: [PATCH] feat(eval): unanswerable questions, and an explicit split manifest MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The corpus contract corpus v2 needs. Without it, two of the four things #192 asks the dataset to cover cannot be represented at all: a question whose right answer is "your sources do not say", and a validation/test assignment that is chosen rather than computed. **Unanswerable questions.** `answerable: false` with `relevant: []`. They are excluded from `metrics` and `byType` — `recallAtK` treats no ground truth as 0/0, and averaging "correctly refused" into "missed" would corrupt the table — and reported as their own group: `noResultCount`, `noResultRate` and `meanRetrieved`. On this row **higher no-result is better**, which is the opposite of how it reads everywhere else. `assertQuestionShape` refuses the reverse combination in either direction, because both are silent: an answerable question with no ground truth reads as a permanent miss, and an unanswerable one carrying ground truth reads as a normal hit. **An explicit split manifest.** `eval/splits.json` replaces the hash of the question id. A hash is reproducible but not *stable*: adding a question moved others between the sides, and a rare query type could end up entirely on one side without anyone choosing that — which is what had happened (`multi-hop` and `cross-lingual` were both entirely on `test`). A question with no manifest entry is now **refused** rather than defaulted, so a new question cannot leak into the reporting side. **Six unanswerable questions** are added (three topic-adjacent, three plainly out-of-scope). Their specifics were checked absent from the corpus before being written, so they are genuinely unanswerable rather than believed to be. ## What the first measurement shows ``` unanswerable: { questions: 6, noResultRate: 0, meanRetrieved: 19 } ``` Every unanswerable question returns **the entire corpus** — 19 of 19 chunks — at `threshold: 0.5`, and the threshold sweep is flat all the way to `0.6`: refusal stays `0/6` and `meanRetrieved` stays at the cap. So on this corpus the threshold cannot separate a relevant passage from an unrelated one, and "Who won the 2018 FIFA World Cup?" is answered with 19 passages of river monitoring, tidal energy and tea storage. That is the evidence the threshold decision did not have before, and it is also the corpus limitation #192 describes: it does not say `0.5` is wrong, it says this corpus cannot tell. The reports say so rather than recommending a change. Retrieval metrics are unchanged (`recallAt5` 1.0000), because the six new questions are correctly kept out of them. ## Testing - `npm run typecheck` — clean - `npm test` — 504 pass, including the manifest contract (unknown side rejected, missing entry refused, `all` needs no manifest), the answerable/unanswerable tie, and the harness's window invariant - `npm run eval` twice — byte-identical `docs/eval/baseline-v1.6.json` - `npm run eval:retrieval`, `eval:threshold`, `eval:sweep` — all regenerated; the threshold rule now gates on answerable quality and then maximises unanswerable refusal ## Scope This is the **contract**, plus the smallest honest corpus increment that exercises it. Hard negatives and near-duplicate documents — the thing that would bring `Recall@5` off 1.0000 and give `candidateK` something to do — are the next step, not this PR. Part of #192 (child 2, stage 1 of the corpus). --- docs/eval/baseline-v1.6.json | 224 +++++++++++++++++++++++++++++++++- docs/eval/baseline-v1.6.md | 20 ++- docs/eval/retrieval-v1.6.json | 21 +++- docs/eval/retrieval-v1.6.md | 11 +- docs/eval/sweep-v1.6.json | 112 +++++++++++++---- docs/eval/sweep-v1.6.md | 55 +++++---- docs/eval/threshold-v1.6.json | 200 +++++++++++++++++++++--------- docs/eval/threshold-v1.6.md | 49 ++++---- eval/README.md | 24 +++- eval/questions.jsonl | 6 + eval/splits.json | 38 ++++++ scripts/eval-retrieval.mjs | 8 +- scripts/eval-sweep.mjs | 27 ++-- scripts/eval-threshold.mjs | 127 +++++++++++-------- src/main/eval/harness.ts | 58 +++++++-- src/main/eval/report.ts | 26 +++- src/main/eval/run.ts | 2 + src/main/eval/types.ts | 106 ++++++++++++++-- test/evalHarness.test.ts | 1 + test/evalSplit.test.ts | 112 +++++++++++------ 20 files changed, 966 insertions(+), 261 deletions(-) create mode 100644 eval/splits.json diff --git a/docs/eval/baseline-v1.6.json b/docs/eval/baseline-v1.6.json index 43b77d7..5f4b48a 100644 --- a/docs/eval/baseline-v1.6.json +++ b/docs/eval/baseline-v1.6.json @@ -17,7 +17,7 @@ "threshold": 0.5, "corpus": "eval/corpus", "documents": 13, - "questions": 30, + "questions": 36, "chunkCount": 19 }, "metrics": { @@ -108,11 +108,18 @@ } } ], + "unanswerable": { + "questions": 6, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 19 + }, "perQuestion": [ { "id": "q001", "question": "Why is bedload harder to measure than suspended sediment?", "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -145,6 +152,7 @@ "id": "q002", "question": "How many replicate samples are collected at each river station?", "type": "exact", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -177,6 +185,7 @@ "id": "q003", "question": "What is the central trade-off in lithium-ion cell design?", "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -209,6 +218,7 @@ "id": "q004", "question": "Why do nickel-rich battery packs need more aggressive thermal management?", "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -241,6 +251,7 @@ "id": "q005", "question": "What happens once the separator in a battery cell melts?", "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -273,6 +284,7 @@ "id": "q006", "question": "At what temperature do honeybees begin to forage?", "type": "exact", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -305,6 +317,7 @@ "id": "q007", "question": "What does a late frost damage during full bloom?", "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -337,6 +350,7 @@ "id": "q008", "question": "Why is a continuous tree canopy more effective at cooling than isolated trees?", "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -369,6 +383,7 @@ "id": "q009", "question": "Why are trees with aggressive surface roots unsuitable for narrow verges?", "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -401,6 +416,7 @@ "id": "q010", "question": "At what temperature is lactic acid fermentation fastest?", "type": "exact", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -433,6 +449,7 @@ "id": "q011", "question": "Is the salt percentage in fermentation based on vegetable weight or water weight?", "type": "exact", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -465,6 +482,7 @@ "id": "q012", "question": "Why must tidal turbines be sited in places with very fast currents?", "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -497,6 +515,7 @@ "id": "q013", "question": "What is the main environmental concern for tidal energy installations?", "type": "semantic", + "answerable": true, "firstRelevantRank": 3, "relevantCount": 1, "retrievedCount": 19, @@ -529,6 +548,7 @@ "id": "q014", "question": "In lake monitoring, how is the sampling depth actually recorded?", "type": "exact", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -561,6 +581,7 @@ "id": "q015", "question": "Why does deep-water oxygen fall while a lake remains stratified?", "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -593,6 +614,7 @@ "id": "q016", "question": "How do supercapacitors hold their charge?", "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -625,6 +647,7 @@ "id": "q017", "question": "Why can solitary bees pollinate a bloom week that is too cold for honeybee hives?", "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -657,6 +680,7 @@ "id": "q018", "question": "Why is one continuous planted roof layer better than several isolated planted beds?", "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -689,6 +713,7 @@ "id": "q019", "question": "Where do acetic acid bacteria sit in a vinegar culture?", "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -721,6 +746,7 @@ "id": "q020", "question": "What happens if a vinegar culture is sealed airtight?", "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -753,6 +779,7 @@ "id": "q021", "question": "Why is wave energy harder to schedule ahead than tidal energy?", "type": "semantic", + "answerable": true, "firstRelevantRank": 2, "relevantCount": 1, "retrievedCount": 19, @@ -785,6 +812,7 @@ "id": "q022", "question": "Where does siting for wave energy devices concentrate, and where does it not?", "type": "exact", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -817,6 +845,7 @@ "id": "q023", "question": "How does the river sampling protocol differ from the lake sampling protocol?", "type": "multi-hop", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 2, "retrievedCount": 19, @@ -851,6 +880,7 @@ "id": "q024", "question": "A street canopy and a green roof are both said to cool; what surface does each one shade?", "type": "multi-hop", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 2, "retrievedCount": 19, @@ -885,6 +915,7 @@ "id": "q025", "question": "Which preservation method depends on keeping air away from the food?", "type": "semantic", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -917,6 +948,7 @@ "id": "q026", "question": "Why can one cold morning cost a grower the whole crop even when colonies are brought in?", "type": "semantic", + "answerable": true, "firstRelevantRank": 2, "relevantCount": 1, "retrievedCount": 19, @@ -949,6 +981,7 @@ "id": "q027", "question": "绿茶应该怎样保存才能减缓氧化?", "type": "zh", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -981,6 +1014,7 @@ "id": "q028", "question": "茶叶储存的相对湿度上限是多少?", "type": "zh", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -1013,6 +1047,7 @@ "id": "q029", "question": "为什么冷冻保存的茶叶取出后不能立刻打开包装?", "type": "zh", + "answerable": true, "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -1045,6 +1080,7 @@ "id": "q030", "question": "为什么潮汐能比风能和太阳能更容易提前安排发电?", "type": "cross-lingual", + "answerable": true, "firstRelevantRank": 2, "relevantCount": 1, "retrievedCount": 19, @@ -1072,6 +1108,192 @@ [], [] ] + }, + { + "id": "q031", + "question": "What is the installed capacity of the tidal energy installation described?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 19, + "contextChars": 2488, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q032", + "question": "Which laboratory published the river monitoring protocol?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 19, + "contextChars": 2257, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q033", + "question": "What is the boiling point of mercury?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 19, + "contextChars": 2043, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q034", + "question": "What is the retail price of the lithium-ion cells discussed?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 19, + "contextChars": 2588, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q035", + "question": "How many megawatts does the tidal array generate?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 19, + "contextChars": 2488, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q036", + "question": "Who won the 2018 FIFA World Cup?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 19, + "contextChars": 2071, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] } ] } diff --git a/docs/eval/baseline-v1.6.md b/docs/eval/baseline-v1.6.md index 5c7fc85..23fffc2 100644 --- a/docs/eval/baseline-v1.6.md +++ b/docs/eval/baseline-v1.6.md @@ -12,7 +12,7 @@ Generated by `npm run eval`. The numbers below are harness output — do not edi | Ranks | `candidateK=20, threshold=0.5` | | Context width | `contextK=3` | | Corpus | `eval/corpus` (13 documents) | -| Split | `all` (30 questions) | +| Split | `all` (36 questions, 30 answerable) | | Index size | 19 chunks | ## Metrics @@ -43,8 +43,22 @@ The type comes from `type` in `questions.jsonl`; untagged questions report as | semantic | 18 | 1.0000 | 0.9312 | 1.0000 | 0.9074 | | zh | 3 | 1.0000 | 1.0000 | 1.0000 | 1.0000 | -Timing is informational only and is **not** frozen: indexing 2015 ms, query -p50 71.63 ms, p95 94.77 ms on the +### Unanswerable questions + +These carry no ground truth, so the correct outcome is that retrieval finds nothing. They +are excluded from every metric above — a missing ground truth is not a miss — and reported +here instead. A higher **no-results** rate is better on this row, which is the opposite of +how it reads everywhere else, and `mean passages retrieved` is how much irrelevant context +was pulled in anyway. This is the row a threshold decision should move. + +| Metric | Value | +| --- | --- | +| Unanswerable questions | 6 | +| Returned no results | 0.0000 (0/6) | +| Mean passages retrieved | 19.00 | + +Timing is informational only and is **not** frozen: indexing 1496 ms, query +p50 11.42 ms, p95 17.31 ms on the machine that produced this file. Timing and index size depend on hardware and on the corpus, so they must never be the reason two runs differ. diff --git a/docs/eval/retrieval-v1.6.json b/docs/eval/retrieval-v1.6.json index 6782582..9bf0d74 100644 --- a/docs/eval/retrieval-v1.6.json +++ b/docs/eval/retrieval-v1.6.json @@ -7,6 +7,9 @@ "label": "dense (vector)", "chunking": "1000/100", "chunkCount": 19, + "split": "all", + "questions": 36, + "answerableCount": 30, "recallAt1": 0.833333, "recallAt5": 1, "recallAt10": 1, @@ -16,14 +19,17 @@ "mapAt10": 0.922222, "contextPrecision": 0.355556, "contextRecall": 1, - "indexingMs": 1548, - "latencyP95Ms": 20.51 + "indexingMs": 1518, + "latencyP95Ms": 14.2 }, { "id": "sparse", "label": "sparse (BM25)", "chunking": "1000/100", "chunkCount": 19, + "split": "all", + "questions": 36, + "answerableCount": 30, "recallAt1": 0.733333, "recallAt5": 0.866667, "recallAt10": 0.866667, @@ -33,14 +39,17 @@ "mapAt10": 0.8, "contextPrecision": 0.311111, "contextRecall": 0.866667, - "indexingMs": 1490, - "latencyP95Ms": 1.96 + "indexingMs": 1513, + "latencyP95Ms": 2.46 }, { "id": "hybrid", "label": "hybrid (RRF of dense + BM25)", "chunking": "1000/100", "chunkCount": 19, + "split": "all", + "questions": 36, + "answerableCount": 30, "recallAt1": 0.866667, "recallAt5": 1, "recallAt10": 1, @@ -50,8 +59,8 @@ "mapAt10": 0.938889, "contextPrecision": 0.355556, "contextRecall": 1, - "indexingMs": 1507, - "latencyP95Ms": 20.66 + "indexingMs": 1557, + "latencyP95Ms": 19.55 } ] } diff --git a/docs/eval/retrieval-v1.6.md b/docs/eval/retrieval-v1.6.md index c13d4e4..a091483 100644 --- a/docs/eval/retrieval-v1.6.md +++ b/docs/eval/retrieval-v1.6.md @@ -4,14 +4,15 @@ Generated by `node scripts/eval-retrieval.mjs`. Numbers are harness output; do n ## What was measured -Every strategy runs the real RAG eval harness against the same corpus and the same 30 -questions as `baseline-v1.6.json`, with chunking held fixed at 1000/100. Only the retrieval strategy changes. +Every strategy runs the real RAG eval harness against the same corpus and the same questions +as `baseline-v1.6.json` (split `all`, 36 questions of which +30 are answerable), with chunking held fixed at 1000/100. Only the retrieval strategy changes. | Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Context P | Query p95 | | --- | --- | --- | --- | --- | --- | --- | --- | -| dense (vector) | 0.8333 | 1.0000 | 0.9278 | 0.9437 | 0.9222 | 0.3556 | 20.51 ms | -| sparse (BM25) | 0.7333 | 0.8667 | 0.8056 | 0.8184 | 0.8000 | 0.3111 | 1.96 ms | -| hybrid (RRF of dense + BM25) | 0.8667 | 1.0000 | 0.9444 | 0.9561 | 0.9389 | 0.3556 | 20.66 ms | +| dense (vector) | 0.8333 | 1.0000 | 0.9278 | 0.9437 | 0.9222 | 0.3556 | 14.20 ms | +| sparse (BM25) | 0.7333 | 0.8667 | 0.8056 | 0.8184 | 0.8000 | 0.3111 | 2.46 ms | +| hybrid (RRF of dense + BM25) | 0.8667 | 1.0000 | 0.9444 | 0.9561 | 0.9389 | 0.3556 | 19.55 ms | ## Not evaluated diff --git a/docs/eval/sweep-v1.6.json b/docs/eval/sweep-v1.6.json index c6e7860..67cdb5b 100644 --- a/docs/eval/sweep-v1.6.json +++ b/docs/eval/sweep-v1.6.json @@ -12,8 +12,11 @@ "contextRecall": 1, "noResultRate": 0, "meanContextChars": 2078.9333333333334, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 5, "chunkCount": 19, - "latencyP95Ms": 13.28 + "latencyP95Ms": 98.32 }, { "strategy": "dense", @@ -26,8 +29,11 @@ "contextRecall": 1, "noResultRate": 0, "meanContextChars": 3326.8333333333335, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 5, "chunkCount": 19, - "latencyP95Ms": 13.14 + "latencyP95Ms": 96.94 }, { "strategy": "dense", @@ -40,8 +46,11 @@ "contextRecall": 1, "noResultRate": 0, "meanContextChars": 2078.9333333333334, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 10, "chunkCount": 19, - "latencyP95Ms": 13.97 + "latencyP95Ms": 84.19 }, { "strategy": "dense", @@ -54,8 +63,11 @@ "contextRecall": 1, "noResultRate": 0, "meanContextChars": 3326.8333333333335, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 10, "chunkCount": 19, - "latencyP95Ms": 14.81 + "latencyP95Ms": 95.07 }, { "strategy": "dense", @@ -68,8 +80,11 @@ "contextRecall": 1, "noResultRate": 0, "meanContextChars": 5311.633333333333, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 10, "chunkCount": 19, - "latencyP95Ms": 15.14 + "latencyP95Ms": 91.67 }, { "strategy": "dense", @@ -82,8 +97,11 @@ "contextRecall": 1, "noResultRate": 0, "meanContextChars": 2078.9333333333334, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 16.19 + "latencyP95Ms": 16.2 }, { "strategy": "dense", @@ -96,8 +114,11 @@ "contextRecall": 1, "noResultRate": 0, "meanContextChars": 3326.8333333333335, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 18.66 + "latencyP95Ms": 14.19 }, { "strategy": "dense", @@ -110,8 +131,11 @@ "contextRecall": 1, "noResultRate": 0, "meanContextChars": 5311.633333333333, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 14.61 + "latencyP95Ms": 20.67 }, { "strategy": "dense", @@ -124,8 +148,11 @@ "contextRecall": 1, "noResultRate": 0, "meanContextChars": 2078.9333333333334, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 17.65 + "latencyP95Ms": 19.91 }, { "strategy": "dense", @@ -138,8 +165,11 @@ "contextRecall": 1, "noResultRate": 0, "meanContextChars": 3326.8333333333335, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 19.74 + "latencyP95Ms": 18.65 }, { "strategy": "dense", @@ -152,8 +182,11 @@ "contextRecall": 1, "noResultRate": 0, "meanContextChars": 5311.633333333333, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 17.64 + "latencyP95Ms": 18.52 }, { "strategy": "hybrid", @@ -166,8 +199,11 @@ "contextRecall": 1, "noResultRate": 0, "meanContextChars": 2044.1333333333334, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 5, "chunkCount": 19, - "latencyP95Ms": 13.37 + "latencyP95Ms": 15 }, { "strategy": "hybrid", @@ -180,8 +216,11 @@ "contextRecall": 1, "noResultRate": 0, "meanContextChars": 3361.6, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 5, "chunkCount": 19, - "latencyP95Ms": 14.7 + "latencyP95Ms": 14.4 }, { "strategy": "hybrid", @@ -194,8 +233,11 @@ "contextRecall": 1, "noResultRate": 0, "meanContextChars": 2062.733333333333, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 10, "chunkCount": 19, - "latencyP95Ms": 22.31 + "latencyP95Ms": 18.54 }, { "strategy": "hybrid", @@ -208,8 +250,11 @@ "contextRecall": 1, "noResultRate": 0, "meanContextChars": 3525.633333333333, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 10, "chunkCount": 19, - "latencyP95Ms": 18.66 + "latencyP95Ms": 16.13 }, { "strategy": "hybrid", @@ -221,9 +266,12 @@ "contextPrecision": 0.133333, "contextRecall": 1, "noResultRate": 0, - "meanContextChars": 5509.8, + "meanContextChars": 5525.166666666667, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 10, "chunkCount": 19, - "latencyP95Ms": 15.84 + "latencyP95Ms": 18.95 }, { "strategy": "hybrid", @@ -236,8 +284,11 @@ "contextRecall": 1, "noResultRate": 0, "meanContextChars": 2056.9333333333334, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 20.78 + "latencyP95Ms": 21.22 }, { "strategy": "hybrid", @@ -250,8 +301,11 @@ "contextRecall": 1, "noResultRate": 0, "meanContextChars": 3445.866666666667, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 23.35 + "latencyP95Ms": 23.23 }, { "strategy": "hybrid", @@ -264,8 +318,11 @@ "contextRecall": 1, "noResultRate": 0, "meanContextChars": 5572.6, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 22.32 + "latencyP95Ms": 21.75 }, { "strategy": "hybrid", @@ -278,8 +335,11 @@ "contextRecall": 1, "noResultRate": 0, "meanContextChars": 2056.9333333333334, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 23.47 + "latencyP95Ms": 20.35 }, { "strategy": "hybrid", @@ -292,8 +352,11 @@ "contextRecall": 1, "noResultRate": 0, "meanContextChars": 3445.866666666667, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 21.71 + "latencyP95Ms": 20.35 }, { "strategy": "hybrid", @@ -306,8 +369,11 @@ "contextRecall": 1, "noResultRate": 0, "meanContextChars": 5572.6, + "unanswerableQuestions": 6, + "unanswerableNoResultRate": 0, + "unanswerableMeanRetrieved": 19, "chunkCount": 19, - "latencyP95Ms": 21.68 + "latencyP95Ms": 20.91 } ] } diff --git a/docs/eval/sweep-v1.6.md b/docs/eval/sweep-v1.6.md index 6af90da..bfd5e48 100644 --- a/docs/eval/sweep-v1.6.md +++ b/docs/eval/sweep-v1.6.md @@ -10,30 +10,30 @@ Each row differs from its neighbour in one parameter. 2 further cell(s) were **skipped** because `contextK > candidateK`; see below. -| Strategy | candidateK | contextK | Recall@5 | nDCG@10 | MAP@10 | Context P | Context R | No-result | Context chars | Index | p95 | -| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | -| dense | 5 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 19 | 13.28 ms | -| dense | 5 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 19 | 13.14 ms | -| dense | 10 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 19 | 13.97 ms | -| dense | 10 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 19 | 14.81 ms | -| dense | 10 | 8 | 1.0000 | 0.9437 | 0.9222 | 0.1333 | 1.0000 | 0.0000 | 5312 | 19 | 15.14 ms | -| dense | 20 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 19 | 16.19 ms | -| dense | 20 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 19 | 18.66 ms | -| dense | 20 | 8 | 1.0000 | 0.9437 | 0.9222 | 0.1333 | 1.0000 | 0.0000 | 5312 | 19 | 14.61 ms | -| dense | 40 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 19 | 17.65 ms | -| dense | 40 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 19 | 19.74 ms | -| dense | 40 | 8 | 1.0000 | 0.9437 | 0.9222 | 0.1333 | 1.0000 | 0.0000 | 5312 | 19 | 17.64 ms | -| hybrid | 5 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2044 | 19 | 13.37 ms | -| hybrid | 5 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3362 | 19 | 14.70 ms | -| hybrid | 10 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2063 | 19 | 22.31 ms | -| hybrid | 10 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3526 | 19 | 18.66 ms | -| hybrid | 10 | 8 | 1.0000 | 0.9561 | 0.9389 | 0.1333 | 1.0000 | 0.0000 | 5510 | 19 | 15.84 ms | -| hybrid | 20 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2057 | 19 | 20.78 ms | -| hybrid | 20 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3446 | 19 | 23.35 ms | -| hybrid | 20 | 8 | 1.0000 | 0.9561 | 0.9389 | 0.1333 | 1.0000 | 0.0000 | 5573 | 19 | 22.32 ms | -| hybrid | 40 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2057 | 19 | 23.47 ms | -| hybrid | 40 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3446 | 19 | 21.71 ms | -| hybrid | 40 | 8 | 1.0000 | 0.9561 | 0.9389 | 0.1333 | 1.0000 | 0.0000 | 5573 | 19 | 21.68 ms | +| Strategy | candidateK | contextK | Recall@5 | nDCG@10 | MAP@10 | Context P | Context R | No-result | Context chars | Unans. no-result | Unans. retrieved | Index | p95 | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | +| dense | 5 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 0.0000 | 5.0 | 19 | 98.32 ms | +| dense | 5 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 0.0000 | 5.0 | 19 | 96.94 ms | +| dense | 10 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 0.0000 | 10.0 | 19 | 84.19 ms | +| dense | 10 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 0.0000 | 10.0 | 19 | 95.07 ms | +| dense | 10 | 8 | 1.0000 | 0.9437 | 0.9222 | 0.1333 | 1.0000 | 0.0000 | 5312 | 0.0000 | 10.0 | 19 | 91.67 ms | +| dense | 20 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 0.0000 | 19.0 | 19 | 16.20 ms | +| dense | 20 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 0.0000 | 19.0 | 19 | 14.19 ms | +| dense | 20 | 8 | 1.0000 | 0.9437 | 0.9222 | 0.1333 | 1.0000 | 0.0000 | 5312 | 0.0000 | 19.0 | 19 | 20.67 ms | +| dense | 40 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 0.0000 | 19.0 | 19 | 19.91 ms | +| dense | 40 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 0.0000 | 19.0 | 19 | 18.65 ms | +| dense | 40 | 8 | 1.0000 | 0.9437 | 0.9222 | 0.1333 | 1.0000 | 0.0000 | 5312 | 0.0000 | 19.0 | 19 | 18.52 ms | +| hybrid | 5 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2044 | 0.0000 | 5.0 | 19 | 15.00 ms | +| hybrid | 5 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3362 | 0.0000 | 5.0 | 19 | 14.40 ms | +| hybrid | 10 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2063 | 0.0000 | 10.0 | 19 | 18.54 ms | +| hybrid | 10 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3526 | 0.0000 | 10.0 | 19 | 16.13 ms | +| hybrid | 10 | 8 | 1.0000 | 0.9561 | 0.9389 | 0.1333 | 1.0000 | 0.0000 | 5525 | 0.0000 | 10.0 | 19 | 18.95 ms | +| hybrid | 20 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2057 | 0.0000 | 19.0 | 19 | 21.22 ms | +| hybrid | 20 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3446 | 0.0000 | 19.0 | 19 | 23.23 ms | +| hybrid | 20 | 8 | 1.0000 | 0.9561 | 0.9389 | 0.1333 | 1.0000 | 0.0000 | 5573 | 0.0000 | 19.0 | 19 | 21.75 ms | +| hybrid | 40 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2057 | 0.0000 | 19.0 | 19 | 20.35 ms | +| hybrid | 40 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3446 | 0.0000 | 19.0 | 19 | 20.35 ms | +| hybrid | 40 | 8 | 1.0000 | 0.9561 | 0.9389 | 0.1333 | 1.0000 | 0.0000 | 5573 | 0.0000 | 19.0 | 19 | 20.91 ms | ## Skipped cells @@ -58,7 +58,12 @@ The harness refuses the same combination at the flag level, so a typo fails loud columns are here so it is visible rather than argued about. - **Context chars** is a proxy for prompt size, not a token count: the harness pins the embedding model, not any generation model's tokenizer. -- **No-result** is the share of questions whose retrieval returned nothing at all. +- **No-result** is the share of *answerable* questions whose retrieval returned nothing — + a miss, and the lower the better. +- **Unans. no-result / retrieved** are the same idea for the *unanswerable* questions, + where the direction flips: there is no ground truth, so returning nothing is correct and + `retrieved` is how much irrelevant context was pulled in anyway. These two are the + columns a threshold decision should move, and they are kept out of every other column. Best nDCG@10 in this grid: `hybrid` candidateK=5, contextK=3 (0.9561). diff --git a/docs/eval/threshold-v1.6.json b/docs/eval/threshold-v1.6.json index 636e5b9..7b4931f 100644 --- a/docs/eval/threshold-v1.6.json +++ b/docs/eval/threshold-v1.6.json @@ -7,175 +7,255 @@ { "threshold": 0, "validation": { - "questions": 10, + "questions": 13, + "answerableQuestions": 10, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 19, + "unanswerable": { + "questions": 3, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 19 + }, "metrics": { - "recallAt1": 0.9, + "recallAt1": 0.75, "recallAt5": 1, "recallAt10": 1, - "mrr": 0.933333, - "ndcgAt10": 0.95, + "mrr": 0.9, + "ndcgAt10": 0.918158, "hitRateAt5": 1, - "mapAt10": 0.933333, - "evidencePrecisionAt5": 0.2 + "mapAt10": 0.883333, + "contextPrecision": 0.366667, + "contextRecall": 1 } }, "test": { - "questions": 20, + "questions": 23, + "answerableQuestions": 20, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 19, + "unanswerable": { + "questions": 3, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 19 + }, "metrics": { - "recallAt1": 0.8, + "recallAt1": 0.875, "recallAt5": 1, "recallAt10": 1, - "mrr": 0.925, - "ndcgAt10": 0.940626, + "mrr": 0.941667, + "ndcgAt10": 0.956546, "hitRateAt5": 1, - "mapAt10": 0.916667, - "evidencePrecisionAt5": 0.22 + "mapAt10": 0.941667, + "contextPrecision": 0.35, + "contextRecall": 1 } } }, { "threshold": 0.3, "validation": { - "questions": 10, + "questions": 13, + "answerableQuestions": 10, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 19, + "unanswerable": { + "questions": 3, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 19 + }, "metrics": { - "recallAt1": 0.9, + "recallAt1": 0.75, "recallAt5": 1, "recallAt10": 1, - "mrr": 0.933333, - "ndcgAt10": 0.95, + "mrr": 0.9, + "ndcgAt10": 0.918158, "hitRateAt5": 1, - "mapAt10": 0.933333, - "evidencePrecisionAt5": 0.2 + "mapAt10": 0.883333, + "contextPrecision": 0.366667, + "contextRecall": 1 } }, "test": { - "questions": 20, + "questions": 23, + "answerableQuestions": 20, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 19, + "unanswerable": { + "questions": 3, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 19 + }, "metrics": { - "recallAt1": 0.8, + "recallAt1": 0.875, "recallAt5": 1, "recallAt10": 1, - "mrr": 0.925, - "ndcgAt10": 0.940626, + "mrr": 0.941667, + "ndcgAt10": 0.956546, "hitRateAt5": 1, - "mapAt10": 0.916667, - "evidencePrecisionAt5": 0.22 + "mapAt10": 0.941667, + "contextPrecision": 0.35, + "contextRecall": 1 } } }, { "threshold": 0.4, "validation": { - "questions": 10, + "questions": 13, + "answerableQuestions": 10, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 19, + "unanswerable": { + "questions": 3, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 19 + }, "metrics": { - "recallAt1": 0.9, + "recallAt1": 0.75, "recallAt5": 1, "recallAt10": 1, - "mrr": 0.933333, - "ndcgAt10": 0.95, + "mrr": 0.9, + "ndcgAt10": 0.918158, "hitRateAt5": 1, - "mapAt10": 0.933333, - "evidencePrecisionAt5": 0.2 + "mapAt10": 0.883333, + "contextPrecision": 0.366667, + "contextRecall": 1 } }, "test": { - "questions": 20, + "questions": 23, + "answerableQuestions": 20, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 19, + "unanswerable": { + "questions": 3, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 19 + }, "metrics": { - "recallAt1": 0.8, + "recallAt1": 0.875, "recallAt5": 1, "recallAt10": 1, - "mrr": 0.925, - "ndcgAt10": 0.940626, + "mrr": 0.941667, + "ndcgAt10": 0.956546, "hitRateAt5": 1, - "mapAt10": 0.916667, - "evidencePrecisionAt5": 0.22 + "mapAt10": 0.941667, + "contextPrecision": 0.35, + "contextRecall": 1 } } }, { "threshold": 0.5, "validation": { - "questions": 10, + "questions": 13, + "answerableQuestions": 10, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 19, + "unanswerable": { + "questions": 3, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 19 + }, "metrics": { - "recallAt1": 0.9, + "recallAt1": 0.75, "recallAt5": 1, "recallAt10": 1, - "mrr": 0.933333, - "ndcgAt10": 0.95, + "mrr": 0.9, + "ndcgAt10": 0.918158, "hitRateAt5": 1, - "mapAt10": 0.933333, - "evidencePrecisionAt5": 0.2 + "mapAt10": 0.883333, + "contextPrecision": 0.366667, + "contextRecall": 1 } }, "test": { - "questions": 20, + "questions": 23, + "answerableQuestions": 20, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 19, + "unanswerable": { + "questions": 3, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 19 + }, "metrics": { - "recallAt1": 0.8, + "recallAt1": 0.875, "recallAt5": 1, "recallAt10": 1, - "mrr": 0.925, - "ndcgAt10": 0.940626, + "mrr": 0.941667, + "ndcgAt10": 0.956546, "hitRateAt5": 1, - "mapAt10": 0.916667, - "evidencePrecisionAt5": 0.22 + "mapAt10": 0.941667, + "contextPrecision": 0.35, + "contextRecall": 1 } } }, { "threshold": 0.6, "validation": { - "questions": 10, + "questions": 13, + "answerableQuestions": 10, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 19, + "unanswerable": { + "questions": 3, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 19 + }, "metrics": { - "recallAt1": 0.9, + "recallAt1": 0.75, "recallAt5": 1, "recallAt10": 1, - "mrr": 0.933333, - "ndcgAt10": 0.95, + "mrr": 0.9, + "ndcgAt10": 0.918158, "hitRateAt5": 1, - "mapAt10": 0.933333, - "evidencePrecisionAt5": 0.2 + "mapAt10": 0.883333, + "contextPrecision": 0.366667, + "contextRecall": 1 } }, "test": { - "questions": 20, + "questions": 23, + "answerableQuestions": 20, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 19, + "unanswerable": { + "questions": 3, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 19 + }, "metrics": { - "recallAt1": 0.8, + "recallAt1": 0.875, "recallAt5": 1, "recallAt10": 1, - "mrr": 0.925, - "ndcgAt10": 0.940626, + "mrr": 0.941667, + "ndcgAt10": 0.956546, "hitRateAt5": 1, - "mapAt10": 0.916667, - "evidencePrecisionAt5": 0.22 + "mapAt10": 0.941667, + "contextPrecision": 0.35, + "contextRecall": 1 } } } diff --git a/docs/eval/threshold-v1.6.md b/docs/eval/threshold-v1.6.md index c9f5c1f..700c45c 100644 --- a/docs/eval/threshold-v1.6.md +++ b/docs/eval/threshold-v1.6.md @@ -6,41 +6,42 @@ Generated by `node scripts/eval-threshold.mjs`. Numbers are harness output; do n The real harness, the same corpus and the production retrieval config (`candidateK=20, contextK=3`), once per candidate threshold. The **validation** -split selects; the **test** split reports. Both come from the same -`--eval-split=` code path, and the split is a deterministic function of the -question id, so this is reproducible. - -| Threshold | n (val) | Recall@5 (val) | nDCG@10 (val) | MAP@10 (val) | No-result (val) | nDCG@10 (test) | No-result (test) | -| --- | --- | --- | --- | --- | --- | --- | --- | -| 0 | 10 | 1.0000 | 0.9500 | 0.9333 | 0.0000 | 0.9406 | 0.0000 | -| 0.3 | 10 | 1.0000 | 0.9500 | 0.9333 | 0.0000 | 0.9406 | 0.0000 | -| 0.4 | 10 | 1.0000 | 0.9500 | 0.9333 | 0.0000 | 0.9406 | 0.0000 | -| 0.5 | 10 | 1.0000 | 0.9500 | 0.9333 | 0.0000 | 0.9406 | 0.0000 | -| 0.6 | 10 | 1.0000 | 0.9500 | 0.9333 | 0.0000 | 0.9406 | 0.0000 | +split selects; the **test** split reports. The split is the committed manifest +`eval/splits.json`, so the same questions are on the same side on every machine. + +Quality columns cover the answerable questions only; **Unans.** columns cover the +unanswerable ones, where returning nothing is the desired outcome and so a *higher* +no-result rate is better. + +| Threshold | n (val) | Recall@5 (val) | nDCG@10 (val) | No-result (val) | Unans. no-result (val) | Unans. retrieved (val) | nDCG@10 (test) | Unans. no-result (test) | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | +| 0 | 10 | 1.0000 | 0.9182 | 0.0000 | 0.0000 | 19.0 | 0.9565 | 0.0000 | +| 0.3 | 10 | 1.0000 | 0.9182 | 0.0000 | 0.0000 | 19.0 | 0.9565 | 0.0000 | +| 0.4 | 10 | 1.0000 | 0.9182 | 0.0000 | 0.0000 | 19.0 | 0.9565 | 0.0000 | +| 0.5 | 10 | 1.0000 | 0.9182 | 0.0000 | 0.0000 | 19.0 | 0.9565 | 0.0000 | +| 0.6 | 10 | 1.0000 | 0.9182 | 0.0000 | 0.0000 | 19.0 | 0.9565 | 0.0000 | ## Selection rule -Best validation nDCG@10, then fewest validation no-results, then the **lowest** -threshold — the first stage is supposed to favour recall and let a later stage -filter, so among equals the wider one is the safer default. +Hold the answerable quality line — validation nDCG@10 and context recall must not +regress versus `threshold = 0` — then take the threshold that refuses the most +unanswerable questions. Tie-break on the lowest threshold. -## Outcome +Raising a threshold is only worth anything if it refuses what the sources do not answer; +the quality gate is there so a refusal gain can never be bought with a retrieval loss. -The sweep is **flat**: every threshold from 0 to 0.6 produces the same validation nDCG@10 (0.9500), the same Recall@5 (1.0000) and a no-result rate of 0.0000. On this corpus the threshold is simply **non-binding** — E5 never scores these query/chunk pairs below the top of the swept range, so no passage is ever filtered out. +## Outcome -**No evidence to change `threshold = 0.5`.** The tie-break rule nominates `0` only because it prefers the widest threshold among equals; that is a tie-break, not a finding. What this run establishes is that the current value cannot be validated *or* falsified here, which is a property of the corpus, not of the threshold. Re-run after #192 child 2 grows it. +The sweep is **flat**: every threshold from 0 to 0.6 produces the same validation nDCG@10 (0.9182), the same Recall@5 (1.0000) and the same unanswerable refusal rate (0/3). No passage is ever filtered out, so the threshold is **non-binding** on this corpus — E5 does not score these query/chunk pairs below the top of the swept range. -The app currently ships `threshold = 0.5`: validation nDCG@10 -0.9500, test nDCG@10 -0.9406, test no-result rate -0.0000. +**No evidence to change `threshold = 0.5`.** All thresholds hold the line equally; picking one would be arbitrary. The current value can be neither validated nor falsified here, which is a property of the corpus rather than of the threshold. ## Caveat on this corpus The split removes the most obvious form of overfitting, but 10 -validation questions is a thin basis for a decision, and the corpus is still small. A -threshold is a product decision with a **no-result-rate** cost attached, so a -recommendation here is only as good as the corpus behind it. Re-run this after the +answerable questions on the validation side is a thin basis for a decision, and the corpus +is still small. A threshold is a product decision with a **refusal-rate** cost attached, so +a recommendation here is only as good as the corpus behind it. Re-run this after the corpus grows (#192 child 2). ## Reproduce diff --git a/eval/README.md b/eval/README.md index 290648f..f01989c 100644 --- a/eval/README.md +++ b/eval/README.md @@ -36,9 +36,14 @@ report describes the whole online path. ### Swept parameters are chosen on `validation`, reported on `test` A parameter picked on the same questions it is scored on is a fitted number, not a -result. `--eval-split=validation` selects roughly a third of the questions by a -deterministic hash of the id; `test` is the rest. `npm run eval:threshold` uses this -to pick a similarity threshold on `validation` and report it on `test`. +result. `eval/splits.json` is the committed assignment; `--eval-split=validation` selects +from it and `test` is the rest. + +It is an explicit manifest rather than a hash of the question id. A hash is reproducible +but not *stable*: adding a question moves others between the sides, and a rare query type +can end up entirely on one side without anyone choosing that. With a manifest, a question +with no entry is **refused** rather than defaulted, so a new question is assigned +deliberately instead of leaking into `test`. `npm run eval:sweep` runs a bounded grid (`strategy × candidateK × contextK`) and writes one dashboard with quality, context precision/recall, prompt size, index size @@ -70,6 +75,7 @@ does not mean "no download". The 134 MB of weights are not committed. eval/ corpus/ first-party documents (markdown today) questions.jsonl one question per line; committed with the corpus + splits.json the committed validation/test assignment ``` The corpus is authored for this repository and carries the repository's GPL-3.0 licence, @@ -83,6 +89,7 @@ dataset needs a new question. "id": "q001", "question": "Why is bedload harder to measure than suspended sediment?", "type": "semantic", + "answerable": true, "relevant": [ { "document": "river-monitoring.md", @@ -95,6 +102,11 @@ dataset needs a new question. } ``` +An **unanswerable** question is the same shape with `"answerable": false` and `"relevant": []`. +The harness refuses the reverse combination in either direction — an answerable question +with no ground truth, or an unanswerable one carrying some — because both are silent: the +first reads as a permanent miss, the second as a normal hit. + Ground truth uses **corpus identity, never database identity**: - `document` is the corpus-relative path. @@ -163,6 +175,12 @@ from unanswerable queries. `semantic`, `multi-hop`, `cross-lingual`, `zh`). A single average hides a change that helps one kind of question and hurts another; the current baseline already shows this, with `cross-lingual` at nDCG 0.63 against 0.93–1.00 elsewhere. +- **Unanswerable questions** — a separate group, never averaged in. They have no ground + truth, so `Recall` on them is 0/0 rather than 0, and the correct outcome is that + retrieval finds nothing. The reported **no-results** rate is the opposite of a miss: +higher is better, and `mean passages retrieved` is how much irrelevant context was pulled + in anyway. This is the only metric a similarity-threshold decision should move, which is + why the threshold sweep reports it separately. - **Latency p50/p95** — informational only. Timing is **not** frozen, and the committed JSON excludes it so two runs diff cleanly. diff --git a/eval/questions.jsonl b/eval/questions.jsonl index de15b0c..d762dd3 100644 --- a/eval/questions.jsonl +++ b/eval/questions.jsonl @@ -28,3 +28,9 @@ {"id":"q028","question":"茶叶储存的相对湿度上限是多少?","relevant":[{"document":"chinese-tea-storage.md","page":null,"block":4,"quote":"相对湿度应保持在百分之五十以下"}],"type":"zh"} {"id":"q029","question":"为什么冷冻保存的茶叶取出后不能立刻打开包装?","relevant":[{"document":"chinese-tea-storage.md","page":null,"block":6,"quote":"冷凝水会直接落在茶叶上"}],"type":"zh"} {"id":"q030","question":"为什么潮汐能比风能和太阳能更容易提前安排发电?","relevant":[{"document":"tidal-energy.md","page":null,"block":2,"quote":"predictable decades ahead, unlike wind or solar"}],"type":"cross-lingual"} +{"id":"q031","question":"What is the installed capacity of the tidal energy installation described?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q032","question":"Which laboratory published the river monitoring protocol?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q033","question":"What is the boiling point of mercury?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q034","question":"What is the retail price of the lithium-ion cells discussed?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q035","question":"How many megawatts does the tidal array generate?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q036","question":"Who won the 2018 FIFA World Cup?","type":"unanswerable","answerable":false,"relevant":[]} diff --git a/eval/splits.json b/eval/splits.json new file mode 100644 index 0000000..a608f76 --- /dev/null +++ b/eval/splits.json @@ -0,0 +1,38 @@ +{ + "q001": "test", + "q002": "validation", + "q003": "test", + "q004": "validation", + "q005": "test", + "q006": "test", + "q007": "test", + "q008": "test", + "q009": "validation", + "q010": "test", + "q011": "validation", + "q012": "test", + "q013": "test", + "q014": "test", + "q015": "validation", + "q016": "test", + "q017": "test", + "q018": "validation", + "q019": "test", + "q020": "test", + "q021": "validation", + "q022": "test", + "q023": "validation", + "q024": "test", + "q025": "test", + "q026": "test", + "q027": "test", + "q028": "validation", + "q029": "test", + "q030": "validation", + "q031": "validation", + "q032": "test", + "q033": "validation", + "q034": "test", + "q035": "validation", + "q036": "test" +} diff --git a/scripts/eval-retrieval.mjs b/scripts/eval-retrieval.mjs index 5964a28..0ed9a6c 100644 --- a/scripts/eval-retrieval.mjs +++ b/scripts/eval-retrieval.mjs @@ -126,6 +126,9 @@ try { label: strategy.label, chunking: `${report.config.chunking.chunkSize}/${report.config.chunking.chunkOverlap}`, chunkCount: report.config.chunkCount, + split: report.config.split, + questions: report.config.questions, + answerableCount: report.byType.reduce((total, entry) => total + entry.questions, 0), ...metrics, ...readTiming(join(outDir, 'baseline-v1.6.md')) }) @@ -176,8 +179,9 @@ Generated by \`node scripts/eval-retrieval.mjs\`. Numbers are harness output; do ## What was measured -Every strategy runs the real RAG eval harness against the same corpus and the same 30 -questions as \`baseline-v1.6.json\`, with chunking held fixed at ${baseline.chunking}. Only the retrieval strategy changes. +Every strategy runs the real RAG eval harness against the same corpus and the same questions +as \`baseline-v1.6.json\` (split \`${baseline.split}\`, ${baseline.questions} questions of which +${baseline.answerableCount} are answerable), with chunking held fixed at ${baseline.chunking}. Only the retrieval strategy changes. | Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Context P | Query p95 | | --- | --- | --- | --- | --- | --- | --- | --- | diff --git a/scripts/eval-sweep.mjs b/scripts/eval-sweep.mjs index 46faefd..45a3eb0 100644 --- a/scripts/eval-sweep.mjs +++ b/scripts/eval-sweep.mjs @@ -141,7 +141,11 @@ try { mkdirSync(outDir, { recursive: true }) const report = await runOne(strategy, candidateK, contextK, outDir) const perQuestion = report.perQuestion ?? [] - const noResult = perQuestion.filter((q) => q.retrievedCount === 0).length + // 质量列只看可答的问题,拒答列只看不可答的问题。把两者平均到一起,会给出一个 + // 看起来很干净的检索分数,即使同一份语料把“谁赢了 2018 世界杯”用 19 条上下文 + // 回答了(#192 评审)。 + const answerable = perQuestion.filter((q) => q.answerable) + const noResult = answerable.filter((q) => q.retrievedCount === 0).length rows.push({ strategy, @@ -152,8 +156,11 @@ try { mapAt10: report.metrics.mapAt10, contextPrecision: report.metrics.contextPrecision, contextRecall: report.metrics.contextRecall, - noResultRate: perQuestion.length === 0 ? 0 : noResult / perQuestion.length, - meanContextChars: mean(perQuestion.map((q) => q.contextChars)), + noResultRate: answerable.length === 0 ? 0 : noResult / answerable.length, + meanContextChars: mean(answerable.map((q) => q.contextChars)), + unanswerableQuestions: report.unanswerable.questions, + unanswerableNoResultRate: report.unanswerable.noResultRate, + unanswerableMeanRetrieved: report.unanswerable.meanRetrieved, chunkCount: report.config.chunkCount, ...readTiming(join(outDir, 'baseline-v1.6.md')) }) @@ -170,7 +177,8 @@ const tableRows = rows `| ${row.strategy} | ${row.candidateK} | ${row.contextK} | ${format4(row.recallAt5)} | ` + `${format4(row.ndcgAt10)} | ${format4(row.mapAt10)} | ${format4(row.contextPrecision)} | ` + `${format4(row.contextRecall)} | ${format4(row.noResultRate)} | ` + - `${Math.round(row.meanContextChars)} | ${row.chunkCount} | ${row.latencyP95Ms?.toFixed(2) ?? '—'} ms |` + `${Math.round(row.meanContextChars)} | ${format4(row.unanswerableNoResultRate)} | ` + + `${row.unanswerableMeanRetrieved.toFixed(1)} | ${row.chunkCount} | ${row.latencyP95Ms?.toFixed(2) ?? '—'} ms |` ) .join('\n') @@ -203,8 +211,8 @@ The real harness, the same corpus, chunking held fixed, over ${STRATEGIES.join(' / ')} × candidateK {${CANDIDATE_KS.join(', ')}} × contextK {${CONTEXT_KS.join(', ')}} — ${rows.length} runs. Each row differs from its neighbour in one parameter.${skipped.length > 0 ? `\n\n${skipped.length} further cell(s) were **skipped** because \`contextK > candidateK\`; see below.` : ''} -| Strategy | candidateK | contextK | Recall@5 | nDCG@10 | MAP@10 | Context P | Context R | No-result | Context chars | Index | p95 | -| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | +| Strategy | candidateK | contextK | Recall@5 | nDCG@10 | MAP@10 | Context P | Context R | No-result | Context chars | Unans. no-result | Unans. retrieved | Index | p95 | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | ${tableRows} ${skippedNote} @@ -218,7 +226,12 @@ ${skippedNote} columns are here so it is visible rather than argued about. - **Context chars** is a proxy for prompt size, not a token count: the harness pins the embedding model, not any generation model's tokenizer. -- **No-result** is the share of questions whose retrieval returned nothing at all. +- **No-result** is the share of *answerable* questions whose retrieval returned nothing — + a miss, and the lower the better. +- **Unans. no-result / retrieved** are the same idea for the *unanswerable* questions, + where the direction flips: there is no ground truth, so returning nothing is correct and + \`retrieved\` is how much irrelevant context was pulled in anyway. These two are the + columns a threshold decision should move, and they are kept out of every other column. Best nDCG@10 in this grid: \`${bestNdcg.strategy}\` candidateK=${bestNdcg.candidateK}, contextK=${bestNdcg.contextK} (${format4(bestNdcg.ndcgAt10)}). diff --git a/scripts/eval-threshold.mjs b/scripts/eval-threshold.mjs index b6d1c73..8eb9697 100644 --- a/scripts/eval-threshold.mjs +++ b/scripts/eval-threshold.mjs @@ -88,22 +88,29 @@ function runOne(threshold, split, outDir) { } /** - * The frozen metrics plus the one this experiment needs and the baseline does not - * carry: how often a threshold turns a question into "no results at all". + * The frozen metrics plus the two the experiment exists for: how often the threshold + * turns an answerable question into "no results at all", and how often it makes an + * unanswerable one return nothing (which is the desired outcome for those). * - * A higher threshold can look better on ranking metrics while quietly making the - * product answer "not in your sources" more often, and that trade is invisible - * unless it is counted. + * A threshold that scores well on ranking metrics but keeps returning the whole corpus + * for a question the sources do not answer is invisible unless the second number is + * counted, and the two have to be split: one is a miss, the other is a correct refusal. */ function summarize(report) { const perQuestion = report.perQuestion ?? [] - const noResult = perQuestion.filter((q) => q.retrievedCount === 0).length - const retrieved = perQuestion.map((q) => q.retrievedCount) + const answerable = perQuestion.filter((q) => q.answerable) + const noResult = answerable.filter((q) => q.retrievedCount === 0).length return { questions: perQuestion.length, + answerableQuestions: answerable.length, noResultCount: noResult, - noResultRate: perQuestion.length === 0 ? 0 : noResult / perQuestion.length, - meanRetrieved: retrieved.length === 0 ? 0 : retrieved.reduce((a, b) => a + b, 0) / retrieved.length, + noResultRate: answerable.length === 0 ? 0 : noResult / answerable.length, + meanRetrieved: + answerable.length === 0 + ? 0 + : answerable.reduce((a, q) => a + q.retrievedCount, 0) / answerable.length, + /** 不可答问题时希望返回空,所以这里的“高”是好事。 */ + unanswerable: report.unanswerable, metrics: report.metrics } } @@ -132,42 +139,63 @@ try { } /** - * Selection rule, stated so it can be argued with: best validation nDCG@10, then - * fewest validation no-results, then the widest (lowest) threshold — because the - * first stage is supposed to favour recall and let a later stage filter. + * Selection rule, stated so it can be argued with (#192). + * + * Raising the threshold is only worth it if it refuses more of what the sources do not + * answer. So: take the widest threshold that does not regress the answerable quality + * metrics (nDCG@10 and context recall on validation) as the reference, then among the + * thresholds that hold that line, pick the one that refuses the most unanswerable + * questions; break ties on the lowest threshold. + * + * `threshold = 0` is the reference, because it is the arm that keeps the ranking intact. */ -const ranked = [...rows].sort( +const reference = rows.find((row) => row.threshold === 0) ?? rows[0] +const EPSILON = 1e-9 +const holdsTheLine = (row) => + row.validation.metrics.ndcgAt10 >= reference.validation.metrics.ndcgAt10 - EPSILON && + row.validation.metrics.contextRecall >= reference.validation.metrics.contextRecall - EPSILON + +const eligible = rows.filter(holdsTheLine) +const ranked = [...eligible].sort( (a, b) => - b.validation.metrics.ndcgAt10 - a.validation.metrics.ndcgAt10 || - a.validation.noResultRate - b.validation.noResultRate || + b.validation.unanswerable.noResultRate - a.validation.unanswerable.noResultRate || a.threshold - b.threshold ) -const winner = ranked[0] -const production = rows.find((row) => row.threshold === PRODUCTION_THRESHOLD) +/** + * A flat sweep is not a weak recommendation, it is no recommendation: if no threshold + * changes either the answerable quality or the unanswerable refusal, the corpus cannot + * tell them apart and moving a product parameter on that evidence would be noise dressed + * as a result. + */ +const flat = rows.every( + (row) => + row.validation.metrics.ndcgAt10 === reference.validation.metrics.ndcgAt10 && + row.validation.unanswerable.noResultRate === reference.validation.unanswerable.noResultRate +) /** - * A flat sweep is not a weak recommendation, it is no recommendation: if every - * threshold scores the same on both metrics, the corpus cannot tell them apart and - * moving a product parameter on that evidence would be noise dressed as a result. + * Refusals are the point of the second number: `noResultCount` of the unanswerable + * questions came back empty, which is the correct outcome. The rest returned passages the + * sources cannot support. */ -const flat = - rows.every( - (row) => - row.validation.metrics.ndcgAt10 === rows[0].validation.metrics.ndcgAt10 && - row.validation.noResultRate === rows[0].validation.noResultRate - ) +const refusals = (row) => `${row.unanswerable.noResultCount}/${row.unanswerable.questions}` +const describe = (row) => + `nDCG@10 ${format4(row.metrics.ndcgAt10)}, Recall@5 ${format4(row.metrics.recallAt5)}, ` + + `answerable no-result ${format4(row.noResultRate)}, ` + + `unanswerable refused ${refusals(row)}` const outcome = flat - ? `The sweep is **flat**: every threshold from ${THRESHOLDS[0]} to ${THRESHOLDS[THRESHOLDS.length - 1]} produces the same validation nDCG@10 (${format4(rows[0].validation.metrics.ndcgAt10)}), the same Recall@5 (${format4(rows[0].validation.metrics.recallAt5)}) and a no-result rate of ${format4(rows[0].validation.noResultRate)}. On this corpus the threshold is simply **non-binding** — E5 never scores these query/chunk pairs below the top of the swept range, so no passage is ever filtered out.\n\n**No evidence to change \`threshold = ${PRODUCTION_THRESHOLD}\`.** The tie-break rule nominates \`${winner.threshold}\` only because it prefers the widest threshold among equals; that is a tie-break, not a finding. What this run establishes is that the current value cannot be validated *or* falsified here, which is a property of the corpus, not of the threshold. Re-run after #192 child 2 grows it.` - : `**Recommended: \`threshold = ${winner.threshold}\`.**\n\n- Validation: nDCG@10 ${format4(winner.validation.metrics.ndcgAt10)}, Recall@5 ${format4(winner.validation.metrics.recallAt5)}, no-result rate ${format4(winner.validation.noResultRate)} (${winner.validation.noResultCount}/${winner.validation.questions})\n- Test: nDCG@10 ${format4(winner.test.metrics.ndcgAt10)}, Recall@5 ${format4(winner.test.metrics.recallAt5)}, no-result rate ${format4(winner.test.noResultRate)} (${winner.test.noResultCount}/${winner.test.questions})\n- Mean retrieved per question: ${winner.test.meanRetrieved.toFixed(2)} (validation ${winner.validation.meanRetrieved.toFixed(2)})` + ? `The sweep is **flat**: every threshold from ${THRESHOLDS[0]} to ${THRESHOLDS[THRESHOLDS.length - 1]} produces the same validation nDCG@10 (${format4(reference.validation.metrics.ndcgAt10)}), the same Recall@5 (${format4(reference.validation.metrics.recallAt5)}) and the same unanswerable refusal rate (${refusals(reference.validation)}). No passage is ever filtered out, so the threshold is **non-binding** on this corpus — E5 does not score these query/chunk pairs below the top of the swept range.\n\n**No evidence to change \`threshold = ${PRODUCTION_THRESHOLD}\`.** All thresholds hold the line equally; picking one would be arbitrary. The current value can be neither validated nor falsified here, which is a property of the corpus rather than of the threshold.` + : `**Recommended: \`threshold = ${winner.threshold}\`.**\n\n- **Validation**: ${describe(winner.validation)}\n- **Test**: ${describe(winner.test)}\n- Production ships \`${PRODUCTION_THRESHOLD}\`: validation ${describe(production.validation)}.\n\nThe rule held answerable quality at the \`threshold = 0\` level (nDCG@10 and context recall must not regress, on the validation split) and then took the threshold that refuses the most unanswerable questions. So this is a refusal gain, not a quality gain — if answerable quality had fallen, the threshold would have been ineligible regardless of how much it refused.` const tableRows = rows .map( (row) => - `| ${row.threshold} | ${row.validation.questions} | ${format4(row.validation.metrics.recallAt5)} | ` + - `${format4(row.validation.metrics.ndcgAt10)} | ${format4(row.validation.metrics.mapAt10)} | ` + - `${format4(row.validation.noResultRate)} | ${format4(row.test.metrics.ndcgAt10)} | ` + - `${format4(row.test.noResultRate)} |` + `| ${row.threshold} | ${row.validation.answerableQuestions} | ${format4(row.validation.metrics.recallAt5)} | ` + + `${format4(row.validation.metrics.ndcgAt10)} | ${format4(row.validation.noResultRate)} | ` + + `${format4(row.validation.unanswerable.noResultRate)} | ` + + `${row.validation.unanswerable.meanRetrieved.toFixed(1)} | ${format4(row.test.metrics.ndcgAt10)} | ` + + `${format4(row.test.unanswerable.noResultRate)} |` ) .join('\n') @@ -179,35 +207,36 @@ Generated by \`node scripts/eval-threshold.mjs\`. Numbers are harness output; do The real harness, the same corpus and the production retrieval config (\`candidateK=20, contextK=3\`), once per candidate threshold. The **validation** -split selects; the **test** split reports. Both come from the same -\`--eval-split=\` code path, and the split is a deterministic function of the -question id, so this is reproducible. +split selects; the **test** split reports. The split is the committed manifest +\`eval/splits.json\`, so the same questions are on the same side on every machine. + +Quality columns cover the answerable questions only; **Unans.** columns cover the +unanswerable ones, where returning nothing is the desired outcome and so a *higher* +no-result rate is better. -| Threshold | n (val) | Recall@5 (val) | nDCG@10 (val) | MAP@10 (val) | No-result (val) | nDCG@10 (test) | No-result (test) | -| --- | --- | --- | --- | --- | --- | --- | --- | +| Threshold | n (val) | Recall@5 (val) | nDCG@10 (val) | No-result (val) | Unans. no-result (val) | Unans. retrieved (val) | nDCG@10 (test) | Unans. no-result (test) | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | ${tableRows} ## Selection rule -Best validation nDCG@10, then fewest validation no-results, then the **lowest** -threshold — the first stage is supposed to favour recall and let a later stage -filter, so among equals the wider one is the safer default. +Hold the answerable quality line — validation nDCG@10 and context recall must not +regress versus \`threshold = 0\` — then take the threshold that refuses the most +unanswerable questions. Tie-break on the lowest threshold. + +Raising a threshold is only worth anything if it refuses what the sources do not answer; +the quality gate is there so a refusal gain can never be bought with a retrieval loss. ## Outcome ${outcome} -The app currently ships \`threshold = ${PRODUCTION_THRESHOLD}\`: validation nDCG@10 -${format4(production.validation.metrics.ndcgAt10)}, test nDCG@10 -${format4(production.test.metrics.ndcgAt10)}, test no-result rate -${format4(production.test.noResultRate)}. - ## Caveat on this corpus -The split removes the most obvious form of overfitting, but ${rows[0].validation.questions} -validation questions is a thin basis for a decision, and the corpus is still small. A -threshold is a product decision with a **no-result-rate** cost attached, so a -recommendation here is only as good as the corpus behind it. Re-run this after the +The split removes the most obvious form of overfitting, but ${reference.validation.answerableQuestions} +answerable questions on the validation side is a thin basis for a decision, and the corpus +is still small. A threshold is a product decision with a **refusal-rate** cost attached, so +a recommendation here is only as good as the corpus behind it. Re-run this after the corpus grows (#192 child 2). ## Reproduce diff --git a/src/main/eval/harness.ts b/src/main/eval/harness.ts index 418ddf7..3297cc0 100644 --- a/src/main/eval/harness.ts +++ b/src/main/eval/harness.ts @@ -12,7 +12,7 @@ */ import { readdir, readFile } from 'fs/promises' -import { join, posix } from 'path' +import { join, posix, relative } from 'path' import { and, eq } from 'drizzle-orm' import { documentBlocks, notebooks, chunks } from '../db/schema' import type { getDatabase } from '../db' @@ -40,9 +40,10 @@ import type { EvalSplit, EvalTypeBreakdown, QuestionReport, - ResolvedGroundTruth + ResolvedGroundTruth, + SplitAssignment } from './types' -import { selectSplit } from './types' +import { assertQuestionShape, parseSplitAssignment, selectSplit } from './types' type Db = ReturnType @@ -52,6 +53,8 @@ export interface EvalHarnessOptions { /** Repo-relative label recorded in the report, so the JSON is machine-independent. */ corpusLabel: string questionsPath: string + /** 切分清单(`eval/splits.json`)。`split: 'all'` 时不读。 */ + splitsPath: string baseline: string /** 本次只评这一份切分(#192);缺省 `all`。 */ split: EvalSplit @@ -212,6 +215,24 @@ async function indexCorpus( return { documentIds, chunkCount, indexingMs: performance.now() - indexingStarted } } +/** + * 读切分清单。`all` 不需要清单,所以 `all` 的运行不会因为缺清单而失败。 + * + * JSON 解析错误会把文件路径带上:清单是提交在仓库里的,一份写坏的清单应该指向它自己。 + */ +async function loadSplitAssignment(options: EvalHarnessOptions): Promise { + if (options.split === 'all') return {} + + const label = relative(process.cwd(), options.splitsPath).split(/[\\/]/).join('/') + let parsed: unknown + try { + parsed = JSON.parse(await readFile(options.splitsPath, 'utf-8')) + } catch (error) { + throw new Error(`${label} could not be read as JSON: ${String(error)}`) + } + return parseSplitAssignment(parsed, label) +} + export async function runEvalHarness( db: Db, knowledgeService: KnowledgeService, @@ -238,7 +259,12 @@ export async function runEvalHarness( options.chunkOptions ) const allQuestions = parseQuestions(await readFile(options.questionsPath, 'utf-8')) - const questions = selectSplit(allQuestions, options.split) + // 数据集自身的契约先校验:可答必须有 ground truth,不可答必须没有。搞反时指标不会 + // 报错,只会静静地失去意义。 + for (const question of allQuestions) assertQuestionShape(question) + + const assignment = await loadSplitAssignment(options) + const questions = selectSplit(allQuestions, options.split, assignment) if (questions.length === 0) { throw new Error(`eval split "${options.split}" selected no questions from ${options.questionsPath}`) } @@ -276,6 +302,7 @@ export async function runEvalHarness( id: question.id, question: question.question, type: question.type ?? UNTAGGED, + answerable: question.answerable ?? true, firstRelevantRank: firstRelevantRank(matchesByRank), relevantCount: groundTruth.length, retrievedCount: results.length, @@ -286,16 +313,31 @@ export async function runEvalHarness( }) } - const metrics = summarize(perQuestion, options.contextK) + // 不可答的问题不进排名指标:它们没有 ground truth,`recallAtK` 对它们返回的是 0/0 + // 而不是 0,把“该拒答”算成“漏报”会让整张表失真。它们自成一组。 + const answerable = perQuestion.filter((q) => q.answerable) + const unanswerableQuestions = perQuestion.filter((q) => !q.answerable) + + const metrics = summarize(answerable, options.contextK) // 每个类别一行,按类别名排序,所以同一个 JSON 在两次运行之间可 diff。 - const byType: EvalTypeBreakdown[] = [...new Set(perQuestion.map((q) => q.type))] + const byType: EvalTypeBreakdown[] = [...new Set(answerable.map((q) => q.type))] .sort() .map((type) => { - const group = perQuestion.filter((q) => q.type === type) + const group = answerable.filter((q) => q.type === type) return { type, questions: group.length, metrics: summarize(group, options.contextK) } }) + const unanswerableNoResults = unanswerableQuestions.filter((q) => q.retrievedCount === 0).length + const unanswerable = { + questions: unanswerableQuestions.length, + noResultCount: unanswerableNoResults, + // 目标方向与其他指标相反:没有相关资料时,返回空才是对的。 + noResultRate: + unanswerableQuestions.length === 0 ? 0 : unanswerableNoResults / unanswerableQuestions.length, + meanRetrieved: mean(unanswerableQuestions.map((q) => q.retrievedCount)) + } + const chunking = { ...DEFAULT_CHUNK_OPTIONS, ...options.chunkOptions } return { baseline: options.baseline, @@ -321,6 +363,7 @@ export async function runEvalHarness( }, metrics, byType, + unanswerable, timing: { latencyP50Ms: percentile(latencies, 50), latencyP95Ms: percentile(latencies, 95), @@ -371,6 +414,7 @@ export function toDeterministicReport(report: EvalReport): EvalDeterministicRepo config: report.config, metrics: report.metrics, byType: report.byType, + unanswerable: report.unanswerable, perQuestion: report.perQuestion } } diff --git a/src/main/eval/report.ts b/src/main/eval/report.ts index 8f6dfaa..9c2b96f 100644 --- a/src/main/eval/report.ts +++ b/src/main/eval/report.ts @@ -24,6 +24,18 @@ export function renderMarkdown(report: EvalReport): string { ) .join('\n') + const unanswerable = report.unanswerable + // byType only ever covers answerable questions, so its sizes add up to that count. + const answerableCount = report.byType.reduce((total, entry) => total + entry.questions, 0) + const unanswerableNote = + unanswerable.questions === 0 + ? 'This corpus carries **no** unanswerable question yet, so refusal is not measured.\n' + + 'The threshold cannot be tuned against it either: every question is answerable, so\n' + + 'every threshold returns something.' + : `| Unanswerable questions | ${unanswerable.questions} |\n` + + `| Returned no results | ${format(unanswerable.noResultRate)} (${unanswerable.noResultCount}/${unanswerable.questions}) |\n` + + `| Mean passages retrieved | ${unanswerable.meanRetrieved.toFixed(2)} |` + return `# RAG eval baseline — ${report.baseline} Generated by \`${report.generatedBy}\`. The numbers below are harness output — do not edit them by hand. @@ -38,7 +50,7 @@ Generated by \`${report.generatedBy}\`. The numbers below are harness output — | Ranks | \`candidateK=${config.candidateK}, threshold=${config.threshold}\` | | Context width | \`contextK=${config.contextK}\` | | Corpus | \`${config.corpus}\` (${config.documents} documents) | -| Split | \`${config.split}\` (${config.questions} questions) | +| Split | \`${config.split}\` (${config.questions} questions, ${answerableCount} answerable) | | Index size | ${config.chunkCount} chunks | ## Metrics @@ -65,6 +77,18 @@ The type comes from \`type\` in \`questions.jsonl\`; untagged questions report a | --- | --- | --- | --- | --- | --- | ${typeRows} +### Unanswerable questions + +These carry no ground truth, so the correct outcome is that retrieval finds nothing. They +are excluded from every metric above — a missing ground truth is not a miss — and reported +here instead. A higher **no-results** rate is better on this row, which is the opposite of +how it reads everywhere else, and \`mean passages retrieved\` is how much irrelevant context +was pulled in anyway. This is the row a threshold decision should move. + +| Metric | Value | +| --- | --- | +${unanswerableNote} + Timing is informational only and is **not** frozen: indexing ${timing.indexingMs} ms, query p50 ${timing.latencyP50Ms.toFixed(2)} ms, p95 ${timing.latencyP95Ms.toFixed(2)} ms on the machine that produced this file. Timing and index size depend on hardware and on the diff --git a/src/main/eval/run.ts b/src/main/eval/run.ts index d927286..20c13fa 100644 --- a/src/main/eval/run.ts +++ b/src/main/eval/run.ts @@ -141,6 +141,7 @@ export async function runEvalCli(argv: readonly string[] = process.argv): Promis const prepare = argv.includes(EVAL_PREPARE_FLAG) const corpusDir = resolve(readOption(argv, '--eval-corpus=', 'eval/corpus')) const questionsPath = resolve(readOption(argv, '--eval-questions=', 'eval/questions.jsonl')) + const splitsPath = resolve(readOption(argv, '--eval-splits=', 'eval/splits.json')) const outDir = resolve(readOption(argv, '--eval-out=', 'docs/eval')) // The real profile is captured before redirecting: the model cache lives under @@ -190,6 +191,7 @@ export async function runEvalCli(argv: readonly string[] = process.argv): Promis // committed JSON is identical on every machine and checkout. corpusLabel: repoRelative(corpusDir) || 'eval/corpus', questionsPath, + splitsPath, baseline: readOption(argv, '--eval-baseline=', 'v1.6'), split: readSplit(argv), // 默认就是生产配置(#77):先取宽,融合,再把 contextK 条送进 prompt。一个不镜像 diff --git a/src/main/eval/types.ts b/src/main/eval/types.ts index bb7f00e..de162bf 100644 --- a/src/main/eval/types.ts +++ b/src/main/eval/types.ts @@ -24,6 +24,14 @@ export interface EvalQuestion { question: string relevant: EvalRelevantLocation[] goldAnswer?: string + /** + * 这份语料里能不能回答(#192)。缺省 `true`。 + * + * `false` 的问题必须 `relevant: []`:它的正确答案是“资料里没有”,所以既不能拿 + * Recall 去惩罚它,也不能让它的“命中”看起来像成功。它评的是另一件事:该拒答的 + * 时候,检索有没有硬找出一堆相似但无关的上下文。 + */ + answerable?: boolean /** * 查询类别(#192)。自由字符串,因为语料还会长出新类别;报告按出现过的值分组, * 缺省归入 `untagged`。 @@ -43,18 +51,78 @@ export interface EvalQuestion { */ export type EvalSplit = 'all' | 'validation' | 'test' -/** id 分桶,0/1/2;只用于切分,不参与检索。 */ -export function splitBucket(id: string): number { - let hash = 0 - for (const character of id) hash = (hash * 31 + character.charCodeAt(0)) >>> 0 - return hash % 3 +/** 问题 id → 它属于切分的哪一边。 */ +export type SplitAssignment = Record + +/** + * 解析提交在仓库里的切分清单(`eval/splits.json`)。 + * + * 用显式清单而不是 id 哈希(#192 评审):哈希看着确定,但它的确定是“每次结果一样”, + * 不是“每次划分一样”——往 `questions.jsonl` 里加一道题,会把其它题在 validation / + * test 之间挪动,而一个稀有类别(multi-hop、cross-lingual)可以在无人选择的情况下整体 + * 落到某一边。清单让划分是被 review 的,不是被算出来的。 + */ +export function parseSplitAssignment(raw: unknown, source: string): SplitAssignment { + if (!raw || typeof raw !== 'object' || Array.isArray(raw)) { + throw new Error( + `${source} must be a JSON object mapping a question id to "validation" or "test"` + ) + } + + const assignment: SplitAssignment = {} + for (const [id, side] of Object.entries(raw as Record)) { + if (side !== 'validation' && side !== 'test') { + throw new Error( + `${source}: question ${id} is assigned ${JSON.stringify(side)}; ` + + 'expected "validation" or "test"' + ) + } + assignment[id] = side + } + return assignment } -/** validation 是 bucket 0(约 1/3),test 是其余(约 2/3)。 */ -export function selectSplit(questions: EvalQuestion[], split: EvalSplit): EvalQuestion[] { - if (split === 'all') return questions - const wantValidation = split === 'validation' - return questions.filter((question) => (splitBucket(question.id) === 0) === wantValidation) +/** + * 选出一份切分。`all` 不需要清单;`validation`/`test` **必须**在清单里有条目。 + * + * 缺条目就报错,而不是默认归入某一边:一个刚加进来的问题应当先被人工决定属于哪一边, + * 而不是静静泄漏进 test。 + */ +export function selectSplit( + questions: readonly EvalQuestion[], + split: EvalSplit, + assignment: SplitAssignment +): EvalQuestion[] { + if (split === 'all') return [...questions] + return questions.filter((question) => { + const side = assignment[question.id] + if (side === undefined) { + throw new Error( + `question ${question.id} has no entry in the split manifest; assign it to ` + + '"validation" or "test" deliberately rather than letting it default into a side' + ) + } + return side === split + }) +} + +/** + * 数据集自身的契约(#192):可答的必须有 ground truth,不可答的必须没有。 + * + * 两者搞反时指标不会报错,只会静静地失去意义 —— 一个可答但没有 ground truth 的问题 + * 会被当成永远漏报,一个不可答却带着 ground truth 的问题会被当成正常命中。 + */ +export function assertQuestionShape(question: EvalQuestion): void { + const answerable = question.answerable ?? true + if (answerable && question.relevant.length === 0) { + throw new Error(`question ${question.id} is answerable but has no ground truth`) + } + if (!answerable && question.relevant.length > 0) { + throw new Error( + `question ${question.id} is unanswerable but carries ${question.relevant.length} ` + + 'ground-truth location(s)' + ) + } } /** One resolved ground-truth location, after runtime id mapping. */ @@ -109,6 +177,8 @@ export interface QuestionReport { question: string /** 查询类别,与 `EvalQuestion.type` 一致;缺省为 `untagged`。 */ type: string + /** 与 `EvalQuestion.answerable` 一致(缺省 true)。 */ + answerable: boolean firstRelevantRank: number relevantCount: number retrievedCount: number @@ -159,8 +229,21 @@ export interface EvalReport { chunkCount: number } metrics: EvalMetrics - /** 每个查询类别一行;类别来自 `questions.jsonl` 的 `type`。 */ + /** 每个查询类别一行;类别来自 `questions.jsonl` 的 `type`。只含可答的问题。 */ byType: EvalTypeBreakdown[] + /** + * 不可答问题的单独一组(#192)。 + * + * 它们不进 `metrics`/`byType`:没有 ground truth,Recall 对它们是 0/0 而不是 0。 + * 它们评的是“该拒答时有没有硬找”——`noResultRate` 越接近 1 越好(在真的没有相关 + * 资料时返回空),`meanRetrieved` 则是“硬找了多少条相似但无关的上下文”。 + */ + unanswerable: { + questions: number + noResultCount: number + noResultRate: number + meanRetrieved: number + } /** `indexingMs` 只用于 #78 的吞吐比较;它不在确定报告里,也不该成为差异原因。 */ timing: { latencyP50Ms: number; latencyP95Ms: number; indexingMs: number } perQuestion: QuestionReport[] @@ -177,5 +260,6 @@ export interface EvalDeterministicReport { config: EvalReport['config'] metrics: EvalMetrics byType: EvalTypeBreakdown[] + unanswerable: EvalReport['unanswerable'] perQuestion: QuestionReport[] } diff --git a/test/evalHarness.test.ts b/test/evalHarness.test.ts index a5e650b..65a51e1 100644 --- a/test/evalHarness.test.ts +++ b/test/evalHarness.test.ts @@ -12,6 +12,7 @@ const options = (overrides: Record = {}): Record ({ @@ -20,42 +23,83 @@ const question = (id: string): EvalQuestion => ({ relevant: [{ document: 'a.md', page: null, block: 0 }] }) -const questions = Array.from({ length: 30 }, (_, index) => question(`q${String(index + 1).padStart(3, '0')}`)) +const assignment = { q1: 'validation', q2: 'test', q3: 'test' } as const -test('the split buckets only ever return 0, 1 or 2', () => { - for (const q of questions) { - const bucket = splitBucket(q.id) - assert.ok(bucket === 0 || bucket === 1 || bucket === 2, `${q.id} -> ${bucket}`) +test('the manifest maps ids to one of the two sides', () => { + assert.deepEqual(parseSplitAssignment({ q1: 'validation', q2: 'test' }, 'splits.json'), { + q1: 'validation', + q2: 'test' + }) +}) + +test('a manifest that is not an object is rejected, naming the file', () => { + for (const bad of [null, undefined, [], 'validation', 3]) { + assert.throws( + () => parseSplitAssignment(bad, 'eval/splits.json'), + /eval\/splits\.json must be a JSON object/ + ) } }) -test('the split is deterministic, not random', () => { - const first = selectSplit(questions, 'validation').map((q) => q.id) - const second = selectSplit(questions, 'validation').map((q) => q.id) - assert.deepEqual(first, second) +test('an unknown side is rejected and names the question', () => { + assert.throws( + () => parseSplitAssignment({ q7: 'train' }, 'eval/splits.json'), + /question q7 is assigned "train"/ + ) +}) + +test('`all` is the whole set and needs no manifest', () => { + const questions = [question('q1'), question('q2')] + assert.deepEqual( + selectSplit(questions, 'all', {}).map((q) => q.id), + ['q1', 'q2'] + ) }) -test('validation and test partition the questions without overlap', () => { - const validation = selectSplit(questions, 'validation').map((q) => q.id) - const test = selectSplit(questions, 'test').map((q) => q.id) +test('the two sides partition the questions and keep their order', () => { + const questions = [question('q1'), question('q2'), question('q3')] + const validation = selectSplit(questions, 'validation', assignment) + const testSide = selectSplit(questions, 'test', assignment) - assert.equal(validation.length + test.length, questions.length) - assert.equal(new Set([...validation, ...test]).size, questions.length) - // Both sides are non-empty on a 30-question set, so neither arm is vacuous. - assert.ok(validation.length > 0 && test.length > 0) + assert.deepEqual(validation.map((q) => q.id), ['q1']) + assert.deepEqual(testSide.map((q) => q.id), ['q2', 'q3']) + assert.equal(validation.length + testSide.length, questions.length) }) -test('`all` is the whole set, and the selected questions keep their order', () => { - assert.deepEqual( - selectSplit(questions, 'all').map((q) => q.id), - questions.map((q) => q.id) +/** + * A new question must be assigned deliberately. Defaulting it into `test` would leak it + * into the reporting side, which is the side a choice must not be fitted to. + */ +test('a question with no manifest entry is refused, not defaulted', () => { + assert.throws( + () => selectSplit([question('q9')], 'validation', assignment), + /question q9 has no entry in the split manifest/ ) +}) - for (const split of ['validation', 'test'] as EvalSplit[]) { - const selected = selectSplit(questions, split).map((q) => q.id) - const expected = questions - .filter((q) => (splitBucket(q.id) === 0) === (split === 'validation')) - .map((q) => q.id) - assert.deepEqual(selected, expected) - } +/** + * Reversing these two is silent: an answerable question with no ground truth reads as a + * permanent miss, and an unanswerable one with ground truth reads as a normal hit. + */ +test('the dataset contract ties answerability to having ground truth', () => { + const answerable = question('q1') + const unanswerable: EvalQuestion = { id: 'q2', question: 'q2', answerable: false, relevant: [] } + + assert.doesNotThrow(() => assertQuestionShape(answerable)) + assert.doesNotThrow(() => assertQuestionShape(unanswerable)) + + assert.throws( + () => assertQuestionShape({ ...answerable, relevant: [] }), + /question q1 is answerable but has no ground truth/ + ) + assert.throws( + () => assertQuestionShape({ ...unanswerable, relevant: [answerable.relevant[0]] }), + /question q2 is unanswerable but carries 1 ground-truth location/ + ) +}) + +test('omitting `answerable` means answerable', () => { + // The pre-#192 dataset has no `answerable` field at all, and every one of those + // questions is answerable; the default has to preserve that. + assert.doesNotThrow(() => assertQuestionShape(question('q1'))) })