From 54c892fb50cef4325e254fbe499ab228ad176239 Mon Sep 17 00:00:00 2001 From: MrSibe Date: Wed, 30 Sep 2026 16:49:34 +0800 Subject: [PATCH] feat(eval): hit rate, MAP, and per-query-type metrics MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Recall@K, MRR and nDCG@10 could not express three things the epic needs, and a single aggregate average was hiding a gap the corpus already contained. Child 3 of #192. **Two metrics added.** `hitRate@5` is whether *any* of the first five passages covers ground truth; `MAP@10` combines ranking position with coverage, so pulling a second relevant passage from rank 9 to rank 2 moves it while Recall@5 sits still. Hit rate is deliberately blunt next to Recall@5: a two-passage question that finds one scores 1.0 and 0.5 respectively, and both facts matter — "the model had a chance" is not "the material was complete". **Every metric is now reported per query type.** `questions.jsonl` gained an optional `type`, the harness groups by it, and the report renders a table. The aggregate was already concealing something: | Type | n | Recall@5 | nDCG@10 | Hit rate@5 | MAP@10 | | --- | --- | --- | --- | --- | --- | | cross-lingual | 1 | 1.0000 | **0.6309** | 1.0000 | **0.5000** | | exact | 6 | 1.0000 | 1.0000 | 1.0000 | 1.0000 | | multi-hop | 2 | 1.0000 | 0.9599 | 1.0000 | 0.9167 | | semantic | 18 | 1.0000 | 0.9312 | 1.0000 | 0.9074 | | zh | 3 | 1.0000 | 1.0000 | 1.0000 | 1.0000 | | **all** | 30 | 1.0000 | 0.9437 | 1.0000 | 0.9222 | The one cross-lingual question (`q030`, Chinese over an English source) ranks far worse than everything else. Recall@5 = 1.0000 reported that as a success; nDCG@10 and MAP@10 are what make the multilingual gap visible. That is the metric doing its job on the existing 30 questions, before the corpus grows. Untagged questions group under `untagged` rather than being dropped, and the types in use are documented in `eval/README.md`. ## Testing - `npm run typecheck` — clean - `npm test` — 485 pass, 4 new: hit rate vs recall, hit rate@k bounds, AP position sensitivity, AP's repeat-counts-once rule - `npm run eval` twice — byte-identical `docs/eval/baseline-v1.6.json` Part of #192 (child 3). --- docs/eval/baseline-v1.6.json | 104 +++++++++++++++++++++++++++++++++++ docs/eval/baseline-v1.6.md | 20 ++++++- eval/README.md | 14 +++++ eval/questions.jsonl | 60 ++++++++++---------- src/main/eval/harness.ts | 80 ++++++++++++++++++++------- src/main/eval/metrics.ts | 50 +++++++++++++++++ src/main/eval/report.ts | 23 ++++++++ src/main/eval/types.ts | 29 ++++++++++ test/evalMetrics.test.ts | 38 +++++++++++++ 9 files changed, 365 insertions(+), 53 deletions(-) diff --git a/docs/eval/baseline-v1.6.json b/docs/eval/baseline-v1.6.json index eb15f14..ca5f93b 100644 --- a/docs/eval/baseline-v1.6.json +++ b/docs/eval/baseline-v1.6.json @@ -26,12 +26,87 @@ "recallAt10": 1, "mrr": 0.927778, "ndcgAt10": 0.94375, + "hitRateAt5": 1, + "mapAt10": 0.922222, "evidencePrecisionAt5": 0.213333 }, + "byType": [ + { + "type": "cross-lingual", + "questions": 1, + "metrics": { + "recallAt1": 0, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 0.5, + "ndcgAt10": 0.63093, + "hitRateAt5": 1, + "mapAt10": 0.5, + "evidencePrecisionAt5": 0.2 + } + }, + { + "type": "exact", + "questions": 6, + "metrics": { + "recallAt1": 1, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 1, + "ndcgAt10": 1, + "hitRateAt5": 1, + "mapAt10": 1, + "evidencePrecisionAt5": 0.2 + } + }, + { + "type": "multi-hop", + "questions": 2, + "metrics": { + "recallAt1": 0.5, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 1, + "ndcgAt10": 0.95986, + "hitRateAt5": 1, + "mapAt10": 0.916667, + "evidencePrecisionAt5": 0.4 + } + }, + { + "type": "semantic", + "questions": 18, + "metrics": { + "recallAt1": 0.833333, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 0.907407, + "ndcgAt10": 0.931214, + "hitRateAt5": 1, + "mapAt10": 0.907407, + "evidencePrecisionAt5": 0.2 + } + }, + { + "type": "zh", + "questions": 3, + "metrics": { + "recallAt1": 1, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 1, + "ndcgAt10": 1, + "hitRateAt5": 1, + "mapAt10": 1, + "evidencePrecisionAt5": 0.2 + } + } + ], "perQuestion": [ { "id": "q001", "question": "Why is bedload harder to measure than suspended sediment?", + "type": "semantic", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -62,6 +137,7 @@ { "id": "q002", "question": "How many replicate samples are collected at each river station?", + "type": "exact", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -92,6 +168,7 @@ { "id": "q003", "question": "What is the central trade-off in lithium-ion cell design?", + "type": "semantic", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -122,6 +199,7 @@ { "id": "q004", "question": "Why do nickel-rich battery packs need more aggressive thermal management?", + "type": "semantic", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -152,6 +230,7 @@ { "id": "q005", "question": "What happens once the separator in a battery cell melts?", + "type": "semantic", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -182,6 +261,7 @@ { "id": "q006", "question": "At what temperature do honeybees begin to forage?", + "type": "exact", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -212,6 +292,7 @@ { "id": "q007", "question": "What does a late frost damage during full bloom?", + "type": "semantic", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -242,6 +323,7 @@ { "id": "q008", "question": "Why is a continuous tree canopy more effective at cooling than isolated trees?", + "type": "semantic", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -272,6 +354,7 @@ { "id": "q009", "question": "Why are trees with aggressive surface roots unsuitable for narrow verges?", + "type": "semantic", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -302,6 +385,7 @@ { "id": "q010", "question": "At what temperature is lactic acid fermentation fastest?", + "type": "exact", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -332,6 +416,7 @@ { "id": "q011", "question": "Is the salt percentage in fermentation based on vegetable weight or water weight?", + "type": "exact", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -362,6 +447,7 @@ { "id": "q012", "question": "Why must tidal turbines be sited in places with very fast currents?", + "type": "semantic", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -392,6 +478,7 @@ { "id": "q013", "question": "What is the main environmental concern for tidal energy installations?", + "type": "semantic", "firstRelevantRank": 3, "relevantCount": 1, "retrievedCount": 19, @@ -422,6 +509,7 @@ { "id": "q014", "question": "In lake monitoring, how is the sampling depth actually recorded?", + "type": "exact", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -452,6 +540,7 @@ { "id": "q015", "question": "Why does deep-water oxygen fall while a lake remains stratified?", + "type": "semantic", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -482,6 +571,7 @@ { "id": "q016", "question": "How do supercapacitors hold their charge?", + "type": "semantic", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -512,6 +602,7 @@ { "id": "q017", "question": "Why can solitary bees pollinate a bloom week that is too cold for honeybee hives?", + "type": "semantic", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -542,6 +633,7 @@ { "id": "q018", "question": "Why is one continuous planted roof layer better than several isolated planted beds?", + "type": "semantic", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -572,6 +664,7 @@ { "id": "q019", "question": "Where do acetic acid bacteria sit in a vinegar culture?", + "type": "semantic", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -602,6 +695,7 @@ { "id": "q020", "question": "What happens if a vinegar culture is sealed airtight?", + "type": "semantic", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -632,6 +726,7 @@ { "id": "q021", "question": "Why is wave energy harder to schedule ahead than tidal energy?", + "type": "semantic", "firstRelevantRank": 2, "relevantCount": 1, "retrievedCount": 19, @@ -662,6 +757,7 @@ { "id": "q022", "question": "Where does siting for wave energy devices concentrate, and where does it not?", + "type": "exact", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -692,6 +788,7 @@ { "id": "q023", "question": "How does the river sampling protocol differ from the lake sampling protocol?", + "type": "multi-hop", "firstRelevantRank": 1, "relevantCount": 2, "retrievedCount": 19, @@ -724,6 +821,7 @@ { "id": "q024", "question": "A street canopy and a green roof are both said to cool; what surface does each one shade?", + "type": "multi-hop", "firstRelevantRank": 1, "relevantCount": 2, "retrievedCount": 19, @@ -756,6 +854,7 @@ { "id": "q025", "question": "Which preservation method depends on keeping air away from the food?", + "type": "semantic", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -786,6 +885,7 @@ { "id": "q026", "question": "Why can one cold morning cost a grower the whole crop even when colonies are brought in?", + "type": "semantic", "firstRelevantRank": 2, "relevantCount": 1, "retrievedCount": 19, @@ -816,6 +916,7 @@ { "id": "q027", "question": "绿茶应该怎样保存才能减缓氧化?", + "type": "zh", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -846,6 +947,7 @@ { "id": "q028", "question": "茶叶储存的相对湿度上限是多少?", + "type": "zh", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -876,6 +978,7 @@ { "id": "q029", "question": "为什么冷冻保存的茶叶取出后不能立刻打开包装?", + "type": "zh", "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, @@ -906,6 +1009,7 @@ { "id": "q030", "question": "为什么潮汐能比风能和太阳能更容易提前安排发电?", + "type": "cross-lingual", "firstRelevantRank": 2, "relevantCount": 1, "retrievedCount": 19, diff --git a/docs/eval/baseline-v1.6.md b/docs/eval/baseline-v1.6.md index 5418dd6..df1c22c 100644 --- a/docs/eval/baseline-v1.6.md +++ b/docs/eval/baseline-v1.6.md @@ -24,10 +24,26 @@ Generated by `npm run eval`. The numbers below are harness output — do not edi | Recall@10 | 1.0000 | | MRR | 0.9278 | | nDCG@10 | 0.9437 | +| Hit rate@5 | 1.0000 | +| MAP@10 | 0.9222 | | Evidence precision@5 | 0.2133 | -Timing is informational only and is **not** frozen: indexing 1582 ms, query -p50 11.35 ms, p95 18.82 ms on the +### By query type + +A single average hides a change that helps one kind of question and hurts another. +The type comes from `type` in `questions.jsonl`; untagged questions report as +`untagged` rather than disappearing. + +| Type | Questions | Recall@5 | nDCG@10 | Hit rate@5 | MAP@10 | +| --- | --- | --- | --- | --- | --- | +| cross-lingual | 1 | 1.0000 | 0.6309 | 1.0000 | 0.5000 | +| exact | 6 | 1.0000 | 1.0000 | 1.0000 | 1.0000 | +| multi-hop | 2 | 1.0000 | 0.9599 | 1.0000 | 0.9167 | +| semantic | 18 | 1.0000 | 0.9312 | 1.0000 | 0.9074 | +| zh | 3 | 1.0000 | 1.0000 | 1.0000 | 1.0000 | + +Timing is informational only and is **not** frozen: indexing 1545 ms, query +p50 11.04 ms, p95 14.14 ms on the machine that produced this file. Timing and index size depend on hardware and on the corpus, so they must never be the reason two runs differ. diff --git a/eval/README.md b/eval/README.md index 191ac61..c8964f1 100644 --- a/eval/README.md +++ b/eval/README.md @@ -59,6 +59,7 @@ dataset needs a new question. { "id": "q001", "question": "Why is bedload harder to measure than suspended sediment?", + "type": "semantic", "relevant": [ { "document": "river-monitoring.md", @@ -78,6 +79,10 @@ Ground truth uses **corpus identity, never database identity**: - `page` is `null` for unpaginated sources. - `quote` is an optional excerpt. The runner fails if the referenced block no longer contains it, so a parser change cannot silently move the ground truth. +- `type` is an optional query class; the report groups every metric by it. Values in use: + `exact` (number/name/detail), `semantic` (why/how), `multi-hop` (two or more blocks), + `cross-lingual` (question language differs from the source), `zh` (Chinese over a + Chinese source). Untagged questions report as `untagged`. Runtime `documentId`s are random and `blockId`s embed them, so neither may appear here. This is what lets #78 change chunking without invalidating the dataset: the ground truth @@ -115,11 +120,20 @@ from unanswerable queries. - **Recall@1/5/10** — share of ground-truth blocks covered by the first k passages. - **MRR** — reciprocal rank of the first relevant passage. - **nDCG@10** — binary-gain discounted cumulative gain. +- **Hit rate@5** — share of questions with at least one relevant passage in the first 5. + Deliberately blunt: it says the answer was *reachable*, where Recall@5 says the material + was *complete*. A two-passage question that finds one scores 1.0 and 0.5 respectively. +- **MAP@10** — mean average precision. The one metric here that combines ranking position + with coverage, so pulling a second relevant passage from rank 9 to rank 2 moves it. - **Evidence precision@5** — of the first 5 retrieved passages, the share that cover a ground-truth block. This is **retrieval precision, not answer citation recall**: the harness runs no model and produces no answer. Answer-level citation correctness is covered by the resolver (#70); a model-driven answer eval would be a separate deliverable. +- **By query type** — the same metrics per `type` in `questions.jsonl` (`exact`, + `semantic`, `multi-hop`, `cross-lingual`, `zh`). A single average hides a change that + helps one kind of question and hurts another; the current baseline already shows this, + with `cross-lingual` at nDCG 0.63 against 0.93–1.00 elsewhere. - **Latency p50/p95** — informational only. Timing is **not** frozen, and the committed JSON excludes it so two runs diff cleanly. diff --git a/eval/questions.jsonl b/eval/questions.jsonl index e125d9c..de15b0c 100644 --- a/eval/questions.jsonl +++ b/eval/questions.jsonl @@ -1,30 +1,30 @@ -{"id":"q001","question":"Why is bedload harder to measure than suspended sediment?","relevant":[{"document":"river-monitoring.md","page":null,"block":4,"quote":"Bedload is the harder fraction to measure"}]} -{"id":"q002","question":"How many replicate samples are collected at each river station?","relevant":[{"document":"river-monitoring.md","page":null,"block":6,"quote":"three replicate samples at each station"}]} -{"id":"q003","question":"What is the central trade-off in lithium-ion cell design?","relevant":[{"document":"battery-chemistry.md","page":null,"block":2,"quote":"trade energy density against thermal stability"}]} -{"id":"q004","question":"Why do nickel-rich battery packs need more aggressive thermal management?","relevant":[{"document":"battery-chemistry.md","page":null,"block":4,"quote":"release oxygen at lower temperatures than iron phosphate"}]} -{"id":"q005","question":"What happens once the separator in a battery cell melts?","relevant":[{"document":"battery-chemistry.md","page":null,"block":6,"quote":"Once the separator melts, the cell shorts internally"}]} -{"id":"q006","question":"At what temperature do honeybees begin to forage?","relevant":[{"document":"orchard-pollination.md","page":null,"block":4,"quote":"above roughly twelve degrees Celsius"}]} -{"id":"q007","question":"What does a late frost damage during full bloom?","relevant":[{"document":"orchard-pollination.md","page":null,"block":6,"quote":"destroys the flower's ovary rather than the petals"}]} -{"id":"q008","question":"Why is a continuous tree canopy more effective at cooling than isolated trees?","relevant":[{"document":"urban-canopy.md","page":null,"block":4,"quote":"asphalt stores far more heat during the day"}]} -{"id":"q009","question":"Why are trees with aggressive surface roots unsuitable for narrow verges?","relevant":[{"document":"urban-canopy.md","page":null,"block":6,"quote":"aggressive surface roots lift pavements"}]} -{"id":"q010","question":"At what temperature is lactic acid fermentation fastest?","relevant":[{"document":"fermentation.md","page":null,"block":4,"quote":"fastest between twenty and twenty-four degrees Celsius"}]} -{"id":"q011","question":"Is the salt percentage in fermentation based on vegetable weight or water weight?","relevant":[{"document":"fermentation.md","page":null,"block":6,"quote":"percentage of the vegetable weight, not the water weight"}]} -{"id":"q012","question":"Why must tidal turbines be sited in places with very fast currents?","relevant":[{"document":"tidal-energy.md","page":null,"block":4,"quote":"power scales with the cube of velocity"}]} -{"id":"q013","question":"What is the main environmental concern for tidal energy installations?","relevant":[{"document":"tidal-energy.md","page":null,"block":6,"quote":"change in sediment transport"}]} -{"id":"q014","question":"In lake monitoring, how is the sampling depth actually recorded?","relevant":[{"document":"lake-monitoring.md","page":null,"block":6,"quote":"Depth is recorded from the profiling sonde"}]} -{"id":"q015","question":"Why does deep-water oxygen fall while a lake remains stratified?","relevant":[{"document":"lake-monitoring.md","page":null,"block":4,"quote":"Oxygen in the hypolimnion is not replenished"}]} -{"id":"q016","question":"How do supercapacitors hold their charge?","relevant":[{"document":"supercapacitors.md","page":null,"block":4,"quote":"electric double layer formed at the surface of a porous carbon electrode"}]} -{"id":"q017","question":"Why can solitary bees pollinate a bloom week that is too cold for honeybee hives?","relevant":[{"document":"wild-pollinators.md","page":null,"block":4,"quote":"forage at lower temperatures than honeybees"}]} -{"id":"q018","question":"Why is one continuous planted roof layer better than several isolated planted beds?","relevant":[{"document":"green-roofs.md","page":null,"block":4,"quote":"an exposed membrane absorbs far more heat during the day"}]} -{"id":"q019","question":"Where do acetic acid bacteria sit in a vinegar culture?","relevant":[{"document":"vinegar-production.md","page":null,"block":4,"quote":"surface of the liquid where oxygen is available"}]} -{"id":"q020","question":"What happens if a vinegar culture is sealed airtight?","relevant":[{"document":"vinegar-production.md","page":null,"block":6,"quote":"Sealing a vinegar culture airtight stops the conversion entirely"}]} -{"id":"q021","question":"Why is wave energy harder to schedule ahead than tidal energy?","relevant":[{"document":"wave-energy.md","page":null,"block":2,"quote":"driven by wind rather than by the moon"}]} -{"id":"q022","question":"Where does siting for wave energy devices concentrate, and where does it not?","relevant":[{"document":"wave-energy.md","page":null,"block":4,"quote":"exposed headlands rather than on sheltered channels"}]} -{"id":"q023","question":"How does the river sampling protocol differ from the lake sampling protocol?","relevant":[{"document":"river-monitoring.md","page":null,"block":6,"quote":"three replicate samples at each station"},{"document":"lake-monitoring.md","page":null,"block":6,"quote":"Depth is recorded from the profiling sonde"}]} -{"id":"q024","question":"A street canopy and a green roof are both said to cool; what surface does each one shade?","relevant":[{"document":"urban-canopy.md","page":null,"block":4,"quote":"asphalt stores far more heat during the day"},{"document":"green-roofs.md","page":null,"block":4,"quote":"an exposed membrane absorbs far more heat during the day"}]} -{"id":"q025","question":"Which preservation method depends on keeping air away from the food?","relevant":[{"document":"fermentation.md","page":null,"block":2,"quote":"an anaerobic environment"}]} -{"id":"q026","question":"Why can one cold morning cost a grower the whole crop even when colonies are brought in?","relevant":[{"document":"orchard-pollination.md","page":null,"block":4,"quote":"above roughly twelve degrees Celsius"}]} -{"id":"q027","question":"绿茶应该怎样保存才能减缓氧化?","relevant":[{"document":"chinese-tea-storage.md","page":null,"block":2,"quote":"绿茶最容易氧化变质,需要低温密封保存"}]} -{"id":"q028","question":"茶叶储存的相对湿度上限是多少?","relevant":[{"document":"chinese-tea-storage.md","page":null,"block":4,"quote":"相对湿度应保持在百分之五十以下"}]} -{"id":"q029","question":"为什么冷冻保存的茶叶取出后不能立刻打开包装?","relevant":[{"document":"chinese-tea-storage.md","page":null,"block":6,"quote":"冷凝水会直接落在茶叶上"}]} -{"id":"q030","question":"为什么潮汐能比风能和太阳能更容易提前安排发电?","relevant":[{"document":"tidal-energy.md","page":null,"block":2,"quote":"predictable decades ahead, unlike wind or solar"}]} +{"id":"q001","question":"Why is bedload harder to measure than suspended sediment?","relevant":[{"document":"river-monitoring.md","page":null,"block":4,"quote":"Bedload is the harder fraction to measure"}],"type":"semantic"} +{"id":"q002","question":"How many replicate samples are collected at each river station?","relevant":[{"document":"river-monitoring.md","page":null,"block":6,"quote":"three replicate samples at each station"}],"type":"exact"} +{"id":"q003","question":"What is the central trade-off in lithium-ion cell design?","relevant":[{"document":"battery-chemistry.md","page":null,"block":2,"quote":"trade energy density against thermal stability"}],"type":"semantic"} +{"id":"q004","question":"Why do nickel-rich battery packs need more aggressive thermal management?","relevant":[{"document":"battery-chemistry.md","page":null,"block":4,"quote":"release oxygen at lower temperatures than iron phosphate"}],"type":"semantic"} +{"id":"q005","question":"What happens once the separator in a battery cell melts?","relevant":[{"document":"battery-chemistry.md","page":null,"block":6,"quote":"Once the separator melts, the cell shorts internally"}],"type":"semantic"} +{"id":"q006","question":"At what temperature do honeybees begin to forage?","relevant":[{"document":"orchard-pollination.md","page":null,"block":4,"quote":"above roughly twelve degrees Celsius"}],"type":"exact"} +{"id":"q007","question":"What does a late frost damage during full bloom?","relevant":[{"document":"orchard-pollination.md","page":null,"block":6,"quote":"destroys the flower's ovary rather than the petals"}],"type":"semantic"} +{"id":"q008","question":"Why is a continuous tree canopy more effective at cooling than isolated trees?","relevant":[{"document":"urban-canopy.md","page":null,"block":4,"quote":"asphalt stores far more heat during the day"}],"type":"semantic"} +{"id":"q009","question":"Why are trees with aggressive surface roots unsuitable for narrow verges?","relevant":[{"document":"urban-canopy.md","page":null,"block":6,"quote":"aggressive surface roots lift pavements"}],"type":"semantic"} +{"id":"q010","question":"At what temperature is lactic acid fermentation fastest?","relevant":[{"document":"fermentation.md","page":null,"block":4,"quote":"fastest between twenty and twenty-four degrees Celsius"}],"type":"exact"} +{"id":"q011","question":"Is the salt percentage in fermentation based on vegetable weight or water weight?","relevant":[{"document":"fermentation.md","page":null,"block":6,"quote":"percentage of the vegetable weight, not the water weight"}],"type":"exact"} +{"id":"q012","question":"Why must tidal turbines be sited in places with very fast currents?","relevant":[{"document":"tidal-energy.md","page":null,"block":4,"quote":"power scales with the cube of velocity"}],"type":"semantic"} +{"id":"q013","question":"What is the main environmental concern for tidal energy installations?","relevant":[{"document":"tidal-energy.md","page":null,"block":6,"quote":"change in sediment transport"}],"type":"semantic"} +{"id":"q014","question":"In lake monitoring, how is the sampling depth actually recorded?","relevant":[{"document":"lake-monitoring.md","page":null,"block":6,"quote":"Depth is recorded from the profiling sonde"}],"type":"exact"} +{"id":"q015","question":"Why does deep-water oxygen fall while a lake remains stratified?","relevant":[{"document":"lake-monitoring.md","page":null,"block":4,"quote":"Oxygen in the hypolimnion is not replenished"}],"type":"semantic"} +{"id":"q016","question":"How do supercapacitors hold their charge?","relevant":[{"document":"supercapacitors.md","page":null,"block":4,"quote":"electric double layer formed at the surface of a porous carbon electrode"}],"type":"semantic"} +{"id":"q017","question":"Why can solitary bees pollinate a bloom week that is too cold for honeybee hives?","relevant":[{"document":"wild-pollinators.md","page":null,"block":4,"quote":"forage at lower temperatures than honeybees"}],"type":"semantic"} +{"id":"q018","question":"Why is one continuous planted roof layer better than several isolated planted beds?","relevant":[{"document":"green-roofs.md","page":null,"block":4,"quote":"an exposed membrane absorbs far more heat during the day"}],"type":"semantic"} +{"id":"q019","question":"Where do acetic acid bacteria sit in a vinegar culture?","relevant":[{"document":"vinegar-production.md","page":null,"block":4,"quote":"surface of the liquid where oxygen is available"}],"type":"semantic"} +{"id":"q020","question":"What happens if a vinegar culture is sealed airtight?","relevant":[{"document":"vinegar-production.md","page":null,"block":6,"quote":"Sealing a vinegar culture airtight stops the conversion entirely"}],"type":"semantic"} +{"id":"q021","question":"Why is wave energy harder to schedule ahead than tidal energy?","relevant":[{"document":"wave-energy.md","page":null,"block":2,"quote":"driven by wind rather than by the moon"}],"type":"semantic"} +{"id":"q022","question":"Where does siting for wave energy devices concentrate, and where does it not?","relevant":[{"document":"wave-energy.md","page":null,"block":4,"quote":"exposed headlands rather than on sheltered channels"}],"type":"exact"} +{"id":"q023","question":"How does the river sampling protocol differ from the lake sampling protocol?","relevant":[{"document":"river-monitoring.md","page":null,"block":6,"quote":"three replicate samples at each station"},{"document":"lake-monitoring.md","page":null,"block":6,"quote":"Depth is recorded from the profiling sonde"}],"type":"multi-hop"} +{"id":"q024","question":"A street canopy and a green roof are both said to cool; what surface does each one shade?","relevant":[{"document":"urban-canopy.md","page":null,"block":4,"quote":"asphalt stores far more heat during the day"},{"document":"green-roofs.md","page":null,"block":4,"quote":"an exposed membrane absorbs far more heat during the day"}],"type":"multi-hop"} +{"id":"q025","question":"Which preservation method depends on keeping air away from the food?","relevant":[{"document":"fermentation.md","page":null,"block":2,"quote":"an anaerobic environment"}],"type":"semantic"} +{"id":"q026","question":"Why can one cold morning cost a grower the whole crop even when colonies are brought in?","relevant":[{"document":"orchard-pollination.md","page":null,"block":4,"quote":"above roughly twelve degrees Celsius"}],"type":"semantic"} +{"id":"q027","question":"绿茶应该怎样保存才能减缓氧化?","relevant":[{"document":"chinese-tea-storage.md","page":null,"block":2,"quote":"绿茶最容易氧化变质,需要低温密封保存"}],"type":"zh"} +{"id":"q028","question":"茶叶储存的相对湿度上限是多少?","relevant":[{"document":"chinese-tea-storage.md","page":null,"block":4,"quote":"相对湿度应保持在百分之五十以下"}],"type":"zh"} +{"id":"q029","question":"为什么冷冻保存的茶叶取出后不能立刻打开包装?","relevant":[{"document":"chinese-tea-storage.md","page":null,"block":6,"quote":"冷凝水会直接落在茶叶上"}],"type":"zh"} +{"id":"q030","question":"为什么潮汐能比风能和太阳能更容易提前安排发电?","relevant":[{"document":"tidal-energy.md","page":null,"block":2,"quote":"predictable decades ahead, unlike wind or solar"}],"type":"cross-lingual"} diff --git a/src/main/eval/harness.ts b/src/main/eval/harness.ts index b23775c..02e5bb4 100644 --- a/src/main/eval/harness.ts +++ b/src/main/eval/harness.ts @@ -21,8 +21,10 @@ import { DEFAULT_CHUNK_OPTIONS, type ChunkOptions } from '../services/ChunkingSe import type { RetrievalStrategy } from '../services/retrieval' import { LOCAL_EMBEDDING_MODEL } from '../embedding/localModel' import { + averagePrecisionAtK, evidencePrecisionAtK, firstRelevantRank, + hitRateAtK, mean, ndcgAtK, percentile, @@ -31,9 +33,11 @@ import { } from './metrics' import type { EvalDeterministicReport, + EvalMetrics, EvalQuestion, EvalReport, EvalRelevantLocation, + EvalTypeBreakdown, QuestionReport, ResolvedGroundTruth } from './types' @@ -69,6 +73,30 @@ export interface EvalHarnessOptions { const NOTEBOOK_ID = 'eval-notebook' +/** A question with no `type` is grouped here rather than dropped from the report. */ +const UNTAGGED = 'untagged' + +/** + * 同一套指标既算总平均,也算每个查询类别(#192)。用一个函数是因为分组平均必须与 + * 总平均是同一个定义,否则两个数就不可比。 + */ +function summarize(perQuestion: readonly QuestionReport[], evidenceK: number): EvalMetrics { + return { + recallAt1: mean(perQuestion.map((q) => recallAtK(q.matchesByRank, q.relevantCount, 1))), + recallAt5: mean(perQuestion.map((q) => recallAtK(q.matchesByRank, q.relevantCount, 5))), + recallAt10: mean(perQuestion.map((q) => recallAtK(q.matchesByRank, q.relevantCount, 10))), + mrr: mean(perQuestion.map((q) => reciprocalRank(q.matchesByRank))), + ndcgAt10: mean(perQuestion.map((q) => ndcgAtK(q.matchesByRank, q.relevantCount, 10))), + hitRateAt5: mean(perQuestion.map((q) => hitRateAtK(q.matchesByRank, 5))), + mapAt10: mean( + perQuestion.map((q) => averagePrecisionAtK(q.matchesByRank, q.relevantCount, 10)) + ), + evidencePrecisionAt5: mean( + perQuestion.map((q) => evidencePrecisionAtK(q.matchesByRank, evidenceK)) + ) + } +} + /** Normalised comparison for the optional quote drift check. */ const normalize = (text: string): string => text.toLowerCase().replace(/\s+/g, ' ').trim() @@ -222,6 +250,7 @@ export async function runEvalHarness( perQuestion.push({ id: question.id, question: question.question, + type: question.type ?? UNTAGGED, firstRelevantRank: firstRelevantRank(matchesByRank), relevantCount: groundTruth.length, retrievedCount: results.length, @@ -229,16 +258,15 @@ export async function runEvalHarness( }) } - const metrics = { - recallAt1: mean(perQuestion.map((q) => recallAtK(q.matchesByRank, q.relevantCount, 1))), - recallAt5: mean(perQuestion.map((q) => recallAtK(q.matchesByRank, q.relevantCount, 5))), - recallAt10: mean(perQuestion.map((q) => recallAtK(q.matchesByRank, q.relevantCount, 10))), - mrr: mean(perQuestion.map((q) => reciprocalRank(q.matchesByRank))), - ndcgAt10: mean(perQuestion.map((q) => ndcgAtK(q.matchesByRank, q.relevantCount, 10))), - evidencePrecisionAt5: mean( - perQuestion.map((q) => evidencePrecisionAtK(q.matchesByRank, options.evidenceK)) - ) - } + const metrics = summarize(perQuestion, options.evidenceK) + + // 每个类别一行,按类别名排序,所以同一个 JSON 在两次运行之间可 diff。 + const byType: EvalTypeBreakdown[] = [...new Set(perQuestion.map((q) => q.type))] + .sort() + .map((type) => { + const group = perQuestion.filter((q) => q.type === type) + return { type, questions: group.length, metrics: summarize(group, options.evidenceK) } + }) const chunking = { ...DEFAULT_CHUNK_OPTIONS, ...options.chunkOptions } return { @@ -264,6 +292,7 @@ export async function runEvalHarness( chunkCount }, metrics, + byType, timing: { latencyP50Ms: percentile(latencies, 50), latencyP95Ms: percentile(latencies, 95), @@ -274,21 +303,29 @@ export async function runEvalHarness( } /** Round metrics to a stable number of decimals so the JSON diffs cleanly. */ +const roundMetric = (value: number): number => Number(value.toFixed(6)) + +function roundMetrics(metrics: EvalMetrics): EvalMetrics { + return { + recallAt1: roundMetric(metrics.recallAt1), + recallAt5: roundMetric(metrics.recallAt5), + recallAt10: roundMetric(metrics.recallAt10), + mrr: roundMetric(metrics.mrr), + ndcgAt10: roundMetric(metrics.ndcgAt10), + hitRateAt5: roundMetric(metrics.hitRateAt5), + mapAt10: roundMetric(metrics.mapAt10), + evidencePrecisionAt5: roundMetric(metrics.evidencePrecisionAt5) + } +} + export function stabilize(report: EvalReport): EvalReport { - const round = (value: number): number => Number(value.toFixed(6)) return { ...report, - metrics: { - recallAt1: round(report.metrics.recallAt1), - recallAt5: round(report.metrics.recallAt5), - recallAt10: round(report.metrics.recallAt10), - mrr: round(report.metrics.mrr), - ndcgAt10: round(report.metrics.ndcgAt10), - evidencePrecisionAt5: round(report.metrics.evidencePrecisionAt5) - }, + metrics: roundMetrics(report.metrics), + byType: report.byType.map((entry) => ({ ...entry, metrics: roundMetrics(entry.metrics) })), timing: { - latencyP50Ms: round(report.timing.latencyP50Ms), - latencyP95Ms: round(report.timing.latencyP95Ms), + latencyP50Ms: roundMetric(report.timing.latencyP50Ms), + latencyP95Ms: roundMetric(report.timing.latencyP95Ms), // Throughput is informational and excluded from the deterministic report; the // full report keeps it for the #78 comparison. indexingMs: Math.round(report.timing.indexingMs) @@ -304,6 +341,7 @@ export function toDeterministicReport(report: EvalReport): EvalDeterministicRepo generatedBy: report.generatedBy, config: report.config, metrics: report.metrics, + byType: report.byType, perQuestion: report.perQuestion } } diff --git a/src/main/eval/metrics.ts b/src/main/eval/metrics.ts index 41f0d23..2dfda96 100644 --- a/src/main/eval/metrics.ts +++ b/src/main/eval/metrics.ts @@ -32,6 +32,56 @@ export function reciprocalRank(matchesByRank: MatchMatrix): number { return rank === 0 ? 0 : 1 / rank } +/** + * Hit rate@k: 1 when any of the first `k` ranks covers ground truth, else 0. + * + * Deliberately the bluntest metric here. Recall@k says *how much* of the ground + * truth was found; hit rate says only whether the answer was findable at all. When + * a question needs two passages, a run that finds one scores 0.5 recall and 1.0 hit + * rate — and the second number is the one that says "the model had a chance". + */ +export function hitRateAtK(matchesByRank: MatchMatrix, k: number): number { + return matchesByRank.slice(0, k).some((matches) => matches.length > 0) ? 1 : 0 +} + +/** + * Average precision@k, averaged over the ground-truth locations. + * + * Precision is measured at each rank that recovers a *fresh* ground-truth location + * (the same rule nDCG uses), then normalised by the total number of ground-truth + * locations. That makes AP the one metric here that combines ranking position with + * coverage: pulling a second relevant passage from rank 9 to rank 2 moves it, while + * recall@5 sits still. + */ +export function averagePrecisionAtK( + matchesByRank: MatchMatrix, + groundTruthCount: number, + k: number +): number { + if (groundTruthCount === 0) return 0 + + const covered = new Set() + let found = 0 + let sum = 0 + const limit = Math.min(matchesByRank.length, k) + + for (let i = 0; i < limit; i++) { + let fresh = false + for (const match of matchesByRank[i]) { + if (match < groundTruthCount && !covered.has(match)) { + covered.add(match) + fresh = true + } + } + if (fresh) { + found += 1 + sum += found / (i + 1) + } + } + + return sum / groundTruthCount +} + /** * nDCG@k with binary gains. The ideal ranking puts every ground-truth location * first, so the discount is a plain log base 2. diff --git a/src/main/eval/report.ts b/src/main/eval/report.ts index b596704..34d9688 100644 --- a/src/main/eval/report.ts +++ b/src/main/eval/report.ts @@ -13,6 +13,17 @@ export function renderMarkdown(report: EvalReport): string { const { config, metrics, timing } = report const chunking = config.chunking + // Built before the template because a nested template literal would terminate the + // outer one; the map is a statement here, not an interpolation. + const typeRows = report.byType + .map( + (entry) => + `| ${entry.type} | ${entry.questions} | ${format(entry.metrics.recallAt5)} | ` + + `${format(entry.metrics.ndcgAt10)} | ${format(entry.metrics.hitRateAt5)} | ` + + `${format(entry.metrics.mapAt10)} |` + ) + .join('\n') + return `# RAG eval baseline — ${report.baseline} Generated by \`${report.generatedBy}\`. The numbers below are harness output — do not edit them by hand. @@ -39,8 +50,20 @@ Generated by \`${report.generatedBy}\`. The numbers below are harness output — | Recall@10 | ${format(metrics.recallAt10)} | | MRR | ${format(metrics.mrr)} | | nDCG@10 | ${format(metrics.ndcgAt10)} | +| Hit rate@5 | ${format(metrics.hitRateAt5)} | +| MAP@10 | ${format(metrics.mapAt10)} | | Evidence precision@${config.evidenceK} | ${format(metrics.evidencePrecisionAt5)} | +### By query type + +A single average hides a change that helps one kind of question and hurts another. +The type comes from \`type\` in \`questions.jsonl\`; untagged questions report as +\`untagged\` rather than disappearing. + +| Type | Questions | Recall@5 | nDCG@10 | Hit rate@5 | MAP@10 | +| --- | --- | --- | --- | --- | --- | +${typeRows} + Timing is informational only and is **not** frozen: indexing ${timing.indexingMs} ms, query p50 ${timing.latencyP50Ms.toFixed(2)} ms, p95 ${timing.latencyP95Ms.toFixed(2)} ms on the machine that produced this file. Timing and index size depend on hardware and on the diff --git a/src/main/eval/types.ts b/src/main/eval/types.ts index c77fb90..dbb692b 100644 --- a/src/main/eval/types.ts +++ b/src/main/eval/types.ts @@ -24,6 +24,14 @@ export interface EvalQuestion { question: string relevant: EvalRelevantLocation[] goldAnswer?: string + /** + * 查询类别(#192)。自由字符串,因为语料还会长出新类别;报告按出现过的值分组, + * 缺省归入 `untagged`。 + * + * 分类的意义是:一个提升实体查询、却弄坏释义查询的策略,不该在总平均上显示成 + * 「没变化」。 + */ + type?: string } /** One resolved ground-truth location, after runtime id mapping. */ @@ -41,6 +49,15 @@ export interface EvalMetrics { recallAt10: number mrr: number ndcgAt10: number + /** + * 前 5 名至少命中一个 ground-truth 块的问题占比。 + * + * 与 Recall@5 并列而不是替代:多块问题只要命中一块,hit rate 就是 1, + * 而 Recall@5 只有 0.5。「模型有没有机会」和「材料齐不齐」是两件事。 + */ + hitRateAt5: number + /** AP@10:把「排序位置」和「覆盖面」合成一个数的那个指标。 */ + mapAt10: number /** * Share of the first `evidenceK` retrieved passages that cover ground truth. * @@ -51,9 +68,18 @@ export interface EvalMetrics { evidencePrecisionAt5: number } +/** 按查询类别聚合的同一套指标(#192)。总平均会掩盖方向相反的两个变化。 */ +export interface EvalTypeBreakdown { + type: string + questions: number + metrics: EvalMetrics +} + export interface QuestionReport { id: string question: string + /** 查询类别,与 `EvalQuestion.type` 一致;缺省为 `untagged`。 */ + type: string firstRelevantRank: number relevantCount: number retrievedCount: number @@ -97,6 +123,8 @@ export interface EvalReport { chunkCount: number } metrics: EvalMetrics + /** 每个查询类别一行;类别来自 `questions.jsonl` 的 `type`。 */ + byType: EvalTypeBreakdown[] /** `indexingMs` 只用于 #78 的吞吐比较;它不在确定报告里,也不该成为差异原因。 */ timing: { latencyP50Ms: number; latencyP95Ms: number; indexingMs: number } perQuestion: QuestionReport[] @@ -112,5 +140,6 @@ export interface EvalDeterministicReport { generatedBy: string config: EvalReport['config'] metrics: EvalMetrics + byType: EvalTypeBreakdown[] perQuestion: QuestionReport[] } diff --git a/test/evalMetrics.test.ts b/test/evalMetrics.test.ts index 1520dc6..3cd8d0e 100644 --- a/test/evalMetrics.test.ts +++ b/test/evalMetrics.test.ts @@ -1,8 +1,10 @@ import { test } from 'node:test' import assert from 'node:assert/strict' import { + averagePrecisionAtK, evidencePrecisionAtK, firstRelevantRank, + hitRateAtK, mean, type MatchMatrix, ndcgAtK, @@ -101,6 +103,42 @@ test('evidence precision counts grounded passages over retrieved passages', () = assert.equal(evidencePrecisionAtK([[0], [0], [0]], 3), 1) }) +test('hit rate@k says whether the answer was reachable at all', () => { + assert.equal(hitRateAtK([[0], [], []], 5), 1) + assert.equal(hitRateAtK([[], [], [1]], 1), 0) + assert.equal(hitRateAtK([[], [], [1]], 3), 1) + assert.equal(hitRateAtK([[], []], 5), 0) + assert.equal(hitRateAtK([], 5), 0) +}) + +/** + * The distinction the metric exists for: a two-passage question that finds only + * one has 0.5 recall but full hit rate — the model had a chance, and recall is what + * says the material was incomplete. + */ +test('hit rate can be 1 while recall@k is only half', () => { + const matches: MatchMatrix = [[0], []] + assert.equal(hitRateAtK(matches, 5), 1) + assert.equal(recallAtK(matches, 2, 5), 0.5) +}) + +test('average precision rewards finding the same ground truth earlier', () => { + // One relevant at rank 1: AP = 1. + assert.equal(averagePrecisionAtK([[0]], 1, 10), 1) + // One relevant at rank 2: P@1 = 0, P@2 = 1/2, so AP = 1/2. + assert.equal(averagePrecisionAtK([[], [0]], 1, 10), 0.5) + // Two relevant at ranks 1 and 2: (1 + 2/2) / 2 = 1. + assert.equal(averagePrecisionAtK([[0], [1]], 2, 10), 1) + // Two relevant, but the second is only found at rank 4: (1 + 2/4) / 2 = 0.75. + assert.equal(averagePrecisionAtK([[0], [], [], [1]], 2, 10), 0.75) +}) + +test('average precision counts a repeated match once and never exceeds 1', () => { + assert.equal(averagePrecisionAtK([[0], [0], [0]], 1, 10), 1) + assert.equal(averagePrecisionAtK([[], []], 0, 10), 0) + assert.equal(averagePrecisionAtK([], 3, 10), 0) +}) + test('mean and percentile handle the empty and single cases', () => { assert.equal(mean([]), 0) assert.equal(mean([1, 2, 3]), 2)