From 8a99107bf408ec1f83126bbf6ee3b42f46f2dd4e Mon Sep 17 00:00:00 2001 From: MrSibe Date: Wed, 30 Sep 2026 18:16:49 +0800 Subject: [PATCH] feat(eval): per-question paired deltas, and hybrid's gain is five questions MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `npm run eval:paired` reports the dense ↔ hybrid pair per question and the classification per query type. It answers the one thing the aggregate comparison could not. The open question was: hybrid is ahead on the full set and level on `validation`, and an average cannot say whether that is a broad small gain or a handful of rescued cases. Those two readings imply different next steps. Child 2 of #192. ## What it found ``` validation 19 questions improved 2 / tied 16 / regressed 1 mean ΔnDCG@10 +0.0047 test 34 questions improved 5 / tied 29 / regressed 0 mean ΔnDCG@10 +0.0432 ``` **The gain is not broad — it is five questions on `test` and two on `validation`.** On the reporting side, 29 of 34 questions are untouched. The mean ΔnDCG@10 of +0.0432 is carried by `q013` (rank 11 → 4, ΔnDCG +0.43), `q059` (5 → 2), `q026` (2 → 1), `q049` (3 → 2) and `q052` (2 → 1). ## And it is not the story we had started telling ourselves The plausible story was "hybrid helps the hard negatives". The breakdown says otherwise: | type | n (test) | improved | tied | regressed | | --- | --- | --- | --- | --- | | cross-lingual | 5 | **0** | 5 | 0 | | hard-negative | 5 | **1** | 4 | 0 | | multi-hop | 3 | 1 | 2 | 0 | | semantic | 15 | **3** | 12 | 0 | | exact | 4 | 0 | 4 | 0 | Three of the five wins are `semantic`, one is `hard-negative` and one `multi-hop`. With n=1 of 5, "hybrid helps hard-negative questions" is **not** supported at this size — which is exactly the pattern that would have been asserted from the aggregate number alone. **`cross-lingual` is untouched on both splits: 0 improved, 0 regressed.** Adding a sparse channel does nothing for the category that is measurably weakest, which is itself a useful negative result and a reason not to reach for hybrid as the answer to it. No question was found by one strategy and missed by the other, on either split, so no regression hides behind the `0 = not found` sentinel. ## How Metrics come from `src/main/eval/metrics.ts` — imported, not reimplemented — so a delta cannot disagree with the metric it is a delta of. `firstRelevantRank` is `0` for "never found", so the miss cases are classified explicitly rather than folded into an arithmetic delta that cannot express them. `validation` is the selecting side; the `test` breakdown **explains** the observed difference and is labelled as not-for-selection in the report itself. Nothing here changes the shipped strategy. ## Testing - `npm run typecheck` — clean - `npm run eval:paired` — 4 harness runs, writes `docs/eval/paired-v1.6.{json,md}` - The committed baseline is unchanged Part of #192 (child 2: corpus diagnostics). --- .prettierignore | 4 + docs/eval/paired-v1.6.json | 1254 ++++++++++++++++++++++++++++++++++++ docs/eval/paired-v1.6.md | 97 +++ eval/README.md | 1 + package.json | 3 +- scripts/eval-paired.mjs | 368 +++++++++++ 6 files changed, 1726 insertions(+), 1 deletion(-) create mode 100644 docs/eval/paired-v1.6.json create mode 100644 docs/eval/paired-v1.6.md create mode 100644 scripts/eval-paired.mjs diff --git a/.prettierignore b/.prettierignore index da32ca2..f418484 100644 --- a/.prettierignore +++ b/.prettierignore @@ -33,6 +33,10 @@ docs/eval/sweep-*.md docs/eval/scores-*.json docs/eval/scores-*.md +# And by `scripts/eval-paired.mjs`. Regenerate with `npm run eval:paired`. +docs/eval/paired-*.json +docs/eval/paired-*.md + # And the same again for `scripts/eval-retrieval.mjs` (#77). docs/eval/retrieval-*.json docs/eval/retrieval-*.md diff --git a/docs/eval/paired-v1.6.json b/docs/eval/paired-v1.6.json new file mode 100644 index 0000000..dd1bfc2 --- /dev/null +++ b/docs/eval/paired-v1.6.json @@ -0,0 +1,1254 @@ +{ + "baseline": "v1.6", + "strategies": [ + "dense", + "hybrid" + ], + "splits": { + "validation": { + "questions": 19, + "meanDelta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0.004654087156011809 + }, + "byType": [ + { + "type": "cross-lingual", + "questions": 4, + "improved": 0, + "tied": 4, + "regressed": 0, + "meanDeltaNdcg10": 0, + "meanDeltaRank": 0 + }, + { + "type": "exact", + "questions": 4, + "improved": 2, + "tied": 2, + "regressed": 0, + "meanDeltaNdcg10": 0.05006329887451612, + "meanDeltaRank": 0.5 + }, + { + "type": "hard-negative", + "questions": 2, + "improved": 0, + "tied": 2, + "regressed": 0, + "meanDeltaNdcg10": 0, + "meanDeltaRank": 0 + }, + { + "type": "multi-hop", + "questions": 2, + "improved": 0, + "tied": 2, + "regressed": 0, + "meanDeltaNdcg10": 0.08861964339202388, + "meanDeltaRank": 0 + }, + { + "type": "semantic", + "questions": 6, + "improved": 0, + "tied": 5, + "regressed": 1, + "meanDeltaNdcg10": -0.04817747105298131, + "meanDeltaRank": -0.3333333333333333 + }, + { + "type": "zh", + "questions": 1, + "improved": 0, + "tied": 1, + "regressed": 0, + "meanDeltaNdcg10": 0, + "meanDeltaRank": 0 + } + ], + "pairs": [ + { + "id": "q002", + "type": "exact", + "question": "How many replicate samples are collected at each river station?", + "dense": { + "firstRelevantRank": 3, + "recallAt5": 1, + "ndcgAt10": 0.5 + }, + "hybrid": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "delta": { + "rank": 1, + "recallAt5": 0, + "ndcgAt10": 0.13092975357145753 + }, + "classification": "improved" + }, + { + "id": "q004", + "type": "semantic", + "question": "Why do nickel-rich battery packs need more aggressive thermal management?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q009", + "type": "semantic", + "question": "Why are trees with aggressive surface roots unsuitable for narrow verges?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q011", + "type": "exact", + "question": "Is the salt percentage in fermentation based on vegetable weight or water weight?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q015", + "type": "semantic", + "question": "Why does deep-water oxygen fall while a lake remains stratified?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q018", + "type": "semantic", + "question": "Why is one continuous planted roof layer better than several isolated planted beds?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q021", + "type": "semantic", + "question": "Why is wave energy harder to schedule ahead than tidal energy?", + "dense": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "hybrid": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q023", + "type": "multi-hop", + "question": "How does the river sampling protocol differ from the lake sampling protocol?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 0.5, + "ndcgAt10": 0.6131471927654584 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 0.5, + "ndcgAt10": 0.7903864795495061 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0.17723928678404777 + }, + "classification": "tied" + }, + { + "id": "q028", + "type": "zh", + "question": "茶叶储存的相对湿度上限是多少?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q030", + "type": "cross-lingual", + "question": "为什么潮汐能比风能和太阳能更容易提前安排发电?", + "dense": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "hybrid": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q037", + "type": "cross-lingual", + "question": "推移质为什么比悬移质更难测?", + "dense": { + "firstRelevantRank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "hybrid": { + "firstRelevantRank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "delta": { + "rank": null, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q040", + "type": "cross-lingual", + "question": "富镍电池为什么对散热要求更高?", + "dense": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "hybrid": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q043", + "type": "cross-lingual", + "question": "花期遭遇晚霜,主要受损的是什么?", + "dense": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "hybrid": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q045", + "type": "exact", + "question": "What is the default retry count in Gateway API v2?", + "dense": { + "firstRelevantRank": 4, + "recallAt5": 1, + "ndcgAt10": 0.43067655807339306 + }, + "hybrid": { + "firstRelevantRank": 3, + "recallAt5": 1, + "ndcgAt10": 0.5 + }, + "delta": { + "rank": 1, + "recallAt5": 0, + "ndcgAt10": 0.06932344192660694 + }, + "classification": "improved" + }, + { + "id": "q047", + "type": "hard-negative", + "question": "What is the context window of Gateway API v3, in tokens?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q050", + "type": "semantic", + "question": "Which Gateway API version waits longest between retries after a failed request?", + "dense": { + "firstRelevantRank": 10, + "recallAt5": 0, + "ndcgAt10": 0.2890648263178879 + }, + "hybrid": { + "firstRelevantRank": 12, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "delta": { + "rank": -2, + "recallAt5": 0, + "ndcgAt10": -0.2890648263178879 + }, + "classification": "regressed" + }, + { + "id": "q055", + "type": "exact", + "question": "How many replicate samples are collected at each reservoir monitoring station?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q057", + "type": "hard-negative", + "question": "At what depth below the surface does the coastal programme place its shore-station sensor?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q060", + "type": "multi-hop", + "question": "Compare the sampling interval and replicate count of the reservoir and groundwater programmes.", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + } + ] + }, + "test": { + "questions": 34, + "meanDelta": { + "rank": 0.3939393939393939, + "recallAt5": 0.029411764705882353, + "ndcgAt10": 0.04318215877040841 + }, + "byType": [ + { + "type": "cross-lingual", + "questions": 5, + "improved": 0, + "tied": 5, + "regressed": 0, + "meanDeltaNdcg10": 0, + "meanDeltaRank": 0 + }, + { + "type": "exact", + "questions": 4, + "improved": 0, + "tied": 4, + "regressed": 0, + "meanDeltaNdcg10": 0, + "meanDeltaRank": 0 + }, + { + "type": "hard-negative", + "questions": 5, + "improved": 1, + "tied": 4, + "regressed": 0, + "meanDeltaNdcg10": 0.026185950714291507, + "meanDeltaRank": 0.2 + }, + { + "type": "multi-hop", + "questions": 3, + "improved": 1, + "tied": 2, + "regressed": 0, + "meanDeltaNdcg10": 0.09781329792785898, + "meanDeltaRank": 0.3333333333333333 + }, + { + "type": "semantic", + "questions": 15, + "improved": 3, + "tied": 12, + "regressed": 0, + "meanDeltaNdcg10": 0.06958825005592344, + "meanDeltaRank": 0.7333333333333333 + }, + { + "type": "zh", + "questions": 2, + "improved": 0, + "tied": 2, + "regressed": 0, + "meanDeltaNdcg10": 0, + "meanDeltaRank": 0 + } + ], + "pairs": [ + { + "id": "q001", + "type": "semantic", + "question": "Why is bedload harder to measure than suspended sediment?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q003", + "type": "semantic", + "question": "What is the central trade-off in lithium-ion cell design?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q005", + "type": "semantic", + "question": "What happens once the separator in a battery cell melts?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q006", + "type": "exact", + "question": "At what temperature do honeybees begin to forage?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q007", + "type": "semantic", + "question": "What does a late frost damage during full bloom?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q008", + "type": "semantic", + "question": "Why is a continuous tree canopy more effective at cooling than isolated trees?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q010", + "type": "exact", + "question": "At what temperature is lactic acid fermentation fastest?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q012", + "type": "semantic", + "question": "Why must tidal turbines be sited in places with very fast currents?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q013", + "type": "semantic", + "question": "What is the main environmental concern for tidal energy installations?", + "dense": { + "firstRelevantRank": 11, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "hybrid": { + "firstRelevantRank": 4, + "recallAt5": 1, + "ndcgAt10": 0.43067655807339306 + }, + "delta": { + "rank": 7, + "recallAt5": 1, + "ndcgAt10": 0.43067655807339306 + }, + "classification": "improved" + }, + { + "id": "q014", + "type": "exact", + "question": "In lake monitoring, how is the sampling depth actually recorded?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q016", + "type": "semantic", + "question": "How do supercapacitors hold their charge?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q017", + "type": "semantic", + "question": "Why can solitary bees pollinate a bloom week that is too cold for honeybee hives?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q019", + "type": "semantic", + "question": "Where do acetic acid bacteria sit in a vinegar culture?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q020", + "type": "semantic", + "question": "What happens if a vinegar culture is sealed airtight?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q022", + "type": "exact", + "question": "Where does siting for wave energy devices concentrate, and where does it not?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q024", + "type": "multi-hop", + "question": "A street canopy and a green roof are both said to cool; what surface does each one shade?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q025", + "type": "semantic", + "question": "Which preservation method depends on keeping air away from the food?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q026", + "type": "semantic", + "question": "Why can one cold morning cost a grower the whole crop even when colonies are brought in?", + "dense": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 1, + "recallAt5": 0, + "ndcgAt10": 0.36907024642854247 + }, + "classification": "improved" + }, + { + "id": "q027", + "type": "zh", + "question": "绿茶应该怎样保存才能减缓氧化?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q029", + "type": "zh", + "question": "为什么冷冻保存的茶叶取出后不能立刻打开包装?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q038", + "type": "cross-lingual", + "question": "河流监测中,每个站点要采几份平行样?", + "dense": { + "firstRelevantRank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "hybrid": { + "firstRelevantRank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "delta": { + "rank": null, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q039", + "type": "cross-lingual", + "question": "设计锂离子电芯时最核心的取舍是什么?", + "dense": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "hybrid": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q041", + "type": "cross-lingual", + "question": "电芯隔膜一旦熔化会导致什么后果?", + "dense": { + "firstRelevantRank": 4, + "recallAt5": 1, + "ndcgAt10": 0.43067655807339306 + }, + "hybrid": { + "firstRelevantRank": 4, + "recallAt5": 1, + "ndcgAt10": 0.43067655807339306 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q042", + "type": "cross-lingual", + "question": "蜜蜂大致从什么温度开始出巢觅食?", + "dense": { + "firstRelevantRank": 3, + "recallAt5": 1, + "ndcgAt10": 0.5 + }, + "hybrid": { + "firstRelevantRank": 3, + "recallAt5": 1, + "ndcgAt10": 0.5 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q044", + "type": "cross-lingual", + "question": "为什么成片的树冠比孤立的树降温效果更好?", + "dense": { + "firstRelevantRank": 9, + "recallAt5": 0, + "ndcgAt10": 0.3010299956639812 + }, + "hybrid": { + "firstRelevantRank": 9, + "recallAt5": 0, + "ndcgAt10": 0.3010299956639812 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q046", + "type": "hard-negative", + "question": "What is the default request timeout of Gateway API v1?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q048", + "type": "hard-negative", + "question": "What is the default rate limit of Gateway API v2?", + "dense": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "hybrid": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q049", + "type": "hard-negative", + "question": "What is the default request timeout of Gateway API v3?", + "dense": { + "firstRelevantRank": 3, + "recallAt5": 1, + "ndcgAt10": 0.5 + }, + "hybrid": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "delta": { + "rank": 1, + "recallAt5": 0, + "ndcgAt10": 0.13092975357145753 + }, + "classification": "improved" + }, + { + "id": "q051", + "type": "multi-hop", + "question": "Compare the default timeouts and rate limits of Gateway API v1 and v3.", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 0.75, + "ndcgAt10": 0.6366824387328317 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 0.75, + "ndcgAt10": 0.6366824387328317 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q052", + "type": "multi-hop", + "question": "Compare the retry count and context window of Gateway API v2 and v3.", + "dense": { + "firstRelevantRank": 2, + "recallAt5": 0.25, + "ndcgAt10": 0.35914753008966527 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 0.25, + "ndcgAt10": 0.6525874238732422 + }, + "delta": { + "rank": 1, + "recallAt5": 0, + "ndcgAt10": 0.29343989378357693 + }, + "classification": "improved" + }, + { + "id": "q056", + "type": "hard-negative", + "question": "How many replicate samples are collected at each estuary transect station?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q058", + "type": "hard-negative", + "question": "At what depth below the water table do groundwater boreholes carry their pressure transducer?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q059", + "type": "semantic", + "question": "Which monitoring programme samples most frequently?", + "dense": { + "firstRelevantRank": 5, + "recallAt5": 1, + "ndcgAt10": 0.38685280723454163 + }, + "hybrid": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "delta": { + "rank": 3, + "recallAt5": 0, + "ndcgAt10": 0.2440769463369159 + }, + "classification": "improved" + }, + { + "id": "q061", + "type": "semantic", + "question": "Why does the estuary programme take more replicate samples than the other monitoring programmes?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + } + ] + } + } +} diff --git a/docs/eval/paired-v1.6.md b/docs/eval/paired-v1.6.md new file mode 100644 index 0000000..dc185dd --- /dev/null +++ b/docs/eval/paired-v1.6.md @@ -0,0 +1,97 @@ +# Paired dense ↔ hybrid deltas — v1.6 (#192) + +Generated by `node scripts/eval-paired.mjs`. Numbers are harness output; do not edit them by hand. + +## What this answers + +The aggregate comparison says hybrid is ahead on the full set and level on `validation`. +An average cannot say whether that is a broad small gain or a few rescued cases, and those +two readings imply different next steps. This is the per-question pair. + +Metrics come from `src/main/eval/metrics.ts`, so a delta cannot disagree with the metric it +is a delta of. Unanswerable questions are excluded — they have no rank to compare, and the +abstention diagnostics cover them. + +`validation` is the side that **selects**; the `test` breakdown below **explains** the +difference that was observed and must not be used to choose. Nothing here changes the +shipped strategy. + +## Validation (this side selects) + +Chosen on this side, so a win here is eligible to inform a decision. + +Questions compared: 19. Mean ΔnDCG@10 +0.0047, +mean Δrank 0.00. + +### By query type + +| Type | n | improved | tied | regressed | mean ΔnDCG@10 | mean Δrank | +| --- | --- | --- | --- | --- | --- | --- | +| cross-lingual | 4 | 0 | 4 | 0 | 0.0000 | 0.00 | +| exact | 4 | 2 | 2 | 0 | +0.0501 | +0.50 | +| hard-negative | 2 | 0 | 2 | 0 | 0.0000 | 0.00 | +| multi-hop | 2 | 0 | 2 | 0 | +0.0886 | 0.00 | +| semantic | 6 | 0 | 5 | 1 | -0.0482 | -0.33 | +| zh | 1 | 0 | 1 | 0 | 0.0000 | 0.00 | + +### Biggest rank wins + +| id | type | dense rank | hybrid rank | Δrank | ΔnDCG@10 | +| --- | --- | --- | --- | --- | --- | +| q002 | exact | 3 | 2 | +1 | +0.1309 | +| q045 | exact | 4 | 3 | +1 | +0.0693 | + +### Biggest rank regressions + +| id | type | dense rank | hybrid rank | Δrank | ΔnDCG@10 | +| --- | --- | --- | --- | --- | --- | +| q050 | semantic | 10 | 12 | -2 | -0.2891 | + +**Found by hybrid, missed by dense**: none. + +**Lost by hybrid, found by dense**: none. + +## Test (explanatory only) + +Not used to choose anything. Present because the aggregate difference this is meant to explain was measured here. + +Questions compared: 34. Mean ΔnDCG@10 +0.0432, +mean Δrank +0.39. + +### By query type + +| Type | n | improved | tied | regressed | mean ΔnDCG@10 | mean Δrank | +| --- | --- | --- | --- | --- | --- | --- | +| cross-lingual | 5 | 0 | 5 | 0 | 0.0000 | 0.00 | +| exact | 4 | 0 | 4 | 0 | 0.0000 | 0.00 | +| hard-negative | 5 | 1 | 4 | 0 | +0.0262 | +0.20 | +| multi-hop | 3 | 1 | 2 | 0 | +0.0978 | +0.33 | +| semantic | 15 | 3 | 12 | 0 | +0.0696 | +0.73 | +| zh | 2 | 0 | 2 | 0 | 0.0000 | 0.00 | + +### Biggest rank wins + +| id | type | dense rank | hybrid rank | Δrank | ΔnDCG@10 | +| --- | --- | --- | --- | --- | --- | +| q013 | semantic | 11 | 4 | +7 | +0.4307 | +| q059 | semantic | 5 | 2 | +3 | +0.2441 | +| q026 | semantic | 2 | 1 | +1 | +0.3691 | +| q049 | hard-negative | 3 | 2 | +1 | +0.1309 | +| q052 | multi-hop | 2 | 1 | +1 | +0.2934 | + +### Biggest rank regressions + +| id | type | dense rank | hybrid rank | Δrank | ΔnDCG@10 | +| --- | --- | --- | --- | --- | --- | +| — | — | — | — | + +**Found by hybrid, missed by dense**: none. + +**Lost by hybrid, found by dense**: none. + +## Reproduce + +```bash +npm run eval:prepare # one-time, networked model bootstrap +npm run eval:paired # offline; rewrites this file +``` diff --git a/eval/README.md b/eval/README.md index 299eca7..22afea3 100644 --- a/eval/README.md +++ b/eval/README.md @@ -13,6 +13,7 @@ npm run eval:retrieval # strategy comparison (#77); validation selects, test re npm run eval:threshold # derive the similarity threshold on validation, report on test npm run eval:sweep # bounded grid over strategy × candidateK × contextK, one dashboard npm run eval:scores # dense score distribution: can one threshold separate relevant from not? +npm run eval:paired # per-question dense ↔ hybrid deltas, by query type npm run eval:blocks eval/corpus/foo.md # print the block ordinals ground truth must use ``` diff --git a/package.json b/package.json index 418c579..8f04662 100644 --- a/package.json +++ b/package.json @@ -45,7 +45,8 @@ "eval:threshold": "npm run build && node scripts/eval-threshold.mjs", "eval:sweep": "npm run build && node scripts/eval-sweep.mjs", "eval:blocks": "node --experimental-transform-types --import ./test/ts-resolve.mjs --disable-warning=ExperimentalWarning --disable-warning=MODULE_TYPELESS_PACKAGE_JSON scripts/eval-blocks.mjs", - "eval:scores": "npm run build && node scripts/eval-scores.mjs" + "eval:scores": "npm run build && node scripts/eval-scores.mjs", + "eval:paired": "npm run build && node --experimental-transform-types --import ./test/ts-resolve.mjs --disable-warning=ExperimentalWarning --disable-warning=MODULE_TYPELESS_PACKAGE_JSON scripts/eval-paired.mjs" }, "//test": [ "`node --test` strips TypeScript types rather than compiling them, and strip-only", diff --git a/scripts/eval-paired.mjs b/scripts/eval-paired.mjs new file mode 100644 index 0000000..b097947 --- /dev/null +++ b/scripts/eval-paired.mjs @@ -0,0 +1,368 @@ +#!/usr/bin/env node +/** + * Per-query paired deltas for dense ↔ hybrid (#192). + * + * The aggregate comparison left a question it cannot answer: hybrid is ahead on the full + * set and level on `validation`, and an average cannot say whether that is a broad small + * gain or a handful of cases it rescued. A mean over forty questions cannot distinguish + * "helps a little everywhere" from "helps a lot twice". + * + * So this reports the pair per question and the classification per query type: + * + * improved / tied / regressed, with the rank move and the nDCG@10 delta + * + * `validation` is the side that selects; the `test` breakdown **explains** the observed + * difference and must not be used to choose. Nothing here changes the shipped strategy. + * + * Metrics come from `src/main/eval/metrics.ts` rather than being reimplemented, so a delta + * cannot disagree with the metric it is a delta of. + * + * Usage: + * node scripts/eval-paired.mjs + */ + +import { spawn } from 'node:child_process' +import { existsSync, mkdtempSync, mkdirSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join, resolve } from 'node:path' + +const OUT_MD = resolve(readArg('--out=', 'docs/eval/paired-v1.6.md')) +const OUT_JSON = OUT_MD.replace(/\.md$/, '.json') + +const SPLITS = ['validation', 'test'] +const STRATEGIES = [ + { id: 'dense', label: 'dense (vector)' }, + { id: 'hybrid', label: 'hybrid (RRF of dense + BM25)' } +] + +/** How many rows to show in each mover table. */ +const MOVER_LIMIT = 8 + +function readArg(prefix, fallback) { + const arg = process.argv.find((value) => value.startsWith(prefix)) + return arg ? arg.slice(prefix.length) : fallback +} + +// Node 24 refuses to spawn a `.cmd`/`.bat` without `shell: true` (EINVAL), and the +// `.bin` entry is exactly that on Windows. Use the real binary the wrapper runs. +const { default: electronBinary } = await import('electron') +const executable = resolve(electronBinary) + +if (!existsSync(executable)) { + console.error('[paired] could not find the electron binary. Run `npm install` first.') + process.exit(1) +} + +const { ndcgAtK, recallAtK } = await import('../src/main/eval/metrics.ts') + +function runStrategy(strategy, split, outDir) { + return new Promise((resolvePromise, reject) => { + const args = [ + '.', + '--eval-harness', + '--eval-baseline=paired', + `--eval-out=${outDir}`, + `--eval-split=${split}`, + `--eval-retrieval=${strategy.id}` + ] + + const isRoot = typeof process.getuid === 'function' && process.getuid() === 0 + if (isRoot || process.env.CI) args.push('--no-sandbox') + + const child = spawn(executable, args, { + stdio: ['ignore', 'pipe', 'pipe'], + env: { ...process.env, ELECTRON_DISABLE_SECURITY_WARNINGS: '1' } + }) + + child.on('error', reject) + child.on('exit', (code) => { + if (code !== 0) { + reject(new Error(`${strategy.id} (${split}) exited with code ${code}`)) + return + } + const reportPath = join(outDir, 'baseline-paired.json') + if (!existsSync(reportPath)) { + reject(new Error(`${strategy.id} (${split}) wrote no report`)) + return + } + resolvePromise(JSON.parse(readFileSync(reportPath, 'utf8'))) + }) + }) +} + +/** One side's per-question view, with the metrics the comparison is made of. */ +function sideOf(question) { + return { + firstRelevantRank: question.firstRelevantRank, + recallAt5: recallAtK(question.matchesByRank, question.relevantCount, 5), + ndcgAt10: ndcgAtK(question.matchesByRank, question.relevantCount, 10) + } +} + +/** + * Classify the pair. `firstRelevantRank` is 0 for "never found", so the miss cases cannot be + * folded into an arithmetic delta: going from found to not-found is a regression no rank + * number expresses. + */ +function classify(dense, hybrid) { + const denseFound = dense.firstRelevantRank > 0 + const hybridFound = hybrid.firstRelevantRank > 0 + + if (denseFound && hybridFound) { + const rankDelta = dense.firstRelevantRank - hybrid.firstRelevantRank + return { + classification: rankDelta > 0 ? 'improved' : rankDelta < 0 ? 'regressed' : 'tied', + rankDelta + } + } + if (!denseFound && hybridFound) return { classification: 'improved', rankDelta: null } + if (denseFound && !hybridFound) return { classification: 'regressed', rankDelta: null } + return { classification: 'tied', rankDelta: null } +} + +/** Human-readable rank move that survives the 0 = "not found" sentinel. */ +const rankText = (rank) => (rank > 0 ? String(rank) : 'not found') + +const format4 = (value) => (value === null ? '—' : value.toFixed(4)) +const signed = (value, digits = 4) => + value === null ? '—' : `${value > 0 ? '+' : ''}${value.toFixed(digits)}` + +const workDir = mkdtempSync(join(tmpdir(), 'knownote-paired-')) +const bySplit = {} + +try { + for (const split of SPLITS) { + const reports = {} + for (const strategy of STRATEGIES) { + console.log(`[paired] ${split}: ${strategy.label}`) + const outDir = join(workDir, `${split}-${strategy.id}`) + mkdirSync(outDir, { recursive: true }) + reports[strategy.id] = await runStrategy(strategy, split, outDir) + } + + const denseById = new Map(reports.dense.perQuestion.map((q) => [q.id, q])) + const pairs = [] + for (const hybridQuestion of reports.hybrid.perQuestion) { + const denseQuestion = denseById.get(hybridQuestion.id) + if (!denseQuestion) continue + // Unanswerable questions have no rank to compare; they are measured by the + // abstention diagnostics, not here. + if (!denseQuestion.answerable) continue + + const dense = sideOf(denseQuestion) + const hybrid = sideOf(hybridQuestion) + const { classification, rankDelta } = classify(dense, hybrid) + pairs.push({ + id: hybridQuestion.id, + type: hybridQuestion.type, + question: hybridQuestion.question, + dense, + hybrid, + delta: { + rank: rankDelta, + recallAt5: hybrid.recallAt5 - dense.recallAt5, + ndcgAt10: hybrid.ndcgAt10 - dense.ndcgAt10 + }, + classification + }) + } + + bySplit[split] = { + questions: pairs.length, + pairs, + byType: aggregateByType(pairs), + meanDelta: { + rank: mean(pairs.map((p) => p.delta.rank).filter((value) => value !== null)), + recallAt5: mean(pairs.map((p) => p.delta.recallAt5)), + ndcgAt10: mean(pairs.map((p) => p.delta.ndcgAt10)) + } + } + } +} finally { + rmSync(workDir, { recursive: true, force: true }) +} + +function mean(values) { + if (values.length === 0) return null + return values.reduce((a, b) => a + b, 0) / values.length +} + +function aggregateByType(pairs) { + const types = [...new Set(pairs.map((p) => p.type))].sort() + return types.map((type) => { + const group = pairs.filter((p) => p.type === type) + const count = (label) => group.filter((p) => p.classification === label).length + return { + type, + questions: group.length, + improved: count('improved'), + tied: count('tied'), + regressed: count('regressed'), + meanDeltaNdcg10: mean(group.map((p) => p.delta.ndcgAt10)), + meanDeltaRank: mean(group.map((p) => p.delta.rank).filter((value) => value !== null)) + } + }) +} + +/** Wins and regressions, by rank move. Misses are listed separately and always first. */ +function movers(pairs) { + const withRank = pairs.filter((p) => p.delta.rank !== null && p.delta.rank !== 0) + const wins = withRank + .filter((p) => p.delta.rank > 0) + .sort((a, b) => b.delta.rank - a.delta.rank) + .slice(0, MOVER_LIMIT) + const losses = withRank + .filter((p) => p.delta.rank < 0) + .sort((a, b) => a.delta.rank - b.delta.rank) + .slice(0, MOVER_LIMIT) + const foundByHybrid = pairs.filter( + (p) => p.dense.firstRelevantRank === 0 && p.hybrid.firstRelevantRank > 0 + ) + const lostByHybrid = pairs.filter( + (p) => p.dense.firstRelevantRank > 0 && p.hybrid.firstRelevantRank === 0 + ) + return { wins, losses, foundByHybrid, lostByHybrid } +} + +const typeTable = (rows) => + rows + .map( + (row) => + `| ${row.type} | ${row.questions} | ${row.improved} | ${row.tied} | ${row.regressed} | ` + + `${signed(row.meanDeltaNdcg10)} | ${signed(row.meanDeltaRank, 2)} |` + ) + .join('\n') + +const moverTable = (rows) => + rows.length === 0 + ? '| — | — | — | — |' + : rows + .map( + (p) => + `| ${p.id} | ${p.type} | ${rankText(p.dense.firstRelevantRank)} | ` + + `${rankText(p.hybrid.firstRelevantRank)} | ${signed(p.delta.rank, 0)} | ` + + `${signed(p.delta.ndcgAt10)} |` + ) + .join('\n') + +const splitSection = (name, data, note) => { + const { wins, losses, foundByHybrid, lostByHybrid } = movers(data.pairs) + return `## ${name} + +${note} + +Questions compared: ${data.questions}. Mean ΔnDCG@10 ${signed(data.meanDelta.ndcgAt10)}, +mean Δrank ${signed(data.meanDelta.rank, 2)}. + +### By query type + +| Type | n | improved | tied | regressed | mean ΔnDCG@10 | mean Δrank | +| --- | --- | --- | --- | --- | --- | --- | +${typeTable(data.byType)} + +### Biggest rank wins + +| id | type | dense rank | hybrid rank | Δrank | ΔnDCG@10 | +| --- | --- | --- | --- | --- | --- | +${moverTable(wins)} + +### Biggest rank regressions + +| id | type | dense rank | hybrid rank | Δrank | ΔnDCG@10 | +| --- | --- | --- | --- | --- | --- | +${moverTable(losses)} + +${ + foundByHybrid.length > 0 + ? `**Found by hybrid, missed by dense** (${foundByHybrid.length}): ${foundByHybrid + .map((p) => `${p.id} (${p.type})`) + .join(', ')}` + : '**Found by hybrid, missed by dense**: none.' +} + +${ + lostByHybrid.length > 0 + ? `**Lost by hybrid, found by dense** (${lostByHybrid.length}): ${lostByHybrid + .map((p) => `${p.id} (${p.type})`) + .join(', ')}` + : '**Lost by hybrid, found by dense**: none.' +} +` +} + +const markdown = `# Paired dense ↔ hybrid deltas — v1.6 (#192) + +Generated by \`node scripts/eval-paired.mjs\`. Numbers are harness output; do not edit them by hand. + +## What this answers + +The aggregate comparison says hybrid is ahead on the full set and level on \`validation\`. +An average cannot say whether that is a broad small gain or a few rescued cases, and those +two readings imply different next steps. This is the per-question pair. + +Metrics come from \`src/main/eval/metrics.ts\`, so a delta cannot disagree with the metric it +is a delta of. Unanswerable questions are excluded — they have no rank to compare, and the +abstention diagnostics cover them. + +\`validation\` is the side that **selects**; the \`test\` breakdown below **explains** the +difference that was observed and must not be used to choose. Nothing here changes the +shipped strategy. + +${splitSection( + 'Validation (this side selects)', + bySplit.validation, + 'Chosen on this side, so a win here is eligible to inform a decision.' +)} +${splitSection( + 'Test (explanatory only)', + bySplit.test, + 'Not used to choose anything. Present because the aggregate difference this is meant to explain was measured here.' +)} +## Reproduce + +\`\`\`bash +npm run eval:prepare # one-time, networked model bootstrap +npm run eval:paired # offline; rewrites this file +\`\`\` +` + +mkdirSync(resolve(OUT_MD, '..'), { recursive: true }) +writeFileSync( + OUT_JSON, + `${JSON.stringify( + { + baseline: 'v1.6', + strategies: ['dense', 'hybrid'], + splits: Object.fromEntries( + SPLITS.map((split) => [ + split, + { + questions: bySplit[split].questions, + meanDelta: bySplit[split].meanDelta, + byType: bySplit[split].byType, + pairs: bySplit[split].pairs + } + ]) + ) + }, + null, + 2 + )}\n` +) +writeFileSync(OUT_MD, markdown) + +for (const split of SPLITS) { + const data = bySplit[split] + const counts = data.byType.reduce( + (acc, row) => ({ + improved: acc.improved + row.improved, + tied: acc.tied + row.tied, + regressed: acc.regressed + row.regressed + }), + { improved: 0, tied: 0, regressed: 0 } + ) + console.log( + `[paired] ${split}: ${data.questions} questions, improved ${counts.improved} / tied ${counts.tied} / regressed ${counts.regressed}, mean ΔnDCG@10 ${signed(data.meanDelta.ndcgAt10)}` + ) +} +console.log(`[paired] wrote ${OUT_JSON} and ${OUT_MD}`)