diff --git a/.prettierignore b/.prettierignore index da32ca2..f418484 100644 --- a/.prettierignore +++ b/.prettierignore @@ -33,6 +33,10 @@ docs/eval/sweep-*.md docs/eval/scores-*.json docs/eval/scores-*.md +# And by `scripts/eval-paired.mjs`. Regenerate with `npm run eval:paired`. +docs/eval/paired-*.json +docs/eval/paired-*.md + # And the same again for `scripts/eval-retrieval.mjs` (#77). docs/eval/retrieval-*.json docs/eval/retrieval-*.md diff --git a/docs/eval/paired-v1.6.json b/docs/eval/paired-v1.6.json new file mode 100644 index 0000000..dd1bfc2 --- /dev/null +++ b/docs/eval/paired-v1.6.json @@ -0,0 +1,1254 @@ +{ + "baseline": "v1.6", + "strategies": [ + "dense", + "hybrid" + ], + "splits": { + "validation": { + "questions": 19, + "meanDelta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0.004654087156011809 + }, + "byType": [ + { + "type": "cross-lingual", + "questions": 4, + "improved": 0, + "tied": 4, + "regressed": 0, + "meanDeltaNdcg10": 0, + "meanDeltaRank": 0 + }, + { + "type": "exact", + "questions": 4, + "improved": 2, + "tied": 2, + "regressed": 0, + "meanDeltaNdcg10": 0.05006329887451612, + "meanDeltaRank": 0.5 + }, + { + "type": "hard-negative", + "questions": 2, + "improved": 0, + "tied": 2, + "regressed": 0, + "meanDeltaNdcg10": 0, + "meanDeltaRank": 0 + }, + { + "type": "multi-hop", + "questions": 2, + "improved": 0, + "tied": 2, + "regressed": 0, + "meanDeltaNdcg10": 0.08861964339202388, + "meanDeltaRank": 0 + }, + { + "type": "semantic", + "questions": 6, + "improved": 0, + "tied": 5, + "regressed": 1, + "meanDeltaNdcg10": -0.04817747105298131, + "meanDeltaRank": -0.3333333333333333 + }, + { + "type": "zh", + "questions": 1, + "improved": 0, + "tied": 1, + "regressed": 0, + "meanDeltaNdcg10": 0, + "meanDeltaRank": 0 + } + ], + "pairs": [ + { + "id": "q002", + "type": "exact", + "question": "How many replicate samples are collected at each river station?", + "dense": { + "firstRelevantRank": 3, + "recallAt5": 1, + "ndcgAt10": 0.5 + }, + "hybrid": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "delta": { + "rank": 1, + "recallAt5": 0, + "ndcgAt10": 0.13092975357145753 + }, + "classification": "improved" + }, + { + "id": "q004", + "type": "semantic", + "question": "Why do nickel-rich battery packs need more aggressive thermal management?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q009", + "type": "semantic", + "question": "Why are trees with aggressive surface roots unsuitable for narrow verges?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q011", + "type": "exact", + "question": "Is the salt percentage in fermentation based on vegetable weight or water weight?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q015", + "type": "semantic", + "question": "Why does deep-water oxygen fall while a lake remains stratified?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q018", + "type": "semantic", + "question": "Why is one continuous planted roof layer better than several isolated planted beds?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q021", + "type": "semantic", + "question": "Why is wave energy harder to schedule ahead than tidal energy?", + "dense": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "hybrid": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q023", + "type": "multi-hop", + "question": "How does the river sampling protocol differ from the lake sampling protocol?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 0.5, + "ndcgAt10": 0.6131471927654584 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 0.5, + "ndcgAt10": 0.7903864795495061 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0.17723928678404777 + }, + "classification": "tied" + }, + { + "id": "q028", + "type": "zh", + "question": "茶叶储存的相对湿度上限是多少?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q030", + "type": "cross-lingual", + "question": "为什么潮汐能比风能和太阳能更容易提前安排发电?", + "dense": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "hybrid": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q037", + "type": "cross-lingual", + "question": "推移质为什么比悬移质更难测?", + "dense": { + "firstRelevantRank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "hybrid": { + "firstRelevantRank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "delta": { + "rank": null, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q040", + "type": "cross-lingual", + "question": "富镍电池为什么对散热要求更高?", + "dense": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "hybrid": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q043", + "type": "cross-lingual", + "question": "花期遭遇晚霜,主要受损的是什么?", + "dense": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "hybrid": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q045", + "type": "exact", + "question": "What is the default retry count in Gateway API v2?", + "dense": { + "firstRelevantRank": 4, + "recallAt5": 1, + "ndcgAt10": 0.43067655807339306 + }, + "hybrid": { + "firstRelevantRank": 3, + "recallAt5": 1, + "ndcgAt10": 0.5 + }, + "delta": { + "rank": 1, + "recallAt5": 0, + "ndcgAt10": 0.06932344192660694 + }, + "classification": "improved" + }, + { + "id": "q047", + "type": "hard-negative", + "question": "What is the context window of Gateway API v3, in tokens?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q050", + "type": "semantic", + "question": "Which Gateway API version waits longest between retries after a failed request?", + "dense": { + "firstRelevantRank": 10, + "recallAt5": 0, + "ndcgAt10": 0.2890648263178879 + }, + "hybrid": { + "firstRelevantRank": 12, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "delta": { + "rank": -2, + "recallAt5": 0, + "ndcgAt10": -0.2890648263178879 + }, + "classification": "regressed" + }, + { + "id": "q055", + "type": "exact", + "question": "How many replicate samples are collected at each reservoir monitoring station?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q057", + "type": "hard-negative", + "question": "At what depth below the surface does the coastal programme place its shore-station sensor?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q060", + "type": "multi-hop", + "question": "Compare the sampling interval and replicate count of the reservoir and groundwater programmes.", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + } + ] + }, + "test": { + "questions": 34, + "meanDelta": { + "rank": 0.3939393939393939, + "recallAt5": 0.029411764705882353, + "ndcgAt10": 0.04318215877040841 + }, + "byType": [ + { + "type": "cross-lingual", + "questions": 5, + "improved": 0, + "tied": 5, + "regressed": 0, + "meanDeltaNdcg10": 0, + "meanDeltaRank": 0 + }, + { + "type": "exact", + "questions": 4, + "improved": 0, + "tied": 4, + "regressed": 0, + "meanDeltaNdcg10": 0, + "meanDeltaRank": 0 + }, + { + "type": "hard-negative", + "questions": 5, + "improved": 1, + "tied": 4, + "regressed": 0, + "meanDeltaNdcg10": 0.026185950714291507, + "meanDeltaRank": 0.2 + }, + { + "type": "multi-hop", + "questions": 3, + "improved": 1, + "tied": 2, + "regressed": 0, + "meanDeltaNdcg10": 0.09781329792785898, + "meanDeltaRank": 0.3333333333333333 + }, + { + "type": "semantic", + "questions": 15, + "improved": 3, + "tied": 12, + "regressed": 0, + "meanDeltaNdcg10": 0.06958825005592344, + "meanDeltaRank": 0.7333333333333333 + }, + { + "type": "zh", + "questions": 2, + "improved": 0, + "tied": 2, + "regressed": 0, + "meanDeltaNdcg10": 0, + "meanDeltaRank": 0 + } + ], + "pairs": [ + { + "id": "q001", + "type": "semantic", + "question": "Why is bedload harder to measure than suspended sediment?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q003", + "type": "semantic", + "question": "What is the central trade-off in lithium-ion cell design?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q005", + "type": "semantic", + "question": "What happens once the separator in a battery cell melts?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q006", + "type": "exact", + "question": "At what temperature do honeybees begin to forage?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q007", + "type": "semantic", + "question": "What does a late frost damage during full bloom?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q008", + "type": "semantic", + "question": "Why is a continuous tree canopy more effective at cooling than isolated trees?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q010", + "type": "exact", + "question": "At what temperature is lactic acid fermentation fastest?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q012", + "type": "semantic", + "question": "Why must tidal turbines be sited in places with very fast currents?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q013", + "type": "semantic", + "question": "What is the main environmental concern for tidal energy installations?", + "dense": { + "firstRelevantRank": 11, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "hybrid": { + "firstRelevantRank": 4, + "recallAt5": 1, + "ndcgAt10": 0.43067655807339306 + }, + "delta": { + "rank": 7, + "recallAt5": 1, + "ndcgAt10": 0.43067655807339306 + }, + "classification": "improved" + }, + { + "id": "q014", + "type": "exact", + "question": "In lake monitoring, how is the sampling depth actually recorded?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q016", + "type": "semantic", + "question": "How do supercapacitors hold their charge?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q017", + "type": "semantic", + "question": "Why can solitary bees pollinate a bloom week that is too cold for honeybee hives?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q019", + "type": "semantic", + "question": "Where do acetic acid bacteria sit in a vinegar culture?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q020", + "type": "semantic", + "question": "What happens if a vinegar culture is sealed airtight?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q022", + "type": "exact", + "question": "Where does siting for wave energy devices concentrate, and where does it not?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q024", + "type": "multi-hop", + "question": "A street canopy and a green roof are both said to cool; what surface does each one shade?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q025", + "type": "semantic", + "question": "Which preservation method depends on keeping air away from the food?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q026", + "type": "semantic", + "question": "Why can one cold morning cost a grower the whole crop even when colonies are brought in?", + "dense": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 1, + "recallAt5": 0, + "ndcgAt10": 0.36907024642854247 + }, + "classification": "improved" + }, + { + "id": "q027", + "type": "zh", + "question": "绿茶应该怎样保存才能减缓氧化?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q029", + "type": "zh", + "question": "为什么冷冻保存的茶叶取出后不能立刻打开包装?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q038", + "type": "cross-lingual", + "question": "河流监测中,每个站点要采几份平行样?", + "dense": { + "firstRelevantRank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "hybrid": { + "firstRelevantRank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "delta": { + "rank": null, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q039", + "type": "cross-lingual", + "question": "设计锂离子电芯时最核心的取舍是什么?", + "dense": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "hybrid": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q041", + "type": "cross-lingual", + "question": "电芯隔膜一旦熔化会导致什么后果?", + "dense": { + "firstRelevantRank": 4, + "recallAt5": 1, + "ndcgAt10": 0.43067655807339306 + }, + "hybrid": { + "firstRelevantRank": 4, + "recallAt5": 1, + "ndcgAt10": 0.43067655807339306 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q042", + "type": "cross-lingual", + "question": "蜜蜂大致从什么温度开始出巢觅食?", + "dense": { + "firstRelevantRank": 3, + "recallAt5": 1, + "ndcgAt10": 0.5 + }, + "hybrid": { + "firstRelevantRank": 3, + "recallAt5": 1, + "ndcgAt10": 0.5 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q044", + "type": "cross-lingual", + "question": "为什么成片的树冠比孤立的树降温效果更好?", + "dense": { + "firstRelevantRank": 9, + "recallAt5": 0, + "ndcgAt10": 0.3010299956639812 + }, + "hybrid": { + "firstRelevantRank": 9, + "recallAt5": 0, + "ndcgAt10": 0.3010299956639812 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q046", + "type": "hard-negative", + "question": "What is the default request timeout of Gateway API v1?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q048", + "type": "hard-negative", + "question": "What is the default rate limit of Gateway API v2?", + "dense": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "hybrid": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q049", + "type": "hard-negative", + "question": "What is the default request timeout of Gateway API v3?", + "dense": { + "firstRelevantRank": 3, + "recallAt5": 1, + "ndcgAt10": 0.5 + }, + "hybrid": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "delta": { + "rank": 1, + "recallAt5": 0, + "ndcgAt10": 0.13092975357145753 + }, + "classification": "improved" + }, + { + "id": "q051", + "type": "multi-hop", + "question": "Compare the default timeouts and rate limits of Gateway API v1 and v3.", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 0.75, + "ndcgAt10": 0.6366824387328317 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 0.75, + "ndcgAt10": 0.6366824387328317 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q052", + "type": "multi-hop", + "question": "Compare the retry count and context window of Gateway API v2 and v3.", + "dense": { + "firstRelevantRank": 2, + "recallAt5": 0.25, + "ndcgAt10": 0.35914753008966527 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 0.25, + "ndcgAt10": 0.6525874238732422 + }, + "delta": { + "rank": 1, + "recallAt5": 0, + "ndcgAt10": 0.29343989378357693 + }, + "classification": "improved" + }, + { + "id": "q056", + "type": "hard-negative", + "question": "How many replicate samples are collected at each estuary transect station?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q058", + "type": "hard-negative", + "question": "At what depth below the water table do groundwater boreholes carry their pressure transducer?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + }, + { + "id": "q059", + "type": "semantic", + "question": "Which monitoring programme samples most frequently?", + "dense": { + "firstRelevantRank": 5, + "recallAt5": 1, + "ndcgAt10": 0.38685280723454163 + }, + "hybrid": { + "firstRelevantRank": 2, + "recallAt5": 1, + "ndcgAt10": 0.6309297535714575 + }, + "delta": { + "rank": 3, + "recallAt5": 0, + "ndcgAt10": 0.2440769463369159 + }, + "classification": "improved" + }, + { + "id": "q061", + "type": "semantic", + "question": "Why does the estuary programme take more replicate samples than the other monitoring programmes?", + "dense": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "hybrid": { + "firstRelevantRank": 1, + "recallAt5": 1, + "ndcgAt10": 1 + }, + "delta": { + "rank": 0, + "recallAt5": 0, + "ndcgAt10": 0 + }, + "classification": "tied" + } + ] + } + } +} diff --git a/docs/eval/paired-v1.6.md b/docs/eval/paired-v1.6.md new file mode 100644 index 0000000..dc185dd --- /dev/null +++ b/docs/eval/paired-v1.6.md @@ -0,0 +1,97 @@ +# Paired dense ↔ hybrid deltas — v1.6 (#192) + +Generated by `node scripts/eval-paired.mjs`. Numbers are harness output; do not edit them by hand. + +## What this answers + +The aggregate comparison says hybrid is ahead on the full set and level on `validation`. +An average cannot say whether that is a broad small gain or a few rescued cases, and those +two readings imply different next steps. This is the per-question pair. + +Metrics come from `src/main/eval/metrics.ts`, so a delta cannot disagree with the metric it +is a delta of. Unanswerable questions are excluded — they have no rank to compare, and the +abstention diagnostics cover them. + +`validation` is the side that **selects**; the `test` breakdown below **explains** the +difference that was observed and must not be used to choose. Nothing here changes the +shipped strategy. + +## Validation (this side selects) + +Chosen on this side, so a win here is eligible to inform a decision. + +Questions compared: 19. Mean ΔnDCG@10 +0.0047, +mean Δrank 0.00. + +### By query type + +| Type | n | improved | tied | regressed | mean ΔnDCG@10 | mean Δrank | +| --- | --- | --- | --- | --- | --- | --- | +| cross-lingual | 4 | 0 | 4 | 0 | 0.0000 | 0.00 | +| exact | 4 | 2 | 2 | 0 | +0.0501 | +0.50 | +| hard-negative | 2 | 0 | 2 | 0 | 0.0000 | 0.00 | +| multi-hop | 2 | 0 | 2 | 0 | +0.0886 | 0.00 | +| semantic | 6 | 0 | 5 | 1 | -0.0482 | -0.33 | +| zh | 1 | 0 | 1 | 0 | 0.0000 | 0.00 | + +### Biggest rank wins + +| id | type | dense rank | hybrid rank | Δrank | ΔnDCG@10 | +| --- | --- | --- | --- | --- | --- | +| q002 | exact | 3 | 2 | +1 | +0.1309 | +| q045 | exact | 4 | 3 | +1 | +0.0693 | + +### Biggest rank regressions + +| id | type | dense rank | hybrid rank | Δrank | ΔnDCG@10 | +| --- | --- | --- | --- | --- | --- | +| q050 | semantic | 10 | 12 | -2 | -0.2891 | + +**Found by hybrid, missed by dense**: none. + +**Lost by hybrid, found by dense**: none. + +## Test (explanatory only) + +Not used to choose anything. Present because the aggregate difference this is meant to explain was measured here. + +Questions compared: 34. Mean ΔnDCG@10 +0.0432, +mean Δrank +0.39. + +### By query type + +| Type | n | improved | tied | regressed | mean ΔnDCG@10 | mean Δrank | +| --- | --- | --- | --- | --- | --- | --- | +| cross-lingual | 5 | 0 | 5 | 0 | 0.0000 | 0.00 | +| exact | 4 | 0 | 4 | 0 | 0.0000 | 0.00 | +| hard-negative | 5 | 1 | 4 | 0 | +0.0262 | +0.20 | +| multi-hop | 3 | 1 | 2 | 0 | +0.0978 | +0.33 | +| semantic | 15 | 3 | 12 | 0 | +0.0696 | +0.73 | +| zh | 2 | 0 | 2 | 0 | 0.0000 | 0.00 | + +### Biggest rank wins + +| id | type | dense rank | hybrid rank | Δrank | ΔnDCG@10 | +| --- | --- | --- | --- | --- | --- | +| q013 | semantic | 11 | 4 | +7 | +0.4307 | +| q059 | semantic | 5 | 2 | +3 | +0.2441 | +| q026 | semantic | 2 | 1 | +1 | +0.3691 | +| q049 | hard-negative | 3 | 2 | +1 | +0.1309 | +| q052 | multi-hop | 2 | 1 | +1 | +0.2934 | + +### Biggest rank regressions + +| id | type | dense rank | hybrid rank | Δrank | ΔnDCG@10 | +| --- | --- | --- | --- | --- | --- | +| — | — | — | — | + +**Found by hybrid, missed by dense**: none. + +**Lost by hybrid, found by dense**: none. + +## Reproduce + +```bash +npm run eval:prepare # one-time, networked model bootstrap +npm run eval:paired # offline; rewrites this file +``` diff --git a/eval/README.md b/eval/README.md index 299eca7..22afea3 100644 --- a/eval/README.md +++ b/eval/README.md @@ -13,6 +13,7 @@ npm run eval:retrieval # strategy comparison (#77); validation selects, test re npm run eval:threshold # derive the similarity threshold on validation, report on test npm run eval:sweep # bounded grid over strategy × candidateK × contextK, one dashboard npm run eval:scores # dense score distribution: can one threshold separate relevant from not? +npm run eval:paired # per-question dense ↔ hybrid deltas, by query type npm run eval:blocks eval/corpus/foo.md # print the block ordinals ground truth must use ``` diff --git a/package.json b/package.json index 418c579..8f04662 100644 --- a/package.json +++ b/package.json @@ -45,7 +45,8 @@ "eval:threshold": "npm run build && node scripts/eval-threshold.mjs", "eval:sweep": "npm run build && node scripts/eval-sweep.mjs", "eval:blocks": "node --experimental-transform-types --import ./test/ts-resolve.mjs --disable-warning=ExperimentalWarning --disable-warning=MODULE_TYPELESS_PACKAGE_JSON scripts/eval-blocks.mjs", - "eval:scores": "npm run build && node scripts/eval-scores.mjs" + "eval:scores": "npm run build && node scripts/eval-scores.mjs", + "eval:paired": "npm run build && node --experimental-transform-types --import ./test/ts-resolve.mjs --disable-warning=ExperimentalWarning --disable-warning=MODULE_TYPELESS_PACKAGE_JSON scripts/eval-paired.mjs" }, "//test": [ "`node --test` strips TypeScript types rather than compiling them, and strip-only", diff --git a/scripts/eval-paired.mjs b/scripts/eval-paired.mjs new file mode 100644 index 0000000..b097947 --- /dev/null +++ b/scripts/eval-paired.mjs @@ -0,0 +1,368 @@ +#!/usr/bin/env node +/** + * Per-query paired deltas for dense ↔ hybrid (#192). + * + * The aggregate comparison left a question it cannot answer: hybrid is ahead on the full + * set and level on `validation`, and an average cannot say whether that is a broad small + * gain or a handful of cases it rescued. A mean over forty questions cannot distinguish + * "helps a little everywhere" from "helps a lot twice". + * + * So this reports the pair per question and the classification per query type: + * + * improved / tied / regressed, with the rank move and the nDCG@10 delta + * + * `validation` is the side that selects; the `test` breakdown **explains** the observed + * difference and must not be used to choose. Nothing here changes the shipped strategy. + * + * Metrics come from `src/main/eval/metrics.ts` rather than being reimplemented, so a delta + * cannot disagree with the metric it is a delta of. + * + * Usage: + * node scripts/eval-paired.mjs + */ + +import { spawn } from 'node:child_process' +import { existsSync, mkdtempSync, mkdirSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join, resolve } from 'node:path' + +const OUT_MD = resolve(readArg('--out=', 'docs/eval/paired-v1.6.md')) +const OUT_JSON = OUT_MD.replace(/\.md$/, '.json') + +const SPLITS = ['validation', 'test'] +const STRATEGIES = [ + { id: 'dense', label: 'dense (vector)' }, + { id: 'hybrid', label: 'hybrid (RRF of dense + BM25)' } +] + +/** How many rows to show in each mover table. */ +const MOVER_LIMIT = 8 + +function readArg(prefix, fallback) { + const arg = process.argv.find((value) => value.startsWith(prefix)) + return arg ? arg.slice(prefix.length) : fallback +} + +// Node 24 refuses to spawn a `.cmd`/`.bat` without `shell: true` (EINVAL), and the +// `.bin` entry is exactly that on Windows. Use the real binary the wrapper runs. +const { default: electronBinary } = await import('electron') +const executable = resolve(electronBinary) + +if (!existsSync(executable)) { + console.error('[paired] could not find the electron binary. Run `npm install` first.') + process.exit(1) +} + +const { ndcgAtK, recallAtK } = await import('../src/main/eval/metrics.ts') + +function runStrategy(strategy, split, outDir) { + return new Promise((resolvePromise, reject) => { + const args = [ + '.', + '--eval-harness', + '--eval-baseline=paired', + `--eval-out=${outDir}`, + `--eval-split=${split}`, + `--eval-retrieval=${strategy.id}` + ] + + const isRoot = typeof process.getuid === 'function' && process.getuid() === 0 + if (isRoot || process.env.CI) args.push('--no-sandbox') + + const child = spawn(executable, args, { + stdio: ['ignore', 'pipe', 'pipe'], + env: { ...process.env, ELECTRON_DISABLE_SECURITY_WARNINGS: '1' } + }) + + child.on('error', reject) + child.on('exit', (code) => { + if (code !== 0) { + reject(new Error(`${strategy.id} (${split}) exited with code ${code}`)) + return + } + const reportPath = join(outDir, 'baseline-paired.json') + if (!existsSync(reportPath)) { + reject(new Error(`${strategy.id} (${split}) wrote no report`)) + return + } + resolvePromise(JSON.parse(readFileSync(reportPath, 'utf8'))) + }) + }) +} + +/** One side's per-question view, with the metrics the comparison is made of. */ +function sideOf(question) { + return { + firstRelevantRank: question.firstRelevantRank, + recallAt5: recallAtK(question.matchesByRank, question.relevantCount, 5), + ndcgAt10: ndcgAtK(question.matchesByRank, question.relevantCount, 10) + } +} + +/** + * Classify the pair. `firstRelevantRank` is 0 for "never found", so the miss cases cannot be + * folded into an arithmetic delta: going from found to not-found is a regression no rank + * number expresses. + */ +function classify(dense, hybrid) { + const denseFound = dense.firstRelevantRank > 0 + const hybridFound = hybrid.firstRelevantRank > 0 + + if (denseFound && hybridFound) { + const rankDelta = dense.firstRelevantRank - hybrid.firstRelevantRank + return { + classification: rankDelta > 0 ? 'improved' : rankDelta < 0 ? 'regressed' : 'tied', + rankDelta + } + } + if (!denseFound && hybridFound) return { classification: 'improved', rankDelta: null } + if (denseFound && !hybridFound) return { classification: 'regressed', rankDelta: null } + return { classification: 'tied', rankDelta: null } +} + +/** Human-readable rank move that survives the 0 = "not found" sentinel. */ +const rankText = (rank) => (rank > 0 ? String(rank) : 'not found') + +const format4 = (value) => (value === null ? '—' : value.toFixed(4)) +const signed = (value, digits = 4) => + value === null ? '—' : `${value > 0 ? '+' : ''}${value.toFixed(digits)}` + +const workDir = mkdtempSync(join(tmpdir(), 'knownote-paired-')) +const bySplit = {} + +try { + for (const split of SPLITS) { + const reports = {} + for (const strategy of STRATEGIES) { + console.log(`[paired] ${split}: ${strategy.label}`) + const outDir = join(workDir, `${split}-${strategy.id}`) + mkdirSync(outDir, { recursive: true }) + reports[strategy.id] = await runStrategy(strategy, split, outDir) + } + + const denseById = new Map(reports.dense.perQuestion.map((q) => [q.id, q])) + const pairs = [] + for (const hybridQuestion of reports.hybrid.perQuestion) { + const denseQuestion = denseById.get(hybridQuestion.id) + if (!denseQuestion) continue + // Unanswerable questions have no rank to compare; they are measured by the + // abstention diagnostics, not here. + if (!denseQuestion.answerable) continue + + const dense = sideOf(denseQuestion) + const hybrid = sideOf(hybridQuestion) + const { classification, rankDelta } = classify(dense, hybrid) + pairs.push({ + id: hybridQuestion.id, + type: hybridQuestion.type, + question: hybridQuestion.question, + dense, + hybrid, + delta: { + rank: rankDelta, + recallAt5: hybrid.recallAt5 - dense.recallAt5, + ndcgAt10: hybrid.ndcgAt10 - dense.ndcgAt10 + }, + classification + }) + } + + bySplit[split] = { + questions: pairs.length, + pairs, + byType: aggregateByType(pairs), + meanDelta: { + rank: mean(pairs.map((p) => p.delta.rank).filter((value) => value !== null)), + recallAt5: mean(pairs.map((p) => p.delta.recallAt5)), + ndcgAt10: mean(pairs.map((p) => p.delta.ndcgAt10)) + } + } + } +} finally { + rmSync(workDir, { recursive: true, force: true }) +} + +function mean(values) { + if (values.length === 0) return null + return values.reduce((a, b) => a + b, 0) / values.length +} + +function aggregateByType(pairs) { + const types = [...new Set(pairs.map((p) => p.type))].sort() + return types.map((type) => { + const group = pairs.filter((p) => p.type === type) + const count = (label) => group.filter((p) => p.classification === label).length + return { + type, + questions: group.length, + improved: count('improved'), + tied: count('tied'), + regressed: count('regressed'), + meanDeltaNdcg10: mean(group.map((p) => p.delta.ndcgAt10)), + meanDeltaRank: mean(group.map((p) => p.delta.rank).filter((value) => value !== null)) + } + }) +} + +/** Wins and regressions, by rank move. Misses are listed separately and always first. */ +function movers(pairs) { + const withRank = pairs.filter((p) => p.delta.rank !== null && p.delta.rank !== 0) + const wins = withRank + .filter((p) => p.delta.rank > 0) + .sort((a, b) => b.delta.rank - a.delta.rank) + .slice(0, MOVER_LIMIT) + const losses = withRank + .filter((p) => p.delta.rank < 0) + .sort((a, b) => a.delta.rank - b.delta.rank) + .slice(0, MOVER_LIMIT) + const foundByHybrid = pairs.filter( + (p) => p.dense.firstRelevantRank === 0 && p.hybrid.firstRelevantRank > 0 + ) + const lostByHybrid = pairs.filter( + (p) => p.dense.firstRelevantRank > 0 && p.hybrid.firstRelevantRank === 0 + ) + return { wins, losses, foundByHybrid, lostByHybrid } +} + +const typeTable = (rows) => + rows + .map( + (row) => + `| ${row.type} | ${row.questions} | ${row.improved} | ${row.tied} | ${row.regressed} | ` + + `${signed(row.meanDeltaNdcg10)} | ${signed(row.meanDeltaRank, 2)} |` + ) + .join('\n') + +const moverTable = (rows) => + rows.length === 0 + ? '| — | — | — | — |' + : rows + .map( + (p) => + `| ${p.id} | ${p.type} | ${rankText(p.dense.firstRelevantRank)} | ` + + `${rankText(p.hybrid.firstRelevantRank)} | ${signed(p.delta.rank, 0)} | ` + + `${signed(p.delta.ndcgAt10)} |` + ) + .join('\n') + +const splitSection = (name, data, note) => { + const { wins, losses, foundByHybrid, lostByHybrid } = movers(data.pairs) + return `## ${name} + +${note} + +Questions compared: ${data.questions}. Mean ΔnDCG@10 ${signed(data.meanDelta.ndcgAt10)}, +mean Δrank ${signed(data.meanDelta.rank, 2)}. + +### By query type + +| Type | n | improved | tied | regressed | mean ΔnDCG@10 | mean Δrank | +| --- | --- | --- | --- | --- | --- | --- | +${typeTable(data.byType)} + +### Biggest rank wins + +| id | type | dense rank | hybrid rank | Δrank | ΔnDCG@10 | +| --- | --- | --- | --- | --- | --- | +${moverTable(wins)} + +### Biggest rank regressions + +| id | type | dense rank | hybrid rank | Δrank | ΔnDCG@10 | +| --- | --- | --- | --- | --- | --- | +${moverTable(losses)} + +${ + foundByHybrid.length > 0 + ? `**Found by hybrid, missed by dense** (${foundByHybrid.length}): ${foundByHybrid + .map((p) => `${p.id} (${p.type})`) + .join(', ')}` + : '**Found by hybrid, missed by dense**: none.' +} + +${ + lostByHybrid.length > 0 + ? `**Lost by hybrid, found by dense** (${lostByHybrid.length}): ${lostByHybrid + .map((p) => `${p.id} (${p.type})`) + .join(', ')}` + : '**Lost by hybrid, found by dense**: none.' +} +` +} + +const markdown = `# Paired dense ↔ hybrid deltas — v1.6 (#192) + +Generated by \`node scripts/eval-paired.mjs\`. Numbers are harness output; do not edit them by hand. + +## What this answers + +The aggregate comparison says hybrid is ahead on the full set and level on \`validation\`. +An average cannot say whether that is a broad small gain or a few rescued cases, and those +two readings imply different next steps. This is the per-question pair. + +Metrics come from \`src/main/eval/metrics.ts\`, so a delta cannot disagree with the metric it +is a delta of. Unanswerable questions are excluded — they have no rank to compare, and the +abstention diagnostics cover them. + +\`validation\` is the side that **selects**; the \`test\` breakdown below **explains** the +difference that was observed and must not be used to choose. Nothing here changes the +shipped strategy. + +${splitSection( + 'Validation (this side selects)', + bySplit.validation, + 'Chosen on this side, so a win here is eligible to inform a decision.' +)} +${splitSection( + 'Test (explanatory only)', + bySplit.test, + 'Not used to choose anything. Present because the aggregate difference this is meant to explain was measured here.' +)} +## Reproduce + +\`\`\`bash +npm run eval:prepare # one-time, networked model bootstrap +npm run eval:paired # offline; rewrites this file +\`\`\` +` + +mkdirSync(resolve(OUT_MD, '..'), { recursive: true }) +writeFileSync( + OUT_JSON, + `${JSON.stringify( + { + baseline: 'v1.6', + strategies: ['dense', 'hybrid'], + splits: Object.fromEntries( + SPLITS.map((split) => [ + split, + { + questions: bySplit[split].questions, + meanDelta: bySplit[split].meanDelta, + byType: bySplit[split].byType, + pairs: bySplit[split].pairs + } + ]) + ) + }, + null, + 2 + )}\n` +) +writeFileSync(OUT_MD, markdown) + +for (const split of SPLITS) { + const data = bySplit[split] + const counts = data.byType.reduce( + (acc, row) => ({ + improved: acc.improved + row.improved, + tied: acc.tied + row.tied, + regressed: acc.regressed + row.regressed + }), + { improved: 0, tied: 0, regressed: 0 } + ) + console.log( + `[paired] ${split}: ${data.questions} questions, improved ${counts.improved} / tied ${counts.tied} / regressed ${counts.regressed}, mean ΔnDCG@10 ${signed(data.meanDelta.ndcgAt10)}` + ) +} +console.log(`[paired] wrote ${OUT_JSON} and ${OUT_MD}`)