diff --git a/.prettierignore b/.prettierignore index ae11386..f6f4ae5 100644 --- a/.prettierignore +++ b/.prettierignore @@ -24,6 +24,11 @@ docs/eval/chunking-*.md docs/eval/threshold-*.json docs/eval/threshold-*.md +# Same reason again: generated by `scripts/eval-sweep.mjs`. Regenerate with +# `npm run eval:sweep`. +docs/eval/sweep-*.json +docs/eval/sweep-*.md + # And the same again for `scripts/eval-retrieval.mjs` (#77). docs/eval/retrieval-*.json docs/eval/retrieval-*.md diff --git a/docs/eval/baseline-v1.6.json b/docs/eval/baseline-v1.6.json index e598a6c..43b77d7 100644 --- a/docs/eval/baseline-v1.6.json +++ b/docs/eval/baseline-v1.6.json @@ -116,6 +116,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2074, "matchesByRank": [ [ 0 @@ -147,6 +148,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2257, "matchesByRank": [ [ 0 @@ -178,6 +180,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2558, "matchesByRank": [ [ 0 @@ -209,6 +212,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2113, "matchesByRank": [ [ 0 @@ -240,6 +244,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2113, "matchesByRank": [ [ 0 @@ -271,6 +276,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 1775, "matchesByRank": [ [ 0 @@ -302,6 +308,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2091, "matchesByRank": [ [ 0 @@ -333,6 +340,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 1970, "matchesByRank": [ [ 0 @@ -364,6 +372,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 1484, "matchesByRank": [ [ 0 @@ -395,6 +404,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2682, "matchesByRank": [ [ 0 @@ -426,6 +436,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2051, "matchesByRank": [ [ 0 @@ -457,6 +468,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2007, "matchesByRank": [ [ 0 @@ -488,6 +500,7 @@ "firstRelevantRank": 3, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2007, "matchesByRank": [ [], [], @@ -519,6 +532,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2257, "matchesByRank": [ [ 0 @@ -550,6 +564,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2621, "matchesByRank": [ [ 0 @@ -581,6 +596,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2558, "matchesByRank": [ [ 0 @@ -612,6 +628,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 1773, "matchesByRank": [ [ 0 @@ -643,6 +660,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 1970, "matchesByRank": [ [ 0 @@ -674,6 +692,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2556, "matchesByRank": [ [ 0 @@ -705,6 +724,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2098, "matchesByRank": [ [ 0 @@ -736,6 +756,7 @@ "firstRelevantRank": 2, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2488, "matchesByRank": [ [], [ @@ -767,6 +788,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2488, "matchesByRank": [ [ 0 @@ -798,6 +820,7 @@ "firstRelevantRank": 1, "relevantCount": 2, "retrievedCount": 19, + "contextChars": 2257, "matchesByRank": [ [ 1 @@ -831,6 +854,7 @@ "firstRelevantRank": 1, "relevantCount": 2, "retrievedCount": 19, + "contextChars": 1970, "matchesByRank": [ [ 0 @@ -864,6 +888,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2098, "matchesByRank": [ [ 0 @@ -895,6 +920,7 @@ "firstRelevantRank": 2, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 1867, "matchesByRank": [ [], [ @@ -926,6 +952,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 1453, "matchesByRank": [ [ 0 @@ -957,6 +984,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 1372, "matchesByRank": [ [ 0 @@ -988,6 +1016,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 1500, "matchesByRank": [ [ 0 @@ -1019,6 +1048,7 @@ "firstRelevantRank": 2, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 1860, "matchesByRank": [ [], [ diff --git a/docs/eval/baseline-v1.6.md b/docs/eval/baseline-v1.6.md index 467fafb..5c7fc85 100644 --- a/docs/eval/baseline-v1.6.md +++ b/docs/eval/baseline-v1.6.md @@ -43,8 +43,8 @@ The type comes from `type` in `questions.jsonl`; untagged questions report as | semantic | 18 | 1.0000 | 0.9312 | 1.0000 | 0.9074 | | zh | 3 | 1.0000 | 1.0000 | 1.0000 | 1.0000 | -Timing is informational only and is **not** frozen: indexing 1499 ms, query -p50 11.58 ms, p95 16.69 ms on the +Timing is informational only and is **not** frozen: indexing 2015 ms, query +p50 71.63 ms, p95 94.77 ms on the machine that produced this file. Timing and index size depend on hardware and on the corpus, so they must never be the reason two runs differ. diff --git a/docs/eval/sweep-v1.6.json b/docs/eval/sweep-v1.6.json new file mode 100644 index 0000000..7f7fce1 --- /dev/null +++ b/docs/eval/sweep-v1.6.json @@ -0,0 +1,341 @@ +{ + "baseline": "v1.6", + "rows": [ + { + "strategy": "dense", + "candidateK": 5, + "contextK": 3, + "recallAt5": 1, + "ndcgAt10": 0.94375, + "mapAt10": 0.922222, + "contextPrecision": 0.355556, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 2078.9333333333334, + "chunkCount": 19, + "latencyP95Ms": 88.15 + }, + { + "strategy": "dense", + "candidateK": 5, + "contextK": 5, + "recallAt5": 1, + "ndcgAt10": 0.94375, + "mapAt10": 0.922222, + "contextPrecision": 0.213333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 3326.8333333333335, + "chunkCount": 19, + "latencyP95Ms": 89.14 + }, + { + "strategy": "dense", + "candidateK": 5, + "contextK": 8, + "recallAt5": 1, + "ndcgAt10": 0.94375, + "mapAt10": 0.922222, + "contextPrecision": 0.213333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 3326.8333333333335, + "chunkCount": 19, + "latencyP95Ms": 98.4 + }, + { + "strategy": "dense", + "candidateK": 10, + "contextK": 3, + "recallAt5": 1, + "ndcgAt10": 0.94375, + "mapAt10": 0.922222, + "contextPrecision": 0.355556, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 2078.9333333333334, + "chunkCount": 19, + "latencyP95Ms": 107.2 + }, + { + "strategy": "dense", + "candidateK": 10, + "contextK": 5, + "recallAt5": 1, + "ndcgAt10": 0.94375, + "mapAt10": 0.922222, + "contextPrecision": 0.213333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 3326.8333333333335, + "chunkCount": 19, + "latencyP95Ms": 77.76 + }, + { + "strategy": "dense", + "candidateK": 10, + "contextK": 8, + "recallAt5": 1, + "ndcgAt10": 0.94375, + "mapAt10": 0.922222, + "contextPrecision": 0.133333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 5311.633333333333, + "chunkCount": 19, + "latencyP95Ms": 91.92 + }, + { + "strategy": "dense", + "candidateK": 20, + "contextK": 3, + "recallAt5": 1, + "ndcgAt10": 0.94375, + "mapAt10": 0.922222, + "contextPrecision": 0.355556, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 2078.9333333333334, + "chunkCount": 19, + "latencyP95Ms": 94.88 + }, + { + "strategy": "dense", + "candidateK": 20, + "contextK": 5, + "recallAt5": 1, + "ndcgAt10": 0.94375, + "mapAt10": 0.922222, + "contextPrecision": 0.213333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 3326.8333333333335, + "chunkCount": 19, + "latencyP95Ms": 85.49 + }, + { + "strategy": "dense", + "candidateK": 20, + "contextK": 8, + "recallAt5": 1, + "ndcgAt10": 0.94375, + "mapAt10": 0.922222, + "contextPrecision": 0.133333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 5311.633333333333, + "chunkCount": 19, + "latencyP95Ms": 84.23 + }, + { + "strategy": "dense", + "candidateK": 40, + "contextK": 3, + "recallAt5": 1, + "ndcgAt10": 0.94375, + "mapAt10": 0.922222, + "contextPrecision": 0.355556, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 2078.9333333333334, + "chunkCount": 19, + "latencyP95Ms": 91.85 + }, + { + "strategy": "dense", + "candidateK": 40, + "contextK": 5, + "recallAt5": 1, + "ndcgAt10": 0.94375, + "mapAt10": 0.922222, + "contextPrecision": 0.213333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 3326.8333333333335, + "chunkCount": 19, + "latencyP95Ms": 86.76 + }, + { + "strategy": "dense", + "candidateK": 40, + "contextK": 8, + "recallAt5": 1, + "ndcgAt10": 0.94375, + "mapAt10": 0.922222, + "contextPrecision": 0.133333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 5311.633333333333, + "chunkCount": 19, + "latencyP95Ms": 89.21 + }, + { + "strategy": "hybrid", + "candidateK": 5, + "contextK": 3, + "recallAt5": 1, + "ndcgAt10": 0.956053, + "mapAt10": 0.938889, + "contextPrecision": 0.355556, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 2044.1333333333334, + "chunkCount": 19, + "latencyP95Ms": 81.29 + }, + { + "strategy": "hybrid", + "candidateK": 5, + "contextK": 5, + "recallAt5": 1, + "ndcgAt10": 0.956053, + "mapAt10": 0.938889, + "contextPrecision": 0.213333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 3361.6, + "chunkCount": 19, + "latencyP95Ms": 79.52 + }, + { + "strategy": "hybrid", + "candidateK": 5, + "contextK": 8, + "recallAt5": 1, + "ndcgAt10": 0.956053, + "mapAt10": 0.938889, + "contextPrecision": 0.213333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 3361.6, + "chunkCount": 19, + "latencyP95Ms": 85.98 + }, + { + "strategy": "hybrid", + "candidateK": 10, + "contextK": 3, + "recallAt5": 1, + "ndcgAt10": 0.956053, + "mapAt10": 0.938889, + "contextPrecision": 0.355556, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 2062.733333333333, + "chunkCount": 19, + "latencyP95Ms": 86.15 + }, + { + "strategy": "hybrid", + "candidateK": 10, + "contextK": 5, + "recallAt5": 1, + "ndcgAt10": 0.956053, + "mapAt10": 0.938889, + "contextPrecision": 0.213333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 3525.633333333333, + "chunkCount": 19, + "latencyP95Ms": 92.74 + }, + { + "strategy": "hybrid", + "candidateK": 10, + "contextK": 8, + "recallAt5": 1, + "ndcgAt10": 0.956053, + "mapAt10": 0.938889, + "contextPrecision": 0.133333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 5509.8, + "chunkCount": 19, + "latencyP95Ms": 89.97 + }, + { + "strategy": "hybrid", + "candidateK": 20, + "contextK": 3, + "recallAt5": 1, + "ndcgAt10": 0.956053, + "mapAt10": 0.938889, + "contextPrecision": 0.355556, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 2056.9333333333334, + "chunkCount": 19, + "latencyP95Ms": 59.56 + }, + { + "strategy": "hybrid", + "candidateK": 20, + "contextK": 5, + "recallAt5": 1, + "ndcgAt10": 0.956053, + "mapAt10": 0.938889, + "contextPrecision": 0.213333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 3445.866666666667, + "chunkCount": 19, + "latencyP95Ms": 77.17 + }, + { + "strategy": "hybrid", + "candidateK": 20, + "contextK": 8, + "recallAt5": 1, + "ndcgAt10": 0.956053, + "mapAt10": 0.938889, + "contextPrecision": 0.133333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 5572.6, + "chunkCount": 19, + "latencyP95Ms": 91.41 + }, + { + "strategy": "hybrid", + "candidateK": 40, + "contextK": 3, + "recallAt5": 1, + "ndcgAt10": 0.956053, + "mapAt10": 0.938889, + "contextPrecision": 0.355556, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 2056.9333333333334, + "chunkCount": 19, + "latencyP95Ms": 89.62 + }, + { + "strategy": "hybrid", + "candidateK": 40, + "contextK": 5, + "recallAt5": 1, + "ndcgAt10": 0.956053, + "mapAt10": 0.938889, + "contextPrecision": 0.213333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 3445.866666666667, + "chunkCount": 19, + "latencyP95Ms": 66.45 + }, + { + "strategy": "hybrid", + "candidateK": 40, + "contextK": 8, + "recallAt5": 1, + "ndcgAt10": 0.956053, + "mapAt10": 0.938889, + "contextPrecision": 0.133333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 5572.6, + "chunkCount": 19, + "latencyP95Ms": 76.65 + } + ] +} diff --git a/docs/eval/sweep-v1.6.md b/docs/eval/sweep-v1.6.md new file mode 100644 index 0000000..2168fa4 --- /dev/null +++ b/docs/eval/sweep-v1.6.md @@ -0,0 +1,64 @@ +# Parameter sweep — v1.6 (#192) + +Generated by `node scripts/eval-sweep.mjs`. Numbers are harness output; do not edit them by hand. + +## What was measured + +The real harness, the same corpus, chunking held fixed, over +dense / hybrid × candidateK {5, 10, 20, 40} × contextK {3, 5, 8} — 24 runs. +Each row differs from its neighbour in one parameter. + +| Strategy | candidateK | contextK | Recall@5 | nDCG@10 | MAP@10 | Context P | Context R | No-result | Context chars | Index | p95 | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | +| dense | 5 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 19 | 88.15 ms | +| dense | 5 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 19 | 89.14 ms | +| dense | 5 | 8 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 19 | 98.40 ms | +| dense | 10 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 19 | 107.20 ms | +| dense | 10 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 19 | 77.76 ms | +| dense | 10 | 8 | 1.0000 | 0.9437 | 0.9222 | 0.1333 | 1.0000 | 0.0000 | 5312 | 19 | 91.92 ms | +| dense | 20 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 19 | 94.88 ms | +| dense | 20 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 19 | 85.49 ms | +| dense | 20 | 8 | 1.0000 | 0.9437 | 0.9222 | 0.1333 | 1.0000 | 0.0000 | 5312 | 19 | 84.23 ms | +| dense | 40 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 19 | 91.85 ms | +| dense | 40 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 19 | 86.76 ms | +| dense | 40 | 8 | 1.0000 | 0.9437 | 0.9222 | 0.1333 | 1.0000 | 0.0000 | 5312 | 19 | 89.21 ms | +| hybrid | 5 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2044 | 19 | 81.29 ms | +| hybrid | 5 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3362 | 19 | 79.52 ms | +| hybrid | 5 | 8 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3362 | 19 | 85.98 ms | +| hybrid | 10 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2063 | 19 | 86.15 ms | +| hybrid | 10 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3526 | 19 | 92.74 ms | +| hybrid | 10 | 8 | 1.0000 | 0.9561 | 0.9389 | 0.1333 | 1.0000 | 0.0000 | 5510 | 19 | 89.97 ms | +| hybrid | 20 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2057 | 19 | 59.56 ms | +| hybrid | 20 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3446 | 19 | 77.17 ms | +| hybrid | 20 | 8 | 1.0000 | 0.9561 | 0.9389 | 0.1333 | 1.0000 | 0.0000 | 5573 | 19 | 91.41 ms | +| hybrid | 40 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2057 | 19 | 89.62 ms | +| hybrid | 40 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3446 | 19 | 66.45 ms | +| hybrid | 40 | 8 | 1.0000 | 0.9561 | 0.9389 | 0.1333 | 1.0000 | 0.0000 | 5573 | 19 | 76.65 ms | + +## How to read it + +- **`candidateK`** moves the ranking metrics and latency: it is how wide the first + stage searches. It cannot change `Context P`/`Context R`, because those look at the + first `contextK` of the fused list and a prefix is unaffected by how deep the list was. +- **`contextK`** moves `Context P` and `Context R` and the context size, not the + ranking metrics. Wider recall rises and precision falls; that is the trade, and both + columns are here so it is visible rather than argued about. +- **Context chars** is a proxy for prompt size, not a token count: the harness pins the + embedding model, not any generation model's tokenizer. +- **No-result** is the share of questions whose retrieval returned nothing at all. + +Best nDCG@10 in this grid: `hybrid` candidateK=5, +contextK=3 (0.9561). +Best context precision: `dense` candidateK=5, +contextK=3 (0.3556). + +These are **not** recommendations. Selecting the grid maximum on the same questions is +how a benchmark becomes a lookup table; the adoption rule in `baseline-v1.6.md` +decides, and the `validation`/`test` split is what keeps that honest. + +## Reproduce + +```bash +npm run eval:prepare # one-time, networked model bootstrap +npm run eval:sweep # offline; rewrites this file +``` diff --git a/eval/README.md b/eval/README.md index 913073e..f715942 100644 --- a/eval/README.md +++ b/eval/README.md @@ -11,6 +11,7 @@ npm run eval:prepare # one-time, networked: download the pinned embedding mod npm run eval # offline and deterministic: run the harness, rewrite the baseline npm run eval:retrieval # strategy comparison (#77) npm run eval:threshold # derive the similarity threshold on validation, report on test +npm run eval:sweep # bounded grid over strategy × candidateK × contextK, one dashboard ``` ### The harness runs the production configuration @@ -39,6 +40,11 @@ result. `--eval-split=validation` selects roughly a third of the questions by a deterministic hash of the id; `test` is the rest. `npm run eval:threshold` uses this to pick a similarity threshold on `validation` and report it on `test`. +`npm run eval:sweep` runs a bounded grid (`strategy × candidateK × contextK`) and +writes one dashboard with quality, context precision/recall, prompt size, index size +and latency side by side. Its grid maximum is labelled as **not** a recommendation: +selecting on the same questions is how a benchmark becomes a lookup table. + `eval:prepare` downloads the pinned `multilingual-e5-small` revision into the app's model cache and verifies it. `eval` never touches the network: if the model is missing it stops with diff --git a/package.json b/package.json index f7b0bcb..4d9134d 100644 --- a/package.json +++ b/package.json @@ -42,7 +42,8 @@ "db:push": "drizzle-kit push", "db:studio": "drizzle-kit studio", "eval:retrieval": "npm run build && node --experimental-transform-types --disable-warning=ExperimentalWarning --disable-warning=MODULE_TYPELESS_PACKAGE_JSON scripts/eval-retrieval.mjs", - "eval:threshold": "npm run build && node scripts/eval-threshold.mjs" + "eval:threshold": "npm run build && node scripts/eval-threshold.mjs", + "eval:sweep": "npm run build && node scripts/eval-sweep.mjs" }, "//test": [ "`node --test` strips TypeScript types rather than compiling them, and strip-only", diff --git a/scripts/eval-sweep.mjs b/scripts/eval-sweep.mjs new file mode 100644 index 0000000..0ed7dc5 --- /dev/null +++ b/scripts/eval-sweep.mjs @@ -0,0 +1,209 @@ +#!/usr/bin/env node +/** + * Parameter sweep + dashboard for #192 (child 7). + * + * Runs the real harness over a bounded grid of `strategy × candidateK × contextK` and + * writes one table. The point is not to find a single number — #192 is explicit that a + * single aggregate "RAG score" says almost nothing — but to make the trade-offs + * visible side by side: quality, the context window's precision/recall, prompt size, + * index size and latency. + * + * Chunking is held fixed so a row differs from its neighbour in one parameter. + * + * Usage: + * node scripts/eval-sweep.mjs + * node scripts/eval-sweep.mjs --strategies=dense --candidate-k=10,20 --context-k=3,5 + * + * The embedding model must already be prepared (`npm run eval:prepare`). + */ + +import { spawn } from 'node:child_process' +import { existsSync, mkdtempSync, mkdirSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join, resolve } from 'node:path' + +const OUT_MD = resolve(readArg('--out=', 'docs/eval/sweep-v1.6.md')) +const OUT_JSON = OUT_MD.replace(/\.md$/, '.json') + +const STRATEGIES = readList('--strategies=', ['dense', 'hybrid']) +const CANDIDATE_KS = readNumberList('--candidate-k=', [5, 10, 20, 40]) +const CONTEXT_KS = readNumberList('--context-k=', [3, 5, 8]) + +function readArg(prefix, fallback) { + const arg = process.argv.find((value) => value.startsWith(prefix)) + return arg ? arg.slice(prefix.length) : fallback +} + +function readList(prefix, fallback) { + const raw = readArg(prefix, null) + return raw ? raw.split(',').filter(Boolean) : fallback +} + +function readNumberList(prefix, fallback) { + const raw = readArg(prefix, null) + if (!raw) return fallback + return raw.split(',').map((value) => { + const parsed = Number(value) + if (!Number.isFinite(parsed)) throw new Error(`${prefix} expects numbers, got ${value}`) + return parsed + }) +} + +// Node 24 refuses to spawn a `.cmd`/`.bat` without `shell: true` (EINVAL), and the +// `.bin` entry is exactly that on Windows. Use the real binary the wrapper runs. +const { default: electronBinary } = await import('electron') +const executable = resolve(electronBinary) + +if (!existsSync(executable)) { + console.error('[sweep] could not find the electron binary. Run `npm install` first.') + process.exit(1) +} + +function runOne(strategy, candidateK, contextK, outDir) { + return new Promise((resolvePromise, reject) => { + const args = [ + '.', + '--eval-harness', + '--eval-baseline=v1.6', + `--eval-out=${outDir}`, + `--eval-retrieval=${strategy}`, + `--eval-candidate-k=${candidateK}`, + `--eval-context-k=${contextK}` + ] + + const isRoot = typeof process.getuid === 'function' && process.getuid() === 0 + if (isRoot || process.env.CI) args.push('--no-sandbox') + + const child = spawn(executable, args, { + stdio: ['ignore', 'pipe', 'pipe'], + env: { ...process.env, ELECTRON_DISABLE_SECURITY_WARNINGS: '1' } + }) + + child.on('error', reject) + child.on('exit', (code) => { + if (code !== 0) { + reject(new Error(`${strategy} candidateK=${candidateK} contextK=${contextK} exited ${code}`)) + return + } + const reportPath = join(outDir, 'baseline-v1.6.json') + if (!existsSync(reportPath)) { + reject(new Error(`${strategy} candidateK=${candidateK} contextK=${contextK} wrote no report`)) + return + } + resolvePromise(JSON.parse(readFileSync(reportPath, 'utf8'))) + }) + }) +} + +/** p95 and index size are informational; the deterministic JSON excludes timing. */ +function readTiming(mdPath) { + if (!existsSync(mdPath)) return { latencyP95Ms: null } + const text = readFileSync(mdPath, 'utf8') + const p95 = /p95 ([\d.]+) ms/.exec(text) + return { latencyP95Ms: p95 ? Number(p95[1]) : null } +} + +const mean = (values) => (values.length === 0 ? 0 : values.reduce((a, b) => a + b, 0) / values.length) + +const workDir = mkdtempSync(join(tmpdir(), 'knownote-sweep-')) +const rows = [] + +try { + for (const strategy of STRATEGIES) { + for (const candidateK of CANDIDATE_KS) { + for (const contextK of CONTEXT_KS) { + const label = `${strategy} candidateK=${candidateK} contextK=${contextK}` + console.log(`[sweep] ${label}`) + const outDir = join(workDir, `${strategy}-${candidateK}-${contextK}`) + mkdirSync(outDir, { recursive: true }) + const report = await runOne(strategy, candidateK, contextK, outDir) + const perQuestion = report.perQuestion ?? [] + const noResult = perQuestion.filter((q) => q.retrievedCount === 0).length + + rows.push({ + strategy, + candidateK, + contextK, + recallAt5: report.metrics.recallAt5, + ndcgAt10: report.metrics.ndcgAt10, + mapAt10: report.metrics.mapAt10, + contextPrecision: report.metrics.contextPrecision, + contextRecall: report.metrics.contextRecall, + noResultRate: perQuestion.length === 0 ? 0 : noResult / perQuestion.length, + meanContextChars: mean(perQuestion.map((q) => q.contextChars)), + chunkCount: report.config.chunkCount, + ...readTiming(join(outDir, 'baseline-v1.6.md')) + }) + } + } + } +} finally { + rmSync(workDir, { recursive: true, force: true }) +} + +const format4 = (value) => value.toFixed(4) + +const tableRows = rows + .map( + (row) => + `| ${row.strategy} | ${row.candidateK} | ${row.contextK} | ${format4(row.recallAt5)} | ` + + `${format4(row.ndcgAt10)} | ${format4(row.mapAt10)} | ${format4(row.contextPrecision)} | ` + + `${format4(row.contextRecall)} | ${format4(row.noResultRate)} | ` + + `${Math.round(row.meanContextChars)} | ${row.chunkCount} | ${row.latencyP95Ms?.toFixed(2) ?? '—'} ms |` + ) + .join('\n') + +const best = (key, filter = () => true) => + rows.filter(filter).reduce((a, b) => (a === null || b[key] > a[key] ? b : a), null) + +const bestNdcg = best('ndcgAt10') +const bestContextPrecision = best('contextPrecision') + +const markdown = `# Parameter sweep — v1.6 (#192) + +Generated by \`node scripts/eval-sweep.mjs\`. Numbers are harness output; do not edit them by hand. + +## What was measured + +The real harness, the same corpus, chunking held fixed, over +${STRATEGIES.join(' / ')} × candidateK {${CANDIDATE_KS.join(', ')}} × contextK {${CONTEXT_KS.join(', ')}} — ${rows.length} runs. +Each row differs from its neighbour in one parameter. + +| Strategy | candidateK | contextK | Recall@5 | nDCG@10 | MAP@10 | Context P | Context R | No-result | Context chars | Index | p95 | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | +${tableRows} + +## How to read it + +- **\`candidateK\`** moves the ranking metrics and latency: it is how wide the first + stage searches. It cannot change \`Context P\`/\`Context R\`, because those look at the + first \`contextK\` of the fused list and a prefix is unaffected by how deep the list was. +- **\`contextK\`** moves \`Context P\` and \`Context R\` and the context size, not the + ranking metrics. Wider recall rises and precision falls; that is the trade, and both + columns are here so it is visible rather than argued about. +- **Context chars** is a proxy for prompt size, not a token count: the harness pins the + embedding model, not any generation model's tokenizer. +- **No-result** is the share of questions whose retrieval returned nothing at all. + +Best nDCG@10 in this grid: \`${bestNdcg.strategy}\` candidateK=${bestNdcg.candidateK}, +contextK=${bestNdcg.contextK} (${format4(bestNdcg.ndcgAt10)}). +Best context precision: \`${bestContextPrecision.strategy}\` candidateK=${bestContextPrecision.candidateK}, +contextK=${bestContextPrecision.contextK} (${format4(bestContextPrecision.contextPrecision)}). + +These are **not** recommendations. Selecting the grid maximum on the same questions is +how a benchmark becomes a lookup table; the adoption rule in \`baseline-v1.6.md\` +decides, and the \`validation\`/\`test\` split is what keeps that honest. + +## Reproduce + +\`\`\`bash +npm run eval:prepare # one-time, networked model bootstrap +npm run eval:sweep # offline; rewrites this file +\`\`\` +` + +mkdirSync(resolve(OUT_MD, '..'), { recursive: true }) +writeFileSync(OUT_JSON, `${JSON.stringify({ baseline: 'v1.6', rows }, null, 2)}\n`) +writeFileSync(OUT_MD, markdown) + +console.log(`[sweep] wrote ${OUT_JSON} and ${OUT_MD}`) diff --git a/src/main/eval/harness.ts b/src/main/eval/harness.ts index c33c43d..727f775 100644 --- a/src/main/eval/harness.ts +++ b/src/main/eval/harness.ts @@ -265,6 +265,9 @@ export async function runEvalHarness( firstRelevantRank: firstRelevantRank(matchesByRank), relevantCount: groundTruth.length, retrievedCount: results.length, + contextChars: results + .slice(0, options.contextK) + .reduce((total, result) => total + result.content.length, 0), matchesByRank }) } diff --git a/src/main/eval/types.ts b/src/main/eval/types.ts index 8833a23..bb7f00e 100644 --- a/src/main/eval/types.ts +++ b/src/main/eval/types.ts @@ -112,6 +112,13 @@ export interface QuestionReport { firstRelevantRank: number relevantCount: number retrievedCount: number + /** + * 送进 context 窗口的前 `contextK` 条证据的字符数(#192 child 7)。 + * + * 是字符数而不是 token 数:不同模型的分词器不同,而 harness 只固定了 embedding + * 模型。把它当 prompt 预算的代理看,不要当成某个模型的 token 数。 + */ + contextChars: number /** Ground-truth indices matched by each retrieved rank, in rank order. */ matchesByRank: number[][] }