From 27fdda27ae860b1c0953fd18ca66730ed32a5ce2 Mon Sep 17 00:00:00 2001 From: MrSibe Date: Wed, 30 Sep 2026 17:08:04 +0800 Subject: [PATCH] feat(eval): bounded parameter sweep with a trade-off dashboard MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The epic asks for a grid over `candidateK` and `contextK` with latency, index size and context size recorded next to quality, and explicitly not for a single aggregate "RAG score". Child 7 of #192. `npm run eval:sweep` runs the real harness over `strategy × candidateK {5,10,20,40} × contextK {3,5,8}` (24 runs, chunking fixed) and writes `docs/eval/sweep-v1.6.{json,md}`. The harness gained `contextChars` per question — the size of the context window, reported as **characters, not tokens**, because the harness pins an embedding model and no generation tokenizer. ## What the dashboard already shows Two of the three axes are decided by this corpus, and one of them decisively: - **`candidateK` changes nothing.** Every metric is identical from 5 to 40, because the corpus is 19 chunks and the relevant passages are already inside the top 5. This is the saturation problem #192 describes, now visible on the axis it affects. - **`contextK` has a clear optimum here: 3.** Context recall is 1.0000 at every width, while precision falls `0.3556 → 0.2133 → 0.1333` and the window grows `2079 → 3327 → 5312` characters as it widens. Wider adds prompt cost and noise for **no** recall. The production default is already 3, and this is the first evidence that it is the right 3 rather than a guess. - **hybrid beats dense** on nDCG@10 (0.9561 vs 0.9437) and MAP@10 (0.9389 vs 0.9222) at every setting, with no metric regressing — consistent with #197. The report labels the grid maximum as **not** a recommendation, because selecting on the same questions is how a benchmark becomes a lookup table; the adoption rule and the `validation`/`test` split are what keep the decision honest. Timing p95 is reported per row but is noisy at this sample size; it is in the table so a latency cost can be seen, not so it can be ranked. ## Testing - `npm run typecheck` — clean - `npm test` — 497 pass - `npm run eval` — regenerated (`contextChars` added to `perQuestion`) - `npm run eval:sweep` — 24 runs, dashboard written Part of #192 (child 7). --- .prettierignore | 5 + docs/eval/baseline-v1.6.json | 30 +++ docs/eval/baseline-v1.6.md | 4 +- docs/eval/sweep-v1.6.json | 341 +++++++++++++++++++++++++++++++++++ docs/eval/sweep-v1.6.md | 64 +++++++ eval/README.md | 6 + package.json | 3 +- scripts/eval-sweep.mjs | 209 +++++++++++++++++++++ src/main/eval/harness.ts | 3 + src/main/eval/types.ts | 7 + 10 files changed, 669 insertions(+), 3 deletions(-) create mode 100644 docs/eval/sweep-v1.6.json create mode 100644 docs/eval/sweep-v1.6.md create mode 100644 scripts/eval-sweep.mjs diff --git a/.prettierignore b/.prettierignore index ae11386..f6f4ae5 100644 --- a/.prettierignore +++ b/.prettierignore @@ -24,6 +24,11 @@ docs/eval/chunking-*.md docs/eval/threshold-*.json docs/eval/threshold-*.md +# Same reason again: generated by `scripts/eval-sweep.mjs`. Regenerate with +# `npm run eval:sweep`. +docs/eval/sweep-*.json +docs/eval/sweep-*.md + # And the same again for `scripts/eval-retrieval.mjs` (#77). docs/eval/retrieval-*.json docs/eval/retrieval-*.md diff --git a/docs/eval/baseline-v1.6.json b/docs/eval/baseline-v1.6.json index e598a6c..43b77d7 100644 --- a/docs/eval/baseline-v1.6.json +++ b/docs/eval/baseline-v1.6.json @@ -116,6 +116,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2074, "matchesByRank": [ [ 0 @@ -147,6 +148,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2257, "matchesByRank": [ [ 0 @@ -178,6 +180,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2558, "matchesByRank": [ [ 0 @@ -209,6 +212,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2113, "matchesByRank": [ [ 0 @@ -240,6 +244,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2113, "matchesByRank": [ [ 0 @@ -271,6 +276,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 1775, "matchesByRank": [ [ 0 @@ -302,6 +308,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2091, "matchesByRank": [ [ 0 @@ -333,6 +340,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 1970, "matchesByRank": [ [ 0 @@ -364,6 +372,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 1484, "matchesByRank": [ [ 0 @@ -395,6 +404,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2682, "matchesByRank": [ [ 0 @@ -426,6 +436,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2051, "matchesByRank": [ [ 0 @@ -457,6 +468,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2007, "matchesByRank": [ [ 0 @@ -488,6 +500,7 @@ "firstRelevantRank": 3, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2007, "matchesByRank": [ [], [], @@ -519,6 +532,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2257, "matchesByRank": [ [ 0 @@ -550,6 +564,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2621, "matchesByRank": [ [ 0 @@ -581,6 +596,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2558, "matchesByRank": [ [ 0 @@ -612,6 +628,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 1773, "matchesByRank": [ [ 0 @@ -643,6 +660,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 1970, "matchesByRank": [ [ 0 @@ -674,6 +692,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2556, "matchesByRank": [ [ 0 @@ -705,6 +724,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2098, "matchesByRank": [ [ 0 @@ -736,6 +756,7 @@ "firstRelevantRank": 2, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2488, "matchesByRank": [ [], [ @@ -767,6 +788,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2488, "matchesByRank": [ [ 0 @@ -798,6 +820,7 @@ "firstRelevantRank": 1, "relevantCount": 2, "retrievedCount": 19, + "contextChars": 2257, "matchesByRank": [ [ 1 @@ -831,6 +854,7 @@ "firstRelevantRank": 1, "relevantCount": 2, "retrievedCount": 19, + "contextChars": 1970, "matchesByRank": [ [ 0 @@ -864,6 +888,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 2098, "matchesByRank": [ [ 0 @@ -895,6 +920,7 @@ "firstRelevantRank": 2, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 1867, "matchesByRank": [ [], [ @@ -926,6 +952,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 1453, "matchesByRank": [ [ 0 @@ -957,6 +984,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 1372, "matchesByRank": [ [ 0 @@ -988,6 +1016,7 @@ "firstRelevantRank": 1, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 1500, "matchesByRank": [ [ 0 @@ -1019,6 +1048,7 @@ "firstRelevantRank": 2, "relevantCount": 1, "retrievedCount": 19, + "contextChars": 1860, "matchesByRank": [ [], [ diff --git a/docs/eval/baseline-v1.6.md b/docs/eval/baseline-v1.6.md index 467fafb..5c7fc85 100644 --- a/docs/eval/baseline-v1.6.md +++ b/docs/eval/baseline-v1.6.md @@ -43,8 +43,8 @@ The type comes from `type` in `questions.jsonl`; untagged questions report as | semantic | 18 | 1.0000 | 0.9312 | 1.0000 | 0.9074 | | zh | 3 | 1.0000 | 1.0000 | 1.0000 | 1.0000 | -Timing is informational only and is **not** frozen: indexing 1499 ms, query -p50 11.58 ms, p95 16.69 ms on the +Timing is informational only and is **not** frozen: indexing 2015 ms, query +p50 71.63 ms, p95 94.77 ms on the machine that produced this file. Timing and index size depend on hardware and on the corpus, so they must never be the reason two runs differ. diff --git a/docs/eval/sweep-v1.6.json b/docs/eval/sweep-v1.6.json new file mode 100644 index 0000000..7f7fce1 --- /dev/null +++ b/docs/eval/sweep-v1.6.json @@ -0,0 +1,341 @@ +{ + "baseline": "v1.6", + "rows": [ + { + "strategy": "dense", + "candidateK": 5, + "contextK": 3, + "recallAt5": 1, + "ndcgAt10": 0.94375, + "mapAt10": 0.922222, + "contextPrecision": 0.355556, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 2078.9333333333334, + "chunkCount": 19, + "latencyP95Ms": 88.15 + }, + { + "strategy": "dense", + "candidateK": 5, + "contextK": 5, + "recallAt5": 1, + "ndcgAt10": 0.94375, + "mapAt10": 0.922222, + "contextPrecision": 0.213333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 3326.8333333333335, + "chunkCount": 19, + "latencyP95Ms": 89.14 + }, + { + "strategy": "dense", + "candidateK": 5, + "contextK": 8, + "recallAt5": 1, + "ndcgAt10": 0.94375, + "mapAt10": 0.922222, + "contextPrecision": 0.213333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 3326.8333333333335, + "chunkCount": 19, + "latencyP95Ms": 98.4 + }, + { + "strategy": "dense", + "candidateK": 10, + "contextK": 3, + "recallAt5": 1, + "ndcgAt10": 0.94375, + "mapAt10": 0.922222, + "contextPrecision": 0.355556, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 2078.9333333333334, + "chunkCount": 19, + "latencyP95Ms": 107.2 + }, + { + "strategy": "dense", + "candidateK": 10, + "contextK": 5, + "recallAt5": 1, + "ndcgAt10": 0.94375, + "mapAt10": 0.922222, + "contextPrecision": 0.213333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 3326.8333333333335, + "chunkCount": 19, + "latencyP95Ms": 77.76 + }, + { + "strategy": "dense", + "candidateK": 10, + "contextK": 8, + "recallAt5": 1, + "ndcgAt10": 0.94375, + "mapAt10": 0.922222, + "contextPrecision": 0.133333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 5311.633333333333, + "chunkCount": 19, + "latencyP95Ms": 91.92 + }, + { + "strategy": "dense", + "candidateK": 20, + "contextK": 3, + "recallAt5": 1, + "ndcgAt10": 0.94375, + "mapAt10": 0.922222, + "contextPrecision": 0.355556, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 2078.9333333333334, + "chunkCount": 19, + "latencyP95Ms": 94.88 + }, + { + "strategy": "dense", + "candidateK": 20, + "contextK": 5, + "recallAt5": 1, + "ndcgAt10": 0.94375, + "mapAt10": 0.922222, + "contextPrecision": 0.213333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 3326.8333333333335, + "chunkCount": 19, + "latencyP95Ms": 85.49 + }, + { + "strategy": "dense", + "candidateK": 20, + "contextK": 8, + "recallAt5": 1, + "ndcgAt10": 0.94375, + "mapAt10": 0.922222, + "contextPrecision": 0.133333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 5311.633333333333, + "chunkCount": 19, + "latencyP95Ms": 84.23 + }, + { + "strategy": "dense", + "candidateK": 40, + "contextK": 3, + "recallAt5": 1, + "ndcgAt10": 0.94375, + "mapAt10": 0.922222, + "contextPrecision": 0.355556, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 2078.9333333333334, + "chunkCount": 19, + "latencyP95Ms": 91.85 + }, + { + "strategy": "dense", + "candidateK": 40, + "contextK": 5, + "recallAt5": 1, + "ndcgAt10": 0.94375, + "mapAt10": 0.922222, + "contextPrecision": 0.213333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 3326.8333333333335, + "chunkCount": 19, + "latencyP95Ms": 86.76 + }, + { + "strategy": "dense", + "candidateK": 40, + "contextK": 8, + "recallAt5": 1, + "ndcgAt10": 0.94375, + "mapAt10": 0.922222, + "contextPrecision": 0.133333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 5311.633333333333, + "chunkCount": 19, + "latencyP95Ms": 89.21 + }, + { + "strategy": "hybrid", + "candidateK": 5, + "contextK": 3, + "recallAt5": 1, + "ndcgAt10": 0.956053, + "mapAt10": 0.938889, + "contextPrecision": 0.355556, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 2044.1333333333334, + "chunkCount": 19, + "latencyP95Ms": 81.29 + }, + { + "strategy": "hybrid", + "candidateK": 5, + "contextK": 5, + "recallAt5": 1, + "ndcgAt10": 0.956053, + "mapAt10": 0.938889, + "contextPrecision": 0.213333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 3361.6, + "chunkCount": 19, + "latencyP95Ms": 79.52 + }, + { + "strategy": "hybrid", + "candidateK": 5, + "contextK": 8, + "recallAt5": 1, + "ndcgAt10": 0.956053, + "mapAt10": 0.938889, + "contextPrecision": 0.213333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 3361.6, + "chunkCount": 19, + "latencyP95Ms": 85.98 + }, + { + "strategy": "hybrid", + "candidateK": 10, + "contextK": 3, + "recallAt5": 1, + "ndcgAt10": 0.956053, + "mapAt10": 0.938889, + "contextPrecision": 0.355556, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 2062.733333333333, + "chunkCount": 19, + "latencyP95Ms": 86.15 + }, + { + "strategy": "hybrid", + "candidateK": 10, + "contextK": 5, + "recallAt5": 1, + "ndcgAt10": 0.956053, + "mapAt10": 0.938889, + "contextPrecision": 0.213333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 3525.633333333333, + "chunkCount": 19, + "latencyP95Ms": 92.74 + }, + { + "strategy": "hybrid", + "candidateK": 10, + "contextK": 8, + "recallAt5": 1, + "ndcgAt10": 0.956053, + "mapAt10": 0.938889, + "contextPrecision": 0.133333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 5509.8, + "chunkCount": 19, + "latencyP95Ms": 89.97 + }, + { + "strategy": "hybrid", + "candidateK": 20, + "contextK": 3, + "recallAt5": 1, + "ndcgAt10": 0.956053, + "mapAt10": 0.938889, + "contextPrecision": 0.355556, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 2056.9333333333334, + "chunkCount": 19, + "latencyP95Ms": 59.56 + }, + { + "strategy": "hybrid", + "candidateK": 20, + "contextK": 5, + "recallAt5": 1, + "ndcgAt10": 0.956053, + "mapAt10": 0.938889, + "contextPrecision": 0.213333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 3445.866666666667, + "chunkCount": 19, + "latencyP95Ms": 77.17 + }, + { + "strategy": "hybrid", + "candidateK": 20, + "contextK": 8, + "recallAt5": 1, + "ndcgAt10": 0.956053, + "mapAt10": 0.938889, + "contextPrecision": 0.133333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 5572.6, + "chunkCount": 19, + "latencyP95Ms": 91.41 + }, + { + "strategy": "hybrid", + "candidateK": 40, + "contextK": 3, + "recallAt5": 1, + "ndcgAt10": 0.956053, + "mapAt10": 0.938889, + "contextPrecision": 0.355556, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 2056.9333333333334, + "chunkCount": 19, + "latencyP95Ms": 89.62 + }, + { + "strategy": "hybrid", + "candidateK": 40, + "contextK": 5, + "recallAt5": 1, + "ndcgAt10": 0.956053, + "mapAt10": 0.938889, + "contextPrecision": 0.213333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 3445.866666666667, + "chunkCount": 19, + "latencyP95Ms": 66.45 + }, + { + "strategy": "hybrid", + "candidateK": 40, + "contextK": 8, + "recallAt5": 1, + "ndcgAt10": 0.956053, + "mapAt10": 0.938889, + "contextPrecision": 0.133333, + "contextRecall": 1, + "noResultRate": 0, + "meanContextChars": 5572.6, + "chunkCount": 19, + "latencyP95Ms": 76.65 + } + ] +} diff --git a/docs/eval/sweep-v1.6.md b/docs/eval/sweep-v1.6.md new file mode 100644 index 0000000..2168fa4 --- /dev/null +++ b/docs/eval/sweep-v1.6.md @@ -0,0 +1,64 @@ +# Parameter sweep — v1.6 (#192) + +Generated by `node scripts/eval-sweep.mjs`. Numbers are harness output; do not edit them by hand. + +## What was measured + +The real harness, the same corpus, chunking held fixed, over +dense / hybrid × candidateK {5, 10, 20, 40} × contextK {3, 5, 8} — 24 runs. +Each row differs from its neighbour in one parameter. + +| Strategy | candidateK | contextK | Recall@5 | nDCG@10 | MAP@10 | Context P | Context R | No-result | Context chars | Index | p95 | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | +| dense | 5 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 19 | 88.15 ms | +| dense | 5 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 19 | 89.14 ms | +| dense | 5 | 8 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 19 | 98.40 ms | +| dense | 10 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 19 | 107.20 ms | +| dense | 10 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 19 | 77.76 ms | +| dense | 10 | 8 | 1.0000 | 0.9437 | 0.9222 | 0.1333 | 1.0000 | 0.0000 | 5312 | 19 | 91.92 ms | +| dense | 20 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 19 | 94.88 ms | +| dense | 20 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 19 | 85.49 ms | +| dense | 20 | 8 | 1.0000 | 0.9437 | 0.9222 | 0.1333 | 1.0000 | 0.0000 | 5312 | 19 | 84.23 ms | +| dense | 40 | 3 | 1.0000 | 0.9437 | 0.9222 | 0.3556 | 1.0000 | 0.0000 | 2079 | 19 | 91.85 ms | +| dense | 40 | 5 | 1.0000 | 0.9437 | 0.9222 | 0.2133 | 1.0000 | 0.0000 | 3327 | 19 | 86.76 ms | +| dense | 40 | 8 | 1.0000 | 0.9437 | 0.9222 | 0.1333 | 1.0000 | 0.0000 | 5312 | 19 | 89.21 ms | +| hybrid | 5 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2044 | 19 | 81.29 ms | +| hybrid | 5 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3362 | 19 | 79.52 ms | +| hybrid | 5 | 8 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3362 | 19 | 85.98 ms | +| hybrid | 10 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2063 | 19 | 86.15 ms | +| hybrid | 10 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3526 | 19 | 92.74 ms | +| hybrid | 10 | 8 | 1.0000 | 0.9561 | 0.9389 | 0.1333 | 1.0000 | 0.0000 | 5510 | 19 | 89.97 ms | +| hybrid | 20 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2057 | 19 | 59.56 ms | +| hybrid | 20 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3446 | 19 | 77.17 ms | +| hybrid | 20 | 8 | 1.0000 | 0.9561 | 0.9389 | 0.1333 | 1.0000 | 0.0000 | 5573 | 19 | 91.41 ms | +| hybrid | 40 | 3 | 1.0000 | 0.9561 | 0.9389 | 0.3556 | 1.0000 | 0.0000 | 2057 | 19 | 89.62 ms | +| hybrid | 40 | 5 | 1.0000 | 0.9561 | 0.9389 | 0.2133 | 1.0000 | 0.0000 | 3446 | 19 | 66.45 ms | +| hybrid | 40 | 8 | 1.0000 | 0.9561 | 0.9389 | 0.1333 | 1.0000 | 0.0000 | 5573 | 19 | 76.65 ms | + +## How to read it + +- **`candidateK`** moves the ranking metrics and latency: it is how wide the first + stage searches. It cannot change `Context P`/`Context R`, because those look at the + first `contextK` of the fused list and a prefix is unaffected by how deep the list was. +- **`contextK`** moves `Context P` and `Context R` and the context size, not the + ranking metrics. Wider recall rises and precision falls; that is the trade, and both + columns are here so it is visible rather than argued about. +- **Context chars** is a proxy for prompt size, not a token count: the harness pins the + embedding model, not any generation model's tokenizer. +- **No-result** is the share of questions whose retrieval returned nothing at all. + +Best nDCG@10 in this grid: `hybrid` candidateK=5, +contextK=3 (0.9561). +Best context precision: `dense` candidateK=5, +contextK=3 (0.3556). + +These are **not** recommendations. Selecting the grid maximum on the same questions is +how a benchmark becomes a lookup table; the adoption rule in `baseline-v1.6.md` +decides, and the `validation`/`test` split is what keeps that honest. + +## Reproduce + +```bash +npm run eval:prepare # one-time, networked model bootstrap +npm run eval:sweep # offline; rewrites this file +``` diff --git a/eval/README.md b/eval/README.md index 913073e..f715942 100644 --- a/eval/README.md +++ b/eval/README.md @@ -11,6 +11,7 @@ npm run eval:prepare # one-time, networked: download the pinned embedding mod npm run eval # offline and deterministic: run the harness, rewrite the baseline npm run eval:retrieval # strategy comparison (#77) npm run eval:threshold # derive the similarity threshold on validation, report on test +npm run eval:sweep # bounded grid over strategy × candidateK × contextK, one dashboard ``` ### The harness runs the production configuration @@ -39,6 +40,11 @@ result. `--eval-split=validation` selects roughly a third of the questions by a deterministic hash of the id; `test` is the rest. `npm run eval:threshold` uses this to pick a similarity threshold on `validation` and report it on `test`. +`npm run eval:sweep` runs a bounded grid (`strategy × candidateK × contextK`) and +writes one dashboard with quality, context precision/recall, prompt size, index size +and latency side by side. Its grid maximum is labelled as **not** a recommendation: +selecting on the same questions is how a benchmark becomes a lookup table. + `eval:prepare` downloads the pinned `multilingual-e5-small` revision into the app's model cache and verifies it. `eval` never touches the network: if the model is missing it stops with diff --git a/package.json b/package.json index f7b0bcb..4d9134d 100644 --- a/package.json +++ b/package.json @@ -42,7 +42,8 @@ "db:push": "drizzle-kit push", "db:studio": "drizzle-kit studio", "eval:retrieval": "npm run build && node --experimental-transform-types --disable-warning=ExperimentalWarning --disable-warning=MODULE_TYPELESS_PACKAGE_JSON scripts/eval-retrieval.mjs", - "eval:threshold": "npm run build && node scripts/eval-threshold.mjs" + "eval:threshold": "npm run build && node scripts/eval-threshold.mjs", + "eval:sweep": "npm run build && node scripts/eval-sweep.mjs" }, "//test": [ "`node --test` strips TypeScript types rather than compiling them, and strip-only", diff --git a/scripts/eval-sweep.mjs b/scripts/eval-sweep.mjs new file mode 100644 index 0000000..0ed7dc5 --- /dev/null +++ b/scripts/eval-sweep.mjs @@ -0,0 +1,209 @@ +#!/usr/bin/env node +/** + * Parameter sweep + dashboard for #192 (child 7). + * + * Runs the real harness over a bounded grid of `strategy × candidateK × contextK` and + * writes one table. The point is not to find a single number — #192 is explicit that a + * single aggregate "RAG score" says almost nothing — but to make the trade-offs + * visible side by side: quality, the context window's precision/recall, prompt size, + * index size and latency. + * + * Chunking is held fixed so a row differs from its neighbour in one parameter. + * + * Usage: + * node scripts/eval-sweep.mjs + * node scripts/eval-sweep.mjs --strategies=dense --candidate-k=10,20 --context-k=3,5 + * + * The embedding model must already be prepared (`npm run eval:prepare`). + */ + +import { spawn } from 'node:child_process' +import { existsSync, mkdtempSync, mkdirSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join, resolve } from 'node:path' + +const OUT_MD = resolve(readArg('--out=', 'docs/eval/sweep-v1.6.md')) +const OUT_JSON = OUT_MD.replace(/\.md$/, '.json') + +const STRATEGIES = readList('--strategies=', ['dense', 'hybrid']) +const CANDIDATE_KS = readNumberList('--candidate-k=', [5, 10, 20, 40]) +const CONTEXT_KS = readNumberList('--context-k=', [3, 5, 8]) + +function readArg(prefix, fallback) { + const arg = process.argv.find((value) => value.startsWith(prefix)) + return arg ? arg.slice(prefix.length) : fallback +} + +function readList(prefix, fallback) { + const raw = readArg(prefix, null) + return raw ? raw.split(',').filter(Boolean) : fallback +} + +function readNumberList(prefix, fallback) { + const raw = readArg(prefix, null) + if (!raw) return fallback + return raw.split(',').map((value) => { + const parsed = Number(value) + if (!Number.isFinite(parsed)) throw new Error(`${prefix} expects numbers, got ${value}`) + return parsed + }) +} + +// Node 24 refuses to spawn a `.cmd`/`.bat` without `shell: true` (EINVAL), and the +// `.bin` entry is exactly that on Windows. Use the real binary the wrapper runs. +const { default: electronBinary } = await import('electron') +const executable = resolve(electronBinary) + +if (!existsSync(executable)) { + console.error('[sweep] could not find the electron binary. Run `npm install` first.') + process.exit(1) +} + +function runOne(strategy, candidateK, contextK, outDir) { + return new Promise((resolvePromise, reject) => { + const args = [ + '.', + '--eval-harness', + '--eval-baseline=v1.6', + `--eval-out=${outDir}`, + `--eval-retrieval=${strategy}`, + `--eval-candidate-k=${candidateK}`, + `--eval-context-k=${contextK}` + ] + + const isRoot = typeof process.getuid === 'function' && process.getuid() === 0 + if (isRoot || process.env.CI) args.push('--no-sandbox') + + const child = spawn(executable, args, { + stdio: ['ignore', 'pipe', 'pipe'], + env: { ...process.env, ELECTRON_DISABLE_SECURITY_WARNINGS: '1' } + }) + + child.on('error', reject) + child.on('exit', (code) => { + if (code !== 0) { + reject(new Error(`${strategy} candidateK=${candidateK} contextK=${contextK} exited ${code}`)) + return + } + const reportPath = join(outDir, 'baseline-v1.6.json') + if (!existsSync(reportPath)) { + reject(new Error(`${strategy} candidateK=${candidateK} contextK=${contextK} wrote no report`)) + return + } + resolvePromise(JSON.parse(readFileSync(reportPath, 'utf8'))) + }) + }) +} + +/** p95 and index size are informational; the deterministic JSON excludes timing. */ +function readTiming(mdPath) { + if (!existsSync(mdPath)) return { latencyP95Ms: null } + const text = readFileSync(mdPath, 'utf8') + const p95 = /p95 ([\d.]+) ms/.exec(text) + return { latencyP95Ms: p95 ? Number(p95[1]) : null } +} + +const mean = (values) => (values.length === 0 ? 0 : values.reduce((a, b) => a + b, 0) / values.length) + +const workDir = mkdtempSync(join(tmpdir(), 'knownote-sweep-')) +const rows = [] + +try { + for (const strategy of STRATEGIES) { + for (const candidateK of CANDIDATE_KS) { + for (const contextK of CONTEXT_KS) { + const label = `${strategy} candidateK=${candidateK} contextK=${contextK}` + console.log(`[sweep] ${label}`) + const outDir = join(workDir, `${strategy}-${candidateK}-${contextK}`) + mkdirSync(outDir, { recursive: true }) + const report = await runOne(strategy, candidateK, contextK, outDir) + const perQuestion = report.perQuestion ?? [] + const noResult = perQuestion.filter((q) => q.retrievedCount === 0).length + + rows.push({ + strategy, + candidateK, + contextK, + recallAt5: report.metrics.recallAt5, + ndcgAt10: report.metrics.ndcgAt10, + mapAt10: report.metrics.mapAt10, + contextPrecision: report.metrics.contextPrecision, + contextRecall: report.metrics.contextRecall, + noResultRate: perQuestion.length === 0 ? 0 : noResult / perQuestion.length, + meanContextChars: mean(perQuestion.map((q) => q.contextChars)), + chunkCount: report.config.chunkCount, + ...readTiming(join(outDir, 'baseline-v1.6.md')) + }) + } + } + } +} finally { + rmSync(workDir, { recursive: true, force: true }) +} + +const format4 = (value) => value.toFixed(4) + +const tableRows = rows + .map( + (row) => + `| ${row.strategy} | ${row.candidateK} | ${row.contextK} | ${format4(row.recallAt5)} | ` + + `${format4(row.ndcgAt10)} | ${format4(row.mapAt10)} | ${format4(row.contextPrecision)} | ` + + `${format4(row.contextRecall)} | ${format4(row.noResultRate)} | ` + + `${Math.round(row.meanContextChars)} | ${row.chunkCount} | ${row.latencyP95Ms?.toFixed(2) ?? '—'} ms |` + ) + .join('\n') + +const best = (key, filter = () => true) => + rows.filter(filter).reduce((a, b) => (a === null || b[key] > a[key] ? b : a), null) + +const bestNdcg = best('ndcgAt10') +const bestContextPrecision = best('contextPrecision') + +const markdown = `# Parameter sweep — v1.6 (#192) + +Generated by \`node scripts/eval-sweep.mjs\`. Numbers are harness output; do not edit them by hand. + +## What was measured + +The real harness, the same corpus, chunking held fixed, over +${STRATEGIES.join(' / ')} × candidateK {${CANDIDATE_KS.join(', ')}} × contextK {${CONTEXT_KS.join(', ')}} — ${rows.length} runs. +Each row differs from its neighbour in one parameter. + +| Strategy | candidateK | contextK | Recall@5 | nDCG@10 | MAP@10 | Context P | Context R | No-result | Context chars | Index | p95 | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | +${tableRows} + +## How to read it + +- **\`candidateK\`** moves the ranking metrics and latency: it is how wide the first + stage searches. It cannot change \`Context P\`/\`Context R\`, because those look at the + first \`contextK\` of the fused list and a prefix is unaffected by how deep the list was. +- **\`contextK\`** moves \`Context P\` and \`Context R\` and the context size, not the + ranking metrics. Wider recall rises and precision falls; that is the trade, and both + columns are here so it is visible rather than argued about. +- **Context chars** is a proxy for prompt size, not a token count: the harness pins the + embedding model, not any generation model's tokenizer. +- **No-result** is the share of questions whose retrieval returned nothing at all. + +Best nDCG@10 in this grid: \`${bestNdcg.strategy}\` candidateK=${bestNdcg.candidateK}, +contextK=${bestNdcg.contextK} (${format4(bestNdcg.ndcgAt10)}). +Best context precision: \`${bestContextPrecision.strategy}\` candidateK=${bestContextPrecision.candidateK}, +contextK=${bestContextPrecision.contextK} (${format4(bestContextPrecision.contextPrecision)}). + +These are **not** recommendations. Selecting the grid maximum on the same questions is +how a benchmark becomes a lookup table; the adoption rule in \`baseline-v1.6.md\` +decides, and the \`validation\`/\`test\` split is what keeps that honest. + +## Reproduce + +\`\`\`bash +npm run eval:prepare # one-time, networked model bootstrap +npm run eval:sweep # offline; rewrites this file +\`\`\` +` + +mkdirSync(resolve(OUT_MD, '..'), { recursive: true }) +writeFileSync(OUT_JSON, `${JSON.stringify({ baseline: 'v1.6', rows }, null, 2)}\n`) +writeFileSync(OUT_MD, markdown) + +console.log(`[sweep] wrote ${OUT_JSON} and ${OUT_MD}`) diff --git a/src/main/eval/harness.ts b/src/main/eval/harness.ts index c33c43d..727f775 100644 --- a/src/main/eval/harness.ts +++ b/src/main/eval/harness.ts @@ -265,6 +265,9 @@ export async function runEvalHarness( firstRelevantRank: firstRelevantRank(matchesByRank), relevantCount: groundTruth.length, retrievedCount: results.length, + contextChars: results + .slice(0, options.contextK) + .reduce((total, result) => total + result.content.length, 0), matchesByRank }) } diff --git a/src/main/eval/types.ts b/src/main/eval/types.ts index 8833a23..bb7f00e 100644 --- a/src/main/eval/types.ts +++ b/src/main/eval/types.ts @@ -112,6 +112,13 @@ export interface QuestionReport { firstRelevantRank: number relevantCount: number retrievedCount: number + /** + * 送进 context 窗口的前 `contextK` 条证据的字符数(#192 child 7)。 + * + * 是字符数而不是 token 数:不同模型的分词器不同,而 harness 只固定了 embedding + * 模型。把它当 prompt 预算的代理看,不要当成某个模型的 token 数。 + */ + contextChars: number /** Ground-truth indices matched by each retrieved rank, in rank order. */ matchesByRank: number[][] }