From 25e1ada24ee07e1d3c36af88c65f7b52aad2b8b3 Mon Sep 17 00:00:00 2001 From: MrSibe Date: Wed, 30 Sep 2026 16:40:03 +0800 Subject: [PATCH 1/2] fix(eval): run the harness on Windows and keep the baseline POSIX MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two gates the v1.5 eval path needs to be trustworthy on every platform. **`npm run eval` could not start on Windows.** The three scripts resolve `node_modules/.bin/electron.cmd` and spawn it. Node 24 refuses to spawn a `.cmd`/`.bat` without `shell: true` and fails with `EINVAL`, so the harness was unrunnable on the platform this project is developed on. The `electron` package exports the path to the real executable the wrapper runs, which spawns directly on every platform and needs no shell. **A Windows run rewrote the committed baseline.** `path.relative` returns backslashes on Windows, so `config.corpus` was written as `eval\corpus` instead of `eval/corpus`. The file's whole contract is that it is identical on every machine — the CI determinism check diffs it — and a Windows run silently broke that. The label is now normalised to POSIX separators. Verified on Windows with the pinned model: `npm run eval` runs and leaves `docs/eval/baseline-v1.5.json` byte-identical to the committed file. --- scripts/eval-chunking.mjs | 8 ++++---- scripts/eval-retrieval.mjs | 8 ++++---- scripts/eval.mjs | 10 ++++++---- src/main/eval/run.ts | 20 ++++++++++++++++---- 4 files changed, 30 insertions(+), 16 deletions(-) diff --git a/scripts/eval-chunking.mjs b/scripts/eval-chunking.mjs index ecc0d4a..70f4e45 100644 --- a/scripts/eval-chunking.mjs +++ b/scripts/eval-chunking.mjs @@ -63,10 +63,10 @@ function readArg(prefix, fallback) { return arg ? arg.slice(prefix.length) : fallback } -const executable = resolve( - 'node_modules/.bin', - process.platform === 'win32' ? 'electron.cmd' : 'electron' -) +// Node 24 refuses to spawn a `.cmd`/`.bat` without `shell: true` (EINVAL), and the +// `.bin` entry is exactly that on Windows. Use the real binary the wrapper runs. +const { default: electronBinary } = await import('electron') +const executable = resolve(electronBinary) if (!existsSync(executable)) { console.error('[chunking] could not find the electron binary. Run `npm install` first.') diff --git a/scripts/eval-retrieval.mjs b/scripts/eval-retrieval.mjs index d573e3b..bccc605 100644 --- a/scripts/eval-retrieval.mjs +++ b/scripts/eval-retrieval.mjs @@ -39,10 +39,10 @@ function readArg(prefix, fallback) { return arg ? arg.slice(prefix.length) : fallback } -const executable = resolve( - 'node_modules/.bin', - process.platform === 'win32' ? 'electron.cmd' : 'electron' -) +// Node 24 refuses to spawn a `.cmd`/`.bat` without `shell: true` (EINVAL), and the +// `.bin` entry is exactly that on Windows. Use the real binary the wrapper runs. +const { default: electronBinary } = await import('electron') +const executable = resolve(electronBinary) if (!existsSync(executable)) { console.error('[retrieval] could not find the electron binary. Run `npm install` first.') diff --git a/scripts/eval.mjs b/scripts/eval.mjs index 0f4d304..8b61229 100644 --- a/scripts/eval.mjs +++ b/scripts/eval.mjs @@ -18,10 +18,12 @@ import { resolve } from 'node:path' const prepare = process.argv.includes('--prepare') const flag = prepare ? '--eval-prepare' : '--eval-harness' -const executable = resolve( - 'node_modules/.bin', - process.platform === 'win32' ? 'electron.cmd' : 'electron' -) +// The `.bin` entry is a shell wrapper (`electron.cmd` on Windows), and Node 24 +// refuses to spawn `.cmd`/`.bat` without `shell: true` — it fails with EINVAL. The +// `electron` package exports the path to the real executable the wrapper runs, so +// spawning that directly works on every platform without a shell. +const { default: electronBinary } = await import('electron') +const executable = resolve(electronBinary) if (!existsSync(executable)) { console.error('[eval] could not find the electron binary. Run `npm install` first.') diff --git a/src/main/eval/run.ts b/src/main/eval/run.ts index 10857a6..58becb8 100644 --- a/src/main/eval/run.ts +++ b/src/main/eval/run.ts @@ -13,7 +13,7 @@ import { app } from 'electron' import { mkdtempSync, rmSync, writeFileSync } from 'fs' import { mkdir } from 'fs/promises' -import { join, relative, resolve } from 'path' +import { join, relative, resolve, sep } from 'path' import { tmpdir } from 'os' import { closeDatabase, getDatabase, initDatabase, initVectorStore, runMigrations } from '../db' import { ConnectionManager } from '../models/ConnectionManager' @@ -62,6 +62,18 @@ function readBoolOption(argv: readonly string[], prefix: string, fallback: boole return value === 'true' } +/** + * Repo-relative path with POSIX separators. + * + * `path.relative` returns backslashes on Windows, and the corpus label is committed + * in the baseline. Without this, a Windows run writes `eval\corpus` and the file + * stops being identical on every machine — which is the one property the committed + * report promises. + */ +function repoRelative(absolutePath: string): string { + return relative(process.cwd(), absolutePath).split(sep).join('/') +} + /** 检索策略(#77)。默认 dense,所以不带 flag 的 `npm run eval` 仍量的是生产默认。 */ function readRetrievalStrategy(argv: readonly string[]): RetrievalStrategy { const raw = readOption(argv, '--eval-retrieval=', 'dense') @@ -162,9 +174,9 @@ export async function runEvalCli(argv: readonly string[] = process.argv): Promis const knowledgeService = new KnowledgeService(embeddingService) const options: EvalHarnessOptions = { corpusDir, - // Recorded in the report as a repo-relative path so the committed JSON is - // identical on every machine and checkout. - corpusLabel: relative(process.cwd(), corpusDir) || 'eval/corpus', + // Recorded in the report as a repo-relative, POSIX-separated path so the + // committed JSON is identical on every machine and checkout. + corpusLabel: repoRelative(corpusDir) || 'eval/corpus', questionsPath, baseline: readOption(argv, '--eval-baseline=', 'v1.5'), topK: 10, From a444a8676fea8c9540c487b9d23654faa3fa5038 Mon Sep 17 00:00:00 2001 From: MrSibe Date: Wed, 30 Sep 2026 16:46:12 +0800 Subject: [PATCH 2/2] feat(eval): separate candidateK from contextK, and run the production config MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The harness measured a retriever nobody runs. Production retrieved at `topK: 3` with `threshold: 0.5`; the harness ran `topK: 10` with `threshold: 0` and reported `Recall@5 = 1.0000` for a pipeline that silently drops rank 4 at cosine 0.47. The first child of #192 asks for the two to be the same configuration. **One K was doing two jobs.** `RetrievalRequest.topK` was both the first-stage width (KNN neighbours, BM25 limit) and the number of passages delivered. In hybrid that made fusion nearly a no-op: dense contributed `topK`, BM25 contributed `topK`, RRF fused at most `2 * topK`, and the result was sliced straight back to `topK`. The two stages are now named: ``` RetrievalRequest: candidateK first-stage width per channel (default 20) topK passages delivered (default 5) ``` `effectiveCandidateK` enforces `candidateK >= topK`, so a caller that asks for more results than the default pool (the search palette, MCP) is never silently capped. `HybridRetriever` fuses the wide pool and truncates once, after fusion. `DenseRetriever` queries `candidateK` and slices to `topK` — equivalent to before for a single strategy, since a prefix of a ranking is the same ranking. The trace and the #157 snapshot gained `candidateK`. A snapshot written before this change backfills it from `topK`, which is what that retrieval actually did. **The harness now defaults to production** (`candidateK: 20`, `contextK: 3`, `threshold: 0.5`), with `--eval-candidate-k=` / `--eval-context-k=` / `--eval-threshold=` to move them deliberately. Ranking metrics are computed at `candidateK` depth, not `contextK`: `Recall@10` needs ten results, and truncation only takes a prefix, so it cannot change the ranking being measured. `contextK` is recorded so the report describes the whole online path. `docs/eval/baseline-v1.6.{json,md}` is the new frozen baseline; v1.5 is kept as history for the #77/#78 deltas, and the CI determinism check moves to v1.6. ## What this did and did not change On the current 19-chunk corpus the production configuration produces the **same** metrics as v1.5 — `threshold: 0.5` is non-binding and `candidateK: 20` exceeds the corpus, so nothing is filtered and no ranking changes. That is the corpus limitation #192 describes, not a result. What changed is that the harness now *states* the production parameters instead of assuming different ones. Hybrid still leads dense on Recall@1 (0.8667 vs 0.8333), MRR (0.9444 vs 0.9278) and nDCG@10 (0.9561 vs 0.9437) with a wide pool; dense stays the default because the adoption rule still cannot move on a saturated Recall@5. ## Testing - `npm run typecheck` — clean - `npm test` — 481 pass (3 new: trace keeps the two Ks apart, the width invariant, the pre-#77 snapshot backfill) - `npm run eval` twice — byte-identical `docs/eval/baseline-v1.6.json` - `npm run eval:retrieval` — dense/sparse/hybrid unchanged in ordering Part of #192 (child 1). --- .github/workflows/verify.yml | 6 +- docs/eval/baseline-v1.6.json | 937 ++++++++++++++++++ docs/eval/baseline-v1.6.md | 65 ++ eval/README.md | 26 +- src/main/eval/harness.ts | 23 +- src/main/eval/report.ts | 10 +- src/main/eval/run.ts | 9 +- src/main/eval/types.ts | 13 +- src/main/ipc/chatHandlers.ts | 5 + src/main/services/KnowledgeService.ts | 6 + src/main/services/retrieval/DenseRetriever.ts | 17 +- .../services/retrieval/HybridRetriever.ts | 25 +- src/main/services/retrieval/trace.ts | 2 + src/main/services/retrieval/types.ts | 42 + src/shared/types/chat.ts | 9 +- src/shared/utils/answerSources.ts | 5 + test/answerSources.test.ts | 15 + test/retrievalContract.test.ts | 36 +- 18 files changed, 1221 insertions(+), 30 deletions(-) create mode 100644 docs/eval/baseline-v1.6.json create mode 100644 docs/eval/baseline-v1.6.md diff --git a/.github/workflows/verify.yml b/.github/workflows/verify.yml index 03f1843..597b1bf 100644 --- a/.github/workflows/verify.yml +++ b/.github/workflows/verify.yml @@ -93,7 +93,7 @@ jobs: if: runner.os != 'Linux' run: npm run smoke:packaged - # The eval baseline (#75) is the reference point every v1.5 experiment is + # The eval baseline (#75) is the reference point every experiment is # measured against, so CI proves the committed numbers still reproduce. The # model cache is keyed on the pinned model file, so the 134 MB download # happens once per pin, not once per run. A hit is required for the run to @@ -110,6 +110,6 @@ jobs: run: | xvfb-run -a npm run eval:prepare xvfb-run -a node scripts/eval.mjs - cp docs/eval/baseline-v1.5.json /tmp/eval-a.json + cp docs/eval/baseline-v1.6.json /tmp/eval-a.json xvfb-run -a node scripts/eval.mjs - diff /tmp/eval-a.json docs/eval/baseline-v1.5.json + diff /tmp/eval-a.json docs/eval/baseline-v1.6.json diff --git a/docs/eval/baseline-v1.6.json b/docs/eval/baseline-v1.6.json new file mode 100644 index 0000000..eb15f14 --- /dev/null +++ b/docs/eval/baseline-v1.6.json @@ -0,0 +1,937 @@ +{ + "baseline": "v1.6", + "generatedBy": "npm run eval", + "config": { + "embedding": "Xenova/multilingual-e5-small@761b726dd34fb83930e26aab4e9ac3899aa1fa78 q8 (384d, local)", + "chunking": { + "chunkSize": 1000, + "chunkOverlap": 100, + "minChunkSize": 100, + "allowSpanPages": false, + "respectHeadings": false + }, + "retrieval": "dense", + "candidateK": 20, + "contextK": 3, + "threshold": 0.5, + "evidenceK": 5, + "corpus": "eval/corpus", + "documents": 13, + "questions": 30, + "chunkCount": 19 + }, + "metrics": { + "recallAt1": 0.833333, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 0.927778, + "ndcgAt10": 0.94375, + "evidencePrecisionAt5": 0.213333 + }, + "perQuestion": [ + { + "id": "q001", + "question": "Why is bedload harder to measure than suspended sediment?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q002", + "question": "How many replicate samples are collected at each river station?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q003", + "question": "What is the central trade-off in lithium-ion cell design?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q004", + "question": "Why do nickel-rich battery packs need more aggressive thermal management?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q005", + "question": "What happens once the separator in a battery cell melts?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q006", + "question": "At what temperature do honeybees begin to forage?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q007", + "question": "What does a late frost damage during full bloom?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q008", + "question": "Why is a continuous tree canopy more effective at cooling than isolated trees?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q009", + "question": "Why are trees with aggressive surface roots unsuitable for narrow verges?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q010", + "question": "At what temperature is lactic acid fermentation fastest?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q011", + "question": "Is the salt percentage in fermentation based on vegetable weight or water weight?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q012", + "question": "Why must tidal turbines be sited in places with very fast currents?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q013", + "question": "What is the main environmental concern for tidal energy installations?", + "firstRelevantRank": 3, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [], + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q014", + "question": "In lake monitoring, how is the sampling depth actually recorded?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q015", + "question": "Why does deep-water oxygen fall while a lake remains stratified?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q016", + "question": "How do supercapacitors hold their charge?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q017", + "question": "Why can solitary bees pollinate a bloom week that is too cold for honeybee hives?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q018", + "question": "Why is one continuous planted roof layer better than several isolated planted beds?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q019", + "question": "Where do acetic acid bacteria sit in a vinegar culture?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q020", + "question": "What happens if a vinegar culture is sealed airtight?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q021", + "question": "Why is wave energy harder to schedule ahead than tidal energy?", + "firstRelevantRank": 2, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q022", + "question": "Where does siting for wave energy devices concentrate, and where does it not?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q023", + "question": "How does the river sampling protocol differ from the lake sampling protocol?", + "firstRelevantRank": 1, + "relevantCount": 2, + "retrievedCount": 19, + "matchesByRank": [ + [ + 1 + ], + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q024", + "question": "A street canopy and a green roof are both said to cool; what surface does each one shade?", + "firstRelevantRank": 1, + "relevantCount": 2, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [ + 1 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q025", + "question": "Which preservation method depends on keeping air away from the food?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q026", + "question": "Why can one cold morning cost a grower the whole crop even when colonies are brought in?", + "firstRelevantRank": 2, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q027", + "question": "绿茶应该怎样保存才能减缓氧化?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q028", + "question": "茶叶储存的相对湿度上限是多少?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q029", + "question": "为什么冷冻保存的茶叶取出后不能立刻打开包装?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q030", + "question": "为什么潮汐能比风能和太阳能更容易提前安排发电?", + "firstRelevantRank": 2, + "relevantCount": 1, + "retrievedCount": 19, + "matchesByRank": [ + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + } + ] +} diff --git a/docs/eval/baseline-v1.6.md b/docs/eval/baseline-v1.6.md new file mode 100644 index 0000000..5418dd6 --- /dev/null +++ b/docs/eval/baseline-v1.6.md @@ -0,0 +1,65 @@ +# RAG eval baseline — v1.6 + +Generated by `npm run eval`. The numbers below are harness output — do not edit them by hand. + +## Configuration + +| Setting | Value | +| --- | --- | +| Embedding | `Xenova/multilingual-e5-small@761b726dd34fb83930e26aab4e9ac3899aa1fa78 q8 (384d, local)` | +| Chunking | `chunkSize=1000, chunkOverlap=100, minChunkSize=100, allowSpanPages=false, respectHeadings=false` | +| Retrieval | `dense` | +| Ranks | `candidateK=20, threshold=0.5` | +| Context width | `contextK=3` | +| Evidence per query | `evidenceK=5` | +| Corpus | `eval/corpus` (13 documents, 30 questions) | +| Index size | 19 chunks | + +## Metrics + +| Metric | Value | +| --- | --- | +| Recall@1 | 0.8333 | +| Recall@5 | 1.0000 | +| Recall@10 | 1.0000 | +| MRR | 0.9278 | +| nDCG@10 | 0.9437 | +| Evidence precision@5 | 0.2133 | + +Timing is informational only and is **not** frozen: indexing 1582 ms, query +p50 11.35 ms, p95 18.82 ms on the +machine that produced this file. Timing and index size depend on hardware and on the +corpus, so they must never be the reason two runs differ. + +## Definitions + +- A retrieved passage is relevant when its provenance covers a ground-truth block. +- **Recall@k** is the share of ground-truth blocks covered by the first `k` passages. +- **Evidence precision@5** is the share of the first `5` + retrieved passages that cover a ground-truth block. This is **retrieval precision**, not + answer citation recall: the harness runs no model and produces no answer. Answer-level + citation correctness is covered by the resolver (#70); a model-driven answer eval is a + separate deliverable. +- Ground truth is expressed in corpus identity (`document` relative path + `block` + ordinal + optional `quote`), never a runtime `documentId`/`blockId`. + +## Comparison protocol + +Experiments (#77, #78) are reported as a **delta against this file**. From v1.6 on, +the harness runs the **production configuration** by default: `candidateK` is the +first-stage width per channel, `contextK` is how many passages the chat prompt takes, +and `threshold` is the similarity floor the app ships. A benchmark that does not +mirror those parameters measures a retriever nobody runs. The adopted-change rule is: + +> Adopt a change only if Recall@5 improves and nDCG@10 does not regress. A change +> that trades a large latency increase for a marginal recall gain is a product +> decision, not an automatic win, and must be stated as such. + +A changed result must be reproducible with: + +```bash +npm run eval:prepare # one-time, networked model bootstrap +npm run eval # offline and deterministic +``` + +The raw report is committed next to this summary as `baseline-v1.6.json`. diff --git a/eval/README.md b/eval/README.md index 095b72a..191ac61 100644 --- a/eval/README.md +++ b/eval/README.md @@ -1,8 +1,8 @@ # RAG eval harness Measures retrieval quality so "did this change make retrieval better?" has an answer. -The current numbers are frozen in [`docs/eval/baseline-v1.4.md`](../docs/eval/baseline-v1.4.md); -every v1.5 experiment (#77, #78) is reported as a delta against that file. +The current numbers are frozen in [`docs/eval/baseline-v1.6.md`](../docs/eval/baseline-v1.6.md); +every experiment (#77, #78) is reported as a delta against that file. ## Commands @@ -11,6 +11,24 @@ npm run eval:prepare # one-time, networked: download the pinned embedding mode npm run eval # offline and deterministic: run the harness, rewrite the baseline ``` +### The harness runs the production configuration + +From v1.6 the harness defaults to the parameters the app ships, so its numbers describe +the product rather than a research setup. Two Ks, because they answer different questions: + +| Flag | Default | Meaning | +| --- | --- | --- | +| `--eval-candidate-k=` | `20` | first-stage width per channel (KNN neighbours, BM25 limit) | +| `--eval-context-k=` | `3` | passages the chat prompt actually takes (`chatHandlers.ts`) | +| `--eval-threshold=` | `0.5` | the similarity floor the app ships | +| `--eval-retrieval=` | `dense` | `dense`, `sparse`, or `hybrid` | +| `--eval-baseline=` | `v1.6` | name written into `docs/eval/baseline-.{json,md}` | + +Ranking metrics are computed at `candidateK` depth, not at `contextK`: `Recall@10` needs +at least ten results, and truncation only takes a prefix of the candidate list, so the +truncation cannot change the ranking it is measured on. `contextK` is recorded so the +report describes the whole online path. + `eval:prepare` downloads the pinned `multilingual-e5-small` revision into the app's model cache and verifies it. `eval` never touches the network: if the model is missing it stops with @@ -87,7 +105,9 @@ from unanswerable queries. 2. Indexes the corpus through the normal ingestion path (`addDocumentFromFile`), so blocks, chunking and embeddings are the real ones. 3. Maps each ground-truth `document`/`block` to the run's runtime ids. -4. Runs the real `Retriever` (`KnowledgeService.search` → `DenseRetriever`). +4. Runs the real `Retriever` (`KnowledgeService.search`) at the configured + `candidateK` depth; `RetrievalRequest` splits the first-stage width from the final + `topK` so the two are not silently the same number. 5. Writes `docs/eval/baseline-.json` (deterministic) and `.md` (with timing). ## Metrics diff --git a/src/main/eval/harness.ts b/src/main/eval/harness.ts index a992415..b23775c 100644 --- a/src/main/eval/harness.ts +++ b/src/main/eval/harness.ts @@ -47,8 +47,16 @@ export interface EvalHarnessOptions { corpusLabel: string questionsPath: string baseline: string - /** Ranks to compute Recall@k for. */ - topK: number + /** + * 第一阶段每个通道的宽度,也是排名指标的评估深度(#77)。 + * + * 与 `contextK` 分开:`candidateK` 决定“找了多宽”(召回),`contextK` 决定“交给 + * LLM 多少”(预算)。只用一个 K 时两者被绑死,benchmark 也无法在足够深的地方算 + * Recall@10。 + */ + candidateK: number + /** 生产 prompt 实际取用的条数(chat 里的 `topK`)。 */ + contextK: number /** Similarity floor; 0 keeps the ranking intact for ranking metrics. */ threshold: number /** How many retrieved passages the evidence-precision metric looks at. */ @@ -190,8 +198,14 @@ export async function runEvalHarness( const groundTruthIds = groundTruth.map((entry) => entry.blockId) const started = performance.now() + // 排名指标在完整的候选深度上计算,不先截到 `contextK`: + // + // - Recall@10 需要至少 10 条结果,而生产的 `contextK` 是 3; + // - 截断只取候选列表的前缀,前缀的排序与截断前一致,所以用宽的结果算排名不等于 + // 把两件事混在一个数里。 const results = await knowledgeService.search(NOTEBOOK_ID, question.question, { - topK: options.topK, + candidateK: options.candidateK, + topK: options.candidateK, threshold: options.threshold, strategy: options.strategy }) @@ -240,7 +254,8 @@ export async function runEvalHarness( respectHeadings: chunking.respectHeadings }, retrieval: options.strategy, - topK: options.topK, + candidateK: options.candidateK, + contextK: options.contextK, threshold: options.threshold, evidenceK: options.evidenceK, corpus: options.corpusLabel, diff --git a/src/main/eval/report.ts b/src/main/eval/report.ts index 1dfbd04..b596704 100644 --- a/src/main/eval/report.ts +++ b/src/main/eval/report.ts @@ -24,7 +24,8 @@ Generated by \`${report.generatedBy}\`. The numbers below are harness output — | Embedding | \`${config.embedding}\` | | Chunking | \`chunkSize=${chunking.chunkSize}, chunkOverlap=${chunking.chunkOverlap}, minChunkSize=${chunking.minChunkSize}, allowSpanPages=${chunking.allowSpanPages}, respectHeadings=${chunking.respectHeadings}\` | | Retrieval | \`${config.retrieval}\` | -| Ranks | \`topK=${config.topK}, threshold=${config.threshold}\` | +| Ranks | \`candidateK=${config.candidateK}, threshold=${config.threshold}\` | +| Context width | \`contextK=${config.contextK}\` | | Evidence per query | \`evidenceK=${config.evidenceK}\` | | Corpus | \`${config.corpus}\` (${config.documents} documents, ${config.questions} questions) | | Index size | ${config.chunkCount} chunks | @@ -59,8 +60,11 @@ corpus, so they must never be the reason two runs differ. ## Comparison protocol -v1.5 experiments (#77, #78) are reported as a **delta against this file**. The -adopted-change rule is: +Experiments (#77, #78) are reported as a **delta against this file**. From v1.6 on, +the harness runs the **production configuration** by default: \`candidateK\` is the +first-stage width per channel, \`contextK\` is how many passages the chat prompt takes, +and \`threshold\` is the similarity floor the app ships. A benchmark that does not +mirror those parameters measures a retriever nobody runs. The adopted-change rule is: > Adopt a change only if Recall@5 improves and nDCG@10 does not regress. A change > that trades a large latency increase for a marginal recall gain is a product diff --git a/src/main/eval/run.ts b/src/main/eval/run.ts index 58becb8..29085ad 100644 --- a/src/main/eval/run.ts +++ b/src/main/eval/run.ts @@ -178,9 +178,12 @@ export async function runEvalCli(argv: readonly string[] = process.argv): Promis // committed JSON is identical on every machine and checkout. corpusLabel: repoRelative(corpusDir) || 'eval/corpus', questionsPath, - baseline: readOption(argv, '--eval-baseline=', 'v1.5'), - topK: 10, - threshold: 0, + baseline: readOption(argv, '--eval-baseline=', 'v1.6'), + // 默认就是生产配置(#77):先取宽,融合,再把 contextK 条送进 prompt。一个不镜像 + // 线上参数的 benchmark 量的是用户永远不会跑的检索器。 + candidateK: readNumberOption(argv, '--eval-candidate-k=', 20), + contextK: readNumberOption(argv, '--eval-context-k=', 3), + threshold: readNumberOption(argv, '--eval-threshold=', 0.5), evidenceK: 5, chunkOptions: readChunkOptions(argv), strategy: readRetrievalStrategy(argv) diff --git a/src/main/eval/types.ts b/src/main/eval/types.ts index 9d3e4e4..c77fb90 100644 --- a/src/main/eval/types.ts +++ b/src/main/eval/types.ts @@ -75,7 +75,18 @@ export interface EvalReport { respectHeadings: boolean } retrieval: string - topK: number + /** + * 第一阶段每个通道的宽度(#77)。排名指标(Recall@K / MRR / nDCG@K)在这个深度上 + * 计算,所以它必须 ≥ 指标里最大的 K。 + */ + candidateK: number + /** + * 生产 prompt 实际取用的证据条数(#77)。 + * + * 快照里记它是为了让 benchmark 描述整条线上链路,而不只是检索器;它不影响排名 + * 指标 —— 截断只是取候选列表的前缀,前缀的排序不变。 + */ + contextK: number threshold: number /** How many retrieved passages the evidence-precision metric looks at. */ evidenceK: number diff --git a/src/main/ipc/chatHandlers.ts b/src/main/ipc/chatHandlers.ts index 6ba6393..00105e9 100644 --- a/src/main/ipc/chatHandlers.ts +++ b/src/main/ipc/chatHandlers.ts @@ -99,9 +99,14 @@ export function registerChatHandlers( // 直接走 `retrieve()` 而不是 `search()`:trace(#157)只有它有,用它再映射出 // SearchResult,不必为了记录参数多检索一次。 + // + // 生产配置就是这两个 K:先取宽(candidateK,每通道 20),融合后只把 3 条送进 + // prompt。这个组合被 eval harness 的默认运行原样量到 —— 否则 benchmark 测的是 + // 一个用户永远不会跑的检索器。 const { evidence, trace } = await knowledgeService.retrieve({ notebookId, query, + candidateK: 20, topK: 3, threshold: 0.5, filter: documentIds ? { documentIds } : undefined diff --git a/src/main/services/KnowledgeService.ts b/src/main/services/KnowledgeService.ts index 9165e2b..11272b1 100644 --- a/src/main/services/KnowledgeService.ts +++ b/src/main/services/KnowledgeService.ts @@ -95,6 +95,11 @@ export interface AddDocumentOptions { */ export interface SearchOptions { topK?: number // 返回结果数量,默认 5 + /** + * 第一阶段每个通道的宽度(#77)。缺省 `DEFAULT_CANDIDATE_K`,且不会小于 `topK`。 + * 搜索面板与 eval harness 用它把「找多宽」和「交多少」分开。 + */ + candidateK?: number threshold?: number // 相似度阈值,默认 0.5 includeContent?: boolean // 是否包含 chunk 内容,默认 true /** 只在这些来源里检索(#94);为空/缺省表示整个 notebook。 */ @@ -1237,6 +1242,7 @@ export class KnowledgeService { const { evidence } = await this.retrieve({ notebookId, query, + candidateK: options.candidateK, topK: options.topK, threshold: options.threshold, strategy: options.strategy, diff --git a/src/main/services/retrieval/DenseRetriever.ts b/src/main/services/retrieval/DenseRetriever.ts index a6187a7..95fece4 100644 --- a/src/main/services/retrieval/DenseRetriever.ts +++ b/src/main/services/retrieval/DenseRetriever.ts @@ -12,6 +12,7 @@ import type { EmbeddingService } from '../EmbeddingService' import type { CandidateHit } from './candidates' import { hydrateEvidence } from './evidence' import { buildRetrievalTrace } from './trace' +import { effectiveCandidateK, DEFAULT_TOP_K } from './types' import type { RetrievalRequest, RetrievalResult, Retriever } from './types' const STRATEGY = 'dense' @@ -27,7 +28,9 @@ export class DenseRetriever implements Retriever { * embed + KNN 抄一遍。 */ async candidateHits(request: RetrievalRequest): Promise { - const topK = request.topK ?? 5 + // 第一阶段按 `candidateK` 取宽;`topK` 的截断由调用方决定,因为 hybrid 需要的是 + // 比最终交付更宽的一池子候选。 + const candidateK = effectiveCandidateK(request) const threshold = request.threshold ?? 0.5 // E5 要求 query 前缀,与索引时的 document 前缀区分 @@ -35,9 +38,9 @@ export class DenseRetriever implements Retriever { const queryEmbedding = await this.embeddingService.embed(request.query, 'query') const vectorStore = await vectorStoreManager.getStore(request.notebookId) - // scope 过滤由向量库在 KNN 之前执行(#94),不是取回 topK 之后再筛。 + // scope 过滤由向量库在 KNN 之前执行(#94),不是取回 candidateK 之后再筛。 const hits = await vectorStore.query(queryEmbedding.embedding, { - topK, + topK: candidateK, threshold, filter: request.filter }) @@ -46,11 +49,14 @@ export class DenseRetriever implements Retriever { } async search(request: RetrievalRequest): Promise { - const topK = request.topK ?? 5 + const topK = request.topK ?? DEFAULT_TOP_K + const candidateK = effectiveCandidateK(request) const threshold = request.threshold ?? 0.5 const startedAt = performance.now() - const hits = await this.candidateHits(request) + // 单策略没有可精排的下游,取宽再截到 `topK` 与直接按 `topK` 查 KNN 等价; + // 两个 K 分开是为了让契约统一,而不是在这里制造差异。 + const hits = (await this.candidateHits(request)).slice(0, topK) const evidence = hits.length === 0 ? [] : hydrateEvidence(getDatabase(), hits) return { @@ -58,6 +64,7 @@ export class DenseRetriever implements Retriever { trace: buildRetrievalTrace({ strategy: STRATEGY, filter: request.filter, + candidateK, topK, threshold, durationMs: performance.now() - startedAt diff --git a/src/main/services/retrieval/HybridRetriever.ts b/src/main/services/retrieval/HybridRetriever.ts index bb899f6..d0a78b9 100644 --- a/src/main/services/retrieval/HybridRetriever.ts +++ b/src/main/services/retrieval/HybridRetriever.ts @@ -5,7 +5,14 @@ import { rrfFuse, type CandidateHit } from './candidates' import { DenseRetriever } from './DenseRetriever' import { hydrateEvidence } from './evidence' import { buildRetrievalTrace } from './trace' -import type { RetrievalRequest, RetrievalResult, RetrievalStrategy, Retriever } from './types' +import { + effectiveCandidateK, + DEFAULT_TOP_K, + type RetrievalRequest, + type RetrievalResult, + type RetrievalStrategy, + type Retriever +} from './types' /** * HybridRetriever (#77) @@ -18,6 +25,10 @@ import type { RetrievalRequest, RetrievalResult, RetrievalStrategy, Retriever } * * 融合在候选层完成(`chunkId + rank`),证据只补齐一次 —— 见 `candidates.ts`。 * 两个通道的分数(cosine 与 BM25)不可比,所以只用 rank,这正是 RRF 的意义。 + * + * 两个 K 是两个阶段:每个通道先按 `candidateK` 取宽(融合池最多 `2 * candidateK`), + * 融合后再截到 `topK`。如果两个通道都只取 `topK`,融合池最多 `2 * topK` 且结果被截回 + * `topK`,融合几乎没有发生空间 —— 那就不是混合检索,只是一个更慢的单路检索。 */ export class HybridRetriever implements Retriever { private readonly dense: DenseRetriever @@ -28,7 +39,8 @@ export class HybridRetriever implements Retriever { async search(request: RetrievalRequest): Promise { const strategy: RetrievalStrategy = request.strategy ?? 'dense' - const topK = request.topK ?? 5 + const candidateK = effectiveCandidateK(request) + const topK = request.topK ?? DEFAULT_TOP_K const startedAt = performance.now() let hits: CandidateHit[] @@ -38,19 +50,19 @@ export class HybridRetriever implements Retriever { if (strategy === 'sparse') { hits = searchChunksFts(request.notebookId, request.query, { - limit: topK, + limit: candidateK, documentIds: request.filter?.documentIds - }) + }).slice(0, topK) } else if (strategy === 'hybrid') { const denseHits = await this.dense.candidateHits(request) const sparseHits = searchChunksFts(request.notebookId, request.query, { - limit: topK, + limit: candidateK, documentIds: request.filter?.documentIds }) hits = rrfFuse([denseHits, sparseHits]).slice(0, topK) } else { threshold = request.threshold ?? 0.5 - hits = await this.dense.candidateHits({ ...request, threshold }) + hits = (await this.dense.candidateHits({ ...request, threshold })).slice(0, topK) } const evidence = hits.length === 0 ? [] : hydrateEvidence(getDatabase(), hits) @@ -60,6 +72,7 @@ export class HybridRetriever implements Retriever { trace: buildRetrievalTrace({ strategy, filter: request.filter, + candidateK, topK, threshold, durationMs: performance.now() - startedAt diff --git a/src/main/services/retrieval/trace.ts b/src/main/services/retrieval/trace.ts index 15243ac..ae788cc 100644 --- a/src/main/services/retrieval/trace.ts +++ b/src/main/services/retrieval/trace.ts @@ -3,6 +3,7 @@ import type { RetrievalFilter, RetrievalTrace } from './types' export interface RetrievalTraceInput { strategy: string filter?: RetrievalFilter + candidateK: number topK: number threshold?: number durationMs: number @@ -24,6 +25,7 @@ export function buildRetrievalTrace(input: RetrievalTraceInput): RetrievalTrace const trace: RetrievalTrace = { strategy: input.strategy, scope, + candidateK: input.candidateK, topK: input.topK, durationMs: input.durationMs } diff --git a/src/main/services/retrieval/types.ts b/src/main/services/retrieval/types.ts index 060134b..921a345 100644 --- a/src/main/services/retrieval/types.ts +++ b/src/main/services/retrieval/types.ts @@ -63,10 +63,26 @@ export type RetrievalStrategy = 'dense' | 'sparse' | 'hybrid' * * 取代旧的 `(notebookId, query, options)` 位置参数:#94 的 scope、#77 的策略参数 * 与 #157 要快照的 trace 都要挂在这一个对象上,而不是散落在调用点。 + * + * 两个 K 是分开的,它们回答的是不同的问题: + * + * candidateK 第一阶段每个通道取多宽(KNN 邻居数 / BM25 limit)—— 偏向召回 + * topK 最终交付多少条证据(chat 里就是送进 prompt 的 context 宽度)—— 偏向精度 + * + * 合并成一个 K 会让「向量库直接搜几条」冒充两阶段检索:hybrid 时两个通道各取 + * `topK` 条,融合池最多只有 `2 * topK`,再截回 `topK`,融合几乎没有发生空间。 */ export interface RetrievalRequest { notebookId: string query: string + /** + * 第一阶段候选宽度。缺省 `DEFAULT_CANDIDATE_K`。 + * + * 实际生效值不会小于 `topK`(见 `effectiveCandidateK`):一个 `topK=50` 的调用方 + * 不该因为没写 `candidateK` 而只拿到 20 条。 + */ + candidateK?: number + /** 最终证据条数。缺省 `DEFAULT_TOP_K`。 */ topK?: number threshold?: number filter?: RetrievalFilter @@ -74,16 +90,42 @@ export interface RetrievalRequest { strategy?: RetrievalStrategy } +/** 没有显式指定时的第一阶最宽度(#77)。与 chat 的生产配置保持一致。 */ +export const DEFAULT_CANDIDATE_K = 20 + +/** 没有显式指定时的最终证据条数,与引入 `candidateK` 之前一致。 */ +export const DEFAULT_TOP_K = 5 + +/** + * 第一阶段的真实宽度:`candidateK`,但不小于 `topK`。 + * + * 没有这条不变式,「取出比交付更宽的一池子」只是多数时候成立:任何 `topK > candidateK` + * 的调用(搜索面板的 limit、MCP 的 topK)都会静默地少返结果。 + */ +export function effectiveCandidateK(request: { + candidateK?: number + topK?: number +}): number { + const topK = request.topK ?? DEFAULT_TOP_K + return Math.max(request.candidateK ?? DEFAULT_CANDIDATE_K, topK) +} + /** * 一次检索实际生效的参数。 * * 它会随回答一起被快照(#157),因此必须由检索层产出、而不是调用方猜: * `strategy` 说明用的是哪条检索路径,`scope` 说明结果被限制在哪些来源。 * `threshold` 缺省表示该策略没有阈值,不是「阈值等于 0」。 + * + * `candidateK` 与 `topK` 是两个不同的量,快照里都必须有:只看 `topK` 无法解释 + * 「为什么这次只召回三条」,也无法复现 hybrid 的融合池有多宽。 */ export interface RetrievalTrace { strategy: string scope: { documentIds?: string[] } + /** 第一阶段每个通道的实际宽度(已应用 `effectiveCandidateK`)。 */ + candidateK: number + /** 最终交付的证据条数。 */ topK: number threshold?: number durationMs: number diff --git a/src/shared/types/chat.ts b/src/shared/types/chat.ts index a358ab9..55e7170 100644 --- a/src/shared/types/chat.ts +++ b/src/shared/types/chat.ts @@ -154,6 +154,13 @@ export interface RetrievalSnapshot { strategy: string /** 空对象 = 整个 notebook。 */ scope: { documentIds?: string[] } + /** + * 第一阶段每个通道的宽度(#77)。 + * + * 引入两个 K 之前的快照没有这个字段:那时候选宽度就是 `topK`,所以解析时按 + * `topK` 回填——那是历史事实,不是拿默认值冒充。 + */ + candidateK: number topK: number /** 缺省表示该策略没有阈值,不是「阈值等于 0」。 */ threshold?: number @@ -172,7 +179,7 @@ export interface RetrievalSnapshot { export interface ChatMessageMetadata { sources?: AnswerSource[] retrieval?: RetrievalStatus - /** 本次检索用了什么参数(#157):策略、生效范围、topK、耗时。 */ + /** 本次检索用了什么参数(#157):策略、生效范围、候选宽度、交付条数、耗时。 */ retrievalSnapshot?: RetrievalSnapshot /** * The structured provenance of this answer (#69). `sources` says *what* the diff --git a/src/shared/utils/answerSources.ts b/src/shared/utils/answerSources.ts index 41b51d3..ffb7271 100644 --- a/src/shared/utils/answerSources.ts +++ b/src/shared/utils/answerSources.ts @@ -93,9 +93,14 @@ export const parseRetrievalSnapshot = (metadata: unknown): RetrievalSnapshot | n const durationMs = toFiniteNumber(candidate.durationMs) if (topK === undefined || durationMs === undefined) return null + // Snapshots written before #77 had a single K, so the candidate width *was* topK. + // Backfilling it states what the old retrieval actually did; it is not a guess. + const candidateK = toFiniteNumber(candidate.candidateK) ?? topK + const snapshot: RetrievalSnapshot = { strategy: candidate.strategy, scope: {}, + candidateK, topK, durationMs } diff --git a/test/answerSources.test.ts b/test/answerSources.test.ts index 29e23eb..5de2726 100644 --- a/test/answerSources.test.ts +++ b/test/answerSources.test.ts @@ -151,6 +151,7 @@ test('a retrieval snapshot round-trips', () => { retrievalSnapshot: { strategy: 'dense', scope: { documentIds: ['doc_1', 'doc_2'] }, + candidateK: 20, topK: 8, threshold: 0.5, durationMs: 42.4 @@ -160,6 +161,7 @@ test('a retrieval snapshot round-trips', () => { assert.deepEqual(snapshot, { strategy: 'dense', scope: { documentIds: ['doc_1', 'doc_2'] }, + candidateK: 20, topK: 8, threshold: 0.5, durationMs: 42.4 @@ -175,6 +177,19 @@ test('a snapshot without a scope means the whole notebook', () => { assert.equal(snapshot?.threshold, undefined) }) +/** + * Before #77 there was one K, so the first-stage width *was* topK. Backfilling it + * says what that retrieval did; it is not an invented default. + */ +test('a snapshot written before the two Ks backfills candidateK from topK', () => { + const snapshot = parseRetrievalSnapshot({ + retrievalSnapshot: { strategy: 'dense', scope: {}, topK: 5, durationMs: 1 } + }) + + assert.equal(snapshot?.candidateK, 5) + assert.equal(snapshot?.topK, 5) +}) + test('a snapshot that cannot be rendered is dropped, not guessed', () => { for (const bad of [ undefined, diff --git a/test/retrievalContract.test.ts b/test/retrievalContract.test.ts index 4cfd2a9..9d32292 100644 --- a/test/retrievalContract.test.ts +++ b/test/retrievalContract.test.ts @@ -1,6 +1,10 @@ import { test } from 'node:test' import assert from 'node:assert/strict' import { buildRetrievalTrace } from '../src/main/services/retrieval/trace.ts' +import { + DEFAULT_CANDIDATE_K, + effectiveCandidateK +} from '../src/main/services/retrieval/types.ts' /** * The retrieval contract (#160) is the seam #94, #157 and #77 build on. These pin @@ -14,6 +18,7 @@ import { buildRetrievalTrace } from '../src/main/services/retrieval/trace.ts' test('a trace carries the effective search parameters and no empty fields', () => { const trace = buildRetrievalTrace({ strategy: 'dense', + candidateK: 20, topK: 5, threshold: 0.5, durationMs: 12.5 @@ -22,21 +27,50 @@ test('a trace carries the effective search parameters and no empty fields', () = assert.deepEqual(trace, { strategy: 'dense', scope: {}, + candidateK: 20, topK: 5, threshold: 0.5, durationMs: 12.5 }) // An unset threshold means "no threshold", not "threshold 0". assert.equal( - 'threshold' in buildRetrievalTrace({ strategy: 'dense', topK: 5, durationMs: 1 }), + 'threshold' in buildRetrievalTrace({ strategy: 'dense', candidateK: 20, topK: 5, durationMs: 1 }), false ) }) +test('the trace keeps the first-stage width and the final count apart', () => { + const trace = buildRetrievalTrace({ + strategy: 'hybrid', + candidateK: 20, + topK: 3, + durationMs: 9 + }) + + // #77: a snapshot that only says `topK` cannot explain the fused candidate pool. + assert.equal(trace.candidateK, 20) + assert.equal(trace.topK, 3) +}) + +/** + * #77: the first stage takes a candidate pool, the delivery stage takes `topK`. + * The invariant that matters is that the pool is never narrower than what is + * supposed to come out of it. + */ +test('the first-stage width is never narrower than the final count', () => { + assert.equal(effectiveCandidateK({}), DEFAULT_CANDIDATE_K) + assert.equal(effectiveCandidateK({ topK: 3 }), DEFAULT_CANDIDATE_K) + assert.equal(effectiveCandidateK({ candidateK: 40, topK: 3 }), 40) + // A caller that asks for more results than the default pool still gets them: the + // search palette and MCP pass their own `topK` and must not be silently capped. + assert.equal(effectiveCandidateK({ topK: 50 }), 50) +}) + test('a filter becomes the recorded scope of the trace', () => { const trace = buildRetrievalTrace({ strategy: 'dense', filter: { documentIds: ['doc_a', 'doc_b'] }, + candidateK: 20, topK: 8, durationMs: 3 })