diff --git a/.prettierignore b/.prettierignore index 1a115ff..ae11386 100644 --- a/.prettierignore +++ b/.prettierignore @@ -19,6 +19,11 @@ docs/eval/baseline-*.md docs/eval/chunking-*.json docs/eval/chunking-*.md +# Same reason again: generated by `scripts/eval-threshold.mjs`. Regenerate with +# `npm run eval:threshold`. +docs/eval/threshold-*.json +docs/eval/threshold-*.md + # And the same again for `scripts/eval-retrieval.mjs` (#77). docs/eval/retrieval-*.json docs/eval/retrieval-*.md diff --git a/docs/eval/baseline-v1.6.json b/docs/eval/baseline-v1.6.json index ca5f93b..cfb0455 100644 --- a/docs/eval/baseline-v1.6.json +++ b/docs/eval/baseline-v1.6.json @@ -11,6 +11,7 @@ "respectHeadings": false }, "retrieval": "dense", + "split": "all", "candidateK": 20, "contextK": 3, "threshold": 0.5, diff --git a/docs/eval/baseline-v1.6.md b/docs/eval/baseline-v1.6.md index df1c22c..605c29a 100644 --- a/docs/eval/baseline-v1.6.md +++ b/docs/eval/baseline-v1.6.md @@ -12,7 +12,8 @@ Generated by `npm run eval`. The numbers below are harness output — do not edi | Ranks | `candidateK=20, threshold=0.5` | | Context width | `contextK=3` | | Evidence per query | `evidenceK=5` | -| Corpus | `eval/corpus` (13 documents, 30 questions) | +| Corpus | `eval/corpus` (13 documents) | +| Split | `all` (30 questions) | | Index size | 19 chunks | ## Metrics @@ -42,8 +43,8 @@ The type comes from `type` in `questions.jsonl`; untagged questions report as | semantic | 18 | 1.0000 | 0.9312 | 1.0000 | 0.9074 | | zh | 3 | 1.0000 | 1.0000 | 1.0000 | 1.0000 | -Timing is informational only and is **not** frozen: indexing 1545 ms, query -p50 11.04 ms, p95 14.14 ms on the +Timing is informational only and is **not** frozen: indexing 1542 ms, query +p50 15.01 ms, p95 69.16 ms on the machine that produced this file. Timing and index size depend on hardware and on the corpus, so they must never be the reason two runs differ. diff --git a/docs/eval/threshold-v1.6.json b/docs/eval/threshold-v1.6.json new file mode 100644 index 0000000..636e5b9 --- /dev/null +++ b/docs/eval/threshold-v1.6.json @@ -0,0 +1,183 @@ +{ + "baseline": "v1.6", + "productionThreshold": 0.5, + "flat": true, + "recommended": 0.5, + "rows": [ + { + "threshold": 0, + "validation": { + "questions": 10, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 19, + "metrics": { + "recallAt1": 0.9, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 0.933333, + "ndcgAt10": 0.95, + "hitRateAt5": 1, + "mapAt10": 0.933333, + "evidencePrecisionAt5": 0.2 + } + }, + "test": { + "questions": 20, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 19, + "metrics": { + "recallAt1": 0.8, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 0.925, + "ndcgAt10": 0.940626, + "hitRateAt5": 1, + "mapAt10": 0.916667, + "evidencePrecisionAt5": 0.22 + } + } + }, + { + "threshold": 0.3, + "validation": { + "questions": 10, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 19, + "metrics": { + "recallAt1": 0.9, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 0.933333, + "ndcgAt10": 0.95, + "hitRateAt5": 1, + "mapAt10": 0.933333, + "evidencePrecisionAt5": 0.2 + } + }, + "test": { + "questions": 20, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 19, + "metrics": { + "recallAt1": 0.8, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 0.925, + "ndcgAt10": 0.940626, + "hitRateAt5": 1, + "mapAt10": 0.916667, + "evidencePrecisionAt5": 0.22 + } + } + }, + { + "threshold": 0.4, + "validation": { + "questions": 10, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 19, + "metrics": { + "recallAt1": 0.9, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 0.933333, + "ndcgAt10": 0.95, + "hitRateAt5": 1, + "mapAt10": 0.933333, + "evidencePrecisionAt5": 0.2 + } + }, + "test": { + "questions": 20, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 19, + "metrics": { + "recallAt1": 0.8, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 0.925, + "ndcgAt10": 0.940626, + "hitRateAt5": 1, + "mapAt10": 0.916667, + "evidencePrecisionAt5": 0.22 + } + } + }, + { + "threshold": 0.5, + "validation": { + "questions": 10, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 19, + "metrics": { + "recallAt1": 0.9, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 0.933333, + "ndcgAt10": 0.95, + "hitRateAt5": 1, + "mapAt10": 0.933333, + "evidencePrecisionAt5": 0.2 + } + }, + "test": { + "questions": 20, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 19, + "metrics": { + "recallAt1": 0.8, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 0.925, + "ndcgAt10": 0.940626, + "hitRateAt5": 1, + "mapAt10": 0.916667, + "evidencePrecisionAt5": 0.22 + } + } + }, + { + "threshold": 0.6, + "validation": { + "questions": 10, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 19, + "metrics": { + "recallAt1": 0.9, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 0.933333, + "ndcgAt10": 0.95, + "hitRateAt5": 1, + "mapAt10": 0.933333, + "evidencePrecisionAt5": 0.2 + } + }, + "test": { + "questions": 20, + "noResultCount": 0, + "noResultRate": 0, + "meanRetrieved": 19, + "metrics": { + "recallAt1": 0.8, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 0.925, + "ndcgAt10": 0.940626, + "hitRateAt5": 1, + "mapAt10": 0.916667, + "evidencePrecisionAt5": 0.22 + } + } + } + ] +} diff --git a/docs/eval/threshold-v1.6.md b/docs/eval/threshold-v1.6.md new file mode 100644 index 0000000..c9f5c1f --- /dev/null +++ b/docs/eval/threshold-v1.6.md @@ -0,0 +1,51 @@ +# Threshold derivation — v1.6 (#192) + +Generated by `node scripts/eval-threshold.mjs`. Numbers are harness output; do not edit them by hand. + +## What was measured + +The real harness, the same corpus and the production retrieval config +(`candidateK=20, contextK=3`), once per candidate threshold. The **validation** +split selects; the **test** split reports. Both come from the same +`--eval-split=` code path, and the split is a deterministic function of the +question id, so this is reproducible. + +| Threshold | n (val) | Recall@5 (val) | nDCG@10 (val) | MAP@10 (val) | No-result (val) | nDCG@10 (test) | No-result (test) | +| --- | --- | --- | --- | --- | --- | --- | --- | +| 0 | 10 | 1.0000 | 0.9500 | 0.9333 | 0.0000 | 0.9406 | 0.0000 | +| 0.3 | 10 | 1.0000 | 0.9500 | 0.9333 | 0.0000 | 0.9406 | 0.0000 | +| 0.4 | 10 | 1.0000 | 0.9500 | 0.9333 | 0.0000 | 0.9406 | 0.0000 | +| 0.5 | 10 | 1.0000 | 0.9500 | 0.9333 | 0.0000 | 0.9406 | 0.0000 | +| 0.6 | 10 | 1.0000 | 0.9500 | 0.9333 | 0.0000 | 0.9406 | 0.0000 | + +## Selection rule + +Best validation nDCG@10, then fewest validation no-results, then the **lowest** +threshold — the first stage is supposed to favour recall and let a later stage +filter, so among equals the wider one is the safer default. + +## Outcome + +The sweep is **flat**: every threshold from 0 to 0.6 produces the same validation nDCG@10 (0.9500), the same Recall@5 (1.0000) and a no-result rate of 0.0000. On this corpus the threshold is simply **non-binding** — E5 never scores these query/chunk pairs below the top of the swept range, so no passage is ever filtered out. + +**No evidence to change `threshold = 0.5`.** The tie-break rule nominates `0` only because it prefers the widest threshold among equals; that is a tie-break, not a finding. What this run establishes is that the current value cannot be validated *or* falsified here, which is a property of the corpus, not of the threshold. Re-run after #192 child 2 grows it. + +The app currently ships `threshold = 0.5`: validation nDCG@10 +0.9500, test nDCG@10 +0.9406, test no-result rate +0.0000. + +## Caveat on this corpus + +The split removes the most obvious form of overfitting, but 10 +validation questions is a thin basis for a decision, and the corpus is still small. A +threshold is a product decision with a **no-result-rate** cost attached, so a +recommendation here is only as good as the corpus behind it. Re-run this after the +corpus grows (#192 child 2). + +## Reproduce + +```bash +npm run eval:prepare # one-time, networked model bootstrap +npm run eval:threshold # offline; rewrites this file +``` diff --git a/eval/README.md b/eval/README.md index c8964f1..2333a26 100644 --- a/eval/README.md +++ b/eval/README.md @@ -7,8 +7,10 @@ every experiment (#77, #78) is reported as a delta against that file. ## Commands ```bash -npm run eval:prepare # one-time, networked: download the pinned embedding model -npm run eval # offline and deterministic: run the harness, rewrite the baseline +npm run eval:prepare # one-time, networked: download the pinned embedding model +npm run eval # offline and deterministic: run the harness, rewrite the baseline +npm run eval:retrieval # strategy comparison (#77) +npm run eval:threshold # derive the similarity threshold on validation, report on test ``` ### The harness runs the production configuration @@ -22,6 +24,7 @@ the product rather than a research setup. Two Ks, because they answer different | `--eval-context-k=` | `3` | passages the chat prompt actually takes (`chatHandlers.ts`) | | `--eval-threshold=` | `0.5` | the similarity floor the app ships | | `--eval-retrieval=` | `dense` | `dense`, `sparse`, or `hybrid` | +| `--eval-split=` | `all` | `all`, `validation`, or `test` — a deterministic id-based split | | `--eval-baseline=` | `v1.6` | name written into `docs/eval/baseline-.{json,md}` | Ranking metrics are computed at `candidateK` depth, not at `contextK`: `Recall@10` needs @@ -29,6 +32,13 @@ at least ten results, and truncation only takes a prefix of the candidate list, truncation cannot change the ranking it is measured on. `contextK` is recorded so the report describes the whole online path. +### Swept parameters are chosen on `validation`, reported on `test` + +A parameter picked on the same questions it is scored on is a fitted number, not a +result. `--eval-split=validation` selects roughly a third of the questions by a +deterministic hash of the id; `test` is the rest. `npm run eval:threshold` uses this +to pick a similarity threshold on `validation` and report it on `test`. + `eval:prepare` downloads the pinned `multilingual-e5-small` revision into the app's model cache and verifies it. `eval` never touches the network: if the model is missing it stops with diff --git a/package.json b/package.json index 7bcf86c..b93f550 100644 --- a/package.json +++ b/package.json @@ -41,7 +41,8 @@ "db:migrate": "drizzle-kit migrate", "db:push": "drizzle-kit push", "db:studio": "drizzle-kit studio", - "eval:retrieval": "npm run build && node scripts/eval-retrieval.mjs" + "eval:retrieval": "npm run build && node scripts/eval-retrieval.mjs", + "eval:threshold": "npm run build && node scripts/eval-threshold.mjs" }, "//test": [ "`node --test` strips TypeScript types rather than compiling them, and strip-only", diff --git a/scripts/eval-threshold.mjs b/scripts/eval-threshold.mjs new file mode 100644 index 0000000..b6d1c73 --- /dev/null +++ b/scripts/eval-threshold.mjs @@ -0,0 +1,233 @@ +#!/usr/bin/env node +/** + * Threshold derivation for #192 (child 4). + * + * `threshold: 0.5` was hand-picked, and a cosine score has no universal meaning: + * the distribution depends on the embedding model, the language, the query type and + * the chunk length. This runs the real harness once per candidate threshold on a + * **validation** split, picks a winner there, and then reports that winner on the + * **test** split — so the number that justifies the choice is not the number the + * choice was fitted to. + * + * The harness owns the split (`--eval-split=`), so both sides are measured by the + * same code path that produces the frozen baseline. + * + * Usage: + * node scripts/eval-threshold.mjs + * + * The embedding model must already be prepared (`npm run eval:prepare`). + */ + +import { spawn } from 'node:child_process' +import { existsSync, mkdtempSync, mkdirSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join, resolve } from 'node:path' + +const OUT_MD = resolve(readArg('--out=', 'docs/eval/threshold-v1.6.md')) +const OUT_JSON = OUT_MD.replace(/\.md$/, '.json') + +/** + * 0 is the "no floor" arm: it keeps the ranking intact and lets a downstream stage + * filter. It is here because #192 says the first stage may legitimately run with no + * threshold at all. + */ +const THRESHOLDS = [0, 0.3, 0.4, 0.5, 0.6] + +/** The threshold the app currently ships, so the report can say whether it holds up. */ +const PRODUCTION_THRESHOLD = 0.5 + +function readArg(prefix, fallback) { + const arg = process.argv.find((value) => value.startsWith(prefix)) + return arg ? arg.slice(prefix.length) : fallback +} + +// Node 24 refuses to spawn a `.cmd`/`.bat` without `shell: true` (EINVAL), and the +// `.bin` entry is exactly that on Windows. Use the real binary the wrapper runs. +const { default: electronBinary } = await import('electron') +const executable = resolve(electronBinary) + +if (!existsSync(executable)) { + console.error('[threshold] could not find the electron binary. Run `npm install` first.') + process.exit(1) +} + +function runOne(threshold, split, outDir) { + return new Promise((resolvePromise, reject) => { + const args = [ + '.', + '--eval-harness', + '--eval-baseline=v1.6', + `--eval-out=${outDir}`, + `--eval-split=${split}`, + `--eval-threshold=${threshold}` + ] + + const isRoot = typeof process.getuid === 'function' && process.getuid() === 0 + if (isRoot || process.env.CI) args.push('--no-sandbox') + + const child = spawn(executable, args, { + stdio: ['ignore', 'pipe', 'pipe'], + env: { ...process.env, ELECTRON_DISABLE_SECURITY_WARNINGS: '1' } + }) + + child.on('error', reject) + child.on('exit', (code) => { + if (code !== 0) { + reject(new Error(`threshold ${threshold} (${split}) exited with code ${code}`)) + return + } + const reportPath = join(outDir, 'baseline-v1.6.json') + if (!existsSync(reportPath)) { + reject(new Error(`threshold ${threshold} (${split}) wrote no report`)) + return + } + const report = JSON.parse(readFileSync(reportPath, 'utf8')) + resolvePromise(summarize(report)) + }) + }) +} + +/** + * The frozen metrics plus the one this experiment needs and the baseline does not + * carry: how often a threshold turns a question into "no results at all". + * + * A higher threshold can look better on ranking metrics while quietly making the + * product answer "not in your sources" more often, and that trade is invisible + * unless it is counted. + */ +function summarize(report) { + const perQuestion = report.perQuestion ?? [] + const noResult = perQuestion.filter((q) => q.retrievedCount === 0).length + const retrieved = perQuestion.map((q) => q.retrievedCount) + return { + questions: perQuestion.length, + noResultCount: noResult, + noResultRate: perQuestion.length === 0 ? 0 : noResult / perQuestion.length, + meanRetrieved: retrieved.length === 0 ? 0 : retrieved.reduce((a, b) => a + b, 0) / retrieved.length, + metrics: report.metrics + } +} + +const format4 = (value) => value.toFixed(4) + +const workDir = mkdtempSync(join(tmpdir(), 'knownote-threshold-')) +const rows = [] + +try { + for (const threshold of THRESHOLDS) { + const validationDir = join(workDir, `${threshold}-validation`) + const testDir = join(workDir, `${threshold}-test`) + mkdirSync(validationDir, { recursive: true }) + mkdirSync(testDir, { recursive: true }) + + console.log(`[threshold] threshold ${threshold}: validation`) + const validation = await runOne(threshold, 'validation', validationDir) + console.log(`[threshold] threshold ${threshold}: test`) + const test = await runOne(threshold, 'test', testDir) + + rows.push({ threshold, validation, test }) + } +} finally { + rmSync(workDir, { recursive: true, force: true }) +} + +/** + * Selection rule, stated so it can be argued with: best validation nDCG@10, then + * fewest validation no-results, then the widest (lowest) threshold — because the + * first stage is supposed to favour recall and let a later stage filter. + */ +const ranked = [...rows].sort( + (a, b) => + b.validation.metrics.ndcgAt10 - a.validation.metrics.ndcgAt10 || + a.validation.noResultRate - b.validation.noResultRate || + a.threshold - b.threshold +) +const winner = ranked[0] +const production = rows.find((row) => row.threshold === PRODUCTION_THRESHOLD) + +/** + * A flat sweep is not a weak recommendation, it is no recommendation: if every + * threshold scores the same on both metrics, the corpus cannot tell them apart and + * moving a product parameter on that evidence would be noise dressed as a result. + */ +const flat = + rows.every( + (row) => + row.validation.metrics.ndcgAt10 === rows[0].validation.metrics.ndcgAt10 && + row.validation.noResultRate === rows[0].validation.noResultRate + ) + +const outcome = flat + ? `The sweep is **flat**: every threshold from ${THRESHOLDS[0]} to ${THRESHOLDS[THRESHOLDS.length - 1]} produces the same validation nDCG@10 (${format4(rows[0].validation.metrics.ndcgAt10)}), the same Recall@5 (${format4(rows[0].validation.metrics.recallAt5)}) and a no-result rate of ${format4(rows[0].validation.noResultRate)}. On this corpus the threshold is simply **non-binding** — E5 never scores these query/chunk pairs below the top of the swept range, so no passage is ever filtered out.\n\n**No evidence to change \`threshold = ${PRODUCTION_THRESHOLD}\`.** The tie-break rule nominates \`${winner.threshold}\` only because it prefers the widest threshold among equals; that is a tie-break, not a finding. What this run establishes is that the current value cannot be validated *or* falsified here, which is a property of the corpus, not of the threshold. Re-run after #192 child 2 grows it.` + : `**Recommended: \`threshold = ${winner.threshold}\`.**\n\n- Validation: nDCG@10 ${format4(winner.validation.metrics.ndcgAt10)}, Recall@5 ${format4(winner.validation.metrics.recallAt5)}, no-result rate ${format4(winner.validation.noResultRate)} (${winner.validation.noResultCount}/${winner.validation.questions})\n- Test: nDCG@10 ${format4(winner.test.metrics.ndcgAt10)}, Recall@5 ${format4(winner.test.metrics.recallAt5)}, no-result rate ${format4(winner.test.noResultRate)} (${winner.test.noResultCount}/${winner.test.questions})\n- Mean retrieved per question: ${winner.test.meanRetrieved.toFixed(2)} (validation ${winner.validation.meanRetrieved.toFixed(2)})` + +const tableRows = rows + .map( + (row) => + `| ${row.threshold} | ${row.validation.questions} | ${format4(row.validation.metrics.recallAt5)} | ` + + `${format4(row.validation.metrics.ndcgAt10)} | ${format4(row.validation.metrics.mapAt10)} | ` + + `${format4(row.validation.noResultRate)} | ${format4(row.test.metrics.ndcgAt10)} | ` + + `${format4(row.test.noResultRate)} |` + ) + .join('\n') + +const markdown = `# Threshold derivation — v1.6 (#192) + +Generated by \`node scripts/eval-threshold.mjs\`. Numbers are harness output; do not edit them by hand. + +## What was measured + +The real harness, the same corpus and the production retrieval config +(\`candidateK=20, contextK=3\`), once per candidate threshold. The **validation** +split selects; the **test** split reports. Both come from the same +\`--eval-split=\` code path, and the split is a deterministic function of the +question id, so this is reproducible. + +| Threshold | n (val) | Recall@5 (val) | nDCG@10 (val) | MAP@10 (val) | No-result (val) | nDCG@10 (test) | No-result (test) | +| --- | --- | --- | --- | --- | --- | --- | --- | +${tableRows} + +## Selection rule + +Best validation nDCG@10, then fewest validation no-results, then the **lowest** +threshold — the first stage is supposed to favour recall and let a later stage +filter, so among equals the wider one is the safer default. + +## Outcome + +${outcome} + +The app currently ships \`threshold = ${PRODUCTION_THRESHOLD}\`: validation nDCG@10 +${format4(production.validation.metrics.ndcgAt10)}, test nDCG@10 +${format4(production.test.metrics.ndcgAt10)}, test no-result rate +${format4(production.test.noResultRate)}. + +## Caveat on this corpus + +The split removes the most obvious form of overfitting, but ${rows[0].validation.questions} +validation questions is a thin basis for a decision, and the corpus is still small. A +threshold is a product decision with a **no-result-rate** cost attached, so a +recommendation here is only as good as the corpus behind it. Re-run this after the +corpus grows (#192 child 2). + +## Reproduce + +\`\`\`bash +npm run eval:prepare # one-time, networked model bootstrap +npm run eval:threshold # offline; rewrites this file +\`\`\` +` + +mkdirSync(resolve(OUT_MD, '..'), { recursive: true }) +writeFileSync( + OUT_JSON, + `${JSON.stringify({ baseline: 'v1.6', productionThreshold: PRODUCTION_THRESHOLD, flat, recommended: flat ? PRODUCTION_THRESHOLD : winner.threshold, rows }, null, 2)}\n` +) +writeFileSync(OUT_MD, markdown) + +console.log( + flat + ? `[threshold] flat sweep; no evidence to move off ${PRODUCTION_THRESHOLD}` + : `[threshold] recommended ${winner.threshold} (validation nDCG@10 ${format4(winner.validation.metrics.ndcgAt10)})` +) +console.log(`[threshold] wrote ${OUT_JSON} and ${OUT_MD}`) diff --git a/src/main/eval/harness.ts b/src/main/eval/harness.ts index 02e5bb4..7ae5114 100644 --- a/src/main/eval/harness.ts +++ b/src/main/eval/harness.ts @@ -37,10 +37,12 @@ import type { EvalQuestion, EvalReport, EvalRelevantLocation, + EvalSplit, EvalTypeBreakdown, QuestionReport, ResolvedGroundTruth } from './types' +import { selectSplit } from './types' type Db = ReturnType @@ -51,6 +53,8 @@ export interface EvalHarnessOptions { corpusLabel: string questionsPath: string baseline: string + /** 本次只评这一份切分(#192);缺省 `all`。 */ + split: EvalSplit /** * 第一阶段每个通道的宽度,也是排名指标的评估深度(#77)。 * @@ -216,7 +220,11 @@ export async function runEvalHarness( options.corpusDir, options.chunkOptions ) - const questions = parseQuestions(await readFile(options.questionsPath, 'utf-8')) + const allQuestions = parseQuestions(await readFile(options.questionsPath, 'utf-8')) + const questions = selectSplit(allQuestions, options.split) + if (questions.length === 0) { + throw new Error(`eval split "${options.split}" selected no questions from ${options.questionsPath}`) + } const perQuestion: QuestionReport[] = [] const latencies: number[] = [] @@ -282,6 +290,7 @@ export async function runEvalHarness( respectHeadings: chunking.respectHeadings }, retrieval: options.strategy, + split: options.split, candidateK: options.candidateK, contextK: options.contextK, threshold: options.threshold, diff --git a/src/main/eval/report.ts b/src/main/eval/report.ts index 34d9688..5b8897c 100644 --- a/src/main/eval/report.ts +++ b/src/main/eval/report.ts @@ -38,7 +38,8 @@ Generated by \`${report.generatedBy}\`. The numbers below are harness output — | Ranks | \`candidateK=${config.candidateK}, threshold=${config.threshold}\` | | Context width | \`contextK=${config.contextK}\` | | Evidence per query | \`evidenceK=${config.evidenceK}\` | -| Corpus | \`${config.corpus}\` (${config.documents} documents, ${config.questions} questions) | +| Corpus | \`${config.corpus}\` (${config.documents} documents) | +| Split | \`${config.split}\` (${config.questions} questions) | | Index size | ${config.chunkCount} chunks | ## Metrics diff --git a/src/main/eval/run.ts b/src/main/eval/run.ts index 29085ad..802cf5e 100644 --- a/src/main/eval/run.ts +++ b/src/main/eval/run.ts @@ -22,6 +22,7 @@ import { KnowledgeService } from '../services/KnowledgeService' import { isModelInstalled } from '../embedding/ModelRegistry' import { DEFAULT_CHUNK_OPTIONS, type ChunkOptions } from '../services/ChunkingService' import type { RetrievalStrategy } from '../services/retrieval' +import type { EvalSplit } from './types' import { runEvalHarness, stabilize, @@ -83,6 +84,17 @@ function readRetrievalStrategy(argv: readonly string[]): RetrievalStrategy { return raw } +/** + * 本次评估的切分(#192)。默认 `all`;阈值这类扫参要用 `validation` 选、`test` 报。 + */ +function readSplit(argv: readonly string[]): EvalSplit { + const raw = readOption(argv, '--eval-split=', 'all') + if (raw !== 'all' && raw !== 'validation' && raw !== 'test') { + throw new Error(`--eval-split expects all, validation or test, got ${JSON.stringify(raw)}`) + } + return raw +} + /** * Chunking config for one run (#78). The defaults are the production defaults, so * `npm run eval` with no flags still measures what ships. @@ -179,6 +191,7 @@ export async function runEvalCli(argv: readonly string[] = process.argv): Promis corpusLabel: repoRelative(corpusDir) || 'eval/corpus', questionsPath, baseline: readOption(argv, '--eval-baseline=', 'v1.6'), + split: readSplit(argv), // 默认就是生产配置(#77):先取宽,融合,再把 contextK 条送进 prompt。一个不镜像 // 线上参数的 benchmark 量的是用户永远不会跑的检索器。 candidateK: readNumberOption(argv, '--eval-candidate-k=', 20), diff --git a/src/main/eval/types.ts b/src/main/eval/types.ts index dbb692b..3a075f4 100644 --- a/src/main/eval/types.ts +++ b/src/main/eval/types.ts @@ -34,6 +34,29 @@ export interface EvalQuestion { type?: string } +/** + * 评估切分(#192)。 + * + * 阈值这类参数必须在**没参与选择**的问题上报数,否则扫参的结果只是把测试集背下来 + * 了。切分按 question id 确定性计算,所以同一份 `questions.jsonl` 在任何机器上切出 + * 同一份 validation / test。 + */ +export type EvalSplit = 'all' | 'validation' | 'test' + +/** id 分桶,0/1/2;只用于切分,不参与检索。 */ +export function splitBucket(id: string): number { + let hash = 0 + for (const character of id) hash = (hash * 31 + character.charCodeAt(0)) >>> 0 + return hash % 3 +} + +/** validation 是 bucket 0(约 1/3),test 是其余(约 2/3)。 */ +export function selectSplit(questions: EvalQuestion[], split: EvalSplit): EvalQuestion[] { + if (split === 'all') return questions + const wantValidation = split === 'validation' + return questions.filter((question) => (splitBucket(question.id) === 0) === wantValidation) +} + /** One resolved ground-truth location, after runtime id mapping. */ export interface ResolvedGroundTruth { document: string @@ -101,6 +124,8 @@ export interface EvalReport { respectHeadings: boolean } retrieval: string + /** 本次评估用了哪一份切分(#192):`all` / `validation` / `test`。 */ + split: string /** * 第一阶段每个通道的宽度(#77)。排名指标(Recall@K / MRR / nDCG@K)在这个深度上 * 计算,所以它必须 ≥ 指标里最大的 K。 diff --git a/test/evalSplit.test.ts b/test/evalSplit.test.ts new file mode 100644 index 0000000..2de5a02 --- /dev/null +++ b/test/evalSplit.test.ts @@ -0,0 +1,61 @@ +import { test } from 'node:test' +import assert from 'node:assert/strict' +import { + selectSplit, + splitBucket, + type EvalQuestion, + type EvalSplit +} from '../src/main/eval/types.ts' + +/** + * The eval split (#192). Thresholds and other swept parameters have to be chosen on + * questions that did not take part in the choice, so the split has to be + * deterministic: the same `questions.jsonl` must produce the same validation and + * test sets on every machine. + */ + +const question = (id: string): EvalQuestion => ({ + id, + question: id, + relevant: [{ document: 'a.md', page: null, block: 0 }] +}) + +const questions = Array.from({ length: 30 }, (_, index) => question(`q${String(index + 1).padStart(3, '0')}`)) + +test('the split buckets only ever return 0, 1 or 2', () => { + for (const q of questions) { + const bucket = splitBucket(q.id) + assert.ok(bucket === 0 || bucket === 1 || bucket === 2, `${q.id} -> ${bucket}`) + } +}) + +test('the split is deterministic, not random', () => { + const first = selectSplit(questions, 'validation').map((q) => q.id) + const second = selectSplit(questions, 'validation').map((q) => q.id) + assert.deepEqual(first, second) +}) + +test('validation and test partition the questions without overlap', () => { + const validation = selectSplit(questions, 'validation').map((q) => q.id) + const test = selectSplit(questions, 'test').map((q) => q.id) + + assert.equal(validation.length + test.length, questions.length) + assert.equal(new Set([...validation, ...test]).size, questions.length) + // Both sides are non-empty on a 30-question set, so neither arm is vacuous. + assert.ok(validation.length > 0 && test.length > 0) +}) + +test('`all` is the whole set, and the selected questions keep their order', () => { + assert.deepEqual( + selectSplit(questions, 'all').map((q) => q.id), + questions.map((q) => q.id) + ) + + for (const split of ['validation', 'test'] as EvalSplit[]) { + const selected = selectSplit(questions, split).map((q) => q.id) + const expected = questions + .filter((q) => (splitBucket(q.id) === 0) === (split === 'validation')) + .map((q) => q.id) + assert.deepEqual(selected, expected) + } +})