From a52b9b7cf7261049580797c0601b1780f5c0a616 Mon Sep 17 00:00:00 2001 From: MrSibe Date: Wed, 30 Sep 2026 16:58:46 +0800 Subject: [PATCH] feat(eval): adopt on the first metric with headroom, not a saturated one MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The v1.5 rule was "Recall@5 must improve and nDCG@10 must not regress". On a corpus where dense already scores Recall@5 = 1.0000, no strategy can improve Recall@5, so the rule was not strict — it was **unsatisfiable**. Every comparison came back "inconclusive", including a hybrid that was better on Recall@1, MRR and nDCG@10. The default never moved, not because hybrid lost but because the rule could not return a verdict. Child 10 of #192. **The amendment.** A metric at its maximum has no headroom and is not allowed to decide. The deciding metric is the first one with headroom, in the order `recallAt5`, `ndcgAt10`, `mrr`, `mapAt10`; a strategy clears the rule when it improves that metric and regresses none of the others. Saturation is detected and reported rather than silently blocking every change. **The rule is code, not prose.** It lives in `src/main/eval/adoption.ts` with 8 unit tests, because it decides whether a shipped default moves and testing it by reading the sentence the script prints would only test the sentence. The experiment script imports it, so there is one definition. ## The v1.5 stalemate resolves ``` | Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Query p95 | | dense (vector) | 0.8333 | 1.0000 | 0.9278 | 0.9437 | 0.9222 | 14.06 ms | | sparse (BM25) | 0.7333 | 0.8667 | 0.8056 | 0.8184 | 0.8000 | 2.63 ms | | hybrid (RRF dense + BM25) | 0.8667 | 1.0000 | 0.9444 | 0.9561 | 0.9389 | 22.32 ms | ``` `recallAt5` is reported as saturated; the deciding metric is `ndcgAt10`; **hybrid clears the rule** and regresses none of the four metrics, at a p95 cost of ~8 ms. The script reports the measurement and does not flip the default — changing the shipped strategy is a product decision, and it is stated that way in the output rather than implied by a green checkmark. `docs/eval/retrieval-v1.6.{json,md}` is the regenerated experiment against the v1.6 baseline; `retrieval-v1.5.*` is kept as the record of the superseded rule. ## Testing - `npm run typecheck` — clean - `npm test` — 497 pass, 8 new: saturation detection, deciding-metric priority, the float-slack regression check, the resolved stalemate, a trade that regresses another metric being refused, the baseline not clearing against itself, all-saturated, and best-candidate selection - `npm run eval` + `npm run eval:retrieval` — regenerated and re-run Part of #192 (child 10). --- docs/eval/baseline-v1.6.md | 14 ++++-- docs/eval/retrieval-v1.6.json | 54 ++++++++++++++++++++++ docs/eval/retrieval-v1.6.md | 49 ++++++++++++++++++++ package.json | 2 +- scripts/eval-retrieval.mjs | 76 ++++++++++++++++-------------- src/main/eval/adoption.ts | 87 +++++++++++++++++++++++++++++++++++ src/main/eval/report.ts | 10 ++-- test/evalAdoption.test.ts | 86 ++++++++++++++++++++++++++++++++++ 8 files changed, 335 insertions(+), 43 deletions(-) create mode 100644 docs/eval/retrieval-v1.6.json create mode 100644 docs/eval/retrieval-v1.6.md create mode 100644 src/main/eval/adoption.ts create mode 100644 test/evalAdoption.test.ts diff --git a/docs/eval/baseline-v1.6.md b/docs/eval/baseline-v1.6.md index 605c29a..ce25b05 100644 --- a/docs/eval/baseline-v1.6.md +++ b/docs/eval/baseline-v1.6.md @@ -43,8 +43,8 @@ The type comes from `type` in `questions.jsonl`; untagged questions report as | semantic | 18 | 1.0000 | 0.9312 | 1.0000 | 0.9074 | | zh | 3 | 1.0000 | 1.0000 | 1.0000 | 1.0000 | -Timing is informational only and is **not** frozen: indexing 1542 ms, query -p50 15.01 ms, p95 69.16 ms on the +Timing is informational only and is **not** frozen: indexing 1551 ms, query +p50 11.73 ms, p95 18.71 ms on the machine that produced this file. Timing and index size depend on hardware and on the corpus, so they must never be the reason two runs differ. @@ -68,9 +68,13 @@ first-stage width per channel, `contextK` is how many passages the chat prompt t and `threshold` is the similarity floor the app ships. A benchmark that does not mirror those parameters measures a retriever nobody runs. The adopted-change rule is: -> Adopt a change only if Recall@5 improves and nDCG@10 does not regress. A change -> that trades a large latency increase for a marginal recall gain is a product -> decision, not an automatic win, and must be stated as such. +> Adopt a strategy when it improves the **first metric with headroom** — in the order +> Recall@5, nDCG@10, MRR, MAP@10 — and regresses none of the others. A metric already +> at its maximum has no headroom and cannot decide anything; a rule that depends on +> one is unsatisfiable, not strict (#192 child 10). +> +> A change that trades a large latency increase for a marginal quality gain is a +> product decision, not an automatic win, and must be stated as such. A changed result must be reproducible with: diff --git a/docs/eval/retrieval-v1.6.json b/docs/eval/retrieval-v1.6.json new file mode 100644 index 0000000..8fed18f --- /dev/null +++ b/docs/eval/retrieval-v1.6.json @@ -0,0 +1,54 @@ +{ + "baseline": "dense", + "chunking": "1000/100", + "strategies": [ + { + "id": "dense", + "label": "dense (vector)", + "chunking": "1000/100", + "chunkCount": 19, + "recallAt1": 0.833333, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 0.927778, + "ndcgAt10": 0.94375, + "hitRateAt5": 1, + "mapAt10": 0.922222, + "evidencePrecisionAt5": 0.213333, + "indexingMs": 1531, + "latencyP95Ms": 14.81 + }, + { + "id": "sparse", + "label": "sparse (BM25)", + "chunking": "1000/100", + "chunkCount": 19, + "recallAt1": 0.733333, + "recallAt5": 0.866667, + "recallAt10": 0.866667, + "mrr": 0.805556, + "ndcgAt10": 0.818355, + "hitRateAt5": 0.866667, + "mapAt10": 0.8, + "evidencePrecisionAt5": 0.186667, + "indexingMs": 1503, + "latencyP95Ms": 1.89 + }, + { + "id": "hybrid", + "label": "hybrid (RRF of dense + BM25)", + "chunking": "1000/100", + "chunkCount": 19, + "recallAt1": 0.866667, + "recallAt5": 1, + "recallAt10": 1, + "mrr": 0.944444, + "ndcgAt10": 0.956053, + "hitRateAt5": 1, + "mapAt10": 0.938889, + "evidencePrecisionAt5": 0.213333, + "indexingMs": 1548, + "latencyP95Ms": 19.15 + } + ] +} diff --git a/docs/eval/retrieval-v1.6.md b/docs/eval/retrieval-v1.6.md new file mode 100644 index 0000000..9651e44 --- /dev/null +++ b/docs/eval/retrieval-v1.6.md @@ -0,0 +1,49 @@ +# Retrieval experiments — v1.6 (#77, #192) + +Generated by `node scripts/eval-retrieval.mjs`. Numbers are harness output; do not edit them by hand. + +## What was measured + +Every strategy runs the real RAG eval harness against the same corpus and the same 30 +questions as `baseline-v1.6.json`, with chunking held fixed at 1000/100. Only the retrieval strategy changes. + +| Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Evidence P@5 | Query p95 | +| --- | --- | --- | --- | --- | --- | --- | --- | +| dense (vector) | 0.8333 | 1.0000 | 0.9278 | 0.9437 | 0.9222 | 0.2133 | 14.81 ms | +| sparse (BM25) | 0.7333 | 0.8667 | 0.8056 | 0.8184 | 0.8000 | 0.1867 | 1.89 ms | +| hybrid (RRF of dense + BM25) | 0.8667 | 1.0000 | 0.9444 | 0.9561 | 0.9389 | 0.2133 | 19.15 ms | + +## Not evaluated + +**Reranking.** The issue lists "hybrid + reranker" as a step, but a cross-encoder +model is not available offline and inventing its numbers would defeat the point of +the harness. It stays open until a model can be pinned the way the embedding model +is. + +## Saturation + +Saturated (no headroom, so they cannot decide anything): `recallAt5`. The deciding metric on this corpus is `ndcgAt10`. A saturated metric is still reported, because "this corpus cannot move it" is itself +information; it is just not allowed to decide the comparison. + +## Adoption rule + +> Adopt a strategy when it improves the **first metric with headroom** — in the order +> `recallAt5`, `nDCG@10`, `MRR`, `MAP@10` — and regresses none of the others. A metric +> already at its maximum has no headroom and cannot decide anything; a rule that depends +> on one is unsatisfiable, not strict (#192 child 10). +> +> A change that trades a large latency increase for a marginal quality gain is a product +> decision, not an automatic win. + +## Outcome + +`hybrid (RRF of dense + BM25)` **clears the rule**: it improves the deciding metric `ndcgAt10` (0.9561 vs dense 0.9437) and regresses none of `recallAt5`, `ndcgAt10`, `mrr`, `mapAt10`. Saturated (no headroom, so they cannot decide anything): `recallAt5`. + +Changing the shipped default is a separate decision, and this script does not make it — it reports the measurement. + +## Reproduce + +```bash +npm run eval:prepare # one-time, networked model bootstrap +npm run eval:retrieval # offline; runs every strategy and rewrites this file +``` diff --git a/package.json b/package.json index b93f550..f7b0bcb 100644 --- a/package.json +++ b/package.json @@ -41,7 +41,7 @@ "db:migrate": "drizzle-kit migrate", "db:push": "drizzle-kit push", "db:studio": "drizzle-kit studio", - "eval:retrieval": "npm run build && node scripts/eval-retrieval.mjs", + "eval:retrieval": "npm run build && node --experimental-transform-types --disable-warning=ExperimentalWarning --disable-warning=MODULE_TYPELESS_PACKAGE_JSON scripts/eval-retrieval.mjs", "eval:threshold": "npm run build && node scripts/eval-threshold.mjs" }, "//test": [ diff --git a/scripts/eval-retrieval.mjs b/scripts/eval-retrieval.mjs index bccc605..d52b5bd 100644 --- a/scripts/eval-retrieval.mjs +++ b/scripts/eval-retrieval.mjs @@ -3,11 +3,14 @@ * Retrieval experiments for #77. * * Runs the real RAG eval harness once per retrieval strategy against the frozen - * chunk baseline (`baseline-v1.5.json`, 1000/100), holding chunking fixed, and + * chunk baseline (`baseline-v1.6.json`, 1000/100), holding chunking fixed, and * writes the comparison the issue asks for as a delta against dense. * * The harness does the measuring; this script only orchestrates and tabulates. * + * The adoption rule is the amended one from #192 child 10: the deciding metric is the + * first metric with headroom, not a metric that the corpus has already maxed out. + * * Usage: * node scripts/eval-retrieval.mjs * @@ -19,7 +22,7 @@ import { existsSync, mkdtempSync, mkdirSync, readFileSync, rmSync, writeFileSync import { tmpdir } from 'node:os' import { join, resolve } from 'node:path' -const OUT_MD = resolve(readArg('--out=', 'docs/eval/retrieval-v1.5.md')) +const OUT_MD = resolve(readArg('--out=', 'docs/eval/retrieval-v1.6.md')) const OUT_JSON = OUT_MD.replace(/\.md$/, '.json') /** @@ -54,7 +57,7 @@ function runStrategy(strategy, outDir) { const args = [ '.', '--eval-harness', - '--eval-baseline=v1.5', + '--eval-baseline=v1.6', `--eval-out=${outDir}`, `--eval-retrieval=${strategy.id}` ] @@ -117,14 +120,14 @@ try { console.log(`[retrieval] running ${strategy.label}`) const metrics = await runStrategy(strategy, outDir) - const report = JSON.parse(readFileSync(join(outDir, 'baseline-v1.5.json'), 'utf8')) + const report = JSON.parse(readFileSync(join(outDir, 'baseline-v1.6.json'), 'utf8')) results.push({ id: strategy.id, label: strategy.label, chunking: `${report.config.chunking.chunkSize}/${report.config.chunking.chunkOverlap}`, chunkCount: report.config.chunkCount, ...metrics, - ...readTiming(join(outDir, 'baseline-v1.5.md')) + ...readTiming(join(outDir, 'baseline-v1.6.md')) }) } } finally { @@ -138,43 +141,46 @@ const format4 = (value) => value.toFixed(4) const rows = results.map( (result) => - `| ${result.label} | ${format4(result.recallAt1)} | ${format4(result.recallAt5)} | ${format4(result.mrr)} | ${format4(result.ndcgAt10)} | ${format4(result.evidencePrecisionAt5)} | ${result.latencyP95Ms?.toFixed(2)} ms |` + `| ${result.label} | ${format4(result.recallAt1)} | ${format4(result.recallAt5)} | ${format4(result.mrr)} | ${format4(result.ndcgAt10)} | ${format4(result.mapAt10)} | ${format4(result.evidencePrecisionAt5)} | ${result.latencyP95Ms?.toFixed(2)} ms |` ) /** - * The rule frozen in the baseline report: Recall@5 must improve and nDCG@10 must - * not regress. Latency is reported so a win that costs 5x latency is stated as a - * trade-off, not hidden. + * The rule lives in `src/main/eval/adoption.ts` so it can be unit tested: it decides + * whether a shipped default moves, and testing it by reading the sentence this script + * prints would be a test of the sentence. + * + * Latency stays in the table so a win that costs 5x latency is stated as a trade-off, + * not hidden. */ -const adopted = results.filter( - (result) => - result.id !== 'dense' && - result.recallAt5 > baseline.recallAt5 && - result.ndcgAt10 >= baseline.ndcgAt10 -) -const winner = - adopted.sort((a, b) => b.recallAt5 - a.recallAt5 || b.ndcgAt10 - a.ndcgAt10)[0] ?? null +const { ADOPTION_METRICS, decideAdoption } = await import('../src/main/eval/adoption.ts') +const { primary, saturated, winner } = decideAdoption(baseline, results) + +const saturationNote = saturated.length + ? `Saturated (no headroom, so they cannot decide anything): ${saturated + .map((key) => `\`${key}\``) + .join(', ')}.` + : 'No metric in the rule is saturated on this corpus.' let outcome if (winner) { - outcome = `\`${winner.label}\` clears the rule (Recall@5 ${format4(winner.recallAt5)} vs dense ${format4(baseline.recallAt5)}, nDCG@10 ${format4(winner.ndcgAt10)} vs ${format4(baseline.ndcgAt10)}).` -} else if (baseline.recallAt5 === 1) { - outcome = `Recall@5 is saturated at 1.0000, so the rule's first condition cannot be met by any strategy. **Dense stays the default**, and the non-dense strategies are reported as inconclusive rather than adopted or rejected on a metric that cannot move.` + outcome = `\`${winner.label}\` **clears the rule**: it improves the deciding metric \`${primary}\` (${format4(winner[primary])} vs dense ${format4(baseline[primary])}) and regresses none of ${ADOPTION_METRICS.map((key) => `\`${key}\``).join(', ')}. ${saturationNote}\n\nChanging the shipped default is a separate decision, and this script does not make it — it reports the measurement.` +} else if (primary === null) { + outcome = `Every metric in the rule is already at its maximum on this corpus, so no strategy can clear any of them. **Dense stays the default**; the comparison is inconclusive by construction, not negative.` } else { - outcome = `No strategy cleared the rule. **Dense stays the default.** A negative result is the point of the experiment: it is the measurement that says the extra machinery is not worth its cost on this corpus, not a failure to deliver.` + outcome = `No strategy cleared the rule. The deciding metric was \`${primary}\` (dense ${format4(baseline[primary])}); the strategies either failed to improve it or regressed another metric. **Dense stays the default.** A negative result is the point of the experiment: it is the measurement that says the extra machinery is not worth its cost on this corpus, not a failure to deliver. ${saturationNote}` } -const markdown = `# Retrieval experiments — v1.5 (#77) +const markdown = `# Retrieval experiments — v1.6 (#77, #192) Generated by \`node scripts/eval-retrieval.mjs\`. Numbers are harness output; do not edit them by hand. ## What was measured Every strategy runs the real RAG eval harness against the same corpus and the same 30 -questions as \`baseline-v1.5.json\`, with chunking held fixed at ${baseline.chunking}. Only the retrieval strategy changes. +questions as \`baseline-v1.6.json\`, with chunking held fixed at ${baseline.chunking}. Only the retrieval strategy changes. -| Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | Evidence P@5 | Query p95 | -| --- | --- | --- | --- | --- | --- | --- | +| Strategy | Recall@1 | Recall@5 | MRR | nDCG@10 | MAP@10 | Evidence P@5 | Query p95 | +| --- | --- | --- | --- | --- | --- | --- | --- | ${rows.join('\n')} ## Not evaluated @@ -184,19 +190,21 @@ model is not available offline and inventing its numbers would defeat the point the harness. It stays open until a model can be pinned the way the embedding model is. -## Corpus limitation +## Saturation -The rule's Recall@5 condition is **saturated** on this corpus: dense already scores -1.0000, so no strategy can improve it and the rule can therefore never be met here. -The metrics that still discriminate are Recall@1, MRR and nDCG@10. A hybrid result -that is better on all three but equal on Recall@5 is therefore *inconclusive*, not a -negative result, and the default is left unchanged until the comparison can run on a -corpus where Recall@5 is not already perfect. +${saturationNote} The deciding metric on this corpus is ${ + primary === null ? 'none — every metric in the rule is already maxed out' : `\`${primary}\`` +}. A saturated metric is still reported, because "this corpus cannot move it" is itself +information; it is just not allowed to decide the comparison. ## Adoption rule -> Adopt a change only if Recall@5 improves and nDCG@10 does not regress. A change -> that trades a large latency increase for a marginal recall gain is a product +> Adopt a strategy when it improves the **first metric with headroom** — in the order +> \`recallAt5\`, \`nDCG@10\`, \`MRR\`, \`MAP@10\` — and regresses none of the others. A metric +> already at its maximum has no headroom and cannot decide anything; a rule that depends +> on one is unsatisfiable, not strict (#192 child 10). +> +> A change that trades a large latency increase for a marginal quality gain is a product > decision, not an automatic win. ## Outcome diff --git a/src/main/eval/adoption.ts b/src/main/eval/adoption.ts new file mode 100644 index 0000000..e63c519 --- /dev/null +++ b/src/main/eval/adoption.ts @@ -0,0 +1,87 @@ +/** + * The adoption rule for retrieval experiments (#192 child 10). + * + * The v1.5 rule was "Recall@5 must improve and nDCG@10 must not regress". On a corpus + * where dense already scores Recall@5 = 1.0000 that condition can never be met, so the + * rule was not strict, it was **unsatisfiable**, and every strategy comparison came + * back "inconclusive" — including a hybrid that was better on every other metric. + * + * The amendment: a metric at its maximum has no headroom and is not allowed to decide. + * The deciding metric is the first one with headroom, and a strategy is adopted when + * it improves that metric and regresses none of the others. + * + * Kept here, not inside the experiment script, so the rule can be unit tested: it + * decides whether a production default moves, and "the script printed a different + * sentence" is not a test. + */ + +/** Priority order. Recall first because a RAG miss cannot be repaired downstream. */ +export const ADOPTION_METRICS = ['recallAt5', 'ndcgAt10', 'mrr', 'mapAt10'] as const + +export type AdoptionMetric = (typeof ADOPTION_METRICS)[number] + +export type MetricBag = Record + +/** Float slack: metrics are rounded to 6 decimals before this runs. */ +export const ADOPTION_EPSILON = 1e-9 + +/** Metrics already at their maximum, which therefore cannot decide a comparison. */ +export function saturatedMetrics( + metrics: MetricBag, + order: readonly string[] = ADOPTION_METRICS +): string[] { + return order.filter((key) => metrics[key] >= 1 - ADOPTION_EPSILON) +} + +/** The first metric in priority order with room to improve, or `null` if none has. */ +export function decidingMetric( + metrics: MetricBag, + order: readonly string[] = ADOPTION_METRICS +): string | null { + return order.find((key) => metrics[key] < 1 - ADOPTION_EPSILON) ?? null +} + +/** Metrics where `candidate` is worse than `baseline` beyond the epsilon. */ +export function regressedMetrics( + candidate: MetricBag, + baseline: MetricBag, + order: readonly string[] = ADOPTION_METRICS +): string[] { + return order.filter((key) => candidate[key] < baseline[key] - ADOPTION_EPSILON) +} + +export interface AdoptionDecision { + /** The metric that decides, or `null` when every metric is already maxed out. */ + primary: string | null + saturated: string[] + /** The best strategy that clears the rule, or `null`. */ + winner: MetricBag | null +} + +/** + * Pick the strategy to recommend. `baseline` is the shipped one and must be part of + * `candidates`; a candidate that improves the deciding metric and regresses nothing + * else clears the rule, and the best such candidate by the deciding metric wins. + */ +export function decideAdoption( + baseline: MetricBag, + candidates: readonly MetricBag[], + order: readonly string[] = ADOPTION_METRICS +): AdoptionDecision { + const saturated = saturatedMetrics(baseline, order) + const primary = decidingMetric(baseline, order) + + if (primary === null) return { primary: null, saturated, winner: null } + + const cleared = candidates.filter( + (candidate) => + candidate !== baseline && + candidate[primary] > baseline[primary] + ADOPTION_EPSILON && + regressedMetrics(candidate, baseline, order).length === 0 + ) + + const winner = + [...cleared].sort((a, b) => b[primary] - a[primary])[0] ?? null + + return { primary, saturated, winner } +} diff --git a/src/main/eval/report.ts b/src/main/eval/report.ts index 5b8897c..7b883f8 100644 --- a/src/main/eval/report.ts +++ b/src/main/eval/report.ts @@ -90,9 +90,13 @@ first-stage width per channel, \`contextK\` is how many passages the chat prompt and \`threshold\` is the similarity floor the app ships. A benchmark that does not mirror those parameters measures a retriever nobody runs. The adopted-change rule is: -> Adopt a change only if Recall@5 improves and nDCG@10 does not regress. A change -> that trades a large latency increase for a marginal recall gain is a product -> decision, not an automatic win, and must be stated as such. +> Adopt a strategy when it improves the **first metric with headroom** — in the order +> Recall@5, nDCG@10, MRR, MAP@10 — and regresses none of the others. A metric already +> at its maximum has no headroom and cannot decide anything; a rule that depends on +> one is unsatisfiable, not strict (#192 child 10). +> +> A change that trades a large latency increase for a marginal quality gain is a +> product decision, not an automatic win, and must be stated as such. A changed result must be reproducible with: diff --git a/test/evalAdoption.test.ts b/test/evalAdoption.test.ts new file mode 100644 index 0000000..ad61e05 --- /dev/null +++ b/test/evalAdoption.test.ts @@ -0,0 +1,86 @@ +import { test } from 'node:test' +import assert from 'node:assert/strict' +import { + ADOPTION_METRICS, + decideAdoption, + decidingMetric, + regressedMetrics, + saturatedMetrics +} from '../src/main/eval/adoption.ts' + +/** + * The adoption rule (#192 child 10) decides whether a shipped default moves, so it is + * pinned here rather than trusted to the sentence the experiment script prints. + * + * The bug it exists for: the v1.5 rule made Recall@5 — already 1.0000 on the corpus — + * its deciding condition, so a hybrid that was better on every other metric came back + * "inconclusive" forever. + */ + +const bag = (overrides: Record = {}): Record => ({ + recallAt5: 0.9, + ndcgAt10: 0.9, + mrr: 0.9, + mapAt10: 0.9, + ...overrides +}) + +test('a metric at its maximum is saturated and cannot decide', () => { + assert.deepEqual(saturatedMetrics(bag({ recallAt5: 1 })), ['recallAt5']) + assert.deepEqual(saturatedMetrics(bag()), []) +}) + +test('the deciding metric is the first one with headroom', () => { + assert.equal(decidingMetric(bag({ recallAt5: 1 })), 'ndcgAt10') + assert.equal(decidingMetric(bag({ recallAt5: 1, ndcgAt10: 1 })), 'mrr') + assert.equal(decidingMetric(bag()), 'recallAt5') + assert.equal(decidingMetric(bag({ recallAt5: 1, ndcgAt10: 1, mrr: 1, mapAt10: 1 })), null) +}) + +test('a regression is measured against the baseline, beyond the float slack', () => { + const baseline = bag() + assert.deepEqual(regressedMetrics(bag({ ndcgAt10: 0.8 }), baseline), ['ndcgAt10']) + // Rounding to 6 decimals must not read as a regression. + assert.deepEqual(regressedMetrics(bag({ ndcgAt10: 0.9 - 1e-12 }), baseline), []) +}) + +test('the v1.5 stalemate is resolved: improving the deciding metric is enough', () => { + const dense = bag({ recallAt5: 1, ndcgAt10: 0.9437, mrr: 0.9278, mapAt10: 0.9222 }) + const hybrid = bag({ recallAt5: 1, ndcgAt10: 0.9561, mrr: 0.9444, mapAt10: 0.9389 }) + + const decision = decideAdoption(dense, [dense, hybrid]) + + assert.deepEqual(decision.saturated, ['recallAt5']) + assert.equal(decision.primary, 'ndcgAt10') + assert.equal(decision.winner, hybrid) +}) + +test('improving one metric while regressing another does not clear the rule', () => { + const dense = bag({ recallAt5: 1, ndcgAt10: 0.9437, mrr: 0.9278 }) + // Better nDCG, worse MRR: exactly the trade the rule refuses to make silently. + const trade = bag({ recallAt5: 1, ndcgAt10: 0.99, mrr: 0.9 }) + + assert.equal(decideAdoption(dense, [dense, trade]).winner, null) +}) + +test('the baseline never clears the rule against itself', () => { + const dense = bag({ recallAt5: 1 }) + assert.equal(decideAdoption(dense, [dense]).winner, null) +}) + +test('when every metric is saturated nothing can be adopted', () => { + const all = bag({ recallAt5: 1, ndcgAt10: 1, mrr: 1, mapAt10: 1 }) + const decision = decideAdoption(all, [all, bag({ ndcgAt10: 1 })]) + + assert.equal(decision.primary, null) + assert.equal(decision.winner, null) + assert.deepEqual(decision.saturated, [...ADOPTION_METRICS]) +}) + +test('the best clearing candidate by the deciding metric wins', () => { + const dense = bag({ recallAt5: 1, ndcgAt10: 0.9 }) + const good = bag({ recallAt5: 1, ndcgAt10: 0.95 }) + const better = bag({ recallAt5: 1, ndcgAt10: 0.98 }) + + assert.equal(decideAdoption(dense, [dense, good, better]).winner, better) +})