From d36883243cb4e424ac39040443c4eac008d56079 Mon Sep 17 00:00:00 2001 From: MrSibe Date: Wed, 30 Sep 2026 18:24:33 +0800 Subject: [PATCH] feat(eval): fifteen near-miss unanswerable questions, and the last eval PR MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Closes RAG Eval v2 Phase 1. This is the final expansion of the eval system — the conclusions below are the point of the phase, not more infrastructure. ## What was added Fifteen unanswerable questions whose **subject is discussed at length in the corpus and whose answer is not there** — "How much GPU memory does Gateway API v3 need?", "Which laboratory is accredited to analyse the estuary transect samples?", "How many staff work on the reservoir monitoring programme?". Every specific term was checked absent before the question was written (`gpu`, `vram`, `sla`, `uptime`, `expiry`, `sdk`, `self-hosted`, `accredit`, `vendor`, `supplier`, `funding`, `grant`, `staff`, `warranty`, `retention`, `encrypt`, `pricing`, `invoice`, `audit` all return nothing). That check is what makes them unanswerable rather than believed to be, and it is the part that cannot be automated away. `validation` unanswerable goes from **5 to 15**, so abstention resolution goes from 20 % per question to 6.7 %. The earlier "Who won the 2018 FIFA World Cup?" questions were a sanity check for a query that is obviously out of scope; these are the realistic shape — the user's actual hallucination risk is asking a question about a document that *is* in the library. ## The conclusion survives the larger sample ``` unanswerable max candidate min 0.8978 (cosine 0.796) p50 0.9279 (0.856) max 0.9435 (0.887) worst required relevant p10 0.8906 (cosine 0.781) ``` Even the **lowest** unanswerable top score is above the p10 of the worst required relevant passage. With n=15 instead of n=5 the distributions still do not separate, and the threshold curve is unchanged in shape: | threshold | raw cosine | answerable full recall | unanswerable abstain | | --- | --- | --- | --- | | 0.875 | 0.750 | 1.0000 | 0.0000 | | 0.900 | 0.800 | 0.8947 | 0.0667 | | 0.925 | 0.850 | 0.6316 | 0.4000 | | 0.950 | 0.900 | 0.1579 | 1.0000 | Abstention still never rises without full recall falling. The finer resolution makes the finding *more* solid, not different. Regenerated to stay consistent: baseline, scores, threshold and sweep. `paired-v1.6` is unchanged, because adding unanswerable questions does not touch any answerable one. ## Where Phase 1 leaves the product 1. The original 19-chunk benchmark could not guide a product decision; the current one (53 chunks, 78 questions, per-type metrics, held-out split) can. 2. **Cross-lingual retrieval is the largest measured gap**: Recall@5 0.6667 and nDCG@10 0.4173 against 0.93–1.00 for same-language questions. Chinese over a Chinese source scores 1.0000, so it is the cross-language matching, not Chinese. 3. **Hybrid has ranking value and no held-out mandate.** Its full-set advantage is five questions out of 34 (`paired-v1.6`), mostly `semantic`, and it does nothing for cross-lingual. Dense stays the default. 4. `contextK = 3` is justified against 5 and 8 on this corpus; `candidateK` above 10 is not distinguishable. 5. `threshold = 0.5` means **cosine ≥ 0**, not cosine ≥ 0.5. 6. **A single embedding similarity threshold cannot decide answerability** — the relevant and non-relevant and unanswerable score distributions overlap, and every threshold that abstains also drops required evidence. (6) is the phase's real result: it says stop tuning this knob and reach for a different mechanism. ## Stop line Phase 1 ends here. Not done, and deliberately not started: public benchmarks, reranker eval, generator faithfulness, citation entailment, growing the corpus to 300 chunks, an `Abstention Signal Evaluation`. Those build a more complete benchmark; they do not unblock KnowNote development, and (6) already says the next useful work is a different retrieval mechanism rather than a better measurement of this one. --- docs/eval/baseline-v1.6.json | 484 +++++++++++++++++++++++++++++++++- docs/eval/baseline-v1.6.md | 10 +- docs/eval/scores-v1.6.json | 16 +- docs/eval/scores-v1.6.md | 8 +- docs/eval/sweep-v1.6.json | 90 +++---- docs/eval/sweep-v1.6.md | 44 ++-- docs/eval/threshold-v1.6.json | 40 +-- docs/eval/threshold-v1.6.md | 2 +- eval/questions.jsonl | 15 ++ eval/splits.json | 17 +- 10 files changed, 618 insertions(+), 108 deletions(-) diff --git a/docs/eval/baseline-v1.6.json b/docs/eval/baseline-v1.6.json index d8171a1..2bddcea 100644 --- a/docs/eval/baseline-v1.6.json +++ b/docs/eval/baseline-v1.6.json @@ -17,7 +17,7 @@ "threshold": 0.5, "corpus": "eval/corpus", "documents": 21, - "questions": 63, + "questions": 78, "chunkCount": 53 }, "metrics": { @@ -124,7 +124,7 @@ } ], "unanswerable": { - "questions": 10, + "questions": 25, "abstentionCount": 0, "retrievalAbstentionRate": 0, "meanCandidatesRetrieved": 20, @@ -2277,6 +2277,486 @@ [], [] ] + }, + { + "id": "q064", + "question": "How much GPU memory does Gateway API v3 need to serve a request?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2908, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q065", + "question": "What is the uptime SLA for Gateway API v2?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2875, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q066", + "question": "How long does an authentication token stay valid before it expires?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2649, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q067", + "question": "Which SDK version should a client use with Gateway API v3?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2908, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q068", + "question": "Can Gateway API v2 be self-hosted on-premise?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2875, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q069", + "question": "How is usage invoiced for Gateway API v2?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2875, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q070", + "question": "Which laboratory is accredited to analyse the estuary transect samples?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2902, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q071", + "question": "Which vendor supplies the coastal monitoring buoys?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2826, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q072", + "question": "What is the annual funding for the groundwater monitoring programme?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2933, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q073", + "question": "How many staff work on the reservoir monitoring programme?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2096, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q074", + "question": "What is the warranty period on the estuary monitoring sensors?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2876, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q075", + "question": "How long are the coastal monitoring readings retained before deletion?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2749, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q076", + "question": "What encryption standard is used to transmit the river monitoring readings?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2746, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q077", + "question": "When was the last external audit of the lake monitoring programme?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2941, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q078", + "question": "What is the unit price of a river monitoring sediment sampler?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2789, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] } ] } diff --git a/docs/eval/baseline-v1.6.md b/docs/eval/baseline-v1.6.md index 65b9366..ffa7f59 100644 --- a/docs/eval/baseline-v1.6.md +++ b/docs/eval/baseline-v1.6.md @@ -12,7 +12,7 @@ Generated by `npm run eval`. The numbers below are harness output — do not edi | Ranks | `candidateK=20, threshold=0.5` | | Context width | `contextK=3` | | Corpus | `eval/corpus` (21 documents) | -| Split | `all` (63 questions, 53 answerable) | +| Split | `all` (78 questions, 53 answerable) | | Index size | 53 chunks | ## Metrics @@ -63,13 +63,13 @@ with a small second one means the threshold filters nothing and the window is al | Metric | Value | | --- | --- | -| Unanswerable questions | 10 | -| Retrieval abstained | 0.0000 (0/10) | +| Unanswerable questions | 25 | +| Retrieval abstained | 0.0000 (0/25) | | Mean candidates passing the threshold | 20.00 | | Mean passages in the context window | 3.00 | -Timing is informational only and is **not** frozen: indexing 2796 ms, query -p50 13.67 ms, p95 16.34 ms on the +Timing is informational only and is **not** frozen: indexing 5921 ms, query +p50 53.92 ms, p95 85.84 ms on the machine that produced this file. Timing and index size depend on hardware and on the corpus, so they must never be the reason two runs differ. diff --git a/docs/eval/scores-v1.6.json b/docs/eval/scores-v1.6.json index 163c779..4cfd04e 100644 --- a/docs/eval/scores-v1.6.json +++ b/docs/eval/scores-v1.6.json @@ -6,7 +6,7 @@ "indexSize": 53, "counts": { "answerable": 19, - "unanswerable": 5 + "unanswerable": 15 }, "distributions": { "bestRelevant": { @@ -47,12 +47,12 @@ }, "unanswerableMax": { "p0": 0.897848, - "p10": 0.897848, - "p25": 0.912751, - "p50": 0.925936, - "p75": 0.934276, + "p10": 0.908616, + "p25": 0.914179, + "p50": 0.927896, + "p75": 0.936526, "p90": 0.942441, - "p100": 0.942441 + "p100": 0.943454 } }, "byType": { @@ -186,7 +186,7 @@ "overlap": { "worstRelevantP10": 0.89056, "bestNonRelevantP90": 0.9386, - "unanswerableMaxP50": 0.925936, + "unanswerableMaxP50": 0.927896, "unanswerableMaxP90": 0.942441 }, "separable": false, @@ -308,7 +308,7 @@ "cosine": 0.8, "hitRate": 0.8947368421052632, "fullRecallRate": 0.8947368421052632, - "abstentionRate": 0.2 + "abstentionRate": 0.06666666666666667 }, { "threshold": 0.925, diff --git a/docs/eval/scores-v1.6.md b/docs/eval/scores-v1.6.md index 18d8928..6bff4b4 100644 --- a/docs/eval/scores-v1.6.md +++ b/docs/eval/scores-v1.6.md @@ -33,7 +33,7 @@ Dense only, `validation` split only, `threshold = 0`, `candidateK = 500` (above index size, so every chunk is scored for every query), `contextK = 3`. - answerable questions: 19 -- unanswerable questions: 5 +- unanswerable questions: 15 - index size: 53 chunks Hybrid is deliberately excluded: its `score` is an RRF value (`1 / (60 + rank)`) and is not @@ -61,7 +61,7 @@ Where a row shows a raw cosine in brackets: an **absolute** score maps as | worst relevant | 19 | 0.8808 (0.762) | 0.8906 (0.781) | 0.9086 (0.817) | 0.9355 (0.871) | 0.9428 (0.886) | 0.9530 (0.906) | 0.9537 (0.907) | | best non-relevant | 19 | 0.8904 (0.781) | 0.9081 (0.816) | 0.9178 (0.836) | 0.9243 (0.849) | 0.9336 (0.867) | 0.9386 (0.877) | 0.9459 (0.892) | | margin (best rel − best non-rel) | 19 | -0.0294 (-0.059) | -0.0272 (-0.054) | -0.0092 (-0.018) | 0.0026 (0.005) | 0.0081 (0.016) | 0.0329 (0.066) | 0.0623 (0.125) | -| unanswerable max candidate | 5 | 0.8978 (0.796) | 0.8978 (0.796) | 0.9128 (0.826) | 0.9259 (0.852) | 0.9343 (0.869) | 0.9424 (0.885) | 0.9424 (0.885) | +| unanswerable max candidate | 15 | 0.8978 (0.796) | 0.9086 (0.817) | 0.9142 (0.828) | 0.9279 (0.856) | 0.9365 (0.873) | 0.9424 (0.885) | 0.9435 (0.887) | ## By query type @@ -101,7 +101,7 @@ relevant passage; **full recall** = share whose every ground-truth block is stil | 0.825 | 0.650 | 1.0000 | 1.0000 | 0.0000 | | 0.850 | 0.700 | 1.0000 | 1.0000 | 0.0000 | | 0.875 | 0.750 | 1.0000 | 1.0000 | 0.0000 | -| 0.900 | 0.800 | 0.8947 | 0.8947 | 0.2000 | +| 0.900 | 0.800 | 0.8947 | 0.8947 | 0.0667 | | 0.925 | 0.850 | 0.6316 | 0.6316 | 0.4000 | | 0.950 | 0.900 | 0.1579 | 0.1579 | 1.0000 | | 0.975 | 0.950 | 0.0000 | 0.0000 | 1.0000 | @@ -115,7 +115,7 @@ The threshold decision turns on whether these overlap: | --- | --- | --- | | worst relevant, p10 | 0.8906 | 0.781 | | best non-relevant, p90 | 0.9386 | 0.877 | -| unanswerable max, p50 | 0.9259 | 0.852 | +| unanswerable max, p50 | 0.9279 | 0.856 | | unanswerable max, p90 | 0.9424 | 0.885 | **The distributions **overlap**, so a higher threshold buys abstention by giving up required relevant passages. If the curve above shows abstention rising only as full recall falls, then the honest conclusion is that **a single dense similarity threshold cannot carry both recall and abstention** — and the next mechanism to evaluate is not a finer threshold grid but a different signal (reranker score, top1−top2 margin, per-query thresholds, or claim-level answerability). diff --git a/docs/eval/sweep-v1.6.json b/docs/eval/sweep-v1.6.json index 0a47cf1..ba157c1 100644 --- a/docs/eval/sweep-v1.6.json +++ b/docs/eval/sweep-v1.6.json @@ -12,12 +12,12 @@ "contextRecall": 0.816038, "noResultRate": 0, "meanContextChars": 2311.811320754717, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 5, "unanswerableContextPassages": 3, "chunkCount": 53, - "latencyP95Ms": 12.3 + "latencyP95Ms": 16.18 }, { "strategy": "dense", @@ -30,12 +30,12 @@ "contextRecall": 0.877358, "noResultRate": 0, "meanContextChars": 4007.0377358490564, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 5, "unanswerableContextPassages": 5, "chunkCount": 53, - "latencyP95Ms": 12.37 + "latencyP95Ms": 86.91 }, { "strategy": "dense", @@ -48,12 +48,12 @@ "contextRecall": 0.816038, "noResultRate": 0, "meanContextChars": 2311.811320754717, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 10, "unanswerableContextPassages": 3, "chunkCount": 53, - "latencyP95Ms": 12.66 + "latencyP95Ms": 100.17 }, { "strategy": "dense", @@ -66,12 +66,12 @@ "contextRecall": 0.877358, "noResultRate": 0, "meanContextChars": 4007.0377358490564, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 10, "unanswerableContextPassages": 5, "chunkCount": 53, - "latencyP95Ms": 13.42 + "latencyP95Ms": 81.38 }, { "strategy": "dense", @@ -84,12 +84,12 @@ "contextRecall": 0.877358, "noResultRate": 0, "meanContextChars": 6658.698113207547, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 10, "unanswerableContextPassages": 8, "chunkCount": 53, - "latencyP95Ms": 12.8 + "latencyP95Ms": 18.17 }, { "strategy": "dense", @@ -102,12 +102,12 @@ "contextRecall": 0.816038, "noResultRate": 0, "meanContextChars": 2311.811320754717, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 20, "unanswerableContextPassages": 3, "chunkCount": 53, - "latencyP95Ms": 13.23 + "latencyP95Ms": 17.01 }, { "strategy": "dense", @@ -120,12 +120,12 @@ "contextRecall": 0.877358, "noResultRate": 0, "meanContextChars": 4007.0377358490564, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 20, "unanswerableContextPassages": 5, "chunkCount": 53, - "latencyP95Ms": 13.55 + "latencyP95Ms": 97.74 }, { "strategy": "dense", @@ -138,12 +138,12 @@ "contextRecall": 0.877358, "noResultRate": 0, "meanContextChars": 6658.698113207547, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 20, "unanswerableContextPassages": 8, "chunkCount": 53, - "latencyP95Ms": 13.75 + "latencyP95Ms": 85.3 }, { "strategy": "dense", @@ -156,12 +156,12 @@ "contextRecall": 0.816038, "noResultRate": 0, "meanContextChars": 2311.811320754717, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 40, "unanswerableContextPassages": 3, "chunkCount": 53, - "latencyP95Ms": 15.44 + "latencyP95Ms": 57.28 }, { "strategy": "dense", @@ -174,12 +174,12 @@ "contextRecall": 0.877358, "noResultRate": 0, "meanContextChars": 4007.0377358490564, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 40, "unanswerableContextPassages": 5, "chunkCount": 53, - "latencyP95Ms": 14.98 + "latencyP95Ms": 16.64 }, { "strategy": "dense", @@ -192,12 +192,12 @@ "contextRecall": 0.877358, "noResultRate": 0, "meanContextChars": 6658.698113207547, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 40, "unanswerableContextPassages": 8, "chunkCount": 53, - "latencyP95Ms": 14.68 + "latencyP95Ms": 99.32 }, { "strategy": "hybrid", @@ -210,12 +210,12 @@ "contextRecall": 0.853774, "noResultRate": 0, "meanContextChars": 2307.3207547169814, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 5, "unanswerableContextPassages": 3, "chunkCount": 53, - "latencyP95Ms": 13.45 + "latencyP95Ms": 90.4 }, { "strategy": "hybrid", @@ -228,12 +228,12 @@ "contextRecall": 0.910377, "noResultRate": 0, "meanContextChars": 3936.566037735849, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 5, "unanswerableContextPassages": 5, "chunkCount": 53, - "latencyP95Ms": 77.74 + "latencyP95Ms": 89.63 }, { "strategy": "hybrid", @@ -246,12 +246,12 @@ "contextRecall": 0.853774, "noResultRate": 0, "meanContextChars": 2321.264150943396, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 10, "unanswerableContextPassages": 3, "chunkCount": 53, - "latencyP95Ms": 14.42 + "latencyP95Ms": 83.59 }, { "strategy": "hybrid", @@ -263,13 +263,13 @@ "contextPrecision": 0.203774, "contextRecall": 0.896226, "noResultRate": 0, - "meanContextChars": 3994.509433962264, - "unanswerableQuestions": 10, + "meanContextChars": 3985.811320754717, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 10, "unanswerableContextPassages": 5, "chunkCount": 53, - "latencyP95Ms": 13.8 + "latencyP95Ms": 89.79 }, { "strategy": "hybrid", @@ -282,12 +282,12 @@ "contextRecall": 0.900943, "noResultRate": 0, "meanContextChars": 6547.641509433963, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 10, "unanswerableContextPassages": 8, "chunkCount": 53, - "latencyP95Ms": 13.04 + "latencyP95Ms": 14.96 }, { "strategy": "hybrid", @@ -300,12 +300,12 @@ "contextRecall": 0.853774, "noResultRate": 0, "meanContextChars": 2322.830188679245, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 20, "unanswerableContextPassages": 3, "chunkCount": 53, - "latencyP95Ms": 15.55 + "latencyP95Ms": 82.31 }, { "strategy": "hybrid", @@ -318,12 +318,12 @@ "contextRecall": 0.896226, "noResultRate": 0, "meanContextChars": 4011.377358490566, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 20, "unanswerableContextPassages": 5, "chunkCount": 53, - "latencyP95Ms": 16.36 + "latencyP95Ms": 18.62 }, { "strategy": "hybrid", @@ -336,12 +336,12 @@ "contextRecall": 0.90566, "noResultRate": 0, "meanContextChars": 6660.264150943396, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 20, "unanswerableContextPassages": 8, "chunkCount": 53, - "latencyP95Ms": 14.66 + "latencyP95Ms": 16.13 }, { "strategy": "hybrid", @@ -354,12 +354,12 @@ "contextRecall": 0.853774, "noResultRate": 0, "meanContextChars": 2322.830188679245, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 40, "unanswerableContextPassages": 3, "chunkCount": 53, - "latencyP95Ms": 14.55 + "latencyP95Ms": 97.73 }, { "strategy": "hybrid", @@ -372,12 +372,12 @@ "contextRecall": 0.896226, "noResultRate": 0, "meanContextChars": 3989.735849056604, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 40, "unanswerableContextPassages": 5, "chunkCount": 53, - "latencyP95Ms": 14.41 + "latencyP95Ms": 100.05 }, { "strategy": "hybrid", @@ -390,12 +390,12 @@ "contextRecall": 0.90566, "noResultRate": 0, "meanContextChars": 6660.792452830188, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 40, "unanswerableContextPassages": 8, "chunkCount": 53, - "latencyP95Ms": 14.47 + "latencyP95Ms": 16.94 } ] } diff --git a/docs/eval/sweep-v1.6.md b/docs/eval/sweep-v1.6.md index 46aeb3e..af8cb92 100644 --- a/docs/eval/sweep-v1.6.md +++ b/docs/eval/sweep-v1.6.md @@ -12,28 +12,28 @@ Each row differs from its neighbour in one parameter. | Strategy | candidateK | contextK | Recall@5 | nDCG@10 | MAP@10 | Context P | Context R | No-result | Context chars | Unans. abstained | Unans. cands | Unans. ctx | Index | p95 | | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | -| dense | 5 | 3 | 0.8774 | 0.7672 | 0.7231 | 0.3082 | 0.8160 | 0.0000 | 2312 | 0.0000 | 5.0 | 3.0 | 53 | 12.30 ms | -| dense | 5 | 5 | 0.8774 | 0.7672 | 0.7231 | 0.2000 | 0.8774 | 0.0000 | 4007 | 0.0000 | 5.0 | 5.0 | 53 | 12.37 ms | -| dense | 10 | 3 | 0.8774 | 0.7804 | 0.7280 | 0.3082 | 0.8160 | 0.0000 | 2312 | 0.0000 | 10.0 | 3.0 | 53 | 12.66 ms | -| dense | 10 | 5 | 0.8774 | 0.7804 | 0.7280 | 0.2000 | 0.8774 | 0.0000 | 4007 | 0.0000 | 10.0 | 5.0 | 53 | 13.42 ms | -| dense | 10 | 8 | 0.8774 | 0.7804 | 0.7280 | 0.1274 | 0.8774 | 0.0000 | 6659 | 0.0000 | 10.0 | 8.0 | 53 | 12.80 ms | -| dense | 20 | 3 | 0.8774 | 0.7804 | 0.7280 | 0.3082 | 0.8160 | 0.0000 | 2312 | 0.0000 | 20.0 | 3.0 | 53 | 13.23 ms | -| dense | 20 | 5 | 0.8774 | 0.7804 | 0.7280 | 0.2000 | 0.8774 | 0.0000 | 4007 | 0.0000 | 20.0 | 5.0 | 53 | 13.55 ms | -| dense | 20 | 8 | 0.8774 | 0.7804 | 0.7280 | 0.1274 | 0.8774 | 0.0000 | 6659 | 0.0000 | 20.0 | 8.0 | 53 | 13.75 ms | -| dense | 40 | 3 | 0.8774 | 0.7804 | 0.7280 | 0.3082 | 0.8160 | 0.0000 | 2312 | 0.0000 | 40.0 | 3.0 | 53 | 15.44 ms | -| dense | 40 | 5 | 0.8774 | 0.7804 | 0.7280 | 0.2000 | 0.8774 | 0.0000 | 4007 | 0.0000 | 40.0 | 5.0 | 53 | 14.98 ms | -| dense | 40 | 8 | 0.8774 | 0.7804 | 0.7280 | 0.1274 | 0.8774 | 0.0000 | 6659 | 0.0000 | 40.0 | 8.0 | 53 | 14.68 ms | -| hybrid | 5 | 3 | 0.9104 | 0.8101 | 0.7682 | 0.3208 | 0.8538 | 0.0000 | 2307 | 0.0000 | 5.0 | 3.0 | 53 | 13.45 ms | -| hybrid | 5 | 5 | 0.9104 | 0.8101 | 0.7682 | 0.2113 | 0.9104 | 0.0000 | 3937 | 0.0000 | 5.0 | 5.0 | 53 | 77.74 ms | -| hybrid | 10 | 3 | 0.8962 | 0.8068 | 0.7580 | 0.3208 | 0.8538 | 0.0000 | 2321 | 0.0000 | 10.0 | 3.0 | 53 | 14.42 ms | -| hybrid | 10 | 5 | 0.8962 | 0.8068 | 0.7580 | 0.2038 | 0.8962 | 0.0000 | 3995 | 0.0000 | 10.0 | 5.0 | 53 | 13.80 ms | -| hybrid | 10 | 8 | 0.8962 | 0.8068 | 0.7580 | 0.1321 | 0.9009 | 0.0000 | 6548 | 0.0000 | 10.0 | 8.0 | 53 | 13.04 ms | -| hybrid | 20 | 3 | 0.8962 | 0.8098 | 0.7605 | 0.3208 | 0.8538 | 0.0000 | 2323 | 0.0000 | 20.0 | 3.0 | 53 | 15.55 ms | -| hybrid | 20 | 5 | 0.8962 | 0.8098 | 0.7605 | 0.2038 | 0.8962 | 0.0000 | 4011 | 0.0000 | 20.0 | 5.0 | 53 | 16.36 ms | -| hybrid | 20 | 8 | 0.8962 | 0.8098 | 0.7605 | 0.1321 | 0.9057 | 0.0000 | 6660 | 0.0000 | 20.0 | 8.0 | 53 | 14.66 ms | -| hybrid | 40 | 3 | 0.8962 | 0.8098 | 0.7605 | 0.3208 | 0.8538 | 0.0000 | 2323 | 0.0000 | 40.0 | 3.0 | 53 | 14.55 ms | -| hybrid | 40 | 5 | 0.8962 | 0.8098 | 0.7605 | 0.2038 | 0.8962 | 0.0000 | 3990 | 0.0000 | 40.0 | 5.0 | 53 | 14.41 ms | -| hybrid | 40 | 8 | 0.8962 | 0.8098 | 0.7605 | 0.1321 | 0.9057 | 0.0000 | 6661 | 0.0000 | 40.0 | 8.0 | 53 | 14.47 ms | +| dense | 5 | 3 | 0.8774 | 0.7672 | 0.7231 | 0.3082 | 0.8160 | 0.0000 | 2312 | 0.0000 | 5.0 | 3.0 | 53 | 16.18 ms | +| dense | 5 | 5 | 0.8774 | 0.7672 | 0.7231 | 0.2000 | 0.8774 | 0.0000 | 4007 | 0.0000 | 5.0 | 5.0 | 53 | 86.91 ms | +| dense | 10 | 3 | 0.8774 | 0.7804 | 0.7280 | 0.3082 | 0.8160 | 0.0000 | 2312 | 0.0000 | 10.0 | 3.0 | 53 | 100.17 ms | +| dense | 10 | 5 | 0.8774 | 0.7804 | 0.7280 | 0.2000 | 0.8774 | 0.0000 | 4007 | 0.0000 | 10.0 | 5.0 | 53 | 81.38 ms | +| dense | 10 | 8 | 0.8774 | 0.7804 | 0.7280 | 0.1274 | 0.8774 | 0.0000 | 6659 | 0.0000 | 10.0 | 8.0 | 53 | 18.17 ms | +| dense | 20 | 3 | 0.8774 | 0.7804 | 0.7280 | 0.3082 | 0.8160 | 0.0000 | 2312 | 0.0000 | 20.0 | 3.0 | 53 | 17.01 ms | +| dense | 20 | 5 | 0.8774 | 0.7804 | 0.7280 | 0.2000 | 0.8774 | 0.0000 | 4007 | 0.0000 | 20.0 | 5.0 | 53 | 97.74 ms | +| dense | 20 | 8 | 0.8774 | 0.7804 | 0.7280 | 0.1274 | 0.8774 | 0.0000 | 6659 | 0.0000 | 20.0 | 8.0 | 53 | 85.30 ms | +| dense | 40 | 3 | 0.8774 | 0.7804 | 0.7280 | 0.3082 | 0.8160 | 0.0000 | 2312 | 0.0000 | 40.0 | 3.0 | 53 | 57.28 ms | +| dense | 40 | 5 | 0.8774 | 0.7804 | 0.7280 | 0.2000 | 0.8774 | 0.0000 | 4007 | 0.0000 | 40.0 | 5.0 | 53 | 16.64 ms | +| dense | 40 | 8 | 0.8774 | 0.7804 | 0.7280 | 0.1274 | 0.8774 | 0.0000 | 6659 | 0.0000 | 40.0 | 8.0 | 53 | 99.32 ms | +| hybrid | 5 | 3 | 0.9104 | 0.8101 | 0.7682 | 0.3208 | 0.8538 | 0.0000 | 2307 | 0.0000 | 5.0 | 3.0 | 53 | 90.40 ms | +| hybrid | 5 | 5 | 0.9104 | 0.8101 | 0.7682 | 0.2113 | 0.9104 | 0.0000 | 3937 | 0.0000 | 5.0 | 5.0 | 53 | 89.63 ms | +| hybrid | 10 | 3 | 0.8962 | 0.8068 | 0.7580 | 0.3208 | 0.8538 | 0.0000 | 2321 | 0.0000 | 10.0 | 3.0 | 53 | 83.59 ms | +| hybrid | 10 | 5 | 0.8962 | 0.8068 | 0.7580 | 0.2038 | 0.8962 | 0.0000 | 3986 | 0.0000 | 10.0 | 5.0 | 53 | 89.79 ms | +| hybrid | 10 | 8 | 0.8962 | 0.8068 | 0.7580 | 0.1321 | 0.9009 | 0.0000 | 6548 | 0.0000 | 10.0 | 8.0 | 53 | 14.96 ms | +| hybrid | 20 | 3 | 0.8962 | 0.8098 | 0.7605 | 0.3208 | 0.8538 | 0.0000 | 2323 | 0.0000 | 20.0 | 3.0 | 53 | 82.31 ms | +| hybrid | 20 | 5 | 0.8962 | 0.8098 | 0.7605 | 0.2038 | 0.8962 | 0.0000 | 4011 | 0.0000 | 20.0 | 5.0 | 53 | 18.62 ms | +| hybrid | 20 | 8 | 0.8962 | 0.8098 | 0.7605 | 0.1321 | 0.9057 | 0.0000 | 6660 | 0.0000 | 20.0 | 8.0 | 53 | 16.13 ms | +| hybrid | 40 | 3 | 0.8962 | 0.8098 | 0.7605 | 0.3208 | 0.8538 | 0.0000 | 2323 | 0.0000 | 40.0 | 3.0 | 53 | 97.73 ms | +| hybrid | 40 | 5 | 0.8962 | 0.8098 | 0.7605 | 0.2038 | 0.8962 | 0.0000 | 3990 | 0.0000 | 40.0 | 5.0 | 53 | 100.05 ms | +| hybrid | 40 | 8 | 0.8962 | 0.8098 | 0.7605 | 0.1321 | 0.9057 | 0.0000 | 6661 | 0.0000 | 40.0 | 8.0 | 53 | 16.94 ms | ## Skipped cells diff --git a/docs/eval/threshold-v1.6.json b/docs/eval/threshold-v1.6.json index 15d8562..fbf9e89 100644 --- a/docs/eval/threshold-v1.6.json +++ b/docs/eval/threshold-v1.6.json @@ -7,13 +7,13 @@ { "threshold": 0, "validation": { - "questions": 24, + "questions": 34, "answerableQuestions": 19, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 20, "unanswerable": { - "questions": 5, + "questions": 15, "abstentionCount": 0, "retrievalAbstentionRate": 0, "meanCandidatesRetrieved": 20, @@ -32,13 +32,13 @@ } }, "test": { - "questions": 39, + "questions": 44, "answerableQuestions": 34, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 20, "unanswerable": { - "questions": 5, + "questions": 10, "abstentionCount": 0, "retrievalAbstentionRate": 0, "meanCandidatesRetrieved": 20, @@ -60,13 +60,13 @@ { "threshold": 0.3, "validation": { - "questions": 24, + "questions": 34, "answerableQuestions": 19, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 20, "unanswerable": { - "questions": 5, + "questions": 15, "abstentionCount": 0, "retrievalAbstentionRate": 0, "meanCandidatesRetrieved": 20, @@ -85,13 +85,13 @@ } }, "test": { - "questions": 39, + "questions": 44, "answerableQuestions": 34, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 20, "unanswerable": { - "questions": 5, + "questions": 10, "abstentionCount": 0, "retrievalAbstentionRate": 0, "meanCandidatesRetrieved": 20, @@ -113,13 +113,13 @@ { "threshold": 0.4, "validation": { - "questions": 24, + "questions": 34, "answerableQuestions": 19, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 20, "unanswerable": { - "questions": 5, + "questions": 15, "abstentionCount": 0, "retrievalAbstentionRate": 0, "meanCandidatesRetrieved": 20, @@ -138,13 +138,13 @@ } }, "test": { - "questions": 39, + "questions": 44, "answerableQuestions": 34, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 20, "unanswerable": { - "questions": 5, + "questions": 10, "abstentionCount": 0, "retrievalAbstentionRate": 0, "meanCandidatesRetrieved": 20, @@ -166,13 +166,13 @@ { "threshold": 0.5, "validation": { - "questions": 24, + "questions": 34, "answerableQuestions": 19, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 20, "unanswerable": { - "questions": 5, + "questions": 15, "abstentionCount": 0, "retrievalAbstentionRate": 0, "meanCandidatesRetrieved": 20, @@ -191,13 +191,13 @@ } }, "test": { - "questions": 39, + "questions": 44, "answerableQuestions": 34, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 20, "unanswerable": { - "questions": 5, + "questions": 10, "abstentionCount": 0, "retrievalAbstentionRate": 0, "meanCandidatesRetrieved": 20, @@ -219,13 +219,13 @@ { "threshold": 0.6, "validation": { - "questions": 24, + "questions": 34, "answerableQuestions": 19, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 20, "unanswerable": { - "questions": 5, + "questions": 15, "abstentionCount": 0, "retrievalAbstentionRate": 0, "meanCandidatesRetrieved": 20, @@ -244,13 +244,13 @@ } }, "test": { - "questions": 39, + "questions": 44, "answerableQuestions": 34, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 20, "unanswerable": { - "questions": 5, + "questions": 10, "abstentionCount": 0, "retrievalAbstentionRate": 0, "meanCandidatesRetrieved": 20, diff --git a/docs/eval/threshold-v1.6.md b/docs/eval/threshold-v1.6.md index 3997609..6dcc74e 100644 --- a/docs/eval/threshold-v1.6.md +++ b/docs/eval/threshold-v1.6.md @@ -34,7 +34,7 @@ retrieval loss. ## Outcome -The sweep is **flat**: every threshold from 0 to 0.6 produces the same validation nDCG@10 (0.7556), the same Recall@5 (0.8684) and the same retrieval abstention rate on unanswerable questions (0/5). No candidate is ever filtered out, so the threshold is **non-binding** on this corpus. +The sweep is **flat**: every threshold from 0 to 0.6 produces the same validation nDCG@10 (0.7556), the same Recall@5 (0.8684) and the same retrieval abstention rate on unanswerable questions (0/15). No candidate is ever filtered out, so the threshold is **non-binding** on this corpus. **No evidence to change `threshold = 0.5`.** All thresholds hold the line equally; picking one would be arbitrary. The current value can be neither validated nor falsified here, which is a property of the corpus rather than of the threshold. Note also what this does *not* establish: abstention is a retrieval-layer statement — whether the model then declines to answer needs a generator eval. diff --git a/eval/questions.jsonl b/eval/questions.jsonl index 10b031a..44fa762 100644 --- a/eval/questions.jsonl +++ b/eval/questions.jsonl @@ -61,3 +61,18 @@ {"id":"q061","type":"semantic","question":"Why does the estuary programme take more replicate samples than the other monitoring programmes?","relevant":[{"document":"estuary-monitoring.md","page":null,"block":8,"quote":"within-station variance"}]} {"id":"q062","type":"unanswerable","answerable":false,"question":"What is the annual operating cost of the coastal monitoring buoys?","relevant":[]} {"id":"q063","type":"unanswerable","answerable":false,"question":"How many litres per second does the reservoir release downstream on a typical day?","relevant":[]} +{"id":"q064","question":"How much GPU memory does Gateway API v3 need to serve a request?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q065","question":"What is the uptime SLA for Gateway API v2?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q066","question":"How long does an authentication token stay valid before it expires?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q067","question":"Which SDK version should a client use with Gateway API v3?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q068","question":"Can Gateway API v2 be self-hosted on-premise?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q069","question":"How is usage invoiced for Gateway API v2?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q070","question":"Which laboratory is accredited to analyse the estuary transect samples?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q071","question":"Which vendor supplies the coastal monitoring buoys?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q072","question":"What is the annual funding for the groundwater monitoring programme?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q073","question":"How many staff work on the reservoir monitoring programme?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q074","question":"What is the warranty period on the estuary monitoring sensors?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q075","question":"How long are the coastal monitoring readings retained before deletion?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q076","question":"What encryption standard is used to transmit the river monitoring readings?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q077","question":"When was the last external audit of the lake monitoring programme?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q078","question":"What is the unit price of a river monitoring sediment sampler?","type":"unanswerable","answerable":false,"relevant":[]} diff --git a/eval/splits.json b/eval/splits.json index 021e265..6e13b15 100644 --- a/eval/splits.json +++ b/eval/splits.json @@ -61,5 +61,20 @@ "q060": "validation", "q061": "test", "q062": "validation", - "q063": "test" + "q063": "test", + "q064": "validation", + "q065": "validation", + "q066": "validation", + "q067": "test", + "q068": "validation", + "q069": "test", + "q070": "validation", + "q071": "validation", + "q072": "test", + "q073": "validation", + "q074": "test", + "q075": "validation", + "q076": "validation", + "q077": "test", + "q078": "validation" }