diff --git a/docs/eval/baseline-v1.6.json b/docs/eval/baseline-v1.6.json index d8171a1..2bddcea 100644 --- a/docs/eval/baseline-v1.6.json +++ b/docs/eval/baseline-v1.6.json @@ -17,7 +17,7 @@ "threshold": 0.5, "corpus": "eval/corpus", "documents": 21, - "questions": 63, + "questions": 78, "chunkCount": 53 }, "metrics": { @@ -124,7 +124,7 @@ } ], "unanswerable": { - "questions": 10, + "questions": 25, "abstentionCount": 0, "retrievalAbstentionRate": 0, "meanCandidatesRetrieved": 20, @@ -2277,6 +2277,486 @@ [], [] ] + }, + { + "id": "q064", + "question": "How much GPU memory does Gateway API v3 need to serve a request?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2908, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q065", + "question": "What is the uptime SLA for Gateway API v2?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2875, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q066", + "question": "How long does an authentication token stay valid before it expires?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2649, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q067", + "question": "Which SDK version should a client use with Gateway API v3?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2908, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q068", + "question": "Can Gateway API v2 be self-hosted on-premise?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2875, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q069", + "question": "How is usage invoiced for Gateway API v2?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2875, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q070", + "question": "Which laboratory is accredited to analyse the estuary transect samples?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2902, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q071", + "question": "Which vendor supplies the coastal monitoring buoys?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2826, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q072", + "question": "What is the annual funding for the groundwater monitoring programme?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2933, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q073", + "question": "How many staff work on the reservoir monitoring programme?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2096, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q074", + "question": "What is the warranty period on the estuary monitoring sensors?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2876, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q075", + "question": "How long are the coastal monitoring readings retained before deletion?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2749, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q076", + "question": "What encryption standard is used to transmit the river monitoring readings?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2746, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q077", + "question": "When was the last external audit of the lake monitoring programme?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2941, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q078", + "question": "What is the unit price of a river monitoring sediment sampler?", + "type": "unanswerable", + "answerable": false, + "firstRelevantRank": 0, + "relevantCount": 0, + "retrievedCount": 20, + "contextChars": 2789, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] } ] } diff --git a/docs/eval/baseline-v1.6.md b/docs/eval/baseline-v1.6.md index 65b9366..ffa7f59 100644 --- a/docs/eval/baseline-v1.6.md +++ b/docs/eval/baseline-v1.6.md @@ -12,7 +12,7 @@ Generated by `npm run eval`. The numbers below are harness output — do not edi | Ranks | `candidateK=20, threshold=0.5` | | Context width | `contextK=3` | | Corpus | `eval/corpus` (21 documents) | -| Split | `all` (63 questions, 53 answerable) | +| Split | `all` (78 questions, 53 answerable) | | Index size | 53 chunks | ## Metrics @@ -63,13 +63,13 @@ with a small second one means the threshold filters nothing and the window is al | Metric | Value | | --- | --- | -| Unanswerable questions | 10 | -| Retrieval abstained | 0.0000 (0/10) | +| Unanswerable questions | 25 | +| Retrieval abstained | 0.0000 (0/25) | | Mean candidates passing the threshold | 20.00 | | Mean passages in the context window | 3.00 | -Timing is informational only and is **not** frozen: indexing 2796 ms, query -p50 13.67 ms, p95 16.34 ms on the +Timing is informational only and is **not** frozen: indexing 5921 ms, query +p50 53.92 ms, p95 85.84 ms on the machine that produced this file. Timing and index size depend on hardware and on the corpus, so they must never be the reason two runs differ. diff --git a/docs/eval/scores-v1.6.json b/docs/eval/scores-v1.6.json index 163c779..4cfd04e 100644 --- a/docs/eval/scores-v1.6.json +++ b/docs/eval/scores-v1.6.json @@ -6,7 +6,7 @@ "indexSize": 53, "counts": { "answerable": 19, - "unanswerable": 5 + "unanswerable": 15 }, "distributions": { "bestRelevant": { @@ -47,12 +47,12 @@ }, "unanswerableMax": { "p0": 0.897848, - "p10": 0.897848, - "p25": 0.912751, - "p50": 0.925936, - "p75": 0.934276, + "p10": 0.908616, + "p25": 0.914179, + "p50": 0.927896, + "p75": 0.936526, "p90": 0.942441, - "p100": 0.942441 + "p100": 0.943454 } }, "byType": { @@ -186,7 +186,7 @@ "overlap": { "worstRelevantP10": 0.89056, "bestNonRelevantP90": 0.9386, - "unanswerableMaxP50": 0.925936, + "unanswerableMaxP50": 0.927896, "unanswerableMaxP90": 0.942441 }, "separable": false, @@ -308,7 +308,7 @@ "cosine": 0.8, "hitRate": 0.8947368421052632, "fullRecallRate": 0.8947368421052632, - "abstentionRate": 0.2 + "abstentionRate": 0.06666666666666667 }, { "threshold": 0.925, diff --git a/docs/eval/scores-v1.6.md b/docs/eval/scores-v1.6.md index 18d8928..6bff4b4 100644 --- a/docs/eval/scores-v1.6.md +++ b/docs/eval/scores-v1.6.md @@ -33,7 +33,7 @@ Dense only, `validation` split only, `threshold = 0`, `candidateK = 500` (above index size, so every chunk is scored for every query), `contextK = 3`. - answerable questions: 19 -- unanswerable questions: 5 +- unanswerable questions: 15 - index size: 53 chunks Hybrid is deliberately excluded: its `score` is an RRF value (`1 / (60 + rank)`) and is not @@ -61,7 +61,7 @@ Where a row shows a raw cosine in brackets: an **absolute** score maps as | worst relevant | 19 | 0.8808 (0.762) | 0.8906 (0.781) | 0.9086 (0.817) | 0.9355 (0.871) | 0.9428 (0.886) | 0.9530 (0.906) | 0.9537 (0.907) | | best non-relevant | 19 | 0.8904 (0.781) | 0.9081 (0.816) | 0.9178 (0.836) | 0.9243 (0.849) | 0.9336 (0.867) | 0.9386 (0.877) | 0.9459 (0.892) | | margin (best rel − best non-rel) | 19 | -0.0294 (-0.059) | -0.0272 (-0.054) | -0.0092 (-0.018) | 0.0026 (0.005) | 0.0081 (0.016) | 0.0329 (0.066) | 0.0623 (0.125) | -| unanswerable max candidate | 5 | 0.8978 (0.796) | 0.8978 (0.796) | 0.9128 (0.826) | 0.9259 (0.852) | 0.9343 (0.869) | 0.9424 (0.885) | 0.9424 (0.885) | +| unanswerable max candidate | 15 | 0.8978 (0.796) | 0.9086 (0.817) | 0.9142 (0.828) | 0.9279 (0.856) | 0.9365 (0.873) | 0.9424 (0.885) | 0.9435 (0.887) | ## By query type @@ -101,7 +101,7 @@ relevant passage; **full recall** = share whose every ground-truth block is stil | 0.825 | 0.650 | 1.0000 | 1.0000 | 0.0000 | | 0.850 | 0.700 | 1.0000 | 1.0000 | 0.0000 | | 0.875 | 0.750 | 1.0000 | 1.0000 | 0.0000 | -| 0.900 | 0.800 | 0.8947 | 0.8947 | 0.2000 | +| 0.900 | 0.800 | 0.8947 | 0.8947 | 0.0667 | | 0.925 | 0.850 | 0.6316 | 0.6316 | 0.4000 | | 0.950 | 0.900 | 0.1579 | 0.1579 | 1.0000 | | 0.975 | 0.950 | 0.0000 | 0.0000 | 1.0000 | @@ -115,7 +115,7 @@ The threshold decision turns on whether these overlap: | --- | --- | --- | | worst relevant, p10 | 0.8906 | 0.781 | | best non-relevant, p90 | 0.9386 | 0.877 | -| unanswerable max, p50 | 0.9259 | 0.852 | +| unanswerable max, p50 | 0.9279 | 0.856 | | unanswerable max, p90 | 0.9424 | 0.885 | **The distributions **overlap**, so a higher threshold buys abstention by giving up required relevant passages. If the curve above shows abstention rising only as full recall falls, then the honest conclusion is that **a single dense similarity threshold cannot carry both recall and abstention** — and the next mechanism to evaluate is not a finer threshold grid but a different signal (reranker score, top1−top2 margin, per-query thresholds, or claim-level answerability). diff --git a/docs/eval/sweep-v1.6.json b/docs/eval/sweep-v1.6.json index 0a47cf1..ba157c1 100644 --- a/docs/eval/sweep-v1.6.json +++ b/docs/eval/sweep-v1.6.json @@ -12,12 +12,12 @@ "contextRecall": 0.816038, "noResultRate": 0, "meanContextChars": 2311.811320754717, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 5, "unanswerableContextPassages": 3, "chunkCount": 53, - "latencyP95Ms": 12.3 + "latencyP95Ms": 16.18 }, { "strategy": "dense", @@ -30,12 +30,12 @@ "contextRecall": 0.877358, "noResultRate": 0, "meanContextChars": 4007.0377358490564, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 5, "unanswerableContextPassages": 5, "chunkCount": 53, - "latencyP95Ms": 12.37 + "latencyP95Ms": 86.91 }, { "strategy": "dense", @@ -48,12 +48,12 @@ "contextRecall": 0.816038, "noResultRate": 0, "meanContextChars": 2311.811320754717, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 10, "unanswerableContextPassages": 3, "chunkCount": 53, - "latencyP95Ms": 12.66 + "latencyP95Ms": 100.17 }, { "strategy": "dense", @@ -66,12 +66,12 @@ "contextRecall": 0.877358, "noResultRate": 0, "meanContextChars": 4007.0377358490564, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 10, "unanswerableContextPassages": 5, "chunkCount": 53, - "latencyP95Ms": 13.42 + "latencyP95Ms": 81.38 }, { "strategy": "dense", @@ -84,12 +84,12 @@ "contextRecall": 0.877358, "noResultRate": 0, "meanContextChars": 6658.698113207547, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 10, "unanswerableContextPassages": 8, "chunkCount": 53, - "latencyP95Ms": 12.8 + "latencyP95Ms": 18.17 }, { "strategy": "dense", @@ -102,12 +102,12 @@ "contextRecall": 0.816038, "noResultRate": 0, "meanContextChars": 2311.811320754717, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 20, "unanswerableContextPassages": 3, "chunkCount": 53, - "latencyP95Ms": 13.23 + "latencyP95Ms": 17.01 }, { "strategy": "dense", @@ -120,12 +120,12 @@ "contextRecall": 0.877358, "noResultRate": 0, "meanContextChars": 4007.0377358490564, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 20, "unanswerableContextPassages": 5, "chunkCount": 53, - "latencyP95Ms": 13.55 + "latencyP95Ms": 97.74 }, { "strategy": "dense", @@ -138,12 +138,12 @@ "contextRecall": 0.877358, "noResultRate": 0, "meanContextChars": 6658.698113207547, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 20, "unanswerableContextPassages": 8, "chunkCount": 53, - "latencyP95Ms": 13.75 + "latencyP95Ms": 85.3 }, { "strategy": "dense", @@ -156,12 +156,12 @@ "contextRecall": 0.816038, "noResultRate": 0, "meanContextChars": 2311.811320754717, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 40, "unanswerableContextPassages": 3, "chunkCount": 53, - "latencyP95Ms": 15.44 + "latencyP95Ms": 57.28 }, { "strategy": "dense", @@ -174,12 +174,12 @@ "contextRecall": 0.877358, "noResultRate": 0, "meanContextChars": 4007.0377358490564, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 40, "unanswerableContextPassages": 5, "chunkCount": 53, - "latencyP95Ms": 14.98 + "latencyP95Ms": 16.64 }, { "strategy": "dense", @@ -192,12 +192,12 @@ "contextRecall": 0.877358, "noResultRate": 0, "meanContextChars": 6658.698113207547, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 40, "unanswerableContextPassages": 8, "chunkCount": 53, - "latencyP95Ms": 14.68 + "latencyP95Ms": 99.32 }, { "strategy": "hybrid", @@ -210,12 +210,12 @@ "contextRecall": 0.853774, "noResultRate": 0, "meanContextChars": 2307.3207547169814, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 5, "unanswerableContextPassages": 3, "chunkCount": 53, - "latencyP95Ms": 13.45 + "latencyP95Ms": 90.4 }, { "strategy": "hybrid", @@ -228,12 +228,12 @@ "contextRecall": 0.910377, "noResultRate": 0, "meanContextChars": 3936.566037735849, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 5, "unanswerableContextPassages": 5, "chunkCount": 53, - "latencyP95Ms": 77.74 + "latencyP95Ms": 89.63 }, { "strategy": "hybrid", @@ -246,12 +246,12 @@ "contextRecall": 0.853774, "noResultRate": 0, "meanContextChars": 2321.264150943396, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 10, "unanswerableContextPassages": 3, "chunkCount": 53, - "latencyP95Ms": 14.42 + "latencyP95Ms": 83.59 }, { "strategy": "hybrid", @@ -263,13 +263,13 @@ "contextPrecision": 0.203774, "contextRecall": 0.896226, "noResultRate": 0, - "meanContextChars": 3994.509433962264, - "unanswerableQuestions": 10, + "meanContextChars": 3985.811320754717, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 10, "unanswerableContextPassages": 5, "chunkCount": 53, - "latencyP95Ms": 13.8 + "latencyP95Ms": 89.79 }, { "strategy": "hybrid", @@ -282,12 +282,12 @@ "contextRecall": 0.900943, "noResultRate": 0, "meanContextChars": 6547.641509433963, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 10, "unanswerableContextPassages": 8, "chunkCount": 53, - "latencyP95Ms": 13.04 + "latencyP95Ms": 14.96 }, { "strategy": "hybrid", @@ -300,12 +300,12 @@ "contextRecall": 0.853774, "noResultRate": 0, "meanContextChars": 2322.830188679245, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 20, "unanswerableContextPassages": 3, "chunkCount": 53, - "latencyP95Ms": 15.55 + "latencyP95Ms": 82.31 }, { "strategy": "hybrid", @@ -318,12 +318,12 @@ "contextRecall": 0.896226, "noResultRate": 0, "meanContextChars": 4011.377358490566, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 20, "unanswerableContextPassages": 5, "chunkCount": 53, - "latencyP95Ms": 16.36 + "latencyP95Ms": 18.62 }, { "strategy": "hybrid", @@ -336,12 +336,12 @@ "contextRecall": 0.90566, "noResultRate": 0, "meanContextChars": 6660.264150943396, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 20, "unanswerableContextPassages": 8, "chunkCount": 53, - "latencyP95Ms": 14.66 + "latencyP95Ms": 16.13 }, { "strategy": "hybrid", @@ -354,12 +354,12 @@ "contextRecall": 0.853774, "noResultRate": 0, "meanContextChars": 2322.830188679245, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 40, "unanswerableContextPassages": 3, "chunkCount": 53, - "latencyP95Ms": 14.55 + "latencyP95Ms": 97.73 }, { "strategy": "hybrid", @@ -372,12 +372,12 @@ "contextRecall": 0.896226, "noResultRate": 0, "meanContextChars": 3989.735849056604, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 40, "unanswerableContextPassages": 5, "chunkCount": 53, - "latencyP95Ms": 14.41 + "latencyP95Ms": 100.05 }, { "strategy": "hybrid", @@ -390,12 +390,12 @@ "contextRecall": 0.90566, "noResultRate": 0, "meanContextChars": 6660.792452830188, - "unanswerableQuestions": 10, + "unanswerableQuestions": 25, "unanswerableAbstentionRate": 0, "unanswerableCandidates": 40, "unanswerableContextPassages": 8, "chunkCount": 53, - "latencyP95Ms": 14.47 + "latencyP95Ms": 16.94 } ] } diff --git a/docs/eval/sweep-v1.6.md b/docs/eval/sweep-v1.6.md index 46aeb3e..af8cb92 100644 --- a/docs/eval/sweep-v1.6.md +++ b/docs/eval/sweep-v1.6.md @@ -12,28 +12,28 @@ Each row differs from its neighbour in one parameter. | Strategy | candidateK | contextK | Recall@5 | nDCG@10 | MAP@10 | Context P | Context R | No-result | Context chars | Unans. abstained | Unans. cands | Unans. ctx | Index | p95 | | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | -| dense | 5 | 3 | 0.8774 | 0.7672 | 0.7231 | 0.3082 | 0.8160 | 0.0000 | 2312 | 0.0000 | 5.0 | 3.0 | 53 | 12.30 ms | -| dense | 5 | 5 | 0.8774 | 0.7672 | 0.7231 | 0.2000 | 0.8774 | 0.0000 | 4007 | 0.0000 | 5.0 | 5.0 | 53 | 12.37 ms | -| dense | 10 | 3 | 0.8774 | 0.7804 | 0.7280 | 0.3082 | 0.8160 | 0.0000 | 2312 | 0.0000 | 10.0 | 3.0 | 53 | 12.66 ms | -| dense | 10 | 5 | 0.8774 | 0.7804 | 0.7280 | 0.2000 | 0.8774 | 0.0000 | 4007 | 0.0000 | 10.0 | 5.0 | 53 | 13.42 ms | -| dense | 10 | 8 | 0.8774 | 0.7804 | 0.7280 | 0.1274 | 0.8774 | 0.0000 | 6659 | 0.0000 | 10.0 | 8.0 | 53 | 12.80 ms | -| dense | 20 | 3 | 0.8774 | 0.7804 | 0.7280 | 0.3082 | 0.8160 | 0.0000 | 2312 | 0.0000 | 20.0 | 3.0 | 53 | 13.23 ms | -| dense | 20 | 5 | 0.8774 | 0.7804 | 0.7280 | 0.2000 | 0.8774 | 0.0000 | 4007 | 0.0000 | 20.0 | 5.0 | 53 | 13.55 ms | -| dense | 20 | 8 | 0.8774 | 0.7804 | 0.7280 | 0.1274 | 0.8774 | 0.0000 | 6659 | 0.0000 | 20.0 | 8.0 | 53 | 13.75 ms | -| dense | 40 | 3 | 0.8774 | 0.7804 | 0.7280 | 0.3082 | 0.8160 | 0.0000 | 2312 | 0.0000 | 40.0 | 3.0 | 53 | 15.44 ms | -| dense | 40 | 5 | 0.8774 | 0.7804 | 0.7280 | 0.2000 | 0.8774 | 0.0000 | 4007 | 0.0000 | 40.0 | 5.0 | 53 | 14.98 ms | -| dense | 40 | 8 | 0.8774 | 0.7804 | 0.7280 | 0.1274 | 0.8774 | 0.0000 | 6659 | 0.0000 | 40.0 | 8.0 | 53 | 14.68 ms | -| hybrid | 5 | 3 | 0.9104 | 0.8101 | 0.7682 | 0.3208 | 0.8538 | 0.0000 | 2307 | 0.0000 | 5.0 | 3.0 | 53 | 13.45 ms | -| hybrid | 5 | 5 | 0.9104 | 0.8101 | 0.7682 | 0.2113 | 0.9104 | 0.0000 | 3937 | 0.0000 | 5.0 | 5.0 | 53 | 77.74 ms | -| hybrid | 10 | 3 | 0.8962 | 0.8068 | 0.7580 | 0.3208 | 0.8538 | 0.0000 | 2321 | 0.0000 | 10.0 | 3.0 | 53 | 14.42 ms | -| hybrid | 10 | 5 | 0.8962 | 0.8068 | 0.7580 | 0.2038 | 0.8962 | 0.0000 | 3995 | 0.0000 | 10.0 | 5.0 | 53 | 13.80 ms | -| hybrid | 10 | 8 | 0.8962 | 0.8068 | 0.7580 | 0.1321 | 0.9009 | 0.0000 | 6548 | 0.0000 | 10.0 | 8.0 | 53 | 13.04 ms | -| hybrid | 20 | 3 | 0.8962 | 0.8098 | 0.7605 | 0.3208 | 0.8538 | 0.0000 | 2323 | 0.0000 | 20.0 | 3.0 | 53 | 15.55 ms | -| hybrid | 20 | 5 | 0.8962 | 0.8098 | 0.7605 | 0.2038 | 0.8962 | 0.0000 | 4011 | 0.0000 | 20.0 | 5.0 | 53 | 16.36 ms | -| hybrid | 20 | 8 | 0.8962 | 0.8098 | 0.7605 | 0.1321 | 0.9057 | 0.0000 | 6660 | 0.0000 | 20.0 | 8.0 | 53 | 14.66 ms | -| hybrid | 40 | 3 | 0.8962 | 0.8098 | 0.7605 | 0.3208 | 0.8538 | 0.0000 | 2323 | 0.0000 | 40.0 | 3.0 | 53 | 14.55 ms | -| hybrid | 40 | 5 | 0.8962 | 0.8098 | 0.7605 | 0.2038 | 0.8962 | 0.0000 | 3990 | 0.0000 | 40.0 | 5.0 | 53 | 14.41 ms | -| hybrid | 40 | 8 | 0.8962 | 0.8098 | 0.7605 | 0.1321 | 0.9057 | 0.0000 | 6661 | 0.0000 | 40.0 | 8.0 | 53 | 14.47 ms | +| dense | 5 | 3 | 0.8774 | 0.7672 | 0.7231 | 0.3082 | 0.8160 | 0.0000 | 2312 | 0.0000 | 5.0 | 3.0 | 53 | 16.18 ms | +| dense | 5 | 5 | 0.8774 | 0.7672 | 0.7231 | 0.2000 | 0.8774 | 0.0000 | 4007 | 0.0000 | 5.0 | 5.0 | 53 | 86.91 ms | +| dense | 10 | 3 | 0.8774 | 0.7804 | 0.7280 | 0.3082 | 0.8160 | 0.0000 | 2312 | 0.0000 | 10.0 | 3.0 | 53 | 100.17 ms | +| dense | 10 | 5 | 0.8774 | 0.7804 | 0.7280 | 0.2000 | 0.8774 | 0.0000 | 4007 | 0.0000 | 10.0 | 5.0 | 53 | 81.38 ms | +| dense | 10 | 8 | 0.8774 | 0.7804 | 0.7280 | 0.1274 | 0.8774 | 0.0000 | 6659 | 0.0000 | 10.0 | 8.0 | 53 | 18.17 ms | +| dense | 20 | 3 | 0.8774 | 0.7804 | 0.7280 | 0.3082 | 0.8160 | 0.0000 | 2312 | 0.0000 | 20.0 | 3.0 | 53 | 17.01 ms | +| dense | 20 | 5 | 0.8774 | 0.7804 | 0.7280 | 0.2000 | 0.8774 | 0.0000 | 4007 | 0.0000 | 20.0 | 5.0 | 53 | 97.74 ms | +| dense | 20 | 8 | 0.8774 | 0.7804 | 0.7280 | 0.1274 | 0.8774 | 0.0000 | 6659 | 0.0000 | 20.0 | 8.0 | 53 | 85.30 ms | +| dense | 40 | 3 | 0.8774 | 0.7804 | 0.7280 | 0.3082 | 0.8160 | 0.0000 | 2312 | 0.0000 | 40.0 | 3.0 | 53 | 57.28 ms | +| dense | 40 | 5 | 0.8774 | 0.7804 | 0.7280 | 0.2000 | 0.8774 | 0.0000 | 4007 | 0.0000 | 40.0 | 5.0 | 53 | 16.64 ms | +| dense | 40 | 8 | 0.8774 | 0.7804 | 0.7280 | 0.1274 | 0.8774 | 0.0000 | 6659 | 0.0000 | 40.0 | 8.0 | 53 | 99.32 ms | +| hybrid | 5 | 3 | 0.9104 | 0.8101 | 0.7682 | 0.3208 | 0.8538 | 0.0000 | 2307 | 0.0000 | 5.0 | 3.0 | 53 | 90.40 ms | +| hybrid | 5 | 5 | 0.9104 | 0.8101 | 0.7682 | 0.2113 | 0.9104 | 0.0000 | 3937 | 0.0000 | 5.0 | 5.0 | 53 | 89.63 ms | +| hybrid | 10 | 3 | 0.8962 | 0.8068 | 0.7580 | 0.3208 | 0.8538 | 0.0000 | 2321 | 0.0000 | 10.0 | 3.0 | 53 | 83.59 ms | +| hybrid | 10 | 5 | 0.8962 | 0.8068 | 0.7580 | 0.2038 | 0.8962 | 0.0000 | 3986 | 0.0000 | 10.0 | 5.0 | 53 | 89.79 ms | +| hybrid | 10 | 8 | 0.8962 | 0.8068 | 0.7580 | 0.1321 | 0.9009 | 0.0000 | 6548 | 0.0000 | 10.0 | 8.0 | 53 | 14.96 ms | +| hybrid | 20 | 3 | 0.8962 | 0.8098 | 0.7605 | 0.3208 | 0.8538 | 0.0000 | 2323 | 0.0000 | 20.0 | 3.0 | 53 | 82.31 ms | +| hybrid | 20 | 5 | 0.8962 | 0.8098 | 0.7605 | 0.2038 | 0.8962 | 0.0000 | 4011 | 0.0000 | 20.0 | 5.0 | 53 | 18.62 ms | +| hybrid | 20 | 8 | 0.8962 | 0.8098 | 0.7605 | 0.1321 | 0.9057 | 0.0000 | 6660 | 0.0000 | 20.0 | 8.0 | 53 | 16.13 ms | +| hybrid | 40 | 3 | 0.8962 | 0.8098 | 0.7605 | 0.3208 | 0.8538 | 0.0000 | 2323 | 0.0000 | 40.0 | 3.0 | 53 | 97.73 ms | +| hybrid | 40 | 5 | 0.8962 | 0.8098 | 0.7605 | 0.2038 | 0.8962 | 0.0000 | 3990 | 0.0000 | 40.0 | 5.0 | 53 | 100.05 ms | +| hybrid | 40 | 8 | 0.8962 | 0.8098 | 0.7605 | 0.1321 | 0.9057 | 0.0000 | 6661 | 0.0000 | 40.0 | 8.0 | 53 | 16.94 ms | ## Skipped cells diff --git a/docs/eval/threshold-v1.6.json b/docs/eval/threshold-v1.6.json index 15d8562..fbf9e89 100644 --- a/docs/eval/threshold-v1.6.json +++ b/docs/eval/threshold-v1.6.json @@ -7,13 +7,13 @@ { "threshold": 0, "validation": { - "questions": 24, + "questions": 34, "answerableQuestions": 19, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 20, "unanswerable": { - "questions": 5, + "questions": 15, "abstentionCount": 0, "retrievalAbstentionRate": 0, "meanCandidatesRetrieved": 20, @@ -32,13 +32,13 @@ } }, "test": { - "questions": 39, + "questions": 44, "answerableQuestions": 34, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 20, "unanswerable": { - "questions": 5, + "questions": 10, "abstentionCount": 0, "retrievalAbstentionRate": 0, "meanCandidatesRetrieved": 20, @@ -60,13 +60,13 @@ { "threshold": 0.3, "validation": { - "questions": 24, + "questions": 34, "answerableQuestions": 19, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 20, "unanswerable": { - "questions": 5, + "questions": 15, "abstentionCount": 0, "retrievalAbstentionRate": 0, "meanCandidatesRetrieved": 20, @@ -85,13 +85,13 @@ } }, "test": { - "questions": 39, + "questions": 44, "answerableQuestions": 34, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 20, "unanswerable": { - "questions": 5, + "questions": 10, "abstentionCount": 0, "retrievalAbstentionRate": 0, "meanCandidatesRetrieved": 20, @@ -113,13 +113,13 @@ { "threshold": 0.4, "validation": { - "questions": 24, + "questions": 34, "answerableQuestions": 19, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 20, "unanswerable": { - "questions": 5, + "questions": 15, "abstentionCount": 0, "retrievalAbstentionRate": 0, "meanCandidatesRetrieved": 20, @@ -138,13 +138,13 @@ } }, "test": { - "questions": 39, + "questions": 44, "answerableQuestions": 34, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 20, "unanswerable": { - "questions": 5, + "questions": 10, "abstentionCount": 0, "retrievalAbstentionRate": 0, "meanCandidatesRetrieved": 20, @@ -166,13 +166,13 @@ { "threshold": 0.5, "validation": { - "questions": 24, + "questions": 34, "answerableQuestions": 19, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 20, "unanswerable": { - "questions": 5, + "questions": 15, "abstentionCount": 0, "retrievalAbstentionRate": 0, "meanCandidatesRetrieved": 20, @@ -191,13 +191,13 @@ } }, "test": { - "questions": 39, + "questions": 44, "answerableQuestions": 34, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 20, "unanswerable": { - "questions": 5, + "questions": 10, "abstentionCount": 0, "retrievalAbstentionRate": 0, "meanCandidatesRetrieved": 20, @@ -219,13 +219,13 @@ { "threshold": 0.6, "validation": { - "questions": 24, + "questions": 34, "answerableQuestions": 19, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 20, "unanswerable": { - "questions": 5, + "questions": 15, "abstentionCount": 0, "retrievalAbstentionRate": 0, "meanCandidatesRetrieved": 20, @@ -244,13 +244,13 @@ } }, "test": { - "questions": 39, + "questions": 44, "answerableQuestions": 34, "noResultCount": 0, "noResultRate": 0, "meanRetrieved": 20, "unanswerable": { - "questions": 5, + "questions": 10, "abstentionCount": 0, "retrievalAbstentionRate": 0, "meanCandidatesRetrieved": 20, diff --git a/docs/eval/threshold-v1.6.md b/docs/eval/threshold-v1.6.md index 3997609..6dcc74e 100644 --- a/docs/eval/threshold-v1.6.md +++ b/docs/eval/threshold-v1.6.md @@ -34,7 +34,7 @@ retrieval loss. ## Outcome -The sweep is **flat**: every threshold from 0 to 0.6 produces the same validation nDCG@10 (0.7556), the same Recall@5 (0.8684) and the same retrieval abstention rate on unanswerable questions (0/5). No candidate is ever filtered out, so the threshold is **non-binding** on this corpus. +The sweep is **flat**: every threshold from 0 to 0.6 produces the same validation nDCG@10 (0.7556), the same Recall@5 (0.8684) and the same retrieval abstention rate on unanswerable questions (0/15). No candidate is ever filtered out, so the threshold is **non-binding** on this corpus. **No evidence to change `threshold = 0.5`.** All thresholds hold the line equally; picking one would be arbitrary. The current value can be neither validated nor falsified here, which is a property of the corpus rather than of the threshold. Note also what this does *not* establish: abstention is a retrieval-layer statement — whether the model then declines to answer needs a generator eval. diff --git a/eval/questions.jsonl b/eval/questions.jsonl index 10b031a..44fa762 100644 --- a/eval/questions.jsonl +++ b/eval/questions.jsonl @@ -61,3 +61,18 @@ {"id":"q061","type":"semantic","question":"Why does the estuary programme take more replicate samples than the other monitoring programmes?","relevant":[{"document":"estuary-monitoring.md","page":null,"block":8,"quote":"within-station variance"}]} {"id":"q062","type":"unanswerable","answerable":false,"question":"What is the annual operating cost of the coastal monitoring buoys?","relevant":[]} {"id":"q063","type":"unanswerable","answerable":false,"question":"How many litres per second does the reservoir release downstream on a typical day?","relevant":[]} +{"id":"q064","question":"How much GPU memory does Gateway API v3 need to serve a request?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q065","question":"What is the uptime SLA for Gateway API v2?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q066","question":"How long does an authentication token stay valid before it expires?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q067","question":"Which SDK version should a client use with Gateway API v3?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q068","question":"Can Gateway API v2 be self-hosted on-premise?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q069","question":"How is usage invoiced for Gateway API v2?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q070","question":"Which laboratory is accredited to analyse the estuary transect samples?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q071","question":"Which vendor supplies the coastal monitoring buoys?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q072","question":"What is the annual funding for the groundwater monitoring programme?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q073","question":"How many staff work on the reservoir monitoring programme?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q074","question":"What is the warranty period on the estuary monitoring sensors?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q075","question":"How long are the coastal monitoring readings retained before deletion?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q076","question":"What encryption standard is used to transmit the river monitoring readings?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q077","question":"When was the last external audit of the lake monitoring programme?","type":"unanswerable","answerable":false,"relevant":[]} +{"id":"q078","question":"What is the unit price of a river monitoring sediment sampler?","type":"unanswerable","answerable":false,"relevant":[]} diff --git a/eval/splits.json b/eval/splits.json index 021e265..6e13b15 100644 --- a/eval/splits.json +++ b/eval/splits.json @@ -61,5 +61,20 @@ "q060": "validation", "q061": "test", "q062": "validation", - "q063": "test" + "q063": "test", + "q064": "validation", + "q065": "validation", + "q066": "validation", + "q067": "test", + "q068": "validation", + "q069": "test", + "q070": "validation", + "q071": "validation", + "q072": "test", + "q073": "validation", + "q074": "test", + "q075": "validation", + "q076": "validation", + "q077": "test", + "q078": "validation" }