diff --git a/.github/workflows/verify.yml b/.github/workflows/verify.yml index 967fe4d..740680c 100644 --- a/.github/workflows/verify.yml +++ b/.github/workflows/verify.yml @@ -78,3 +78,24 @@ jobs: - name: Smoke test the packaged app (macOS / Windows) if: runner.os != 'Linux' run: npm run smoke:packaged + + # The eval baseline (#75) is the reference point every v1.5 experiment is + # measured against, so CI proves the committed numbers still reproduce. The + # model cache is keyed on the pinned model file, so the 134 MB download + # happens once per pin, not once per run. A hit is required for the run to + # be offline: `eval:prepare` only downloads when the cache is cold. + - name: Cache eval embedding model + if: runner.os == 'Linux' + uses: actions/cache@v4 + with: + path: ~/.config/knownote/models + key: knownote-eval-model-${{ hashFiles('src/main/embedding/localModel.ts') }} + + - name: Eval harness is deterministic + if: runner.os == 'Linux' + run: | + xvfb-run -a npm run eval:prepare + xvfb-run -a node scripts/eval.mjs + cp docs/eval/baseline-v1.4.json /tmp/eval-a.json + xvfb-run -a node scripts/eval.mjs + diff /tmp/eval-a.json docs/eval/baseline-v1.4.json diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 05390d9..269794b 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -39,18 +39,20 @@ npm run dev ## Commands -| Command | What it does | -| ------------------------ | ------------------------------------------------------------- | -| `npm run dev` | Start the app in development with HMR. | -| `npm run typecheck` | Typecheck the main/preload, renderer and test projects. | -| `npm test` | Run the Node test suite (`test/**/*.test.ts`). | -| `npm run lint` | ESLint (see the note on the current baseline below). | -| `npm run format` | Prettier over the whole repository. | -| `npm run build` | Typecheck, then bundle with electron-vite. | -| `npm run build:unpack` | `build`, then produce an unpacked app in `dist/`. | -| `npm run smoke:packaged` | Launch the packaged app's `--smoke-test` and check it starts. | -| `npm run db:generate` | Generate a Drizzle migration from `src/main/db/schema.ts`. | -| `npm run db:studio` | Inspect the development database. | +| Command | What it does | +| ------------------------ | -------------------------------------------------------------- | +| `npm run dev` | Start the app in development with HMR. | +| `npm run typecheck` | Typecheck the main/preload, renderer and test projects. | +| `npm test` | Run the Node test suite (`test/**/*.test.ts`). | +| `npm run lint` | ESLint (see the note on the current baseline below). | +| `npm run format` | Prettier over the whole repository. | +| `npm run build` | Typecheck, then bundle with electron-vite. | +| `npm run build:unpack` | `build`, then produce an unpacked app in `dist/`. | +| `npm run smoke:packaged` | Launch the packaged app's `--smoke-test` and check it starts. | +| `npm run eval:prepare` | One-time, networked: download the pinned eval embedding model. | +| `npm run eval` | Run the RAG eval harness offline; rewrites `docs/eval/`. | +| `npm run db:generate` | Generate a Drizzle migration from `src/main/db/schema.ts`. | +| `npm run db:studio` | Inspect the development database. | ## Before you open a pull request diff --git a/docs/architecture.md b/docs/architecture.md index 47b1f9f..490ef34 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -78,3 +78,28 @@ Rules: compatibility and delegates to the default retriever. `SearchResult` gained a `locator` field, which is populated from `RetrievedEvidence`; ranking, scores and the existing fields are unchanged. + +## Retrieval measurement + +Retrieval quality is a number before it is an opinion. `eval/` holds a committed +corpus and a ground-truth dataset; `src/main/eval/` runs them through the normal +ingestion path and the real `Retriever`, and writes +`docs/eval/baseline-.json` (deterministic) plus a markdown summary. + +Rules: + +- **Ground truth is corpus identity, not database identity.** A relevant location + is `{ document: , page, block: , quote? }`. Runtime + `documentId`s are random and `blockId`s embed them, so a dataset keyed on them + would break — instead of measuring — a chunking or parser change. +- **The baseline is frozen and the delta is explicit.** v1.5 experiments (#77, + #78) are reported as a delta against the committed baseline, with an + adopted-change threshold. A change that is not measured against it is not + adopted. +- **`npm run eval` is offline.** Only `npm run eval:prepare` may download the + pinned model. The pinned revision is part of the embedding space identity, so a + model change is a baseline change. + +The harness is a main-process entry (`--eval-harness`), like the packaged smoke +test, because the DB layer, vector store and loaders do not exist outside +Electron. See `eval/README.md` for the dataset format and commands. diff --git a/docs/eval/baseline-v1.4.json b/docs/eval/baseline-v1.4.json new file mode 100644 index 0000000..5af2a95 --- /dev/null +++ b/docs/eval/baseline-v1.4.json @@ -0,0 +1,664 @@ +{ + "baseline": "v1.4", + "generatedBy": "npm run eval", + "config": { + "embedding": "Xenova/multilingual-e5-small@761b726dd34fb83930e26aab4e9ac3899aa1fa78 q8 (384d, local)", + "chunking": { + "chunkSize": 500, + "chunkOverlap": 50, + "minChunkSize": 100, + "allowSpanPages": false + }, + "retrieval": "dense", + "topK": 10, + "threshold": 0, + "evidenceK": 5, + "corpus": "eval/corpus", + "documents": 13, + "questions": 30 + }, + "metrics": { + "recallAt1": 0.766667, + "recallAt5": 0.933333, + "recallAt10": 1, + "mrr": 0.849206, + "ndcgAt10": 0.889891, + "evidencePrecisionAt5": 0.2 + }, + "perQuestion": [ + { + "id": "q001", + "question": "Why is bedload harder to measure than suspended sediment?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q002", + "question": "How many replicate samples are collected at each river station?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q003", + "question": "What is the central trade-off in lithium-ion cell design?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q004", + "question": "Why do nickel-rich battery packs need more aggressive thermal management?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q005", + "question": "What happens once the separator in a battery cell melts?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q006", + "question": "At what temperature do honeybees begin to forage?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q007", + "question": "What does a late frost damage during full bloom?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q008", + "question": "Why is a continuous tree canopy more effective at cooling than isolated trees?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q009", + "question": "Why are trees with aggressive surface roots unsuitable for narrow verges?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q010", + "question": "At what temperature is lactic acid fermentation fastest?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q011", + "question": "Is the salt percentage in fermentation based on vegetable weight or water weight?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q012", + "question": "Why must tidal turbines be sited in places with very fast currents?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q013", + "question": "What is the main environmental concern for tidal energy installations?", + "firstRelevantRank": 6, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [], + [], + [], + [], + [], + [ + 0 + ], + [], + [], + [], + [] + ] + }, + { + "id": "q014", + "question": "In lake monitoring, how is the sampling depth actually recorded?", + "firstRelevantRank": 2, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q015", + "question": "Why does deep-water oxygen fall while a lake remains stratified?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q016", + "question": "How do supercapacitors hold their charge?", + "firstRelevantRank": 2, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q017", + "question": "Why can solitary bees pollinate a bloom week that is too cold for honeybee hives?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q018", + "question": "Why is one continuous planted roof layer better than several isolated planted beds?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q019", + "question": "Where do acetic acid bacteria sit in a vinegar culture?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q020", + "question": "What happens if a vinegar culture is sealed airtight?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q021", + "question": "Why is wave energy harder to schedule ahead than tidal energy?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q022", + "question": "Where does siting for wave energy devices concentrate, and where does it not?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q023", + "question": "How does the river sampling protocol differ from the lake sampling protocol?", + "firstRelevantRank": 3, + "relevantCount": 2, + "retrievedCount": 10, + "matchesByRank": [ + [], + [], + [ + 1 + ], + [], + [ + 0 + ], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q024", + "question": "A street canopy and a green roof are both said to cool; what surface does each one shade?", + "firstRelevantRank": 3, + "relevantCount": 2, + "retrievedCount": 10, + "matchesByRank": [ + [], + [], + [ + 0 + ], + [ + 1 + ], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q025", + "question": "Which preservation method depends on keeping air away from the food?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q026", + "question": "Why can one cold morning cost a grower the whole crop even when colonies are brought in?", + "firstRelevantRank": 7, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [], + [], + [], + [], + [], + [], + [ + 0 + ], + [], + [], + [] + ] + }, + { + "id": "q027", + "question": "绿茶应该怎样保存才能减缓氧化?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q028", + "question": "茶叶储存的相对湿度上限是多少?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q029", + "question": "为什么冷冻保存的茶叶取出后不能立刻打开包装?", + "firstRelevantRank": 1, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [], + [] + ] + }, + { + "id": "q030", + "question": "为什么潮汐能比风能和太阳能更容易提前安排发电?", + "firstRelevantRank": 2, + "relevantCount": 1, + "retrievedCount": 10, + "matchesByRank": [ + [], + [ + 0 + ], + [], + [], + [], + [], + [], + [], + [], + [] + ] + } + ] +} diff --git a/docs/eval/baseline-v1.4.md b/docs/eval/baseline-v1.4.md new file mode 100644 index 0000000..72a1394 --- /dev/null +++ b/docs/eval/baseline-v1.4.md @@ -0,0 +1,59 @@ +# RAG eval baseline — v1.4 + +Generated by `npm run eval`. The numbers below are harness output — do not edit them by hand. + +## Configuration + +| Setting | Value | +| --- | --- | +| Embedding | `Xenova/multilingual-e5-small@761b726dd34fb83930e26aab4e9ac3899aa1fa78 q8 (384d, local)` | +| Chunking | `chunkSize=500, chunkOverlap=50, minChunkSize=100, allowSpanPages=false` | +| Retrieval | `dense` | +| Ranks | `topK=10, threshold=0` | +| Evidence per query | `evidenceK=5` | +| Corpus | `eval/corpus` (13 documents, 30 questions) | + +## Metrics + +| Metric | Value | +| --- | --- | +| Recall@1 | 0.7667 | +| Recall@5 | 0.9333 | +| Recall@10 | 1.0000 | +| MRR | 0.8492 | +| nDCG@10 | 0.8899 | +| Evidence precision@5 | 0.2000 | + +Timing is informational only and is **not** frozen: p50 13.42 ms, +p95 23.64 ms on the machine that produced this file. Query +latency depends on hardware and load, so it must never be the reason two runs differ. + +## Definitions + +- A retrieved passage is relevant when its provenance covers a ground-truth block. +- **Recall@k** is the share of ground-truth blocks covered by the first `k` passages. +- **Evidence precision@5** is the share of the first `5` + retrieved passages that cover a ground-truth block. This is **retrieval precision**, not + answer citation recall: the harness runs no model and produces no answer. Answer-level + citation correctness is covered by the resolver (#70); a model-driven answer eval is a + separate deliverable. +- Ground truth is expressed in corpus identity (`document` relative path + `block` + ordinal + optional `quote`), never a runtime `documentId`/`blockId`. + +## Comparison protocol + +v1.5 experiments (#77, #78) are reported as a **delta against this file**. The +adopted-change rule is: + +> Adopt a change only if Recall@5 improves and nDCG@10 does not regress. A change +> that trades a large latency increase for a marginal recall gain is a product +> decision, not an automatic win, and must be stated as such. + +A changed result must be reproducible with: + +```bash +npm run eval:prepare # one-time, networked model bootstrap +npm run eval # offline and deterministic +``` + +The raw report is committed next to this summary as `baseline-v1.4.json`. diff --git a/eval/README.md b/eval/README.md new file mode 100644 index 0000000..095b72a --- /dev/null +++ b/eval/README.md @@ -0,0 +1,116 @@ +# RAG eval harness + +Measures retrieval quality so "did this change make retrieval better?" has an answer. +The current numbers are frozen in [`docs/eval/baseline-v1.4.md`](../docs/eval/baseline-v1.4.md); +every v1.5 experiment (#77, #78) is reported as a delta against that file. + +## Commands + +```bash +npm run eval:prepare # one-time, networked: download the pinned embedding model +npm run eval # offline and deterministic: run the harness, rewrite the baseline +``` + +`eval:prepare` downloads the pinned `multilingual-e5-small` revision into the app's model +cache and verifies it. `eval` never touches the network: if the model is missing it stops +with + +``` +Eval model is not available locally. +Run: npm run eval:prepare +``` + +"Runs without configuration" means no API keys and no model/provider setup in the app — it +does not mean "no download". The 134 MB of weights are not committed. + +## Layout + +```text +eval/ + corpus/ first-party documents (markdown today) + questions.jsonl one question per line; committed with the corpus +``` + +The corpus is authored for this repository and carries the repository's GPL-3.0 licence, +so it is redistributable. Keeping it first-party avoids a licence question every time the +dataset needs a new question. + +`questions.jsonl` schema: + +```json +{ + "id": "q001", + "question": "Why is bedload harder to measure than suspended sediment?", + "relevant": [ + { + "document": "river-monitoring.md", + "page": null, + "block": 4, + "quote": "Bedload is the harder fraction to measure" + } + ], + "goldAnswer": "optional" +} +``` + +Ground truth uses **corpus identity, never database identity**: + +- `document` is the corpus-relative path. +- `block` is the block ordinal inside the document (`document_blocks.order`). +- `page` is `null` for unpaginated sources. +- `quote` is an optional excerpt. The runner fails if the referenced block no longer + contains it, so a parser change cannot silently move the ground truth. + +Runtime `documentId`s are random and `blockId`s embed them, so neither may appear here. +This is what lets #78 change chunking without invalidating the dataset: the ground truth +describes the source, and the runner maps it to whatever ids that run produced. + +### Dataset design + +The dataset is built to avoid saturation, not to produce low scores. A benchmark whose +metric is already 1.0 cannot tell a better retriever from the current one. So the corpus +pairs near-duplicate documents (river vs lake monitoring, street canopy vs green roofs, +tidal vs wave energy), and the questions include: + +- questions that distinguish two similar documents by one detail, +- multi-location ground truth that requires several passages at once, +- paraphrases whose wording does not overlap the source, +- Chinese questions, including over English sources, because the local model is + multilingual. + +Every question is answerable from the corpus; difficulty comes from discrimination, not +from unanswerable queries. + +## What it does + +1. Creates a throwaway database (a temp profile; the developer's own DB is never opened). +2. Indexes the corpus through the normal ingestion path (`addDocumentFromFile`), so blocks, + chunking and embeddings are the real ones. +3. Maps each ground-truth `document`/`block` to the run's runtime ids. +4. Runs the real `Retriever` (`KnowledgeService.search` → `DenseRetriever`). +5. Writes `docs/eval/baseline-.json` (deterministic) and `.md` (with timing). + +## Metrics + +- **Recall@1/5/10** — share of ground-truth blocks covered by the first k passages. +- **MRR** — reciprocal rank of the first relevant passage. +- **nDCG@10** — binary-gain discounted cumulative gain. +- **Evidence precision@5** — of the first 5 retrieved passages, the share that cover a + ground-truth block. This is **retrieval precision, not answer citation recall**: the + harness runs no model and produces no answer. Answer-level citation correctness is + covered by the resolver (#70); a model-driven answer eval would be a separate + deliverable. +- **Latency p50/p95** — informational only. Timing is **not** frozen, and the committed + JSON excludes it so two runs diff cleanly. + +## Determinism + +`npm run eval` twice must print the same `[eval] metrics {...}` line: + +```bash +node scripts/eval.mjs | grep '\[eval\] metrics' +node scripts/eval.mjs | grep '\[eval\] metrics' +``` + +The local embedding backend always runs (`KNOWNOTE_EVAL_MODEL_CACHE` can point CI at its own +cache), so a developer's remote embedding configuration cannot leak into the baseline. diff --git a/eval/corpus/battery-chemistry.md b/eval/corpus/battery-chemistry.md new file mode 100644 index 0000000..059c819 --- /dev/null +++ b/eval/corpus/battery-chemistry.md @@ -0,0 +1,13 @@ +# Battery Chemistry Notes + +## Overview + +Lithium-ion cells trade energy density against thermal stability. The choice of cathode chemistry sets most of that trade-off before any cell is assembled, which is why pack engineers treat the chemistry decision as the point of no return. Anode and electrolyte choices refine the behaviour, but they cannot rescue a cathode that is unsuitable for the duty cycle. + +## Cathode Materials + +Nickel-rich layered oxides store more energy per unit mass, but they release oxygen at lower temperatures than iron phosphate. That released oxygen can feed a cell fire, so nickel-rich packs need more aggressive thermal management. Iron phosphate sacrifices energy density in exchange for a much higher decomposition temperature, which is why it dominates applications where abuse tolerance matters more than range. + +## Thermal Runaway + +Thermal runaway begins when internal heat generation outpaces heat removal. Once the separator melts, the cell shorts internally and the reaction becomes self-sustaining within seconds. Propagation to neighbouring cells is driven by conducted heat, so the spacing and the inter-cell material are as important as the chemistry itself. diff --git a/eval/corpus/chinese-tea-storage.md b/eval/corpus/chinese-tea-storage.md new file mode 100644 index 0000000..7d0bad3 --- /dev/null +++ b/eval/corpus/chinese-tea-storage.md @@ -0,0 +1,13 @@ +# 茶叶储存指南 + +## 概述 + +茶叶储存的核心是控制湿度、温度和氧气。绿茶最容易氧化变质,需要低温密封保存;黑茶等后发酵茶则需要少量空气才能继续转化。 + +## 湿度控制 + +相对湿度应保持在百分之五十以下,湿度过高会让茶叶吸潮并滋生霉菌。干燥剂需要定期更换,因为吸潮饱和后它不再起作用,反而会把水分重新释放回容器。 + +## 温度控制 + +短期存放可以冷藏,但取出后必须等到恢复室温再打开包装,否则冷凝水会直接落在茶叶上。长期存放更适合冷冻,并且分装成小份,避免反复解冻。 diff --git a/eval/corpus/fermentation.md b/eval/corpus/fermentation.md new file mode 100644 index 0000000..c19b388 --- /dev/null +++ b/eval/corpus/fermentation.md @@ -0,0 +1,13 @@ +# Fermentation Field Notes + +## Overview + +Lactic acid fermentation preserves vegetables by lowering the pH below the level where spoilage organisms can grow. The process needs salt, an anaerobic environment and time. The salt is not there for flavour alone; it selects for the bacteria the process depends on and suppresses the ones that would soften the vegetable. + +## Lactic Acid Pathway + +Salt-tolerant lactic acid bacteria dominate the early brine and produce acid that suppresses their competitors. The acidification is fastest between twenty and twenty-four degrees Celsius. Below that range the process still works but proceeds slowly enough that spoilage organisms have time to establish, which is why a cool cellar is a risk as well as an advantage. + +## Salt Ratio + +The salt concentration is usually expressed as a percentage of the vegetable weight, not the water weight. Two percent is the common starting point; a weaker brine risks soft vegetables, a stronger one slows fermentation. The percentage must be weighed rather than measured by volume, because the density of salt varies with how it is packed. diff --git a/eval/corpus/green-roofs.md b/eval/corpus/green-roofs.md new file mode 100644 index 0000000..eba2693 --- /dev/null +++ b/eval/corpus/green-roofs.md @@ -0,0 +1,13 @@ +# Green Roof Design + +## Overview + +A green roof is a planted layer on a building rather than at street level. It cools the building directly by shading the roof membrane and indirectly by evapotranspiration, and it retains rainfall that would otherwise enter the storm sewer. + +## Cooling Effect + +The cooling benefit is largest where the planting covers the roof membrane rather than a small decorative strip, because an exposed membrane absorbs far more heat during the day. A continuous planted layer is therefore more effective than the same planted area broken into isolated beds. + +## Substrate Depth + +Substrate depth decides which species survive a dry summer. Shallow substrate supports sedums that tolerate drought but provide little insulation, while deeper substrate supports grasses and shrubs with more cooling but a much heavier structural load. diff --git a/eval/corpus/lake-monitoring.md b/eval/corpus/lake-monitoring.md new file mode 100644 index 0000000..4a5f997 --- /dev/null +++ b/eval/corpus/lake-monitoring.md @@ -0,0 +1,13 @@ +# Lake Monitoring Handbook + +## Overview + +Lake monitoring relies on depth profiles rather than a single gauge reading, because a lake stratifies into layers that behave differently. The goal is to track temperature, dissolved oxygen and nutrient concentration through the water column over a full seasonal cycle. + +## Thermal Stratification + +In summer the surface layer warms and floats above a cold hypolimnion, separated by a sharp thermocline. Oxygen in the hypolimnion is not replenished while the lake is stratified, so deep-water oxygen declines steadily until autumn mixing resets the column. + +## Sampling Protocol + +Field teams collect three replicate samples at each depth to control for local variability. Samples are refrigerated within one hour and analysed for turbidity before any chemical treatment, since turbidity interferes with most colorimetric assays. Depth is recorded from the profiling sonde, not from the sample line, because the line stretches under its own weight. diff --git a/eval/corpus/orchard-pollination.md b/eval/corpus/orchard-pollination.md new file mode 100644 index 0000000..a09ad8f --- /dev/null +++ b/eval/corpus/orchard-pollination.md @@ -0,0 +1,13 @@ +# Orchard Pollination Guide + +## Overview + +Most apple varieties cannot set fruit from their own pollen. Growers therefore plant compatible polliniser trees and depend on insects to move pollen between them. The polliniser must flower at the same time as the main variety, because pollen that is available a week late is pollen that the main crop cannot use. + +## Bee Activity + +Honeybees forage only when the temperature is above roughly twelve degrees Celsius, so a cold bloom week can leave flowers unpollinated even when hives are present. Mason bees fly at lower temperatures and can partly cover that gap. Growers in frost-prone valleys often stock both species so that the bloom is not lost to a single cold morning. + +## Frost Risk + +A late frost during full bloom destroys the flower's ovary rather than the petals, so damage is invisible for several days. Growers often run wind machines to mix warmer air from above the inversion layer. The machines are worth running only while the inversion exists, which is why they are switched on from a temperature reading rather than a schedule. diff --git a/eval/corpus/river-monitoring.md b/eval/corpus/river-monitoring.md new file mode 100644 index 0000000..f9e1d84 --- /dev/null +++ b/eval/corpus/river-monitoring.md @@ -0,0 +1,13 @@ +# River Monitoring Handbook + +## Overview + +River monitoring combines fixed gauges with periodic field surveys. A gauge gives a continuous stage record at one point, while a survey samples many points at one time. Neither alone is enough: the goal is a continuous record of flow, sediment load and water chemistry that a single survey cannot provide, and a survey that can validate what the gauge inferred between visits. + +## Sediment Transport + +Coarse sediment moves mostly as bedload during high-flow events, while fine sediment travels in suspension even at moderate discharge. Bedload is the harder fraction to measure, because it moves along the bed where instruments are difficult to anchor and the transport is episodic. Suspended sediment, by contrast, can be estimated from a single depth-integrated sample calibrated against turbidity. + +## Sampling Protocol + +Field teams collect three replicate samples at each station to control for local variability. Samples are refrigerated within one hour and analysed for turbidity before any chemical treatment, since turbidity interferes with most colorimetric assays. Replicates are kept separate through analysis so that a disagreement between them is visible as variance rather than averaged away. diff --git a/eval/corpus/supercapacitors.md b/eval/corpus/supercapacitors.md new file mode 100644 index 0000000..0cf347a --- /dev/null +++ b/eval/corpus/supercapacitors.md @@ -0,0 +1,13 @@ +# Supercapacitor Notes + +## Overview + +Supercapacitors store charge at a solid-electrolyte interface rather than in a chemical reaction. The absence of a bulk reaction is why they deliver very high power and tolerate hundreds of thousands of cycles, and also why their energy density is far below a lithium-ion cell. + +## Charge Storage + +Charge is held in an electric double layer formed at the surface of a porous carbon electrode. Because storage is a surface effect, the accessible surface area sets the capacitance, and pores that are too small for the electrolyte ions contribute nothing. + +## Thermal Behaviour + +Supercapacitors generate little heat during cycling, so their thermal management is far simpler than a battery pack's. The failure mode to watch is overvoltage, which decomposes the electrolyte and vents gas rather than producing the self-sustaining reaction a lithium cell can enter. diff --git a/eval/corpus/tidal-energy.md b/eval/corpus/tidal-energy.md new file mode 100644 index 0000000..c4ea57b --- /dev/null +++ b/eval/corpus/tidal-energy.md @@ -0,0 +1,13 @@ +# Tidal Energy Briefing + +## Overview + +Tidal energy is predictable decades ahead, unlike wind or solar. The resource is concentrated at sites with a large tidal range or strong tidal currents. That predictability is the technology's main advantage: a tidal operator can commit to a generation schedule years in advance, which matters to a grid that must otherwise pay for storage to cover the variability of wind. + +## Turbine Siting + +Turbines are sited where peak current speed exceeds two metres per second, because power scales with the cube of velocity. Siting therefore concentrates on narrow channels between headlands. A site with half the current speed yields roughly one eighth of the power, so a marginal channel rarely justifies the fixed cost of the installation. + +## Environmental Impact + +The main concern is not collision but the change in sediment transport, which can alter the seabed over years. Monitoring programmes usually compare the seabed before commissioning with a survey several years later. Blade strike on fish is studied too, but the evidence so far suggests that the far larger effect is the slow reshaping of the channel floor. diff --git a/eval/corpus/urban-canopy.md b/eval/corpus/urban-canopy.md new file mode 100644 index 0000000..b94852d --- /dev/null +++ b/eval/corpus/urban-canopy.md @@ -0,0 +1,13 @@ +# Urban Canopy Planning + +## Overview + +Street trees are infrastructure, not decoration. Their canopy shades asphalt, intercepts rainfall and lowers the air temperature of the surrounding block. A street tree also has a maintenance cost and a service life, so the planning question is not whether to plant but where the same budget buys the most cooling. + +## Cooling Effect + +The cooling benefit is largest where canopy covers the pavement rather than the building, because asphalt stores far more heat during the day. A continuous canopy is therefore more effective than the same number of isolated trees. Interception of rainfall follows the same geometry: a connected canopy slows runoff across a whole street rather than at scattered points. + +## Species Selection + +Species are chosen for drought tolerance and root behaviour rather than appearance. Trees with aggressive surface roots lift pavements and are poor choices for narrow verges. A species that needs summer irrigation also erodes the cooling benefit, because the water it consumes is itself a resource the city is trying to conserve. diff --git a/eval/corpus/vinegar-production.md b/eval/corpus/vinegar-production.md new file mode 100644 index 0000000..ff4f0f4 --- /dev/null +++ b/eval/corpus/vinegar-production.md @@ -0,0 +1,13 @@ +# Vinegar Production Notes + +## Overview + +Vinegar production is a two-stage fermentation: yeast first converts sugar to ethanol, and acetic acid bacteria then oxidise the ethanol to acetic acid. The second stage is aerobic, which is what separates it from the anaerobic lactic fermentation used for preserving vegetables. + +## Acetic Acid Pathway + +Acetic acid bacteria sit at the surface of the liquid where oxygen is available, and they convert ethanol to acetic acid with acetaldehyde as an intermediate. The reaction releases heat and consumes oxygen, so a culture that is stirred too gently stalls for lack of oxygen rather than for lack of substrate. + +## Oxygen Requirement + +The vessel is deliberately kept partly empty so that the surface film has air, and the liquid is never filled to the neck. Sealing a vinegar culture airtight stops the conversion entirely, unlike a lactic fermentation where the seal is required. diff --git a/eval/corpus/wave-energy.md b/eval/corpus/wave-energy.md new file mode 100644 index 0000000..78b6ca3 --- /dev/null +++ b/eval/corpus/wave-energy.md @@ -0,0 +1,13 @@ +# Wave Energy Briefing + +## Overview + +Wave energy is driven by wind rather than by the moon, so unlike tidal energy it is not predictable decades ahead. The resource is concentrated on exposed coastlines where the wave climate is energetic and consistent. + +## Device Siting + +Devices are sited where the average wave power exceeds a threshold that justifies the mooring cost, and where the seabed allows a fixed or moored installation. Siting therefore concentrates on exposed headlands rather than on sheltered channels. + +## Environmental Impact + +The main concern is not collision but the change in sediment transport, which can alter the shoreline over years. Monitoring programmes usually compare the seabed before commissioning with a survey several years later. The effect on surfing breaks has also blocked several projects. diff --git a/eval/corpus/wild-pollinators.md b/eval/corpus/wild-pollinators.md new file mode 100644 index 0000000..0512fa8 --- /dev/null +++ b/eval/corpus/wild-pollinators.md @@ -0,0 +1,13 @@ +# Wild Pollinator Notes + +## Overview + +Wild pollinators include solitary bees, hoverflies and bumblebees, and they do not live in managed hives. Their contribution is easy to overlook because it is not rented, counted or moved between orchards. + +## Solitary Bees + +Solitary bees nest in bare ground or in hollow stems rather than in colonies. They forage at lower temperatures than honeybees and often fly in light rain, so they can pollinate a bloom week that is too cold for a hive. + +## Habitat + +Hedgerows, unsprayed field margins and bare ground are the habitat that sustains them. Removing the margin to gain a few rows of crop usually costs more pollination than the extra rows return. diff --git a/eval/questions.jsonl b/eval/questions.jsonl new file mode 100644 index 0000000..e125d9c --- /dev/null +++ b/eval/questions.jsonl @@ -0,0 +1,30 @@ +{"id":"q001","question":"Why is bedload harder to measure than suspended sediment?","relevant":[{"document":"river-monitoring.md","page":null,"block":4,"quote":"Bedload is the harder fraction to measure"}]} +{"id":"q002","question":"How many replicate samples are collected at each river station?","relevant":[{"document":"river-monitoring.md","page":null,"block":6,"quote":"three replicate samples at each station"}]} +{"id":"q003","question":"What is the central trade-off in lithium-ion cell design?","relevant":[{"document":"battery-chemistry.md","page":null,"block":2,"quote":"trade energy density against thermal stability"}]} +{"id":"q004","question":"Why do nickel-rich battery packs need more aggressive thermal management?","relevant":[{"document":"battery-chemistry.md","page":null,"block":4,"quote":"release oxygen at lower temperatures than iron phosphate"}]} +{"id":"q005","question":"What happens once the separator in a battery cell melts?","relevant":[{"document":"battery-chemistry.md","page":null,"block":6,"quote":"Once the separator melts, the cell shorts internally"}]} +{"id":"q006","question":"At what temperature do honeybees begin to forage?","relevant":[{"document":"orchard-pollination.md","page":null,"block":4,"quote":"above roughly twelve degrees Celsius"}]} +{"id":"q007","question":"What does a late frost damage during full bloom?","relevant":[{"document":"orchard-pollination.md","page":null,"block":6,"quote":"destroys the flower's ovary rather than the petals"}]} +{"id":"q008","question":"Why is a continuous tree canopy more effective at cooling than isolated trees?","relevant":[{"document":"urban-canopy.md","page":null,"block":4,"quote":"asphalt stores far more heat during the day"}]} +{"id":"q009","question":"Why are trees with aggressive surface roots unsuitable for narrow verges?","relevant":[{"document":"urban-canopy.md","page":null,"block":6,"quote":"aggressive surface roots lift pavements"}]} +{"id":"q010","question":"At what temperature is lactic acid fermentation fastest?","relevant":[{"document":"fermentation.md","page":null,"block":4,"quote":"fastest between twenty and twenty-four degrees Celsius"}]} +{"id":"q011","question":"Is the salt percentage in fermentation based on vegetable weight or water weight?","relevant":[{"document":"fermentation.md","page":null,"block":6,"quote":"percentage of the vegetable weight, not the water weight"}]} +{"id":"q012","question":"Why must tidal turbines be sited in places with very fast currents?","relevant":[{"document":"tidal-energy.md","page":null,"block":4,"quote":"power scales with the cube of velocity"}]} +{"id":"q013","question":"What is the main environmental concern for tidal energy installations?","relevant":[{"document":"tidal-energy.md","page":null,"block":6,"quote":"change in sediment transport"}]} +{"id":"q014","question":"In lake monitoring, how is the sampling depth actually recorded?","relevant":[{"document":"lake-monitoring.md","page":null,"block":6,"quote":"Depth is recorded from the profiling sonde"}]} +{"id":"q015","question":"Why does deep-water oxygen fall while a lake remains stratified?","relevant":[{"document":"lake-monitoring.md","page":null,"block":4,"quote":"Oxygen in the hypolimnion is not replenished"}]} +{"id":"q016","question":"How do supercapacitors hold their charge?","relevant":[{"document":"supercapacitors.md","page":null,"block":4,"quote":"electric double layer formed at the surface of a porous carbon electrode"}]} +{"id":"q017","question":"Why can solitary bees pollinate a bloom week that is too cold for honeybee hives?","relevant":[{"document":"wild-pollinators.md","page":null,"block":4,"quote":"forage at lower temperatures than honeybees"}]} +{"id":"q018","question":"Why is one continuous planted roof layer better than several isolated planted beds?","relevant":[{"document":"green-roofs.md","page":null,"block":4,"quote":"an exposed membrane absorbs far more heat during the day"}]} +{"id":"q019","question":"Where do acetic acid bacteria sit in a vinegar culture?","relevant":[{"document":"vinegar-production.md","page":null,"block":4,"quote":"surface of the liquid where oxygen is available"}]} +{"id":"q020","question":"What happens if a vinegar culture is sealed airtight?","relevant":[{"document":"vinegar-production.md","page":null,"block":6,"quote":"Sealing a vinegar culture airtight stops the conversion entirely"}]} +{"id":"q021","question":"Why is wave energy harder to schedule ahead than tidal energy?","relevant":[{"document":"wave-energy.md","page":null,"block":2,"quote":"driven by wind rather than by the moon"}]} +{"id":"q022","question":"Where does siting for wave energy devices concentrate, and where does it not?","relevant":[{"document":"wave-energy.md","page":null,"block":4,"quote":"exposed headlands rather than on sheltered channels"}]} +{"id":"q023","question":"How does the river sampling protocol differ from the lake sampling protocol?","relevant":[{"document":"river-monitoring.md","page":null,"block":6,"quote":"three replicate samples at each station"},{"document":"lake-monitoring.md","page":null,"block":6,"quote":"Depth is recorded from the profiling sonde"}]} +{"id":"q024","question":"A street canopy and a green roof are both said to cool; what surface does each one shade?","relevant":[{"document":"urban-canopy.md","page":null,"block":4,"quote":"asphalt stores far more heat during the day"},{"document":"green-roofs.md","page":null,"block":4,"quote":"an exposed membrane absorbs far more heat during the day"}]} +{"id":"q025","question":"Which preservation method depends on keeping air away from the food?","relevant":[{"document":"fermentation.md","page":null,"block":2,"quote":"an anaerobic environment"}]} +{"id":"q026","question":"Why can one cold morning cost a grower the whole crop even when colonies are brought in?","relevant":[{"document":"orchard-pollination.md","page":null,"block":4,"quote":"above roughly twelve degrees Celsius"}]} +{"id":"q027","question":"绿茶应该怎样保存才能减缓氧化?","relevant":[{"document":"chinese-tea-storage.md","page":null,"block":2,"quote":"绿茶最容易氧化变质,需要低温密封保存"}]} +{"id":"q028","question":"茶叶储存的相对湿度上限是多少?","relevant":[{"document":"chinese-tea-storage.md","page":null,"block":4,"quote":"相对湿度应保持在百分之五十以下"}]} +{"id":"q029","question":"为什么冷冻保存的茶叶取出后不能立刻打开包装?","relevant":[{"document":"chinese-tea-storage.md","page":null,"block":6,"quote":"冷凝水会直接落在茶叶上"}]} +{"id":"q030","question":"为什么潮汐能比风能和太阳能更容易提前安排发电?","relevant":[{"document":"tidal-energy.md","page":null,"block":2,"quote":"predictable decades ahead, unlike wind or solar"}]} diff --git a/package.json b/package.json index 4e3826e..ed669e0 100644 --- a/package.json +++ b/package.json @@ -28,6 +28,8 @@ "postinstall": "electron-builder install-app-deps", "build:unpack": "npm run build && electron-builder --dir", "smoke:packaged": "node scripts/smoke-packaged.mjs", + "eval:prepare": "npm run build && node scripts/eval.mjs --prepare", + "eval": "npm run build && node scripts/eval.mjs", "build:win": "npm run build && electron-builder --win", "build:mac": "npm run build && electron-builder --mac", "build:linux": "npm run build && electron-builder --linux", diff --git a/scripts/eval.mjs b/scripts/eval.mjs new file mode 100644 index 0000000..0f4d304 --- /dev/null +++ b/scripts/eval.mjs @@ -0,0 +1,57 @@ +#!/usr/bin/env node +/** + * Launches the RAG eval harness (#75) through the real Electron main process. + * + * The harness needs the app's database, vector store and document loaders, none + * of which load outside Electron. `--eval-prepare` is the one-time, networked + * model bootstrap; `npm run eval` itself performs no network access. + * + * Usage: + * node scripts/eval.mjs # run the harness (offline) + * node scripts/eval.mjs --prepare # download the pinned model + */ + +import { spawn } from 'node:child_process' +import { existsSync } from 'node:fs' +import { resolve } from 'node:path' + +const prepare = process.argv.includes('--prepare') +const flag = prepare ? '--eval-prepare' : '--eval-harness' + +const executable = resolve( + 'node_modules/.bin', + process.platform === 'win32' ? 'electron.cmd' : 'electron' +) + +if (!existsSync(executable)) { + console.error('[eval] could not find the electron binary. Run `npm install` first.') + process.exit(1) +} + +const args = ['.', flag] + +// The Chromium sandbox needs a setuid helper or user namespaces; containers run +// as root and CI runners often lack both. The harness never renders anything, so +// running unsandboxed is safe here and keeps `npm run eval` working in CI. +const isRoot = typeof process.getuid === 'function' && process.getuid() === 0 +if (isRoot || process.env.CI) { + args.push('--no-sandbox') +} + +const child = spawn(executable, args, { + stdio: 'inherit', + env: { ...process.env, ELECTRON_DISABLE_SECURITY_WARNINGS: '1' } +}) + +child.on('error', (error) => { + console.error('[eval] failed to launch:', error.message) + process.exit(1) +}) + +child.on('exit', (code, signal) => { + if (signal) { + console.error(`[eval] app was killed by ${signal}`) + process.exit(1) + } + process.exit(code ?? 1) +}) diff --git a/src/main/eval/harness.ts b/src/main/eval/harness.ts new file mode 100644 index 0000000..0678a45 --- /dev/null +++ b/src/main/eval/harness.ts @@ -0,0 +1,264 @@ +/** + * RAG eval harness (#75). + * + * Indexes a corpus through the normal ingestion path, runs the real `Retriever`, + * and reports retrieval and citation metrics. It is the reference point every + * v1.5 experiment (#77, #78) must be measured against — so the numbers are + * produced here, never typed by hand. + * + * Ground truth is resolved from corpus identity to runtime identity **after** + * ingestion, because `documentId` is random and `blockId` embeds it. Nothing in + * the committed dataset references a runtime id. + */ + +import { readdir, readFile } from 'fs/promises' +import { join, posix } from 'path' +import { and, eq } from 'drizzle-orm' +import { documentBlocks, notebooks } from '../db/schema' +import type { getDatabase } from '../db' +import type { KnowledgeService } from '../services/KnowledgeService' +import { DEFAULT_CHUNK_OPTIONS } from '../services/ChunkingService' +import { LOCAL_EMBEDDING_MODEL } from '../embedding/localModel' +import { + evidencePrecisionAtK, + firstRelevantRank, + mean, + ndcgAtK, + percentile, + recallAtK, + reciprocalRank +} from './metrics' +import type { + EvalDeterministicReport, + EvalQuestion, + EvalReport, + EvalRelevantLocation, + QuestionReport, + ResolvedGroundTruth +} from './types' + +type Db = ReturnType + +export interface EvalHarnessOptions { + /** Absolute path used to read the corpus. */ + corpusDir: string + /** Repo-relative label recorded in the report, so the JSON is machine-independent. */ + corpusLabel: string + questionsPath: string + baseline: string + /** Ranks to compute Recall@k for. */ + topK: number + /** Similarity floor; 0 keeps the ranking intact for ranking metrics. */ + threshold: number + /** How many retrieved passages the evidence-precision metric looks at. */ + evidenceK: number +} + +const NOTEBOOK_ID = 'eval-notebook' + +/** Normalised comparison for the optional quote drift check. */ +const normalize = (text: string): string => text.toLowerCase().replace(/\s+/g, ' ').trim() + +function parseQuestions(raw: string): EvalQuestion[] { + return raw + .split('\n') + .map((line) => line.trim()) + .filter((line) => line.length > 0) + .map((line, index) => { + try { + return JSON.parse(line) as EvalQuestion + } catch (error) { + throw new Error(`questions.jsonl line ${index + 1} is not valid JSON: ${String(error)}`) + } + }) +} + +/** + * 把 `document`/`block` 顺序解析成运行期的 `documentId`/`blockId`,并用可选 quote + * 校验块顺序没有因为 parser 改动而漂移。 + */ +function resolveGroundTruth( + db: Db, + question: EvalQuestion, + documentIds: Map +): ResolvedGroundTruth[] { + return question.relevant.map((location: EvalRelevantLocation) => { + const documentId = documentIds.get(location.document) + if (!documentId) { + throw new Error( + `question ${question.id} references "${location.document}", which is not in the corpus` + ) + } + + const block = db + .select() + .from(documentBlocks) + .where( + and(eq(documentBlocks.documentId, documentId), eq(documentBlocks.order, location.block)) + ) + .get() + + if (!block) { + throw new Error( + `question ${question.id}: "${location.document}" has no block with ordinal ${location.block}` + ) + } + if (location.quote && !normalize(block.text).includes(normalize(location.quote))) { + throw new Error( + `question ${question.id}: quote drifted — block ${location.block} of "${location.document}" ` + + `does not contain ${JSON.stringify(location.quote)}. Update the ground truth deliberately.` + ) + } + if (location.page !== null && block.page !== location.page) { + throw new Error( + `question ${question.id}: "${location.document}" block ${location.block} is on page ` + + `${block.page}, ground truth says ${location.page}` + ) + } + + return { + document: location.document, + documentId, + blockId: block.id, + page: block.page, + block: location.block + } + }) +} + +/** Index every corpus document and return `corpus-relative path → runtime documentId`. */ +async function indexCorpus( + db: Db, + knowledgeService: KnowledgeService, + corpusDir: string +): Promise> { + const now = new Date() + db.insert(notebooks) + .values({ id: NOTEBOOK_ID, title: 'Eval corpus', createdAt: now, updatedAt: now }) + .run() + + const files = (await readdir(corpusDir)).filter((name) => !name.startsWith('.')).sort() + const documentIds = new Map() + + for (const file of files) { + const documentId = await knowledgeService.addDocumentFromFile( + NOTEBOOK_ID, + join(corpusDir, file) + ) + documentIds.set(posix.normalize(file), documentId) + } + + return documentIds +} + +export async function runEvalHarness( + db: Db, + knowledgeService: KnowledgeService, + options: EvalHarnessOptions +): Promise { + const documentIds = await indexCorpus(db, knowledgeService, options.corpusDir) + const questions = parseQuestions(await readFile(options.questionsPath, 'utf-8')) + + const perQuestion: QuestionReport[] = [] + const latencies: number[] = [] + + for (const question of questions) { + const groundTruth = resolveGroundTruth(db, question, documentIds) + const groundTruthIds = groundTruth.map((entry) => entry.blockId) + + const started = performance.now() + const results = await knowledgeService.search(NOTEBOOK_ID, question.question, { + topK: options.topK, + threshold: options.threshold + }) + const latencyMs = performance.now() - started + latencies.push(latencyMs) + + const matchesByRank = results.map((result) => { + const blockIds = new Set(result.locator.blocks.map((block) => block.blockId)) + return groundTruthIds + .map((blockId, index) => (blockIds.has(blockId) ? index : -1)) + .filter((index) => index >= 0) + }) + + perQuestion.push({ + id: question.id, + question: question.question, + firstRelevantRank: firstRelevantRank(matchesByRank), + relevantCount: groundTruth.length, + retrievedCount: results.length, + matchesByRank + }) + } + + const metrics = { + recallAt1: mean(perQuestion.map((q) => recallAtK(q.matchesByRank, q.relevantCount, 1))), + recallAt5: mean(perQuestion.map((q) => recallAtK(q.matchesByRank, q.relevantCount, 5))), + recallAt10: mean(perQuestion.map((q) => recallAtK(q.matchesByRank, q.relevantCount, 10))), + mrr: mean(perQuestion.map((q) => reciprocalRank(q.matchesByRank))), + ndcgAt10: mean(perQuestion.map((q) => ndcgAtK(q.matchesByRank, q.relevantCount, 10))), + evidencePrecisionAt5: mean( + perQuestion.map((q) => evidencePrecisionAtK(q.matchesByRank, options.evidenceK)) + ) + } + + const chunking = DEFAULT_CHUNK_OPTIONS + return { + baseline: 'v1.4', + generatedBy: 'npm run eval', + config: { + embedding: `${LOCAL_EMBEDDING_MODEL.id}@${LOCAL_EMBEDDING_MODEL.revision} ${LOCAL_EMBEDDING_MODEL.dtype} (${LOCAL_EMBEDDING_MODEL.dimensions}d, local)`, + chunking: { + chunkSize: chunking.chunkSize, + chunkOverlap: chunking.chunkOverlap, + minChunkSize: chunking.minChunkSize, + allowSpanPages: chunking.allowSpanPages + }, + retrieval: 'dense', + topK: options.topK, + threshold: options.threshold, + evidenceK: options.evidenceK, + corpus: options.corpusLabel, + documents: documentIds.size, + questions: questions.length + }, + metrics, + timing: { + latencyP50Ms: percentile(latencies, 50), + latencyP95Ms: percentile(latencies, 95) + }, + perQuestion + } +} + +/** Round metrics to a stable number of decimals so the JSON diffs cleanly. */ +export function stabilize(report: EvalReport): EvalReport { + const round = (value: number): number => Number(value.toFixed(6)) + return { + ...report, + metrics: { + recallAt1: round(report.metrics.recallAt1), + recallAt5: round(report.metrics.recallAt5), + recallAt10: round(report.metrics.recallAt10), + mrr: round(report.metrics.mrr), + ndcgAt10: round(report.metrics.ndcgAt10), + evidencePrecisionAt5: round(report.metrics.evidencePrecisionAt5) + }, + timing: { + latencyP50Ms: round(report.timing.latencyP50Ms), + latencyP95Ms: round(report.timing.latencyP95Ms) + }, + perQuestion: report.perQuestion + } +} + +/** Strip the non-deterministic timing so the committed JSON is diff-stable. */ +export function toDeterministicReport(report: EvalReport): EvalDeterministicReport { + return { + baseline: report.baseline, + generatedBy: report.generatedBy, + config: report.config, + metrics: report.metrics, + perQuestion: report.perQuestion + } +} diff --git a/src/main/eval/metrics.ts b/src/main/eval/metrics.ts new file mode 100644 index 0000000..c7dd764 --- /dev/null +++ b/src/main/eval/metrics.ts @@ -0,0 +1,88 @@ +/** + * Retrieval metrics (#75). + * + * Pure functions over the per-question match matrix, so the arithmetic is unit + * tested and the harness only has to supply retrieval output. Relevance is + * binary (an item either covers a ground-truth block or it does not) and gains + * are not graded — the dataset records locations, not degrees of relevance. + */ + +/** Ground-truth indices matched by each retrieved rank, in rank order. */ +export type MatchMatrix = number[][] + +/** 1-based rank of the first relevant item, or 0 when none is relevant. */ +export function firstRelevantRank(matchesByRank: MatchMatrix): number { + const index = matchesByRank.findIndex((matches) => matches.length > 0) + return index === -1 ? 0 : index + 1 +} + +/** Fraction of ground-truth locations covered by the first `k` ranks. */ +export function recallAtK(matchesByRank: MatchMatrix, groundTruthCount: number, k: number): number { + if (groundTruthCount === 0) return 0 + const covered = new Set() + for (const matches of matchesByRank.slice(0, k)) { + for (const match of matches) covered.add(match) + } + return covered.size / groundTruthCount +} + +/** Mean reciprocal rank of the first relevant item. */ +export function reciprocalRank(matchesByRank: MatchMatrix): number { + const rank = firstRelevantRank(matchesByRank) + return rank === 0 ? 0 : 1 / rank +} + +/** + * nDCG@k with binary gains. The ideal ranking puts every ground-truth location + * first, so the discount is a plain log base 2. + */ +export function ndcgAtK(matchesByRank: MatchMatrix, groundTruthCount: number, k: number): number { + if (groundTruthCount === 0) return 0 + + let dcg = 0 + const limit = Math.min(matchesByRank.length, k) + for (let i = 0; i < limit; i++) { + if (matchesByRank[i].length > 0) dcg += 1 / Math.log2(i + 2) + } + + let idcg = 0 + const ideal = Math.min(groundTruthCount, k) + for (let i = 0; i < ideal; i++) idcg += 1 / Math.log2(i + 2) + + return idcg === 0 ? 0 : dcg / idcg +} + +/** + * Evidence precision@k: of the first `k` retrieved passages, the share that cover + * a ground-truth block. + * + * This is retrieval precision, **not** answer citation recall. No model runs in + * this harness and no answer is produced, so a metric that claims to be about an + * answer's citations would be a false claim. Answer-level citation correctness is + * the resolver's job (#70); a model-driven answer eval is a separate deliverable. + * + * Each retrieved passage is counted once, however many ground-truth blocks it + * covers. Overlapping chunks can cover the same block, and each still counts as a + * separate retrieved passage — a reader would see two citations there, and this + * metric describes what they would be shown. + */ +export function evidencePrecisionAtK(matchesByRank: MatchMatrix, k: number): number { + const retrieved = matchesByRank.slice(0, k) + if (retrieved.length === 0) return 0 + const grounded = retrieved.filter((matches) => matches.length > 0).length + return grounded / retrieved.length +} + +export function mean(values: number[]): number { + if (values.length === 0) return 0 + return values.reduce((sum, value) => sum + value, 0) / values.length +} + +/** Nearest-rank percentile. `p` is 0-100. */ +export function percentile(values: number[], p: number): number { + if (values.length === 0) return 0 + const sorted = [...values].sort((a, b) => a - b) + const rank = Math.ceil((p / 100) * sorted.length) + const index = Math.min(Math.max(rank - 1, 0), sorted.length - 1) + return sorted[index] +} diff --git a/src/main/eval/report.ts b/src/main/eval/report.ts new file mode 100644 index 0000000..bcec7c9 --- /dev/null +++ b/src/main/eval/report.ts @@ -0,0 +1,76 @@ +import type { EvalReport } from './types' + +/** + * Renders the committed baseline report. Kept separate from the runner so the + * document is a pure function of the report object: there is exactly one place + * that decides how a number is displayed, and it cannot disagree with the JSON + * written beside it. + */ + +const format = (value: number): string => value.toFixed(4) + +export function renderMarkdown(report: EvalReport): string { + const { config, metrics, timing } = report + const chunking = config.chunking + + return `# RAG eval baseline — ${report.baseline} + +Generated by \`${report.generatedBy}\`. The numbers below are harness output — do not edit them by hand. + +## Configuration + +| Setting | Value | +| --- | --- | +| Embedding | \`${config.embedding}\` | +| Chunking | \`chunkSize=${chunking.chunkSize}, chunkOverlap=${chunking.chunkOverlap}, minChunkSize=${chunking.minChunkSize}, allowSpanPages=${chunking.allowSpanPages}\` | +| Retrieval | \`${config.retrieval}\` | +| Ranks | \`topK=${config.topK}, threshold=${config.threshold}\` | +| Evidence per query | \`evidenceK=${config.evidenceK}\` | +| Corpus | \`${config.corpus}\` (${config.documents} documents, ${config.questions} questions) | + +## Metrics + +| Metric | Value | +| --- | --- | +| Recall@1 | ${format(metrics.recallAt1)} | +| Recall@5 | ${format(metrics.recallAt5)} | +| Recall@10 | ${format(metrics.recallAt10)} | +| MRR | ${format(metrics.mrr)} | +| nDCG@10 | ${format(metrics.ndcgAt10)} | +| Evidence precision@${config.evidenceK} | ${format(metrics.evidencePrecisionAt5)} | + +Timing is informational only and is **not** frozen: p50 ${timing.latencyP50Ms.toFixed(2)} ms, +p95 ${timing.latencyP95Ms.toFixed(2)} ms on the machine that produced this file. Query +latency depends on hardware and load, so it must never be the reason two runs differ. + +## Definitions + +- A retrieved passage is relevant when its provenance covers a ground-truth block. +- **Recall@k** is the share of ground-truth blocks covered by the first \`k\` passages. +- **Evidence precision@${config.evidenceK}** is the share of the first \`${config.evidenceK}\` + retrieved passages that cover a ground-truth block. This is **retrieval precision**, not + answer citation recall: the harness runs no model and produces no answer. Answer-level + citation correctness is covered by the resolver (#70); a model-driven answer eval is a + separate deliverable. +- Ground truth is expressed in corpus identity (\`document\` relative path + \`block\` + ordinal + optional \`quote\`), never a runtime \`documentId\`/\`blockId\`. + +## Comparison protocol + +v1.5 experiments (#77, #78) are reported as a **delta against this file**. The +adopted-change rule is: + +> Adopt a change only if Recall@5 improves and nDCG@10 does not regress. A change +> that trades a large latency increase for a marginal recall gain is a product +> decision, not an automatic win, and must be stated as such. + +A changed result must be reproducible with: + +\`\`\`bash +npm run eval:prepare # one-time, networked model bootstrap +npm run eval # offline and deterministic +\`\`\` + +The raw report is committed next to this summary as \`baseline-${report.baseline}.json\`. +` +} diff --git a/src/main/eval/run.ts b/src/main/eval/run.ts new file mode 100644 index 0000000..d999722 --- /dev/null +++ b/src/main/eval/run.ts @@ -0,0 +1,144 @@ +/** + * Eval harness entry point (#75), enabled by `--eval-harness` / `--eval-prepare`. + * + * Runs in the real main process because the DB layer, the vector store and the + * loaders only exist there — a plain Node runner would have to reimplement the + * ingestion path it is supposed to measure. It never opens a window. + * + * `--eval-prepare` is the only step allowed to touch the network: it downloads + * the pinned embedding model into the real model cache. `--eval-harness` then + * runs offline against a throwaway database and refuses to download. + */ + +import { app } from 'electron' +import { mkdtempSync, rmSync, writeFileSync } from 'fs' +import { mkdir } from 'fs/promises' +import { join, relative, resolve } from 'path' +import { tmpdir } from 'os' +import { closeDatabase, getDatabase, initDatabase, initVectorStore, runMigrations } from '../db' +import { ConnectionManager } from '../models/ConnectionManager' +import { EmbeddingService } from '../services/EmbeddingService' +import { KnowledgeService } from '../services/KnowledgeService' +import { isModelInstalled } from '../embedding/ModelRegistry' +import { + runEvalHarness, + stabilize, + toDeterministicReport, + type EvalHarnessOptions +} from './harness' +import { renderMarkdown } from './report' + +export const EVAL_FLAG = '--eval-harness' +export const EVAL_PREPARE_FLAG = '--eval-prepare' + +export function isEvalRequested(argv: readonly string[] = process.argv): boolean { + return argv.includes(EVAL_FLAG) || argv.includes(EVAL_PREPARE_FLAG) +} + +function readOption(argv: readonly string[], prefix: string, fallback: string): string { + const arg = argv.find((value) => value.startsWith(prefix)) + return arg ? arg.slice(prefix.length) : fallback +} + +/** + * A ConnectionManager that can never produce a remote backend, so the harness + * always measures the built-in local model regardless of the developer's own + * embedding configuration. Determinism is a property of the baseline, not of the + * machine it was produced on. + */ +function localOnlyConnectionManager(): ConnectionManager { + // SAFETY: only `getEmbeddingClient` and `getConnection('embedding')` are reached by + // `EmbeddingService`; both always answer "no remote connection", so no other member of + // ConnectionManager is ever touched. A new call site fails at runtime, loudly. + return { + getEmbeddingClient: async () => null, + getConnection: async () => null + } as unknown as ConnectionManager +} + +export async function runEvalCli(argv: readonly string[] = process.argv): Promise { + const prepare = argv.includes(EVAL_PREPARE_FLAG) + const corpusDir = resolve(readOption(argv, '--eval-corpus=', 'eval/corpus')) + const questionsPath = resolve(readOption(argv, '--eval-questions=', 'eval/questions.jsonl')) + const outDir = resolve(readOption(argv, '--eval-out=', 'docs/eval')) + + // The real profile is captured before redirecting: the model cache lives under + // it and must survive the throwaway run. + const realUserData = app.getPath('userData') + const workDir = mkdtempSync(join(tmpdir(), 'knownote-eval-')) + app.setPath('userData', workDir) + + // `KNOWNOTE_EVAL_MODEL_CACHE` lets CI point at its own cache instead of the + // developer profile, without changing what production uses. + const modelCacheDir = process.env.KNOWNOTE_EVAL_MODEL_CACHE || join(realUserData, 'models') + + const embeddingService = new EmbeddingService(localOnlyConnectionManager(), { + cacheDir: modelCacheDir + }) + + let databaseInitialized = false + + try { + if (prepare) { + console.log(`[eval] preparing the pinned local embedding model into ${modelCacheDir}`) + await embeddingService.ensureReady((progress) => { + const percent = Math.round(progress.progress * 100) + if (percent > 0 && percent % 10 === 0) { + console.log(`[eval] model download ${percent}%`) + } + }) + console.log('[eval] model ready; `npm run eval` will now run offline') + return 0 + } + + if (!(await isModelInstalled(modelCacheDir))) { + console.error('[eval] Eval model is not available locally.') + console.error('[eval] Run: npm run eval:prepare') + return 1 + } + + initDatabase() + databaseInitialized = true + runMigrations() + initVectorStore() + + const knowledgeService = new KnowledgeService(embeddingService) + const options: EvalHarnessOptions = { + corpusDir, + // Recorded in the report as a repo-relative path so the committed JSON is + // identical on every machine and checkout. + corpusLabel: relative(process.cwd(), corpusDir) || 'eval/corpus', + questionsPath, + baseline: 'v1.4', + topK: 10, + threshold: 0, + evidenceK: 5 + } + + const report = stabilize(await runEvalHarness(getDatabase(), knowledgeService, options)) + + await mkdir(outDir, { recursive: true }) + const jsonPath = join(outDir, `baseline-${report.baseline}.json`) + const markdownPath = join(outDir, `baseline-${report.baseline}.md`) + writeFileSync(jsonPath, `${JSON.stringify(toDeterministicReport(report), null, 2)}\n`) + writeFileSync(markdownPath, renderMarkdown(report)) + + // A single machine-readable line so a determinism check can diff the metrics + // without parsing the whole report (timing is deliberately excluded). + console.log(`[eval] metrics ${JSON.stringify(report.metrics)}`) + console.log(`[eval] wrote ${jsonPath} and ${markdownPath}`) + return 0 + } catch (error) { + console.error('[eval] FAIL', error instanceof Error ? error.message : error) + return 1 + } finally { + if (databaseInitialized) { + try { + closeDatabase() + } catch { + // A close failure must not mask the eval result. + } + } + rmSync(workDir, { recursive: true, force: true }) + } +} diff --git a/src/main/eval/types.ts b/src/main/eval/types.ts new file mode 100644 index 0000000..cdb31e0 --- /dev/null +++ b/src/main/eval/types.ts @@ -0,0 +1,100 @@ +/** + * RAG eval harness types (#75). + * + * Ground truth is expressed in **corpus identity**, not database identity. A + * `document` is the corpus-relative path, a `block` is the block ordinal inside + * that document. Runtime `documentId`s are random, and `blockId` embeds them, so + * neither may appear in the committed ground truth: #78 changes chunking, and a + * ground truth bound to a runtime id would break instead of measuring the change. + */ + +export interface EvalRelevantLocation { + /** Corpus-relative path, e.g. `river-monitoring.md`. */ + document: string + /** Page number when the source is paginated, else null. */ + page: number | null + /** Block ordinal inside the document (`document_blocks.order`). */ + block: number + /** Optional excerpt used to detect parser drift in `block`. */ + quote?: string +} + +export interface EvalQuestion { + id: string + question: string + relevant: EvalRelevantLocation[] + goldAnswer?: string +} + +/** One resolved ground-truth location, after runtime id mapping. */ +export interface ResolvedGroundTruth { + document: string + documentId: string + blockId: string + page: number | null + block: number +} + +export interface EvalMetrics { + recallAt1: number + recallAt5: number + recallAt10: number + mrr: number + ndcgAt10: number + /** + * Share of the first `evidenceK` retrieved passages that cover ground truth. + * + * This is **retrieval precision**, not answer citation recall: no model runs in + * this harness and no answer is produced. Answer-level citation correctness is + * the resolver's job (#70) and would need a separate, model-driven eval. + */ + evidencePrecisionAt5: number +} + +export interface QuestionReport { + id: string + question: string + firstRelevantRank: number + relevantCount: number + retrievedCount: number + /** Ground-truth indices matched by each retrieved rank, in rank order. */ + matchesByRank: number[][] +} + +export interface EvalReport { + baseline: string + generatedBy: string + config: { + embedding: string + chunking: { + chunkSize: number + chunkOverlap: number + minChunkSize: number + allowSpanPages: boolean + } + retrieval: string + topK: number + threshold: number + /** How many retrieved passages the evidence-precision metric looks at. */ + evidenceK: number + corpus: string + documents: number + questions: number + } + metrics: EvalMetrics + timing: { latencyP50Ms: number; latencyP95Ms: number } + perQuestion: QuestionReport[] +} + +/** + * The committed report: everything that is identical between two runs. Wall-clock + * timing is deliberately absent, so `npm run eval` twice produces a byte-identical + * JSON that a PR can actually diff. + */ +export interface EvalDeterministicReport { + baseline: string + generatedBy: string + config: EvalReport['config'] + metrics: EvalMetrics + perQuestion: QuestionReport[] +} diff --git a/src/main/index.ts b/src/main/index.ts index 15c774a..e3ffe3f 100644 --- a/src/main/index.ts +++ b/src/main/index.ts @@ -20,6 +20,7 @@ import { broadcastProgress as broadcastEmbeddingProgress } from './ipc/embedding import { getStore } from './config/store' import { migrateProvidersToConnections } from './config/connectionMigration' import { isSmokeTestRequested, runSmokeTest } from './smokeTest' +import { isEvalRequested, runEvalCli } from './eval/run' import Logger from '../shared/utils/logger' let connectionManager: ConnectionManager | null = null @@ -44,6 +45,15 @@ app.whenReady().then(async () => { return } + // RAG eval harness (#75). Like the smoke test it runs before any window is + // created and exits with the result. + if (isEvalRequested()) { + const code = await runEvalCli() + await new Promise((resolve) => process.stdout.write('', () => resolve())) + app.exit(code) + return + } + // Set app user model id for windows electronApp.setAppUserModelId('com.knownote.app') diff --git a/src/main/services/ChunkingService.ts b/src/main/services/ChunkingService.ts index 5c459fe..5696f0f 100644 --- a/src/main/services/ChunkingService.ts +++ b/src/main/services/ChunkingService.ts @@ -23,6 +23,33 @@ export interface ChunkOptions { allowSpanPages?: boolean // 允许一个 chunk 跨页,默认 false(分页文档在页边界断开) } +/** + * 默认分块参数。导出的原因只有一个:eval baseline 报告必须引用真实值,而不是 + * 把它手抄一遍。改了默认值而没有重新跑 baseline,差异会从报告里直接暴露出来。 + */ +export const DEFAULT_CHUNK_OPTIONS: Required = { + chunkSize: 500, + chunkOverlap: 50, + minChunkSize: 100, + allowSpanPages: false, + separators: [ + '\n\n\n', // 多个空行(章节分隔) + '\n\n', // 段落分隔 + '\n', // 行分隔 + '。', // 中文句号 + '.', // 英文句号 + '!', + '!', + '?', + '?', + ';', + ';', + ',', + ',', + ' ' // 空格(最后手段) + ] +} + /** * 分块器接受的块。字段是 `document_blocks` 的子集。 */ @@ -74,28 +101,7 @@ interface Range { * 支持块感知分块(保留来源结构),以及无块时的字符窗口回退。 */ export class ChunkingService { - private defaultOptions: Required = { - chunkSize: 500, - chunkOverlap: 50, - minChunkSize: 100, - allowSpanPages: false, - separators: [ - '\n\n\n', // 多个空行(章节分隔) - '\n\n', // 段落分隔 - '\n', // 行分隔 - '。', // 中文句号 - '.', // 英文句号 - '!', - '!', - '?', - '?', - ';', - ';', - ',', - ',', - ' ' // 空格(最后手段) - ] - } + private defaultOptions: Required = DEFAULT_CHUNK_OPTIONS /** * 块感知分块。偏移与内容都锚定在 `content` 这份规范字符串上。 diff --git a/test/evalMetrics.test.ts b/test/evalMetrics.test.ts new file mode 100644 index 0000000..10d7e03 --- /dev/null +++ b/test/evalMetrics.test.ts @@ -0,0 +1,71 @@ +import { test } from 'node:test' +import assert from 'node:assert/strict' +import { + evidencePrecisionAtK, + firstRelevantRank, + mean, + ndcgAtK, + percentile, + recallAtK, + reciprocalRank +} from '../src/main/eval/metrics.ts' + +/** + * The eval metrics are the only numbers v1.5 experiments (#77, #78) are allowed + * to argue from, so they are pinned by hand here rather than trusted to the + * harness that calls them. + */ + +test('recall@k covers ground truth within the first k ranks', () => { + // rank 1 matches gt 0, rank 2 matches nothing, rank 3 matches gt 1 + const matches = [[0], [], [1]] + + assert.equal(recallAtK(matches, 2, 1), 0.5) + assert.equal(recallAtK(matches, 2, 2), 0.5) + assert.equal(recallAtK(matches, 2, 3), 1) +}) + +test('a repeated match does not inflate recall past 1', () => { + const matches = [[0], [0], [0]] + assert.equal(recallAtK(matches, 1, 3), 1) +}) + +test('recall is 0 when there is no ground truth', () => { + assert.equal(recallAtK([[]], 0, 5), 0) +}) + +test('first relevant rank is 1-based and 0 when nothing is relevant', () => { + assert.equal(firstRelevantRank([[], [], [2]]), 3) + assert.equal(firstRelevantRank([[], []]), 0) + assert.equal(firstRelevantRank([]), 0) +}) + +test('reciprocal rank is 1/rank of the first hit', () => { + assert.equal(reciprocalRank([[0]]), 1) + assert.equal(reciprocalRank([[], [0]]), 0.5) + assert.equal(reciprocalRank([[], []]), 0) +}) + +test('nDCG@k discounts a later hit and is 1 when the hit is first', () => { + assert.equal(ndcgAtK([[0]], 1, 10), 1) + // A single ground truth at rank 2: 1/log2(3) over the ideal 1/log2(2) + assert.ok(Math.abs(ndcgAtK([[], [0]], 1, 10) - 1 / Math.log2(3)) < 1e-12) + assert.equal(ndcgAtK([[], []], 1, 10), 0) +}) + +test('evidence precision counts grounded passages over retrieved passages', () => { + // 2 of 3 retrieved passages cover a ground-truth block + assert.equal(evidencePrecisionAtK([[0], [], [1]], 3), 2 / 3) + assert.equal(evidencePrecisionAtK([], 5), 0) + // Each retrieved passage counts once, even when several cover the same block + assert.equal(evidencePrecisionAtK([[0], [0], [0]], 3), 1) +}) + +test('mean and percentile handle the empty and single cases', () => { + assert.equal(mean([]), 0) + assert.equal(mean([1, 2, 3]), 2) + assert.equal(percentile([], 50), 0) + assert.equal(percentile([42], 95), 42) + assert.equal(percentile([1, 2, 3, 4, 5], 50), 3) + assert.equal(percentile([5, 1, 4, 2, 3], 95), 5) +})