diff --git a/.claude/state/assess-comments.coverage.md b/.claude/state/assess-comments.coverage.md index d4c3ffc3b1..9badb2c4ca 100644 --- a/.claude/state/assess-comments.coverage.md +++ b/.claude/state/assess-comments.coverage.md @@ -27,23 +27,23 @@ whether to commit the change. | Capability | Confidence | Real encounters | Last verified | Evidence | |---|---|---|---|---| -| Branch/PR identification + arg handling | 🟒 corroborated | 8 | 2026-07-02 | #3415, #3507, #3374, #3536, #3542, #3543, #3573, #2531 | -| Multi-source collection (inline + top-level + reviews) | 🟒 corroborated | 9 | 2026-07-02 | #3415, #3507, #3510, #3374, #3536, #3542, #3543, #3573, #2531 | -| GraphQL thread-resolution pull (`isResolved`/`isOutdated`) | 🟒 corroborated | 7 | 2026-07-02 | #3510 (12/12 resolved), #3374 (15/15), #3536 (2/2 open), #3542 (3 resolved/1 open), #3543 (1/1 open), #3573 (2/2 open), #2531 (r1: 11 open/2 res-outdated/1 open; r2: 11 of 14 outdated after fixes + APPROVED) | -| Source-role tagging (bugbot/security/history/summary/ci/human) | 🟒 corroborated | 8 | 2026-07-02 | #3415, #3507, #3374, #3536, #3542, #3543, #3573, #2531 | -| Open/resolved split | 🟒 corroborated | 7 | 2026-07-02 | #3510, #3374, #3536 (0 resolved/2 open), #3542 (3 resolved/1 open), #3543, #3573 (r1/r2 mixed), #2531 (11 open / 3 resolved) | -| Fix-quality spot-check (genuinely fixed vs silenced) | 🟒 corroborated | 5 | 2026-07-02 | #3510 (term removals landed), #3374 (`num_docs`, dropIndex landed), #3542 (xargs+guard, narrowed exclude, SHA pin landed), #3573 (relpath + `--add` dedup landed), #2531 (r1 found decimal thread 2619563419 reverted; r2 doc fixes landed + engineer ZdravkoDonev **APPROVED**) | +| Branch/PR identification + arg handling | 🟒 corroborated | 9 | 2026-07-24 | #3415, #3507, #3374, #3536, #3542, #3543, #3573, #2531, #3585 | +| Multi-source collection (inline + top-level + reviews) | 🟒 corroborated | 10 | 2026-07-24 | #3415, #3507, #3510, #3374, #3536, #3542, #3543, #3573, #2531, #3585 | +| GraphQL thread-resolution pull (`isResolved`/`isOutdated`) | 🟒 corroborated | 8 | 2026-07-24 | #3510 (12/12 resolved), #3374 (15/15), #3536 (2/2 open), #3542 (3 resolved/1 open), #3543 (1/1 open), #3573 (2/2 open), #2531 (r1: 11 open/2 res-outdated/1 open; r2: 11 of 14 outdated after fixes + APPROVED), #3585 (10 threads: 2 open / 8 resolved, of which 1 resolved+not-outdated) | +| Source-role tagging (bugbot/security/history/summary/ci/human) | 🟒 corroborated | 9 | 2026-07-24 | #3415, #3507, #3374, #3536, #3542, #3543, #3573, #2531, #3585 (bugbot + Jit security + CLA/Jira ci + **Redis Memory history bot** + Cursor summary) | +| Open/resolved split | 🟒 corroborated | 8 | 2026-07-24 | #3510, #3374, #3536 (0 resolved/2 open), #3542 (3 resolved/1 open), #3543, #3573 (r1/r2 mixed), #2531 (11 open / 3 resolved), #3585 (2 open / 8 resolved) | +| Fix-quality spot-check (genuinely fixed vs silenced) | 🟒 corroborated | 6 | 2026-07-24 | #3510 (term removals landed), #3374 (`num_docs`, dropIndex landed), #3542 (xargs+guard, narrowed exclude, SHA pin landed), #3573 (relpath + `--add` dedup landed), #2531 (r1 found decimal thread 2619563419 reverted; r2 doc fixes landed + engineer ZdravkoDonev **APPROVED**), #3585 (dup-id High genuinely fixed in get-page.ts via ambiguous()+candidates; EMBED_MODEL fix landed 85e1d0d17 β€” both real, not silenced; r2: the 2 prior-open findings confirmed genuinely fixed β€” empty-index guard added to index.ts main() [now resolved+not-outdated, Medium], matchingSections deferred past top-k with lexical/hybrid MRR .525/.704 unchanged) | | "Resolved β‰  fixed" flag β€” **legitimate deferral** variant | 🟑 seen once | 1 | 2026-06-23 | #3510 (TS.BGET:122 left pending eng) | | "Resolved β‰  fixed" flag β€” **still-broken / reverted** variant | 🟑 seen once | 1 | 2026-06-30 | #2531 (resolved+outdated thread 2619563419 said decimal default=`string`; a later rewrite reverted current code to `precise`, so the resolved fix is no longer in the code β€” engineer re-raised it as 3496835587). Regression flavour; see worked examples | -| Cross-tool **agreement** | 🟑 seen once | 1 | 2026-06-23 | #3374 (Claude + bugbot independently on `num_docs`) | +| Cross-tool **agreement** | 🟒 corroborated | 2 | 2026-07-24 | #3374 (Claude + bugbot independently on `num_docs`); #3585 (bugbot finding 3639447417 "EMBED_MODEL documented but never read" independently matched the orphaned diff the author had already identified & committed as 85e1d0d17 β€” both flagged the parametrization missing from committed code) | | **Contradiction** detection | 🟒 corroborated | 2 | 2026-07-02 | #3415 (approval vs open bugbot finding); #2531 (RDI engineer's repo ground truth contradicts the page's Debezium-docs claims on β‰₯4 points β€” version, decimal default, temporal pass-through, MariaDB connector β€” **and** engineer-vs-existing-doc on temporal normalization). *(#3507 was an off-branch manual demo β€” not counted.)* | | **Ping-pong loop** detection | ❓ untested | 0 | 2026-07-02 | still no true tool A↔B loop across #3536 (4 rounds), #3542 (r2 "empty-scope"), #3573 (r1 fixed point; r2 independent), or #2531. #3542/#3573 were churn not loops; #2531's nearest reopened-concern was the decimal regression (resolved Dec β†’ reverted by a June rewrite β†’ re-raised) β€” a regression across one rewrite, not a cycle | | **Subsystem churn** detection (repeated findings on one patched area) | 🟒 corroborated | 3 PRs | 2026-07-02 | 3 distinct PRs. #3536 β€” 3 instances (review-handling / churn-feature / cap↔report contract). #3542 β€” 2 instances on the extraction *fail-loud-on-empty* contract (r1 ARG_MAX silent-green β†’ r2 sibling zero-files `exit 0`). #3573 β€” 2 instances on the `--add` virtual-merge mechanism (r1 dup-vs-disk β†’ r2 dropped-under-collapse). Worked examples below | | Approval-over-open-finding cross-check | 🟒 corroborated | 6 | 2026-07-02 | #3415 (dwdougherty), #3374 (low-confidence over open HIGH), #3536 (high-confidence over 2 open Mediums: benign), #3542 (paoloredis "yep go ahead" 7 min after open Medium #3498159511; unacknowledged), #3573 (dwdougherty "Sure, why not?" APPROVED 13:41 over open findings; 2 bot findings landed 13:49 after), #2531 (run1 correct **negative** β€” no approval; run2 **positive** β€” ZdravkoDonev APPROVED 13:13 then bugbot finding 3499796857 landed 15:08, and he approved over 2-3 of his own still-open findings incl. the temporal one) | | Depth cap / prioritisation under load | 🟒 corroborated | 2 | 2026-07-02 | #3374 (19 candidate findings β†’ 4 deep-verified); #2531 (r1: 14 threads β†’ 5 deep-verified, 6 deferred). *(#3542/#3573 were under cap β€” not load tests)* | -| Mandatory deep-verify of resolved+not-outdated HIGH | 🟑 seen once | 1 | 2026-07-02 | #3542 #3467309496 (High "Grep failure skips link check", resolved + isOutdated:false) β€” deep-verified against current code: xargs+guard genuinely present, so legitimately fixed (not still-broken). First real firing of the rule | -| Bot calibration (fixed-vs-dismissed ratio) | 🟒 corroborated | 6 | 2026-07-02 | #3374 (bugbot mostly accepted); #3536 (5/5 valid); #3542 (3/3 valid β€” 2 fixed, 1 open); #3543 (1/1 valid; Jit 0); #3573 (4/4 valid; Jit 0); #2531 (r1 bugbot 0 findings; r2 bugbot 1/1 valid β€” caught the ledger duplicate-rows defect 3499796857; Jit 0) | -| Codex second-opinion availability gate | 🟒 corroborated | 6 | 2026-07-02 | #3415, #3374 (CLI on PATH; #3374 had a real Codex review), #3542, #3543, #3573, #2531 (codex on PATH) | +| Mandatory deep-verify of resolved+not-outdated HIGH | 🟒 corroborated | 2 | 2026-07-24 | #3542 #3467309496 (High "Grep failure skips link check", resolved + isOutdated:false) β€” xargs+guard genuinely present, legitimately fixed. #3585 3513924736 (High "Duplicate page IDs break get_page", resolved + isOutdated:false) β€” deep-verified get-page.ts: non-unique id now returns ambiguous()+candidates, search hits carry unique url; genuinely fixed *elsewhere* than the flagged line (hence not outdated). 2nd distinct PR | +| Bot calibration (fixed-vs-dismissed ratio) | 🟒 corroborated | 7 | 2026-07-24 | #3374 (bugbot mostly accepted); #3536 (5/5 valid); #3542 (3/3 valid β€” 2 fixed, 1 open); #3543 (1/1 valid; Jit 0); #3573 (4/4 valid; Jit 0); #2531 (r1 bugbot 0 findings; r2 bugbot 1/1 valid β€” caught ledger dup-rows 3499796857; Jit 0); #3585 (bugbot 13/13 valid across β‰₯5 rounds β€” r1 8 fixed incl. all 3 High; then empty-index + matchingSections fixed; r2 normalizeUrl dup 3644535220 + eager native import 3644667453 fixed; r3 Float32Array misaligned-buffer 3644788302 Low, valid, open; Jit 0; consistently high trust) | +| Codex second-opinion availability gate | 🟒 corroborated | 7 | 2026-07-24 | #3415, #3374 (CLI on PATH; #3374 had a real Codex review), #3542, #3543, #3573, #2531, #3585 (codex on PATH) | | Ledger self-integrity after `main` merge (no duplicate rows) | 🟑 seen twice | 2 | 2026-07-02 | #2531 r2 β€” bugbot 3499796857 caught the shared ledger gaining duplicate rows when `main` (carrying a #3573-era ledger) merged in and git kept both blocks. **2026-07-02**: merging `main` again produced a real conflict as #3542/#3573 edited the same rows β€” union-merged per capability. Recurring shared-file hazard; see worked examples + step-11 refinement | ## Worked examples library @@ -71,6 +71,35 @@ pushed, and bugbot's next re-scan came back **clean β€” no comments**. A real lo would have spawned another round; this settled. So the "not a loop" judgement is borne out by what happened next: assess β†’ fix β†’ re-scan reached a fixed point. +**Near-miss (NOT a loop) β€” #3585, 2026-07-24.** Bugbot Low #3638621205 +(`matchingSections` eager compute) landed in `search.ts` β€” a file the author had +just edited in Step 4 (added `hitForUrl`, which also calls `matchingSections`). +Superficially loop-shaped (finding in a freshly-edited file). But round-1 search.ts +findings were all *correctness* (dup-ids, suffix match, version) and were resolved; +this is a *new, independent perf* observation, not the same concern reopened or an +A↔B cycle. Normal iteration β†’ fresh cluster, not ping-pong. Also NOT churn: distinct +concern class (perf vs correctness), not repeated patching of one under-specified area. + +*Round-2 re-confirmation (2026-07-24):* fixing the 2 open findings triggered a +re-scan that surfaced 2 **new, independent** findings in the Step-4 code +(normalizeUrl dup across 4 files 3644535220; eager native import 3644667453) β€” +again the fixβ†’rescanβ†’new-findings-in-changed-files pattern, NOT a loop (no reopened +concern, no A↔B). The prior findings stayed resolved. Consistent with the near-miss +rule; still no true ping-pong on this PR. + +*Round-3 (2026-07-24):* fixing those 2 triggered a 3rd re-scan β†’ 1 new independent +finding (Float32Array misaligned-buffer in load-index.mjs 3644788302). **New +distinction worth naming: DRIP-FEED, not churn.** 3 consecutive rounds of findings +in the *same newly-added feature* (the ~7-file Step-4 hybrid code), but each on a +**different file** (index.ts β†’ hybrid.ts+3 β†’ load-index.mjs), each an independent +well-understood nit (import strategy / DRY / buffer alignment), each fix clean & +non-reopening. That is NOT churn (churn = repeated patches to the *same +under-specified area*, each exposing the next adjacent gap in that spot) and NOT +ping-pong (no reopened concern, no A↔B). It's Bugbot draining a backlog from a large +new diff, ~1-2 findings/scan. Right move is NOT redesign but a **proactive +self-review sweep** of the remaining new files to get ahead of the drip. See +step-11 refinement suggestion. + **Near-miss (NOT a loop) β€” #3542, 2026-07-01.** Round-1 bugbot High #3467309496 (ARG_MAX / `|| true` silent-green) was fixed (commit `21f079e1d`: xargs + zero-URL `exit 1` guard). Round-2 re-scan raised Medium #3498159511 ("empty scope diff --git a/build/docs-mcp-server/SPEC.md b/build/docs-mcp-server/SPEC.md new file mode 100644 index 0000000000..95c74d1032 --- /dev/null +++ b/build/docs-mcp-server/SPEC.md @@ -0,0 +1,278 @@ +# Docs MCP server β€” design spec + +**Status:** Draft (investigation, DOC-6809) +**Owner:** Docs +**Scope:** A read-only MCP server that lets AI coding agents query the Redis +documentation corpus as a tool, backed by the JSON/NDJSON feed we already +publish. + +--- + +## 1. Motivation + +We already publish AI-readable docs three ways: `llms.txt` (index), per-page +Markdown (`index.html.md`), and structured JSON/NDJSON with role-tagged +sections. These are all *passive files* β€” an agent must know they exist, fetch +them, and do its own retrieval. + +The gap is an *active, queryable* surface: a tool an agent (Cursor, Claude +Code, ChatGPT, VS Code) can call mid-task to get a **current, sourced** answer +instead of relying on stale training data. That is what this server provides. + +### Explicitly *not* this server + +- **`redis/mcp-redis`** is a *data-plane* server: it connects an agent to a + *running Redis instance* to read/write/query data. It needs a connection + string and can mutate data. +- **This server** is a *knowledge-plane* server: it connects an agent to the + *documentation*. It is read-only, needs no database, no credentials, and its + entire value is returning citations to our docs. + +They are orthogonal and should stay separate products/installs. Bundling doc +lookup into the data-plane server forces the reference-only audience to stand +up a data server and hand it credentials β€” friction that kills adoption for +the exact audience (coding agents) that benefits most. + +## 2. Goals / non-goals + +**Goals** +- Thin retrieval wrapper over the **existing** JSON feed β€” no new content + pipeline. +- Read-only, no secrets, no live DB connection. +- Every response carries a canonical `url` (citations by construction). +- Token-lean: search returns summaries + refs; agents drill down deliberately. +- Version-aware (our docs are versioned; mixing versions is a correctness bug). + +**Non-goals** +- No writes, no code execution, no live Redis access (that's `mcp-redis`). +- No new authoring format β€” we consume `sections[]` / `examples[]` as-is. +- WebMCP / in-browser tool registration β€” out of scope for now (different + layer, different audience; see DOC-6809 discussion). + +## 3. Data source + +No new pipeline. Reuse the current build output: + +``` +Hugo ──► per-page public/**/index.json ──► generate_ndjson.py ──► docs.ndjson +``` + +Document schema (already published on the *AI Agent Resources* page): + +- **Page**: `id`, `title`, `url`, `summary`, `page_type` (`content` | `index`), + `content_hash`, `sections[]`, `examples[]`, `children[]` +- **Section**: `id`, `title`, `role` (`overview` | `syntax` | `parameters` | + `returns` | `example` | …), `text` +- **Example**: `id`, `language`, `code`, `section_id` + +The server loads `docs.ndjson` (or an index built from it) at startup. Because +`content_hash` is deterministic (`sha256` over summary + section text + +example code), it doubles as a cache/freshness key. + +## 4. Tool surface + +Five tools. Names, inputs, and the field of the existing schema each is +projected from: + +### `search_docs` +Rank pages by relevance. Returns **refs only, no full text**. + +- **In:** `query` (string, required); optional `page_type`, `group`, + `version`, `limit` (default 10) +- **Out:** `[{ id, title, url, summary, matching_section_ids[] }]` +- **From:** NDJSON feed; `page_type` filter lets callers skip `index` pages. + +### `get_page` +Fetch one page, optionally filtered to specific section roles. + +- **In:** `id` **or** `url` (required); optional `roles[]` (e.g. + `["syntax","parameters"]`) +- **Out:** `{ id, title, url, summary, page_type, content_hash, sections[] }` + where `sections` is filtered to `roles[]` if given +- **From:** per-page `index.json`. `roles[]` filtering is only possible because + sections are role-tagged β€” big token savings (pull `parameters` without the + overview prose). + +### `get_section` +Return a single role-tagged chunk β€” the retrieval-native unit. + +- **In:** `page_id` (required), `section_id` (required) +- **Out:** `{ page_id, section_id, title, role, text, url }` +- **From:** `sections[]`. + +### `get_examples` +Return runnable code, filterable by language. **The highest-value tool for +coding agents** β€” their most common need is "the go-redis snippet for `XADD`", +not prose. + +- **In:** `query` **or** `command` (one required); optional `language` + (`python` | `go` | `java` | …) +- **Out:** `[{ id, code, language, url, section_id }]` +- **From:** `examples[]` (carries `language` + `section_id` already). + +### `get_command` +Convenience lookup for the highest-traffic page type. + +- **In:** `name` (e.g. `XADD`) +- **Out:** command page with `syntax`, `parameters`, `returns` sections + + `examples[]` + `url` +- **From:** command pages (specialised `get_page`; commands are first-class). + +## 5. Response conventions + +- **Always include `url`.** Agents cite; users click. +- **Search never returns full text.** Force the drill-down path + (`search_docs` β†’ `get_section` / `get_examples`) so context stays small. +- **Return `content_hash` on page/section responses.** Lets agents and our own + eval harness do `If-None-Match`-style freshness checks for free. +- **Truncate defensively.** Cap `text` length per section in responses; expose + a `truncated: true` flag rather than silently cutting. + +## 6. Transport & deployment + +Follow the existing repo pattern (`build/command_api_mapping/mcp-server/`): +TypeScript, `@modelcontextprotocol/sdk`, Zod input schemas. + +- **Local / stdio:** publish an npx-runnable package so developers can add it to + Cursor/Claude Code config. Zero infra. +- **Remote / hosted:** an HTTP+SSE endpoint on `redis.io` (e.g. + `https://redis.io/mcp`) built from the same handlers, so no install is + required. Advertise it on the *AI Agent Resources* page next to `llms.txt`. + +Both modes share one core: load feed β†’ build index β†’ handle tool calls. The +data is public, so the remote endpoint needs no auth (rate-limit only). + +### Search backend vs. transport β€” what needs a datastore + +Whether the server needs a Redis (or any) backend depends entirely on the +search implementation, **not** on the transport. The corpus is small +(`docs.ndjson` β‰ˆ 30 MB / β‰ˆ 5 MB gzipped, β‰ˆ 4,100 docs), which is what makes +the lexical path infra-free. + +| | Lexical (BM25) | Vector (semantic) | +|---|---|---| +| **stdio (client-side)** | βœ… self-contained in-memory index | ❌ impractical (would ship an index + embedding model per install) | +| **remote (hosted)** | βœ… in-process index, no datastore | βœ… needs a vector store (RediSearch / RedisVL) | + +- **Lexical (v1):** build a BM25 index in process memory from `docs.ndjson` at + startup β€” MiniSearch/Lunr (JS) or a simple BM25 (Python). No datastore, even + when hosted; horizontally-scaled instances each just load ~5 MB gzipped at + boot and build their own index. Role-tagged sections and exact command-name + tokens make lexical retrieval unusually strong on this corpus. +- **Vector (v2, hosted only):** needs embeddings computed offline at build + time, a vector index queried at runtime, and query-time embedding on each + request. That implies a real server-side backend β€” the natural fit is + **RediSearch / RedisVL**, which doubles as a Redis showcase. It cannot run + purely client-side, so it's a hosted-endpoint upgrade, not a stdio feature. + +Recommendation: ship v1 lexical in **both** modes with no backend; treat +vector-on-Redis as a later upgrade to the **hosted** endpoint only. + +## 7. Versioning + +- **Planned:** `search_docs` / `get_page` accept an optional `version` (default + `latest`), and responses echo the resolved version so an agent can't silently + blend versions β€” the single most common RAG-over-docs correctness bug. +- **v0 status: deferred, not implemented.** The prototype does **not** expose a + `version` param. An earlier v0 advertised a `latest` default it didn't + enforce (the filter was a no-op, and the live feed is single-version anyway), + which misleads agents. Rather than bake in a `page.url.includes("//")` + heuristic, the param was removed until a committed URL/version model exists + (Bugbot #3585 + Codex review). Add it back with the real filter when the feed + carries multiple version trees. + +## 8. Freshness + +- Rebuild the server's index whenever `docs.ndjson` is regenerated (same build + step). No separate content pipeline to keep in sync. +- `content_hash` per page enables incremental index updates and client caching. + +## 9. Security + +- Read-only. No write tools, no code execution, no connection string. +- Serves only already-public content. +- Remote endpoint: rate-limit, no auth, no PII. + +## 10. Open questions + +- **Search backend:** resolved for v1 β€” lexical (BM25 over NDJSON), no + datastore, runs in both stdio and hosted modes (see Β§6). **v2 vector: measured + and justified.** A measure-first experiment (`vector-eval/`, no Redis: + bge-small embeddings + numpy cosine, scored on the same 35-case eval) shows + vector/hybrid clearly beats lexical β†’ build the hosted RediSearch/RedisVL path. + Two rounds: + - *Page-level chunks (coarse):* hybrid best (overall recall@5 69%β†’86%, MRR + 0.53β†’0.66); vector alone only modest; **concept stayed weak** (hybrid @5 77%). + - *Section-level chunks (feed `sections[]`):* the big lift, and it fixed + concept. **Vector alone becomes the strongest by MRR** (overall 0.72 vs + hybrid 0.67; concept 0.64 vs 0.47 page-level) and best @1/@3; **hybrid RRF is + best by recall@5/@10** (concept @5 92% / @10 100%; overall @5 91%). Vanilla + equal-weight RRF now *dilutes* the top ranks because it fuses the strong + vector retriever with the weaker lexical one. + - **Decision (measured β€” fusion sweep):** build **section-level embeddings** + + **weighted RRF favouring vector ~2–3Γ—**. The sweep (`fusion_sweep.py`) shows + vector-weighted RRF recovers the top-1 precision equal-weight RRF lost *and* + keeps the top-k recall: overall MRR **.73** (vs .72 pure vector, .69 equal + RRF), command MRR **.80**, concept @5 **92%** / @10 **100%**. So the dilution + was specifically *equal* weighting. (n=35 is small, so treat the 2Γ— vs 3Γ— + choice as noise β€” just weight vector above lexical.) +- **Ranking quality (measured via the eval harness):** lexical BM25 with + Porter stemming, stopword removal, title/summary/slug field boosts, and + balanced page-type weighting (demote release-notes/REST-API/references only) + gets, on the 35-case eval (22 command + 13 concept), **command recall@5 73% / + concept 62% / overall 69%, MRR 0.53** β€” up from a **59% / 0.42** un-stemmed, + un-weighted lexical baseline on the command set. Residual misses are pure + semantic gaps ("remove a key" β†’ `flushdb` beats `del`) that only vector search + closes. Net: stemming+weighting **weakened but did not eliminate** the Β§6 + vector-search case β€” the eval now lets that call be made on numbers. +- **Command boost is command-overfit (found after adding concept cases).** With + 13 concept/how-to cases added, per-kind numbers diverge sharply: command + recall@5 86% / MRR 0.65 vs **concept recall@5 46% / MRR 0.29**. The + `/commands/*` Γ—1.5 boost is the cause β€” it ranks command pages above the + canonical concept page when both compete ("configure persistence" β†’ + `bgrewriteaof`; "set up replication" β†’ `cluster-replicate`; "keyspace + notifications" β†’ `expire`), and the blanket `/operate/` demotion drags down + legitimate concept pages (persistence, replication). Ablation (neutralise the + command boost, demote only REST-API/release-notes/references): concept @5 + 46%β†’62%, MRR 0.29β†’0.45; command @5 86%β†’73%. **Resolved: adopted the balanced + weighting** (no command boost, no blanket `/operate/` demotion). A modest + command boost (Γ—1.2) helped command none vs neutral, so lifting command + ranking should come from better lexical handling or vectors, not a bigger + thumb on the scale β€” the next lever, measured against this eval. +- **Section-role vocabulary (found via live MCP test):** the roles the spec + assumed (`syntax`, `parameters`, `returns`, `example`) do **not** all match + the feed. Command pages actually carry `content` / `parameters` / `example` + (singular) / `returns` β€” there is no `syntax` role, and it's `example` not + `examples`. So a `roles: ["examples","syntax"]` filter returns **zero + sections** against the real feed (verified on EXPIREAT). Before building + `get_examples` / `get_command` and documenting `roles`, enumerate the actual + role set across the corpus and align tool params/docs to it (and decide + whether the server should normalise synonyms like `examples`β†’`example`). +- **`get_command` coverage:** command pages *do* carry `parameters`/`returns`/ + `example` roles (confirmed: `get_page('expire')` β†’ roles + `[content, parameters, example, returns]`). Confirm this holds across all + command pages or add a fallback. +- **Package ownership:** does this live here in `docs`, or graduate to its own + repo like `mcp-redis`? +- **Overlap with `mcp-redis`:** does that server already do any doc lookup we + should pull into here and deprecate there? + +## 11. Phased plan + +1. **v0 (prototype):** stdio server, lexical `search_docs` + `get_page` over a + local `docs.ndjson`. Prove the loop in Claude Code. +2. **v1:** add `get_examples`, `get_section`, `get_command`; `version` support; + token-budget guards. +3. **v2:** hosted remote endpoint on `redis.io`; advertise on AI Agent + Resources. +4. **Ongoing:** wire an **AI-answer eval** β€” a fixed set of real questions run + through the server, scored for correctness β€” as a docs-quality regression + gate. (Extends the code-example verification mindset to answer quality.) + +## 12. Success signals + +- A coding agent, given only this server, answers common Redis how-to questions + correctly and with citations. +- Measurable reduction in version-mixing / stale-API answers versus the model's + own training data. +- Adoption: entries in Cursor/Claude Code MCP configs pointing at the endpoint. diff --git a/build/docs-mcp-server/node/.gitignore b/build/docs-mcp-server/node/.gitignore new file mode 100644 index 0000000000..144260fb0f --- /dev/null +++ b/build/docs-mcp-server/node/.gitignore @@ -0,0 +1,5 @@ +node_modules/ +dist/ +*.log +test/eval/docs.ndjson* +local_cache/ diff --git a/build/docs-mcp-server/node/README.md b/build/docs-mcp-server/node/README.md new file mode 100644 index 0000000000..c0ee45d4bb --- /dev/null +++ b/build/docs-mcp-server/node/README.md @@ -0,0 +1,151 @@ +# redis-docs-mcp (v0 prototype) + +A read-only MCP server that lets an AI coding agent query the Redis +documentation corpus as a tool, backed by the `docs.ndjson` feed we already +publish. See [`../SPEC.md`](../SPEC.md) for the full design. + +**This is a v0 prototype.** It ships two tools (`search_docs`, `get_page`) over +an in-memory lexical (BM25 + field-boost) index. No datastore, no credentials, +no live Redis connection. + +## Install & build + +```bash +cd build/docs-mcp-server/node +npm install +npm run build # compiles to dist/ +``` + +## Try it offline (no network) + +```bash +npm run smoke # runs against test/fixture.ndjson +``` + +Point it at the real feed (or any local `.ndjson` / `.ndjson.gz`): + +```bash +DOCS_NDJSON="https://redis.io/docs/latest/docs.ndjson" npm run smoke +``` + +## Run as an MCP server + +The server speaks MCP over stdio. Feed source is set via `DOCS_NDJSON` +(default: `https://redis.io/docs/latest/docs.ndjson`). + +Add to a Claude Code / Cursor MCP config after `npm run build`: + +```json +{ + "mcpServers": { + "redis-docs": { + "command": "node", + "args": ["/ABSOLUTE/PATH/build/docs-mcp-server/node/dist/index.js"], + "env": { "DOCS_NDJSON": "https://redis.io/docs/latest/docs.ndjson" } + } + } +} +``` + +## Hybrid mode (hosted) β€” DOC-6809 Step 4 prototype + +`search_docs` has two backends, chosen by whether `REDIS_URL` is set: + +- **Lexical-only (default):** in-memory BM25, no datastore β€” the stdio mode above. +- **Hybrid (hosted):** when `REDIS_URL` is set, fuses the same BM25 lexical ranker + with **vector KNN from Redis** using weighted reciprocal-rank fusion (vector + ~3Γ— lexical). The query is embedded in-process with `fastembed-js` + (bge-small-en-v1.5). This is the recipe the measure-first work settled on; + Redis is the vector backend only β€” lexical + fusion stay in-process, because + native `FT.HYBRID` can't express the weighted recipe (see `../redis-eval/`). + +Load the vector index once (rebuild whenever `docs.ndjson` changes), then run: + +```bash +# 1. build the vector index in Redis (embeds all section chunks with fastembed-js) +REDIS_URL=redis://localhost:6379 npm run load-index + +# 2. start the server in hybrid mode +REDIS_URL=redis://localhost:6379 DOCS_NDJSON= npm run start +``` + +`npm run eval:hybrid` scores the hybrid path through the real `search_docs` tool. +**Measured (local Redis 8.8, same 35-case eval), hybrid vs lexical-only:** + +| group | lexical MRR | **hybrid MRR** | hybrid @5 / @10 | +|---|---|---|---| +| overall | 0.53 | **0.70** | 91% / 94% | +| command | 0.57 | **0.78** | 91% / 91% | +| concept | 0.45 | **0.57** | 92% / 100% | + +(Hybrid sits within tie-break noise of the offline recipe's .73; the small gap is +FLOAT32 KNN + embedding tie-breaks, not a ranking difference β€” see `../redis-eval/`.) + +## Tools + +| Tool | Inputs | Returns | +|------|--------|---------| +| `search_docs` | `query` (req), `page_type`, `limit` | ranked `[{id, title, url, summary, page_type, score, matching_section_ids}]` β€” refs only, no full text | +| `get_page` | `id` **or** `url` (req), `roles[]` | one page with `content_hash` + `sections`, optionally filtered to the given section roles | + +Typical agent flow: `search_docs` β†’ pick a result β†’ `get_page` with `roles` +(e.g. `["parameters","returns"]`) to pull just what's needed. + +## Measured (real feed, ~2,530 pages) + +- Index build: ~0.3 s. Query latency: ~75–125 ms. This is why v0 needs no + datastore and why Rust/WASM would be premature (see SPEC Β§6). + +## Retrieval eval + +`npm run eval` scores retrieval quality: it runs the questions in +`test/eval/cases.json` (command-lookup questions phrased *without* the command +name) through `search_docs` and reports recall@k / MRR, plus a data-integrity +check that flags any expected url missing from the feed. The feed is read from +`DOCS_NDJSON` or a local cache at `test/eval/docs.ndjson.gz` (gitignored; +`curl -o test/eval/docs.ndjson.gz https://redis.io/docs/latest/docs.ndjson.gz`). + +Cases are tagged `command` (22) or `concept` (13, how-to / concept pages) so the +runner reports recall per kind β€” because command and concept queries behave very +differently. + +**Current results (shipped config: Porter stemming + field boosts + *balanced* +page-type weighting β€” demote REST-API/release-notes/references Γ—0.5, no command +boost, no blanket `/operate/` demotion):** + +| group | recall@1 | @3 | @5 | @10 | MRR | +|---|---|---|---|---|---| +| command (22) | 41% | 64% | 73% | 95% | 0.57 | +| concept (13) | 31% | 54% | 62% | 77% | 0.45 | +| overall (35) | 37% | 60% | 69% | 89% | 0.53 | + +(Un-stemmed, un-weighted lexical baseline on the command set was 59%@5 / 0.42.) + +**Why balanced:** an earlier command-optimised config (`/commands/*` Γ—1.5, +`/operate/` Γ—0.7) scored command @5 86% but only concept @5 46% β€” the boost +ranked command pages above the canonical concept page when both competed +("configure persistence" β†’ `bgrewriteaof`). We chose the balanced weighting: +concept @5 46%β†’62% for command @5 86%β†’73% (SPEC Β§10). The residual misses are +pure semantic gaps that motivate vector search (SPEC Β§6/Β§10) β€” the right lever +for lifting both, rather than a bigger thumb on the scale. + +**Stemmer:** Porter (`src/stem.ts`) is the default. A Paice/Lancaster stemmer +(`src/stem-paice.ts`) is available behind `STEMMER=paice` for comparison. Bake-off +on this eval: Porter wins or ties β€” Porter overall @5 69% / MRR 0.53 vs Paice +63% / 0.52; Paice edges @1 (its aggressive conflation occasionally nabs the exact +top hit) but is worse at @3–@10 and on concept, and over-stems (e.g. +`organization β†’ org`). Kept Porter. (Small margins at n=35, and the stemmer only +really matters for the pure-lexical deployment β€” in the hybrid recipe lexical is +the minority signal.) + +## Known limitations (v0) + +- **Ranking is lexical (BM25 + Porter stemming + field/page-type weighting).** + Now recall@5 86% on command lookups (see eval above), but still lexical: it + can't bridge pure semantic gaps (e.g. "remove a key" β†’ `flushdb` over `del`), + and concept/how-to queries are unmeasured. Closing the remainder is the case + for vector search (SPEC Β§6/Β§10). +- **No version filtering.** The `version` param was removed until a committed + URL/version model exists (the feed is single-version today); see SPEC Β§7. +- Only `search_docs` + `get_page`. `get_examples`, `get_section`, + `get_command` are v1 (SPEC Β§4/Β§11). diff --git a/build/docs-mcp-server/node/package.json b/build/docs-mcp-server/node/package.json new file mode 100644 index 0000000000..e740cf5730 --- /dev/null +++ b/build/docs-mcp-server/node/package.json @@ -0,0 +1,36 @@ +{ + "name": "redis-docs-mcp", + "version": "0.0.1", + "description": "Read-only MCP server that queries the Redis documentation corpus (docs.ndjson) as a tool. v0 prototype: lexical search_docs + get_page.", + "type": "module", + "main": "dist/index.js", + "bin": { + "redis-docs-mcp": "dist/index.js" + }, + "scripts": { + "build": "tsc", + "start": "tsx src/index.ts", + "dev": "tsx watch src/index.ts", + "smoke": "tsx src/smoke.ts", + "eval": "npm run build && node test/eval/run.mjs", + "eval:hybrid": "npm run build && node test/eval/run-hybrid.mjs", + "load-index": "npm run build && node scripts/load-index.mjs" + }, + "keywords": [ + "redis", + "mcp", + "docs" + ], + "license": "MIT", + "dependencies": { + "@modelcontextprotocol/sdk": "^1.0.0", + "fastembed": "^2.1.0", + "redis": "^6.1.0", + "zod": "^3.22.0" + }, + "devDependencies": { + "@types/node": "^20.0.0", + "tsx": "^4.0.0", + "typescript": "^5.0.0" + } +} diff --git a/build/docs-mcp-server/node/scripts/load-index.mjs b/build/docs-mcp-server/node/scripts/load-index.mjs new file mode 100644 index 0000000000..459f46ae11 --- /dev/null +++ b/build/docs-mcp-server/node/scripts/load-index.mjs @@ -0,0 +1,108 @@ +// Build-time index loader for hybrid mode (DOC-6809 Step 4). Chunks the feed +// (section-level), embeds each chunk with fastembed-js, and loads the vectors +// into Redis as the docs_vec FLAT/COSINE index. Run once whenever docs.ndjson +// is regenerated (SPEC freshness model); the server only queries at runtime. +// +// REDIS_URL=redis://localhost:6379 npm run load-index +// REDIS_URL=... node scripts/load-index.mjs --vectors ../redis-eval/vecdump +// +// --vectors seeds precomputed vectors (meta.json/owners.json/vectors.f32) +// instead of embedding in Node. Step 2 proved fastembed-js == those Python +// vectors (cosine 1.0), so seeding is equivalent β€” used to avoid a ~1h local +// re-embed while iterating. Production runs without the flag. +import { readFile } from "node:fs/promises"; +import { fileURLToPath, pathToFileURL } from "node:url"; +import { resolve } from "node:path"; +import { loadFeed } from "../dist/feed.js"; +import { buildChunks } from "../dist/chunk.js"; +import { embedPassages } from "../dist/embed.js"; +import { VectorStore } from "../dist/vector-store.js"; +import { EMBED_DIM } from "../dist/constants.js"; + +const REDIS_URL = process.env.REDIS_URL ?? "redis://localhost:6379"; +const FEED = + process.env.DOCS_NDJSON ?? + fileURLToPath(new URL("../test/eval/docs.ndjson.gz", import.meta.url)); + +function arg(name) { + const i = process.argv.indexOf(name); + return i >= 0 ? process.argv[i + 1] : undefined; +} + +async function fromSeed(dir) { + // Resolve against cwd (intuitive for a CLI arg), not the script location. + const base = pathToFileURL(resolve(process.cwd(), dir) + "/"); + const meta = JSON.parse(await readFile(new URL("meta.json", base), "utf8")); + const owners = JSON.parse(await readFile(new URL("owners.json", base), "utf8")); + const buf = await readFile(new URL("vectors.f32", base)); + // Validate the seed before trusting it: mismatched/truncated vectors would + // otherwise "load" but be silently rejected or misindexed by Redis while the + // loader reports success. + const { n, dim } = meta; + if (!Number.isInteger(n) || !Number.isInteger(dim) || n <= 0 || dim <= 0) { + throw new Error(`seed meta.json invalid: n=${n} dim=${dim}`); + } + if (dim !== EMBED_DIM) { + throw new Error(`seed dim ${dim} != expected EMBED_DIM ${EMBED_DIM}`); + } + if (!Array.isArray(owners) || owners.length !== n) { + throw new Error(`seed owners length ${owners?.length} != n ${n}`); + } + if (buf.byteLength !== n * dim * 4) { + throw new Error(`seed vectors.f32 is ${buf.byteLength} bytes, expected ${n * dim * 4} (n*dim*4)`); + } + // Copy into a fresh 0-offset ArrayBuffer before the Float32Array view: a + // pooled Node Buffer's byteOffset isn't guaranteed 4-byte aligned, and an + // unaligned offset makes `new Float32Array(buffer, offset)` throw RangeError. + const ab = buf.buffer.slice(buf.byteOffset, buf.byteOffset + buf.byteLength); + const floats = new Float32Array(ab); + const chunks = []; + for (let i = 0; i < n; i++) { + chunks.push({ owner: owners[i], vec: floats.subarray(i * dim, (i + 1) * dim) }); + } + console.error(`[load-index] seeded ${n} vectors (dim ${dim}) from ${dir}`); + return chunks; +} + +async function fromEmbed() { + const pages = await loadFeed(FEED); + const chunks = buildChunks(pages); + // Bail before loading the embedder: no point spinning up native ONNX to embed + // nothing (main() guards too, but this avoids the wasted model init on empty). + if (chunks.length === 0) return chunks; + console.error(`[load-index] ${pages.length} pages -> ${chunks.length} chunks; embedding (fastembed-js) ...`); + const vecs = await embedPassages(chunks.map((c) => c.text)); + return chunks.map((c, i) => ({ owner: c.owner, vec: vecs[i] })); +} + +async function main() { + const seedDir = arg("--vectors"); + const chunks = seedDir ? await fromSeed(seedDir) : await fromEmbed(); + // Guard before chunks[0]: an empty/invalid feed or seed would otherwise throw + // an opaque "Cannot read properties of undefined" on the ensureIndex line. + if (chunks.length === 0) { + throw new Error( + `No chunks to load from ${seedDir ?? FEED} β€” aborting (empty or invalid source).`, + ); + } + + const store = new VectorStore(REDIS_URL); + await store.connect(); + // TODO (production hardening, deferred β€” Codex review): this drops the live + // index before loading, so a hosted server reloading under traffic would see + // an absent/partial corpus mid-load (and a partial index if loading fails). + // For zero-downtime reloads, build under a versioned index+prefix, verify it, + // then atomically switch an alias. Fine for the current offline, + // single-instance prototype. + await store.dropIndex(); + await store.ensureIndex(chunks[0].vec.length); + const t0 = Date.now(); + await store.loadChunks(chunks); + console.error(`[load-index] loaded ${chunks.length} chunks into ${REDIS_URL} in ${((Date.now() - t0) / 1000).toFixed(1)}s`); + await store.close(); +} + +main().catch((e) => { + console.error("[load-index] fatal:", e); + process.exit(1); +}); diff --git a/build/docs-mcp-server/node/src/chunk.ts b/build/docs-mcp-server/node/src/chunk.ts new file mode 100644 index 0000000000..1c95be9ed0 --- /dev/null +++ b/build/docs-mcp-server/node/src/chunk.ts @@ -0,0 +1,45 @@ +// Section-level chunking for the vector index. Faithful port of the Python +// vector-eval build_chunks() (mode "section"), so Node-embedded corpus vectors +// correspond 1:1 with the offline experiment's chunks: +// - one "anchor" chunk per page: ". <summary>" +// - up to MAX_SECTIONS section chunks: "<title> β€” <section title>. <body>" +// - body truncated to LEAD_CHARS; sections with <20 chars of body skipped +// - owner of every chunk is the page's normalized url +// Section-level chunking is what fixed the concept-query gap (DOC-6809 SPEC Β§10). +import type { Page } from "./types.js"; +import { normalizeUrl } from "./url.js"; + +const LEAD_CHARS = 1200; +const MAX_SECTIONS = 8; + +export interface Chunk { + text: string; + owner: string; // normalized page url +} + +export function buildChunks(pages: Page[]): Chunk[] { + const chunks: Chunk[] = []; + for (const p of pages) { + const owner = normalizeUrl(p.url); + const title = p.title ?? ""; + const summary = p.summary ?? ""; + const sections = p.sections ?? []; + + const anchor = [title, summary].filter(Boolean).join(". ").trim(); + if (anchor) chunks.push({ text: anchor, owner }); + + let n = 0; + for (const s of sections) { + const body = (s.text ?? "").trim(); + if (body.length < 20) continue; + const st = (s.title ?? "").trim(); + chunks.push({ + text: `${title} β€” ${st}. ${body.slice(0, LEAD_CHARS)}`.trim(), + owner, + }); + if (++n >= MAX_SECTIONS) break; + } + if (!anchor && n === 0) chunks.push({ text: title || owner, owner }); + } + return chunks; +} diff --git a/build/docs-mcp-server/node/src/constants.ts b/build/docs-mcp-server/node/src/constants.ts new file mode 100644 index 0000000000..0aa85d977b --- /dev/null +++ b/build/docs-mcp-server/node/src/constants.ts @@ -0,0 +1,6 @@ +// Dependency-free shared constants. Kept separate from embed.ts so modules that +// only need the dimension (vector-store, the loader's seed path) don't +// transitively import fastembed and load the native ONNX runtime. + +/** bge-small-en-v1.5 embedding dimension. */ +export const EMBED_DIM = 384; diff --git a/build/docs-mcp-server/node/src/embed.ts b/build/docs-mcp-server/node/src/embed.ts new file mode 100644 index 0000000000..6d280e0ae3 --- /dev/null +++ b/build/docs-mcp-server/node/src/embed.ts @@ -0,0 +1,59 @@ +// bge-small-en-v1.5 embedding via fastembed-js (ONNX, in-process). Proven in +// the DOC-6809 Step 2 experiment to reproduce Python fastembed vectors to +// cosine 1.0, so query-time embedding here is interchangeable with the offline +// corpus vectors. IMPORTANT: use plain embed() and prepend the BGE query prefix +// manually for queries β€” do NOT use fastembed's queryEmbed/passageEmbed, whose +// built-in prefix wording differs and silently breaks parity (Step 2 finding). +import { FlagEmbedding, EmbeddingModel } from "fastembed"; +import { EMBED_DIM } from "./constants.js"; + +const QUERY_PREFIX = "Represent this sentence for searching relevant passages: "; + +let modelPromise: Promise<FlagEmbedding> | null = null; + +function model(): Promise<FlagEmbedding> { + // Memoize the loaded model, but DON'T cache a rejected init: a transient + // failure (first-run download timeout, fs perms) must not poison the singleton + // and break hybrid mode until restart. Clear the cache on failure so the next + // call retries. + if (!modelPromise) { + modelPromise = FlagEmbedding.init({ model: EmbeddingModel.BGESmallENV15 }).catch( + (e) => { + modelPromise = null; + throw e; + }, + ); + } + return modelPromise; +} + +function l2normalize(v: number[]): Float32Array { + let n = 0; + for (const x of v) n += x * x; + n = Math.sqrt(n) + 1e-12; + const out = new Float32Array(v.length); + for (let i = 0; i < v.length; i++) out[i] = v[i] / n; + return out; +} + +async function embedAll(texts: string[]): Promise<Float32Array[]> { + const m = await model(); + const out: Float32Array[] = []; + for await (const batch of m.embed(texts, 64)) { + for (const v of batch) out.push(l2normalize(Array.from(v))); + } + return out; +} + +/** Embed a search query (applies the BGE query prefix). Returns a unit vector. */ +export async function embedQuery(text: string): Promise<Float32Array> { + const [v] = await embedAll([QUERY_PREFIX + text]); + return v; +} + +/** Embed corpus passages (no prefix). Returns unit vectors, input order. */ +export async function embedPassages(texts: string[]): Promise<Float32Array[]> { + return embedAll(texts); +} + +export { EMBED_DIM }; diff --git a/build/docs-mcp-server/node/src/feed.ts b/build/docs-mcp-server/node/src/feed.ts new file mode 100644 index 0000000000..7eb01a0a78 --- /dev/null +++ b/build/docs-mcp-server/node/src/feed.ts @@ -0,0 +1,44 @@ +import { readFile } from "node:fs/promises"; +import { gunzipSync } from "node:zlib"; +import type { Page } from "./types.js"; + +/** Load raw feed bytes from a local path or http(s) URL, gunzipping if .gz. */ +async function loadFeedRaw(source: string): Promise<string> { + let buf: Buffer; + if (/^https?:\/\//i.test(source)) { + const res = await fetch(source); + if (!res.ok) { + throw new Error(`Failed to fetch feed ${source}: ${res.status} ${res.statusText}`); + } + buf = Buffer.from(await res.arrayBuffer()); + } else { + buf = await readFile(source); + } + if (source.toLowerCase().endsWith(".gz")) { + buf = gunzipSync(buf); + } + return buf.toString("utf-8"); +} + +/** Parse NDJSON text into pages, skipping blank/invalid lines and non-doc objects. */ +export function parseNdjson(text: string): Page[] { + const pages: Page[] = []; + for (const line of text.split("\n")) { + const trimmed = line.trim(); + if (!trimmed) continue; + try { + const obj = JSON.parse(trimmed); + // Match the same guard generate_ndjson.py uses: must have id + title + url. + if (obj && typeof obj.id === "string" && typeof obj.url === "string" && typeof obj.title === "string") { + pages.push(obj as Page); + } + } catch { + // Not our format β€” skip. + } + } + return pages; +} + +export async function loadFeed(source: string): Promise<Page[]> { + return parseNdjson(await loadFeedRaw(source)); +} diff --git a/build/docs-mcp-server/node/src/hybrid.ts b/build/docs-mcp-server/node/src/hybrid.ts new file mode 100644 index 0000000000..325e3d9807 --- /dev/null +++ b/build/docs-mcp-server/node/src/hybrid.ts @@ -0,0 +1,87 @@ +// Hybrid searcher (hosted mode). Fuses the existing app-side BM25 lexical ranker +// (DocsIndex) with Redis vector KNN using weighted reciprocal-rank fusion, +// favouring the vector signal ~3x. This is the recipe the DOC-6809 measure-first +// work settled on (SPEC Β§6/Β§10): section-level bge-small embeddings + weighted +// RRF. Query embedding happens in-process via fastembed-js (embed.ts). Lexical +// and fusion stay here in the app because native FT.HYBRID can't express the +// weighted recipe and its raw BM25 is much weaker than this ranker (Step 3). +import type { DocsIndex, SearchHit, SearchOptions } from "./search.js"; +import type { VectorStore } from "./vector-store.js"; +import { embedQuery } from "./embed.js"; +import { normalizeUrl } from "./url.js"; + +const RRF_K = 60; +const DEFAULT_VECTOR_WEIGHT = 3; +const LEXICAL_POOL = 50; // lexical candidates fused +const VECTOR_POOL = 200; // vector chunks fetched (deduped to <=50 pages) + +/** Weighted reciprocal-rank fusion. Returns url -> fused score. */ +function weightedRrf( + lists: Array<{ urls: string[]; weight: number }>, +): Map<string, number> { + const scores = new Map<string, number>(); + for (const { urls, weight } of lists) { + urls.forEach((url, rank) => { + scores.set(url, (scores.get(url) ?? 0) + weight / (RRF_K + rank + 1)); + }); + } + return scores; +} + +export class HybridSearcher { + constructor( + private readonly index: DocsIndex, + private readonly store: VectorStore, + private readonly vectorWeight = DEFAULT_VECTOR_WEIGHT, + ) {} + + async search(query: string, opts: SearchOptions = {}): Promise<SearchHit[]> { + const limit = opts.limit ?? 10; + + // Lexical side (already page-type filtered) is computed first and always + // usable. The vector side (query embedding + Redis KNN) can fail transiently + // β€” if it does, degrade to lexical-only rather than failing the whole tool + // call, since hybrid is meant to be an enhancement over a working lexical base. + const lexHits = this.index.search(query, { limit: LEXICAL_POOL, pageType: opts.pageType }); + let vecUrls: string[]; + try { + const qvec = await embedQuery(query); + vecUrls = await this.store.knn(qvec, VECTOR_POOL, LEXICAL_POOL); + } catch (e) { + console.error( + `[redis-docs-mcp] vector search failed, returning lexical-only: ${ + e instanceof Error ? e.message : String(e) + }`, + ); + return lexHits.slice(0, limit); + } + + const lexByUrl = new Map(lexHits.map((h) => [normalizeUrl(h.url), h])); + const lexUrls = lexHits.map((h) => normalizeUrl(h.url)); + + const fused = weightedRrf([ + { urls: vecUrls, weight: this.vectorWeight }, + { urls: lexUrls, weight: 1 }, + ]); + + const ranked = [...fused.entries()].sort((a, b) => b[1] - a[1]); + + const out: SearchHit[] = []; + for (const [url, score] of ranked) { + // Page-type filter: lexical hits are pre-filtered; a vector-only url must + // be checked against the page's type here. + const lex = lexByUrl.get(url); + let hit: SearchHit | undefined; + if (lex) { + hit = { ...lex, score: Number(score.toFixed(4)) }; + } else { + hit = this.index.hitForUrl(url, query, score); + } + if (!hit) continue; + if (opts.pageType && hit.page_type !== opts.pageType) continue; + out.push(hit); + if (out.length >= limit) break; + } + return out; + } +} diff --git a/build/docs-mcp-server/node/src/index.ts b/build/docs-mcp-server/node/src/index.ts new file mode 100644 index 0000000000..f7be55b925 --- /dev/null +++ b/build/docs-mcp-server/node/src/index.ts @@ -0,0 +1,164 @@ +#!/usr/bin/env node +import { Server } from "@modelcontextprotocol/sdk/server/index.js"; +import { + ListToolsRequestSchema, + CallToolRequestSchema, +} from "@modelcontextprotocol/sdk/types.js"; +import { StdioServerTransport } from "@modelcontextprotocol/sdk/server/stdio.js"; + +import { loadFeed } from "./feed.js"; +import { DocsIndex, type Searcher } from "./search.js"; +import { searchDocs, SearchDocsInput } from "./tools/search-docs.js"; +import { getPage, GetPageInput } from "./tools/get-page.js"; +import { toolResult, fail } from "./response.js"; + +// Feed source: local path or http(s) URL, gzip-aware. Defaults to production. +const FEED_SOURCE = + process.env.DOCS_NDJSON ?? "https://redis.io/docs/latest/docs.ndjson"; + +// Hosted hybrid mode activates only when a Redis backend is configured. Without +// REDIS_URL the server stays lexical-only, which needs no datastore and runs +// client-side over stdio (SPEC Β§6). The vector index must be pre-loaded by +// `npm run load-index` against the same REDIS_URL. +const REDIS_URL = process.env.REDIS_URL; + +const TOOLS = [ + { + name: "search_docs", + description: + "Search the Redis documentation and return the most relevant pages as references (title, url, summary, matching section ids). Each hit includes a unique `url` β€” pass that `url` to get_page (a hit's `id` is NOT unique and may be ambiguous). Returns no full text β€” follow up with get_page to read a result.", + inputSchema: { + type: "object" as const, + properties: { + query: { type: "string", description: "Search query" }, + page_type: { + type: "string", + enum: ["content", "index"], + description: "Restrict to prose ('content') or navigation ('index') pages", + }, + limit: { + type: "number", + description: "Max results (default 10, max 50)", + }, + }, + required: ["query"], + }, + }, + { + name: "get_page", + description: + "Fetch a single documentation page. Prefer the unique `url` from a search_docs hit. `id` also works but is NOT unique, so an ambiguous id returns an error listing candidate urls (likewise an ambiguous partial url). Optionally filter to sections with specific roles (e.g. ['syntax','parameters']) to save tokens. Includes content_hash for caching.", + inputSchema: { + type: "object" as const, + properties: { + id: { + type: "string", + description: "Page id (last URL slug); not unique β€” prefer url", + }, + url: { + type: "string", + description: "Full page URL (preferred β€” unique). A partial/suffix URL also works when it matches exactly one page.", + }, + roles: { + type: "array", + items: { type: "string" }, + description: "Only return sections with these roles", + }, + }, + }, + }, +]; + +async function main() { + const pages = await loadFeed(FEED_SOURCE); + // Refuse to start on an empty index (empty file, bad path, or no valid NDJSON + // lines) rather than advertising readiness and serving only empty results. + if (pages.length === 0) { + throw new Error( + `No documents loaded from ${FEED_SOURCE} β€” refusing to start with an empty index.`, + ); + } + const index = new DocsIndex(pages); + // Log to stderr β€” stdout is reserved for the MCP protocol. + console.error(`[redis-docs-mcp] indexed ${index.size} pages from ${FEED_SOURCE}`); + + // search_docs backend: hybrid when Redis is configured, else lexical-only. + // get_page always uses the lexical index (feed lookup, no ranking). + let searcher: Searcher = index; + // Cleanup for the hybrid backend (no-op in lexical-only mode). Called on + // shutdown so an open Redis socket can't keep the stdio process alive. + let closeBackend: () => Promise<void> = async () => {}; + if (REDIS_URL) { + // Load the hybrid path lazily: it pulls in fastembed (native onnxruntime) + // and the redis client, which the default lexical-only stdio mode never + // needs and which would otherwise crash startup on platforms lacking the + // native ONNX binary. Keep the no-REDIS_URL path dependency-free. + const { VectorStore } = await import("./vector-store.js"); + const { HybridSearcher } = await import("./hybrid.js"); + const store = new VectorStore(REDIS_URL); + try { + await store.connect(); + // TODO (deferred β€” Codex review): ensureIndex creates a missing index but + // accepts any existing one without checking it. Verify via FT.INFO (dim, + // metric, prefix, non-zero doc count) so a misconfig can't advertise + // hybrid over an empty/mismatched index β€” or explicitly fall back to + // lexical. Operator runs load-index before serving today, so deferred. + await store.ensureIndex(); + } catch (e) { + // Don't leak the socket if startup fails partway through. + await store.close().catch(() => {}); + throw e; + } + closeBackend = () => store.close(); + searcher = new HybridSearcher(index, store); + console.error(`[redis-docs-mcp] hybrid mode: vector KNN via ${REDIS_URL}`); + } else { + console.error("[redis-docs-mcp] lexical-only mode (no REDIS_URL)"); + } + + const server = new Server( + { name: "redis-docs-mcp", version: "0.0.1" }, + { capabilities: { tools: {} } }, + ); + + server.setRequestHandler(ListToolsRequestSchema, async () => ({ tools: TOOLS })); + + server.setRequestHandler(CallToolRequestSchema, async (req) => { + const { name, arguments: args } = req.params; + try { + switch (name) { + case "search_docs": + return toolResult(await searchDocs(searcher, SearchDocsInput.parse(args ?? {}))); + case "get_page": + return toolResult(getPage(index, GetPageInput.parse(args ?? {}))); + default: + return fail(`Unknown tool: ${name}`); + } + } catch (e) { + return fail(e instanceof Error ? e.message : String(e)); + } + }); + + // Close the Redis backend on shutdown so a live socket can't keep the process + // alive after the client disconnects (stdio EOF fires the connection close). + let shuttingDown = false; + const shutdown = async (reason: string) => { + if (shuttingDown) return; + shuttingDown = true; + console.error(`[redis-docs-mcp] shutting down (${reason})`); + await closeBackend().catch(() => {}); + process.exit(0); + }; + server.onclose = () => void shutdown("connection closed"); + process.on("SIGINT", () => void shutdown("SIGINT")); + process.on("SIGTERM", () => void shutdown("SIGTERM")); + + const transport = new StdioServerTransport(); + await server.connect(transport); + console.error("[redis-docs-mcp] ready on stdio"); +} + +main().catch((e) => { + console.error("[redis-docs-mcp] fatal:", e); + process.exit(1); +}); diff --git a/build/docs-mcp-server/node/src/response.ts b/build/docs-mcp-server/node/src/response.ts new file mode 100644 index 0000000000..6f1db590f1 --- /dev/null +++ b/build/docs-mcp-server/node/src/response.ts @@ -0,0 +1,24 @@ +// MCP tool-response helpers, factored out of index.ts so tests can import them +// without triggering index.ts's main() (which starts the stdio server). + +/** + * Serialise a tool result. If the tool returned an object carrying an `error` + * field (e.g. get_page couldn't resolve the page, or an id/url was ambiguous), + * mark the MCP response as an error so clients don't treat a failed lookup as + * success. + */ +export function toolResult(data: unknown) { + const isError = typeof data === "object" && data !== null && "error" in data; + return { + content: [{ type: "text" as const, text: JSON.stringify(data, null, 2) }], + ...(isError ? { isError: true } : {}), + }; +} + +/** For protocol-level failures (unknown tool, input parse errors). */ +export function fail(message: string) { + return { + content: [{ type: "text" as const, text: JSON.stringify({ error: message }) }], + isError: true, + }; +} diff --git a/build/docs-mcp-server/node/src/search.ts b/build/docs-mcp-server/node/src/search.ts new file mode 100644 index 0000000000..2b1bb580e3 --- /dev/null +++ b/build/docs-mcp-server/node/src/search.ts @@ -0,0 +1,260 @@ +import type { Page } from "./types.js"; +import { stem as stemPorter } from "./stem.js"; +import { stem as stemPaice } from "./stem-paice.js"; +import { normalizeUrl } from "./url.js"; + +// Stemmer is switchable for the eval bake-off (STEMMER=paice|porter). Porter is +// the default/shipped analyzer. +const stem = (process.env.STEMMER ?? "porter").toLowerCase() === "paice" ? stemPaice : stemPorter; + +// Self-contained BM25 lexical index. No external search dependency: at +// ~4,100 docs the whole index builds in-memory in well under a second, which +// is why v0 needs no datastore (see SPEC.md Β§6). + +const K1 = 1.5; +const B = 0.75; + +// Page-type weighting applied to the final score. Demote clearly-secondary +// reference material (release-notes / REST-API / other references) that was +// observed outranking primary docs. Multipliers, not filters. +// +// Deliberately balanced, NOT command-optimised: an earlier config boosted +// /commands/* (x1.5) and demoted all of /operate/ (x0.7), which lifted command +// queries but ranked command pages above the canonical concept page when both +// competed (persistence -> bgrewriteaof) and buried legitimate /operate/ +// concept pages. The eval showed that cost concept recall@5 ~16pts for ~13pts +// of command gain, so we chose the balanced weighting (SPEC Β§10). Lifting +// command ranking further should come from better signal (vectors), not a +// bigger thumb on the scale. +function pageWeight(url: string): number { + const u = url.toLowerCase(); + if (u.includes("/release-notes") || u.includes("/rest-api/") || u.includes("/references/")) { + return 0.5; + } + return 1; +} + +// Field boosts (added on top of the body BM25 score, weighted by term idf). +// A query term appearing in the title/slug/summary is a strong signal that the +// page is *about* that term β€” this lifts canonical command pages (whose summary +// is a one-line definition) above long pages that merely mention the terms. +const W_SUMMARY = 6; // canonical one-line definition β€” strongest signal +const W_TITLE = 4; +const W_SLUG = 2; // lowest: slug word-collisions (set-up-redis, key-specs) mislead + +// Common words carry no topical signal and their title/slug/summary collisions +// distort ranking (e.g. "set"/"key"). Dropped from the query only. +const STOPWORDS = new Set([ + "a", "an", "the", "to", "of", "in", "on", "for", "with", "and", "or", "is", + "are", "how", "do", "i", "my", "me", "can", "what", "when", "which", "you", + "your", "it", "this", "that", "from", "by", "as", "at", "be", "using", "use", +]); + +/** Split on non-alphanumerics so "JSON.SET" -> ["json","set"], "XADD" -> ["xadd"]. */ +function tokenize(text: string): string[] { + return text.toLowerCase().match(/[a-z0-9]+/g) ?? []; +} + +/** Tokenize + stem. Used for everything indexed and for query terms, so word + * forms conflate. Stopwords are filtered on RAW tokens before this (see search). */ +function analyze(text: string): string[] { + return tokenize(text).map(stem); +} + +/** Everything worth matching against for a page: slug, title, summary, section text. */ +function searchableText(p: Page): string { + const parts: string[] = [p.id ?? "", p.title ?? "", p.summary ?? ""]; + for (const s of p.sections ?? []) { + parts.push(s.title ?? "", s.text ?? ""); + } + return parts.join(" "); +} + +/** Section ids whose title/text contain any query term (capped for token budget). */ +function matchingSections(p: Page, qterms: Set<string>): string[] { + const out: string[] = []; + for (const s of p.sections ?? []) { + const toks = new Set(analyze(`${s.title ?? ""} ${s.text ?? ""}`)); + for (const t of qterms) { + if (toks.has(t)) { + out.push(s.id); + break; + } + } + if (out.length >= 5) break; + } + return out; +} + +export interface SearchOptions { + limit?: number; + pageType?: string; +} + +export interface SearchHit { + id: string; + title: string; + url: string; + summary: string; + page_type: string; + score: number; + matching_section_ids: string[]; +} + +/** A ranking backend for search_docs. DocsIndex (lexical) is sync; the hosted + * HybridSearcher is async β€” the tool awaits either. */ +export interface Searcher { + search(query: string, opts?: SearchOptions): SearchHit[] | Promise<SearchHit[]>; +} + +export class DocsIndex { + readonly pages: Page[]; + // The feed's `id` is the last URL path segment (e.g. "config", "acl") and is + // NOT unique β€” ~200 ids map to several pages. So id -> list, and callers must + // disambiguate by url. `url` IS unique, so byUrl stays 1:1. + private byId = new Map<string, Page[]>(); + private byUrl = new Map<string, Page>(); + private docs: Array<{ + page: Page; + tf: Map<string, number>; + len: number; + titleTok: Set<string>; + slugTok: Set<string>; + summaryTok: Set<string>; + }> = []; + private df = new Map<string, number>(); + private avgdl = 0; + private N = 0; + + constructor(pages: Page[]) { + this.pages = pages; + let totalLen = 0; + for (const p of pages) { + const bucket = this.byId.get(p.id); + if (bucket) bucket.push(p); + else this.byId.set(p.id, [p]); + if (p.url) this.byUrl.set(normalizeUrl(p.url), p); + + const tokens = analyze(searchableText(p)); + if (tokens.length === 0) continue; + const tf = new Map<string, number>(); + for (const t of tokens) tf.set(t, (tf.get(t) ?? 0) + 1); + for (const t of tf.keys()) this.df.set(t, (this.df.get(t) ?? 0) + 1); + this.docs.push({ + page: p, + tf, + len: tokens.length, + titleTok: new Set(analyze(p.title ?? "")), + slugTok: new Set(analyze(p.id ?? "")), + summaryTok: new Set(analyze(p.summary ?? "")), + }); + totalLen += tokens.length; + } + this.N = this.docs.length; + this.avgdl = this.N ? totalLen / this.N : 0; + } + + get size(): number { + return this.pages.length; + } + + /** All pages sharing this id (usually one, but the feed's id is not unique). */ + getPagesById(id: string): Page[] { + return this.byId.get(id) ?? []; + } + + getByUrl(url: string): Page | undefined { + return this.byUrl.get(normalizeUrl(url)); + } + + /** + * Fallback lookup when a caller passes a path or partial URL. Returns EVERY + * page whose url ends with the given suffix **at a path-segment boundary**, + * so "get" matches ".../commands/get" but NOT ".../config-get" or + * ".../arget". A suffix can still match several pages (e.g. "/install/" or a + * bare last segment shared by many pages), so the caller must disambiguate. + */ + matchByUrlSuffix(url: string): Page[] { + const target = normalizeUrl(url).replace(/^https?:\/\/[^/]+/, ""); + if (!target) return []; + // Anchor the leading edge to a "/" so we match whole path segments. The + // trailing edge is already anchored: normalizeUrl strips the trailing slash + // and we compare against the end of the string. + const anchored = target.startsWith("/") ? target : `/${target}`; + return this.pages.filter((p) => normalizeUrl(p.url).endsWith(anchored)); + } + + /** + * Build a SearchHit for a page by url, for hybrid fusion β€” a page surfaced by + * vector KNN may not appear in the lexical results, so it has no hit yet. The + * caller supplies the fused score; matching sections are computed from the + * query with the same analyzer as search(). + */ + hitForUrl(url: string, query: string, score: number): SearchHit | undefined { + const p = this.getByUrl(url); + if (!p) return undefined; + const raw = [...new Set(tokenize(query))].filter((t) => !STOPWORDS.has(t)); + const base = raw.length ? raw : [...new Set(tokenize(query))]; + const qset = new Set(base.map(stem)); + return { + id: p.id, + title: p.title, + url: p.url, + summary: p.summary ?? "", + page_type: p.page_type ?? "content", + score: Number(score.toFixed(4)), + matching_section_ids: matchingSections(p, qset), + }; + } + + search(query: string, opts: SearchOptions = {}): SearchHit[] { + // Filter stopwords on RAW tokens (before stemming), then stem + dedupe. + const raw = [...new Set(tokenize(query))]; + let kept = raw.filter((t) => !STOPWORDS.has(t)); + if (kept.length === 0) kept = raw; // query was all stopwords + const qterms = [...new Set(kept.map(stem))]; + if (qterms.length === 0) return []; + + const idf = new Map<string, number>(); + for (const t of qterms) { + const df = this.df.get(t) ?? 0; + idf.set(t, Math.log(1 + (this.N - df + 0.5) / (df + 0.5))); + } + + const qset = new Set(qterms); + // Score first; defer the expensive per-section matchingSections() until + // after the top-k slice, so we only re-analyze section text for the handful + // of pages we actually return, not every positive-score page in the corpus. + const scored: Array<{ page: Page; score: number }> = []; + for (const d of this.docs) { + if (opts.pageType && (d.page.page_type ?? "content") !== opts.pageType) continue; + + let score = 0; + for (const t of qterms) { + const termIdf = idf.get(t) ?? 0; + const tf = d.tf.get(t); + if (tf) { + const denom = tf + K1 * (1 - B + B * (d.len / (this.avgdl || 1))); + score += termIdf * ((tf * (K1 + 1)) / denom); + } + // Field boosts: reward the term appearing in high-signal fields. + if (d.slugTok.has(t)) score += termIdf * W_SLUG; + if (d.titleTok.has(t)) score += termIdf * W_TITLE; + if (d.summaryTok.has(t)) score += termIdf * W_SUMMARY; + } + if (score > 0) { + scored.push({ page: d.page, score: Number((score * pageWeight(d.page.url)).toFixed(4)) }); + } + } + scored.sort((a, b) => b.score - a.score); + return scored.slice(0, opts.limit ?? 10).map(({ page, score }) => ({ + id: page.id, + title: page.title, + url: page.url, + summary: page.summary ?? "", + page_type: page.page_type ?? "content", + score, + matching_section_ids: matchingSections(page, qset), + })); + } +} diff --git a/build/docs-mcp-server/node/src/smoke.ts b/build/docs-mcp-server/node/src/smoke.ts new file mode 100644 index 0000000000..996d119220 --- /dev/null +++ b/build/docs-mcp-server/node/src/smoke.ts @@ -0,0 +1,82 @@ +// Offline smoke test: assertion-based checks of the index + tools + MCP +// response wrapping (no stdio transport). Exits non-zero on any failure. +// npm run smoke +// Runs against test/fixture.ndjson; the assertions encode fixture-specific +// ids/urls (incl. the colliding id="install" pair), so it is not meant to be +// pointed at the live feed. +import { fileURLToPath } from "node:url"; +import { loadFeed } from "./feed.js"; +import { DocsIndex } from "./search.js"; +import { searchDocs } from "./tools/search-docs.js"; +import { getPage } from "./tools/get-page.js"; +import { toolResult } from "./response.js"; + +let failures = 0; +function check(label: string, cond: boolean) { + console.log(`${cond ? "PASS" : "FAIL"} ${label}`); + if (!cond) failures++; +} + +const feed = + process.env.DOCS_NDJSON ?? + fileURLToPath(new URL("../test/fixture.ndjson", import.meta.url)); + +const pages = await loadFeed(feed); +const index = new DocsIndex(pages); +console.log(`loaded ${index.size} pages from ${feed}\n`); + +// --- search_docs --- +const stream = await searchDocs(index, { query: "append an entry to a stream" }); +check("search returns hits", stream.count > 0); +check("search hits carry a url", Boolean(stream.results[0]?.url)); + +// --- get_page happy paths --- +const xadd = getPage(index, { id: "commands/xadd", roles: ["parameters"] }) as any; +check("get_page(unique id) resolves", xadd.id === "commands/xadd"); +check("roles filter returns only 'parameters'", (xadd.sections ?? []).every((s: any) => s.role === "parameters")); + +const exact = getPage(index, { url: "https://redis.io/docs/latest/operate/redisinsight/install/" }) as any; +check("get_page(exact url) resolves the right page", exact.title === "Install Redis Insight"); + +// exact url is authoritative even when a non-unique id is passed alongside it +// (Bugbot round-3 High): search hits carry both id + url, and id "install" is ambiguous. +const exactPlusId = getPage(index, { + url: "https://redis.io/docs/latest/operate/redisinsight/install/", + id: "install", +}) as any; +check("exact url + non-unique id resolves (not ambiguous)", exactPlusId.title === "Install Redis Insight"); + +const suffixUnique = getPage(index, { url: "/commands/xadd/" }) as any; +check("get_page(unambiguous partial url) resolves", suffixUnique.id === "commands/xadd"); + +// --- get_page ambiguity (Bugbot High + Codex Medium) --- +const ambId = getPage(index, { id: "install" }) as any; +check("ambiguous id returns error", typeof ambId.error === "string"); +check("ambiguous id lists candidates", (ambId.candidates ?? []).length === 2); + +const ambUrl = getPage(index, { url: "/install/" }) as any; +check("ambiguous partial url returns error (not silent first match)", typeof ambUrl.error === "string"); +check("ambiguous partial url lists candidates", (ambUrl.candidates ?? []).length === 2); + +// --- boundary-anchored suffix (Bugbot round-2 High): "add" must NOT match "xadd" --- +const boundary = getPage(index, { url: "add" }) as any; +check("partial url matches only on path-segment boundary (add !-> xadd)", typeof boundary.error === "string" && !boundary.candidates); + +// --- conflicting handles (Codex convergence model): url and id point at different pages --- +const conflict = getPage(index, { + url: "https://redis.io/docs/latest/develop/data-types/json/", + id: "commands/xadd", +}) as any; +check("conflicting url+id returns error", typeof conflict.error === "string"); +check("conflicting url+id lists both candidates", (conflict.candidates ?? []).length === 2); + +const missing = getPage(index, { id: "does-not-exist-anywhere" }) as any; +check("missing page returns error", typeof missing.error === "string"); + +// --- MCP response wrapping (Fix 2 / Bugbot Medium) --- +check("toolResult(missing) sets isError", toolResult(missing).isError === true); +check("toolResult(ambiguous id) sets isError", toolResult(ambId).isError === true); +check("toolResult(search) does NOT set isError", toolResult(stream).isError === undefined); + +console.log(`\n${failures === 0 ? "ALL PASSED" : failures + " FAILED"}`); +if (failures > 0) process.exit(1); diff --git a/build/docs-mcp-server/node/src/stem-paice.ts b/build/docs-mcp-server/node/src/stem-paice.ts new file mode 100644 index 0000000000..3958a075e7 --- /dev/null +++ b/build/docs-mcp-server/node/src/stem-paice.ts @@ -0,0 +1,98 @@ +// Paice/Lancaster stemmer β€” an iterative, rule-table-driven stemmer, more +// aggressive than Porter (heavier conflation, editable rules). Included to +// bake off against Porter (src/stem.ts) on the retrieval eval. +// +// Rules use Paice's compact notation, one per line: the leading letters are the +// ending REVERSED; optional `*` = apply only while the word is still intact; +// digits = characters to remove; trailing letters = characters to append; +// `>` = continue (re-scan), `.` = stop. Rule set is the widely-published +// Lancaster default (as used by NLTK's LancasterStemmer). + +const RAW_RULES = [ + "ai*2.", "a*1.", "bb1.", "city3s.", "ci2>", "cn1t>", "dd1.", "dei3y>", + "deec2ss.", "dee1.", "de2>", "dooh4>", "e1>", "feil1v.", "fi2>", "gni3>", + "gai3y.", "ga2>", "gg1.", "ht*2.", "hsiug5ct.", "hsi3>", "i*1.", "i1y>", + "ji1d.", "juf1s.", "ju1d.", "jo1d.", "jeh1r.", "jrev1t.", "jsim2t.", "jn1d.", + "j1s.", "lbaifi6.", "lbai4y.", "lba3>", "lbi3.", "lib2l>", "lc1.", "lufi4y.", + "luf3>", "lu2.", "lai3>", "lau3>", "la2>", "ll1.", "mui3.", "mu*2.", "msi3>", + "mm1.", "nois4j>", "noix4ct.", "noi3>", "nai3>", "na2>", "nee0.", "ne2>", + "nn1.", "pihs4>", "pp1.", "re2>", "rae0.", "ra2.", "ro2>", "ru2>", "rr1.", + "rt1>", "rei3y>", "sei3y>", "sis2.", "si2>", "ssen4>", "ss0.", "suo3>", + "su*2.", "s*1>", "s0.", "tacilp4y.", "ta2>", "tnem4>", "tne3>", "tna3>", + "tpir2b.", "tpro2b.", "tcud1.", "tpmus2.", "tpec2iv.", "tulo2v.", "tsis0.", + "tsi3>", "tt1.", "uqi3.", "ugo1.", "vis3j>", "vie0.", "vi2>", "ylb1>", + "yli3y>", "ylp0.", "yl2>", "ygo1.", "yhp1.", "ymo1.", "ypo1.", "yti3>", + "yte3>", "ytl2.", "yrtsi5.", "yra3>", "yro3>", "yfi3.", "ycn2t>", "yca3>", + "zi2>", "zy1s.", +]; + +interface Rule { + end: string; // actual ending (un-reversed) + intact: boolean; // apply only if the word is unmodified + remove: number; + append: string; + cont: boolean; // continue re-scanning after applying +} + +function parseRule(s: string): Rule { + let i = 0; + let rev = ""; + while (i < s.length && s[i] >= "a" && s[i] <= "z") rev += s[i++]; + const intact = s[i] === "*"; + if (intact) i++; + let num = ""; + while (i < s.length && s[i] >= "0" && s[i] <= "9") num += s[i++]; + let append = ""; + while (i < s.length && s[i] >= "a" && s[i] <= "z") append += s[i++]; + return { + end: rev.split("").reverse().join(""), + intact, + remove: parseInt(num || "0", 10), + append, + cont: s[i] === ">", + }; +} + +// Group rules by the word's last character (= last char of the ending) for +// fast lookup, preserving Paice's rule order within each group. +const RULES_BY_LAST: Map<string, Rule[]> = (() => { + const m = new Map<string, Rule[]>(); + for (const raw of RAW_RULES) { + const r = parseRule(raw); + const key = r.end[r.end.length - 1]; + (m.get(key) ?? m.set(key, []).get(key)!).push(r); + } + return m; +})(); + +function acceptable(stem: string): boolean { + if (stem.length === 0) return false; + if ("aeiou".includes(stem[0])) return stem.length >= 2; + // consonant start: need >= 3 chars and at least one vowel (y counts) + if (stem.length < 3) return false; + for (const c of stem) if ("aeiouy".includes(c)) return true; + return false; +} + +export function stem(word: string): string { + let w = word.toLowerCase(); + let intact = true; + for (let guard = 0; guard < 50; guard++) { + const rules = RULES_BY_LAST.get(w[w.length - 1]); + if (!rules) return w; + let applied = false; + for (const r of rules) { + if (!w.endsWith(r.end)) continue; + if (r.intact && !intact) continue; + const candidate = w.slice(0, w.length - r.remove); + if (!acceptable(candidate)) return w; // matched ending but unacceptable β†’ stop + w = candidate + r.append; + intact = false; + if (!r.cont) return w; + applied = true; + break; // re-scan with the shortened word + } + if (!applied) return w; + } + return w; +} diff --git a/build/docs-mcp-server/node/src/stem.ts b/build/docs-mcp-server/node/src/stem.ts new file mode 100644 index 0000000000..e94296f3b4 --- /dev/null +++ b/build/docs-mcp-server/node/src/stem.ts @@ -0,0 +1,136 @@ +// Compact Porter stemmer (classic algorithm). Applied to BOTH index and query +// tokens so word-form variants conflate (append/appends, prepend/prepending, +// expire/expires, queries/query). Consistency matters more than linguistic +// perfection here β€” the retrieval eval validates the net effect. + +const step2list: Record<string, string> = { + ational: "ate", tional: "tion", enci: "ence", anci: "ance", izer: "ize", + bli: "ble", alli: "al", entli: "ent", eli: "e", ousli: "ous", + ization: "ize", ation: "ate", ator: "ate", alism: "al", iveness: "ive", + fulness: "ful", ousness: "ous", aliti: "al", iviti: "ive", biliti: "ble", + logi: "log", +}; +const step3list: Record<string, string> = { + icate: "ic", ative: "", alize: "al", iciti: "ic", ical: "ic", ful: "", ness: "", +}; + +const c = "[^aeiou]"; +const v = "[aeiouy]"; +const C = c + "[^aeiouy]*"; +const V = v + "[aeiou]*"; +const mgr0 = "^(" + C + ")?" + V + C; +const meq1 = "^(" + C + ")?" + V + C + "(" + V + ")?$"; +const mgr1 = "^(" + C + ")?" + V + C + V + C; +const s_v = "^(" + C + ")?" + v; + +export function stem(w: string): string { + if (w.length < 3) return w; + + let stemmed: string; + let suffix: string; + let re: RegExp; + let re2: RegExp; + let re3: RegExp; + let re4: RegExp; + + const firstch = w.substr(0, 1); + if (firstch === "y") w = firstch.toUpperCase() + w.substr(1); + + // Step 1a + re = /^(.+?)(ss|i)es$/; + re2 = /^(.+?)([^s])s$/; + if (re.test(w)) w = w.replace(re, "$1$2"); + else if (re2.test(w)) w = w.replace(re2, "$1$2"); + + // Step 1b + re = /^(.+?)eed$/; + re2 = /^(.+?)(ed|ing)$/; + if (re.test(w)) { + const fp = re.exec(w)!; + re = new RegExp(mgr0); + if (re.test(fp[1])) { + re = /.$/; + w = w.replace(re, ""); + } + } else if (re2.test(w)) { + const fp = re2.exec(w)!; + stemmed = fp[1]; + re2 = new RegExp(s_v); + if (re2.test(stemmed)) { + w = stemmed; + re2 = /(at|bl|iz)$/; + re3 = new RegExp("([^aeiouylsz])\\1$"); + re4 = new RegExp("^" + C + v + "[^aeiouwxy]$"); + if (re2.test(w)) w = w + "e"; + else if (re3.test(w)) { + re = /.$/; + w = w.replace(re, ""); + } else if (re4.test(w)) w = w + "e"; + } + } + + // Step 1c + re = /^(.+?)y$/; + if (re.test(w)) { + const fp = re.exec(w)!; + stemmed = fp[1]; + re = new RegExp(s_v); + if (re.test(stemmed)) w = stemmed + "i"; + } + + // Step 2 + re = /^(.+?)(ational|tional|enci|anci|izer|bli|alli|entli|eli|ousli|ization|ation|ator|alism|iveness|fulness|ousness|aliti|iviti|biliti|logi)$/; + if (re.test(w)) { + const fp = re.exec(w)!; + stemmed = fp[1]; + suffix = fp[2]; + re = new RegExp(mgr0); + if (re.test(stemmed)) w = stemmed + step2list[suffix]; + } + + // Step 3 + re = /^(.+?)(icate|ative|alize|iciti|ical|ful|ness)$/; + if (re.test(w)) { + const fp = re.exec(w)!; + stemmed = fp[1]; + suffix = fp[2]; + re = new RegExp(mgr0); + if (re.test(stemmed)) w = stemmed + step3list[suffix]; + } + + // Step 4 + re = /^(.+?)(al|ance|ence|er|ic|able|ible|ant|ement|ment|ent|ou|ism|ate|iti|ous|ive|ize)$/; + re2 = /^(.+?)(s|t)(ion)$/; + if (re.test(w)) { + const fp = re.exec(w)!; + stemmed = fp[1]; + re = new RegExp(mgr1); + if (re.test(stemmed)) w = stemmed; + } else if (re2.test(w)) { + const fp = re2.exec(w)!; + stemmed = fp[1] + fp[2]; + re2 = new RegExp(mgr1); + if (re2.test(stemmed)) w = stemmed; + } + + // Step 5a + re = /^(.+?)e$/; + if (re.test(w)) { + const fp = re.exec(w)!; + stemmed = fp[1]; + re = new RegExp(mgr1); + re2 = new RegExp(meq1); + re3 = new RegExp("^" + C + v + "[^aeiouwxy]$"); + if (re.test(stemmed) || (re2.test(stemmed) && !re3.test(stemmed))) w = stemmed; + } + + // Step 5b + re = /ll$/; + re2 = new RegExp(mgr1); + if (re.test(w) && re2.test(w)) { + re = /.$/; + w = w.replace(re, ""); + } + + return w.toLowerCase(); +} diff --git a/build/docs-mcp-server/node/src/tools/get-page.ts b/build/docs-mcp-server/node/src/tools/get-page.ts new file mode 100644 index 0000000000..d3326423b8 --- /dev/null +++ b/build/docs-mcp-server/node/src/tools/get-page.ts @@ -0,0 +1,99 @@ +import { z } from "zod"; +import type { DocsIndex } from "../search.js"; +import type { Page } from "../types.js"; + +export const GetPageInput = z + .object({ + id: z.string().optional(), + url: z.string().optional(), + roles: z.array(z.string()).optional(), + }) + .refine((v) => Boolean(v.id || v.url), { + message: "Provide either 'id' or 'url'.", + }); +export type GetPageInput = z.infer<typeof GetPageInput>; + +/** + * The distinct pages (deduped by unique url) that the supplied handles resolve + * to. `url` is authoritative when it matches an exact page; otherwise it is + * treated as a path-boundary suffix. `id` is the last URL segment and is not + * unique, so it may contribute several pages. Collecting across all handles + * (rather than short-circuiting) means an ambiguous url still lets `id` help, + * and it surfaces the case where url and id point at *different* pages. + */ +function collectCandidates(index: DocsIndex, input: GetPageInput): Page[] { + const byUrl = new Map<string, Page>(); + const add = (p: Page | undefined) => { + if (p) byUrl.set(p.url, p); + }; + + // A boundary-suffix url (only reached when there was no exact match) plus id. + if (input.url) index.matchByUrlSuffix(input.url).forEach(add); + if (input.id) index.getPagesById(input.id).forEach(add); + return [...byUrl.values()]; +} + +function describeHandles(input: GetPageInput): string { + const parts: string[] = []; + if (input.url) parts.push(`url '${input.url}'`); + if (input.id) parts.push(`id '${input.id}'`); + return parts.join(" and "); +} + +function ambiguous(input: GetPageInput, candidates: Page[]) { + return { + error: `Ambiguous lookup: ${describeHandles(input)} matched ${candidates.length} pages. Call get_page again with a single, exact url.`, + candidates: candidates.map((p) => ({ title: p.title, url: p.url })), + }; +} + +function render(page: Page, input: GetPageInput) { + let sections = page.sections ?? []; + if (input.roles && input.roles.length) { + const want = new Set(input.roles.map((r) => r.toLowerCase())); + sections = sections.filter((s) => want.has((s.role ?? "").toLowerCase())); + } + return { + id: page.id, + title: page.title, + url: page.url, + summary: page.summary ?? "", + page_type: page.page_type ?? "content", + content_hash: page.content_hash, + sections, + }; +} + +/** Fetch one page, optionally filtered to sections with the given roles. */ +export function getPage(index: DocsIndex, input: GetPageInput) { + // An exact url is unique and authoritative. Return it directly rather than + // diluting it with a (possibly non-unique) id supplied alongside it β€” UNLESS + // the id points somewhere else entirely, which is a genuine conflict worth + // surfacing. A search hit's id is that page's own id, so the common + // exact-url + its-own-id case resolves cleanly. + if (input.url) { + const exact = index.getByUrl(input.url); + if (exact) { + if (input.id) { + const idPages = index.getPagesById(input.id); + if (idPages.length > 0 && !idPages.some((p) => p.url === exact.url)) { + const byUrl = new Map<string, Page>(); + [exact, ...idPages].forEach((p) => byUrl.set(p.url, p)); + return ambiguous(input, [...byUrl.values()]); + } + } + return render(exact, input); + } + } + + // Otherwise converge across the remaining handles (boundary-suffix url + id) + // and resolve only when they point at exactly one page. + const candidates = collectCandidates(index, input); + if (candidates.length === 0) { + return { error: `Page not found for ${describeHandles(input)}.` }; + } + if (candidates.length > 1) { + return ambiguous(input, candidates); + } + return render(candidates[0], input); +} diff --git a/build/docs-mcp-server/node/src/tools/search-docs.ts b/build/docs-mcp-server/node/src/tools/search-docs.ts new file mode 100644 index 0000000000..e17089a356 --- /dev/null +++ b/build/docs-mcp-server/node/src/tools/search-docs.ts @@ -0,0 +1,19 @@ +import { z } from "zod"; +import type { Searcher } from "../search.js"; + +export const SearchDocsInput = z.object({ + query: z.string().min(1, "query is required"), + page_type: z.enum(["content", "index"]).optional(), + limit: z.number().int().positive().max(50).optional(), +}); +export type SearchDocsInput = z.infer<typeof SearchDocsInput>; + +/** Rank pages by relevance. Returns refs + summaries only β€” never full text. + * The searcher may be lexical (sync) or hybrid (async), so this awaits. */ +export async function searchDocs(searcher: Searcher, input: SearchDocsInput) { + const results = await searcher.search(input.query, { + limit: input.limit ?? 10, + pageType: input.page_type, + }); + return { query: input.query, count: results.length, results }; +} diff --git a/build/docs-mcp-server/node/src/types.ts b/build/docs-mcp-server/node/src/types.ts new file mode 100644 index 0000000000..82ab8cc32d --- /dev/null +++ b/build/docs-mcp-server/node/src/types.ts @@ -0,0 +1,35 @@ +// Document schema as published in docs.ndjson / per-page index.json. +// See content/ai-agent-resources.md for the authoritative field reference. + +export interface Section { + id: string; + title: string; + /** Semantic role: overview | syntax | parameters | returns | example | ... */ + role?: string; + text: string; +} + +export interface Example { + id: string; + language: string; + code: string; + section_id: string; +} + +export interface Child { + title?: string; + url?: string; +} + +export interface Page { + id: string; + title: string; + url: string; + summary?: string; + /** "content" (has prose) or "index" (navigation only) */ + page_type?: string; + content_hash?: string; + sections?: Section[]; + examples?: Example[]; + children?: Child[]; +} diff --git a/build/docs-mcp-server/node/src/url.ts b/build/docs-mcp-server/node/src/url.ts new file mode 100644 index 0000000000..04516dac8e --- /dev/null +++ b/build/docs-mcp-server/node/src/url.ts @@ -0,0 +1,8 @@ +// Canonical URL normalization, shared across the lexical index, chunker, vector +// store, and hybrid fusion. It MUST be the single definition: the hybrid path +// matches pages across modules by normalized url (chunk.ts indexes by it, +// vector-store.ts returns it, hybrid.ts fuses on it, search.ts looks up by it), +// so any divergence would silently break cross-module findability. +export function normalizeUrl(u: string): string { + return u.trim().toLowerCase().replace(/\/+$/, ""); +} diff --git a/build/docs-mcp-server/node/src/vector-store.ts b/build/docs-mcp-server/node/src/vector-store.ts new file mode 100644 index 0000000000..63d7056ccc --- /dev/null +++ b/build/docs-mcp-server/node/src/vector-store.ts @@ -0,0 +1,117 @@ +// Redis vector-KNN backend for hybrid mode. Redis is used ONLY for vector +// search (DOC-6809 Step 3 verdict: keep BM25 lexical + weighted-RRF fusion +// app-side; native FT.HYBRID can't express the weighted recipe). Section chunks +// are stored as HASHes with a FLAT/COSINE vector field; `owner` (page url) is +// unindexed and pulled back via RETURN. FLAT = exact KNN, which reproduced the +// offline numpy ranking to tie-breaking in Step 1; swap to HNSW later if the +// corpus grows enough to need it. +import { + createClient, + SCHEMA_FIELD_TYPE, + SCHEMA_VECTOR_FIELD_ALGORITHM, + type RedisClientType, +} from "redis"; +import { EMBED_DIM } from "./constants.js"; +import { normalizeUrl } from "./url.js"; + +const INDEX = "docs_vec"; +const PREFIX = "docvec:"; + +function vecBuffer(v: Float32Array): Buffer { + return Buffer.from(v.buffer, v.byteOffset, v.byteLength); +} + +export class VectorStore { + private client: RedisClientType; + + constructor(url: string) { + this.client = createClient({ url }); + this.client.on("error", (e) => console.error("[redis-docs-mcp] redis:", e)); + } + + async connect(): Promise<void> { + if (!this.client.isOpen) await this.client.connect(); + } + + async close(): Promise<void> { + if (this.client.isOpen) await this.client.close(); + } + + /** Create the FLAT/COSINE index if absent (idempotent). */ + async ensureIndex(dim = EMBED_DIM): Promise<void> { + try { + await this.client.ft.create( + INDEX, + { + vec: { + type: SCHEMA_FIELD_TYPE.VECTOR, + ALGORITHM: SCHEMA_VECTOR_FIELD_ALGORITHM.FLAT, + TYPE: "FLOAT32", + DIM: dim, + DISTANCE_METRIC: "COSINE", + }, + }, + { ON: "HASH", PREFIX }, + ); + } catch (e) { + const msg = e instanceof Error ? e.message : String(e); + if (!/index already exists/i.test(msg)) throw e; + } + } + + async dropIndex(): Promise<void> { + try { + await this.client.ft.dropIndex(INDEX, { DD: true }); + } catch (e) { + // Swallow ONLY "index doesn't exist" (nothing to drop). Auth, permission, + // or network errors must propagate β€” otherwise a failed drop is silently + // ignored and the loader reuses/mixes into a stale index. + const msg = e instanceof Error ? e.message : String(e); + if (!/unknown index|no such index|not exist/i.test(msg)) throw e; + } + } + + /** Bulk-load chunk vectors. Each chunk becomes one HASH {owner, vec}. */ + async loadChunks(chunks: Array<{ owner: string; vec: Float32Array }>): Promise<void> { + const BATCH = 2000; + for (let i = 0; i < chunks.length; i += BATCH) { + const slice = chunks.slice(i, i + BATCH); + await Promise.all( + slice.map((c, j) => + this.client.hSet(`${PREFIX}${i + j}`, { + owner: c.owner, + vec: vecBuffer(c.vec), + }), + ), + ); + } + } + + /** + * KNN over `k` chunks, deduped to the nearest chunk per page. Returns page + * urls, nearest first (mirrors the offline rank_pages()). + */ + async knn(qvec: Float32Array, k = 200, topn = 50): Promise<string[]> { + const res = await this.client.ft.search( + INDEX, + `*=>[KNN ${k} @vec $BLOB AS score]`, + { + PARAMS: { BLOB: vecBuffer(qvec) }, + SORTBY: { BY: "score", DIRECTION: "ASC" }, // cosine distance: nearest = smallest + RETURN: ["owner"], + LIMIT: { from: 0, size: k }, + DIALECT: 2, + }, + ); + const seen = new Set<string>(); + const pages: string[] = []; + for (const doc of res.documents) { + const owner = normalizeUrl(String((doc.value as { owner?: string }).owner ?? "")); + if (!owner || seen.has(owner)) continue; + seen.add(owner); + pages.push(owner); + if (pages.length >= topn) break; + } + return pages; + } +} diff --git a/build/docs-mcp-server/node/test/embed-dump.mjs b/build/docs-mcp-server/node/test/embed-dump.mjs new file mode 100644 index 0000000000..764d8f78b9 --- /dev/null +++ b/build/docs-mcp-server/node/test/embed-dump.mjs @@ -0,0 +1,55 @@ +// Step 2b: embed the exported parity strings with fastembed-js (bge-small, +// ONNX in Node) and dump the vectors for Python-side comparison. Uses plain +// embed() and prepends the SAME BGE query prefix as the Python eval, so the +// only variable under test is the Node ONNX embedding itself (tokenizer + +// model + pooling), not prefix wording or method choice. +// +// node test/embed-dump.mjs +import { FlagEmbedding, EmbeddingModel } from "fastembed"; +import fs from "node:fs"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; + +const HERE = path.dirname(fileURLToPath(import.meta.url)); +const IN = path.join(HERE, "..", "..", "redis-eval", "parity_texts.json"); +const OUT = path.join(HERE, "..", "..", "redis-eval", "node_vectors.json"); + +function l2normalize(v) { + let n = 0; + for (const x of v) n += x * x; + n = Math.sqrt(n) + 1e-12; + return v.map((x) => x / n); +} + +async function embedAll(model, texts) { + const out = []; + // embed() yields batches of Float32-ish arrays. + for await (const batch of model.embed(texts, 64)) { + for (const v of batch) out.push(l2normalize(Array.from(v))); + } + return out; +} + +const { query_prefix, queries, corpus } = JSON.parse(fs.readFileSync(IN, "utf-8")); + +const model = await FlagEmbedding.init({ model: EmbeddingModel.BGESmallENV15 }); + +console.log(`embedding ${queries.length} queries + ${corpus.length} corpus chunks ...`); +const t0 = Date.now(); +const queryVecs = await embedAll(model, queries.map((q) => query_prefix + q)); +const corpusVecs = await embedAll(model, corpus); +const secs = (Date.now() - t0) / 1000; + +fs.writeFileSync( + OUT, + JSON.stringify({ + model: "fast-bge-small-en-v1.5", + dim: queryVecs[0].length, + queries: queryVecs, + corpus: corpusVecs, + }) +); +console.log( + `wrote ${OUT}: dim ${queryVecs[0].length}, ` + + `${(queries.length + corpus.length) / secs | 0} texts/s` +); diff --git a/build/docs-mcp-server/node/test/eval/cases.json b/build/docs-mcp-server/node/test/eval/cases.json new file mode 100644 index 0000000000..3e24aa50ce --- /dev/null +++ b/build/docs-mcp-server/node/test/eval/cases.json @@ -0,0 +1,38 @@ +[ + { "kind": "command", "q": "append an entry to a stream", "expected": ["https://redis.io/docs/latest/commands/xadd/"] }, + { "kind": "command", "q": "add a member to a sorted set with a score", "expected": ["https://redis.io/docs/latest/commands/zadd/"] }, + { "kind": "command", "q": "set a string value only if the key does not already exist", "expected": ["https://redis.io/docs/latest/commands/setnx/", "https://redis.io/docs/latest/commands/set/"] }, + { "kind": "command", "q": "make a key expire after a given number of seconds", "expected": ["https://redis.io/docs/latest/commands/expire/", "https://redis.io/docs/latest/commands/pexpire/"] }, + { "kind": "command", "q": "atomically increment the integer stored at a key", "expected": ["https://redis.io/docs/latest/commands/incr/", "https://redis.io/docs/latest/commands/incrby/"] }, + { "kind": "command", "q": "remove a key from the database", "expected": ["https://redis.io/docs/latest/commands/del/", "https://redis.io/docs/latest/commands/unlink/"] }, + { "kind": "command", "q": "get all the fields and values stored in a hash", "expected": ["https://redis.io/docs/latest/commands/hgetall/"] }, + { "kind": "command", "q": "publish a message to a channel", "expected": ["https://redis.io/docs/latest/commands/publish/"] }, + { "kind": "command", "q": "listen for messages on a channel", "expected": ["https://redis.io/docs/latest/commands/subscribe/"] }, + { "kind": "command", "q": "prepend an element to the beginning of a list", "expected": ["https://redis.io/docs/latest/commands/lpush/"] }, + { "kind": "command", "q": "read a range of elements from a list", "expected": ["https://redis.io/docs/latest/commands/lrange/"] }, + { "kind": "command", "q": "check how long until a key expires", "expected": ["https://redis.io/docs/latest/commands/ttl/", "https://redis.io/docs/latest/commands/pttl/"] }, + { "kind": "command", "q": "add one or more members to a set", "expected": ["https://redis.io/docs/latest/commands/sadd/"] }, + { "kind": "command", "q": "run a server-side Lua script", "expected": ["https://redis.io/docs/latest/commands/eval/", "https://redis.io/docs/latest/commands/eval_ro/"] }, + { "kind": "command", "q": "retrieve the value of a string key", "expected": ["https://redis.io/docs/latest/commands/get/"] }, + { "kind": "command", "q": "incrementally iterate the keyspace without blocking the server", "expected": ["https://redis.io/docs/latest/commands/scan/"] }, + { "kind": "command", "q": "rename an existing key", "expected": ["https://redis.io/docs/latest/commands/rename/"] }, + { "kind": "command", "q": "set multiple fields on a hash at once", "expected": ["https://redis.io/docs/latest/commands/hset/", "https://redis.io/docs/latest/commands/hmset/"] }, + { "kind": "command", "q": "remove and return the first element of a list", "expected": ["https://redis.io/docs/latest/commands/lpop/", "https://redis.io/docs/latest/commands/blpop/"] }, + { "kind": "command", "q": "count the number of members in a set", "expected": ["https://redis.io/docs/latest/commands/scard/"] }, + { "kind": "command", "q": "store a JSON document at a path", "expected": ["https://redis.io/docs/latest/commands/json.set/"] }, + { "kind": "command", "q": "create a full-text search index", "expected": ["https://redis.io/docs/latest/commands/ft.create/"] }, + + { "kind": "concept", "q": "connect to Redis from a Python application", "expected": ["https://redis.io/docs/latest/develop/clients/redis-py/connect/"] }, + { "kind": "concept", "q": "connect to Redis using the Jedis Java client", "expected": ["https://redis.io/docs/latest/develop/clients/jedis/connect/"] }, + { "kind": "concept", "q": "what are the differences between the Redis data types", "expected": ["https://redis.io/docs/latest/develop/data-types/compare-data-types/", "https://redis.io/docs/latest/develop/data-types/"] }, + { "kind": "concept", "q": "how do Redis transactions work", "expected": ["https://redis.io/docs/latest/develop/using-commands/transactions/"] }, + { "kind": "concept", "q": "send several commands together to reduce round trips", "expected": ["https://redis.io/docs/latest/develop/using-commands/pipelining/"] }, + { "kind": "concept", "q": "configure persistence with RDB snapshots and the append-only file", "expected": ["https://redis.io/docs/latest/operate/oss_and_stack/management/persistence/"] }, + { "kind": "concept", "q": "set up replication between a primary and its replicas", "expected": ["https://redis.io/docs/latest/operate/oss_and_stack/management/replication/"] }, + { "kind": "concept", "q": "run a vector similarity search over embeddings", "expected": ["https://redis.io/docs/latest/develop/ai/search-and-query/query/vector-search/", "https://redis.io/docs/latest/develop/ai/search-and-query/"] }, + { "kind": "concept", "q": "get notified when keys are changed or expire", "expected": ["https://redis.io/docs/latest/develop/pubsub/keyspace-notifications/"] }, + { "kind": "concept", "q": "cache data on the client side to reduce load", "expected": ["https://redis.io/docs/latest/develop/clients/client-side-caching/", "https://redis.io/docs/latest/develop/reference/client-side-caching/"] }, + { "kind": "concept", "q": "work with the string data type", "expected": ["https://redis.io/docs/latest/develop/data-types/strings/"] }, + { "kind": "concept", "q": "use hashes to store object-like records", "expected": ["https://redis.io/docs/latest/develop/data-types/hashes/"] }, + { "kind": "concept", "q": "full-text search and secondary indexing over Redis data", "expected": ["https://redis.io/docs/latest/develop/ai/search-and-query/"] } +] diff --git a/build/docs-mcp-server/node/test/eval/dump-lexical.mjs b/build/docs-mcp-server/node/test/eval/dump-lexical.mjs new file mode 100644 index 0000000000..678b603410 --- /dev/null +++ b/build/docs-mcp-server/node/test/eval/dump-lexical.mjs @@ -0,0 +1,30 @@ +// Dump the REAL lexical ranker's results for every eval case, so the Python +// vector experiment can compare against (and fuse with) the exact production +// ranking rather than a reimplementation. Reuses the built dist/. +// node test/eval/dump-lexical.mjs > test/eval/lexical.json +import { readFile } from "node:fs/promises"; +import { fileURLToPath } from "node:url"; +import { loadFeed } from "../../dist/feed.js"; +import { DocsIndex } from "../../dist/search.js"; +import { searchDocs } from "../../dist/tools/search-docs.js"; + +const TOPN = 50; +const norm = (u) => u.trim().toLowerCase().replace(/\/+$/, ""); +const feedSrc = + process.env.DOCS_NDJSON ?? fileURLToPath(new URL("./docs.ndjson.gz", import.meta.url)); +const cases = JSON.parse(await readFile(fileURLToPath(new URL("./cases.json", import.meta.url)), "utf8")); + +const index = new DocsIndex(await loadFeed(feedSrc)); + +const out = []; +for (const c of cases) { + const { results } = await searchDocs(index, { query: c.q, limit: TOPN }); + out.push({ + q: c.q, + kind: c.kind ?? "command", + expected: c.expected.map(norm), + lexical: results.map((r) => norm(r.url)), + }); +} + +process.stdout.write(JSON.stringify(out, null, 2) + "\n"); diff --git a/build/docs-mcp-server/node/test/eval/run-hybrid.mjs b/build/docs-mcp-server/node/test/eval/run-hybrid.mjs new file mode 100644 index 0000000000..861c5d5218 --- /dev/null +++ b/build/docs-mcp-server/node/test/eval/run-hybrid.mjs @@ -0,0 +1,95 @@ +// Hybrid retrieval eval (DOC-6809 Step 4 acceptance). Runs the 35 cases through +// the REAL search_docs tool function backed by HybridSearcher β€” the exact path +// index.ts calls β€” so this exercises query embedding (fastembed-js) + Redis KNN +// + app-side weighted RRF end to end. Confirms the shipped hybrid mode +// reproduces the offline recipe (overall MRR ~.73). +// +// Prereq: load the vector index first, against the same REDIS_URL: +// REDIS_URL=redis://localhost:6379 node scripts/load-index.mjs --vectors ../redis-eval/vecdump +// REDIS_URL=redis://localhost:6379 node test/eval/run-hybrid.mjs +// +// Imports the BUILT server (dist/) β€” run `npm run build` first (eval:hybrid does). +import { readFile } from "node:fs/promises"; +import { fileURLToPath } from "node:url"; +import { loadFeed } from "../../dist/feed.js"; +import { DocsIndex } from "../../dist/search.js"; +import { HybridSearcher } from "../../dist/hybrid.js"; +import { VectorStore } from "../../dist/vector-store.js"; +import { searchDocs } from "../../dist/tools/search-docs.js"; + +const K = [1, 3, 5, 10]; +const LIMIT = 10; +const REDIS_URL = process.env.REDIS_URL ?? "redis://localhost:6379"; +const norm = (u) => u.trim().toLowerCase().replace(/\/+$/, ""); +const short = (u) => (u ? u.replace("https://redis.io/docs/latest", "") : "β€”"); + +const feedSrc = + process.env.DOCS_NDJSON ?? fileURLToPath(new URL("./docs.ndjson.gz", import.meta.url)); +const cases = JSON.parse(await readFile(fileURLToPath(new URL("./cases.json", import.meta.url)), "utf8")); + +const pages = await loadFeed(feedSrc); +const index = new DocsIndex(pages); +const feedUrls = new Set(pages.map((p) => norm(p.url))); + +const store = new VectorStore(REDIS_URL); +await store.connect(); +await store.ensureIndex(); +const hybrid = new HybridSearcher(index, store); + +const broken = []; +const rows = []; +for (const c of cases) { + const expected = c.expected.map(norm); + if (expected.every((u) => !feedUrls.has(u))) { + broken.push({ q: c.q, missing: expected }); + continue; + } + const { results } = await searchDocs(hybrid, { query: c.q, limit: LIMIT }); + const urls = results.map((r) => norm(r.url)); + let rank = null; + for (let i = 0; i < urls.length; i++) { + if (expected.includes(urls[i])) { + rank = i + 1; + break; + } + } + rows.push({ kind: c.kind ?? "command", q: c.q, rank, top: urls[0] }); +} +await store.close(); + +function metrics(set) { + const n = set.length || 1; + const recall = Object.fromEntries( + K.map((k) => [k, set.filter((r) => r.rank && r.rank <= k).length / n]), + ); + const mrr = set.reduce((s, r) => s + (r.rank ? 1 / r.rank : 0), 0) / n; + return { recall, mrr }; +} + +console.log( + `Hybrid via ${REDIS_URL} | ${pages.length} pages | cases scored: ${rows.length}` + + (broken.length ? ` | ${broken.length} BROKEN` : "") + + "\n", +); +for (const r of rows) { + const tag = r.rank ? `#${r.rank}`.padEnd(5) : "MISS "; + console.log(`${tag} [${r.kind.slice(0, 4)}] ${r.q}${r.rank ? "" : ` [rank-1 was: ${short(r.top)}]`}`); +} + +const groups = [ + ["overall", rows], + ["command", rows.filter((r) => r.kind === "command")], + ["concept", rows.filter((r) => r.kind === "concept")], +]; +console.log("\n--- hybrid retrieval quality (recall@1 / @3 / @5 / @10 | MRR) ---"); +for (const [label, set] of groups) { + if (!set.length) continue; + const m = metrics(set); + const cells = K.map((k) => `${(m.recall[k] * 100).toFixed(0)}%`.padStart(4)).join(" / "); + console.log(`${label.padEnd(8)} (n=${String(set.length).padStart(2)}): ${cells} | ${m.mrr.toFixed(3)}`); +} + +if (broken.length) { + console.log("\n--- BROKEN eval cases ---"); + for (const b of broken) console.log(` "${b.q}" -> ${b.missing.map(short).join(", ")}`); +} diff --git a/build/docs-mcp-server/node/test/eval/run.mjs b/build/docs-mcp-server/node/test/eval/run.mjs new file mode 100644 index 0000000000..b59fbaf882 --- /dev/null +++ b/build/docs-mcp-server/node/test/eval/run.mjs @@ -0,0 +1,84 @@ +// AI-answer eval (retrieval quality) for the docs MCP server. +// Runs each question in cases.json through search_docs and measures whether the +// expected canonical page is retrieved (recall@k, MRR). A data-integrity check +// flags any expected url that isn't in the feed, so a bad ground-truth entry is +// reported rather than silently scored as a miss. +// +// node test/eval/run.mjs # uses cached test/eval/docs.ndjson.gz +// DOCS_NDJSON=<path|url> node test/eval/run.mjs +// +// Imports the BUILT server (dist/), so run `npm run build` first. +import { readFile } from "node:fs/promises"; +import { fileURLToPath } from "node:url"; +import { loadFeed } from "../../dist/feed.js"; +import { DocsIndex } from "../../dist/search.js"; +import { searchDocs } from "../../dist/tools/search-docs.js"; + +const K = [1, 3, 5, 10]; +const LIMIT = 10; +const norm = (u) => u.trim().toLowerCase().replace(/\/+$/, ""); +const short = (u) => (u ? u.replace("https://redis.io/docs/latest", "") : "β€”"); + +const feedSrc = + process.env.DOCS_NDJSON ?? fileURLToPath(new URL("./docs.ndjson.gz", import.meta.url)); +const cases = JSON.parse(await readFile(fileURLToPath(new URL("./cases.json", import.meta.url)), "utf8")); + +const pages = await loadFeed(feedSrc); +const index = new DocsIndex(pages); +const feedUrls = new Set(pages.map((p) => norm(p.url))); + +const broken = []; +const rows = []; +for (const c of cases) { + const expected = c.expected.map(norm); + if (expected.every((u) => !feedUrls.has(u))) { + broken.push({ q: c.q, missing: expected }); + continue; + } + const results = (await searchDocs(index, { query: c.q, limit: LIMIT })).results.map((r) => norm(r.url)); + let rank = null; + for (let i = 0; i < results.length; i++) { + if (expected.includes(results[i])) { + rank = i + 1; + break; + } + } + rows.push({ kind: c.kind ?? "command", q: c.q, rank, top: results[0] }); +} + +function metrics(set) { + const n = set.length || 1; + const recall = Object.fromEntries( + K.map((k) => [k, set.filter((r) => r.rank && r.rank <= k).length / n]), + ); + const mrr = set.reduce((s, r) => s + (r.rank ? 1 / r.rank : 0), 0) / n; + return { recall, mrr }; +} + +console.log( + `Feed: ${pages.length} pages | cases scored: ${rows.length}` + + (broken.length ? ` | ${broken.length} BROKEN (expected url not in feed)` : "") + + "\n", +); +for (const r of rows) { + const tag = r.rank ? `#${r.rank}`.padEnd(5) : "MISS "; + console.log(`${tag} [${r.kind.slice(0, 4)}] ${r.q}${r.rank ? "" : ` [rank-1 was: ${short(r.top)}]`}`); +} + +const groups = [ + ["overall", rows], + ["command", rows.filter((r) => r.kind === "command")], + ["concept", rows.filter((r) => r.kind === "concept")], +]; +console.log("\n--- retrieval quality (recall@1 / @3 / @5 / @10 | MRR) ---"); +for (const [label, set] of groups) { + if (!set.length) continue; + const m = metrics(set); + const cells = K.map((k) => `${(m.recall[k] * 100).toFixed(0)}%`.padStart(4)).join(" / "); + console.log(`${label.padEnd(8)} (n=${String(set.length).padStart(2)}): ${cells} | ${m.mrr.toFixed(3)}`); +} + +if (broken.length) { + console.log("\n--- BROKEN eval cases (fix ground truth) ---"); + for (const b of broken) console.log(` "${b.q}" -> not in feed: ${b.missing.map(short).join(", ")}`); +} diff --git a/build/docs-mcp-server/node/test/fixture.ndjson b/build/docs-mcp-server/node/test/fixture.ndjson new file mode 100644 index 0000000000..95b767477c --- /dev/null +++ b/build/docs-mcp-server/node/test/fixture.ndjson @@ -0,0 +1,5 @@ +{"id":"commands/xadd","title":"XADD","url":"https://redis.io/docs/latest/commands/xadd/","summary":"Appends a new entry to a stream.","page_type":"content","content_hash":"abc123","sections":[{"id":"overview","title":"XADD","role":"overview","text":"Appends the specified stream entry to the stream at the specified key. If the key does not exist, a new stream is created."},{"id":"parameters","title":"Parameters","role":"parameters","text":"key: the name of the stream. NOMKSTREAM: optional, do not create the stream if it does not exist. field value: one or more field-value pairs to add to the stream entry."},{"id":"return","title":"Return value","role":"returns","text":"Returns the ID of the added stream entry."}],"examples":[{"id":"overview-ex0","language":"python","code":"r.xadd('mystream', {'field': 'value'})","section_id":"overview"}],"children":[]} +{"id":"develop/data-types/json","title":"JSON","url":"https://redis.io/docs/latest/develop/data-types/json/","summary":"Store and query JSON documents in Redis.","page_type":"content","content_hash":"def456","sections":[{"id":"overview","title":"Redis JSON","role":"overview","text":"The JSON data type lets you store, update, and retrieve JSON values in a Redis database."},{"id":"example","title":"Example","role":"example","text":"Use JSON.SET to store a document and JSON.GET to retrieve values from it by path."}],"examples":[{"id":"example-ex0","language":"python","code":"r.json().set('doc', '$', {'a': 1})","section_id":"example"}],"children":[]} +{"id":"commands","title":"Commands","url":"https://redis.io/docs/latest/commands/","summary":"Reference documentation for all Redis commands.","page_type":"index","children":[{"title":"XADD","url":"https://redis.io/docs/latest/commands/xadd/"}]} +{"id":"install","title":"Install Redis","url":"https://redis.io/docs/latest/operate/oss_and_stack/install/","summary":"Install Redis Open Source on your platform.","page_type":"content","content_hash":"aaa111","sections":[{"id":"overview","title":"Install Redis","role":"overview","text":"Install Redis Open Source on Linux, macOS, or Windows."}],"examples":[],"children":[]} +{"id":"install","title":"Install Redis Insight","url":"https://redis.io/docs/latest/operate/redisinsight/install/","summary":"Install the Redis Insight GUI.","page_type":"content","content_hash":"bbb222","sections":[{"id":"overview","title":"Install Redis Insight","role":"overview","text":"Download and install the Redis Insight desktop application."}],"examples":[],"children":[]} diff --git a/build/docs-mcp-server/node/test/mcp-client.mjs b/build/docs-mcp-server/node/test/mcp-client.mjs new file mode 100644 index 0000000000..5010723f0e --- /dev/null +++ b/build/docs-mcp-server/node/test/mcp-client.mjs @@ -0,0 +1,44 @@ +// Manual local test: drive the built stdio server through the real MCP client +// (spawn -> initialize -> tools/list -> tools/call), against the local fixture. +// node test/mcp-client.mjs +import { Client } from "@modelcontextprotocol/sdk/client/index.js"; +import { StdioClientTransport } from "@modelcontextprotocol/sdk/client/stdio.js"; +import { fileURLToPath } from "node:url"; + +const feed = + process.env.DOCS_NDJSON ?? fileURLToPath(new URL("./fixture.ndjson", import.meta.url)); +const serverEntry = fileURLToPath(new URL("../dist/index.js", import.meta.url)); + +const transport = new StdioClientTransport({ + command: "node", + args: [serverEntry], + env: { ...process.env, DOCS_NDJSON: feed }, +}); + +const client = new Client({ name: "local-test-client", version: "0.0.1" }, { capabilities: {} }); +await client.connect(transport); +console.log("connected to server\n"); + +const tools = await client.listTools(); +console.log("tools/list ->", tools.tools.map((t) => t.name).join(", "), "\n"); + +const search = await client.callTool({ + name: "search_docs", + arguments: { query: "append an entry to a stream" }, +}); +console.log("tools/call search_docs('append an entry to a stream'):"); +console.log(search.content[0].text, "\n"); + +const amb = await client.callTool({ name: "get_page", arguments: { id: "install" } }); +console.log(`tools/call get_page(id='install') -> isError=${amb.isError}`); +console.log(amb.content[0].text, "\n"); + +const page = await client.callTool({ + name: "get_page", + arguments: { url: "https://redis.io/docs/latest/commands/xadd/", roles: ["parameters"] }, +}); +console.log(`tools/call get_page(url=.../commands/xadd/, roles=['parameters']) -> isError=${page.isError}`); +console.log(page.content[0].text); + +await client.close(); +console.log("\nclosed cleanly"); diff --git a/build/docs-mcp-server/node/tsconfig.json b/build/docs-mcp-server/node/tsconfig.json new file mode 100644 index 0000000000..01b2904d2f --- /dev/null +++ b/build/docs-mcp-server/node/tsconfig.json @@ -0,0 +1,19 @@ +{ + "compilerOptions": { + "target": "ES2022", + "module": "esnext", + "lib": ["ES2022"], + "outDir": "./dist", + "rootDir": "./src", + "strict": true, + "esModuleInterop": true, + "skipLibCheck": true, + "forceConsistentCasingInFileNames": true, + "resolveJsonModule": true, + "declaration": true, + "sourceMap": true, + "moduleResolution": "node" + }, + "include": ["src/**/*"], + "exclude": ["node_modules", "dist", "src/smoke.ts"] +} diff --git a/build/docs-mcp-server/redis-eval/.gitignore b/build/docs-mcp-server/redis-eval/.gitignore new file mode 100644 index 0000000000..44746405a4 --- /dev/null +++ b/build/docs-mcp-server/redis-eval/.gitignore @@ -0,0 +1,6 @@ +# Regenerable parity artifacts (export_texts.py / embed-dump.mjs) +parity_texts.json +node_vectors.json +__pycache__/ +# Portable vector dump for the Step 4 loader seed (dump_vectors_bin.py) +vecdump/ diff --git a/build/docs-mcp-server/redis-eval/README.md b/build/docs-mcp-server/redis-eval/README.md new file mode 100644 index 0000000000..30436bd804 --- /dev/null +++ b/build/docs-mcp-server/redis-eval/README.md @@ -0,0 +1,177 @@ +# redis-eval β€” hosted-phase prototype (DOC-6809) + +Takes the recipe the offline `vector-eval/` experiment settled on (section-level +bge-small chunks + weighted RRF favouring vector ~2–3Γ—) and proves it on a real +Redis 8 Query Engine backend, one step at a time. Each step isolates a single +source of variance so a divergence can only come from one place. + +Runs against a local **Redis 8** with the `search` module on `localhost:6379` +(verified: Redis 8.8.0, RediSearch 8.8). Reuses `../vector-eval/.venv` (adds +`redis-py`) and the cached `../vector-eval/embeddings-section.npz` β€” no +re-embedding. + +## Step 1 β€” Redis retrieval parity (`parity.py`) βœ… PASS + +**Question:** does RediSearch `FLAT`/`COSINE` KNN + app-layer weighted RRF +reproduce the offline numpy-cosine ranking? Embedding is held constant: BOTH the +numpy reference and the Redis path use the **same cached corpus vectors** and the +**same Python-embedded query vectors**. The only moving part is Redis vs numpy. + +Design: load all 15,300 section chunks into a HASH index with a single indexed +`VECTOR FLAT` field (FLOAT32, DIM 384, COSINE); `owner` (page URL) rides on the +hash and comes back via `RETURN` (unindexed). Per query: KNN over the full chunk +population β†’ dedup to best (nearest) chunk per page β†’ top-50 pages, exactly +mirroring numpy `rank_pages`. Fuse with the dumped lexical ranking via weighted +RRF (v2/v3) and score against the same 35-case eval. + +**Result (2529 pages β†’ 15,300 chunks):** + +| | numpy (offline) | redis (this) | +|---|---|---| +| command MRR (v3) | 0.795 | **0.795** (identical) | +| command recall@1/@5 | 73% / 91% | **73% / 91%** (identical) | +| concept MRR (vector) | 0.638 | 0.599 | +| overall MRR (wrrf v3) | 0.731 | 0.712 | +| top-1 page match | β€” | 34/35 | +| top-50 exact order | β€” | 19/35 | +| KNN latency (K=200) | β€” | **p50 4.1 ms, max 29 ms** | + +**Verdict:** parity holds to tie-breaking. Command metrics are bit-identical. The +single concept-MRR delta traces to ONE query β€” *"connect to Redis from a Python +application"* β€” where two chunks (`redis-py/connect`, `ioredis/connect`) have an +**exactly equal** cosine score (gap `0.00e+00`); numpy's stable argsort and +Redis's internal order break the tie differently. Deep-tail order diverges on +more queries (hence top-50 exact only 19/35) but that rarely moves the first +expected hit (top-1 34/35), so recall@k / MRR are unaffected except at that one +tie. `diag_divergence.py` reproduces the tie evidence. + +Side note (not a Redis issue): those client "connect" pages embed identically +because their section anchor text is templated the same across clients β€” the +embedding can't distinguish redis-py from ioredis for a "Python" query. A +chunking/corpus tuning candidate for later. + +## Step 2 β€” Node ONNX embedding parity βœ… PASS + +**Question:** does embedding in Node (`fastembed` npm, ONNX via `onnxruntime-node`) +reproduce Python `fastembed` vectors closely enough to (a) embed queries +in-process in the Node MCP server and (b) keep the cached Python corpus vectors + +offline numbers valid? `fastembed` npm is the Node port of the same Qdrant +fastembed the Python side uses (same `@anush008/tokenizers`, same HF model files). + +Design: Python exports the exact query + 510-chunk corpus-sample strings +(`export_texts.py` β†’ `parity_texts.json`); Node embeds them with plain `embed()` +prepending the same BGE query prefix (`node/test/embed-dump.mjs` β†’ +`node_vectors.json`); Python re-embeds the identical strings and compares +(`embed_parity.py`). Using plain `embed()` (not `queryEmbed`/`passageEmbed`) keeps +the only variable the ONNX embedding itself, not prefix wording. + +**Result:** + +| check | result | +|---|---| +| cosine(node, python), 35 queries | min/p50/mean **1.00000** | +| cosine(node, python), 510 corpus chunks | min/p50/mean **1.00000** | +| texts below 0.999 cosine | **0 / 545** | +| eval, pure-vector MRR (Node vs Python queries) | identical (.717 / .763 / .638) | +| eval, wrrf-v3 command MRR | .795 β†’ .789 | + +**Verdict:** the Node embeddings are identical to Python's (cosine 1.0), so +embedding in-process in the Node server is safe and the cached corpus vectors +remain valid. The lone wrrf-v3 delta (command MRR .795β†’.789, one query) is the +same tie-break sensitivity seen in Step 1: cosine-1.0 is not bit-identical, so a +couple of near-tied corpus chunks reorder in the tail and RRF β€” sensitive to +exact rank positions β€” flips one fused rank, even though the vector-only metric +is unchanged. Noise at n=22, not an embedding discrepancy. + +Node embedding ran ~4 texts/s here (unoptimised local ONNX, same caveat as the +Python side β€” not a production latency signal; the Step 1 KNN latency is). + +## Step 3 β€” Redis-native FT.HYBRID vs app-layer weighted RRF βœ… (resolves the fork) + +**Question:** does Redis's native `FT.HYBRID` (8.4.4+) reproduce our validated +app-layer weighted-RRF recipe (vector ~3Γ— lexical, overall MRR .73)? + +Two structural facts, found in the command spec: +- `COMBINE RRF` exposes only `CONSTANT` + `WINDOW` β€” **no per-retriever weights**. + Native RRF is equal-weight, the variant the fusion sweep already found dilutes + the top ranks. +- `COMBINE LINEAR` takes `ALPHA`/`BETA` but fuses raw **scores** linearly β€” a + different algorithm from our rank-based weighted RRF. +- Native hybrid also uses Redis's **own** BM25 over an indexed TEXT field, not our + Porter-stemmed / field-boosted Node lexical (`lexical.json`). + +`native_hybrid.py` decomposes both axes (fusion + lexical) on the 35-case eval. +Working `FT.HYBRID` invocation (gotchas noted for reuse): +`FT.HYBRID <idx> SEARCH "<terms>" SCORER BM25 VSIM @vec $qv KNN 2 K 200 +[COMBINE RRF 2 CONSTANT 60 | COMBINE LINEAR 4 ALPHA a BETA b] LOAD 1 @owner +LIMIT 0 200 PARAMS 2 qv <blob>` β€” `VSIM @field $param`; KNN/COMBINE counts are +**k/v-pair counts** not the k value; **no `DIALECT`**; project with `LOAD`, not +`RETURN`. + +**Results (overall MRR):** + +| system | overall | command | concept | +|---|---|---|---| +| redis bm25 only | 0.117 | 0.032 | 0.262 | +| our lexical only | 0.530 | 0.571 | 0.461 | +| native rrf (equal-weight) | 0.425 | 0.448 | 0.387 | +| native linear Ξ±.2/Ξ².8 | 0.501 | 0.527 | 0.458 | +| app wrrf, Redis BM25 + vec | 0.431 | 0.396 | 0.491 | +| **app wrrf, our lexical + vec (recipe)** | **0.731** | **0.795** | **0.621** | + +**Verdict β€” resolves the "showcase vs control" fork toward app-layer fusion:** +native `FT.HYBRID` does **not** reproduce the recipe, for two independent reasons. +(1) Its RRF is equal-weight-only; LINEAR weights but underperforms weighted-RRF. +(2) More decisively, its lexical side (Redis raw BM25, MRR .117; **0% command +recall@5**) is far weaker than our Node ranker (.530) β€” command queries like +"append an entry to a stream" carry no `xadd` token, so a plain body-text BM25 +ranks streams *tutorials* above the *XADD command page*; our title/slug field +boosts + page-type weighting are what fix that, and a plain Redis TEXT index +doesn't carry that signal. A subtle corollary: RRF gives every list fixed +rank-reciprocal mass regardless of quality, so it injects a bad retriever's +distractors (hence app-wrrf-over-Redis-BM25 .431 < native-linear .501); RRF wins +only when both retrievers are good, which they are with our Node lexical (.731). + +**Architecture conclusion (matches SPEC Β§6):** use **Redis for vector KNN** +(proven in Step 1, ~4 ms), keep **lexical BM25 + weighted-RRF fusion in the app +(Node) layer** β€” where the lexical path already has to live for the stdio +no-datastore mode. `FT.HYBRID`'s all-in-Redis showcase costs ~0.23 MRR and can't +express the weighted fusion, so it's not the path. + +*Honest caveat:* native BM25's showing is depressed partly by a deliberately +naive OR-of-terms query and by indexing only the section body (no boosted +title/slug fields). A Redis index mirroring the Node analyzer would narrow the +lexical gap β€” but that is re-implementing our lexical ranker inside Redis, with +native RRF still unable to do the weighted fusion. The conclusion holds either +way. (The eval's vector side here is numpy `rank_pages`; Step 1 already showed +Redis KNN β‰ˆ numpy to tie-breaking, so this isolates fusion + lexical cleanly.) + +## Step 4 β€” wired into the MCP server βœ… (vertical slice) + +Implemented in `../node/` behind a `REDIS_URL` feature flag: hybrid `search_docs` += app-side BM25 + Redis vector KNN, fused with weighted RRF (vector 3Γ—), query +embedded in-process via fastembed-js. Loader `node/scripts/load-index.mjs` builds +the `docs_vec` FLAT/COSINE index (embeds with fastembed-js, or `--vectors +<dir>` to seed from `dump_vectors_bin.py` output β€” used here to skip a ~1h local +re-embed, valid because Step 2 proved Node ≑ Python vectors). + +Live eval through the real `search_docs` tool (`npm run eval:hybrid`): overall +MRR **.704**, command **.783**, concept **.570** (concept @10 **100%**) β€” within +tie-break noise of the offline .731 (stacks the Step 1 KNN + Step 2 embedding +tie-breaks). Lexical-only default is unchanged (overall .525) and smoke passes. +Payoff: hybrid lifts overall MRR **.53 β†’ .70**, concept @10 **77% β†’ 100%**. See +`../node/README.md` (Hybrid mode). Stopped here for review β€” no HTTP endpoint / +rate-limiting / deploy yet. + +## Run + +``` +../vector-eval/.venv/bin/python parity.py section # Step 1 +../vector-eval/.venv/bin/python diag_divergence.py section + +../vector-eval/.venv/bin/python export_texts.py # Step 2 +(cd ../node && node test/embed-dump.mjs) # (downloads model 1st run) +../vector-eval/.venv/bin/python embed_parity.py section + +../vector-eval/.venv/bin/python native_hybrid.py section # Step 3 +``` diff --git a/build/docs-mcp-server/redis-eval/diag_divergence.py b/build/docs-mcp-server/redis-eval/diag_divergence.py new file mode 100644 index 0000000000..a2b4eacf21 --- /dev/null +++ b/build/docs-mcp-server/redis-eval/diag_divergence.py @@ -0,0 +1,44 @@ +#!/usr/bin/env python3 +"""Diagnostic: for any query whose Redis top-1 page != numpy top-1 page, print +the numpy top-3 chunk cosine scores. If the top-2 gap is ~1e-5 or less, the +divergence is FLOAT32 tie-break noise (summation order), not a ranking defect. +""" +import os +import sys +import json + +import numpy as np +import redis +from redis.commands.search.query import Query + +HERE = os.path.dirname(os.path.abspath(__file__)) +sys.path.insert(0, os.path.join(HERE, "..", "vector-eval")) +import eval_vector as ev # noqa: E402 +import parity # noqa: E402 + +cases = json.load(open(ev.LEXICAL, encoding="utf-8")) +pages = ev.load_pages() +texts, owners = ev.build_chunks(pages, ev.MODE) +emb, owners = ev.get_corpus_embeddings(texts, owners) +qvecs = ev.embed_batch(ev._model(), [c["q"] for c in cases], is_query=True) + +r = redis.Redis(host="localhost", port=6379, decode_responses=False) +parity.build_index(r, emb.shape[1]) +parity.load_chunks(r, emb, owners) + +for c, qv in zip(cases, qvecs): + np_pages = ev.rank_pages(qv, emb, owners) + rd_pages = parity.redis_knn_pages(r, qv, k=emb.shape[0]) + if np_pages[0] == rd_pages[0]: + continue + sims = emb @ qv + order = np.argsort(-sims)[:3] + print(f"\nQUERY ({c['kind']}): {c['q']}") + print(f" numpy top-1 page: {np_pages[0]}") + print(f" redis top-1 page: {rd_pages[0]}") + print(" numpy top-3 chunk cosine scores:") + for j in order: + print(f" {sims[j]:.8f} {owners[j]}") + print(f" top-1 vs top-2 gap: {sims[order[0]] - sims[order[1]]:.2e}") + +r.ft(parity.INDEX).dropindex(delete_documents=True) diff --git a/build/docs-mcp-server/redis-eval/dump_vectors_bin.py b/build/docs-mcp-server/redis-eval/dump_vectors_bin.py new file mode 100644 index 0000000000..551903d3fe --- /dev/null +++ b/build/docs-mcp-server/redis-eval/dump_vectors_bin.py @@ -0,0 +1,36 @@ +#!/usr/bin/env python3 +"""Dump the cached bge-small section vectors to a portable binary the Node +build-index loader can seed from (--vectors), so the vertical-slice eval need +not re-embed 15k chunks in Node at ~4/s. Step 2 proved Node fastembed-js == these +Python vectors (cosine 1.0), so seeding is equivalent to the Node embed path. + +Writes to redis-eval/vecdump/: + meta.json {n, dim} + owners.json [url, ...] (aligned to rows) + vectors.f32 raw little-endian float32, n*dim + + ../vector-eval/.venv/bin/python dump_vectors_bin.py section +""" +import json +import os +import sys + +import numpy as np + +HERE = os.path.dirname(os.path.abspath(__file__)) +sys.path.insert(0, os.path.join(HERE, "..", "vector-eval")) +import eval_vector as ev # noqa: E402 + +pages = ev.load_pages() +texts, owners = ev.build_chunks(pages, ev.MODE) +emb, owners = ev.get_corpus_embeddings(texts, owners) +emb = np.ascontiguousarray(emb, dtype="<f4") # little-endian float32 + +out = os.path.join(HERE, "vecdump") +os.makedirs(out, exist_ok=True) +emb.tofile(os.path.join(out, "vectors.f32")) +with open(os.path.join(out, "owners.json"), "w", encoding="utf-8") as f: + json.dump([str(o) for o in owners], f) +with open(os.path.join(out, "meta.json"), "w", encoding="utf-8") as f: + json.dump({"n": int(emb.shape[0]), "dim": int(emb.shape[1])}, f) +print(f"wrote {out}: n={emb.shape[0]} dim={emb.shape[1]}") diff --git a/build/docs-mcp-server/redis-eval/embed_parity.py b/build/docs-mcp-server/redis-eval/embed_parity.py new file mode 100644 index 0000000000..596e43abc8 --- /dev/null +++ b/build/docs-mcp-server/redis-eval/embed_parity.py @@ -0,0 +1,100 @@ +#!/usr/bin/env python3 +"""Step 2c: compare Node fastembed-js vectors against Python fastembed on the +SAME strings, and confirm the eval is stable when queries are embedded in Node. + +Two checks: + 1. Per-text cosine(node_vec, python_vec) for the 35 queries + corpus sample. + Near-1.0 means the Node ONNX path reproduces Python embeddings (so the + cached Python corpus vectors and the offline numbers stay valid). + 2. Re-run the 35-case eval with Node-embedded QUERY vectors against the cached + Python corpus (the most likely near-term wiring: build offline in Python, + embed queries in Node at request time). Metrics should match Step 1. + +Run export_texts.py + node test/embed-dump.mjs first. + ../vector-eval/.venv/bin/python embed_parity.py section +""" +import json +import os +import sys + +import numpy as np + +HERE = os.path.dirname(os.path.abspath(__file__)) +sys.path.insert(0, os.path.join(HERE, "..", "vector-eval")) +import eval_vector as ev # noqa: E402 + + +def cosine(a, b): + a = np.asarray(a, dtype=np.float32) + b = np.asarray(b, dtype=np.float32) + return float(a @ b / ((np.linalg.norm(a) * np.linalg.norm(b)) + 1e-12)) + + +def main(): + texts = json.load(open(os.path.join(HERE, "parity_texts.json"), encoding="utf-8")) + node = json.load(open(os.path.join(HERE, "node_vectors.json"), encoding="utf-8")) + prefix = texts["query_prefix"] + + # Python embeddings of the identical strings. + py_q = ev.embed_batch(ev._model(), texts["queries"], is_query=True) + py_c = ev.embed_batch(ev._model(), texts["corpus"], is_query=False) + + q_cos = [cosine(n, p) for n, p in zip(node["queries"], py_q)] + c_cos = [cosine(n, p) for n, p in zip(node["corpus"], py_c)] + + def stat(name, xs): + xs = sorted(xs) + print(f" {name:14} min {xs[0]:.5f} p50 {xs[len(xs)//2]:.5f} " + f"mean {sum(xs)/len(xs):.5f} (n={len(xs)})") + + print(f"Node dim {node['dim']} vs Python dim {py_q.shape[1]}") + print("Per-text cosine(node, python):") + stat("queries", q_cos) + stat("corpus", c_cos) + below = sum(1 for x in q_cos + c_cos if x < 0.999) + print(f" texts below 0.999 cosine: {below}/{len(q_cos)+len(c_cos)}") + + # --- Eval stability: Node query vectors vs cached Python corpus --- + cases = json.load(open(ev.LEXICAL, encoding="utf-8")) + pages = ev.load_pages() + corpus_texts, owners = ev.build_chunks(pages, ev.MODE) + emb, owners = ev.get_corpus_embeddings(corpus_texts, owners) + node_q = np.asarray(node["queries"], dtype=np.float32) + + groups = ["overall", "command", "concept"] + + def run(qvecs, label): + systems = { + "vector": lambda lex, vec: vec, + "wrrf v3": lambda lex, vec: _wrrf([(vec, 3), (lex, 1)]), + } + data = {s: {g: [] for g in groups} for s in systems} + for c, qv in zip(cases, qvecs): + exp = set(c["expected"]) + vec = ev.rank_pages(qv, emb, owners) + for s, fn in systems.items(): + rk = ev.best_rank(fn(c["lexical"], vec), exp) + data[s]["overall"].append(rk) + data[s][c["kind"]].append(rk) + print(f"\n### {label} ###") + for g in groups: + print(f"=== {g} (n={len(data['vector'][g])}) === @1/@3/@5/@10 | MRR") + for s in data: + rec, mrr = ev.metrics(data[s][g]) + cells = " / ".join(f"{rec[k]*100:3.0f}%" for k in ev.KS) + print(f" {s:8} {cells} | {mrr:.3f}") + + run(py_q, "Python query vectors (baseline)") + run(node_q, "Node query vectors (this prototype)") + + +def _wrrf(lists_weights, k=60, topn=50): + scores = {} + for lst, w in lists_weights: + for rank, u in enumerate(lst): + scores[u] = scores.get(u, 0.0) + w / (k + rank + 1) + return [u for u, _ in sorted(scores.items(), key=lambda kv: -kv[1])][:topn] + + +if __name__ == "__main__": + main() diff --git a/build/docs-mcp-server/redis-eval/export_texts.py b/build/docs-mcp-server/redis-eval/export_texts.py new file mode 100644 index 0000000000..577729968e --- /dev/null +++ b/build/docs-mcp-server/redis-eval/export_texts.py @@ -0,0 +1,34 @@ +#!/usr/bin/env python3 +"""Step 2a: export the EXACT strings both embedders must see, so Node vs Python +embedding parity has zero input drift. Writes parity_texts.json: + { "query_prefix": "...", "queries": [...35 raw query strings...], + "corpus": [...sampled chunk texts...] } +The BGE query prefix is emitted here and applied identically on both sides. +""" +import json +import os +import sys + +HERE = os.path.dirname(os.path.abspath(__file__)) +sys.path.insert(0, os.path.join(HERE, "..", "vector-eval")) +import eval_vector as ev # noqa: E402 + +cases = json.load(open(ev.LEXICAL, encoding="utf-8")) +pages = ev.load_pages() +texts, owners = ev.build_chunks(pages, ev.MODE) + +# Spread a ~500-chunk sample across the corpus (deterministic stride). +stride = max(1, len(texts) // 500) +sample_idx = list(range(0, len(texts), stride)) +corpus_sample = [texts[i] for i in sample_idx] + +out = { + "query_prefix": ev.BGE_QUERY_PREFIX, + "queries": [c["q"] for c in cases], + "corpus": corpus_sample, + "corpus_idx": sample_idx, # so the comparator can align to cached vectors +} +path = os.path.join(HERE, "parity_texts.json") +with open(path, "w", encoding="utf-8") as f: + json.dump(out, f) +print(f"wrote {path}: {len(out['queries'])} queries, {len(corpus_sample)} corpus chunks") diff --git a/build/docs-mcp-server/redis-eval/native_hybrid.py b/build/docs-mcp-server/redis-eval/native_hybrid.py new file mode 100644 index 0000000000..7f92de28a2 --- /dev/null +++ b/build/docs-mcp-server/redis-eval/native_hybrid.py @@ -0,0 +1,191 @@ +#!/usr/bin/env python3 +"""Step 3: does Redis-native FT.HYBRID reproduce our validated app-layer +weighted-RRF recipe (vector ~3x lexical, overall MRR .73)? + +FT.HYBRID (Redis 8.4.4+) fuses a text search and a vector search server-side. +Two findings drive the design: + - COMBINE RRF exposes only CONSTANT + WINDOW β€” NO per-retriever weights. Native + RRF is therefore EQUAL-weight, the exact variant our fusion sweep found + dilutes the top ranks (.69 vs .73). + - COMBINE LINEAR takes ALPHA/BETA weights but fuses raw SCORES linearly, a + different algorithm from our rank-based weighted RRF. +Also, native hybrid uses Redis's OWN BM25 over an indexed TEXT field, not our +Porter-stemmed/boosted Node BM25 (lexical.json). So "native" differs on two axes: +fusion AND lexical signal. This harness decomposes both: + + native RRF Redis BM25 + Redis KNN, native equal-weight RRF + native LINEAR a/b Redis BM25 + Redis KNN, native weighted score fusion + app wRRF (redis) Redis BM25 + Redis KNN, OUR weighted-rank RRF [isolates fusion] + app wRRF (ours) lexical.json + Redis KNN, our recipe (= Step 1) [isolates lexical] + + ../vector-eval/.venv/bin/python native_hybrid.py section +""" +import json +import os +import re +import sys + +import numpy as np +import redis + +HERE = os.path.dirname(os.path.abspath(__file__)) +sys.path.insert(0, os.path.join(HERE, "..", "vector-eval")) +import eval_vector as ev # noqa: E402 + +INDEX = "docs_hybrid" +PREFIX = "hybrid:chunk:" +KNN_K = 200 +TOPN = 50 + + +def connect(): + r = redis.Redis(host="localhost", port=6379, decode_responses=False) + r.ping() + return r + + +def build_index(r, dim): + try: + r.execute_command("FT.DROPINDEX", INDEX, "DD") + except redis.ResponseError: + pass + r.execute_command( + "FT.CREATE", INDEX, "ON", "HASH", "PREFIX", "1", PREFIX, "SCHEMA", + "text", "TEXT", + "owner", "TAG", + "vec", "VECTOR", "FLAT", "6", + "TYPE", "FLOAT32", "DIM", dim, "DISTANCE_METRIC", "COSINE", + ) + + +def load_chunks(r, texts, emb, owners): + pipe = r.pipeline(transaction=False) + for i, (t, v, o) in enumerate(zip(texts, emb, owners)): + pipe.hset(f"{PREFIX}{i}", mapping={ + "text": t, "owner": o, + "vec": np.asarray(v, dtype=np.float32).tobytes(), + }) + if i % 2000 == 0: + pipe.execute() + pipe.execute() + + +def or_query(q): + """Natural-language query -> BM25-friendly OR of alphanumeric terms.""" + terms = re.findall(r"[a-z0-9]+", q.lower()) + return " | ".join(terms) if terms else "*" + + +def _dedup_pages(rows): + best = [] + seen = set() + for owner in rows: + if owner not in seen: + seen.add(owner) + best.append(owner) + if len(best) >= TOPN: + break + return best + + +def _owner(doc): + o = doc.get(b"owner") if isinstance(doc, dict) else None + return o.decode() if isinstance(o, bytes) else o + + +def native_hybrid_pages(r, q, blob, combine): + args = ["FT.HYBRID", INDEX, "SEARCH", or_query(q), "SCORER", "BM25", + "VSIM", "@vec", "$qv", "KNN", "2", "K", str(KNN_K)] + args += combine + args += ["LOAD", "1", "@owner", "LIMIT", "0", str(KNN_K), + "PARAMS", "2", "qv", blob] + res = r.execute_command(*args) + rows = res[b"results"] if isinstance(res, dict) else res.get("results", []) + return _dedup_pages([_owner(d) for d in rows]) + + +def redis_bm25_pages(r, q): + res = r.ft(INDEX).search(_bm25_query(or_query(q))) + return _dedup_pages([_owner_obj(d) for d in res.docs]) + + +def _bm25_query(text): + from redis.commands.search.query import Query + return Query(text).scorer("BM25").return_fields("owner").paging(0, KNN_K) + + +def _owner_obj(doc): + o = getattr(doc, "owner", None) + return o.decode() if isinstance(o, bytes) else o + + +def main(): + cases = json.load(open(ev.LEXICAL, encoding="utf-8")) + pages = ev.load_pages() + texts, owners = ev.build_chunks(pages, ev.MODE) + emb, owners = ev.get_corpus_embeddings(texts, owners) + dim = emb.shape[1] + print(f"{len(pages)} pages -> {len(texts)} chunks, dim {dim}") + + r = connect() + build_index(r, dim) + load_chunks(r, texts, emb, owners) + print("index built + loaded") + + qvecs = ev.embed_batch(ev._model(), [c["q"] for c in cases], is_query=True) + + systems = ["redis bm25 only", "our lexical only", + "native rrf", "native lin .2/.8", "native lin .1/.9", + "app wrrf (redis)", "app wrrf (ours)"] + groups = ["overall", "command", "concept"] + data = {s: {g: [] for g in groups} for s in systems} + + for c, qv in zip(cases, qvecs): + exp = set(c["expected"]) + blob = np.asarray(qv, dtype=np.float32).tobytes() + + rankings = { + "native rrf": native_hybrid_pages( + r, c["q"], blob, ["COMBINE", "RRF", "2", "CONSTANT", "60"]), + "native lin .2/.8": native_hybrid_pages( + r, c["q"], blob, + ["COMBINE", "LINEAR", "4", "ALPHA", "0.2", "BETA", "0.8"]), + "native lin .1/.9": native_hybrid_pages( + r, c["q"], blob, + ["COMBINE", "LINEAR", "4", "ALPHA", "0.1", "BETA", "0.9"]), + } + # app-layer fusions: same Redis KNN, two different lexical sources + redis_vec = ev.rank_pages(qv, emb, owners) # matches Step 1 vector side + redis_bm25 = redis_bm25_pages(r, c["q"]) + rankings["redis bm25 only"] = redis_bm25 + rankings["our lexical only"] = c["lexical"] + rankings["app wrrf (redis)"] = _wrrf([(redis_vec, 3), (redis_bm25, 1)]) + rankings["app wrrf (ours)"] = _wrrf([(redis_vec, 3), (c["lexical"], 1)]) + + for s in systems: + rk = ev.best_rank(rankings[s], exp) + data[s]["overall"].append(rk) + data[s][c["kind"]].append(rk) + + for g in groups: + n = len(data[systems[0]][g]) + print(f"\n=== {g} (n={n}) === recall@1 / @3 / @5 / @10 | MRR") + for s in systems: + rec, mrr = ev.metrics(data[s][g]) + cells = " / ".join(f"{rec[k]*100:3.0f}%" for k in ev.KS) + print(f" {s:18} {cells} | {mrr:.3f}") + + r.execute_command("FT.DROPINDEX", INDEX, "DD") + print("\ndropped index; done.") + + +def _wrrf(lists_weights, k=60, topn=TOPN): + scores = {} + for lst, w in lists_weights: + for rank, u in enumerate(lst): + scores[u] = scores.get(u, 0.0) + w / (k + rank + 1) + return [u for u, _ in sorted(scores.items(), key=lambda kv: -kv[1])][:topn] + + +if __name__ == "__main__": + main() diff --git a/build/docs-mcp-server/redis-eval/parity.py b/build/docs-mcp-server/redis-eval/parity.py new file mode 100644 index 0000000000..f329c01129 --- /dev/null +++ b/build/docs-mcp-server/redis-eval/parity.py @@ -0,0 +1,197 @@ +#!/usr/bin/env python3 +""" +Step 1 of the hosted-phase prototype (DOC-6809): Redis retrieval parity. + +Question this isolates: does RediSearch FLAT/COSINE KNN + app-layer weighted RRF +reproduce the OFFLINE numpy-cosine eval ranking (overall MRR ~.73, concept @10 +100%)? To keep embedding out of the picture entirely (that's Step 2 β€” Node ONNX +parity), BOTH the numpy reference path and the Redis path use the SAME cached +bge-small corpus vectors (embeddings-section.npz) and the SAME Python-embedded +query vectors. The only moving part here is Redis vs numpy. + +What it does: + 1. Reuse eval_vector to build the section chunks + load cached corpus vectors. + 2. Load every chunk into a RediSearch HASH index (FLAT, COSINE, FLOAT32, 384). + 3. Embed the 35 eval queries once with Python fastembed (the reference vectors). + 4. Per query: Redis KNN over all chunks -> dedup to best-per-page -> top-50 pages. + Compare that page ranking against numpy rank_pages on the identical vector. + 5. Fuse each ranking with the dumped lexical ranking via weighted RRF and score + recall@k / MRR, side by side numpy-vector vs redis-vector. + 6. Report KNN latency at a realistic K (separate from the full-K parity query). + +Usage: + ../vector-eval/.venv/bin/python parity.py section +Requires: local Redis 8 with search (localhost:6379), cached embeddings-section.npz. +""" +import os +import sys +import time + +import numpy as np +import redis +from redis.commands.search.field import VectorField +from redis.commands.search.index_definition import IndexDefinition, IndexType +from redis.commands.search.query import Query + +HERE = os.path.dirname(os.path.abspath(__file__)) +sys.path.insert(0, os.path.join(HERE, "..", "vector-eval")) +import eval_vector as ev # noqa: E402 (MODE/CACHE resolve from argv[1], default "section") + +INDEX = "docs_parity" +PREFIX = "parity:chunk:" +REALISTIC_K = 200 # for the latency figure; parity uses full-K + + +def connect(): + r = redis.Redis(host="localhost", port=6379, decode_responses=False) + r.ping() + return r + + +def build_index(r, dim): + try: + r.ft(INDEX).dropindex(delete_documents=True) + except redis.ResponseError: + pass + # Only the vector is indexed; `owner` lives on the hash and is pulled back + # via RETURN at query time (RediSearch returns unindexed hash fields fine). + schema = ( + VectorField( + "vector", + "FLAT", + {"TYPE": "FLOAT32", "DIM": dim, "DISTANCE_METRIC": "COSINE"}, + ), + ) + r.ft(INDEX).create_index( + schema, + definition=IndexDefinition(prefix=[PREFIX], index_type=IndexType.HASH), + ) + + +def load_chunks(r, emb, owners): + pipe = r.pipeline(transaction=False) + for i, (vec, owner) in enumerate(zip(emb, owners)): + pipe.hset( + f"{PREFIX}{i}", + mapping={ + "vector": np.asarray(vec, dtype=np.float32).tobytes(), + "owner": owner, + }, + ) + if i % 2000 == 0: + pipe.execute() + pipe.execute() + + +def redis_knn_pages(r, qvec, k, topn=50): + """KNN over `k` chunks, dedup to best (nearest) page, return top-n page urls.""" + blob = np.asarray(qvec, dtype=np.float32).tobytes() + q = ( + Query(f"*=>[KNN {k} @vector $blob AS score]") + .sort_by("score") # COSINE distance, ascending = nearest first + .return_fields("owner", "score") + .paging(0, k) + .dialect(2) + ) + res = r.ft(INDEX).search(q, query_params={"blob": blob}) + best = {} + for doc in res.docs: + owner = doc.owner.decode() if isinstance(doc.owner, bytes) else doc.owner + if owner not in best: # docs are distance-sorted, so first = nearest + best[owner] = float(doc.score) + if len(best) >= topn: + break + return list(best.keys()) + + +def main(): + import json + + cases = json.load(open(ev.LEXICAL)) + print(f"Mode: {ev.MODE}. Building chunks + loading cached corpus vectors ...") + pages = ev.load_pages() + texts, owners = ev.build_chunks(pages, ev.MODE) + emb, owners = ev.get_corpus_embeddings(texts, owners) # cache hit, no re-embed + total, dim = emb.shape + print(f" {len(pages)} pages -> {total} chunks, dim {dim}") + + print("Connecting to Redis + (re)building FLAT index ...") + r = connect() + build_index(r, dim) + t0 = time.time() + load_chunks(r, emb, owners) + print(f" loaded {total} chunks in {time.time()-t0:.1f}s") + + print("Embedding 35 queries with Python fastembed (reference vectors) ...") + qvecs = ev.embed_batch(ev._model(), [c["q"] for c in cases], is_query=True) + + # --- Parity: Redis full-K KNN vs numpy rank_pages on identical vectors --- + print("\nParity check: Redis KNN page ranking vs numpy rank_pages ...") + identical, top1_match, latencies = 0, 0, [] + redis_rank_by_case = [] + for c, qv in zip(cases, qvecs): + np_pages = ev.rank_pages(qv, emb, owners) # numpy reference (all chunks) + rd_pages = redis_knn_pages(r, qv, k=total) # Redis, full-K = same population + redis_rank_by_case.append(rd_pages) + if np_pages[:50] == rd_pages[:50]: + identical += 1 + if np_pages and rd_pages and np_pages[0] == rd_pages[0]: + top1_match += 1 + # realistic-K latency (not the full-K parity query) + t = time.time() + redis_knn_pages(r, qv, k=REALISTIC_K) + latencies.append((time.time() - t) * 1000) + n = len(cases) + print(f" top-50 page order identical: {identical}/{n}") + print(f" top-1 page match: {top1_match}/{n}") + lat = sorted(latencies) + print(f" KNN K={REALISTIC_K} latency: p50 {lat[n//2]:.1f}ms max {lat[-1]:.1f}ms") + + # --- Metrics: numpy-vector vs redis-vector, each fused with lexical --- + def score(vec_ranker): + groups = ["overall", "command", "concept"] + systems = { + "vector": lambda lex, vec: vec, + "wrrf v2": lambda lex, vec: _wrrf([(vec, 2), (lex, 1)]), + "wrrf v3": lambda lex, vec: _wrrf([(vec, 3), (lex, 1)]), + } + data = {s: {g: [] for g in groups} for s in systems} + for c, qv in zip(cases, qvecs): + exp = set(c["expected"]) + lex = c["lexical"] + vec = vec_ranker(c, qv) + for s, fn in systems.items(): + rk = ev.best_rank(fn(lex, vec), exp) + data[s]["overall"].append(rk) + data[s][c["kind"]].append(rk) + return data + + def report(title, data): + print(f"\n### {title} ###") + for g in ["overall", "command", "concept"]: + m = len(data["vector"][g]) + print(f"\n=== {g} (n={m}) === recall@1 / @3 / @5 / @10 | MRR") + for s in data: + rec, mrr = ev.metrics(data[s][g]) + cells = " / ".join(f"{rec[k]*100:3.0f}%" for k in ev.KS) + print(f" {s:8} {cells} | {mrr:.3f}") + + np_idx = {id(c): ev.rank_pages(qv, emb, owners) for c, qv in zip(cases, qvecs)} + report("numpy vector (offline reference)", score(lambda c, qv: np_idx[id(c)])) + rd_idx = {id(c): rd for c, rd in zip(cases, redis_rank_by_case)} + report("redis vector (this prototype)", score(lambda c, qv: rd_idx[id(c)])) + + r.ft(INDEX).dropindex(delete_documents=True) + print("\nDropped index; done.") + + +def _wrrf(lists_weights, k=60, topn=50): + scores = {} + for lst, w in lists_weights: + for rank, u in enumerate(lst): + scores[u] = scores.get(u, 0.0) + w / (k + rank + 1) + return [u for u, _ in sorted(scores.items(), key=lambda kv: -kv[1])][:topn] + + +if __name__ == "__main__": + main() diff --git a/build/docs-mcp-server/redis-eval/requirements.txt b/build/docs-mcp-server/redis-eval/requirements.txt new file mode 100644 index 0000000000..6d85d73e8b --- /dev/null +++ b/build/docs-mcp-server/redis-eval/requirements.txt @@ -0,0 +1,4 @@ +# Reuses ../vector-eval/.venv (fastembed + numpy). Adds the Redis client. +# Local Redis 8 with the `search` module must be reachable on localhost:6379. +-r ../vector-eval/requirements.txt +redis>=8.0 diff --git a/build/docs-mcp-server/vector-eval/.gitignore b/build/docs-mcp-server/vector-eval/.gitignore new file mode 100644 index 0000000000..b741d0feb6 --- /dev/null +++ b/build/docs-mcp-server/vector-eval/.gitignore @@ -0,0 +1,5 @@ +.venv/ +__pycache__/ +embeddings*.npz +lexical.json +*.log diff --git a/build/docs-mcp-server/vector-eval/README.md b/build/docs-mcp-server/vector-eval/README.md new file mode 100644 index 0000000000..4661a73482 --- /dev/null +++ b/build/docs-mcp-server/vector-eval/README.md @@ -0,0 +1,68 @@ +# vector-eval β€” measure-first vector search experiment + +Answers "does vector / hybrid retrieval beat the tuned lexical ranker, and is it +worth the hosted RediSearch infra?" **before** building any of it. No Redis: it +embeds the corpus with an open model (bge-small-en-v1.5 via fastembed/ONNX), +ranks in numpy, and scores against the **same 35-case eval** the lexical harness +uses (`../node/test/eval/cases.json`). + +## Run + +```bash +python3 -m venv .venv && .venv/bin/pip install -r requirements.txt +node ../node/test/eval/dump-lexical.mjs > lexical.json # real lexical rankings +.venv/bin/python eval_vector.py section # or: page +``` + +Embedding is CHECKPOINTED to `embeddings-<mode>.npz` every 1024 chunks (a +crash/kill resumes, doesn't restart). Everything except the scripts is +gitignored (venv, caches, `lexical.json`). Note: embedding runs slowly here +(~6–11 chunks/s on unoptimised CPU ONNX) β€” **not** a production latency signal. + +## Results (recall@5 / MRR) + +| group | | lexical | vector | hybrid (RRF) | +|---|---|---|---|---| +| overall | page-level | 69 / .53 | 74 / .60 | **86 / .66** | +| overall | section-level | 69 / .53 | **83 / .72** | 91 / .67 | +| command | section-level | 73 / .57 | 91 / **.76** | 91 / .76 | +| concept | section-level | 62 / .46 | 69 / **.64** | **92** / .53 | + +(section-level concept: hybrid @10 = 100%, vector @10 = 85%.) + +## Findings + +1. **Section-level chunking (feed `sections[]`) is the big win** β€” especially + for concept queries (vector MRR .47 β†’ .64). Coarse page-level chunks were + what held vector back. +2. **With strong section-level embeddings, vanilla equal-weight RRF hurts the + top ranks**: pure vector is best by MRR / @1–@3, hybrid is best by recall@5/@10. + Fusing a strong retriever with a weaker one dilutes its confident top hits. +3. **Direction:** build section-level embeddings + **weighted RRF favouring + vector ~2–3Γ—** (`fusion_sweep.py`). Weighted RRF recovers the top-1 precision + that equal-weight RRF lost while keeping top-k recall β€” overall MRR .73 (vs + .72 pure vector, .69 equal RRF), concept @5 92% / @10 100%. The dilution was + *equal* weighting, not fusion per se. See `../SPEC.md` Β§6/Β§10. + +## Fusion sweep (section-level, cached embeddings) + +`python fusion_sweep.py section` β€” resolves which fusion to build: + +| system | overall MRR | command MRR | concept @5 / @10 | +|---|---|---|---| +| lexical | .53 | .57 | 62 / 77 | +| pure vector | .72 | .76 | 69 / 85 | +| equal RRF (1:1) | .69 | .76 | 92 / 100 | +| **weighted RRF (3:1)** | **.73** | **.80** | **92 / 100** | + +## Model comparison (bge-small vs bge-base) + +`EMBED_MODEL=BAAI/bge-base-en-v1.5 python eval_vector.py section` (per-model +cache). At weighted RRF 3:1: bge-base overall MRR .74 vs bge-small .73, +command .82 vs .80, concept .61 vs .62 β€” **recall@5 identical** (91/91/92). +The only real gain is command (already strong); concept (the weak spot) is +flat. **Kept bge-small**: bge-base costs ~2Γ— vector size + compute (and embeds +at half the speed here) for a rounding-error overall gain. + +Limitations: small eval (35 cases, 13 concept β€” directional, not definitive); +sections truncated to ~1200 chars. diff --git a/build/docs-mcp-server/vector-eval/eval_vector.py b/build/docs-mcp-server/vector-eval/eval_vector.py new file mode 100644 index 0000000000..b16599350f --- /dev/null +++ b/build/docs-mcp-server/vector-eval/eval_vector.py @@ -0,0 +1,224 @@ +#!/usr/bin/env python3 +""" +Measure-first vector-search experiment for the docs MCP server (no Redis). + +Compares three rankers on the same 35-case eval used by the lexical harness: + - lexical : the production Node ranker's output (read from lexical.json) + - vector : bge-small-en-v1.5 embeddings, best-chunk cosine, ranked in numpy + - hybrid : reciprocal-rank fusion of lexical + vector + +Chunk mode (argv[1], default "section"): + - page : one chunk/page (title + summary + lead section text) ~2.5k chunks + - section : one chunk per section (page title + section title + text) + a + page anchor chunk ~15-18k chunks; better for concept queries. + +Embedding is batched with live progress and CHECKPOINTED every CKPT_EVERY chunks +to embeddings-<mode>.npz, so a crash/kill resumes instead of restarting (the +21.5k-chunk first attempt died after >1h with nothing saved). Model is +bge-small-en-v1.5 (what we'd run via RedisVL in production); loaded here through +fastembed (ONNX, no torch). Embedding is ~6 chunks/s on this CPU β€” an +unoptimised-local artefact, not a production latency signal. + +Usage: + pip install -r requirements.txt + node ../node/test/eval/dump-lexical.mjs > lexical.json + python eval_vector.py section +""" +import gzip +import json +import os +import sys +import time + +import numpy as np + +HERE = os.path.dirname(os.path.abspath(__file__)) +FEED = os.path.join(HERE, "..", "node", "test", "eval", "docs.ndjson.gz") +LEXICAL = os.path.join(HERE, "lexical.json") +DEFAULT_MODEL = "BAAI/bge-small-en-v1.5" +MODEL = os.environ.get("EMBED_MODEL", DEFAULT_MODEL) # e.g. BAAI/bge-base-en-v1.5 +BGE_QUERY_PREFIX = "Represent this sentence for searching relevant passages: " +LEAD_CHARS = 1200 +MAX_SECTIONS = 8 +CKPT_EVERY = 1024 +KS = [1, 3, 5, 10] + +MODE = (sys.argv[1] if len(sys.argv) > 1 else "section").lower() +# Default model keeps its original cache name (back-compat); other models get a +# per-model cache so runs don't clobber each other. +_SLUG = MODEL.split("/")[-1] +CACHE = os.path.join( + HERE, + f"embeddings-{MODE}.npz" if MODEL == DEFAULT_MODEL else f"embeddings-{MODE}-{_SLUG}.npz", +) + + +def norm_url(u): + return u.strip().lower().rstrip("/") + + +def load_pages(): + pages = [] + with gzip.open(FEED, "rt", encoding="utf-8") as f: + for line in f: + line = line.strip() + if not line: + continue + try: + o = json.loads(line) + except json.JSONDecodeError: + continue + if isinstance(o, dict) and o.get("url") and o.get("title"): + pages.append(o) + return pages + + +def build_chunks(pages, mode): + texts, owners = [], [] + for p in pages: + url = norm_url(p["url"]) + title = p.get("title", "") or "" + summary = p.get("summary", "") or "" + sections = p.get("sections") or [] + if mode == "page": + secs = " ".join((s.get("text", "") or "") for s in sections)[:LEAD_CHARS] + parts = [x for x in (title, summary, secs) if x] + texts.append(". ".join(parts).strip() or title or url) + owners.append(url) + else: # section + anchor = ". ".join(x for x in (title, summary) if x).strip() + if anchor: + texts.append(anchor) + owners.append(url) + n = 0 + for s in sections: + body = (s.get("text", "") or "").strip() + if len(body) < 20: + continue + st = (s.get("title", "") or "").strip() + texts.append(f"{title} β€” {st}. {body[:LEAD_CHARS]}".strip()) + owners.append(url) + n += 1 + if n >= MAX_SECTIONS: + break + if not anchor and n == 0: + texts.append(title or url) + owners.append(url) + return texts, np.array(owners) + + +def _model(): + from fastembed import TextEmbedding + + return TextEmbedding(model_name=MODEL, threads=os.cpu_count()) + + +def embed_batch(model, texts, is_query=False): + payload = [BGE_QUERY_PREFIX + t for t in texts] if is_query else texts + arr = np.asarray(list(model.embed(payload, batch_size=64)), dtype=np.float32) + arr /= np.linalg.norm(arr, axis=1, keepdims=True) + 1e-12 + return arr + + +def get_corpus_embeddings(texts, owners): + total = len(texts) + emb = None # allocated once we know the model's embedding dimension + start = 0 + if os.path.exists(CACHE): + d = np.load(CACHE, allow_pickle=True) + if int(d["total"]) == total: + emb = d["emb"] + start = int(d["n_done"]) + if start >= total: + print(f" (full cache: {total} chunks)") + return emb, owners + print(f" resuming from {start}/{total}") + model = _model() + t0 = time.time() + for s in range(start, total, CKPT_EVERY): + e = min(s + CKPT_EVERY, total) + batch = embed_batch(model, texts[s:e]) + if emb is None: + emb = np.zeros((total, batch.shape[1]), dtype=np.float32) + emb[s:e] = batch + rate = (e - start) / (time.time() - t0) + print(f" {e}/{total} chunks ({rate:.0f}/s)", flush=True) + np.savez(CACHE, emb=emb, owners=owners, total=total, n_done=e) + print(f" cached {os.path.basename(CACHE)}") + return emb, owners + + +def rank_pages(qvec, emb, owners, topn=50): + sims = emb @ qvec + best = {} + for i in np.argsort(-sims): + u = owners[i] + if u not in best: + best[u] = float(sims[i]) + if len(best) >= topn: + break + return [u for u, _ in sorted(best.items(), key=lambda kv: -kv[1])] + + +def rrf(lists, k=60, topn=50): + scores = {} + for lst in lists: + for rank, url in enumerate(lst): + scores[url] = scores.get(url, 0.0) + 1.0 / (k + rank + 1) + return [u for u, _ in sorted(scores.items(), key=lambda kv: -kv[1])][:topn] + + +def best_rank(ranking, expected): + for i, u in enumerate(ranking): + if u in expected: + return i + 1 + return None + + +def metrics(ranks): + n = len(ranks) or 1 + rec = {k: sum(1 for r in ranks if r and r <= k) / n for k in KS} + mrr = sum((1.0 / r) if r else 0.0 for r in ranks) / n + return rec, mrr + + +def main(): + if not os.path.exists(LEXICAL): + sys.exit("lexical.json missing β€” run: node ../node/test/eval/dump-lexical.mjs > lexical.json") + cases = json.load(open(LEXICAL)) + + print(f"Mode: {MODE}. Loading feed + building chunks ...") + pages = load_pages() + texts, owners = build_chunks(pages, MODE) + print(f" {len(pages)} pages -> {len(texts)} chunks") + + print("Embedding corpus ...") + emb, owners = get_corpus_embeddings(texts, owners) + + print("Embedding queries ...") + qvecs = embed_batch(_model(), [c["q"] for c in cases], is_query=True) + + systems, groups = ["lexical", "vector", "hybrid"], ["overall", "command", "concept"] + data = {s: {g: [] for g in groups} for s in systems} + for c, qv in zip(cases, qvecs): + expected = set(c["expected"]) + lex, vec = c["lexical"], rank_pages(qv, emb, owners) + ranks = {"lexical": best_rank(lex, expected), + "vector": best_rank(vec, expected), + "hybrid": best_rank(rrf([lex, vec]), expected)} + for s in systems: + data[s]["overall"].append(ranks[s]) + data[s][c["kind"]].append(ranks[s]) + + print(f"\n### chunk mode: {MODE} ###") + for g in groups: + n = len(data["lexical"][g]) + print(f"\n=== {g} (n={n}) === recall@1 / @3 / @5 / @10 | MRR") + for s in systems: + rec, mrr = metrics(data[s][g]) + cells = " / ".join(f"{rec[k]*100:3.0f}%" for k in KS) + print(f" {s:8} {cells} | {mrr:.3f}") + + +if __name__ == "__main__": + main() diff --git a/build/docs-mcp-server/vector-eval/fusion_sweep.py b/build/docs-mcp-server/vector-eval/fusion_sweep.py new file mode 100644 index 0000000000..8bbff41e03 --- /dev/null +++ b/build/docs-mcp-server/vector-eval/fusion_sweep.py @@ -0,0 +1,56 @@ +#!/usr/bin/env python3 +""" +Fusion sweep: with section-level embeddings CACHED, compare lexical / pure +vector / equal-weight RRF / weighted RRF (favouring vector) on the eval, to +resolve which fusion to build. Reuses eval_vector's corpus embeddings (cache +hit β€” no re-embedding) and the dumped lexical rankings. + + python fusion_sweep.py section +""" +import json + +import eval_vector as ev + + +def wrrf(lists_weights, k=60, topn=50): + """Weighted reciprocal-rank fusion. lists_weights: [(ranked_urls, weight)].""" + scores = {} + for lst, w in lists_weights: + for r, u in enumerate(lst): + scores[u] = scores.get(u, 0.0) + w / (k + r + 1) + return [u for u, _ in sorted(scores.items(), key=lambda kv: -kv[1])][:topn] + + +cases = json.load(open(ev.LEXICAL)) +pages = ev.load_pages() +texts, owners = ev.build_chunks(pages, ev.MODE) +emb, owners = ev.get_corpus_embeddings(texts, owners) # cache hit +qvecs = ev.embed_batch(ev._model(), [c["q"] for c in cases], is_query=True) + +systems = { + "lexical": lambda lex, vec: lex, + "vector": lambda lex, vec: vec, + "rrf 1:1": lambda lex, vec: wrrf([(vec, 1), (lex, 1)]), + "wrrf v2": lambda lex, vec: wrrf([(vec, 2), (lex, 1)]), + "wrrf v3": lambda lex, vec: wrrf([(vec, 3), (lex, 1)]), + "wrrf v5": lambda lex, vec: wrrf([(vec, 5), (lex, 1)]), +} +groups = ["overall", "command", "concept"] +data = {s: {g: [] for g in groups} for s in systems} + +for c, q in zip(cases, qvecs): + exp = set(c["expected"]) + lex = c["lexical"] + vec = ev.rank_pages(q, emb, owners) + for s, fn in systems.items(): + r = ev.best_rank(fn(lex, vec), exp) + data[s]["overall"].append(r) + data[s][c["kind"]].append(r) + +for g in groups: + n = len(data["lexical"][g]) + print(f"\n=== {g} (n={n}) === recall@1 / @3 / @5 / @10 | MRR") + for s in systems: + rec, mrr = ev.metrics(data[s][g]) + cells = " / ".join(f"{rec[k] * 100:3.0f}%" for k in ev.KS) + print(f" {s:9} {cells} | {mrr:.3f}") diff --git a/build/docs-mcp-server/vector-eval/requirements.txt b/build/docs-mcp-server/vector-eval/requirements.txt new file mode 100644 index 0000000000..7f7505c751 --- /dev/null +++ b/build/docs-mcp-server/vector-eval/requirements.txt @@ -0,0 +1,2 @@ +fastembed>=0.3 +numpy>=1.24