From bc0f3917ba4cea3c03a131f0df7890dd0021af02 Mon Sep 17 00:00:00 2001 From: Felix Stubner Date: Mon, 21 Sep 2026 01:10:09 +0100 Subject: [PATCH 01/34] fix: close the five findings left open by the audit All four code defects came out of the same audit as #380 and share its shape: a read degrades correctly, and the degraded value is then used as the base for a write or a filter. Antigravity dropped every step whose timestamp fell back to the epoch sentinel. `steps.ts` returns `new Date(0)` when neither the step metadata nor the summary carries a parseable time, `fullSync` reads with `since = new Date(0)`, and `0 <= 0` is true -- so those steps were lost on every path including the rebuild that exists to recover them, silently. Every other scraper guards this; claude-code even spells out why. `disconnect` stripped comments from a TOML config that `setup` refuses to touch. The write path checks `tomlHasComments` and declines, telling the user to edit by hand and keep their comments; the remove path re-serialised and deleted them all, which made following that advice pointless. `disconnect` rewrote an unparseable `.xtctx/config.yaml` from `{}`. Degrading the read is right and deliberate -- disconnect is what someone reaches for when things are already wrong. Using `{}` as the base for the write is not: the file came back holding nothing but `tools:`, and any `storePath` override went with it. Everything else still disconnects; only this rewrite is skipped, with a warning naming the file. Cursor renumbered a composer after a pruned bubble. Two skips did not advance `messageIndex` while the two below them did, and `scan.ts` hashes that index into the row id -- so a bubble Cursor pruned between scans shifted every later turn and re-inserted it under a new id beside the row already stored. The fifth finding is not a fix, because it could not be reproduced. The smoke suite failed four of five cases once under load and passed on every attempt since, including under deliberate contention. What made that unfixable is that nothing drained the spawned server's stderr, so the only evidence was "initialize did not answer within 60s" -- a symptom, with the cause piped into a buffer nobody read. That text is now attached to the failure, and an exited child fails its pending call immediately instead of waiting out the timeout. Each test fails against the previous behaviour: the antigravity one on an empty chunk list, cursor on index 1 where 2 belongs, and both disconnect ones on the destroyed file. --- src/config/disconnect.ts | 31 ++++-- src/config/mcp-config.ts | 18 +++ src/scrapers/antigravity/scraper.ts | 13 ++- src/scrapers/cursor.ts | 13 ++- .../disconnect-preserves-unparsable.test.ts | 75 +++++++++++++ .../antigravity-epoch-timestamps.test.ts | 81 ++++++++++++++ .../cursor-pruned-bubble-index.test.ts | 105 ++++++++++++++++++ tests/smoke/mcp-stdio.smoke.test.ts | 44 +++++++- 8 files changed, 369 insertions(+), 11 deletions(-) create mode 100644 tests/config/disconnect-preserves-unparsable.test.ts create mode 100644 tests/scrapers/antigravity-epoch-timestamps.test.ts create mode 100644 tests/scrapers/cursor-pruned-bubble-index.test.ts diff --git a/src/config/disconnect.ts b/src/config/disconnect.ts index 1fd9320e..19f0d7ef 100644 --- a/src/config/disconnect.ts +++ b/src/config/disconnect.ts @@ -169,14 +169,18 @@ export async function disconnectProject(options: DisconnectOptions = {}): Promis }; } + const projectConfig = await disableToolsInProjectConfig(configPath, tools, projectRoot); writes.push({ path: configPath, kind: "config", // config.yaml is rewritten with the tools disabled, never deleted — it is // the record that xtctx was disconnected. action: "updated", - changed: await disableToolsInProjectConfig(configPath, tools, projectRoot), + changed: projectConfig.changed, }); + if (projectConfig.warning) { + warnings.push(projectConfig.warning); + } const mcpTools = options.globalMcp ? tools : tools.filter((tool) => !GLOBAL_MCP_TOOLS.has(tool)); const mcpSummary = await removeMcpServerConfigs(projectRoot, "xtctx", mcpTools, options.homeDir ? { homeDir: options.homeDir } : {}); @@ -323,21 +327,34 @@ async function disableToolsInProjectConfig( configPath: string, tools: ToolId[], projectRoot: string, -): Promise { +): Promise<{ changed: boolean; warning?: string }> { const raw = await readUtf8IfExists(configPath); if (raw === null) { - return false; + return { changed: false }; } // Every other reader of this file degrades on unparseable YAML rather than // throwing; this one did not, so a stray tab made uninstalling impossible. // Disconnect is the command someone reaches for when things are already // wrong, which is the worst moment to require a well-formed config. + // + // Degrading the READ is right. Using the degraded result as the base for a + // WRITE is not: `{}` plus the disabled tools is a complete document, so a + // file with a stray tab was replaced by one holding nothing but `tools:`, + // and any `storePath` override in it — which `status` treats as a + // deliberate, user-owned setting — went with it. Everything else disconnect + // does still happens; only this rewrite is skipped. let parsed: unknown; try { parsed = parseYaml(raw); - } catch { - parsed = null; + } catch (error) { + return { + changed: false, + warning: + `${configPath} could not be parsed, so its tools were left as they are: ` + + `${error instanceof Error ? error.message : String(error)}. ` + + `Everything else was disconnected; fix the file or delete it to finish.`, + }; } const config = isRecord(parsed) ? parsed : {}; const currentTools = isRecord(config.tools) ? { ...config.tools } : {}; @@ -353,12 +370,12 @@ async function disableToolsInProjectConfig( } if (!changed) { - return false; + return { changed: false }; } config.tools = currentTools; await writeFileAtomic(configPath, stringifyYaml(config), { containWithin: projectRoot }); - return true; + return { changed: true }; } function memoryPathsToDisconnect(projectRoot: string, tools: ToolId[], all: boolean): string[] { diff --git a/src/config/mcp-config.ts b/src/config/mcp-config.ts index 6e9aaf63..20293eee 100644 --- a/src/config/mcp-config.ts +++ b/src/config/mcp-config.ts @@ -520,6 +520,24 @@ async function removeMcpConfig( }; } + // Same refusal the write path makes, for the same reason: the TOML parser + // drops comments, so rewriting a commented file to take one entry out + // deletes every comment with it. Setup already declines here and tells the + // user to edit by hand keeping their comments — and disconnect then + // removed them all, which made following that advice pointless. + if (format === "toml" && tomlHasComments(existingContent)) { + return { + tool, + path: configPath, + scope, + removed: false, + skipped: true, + warning: + `MCP config at ${configPath} contains comments, which xtctx will not rewrite. ` + + `Remove the "${serverName}" entry manually, or remove the comments so xtctx can manage it.`, + }; + } + const existing = parseConfig(existingContent, format); const existingEntries = isRecord(existing[rootKey]) ? { ...(existing[rootKey] as Record) } : {}; diff --git a/src/scrapers/antigravity/scraper.ts b/src/scrapers/antigravity/scraper.ts index 3165953c..8ab77b95 100644 --- a/src/scrapers/antigravity/scraper.ts +++ b/src/scrapers/antigravity/scraper.ts @@ -124,7 +124,13 @@ export class AntigravityScraper extends AbstractScraper { }); for (const [messageIndex, artifact] of artifacts.entries()) { - if (artifact.timestamp <= since) { + // A zero `since` means full sync: emit even epoch-sentinel timestamps. + // `store.ts` returns `new Date(0)` for an artifact whose metadata has + // no `updatedAt` and whose `stat` fails, and `0 <= 0` is true — so + // without this guard those artifacts were dropped on every path, + // including a rebuild that exists to recover them, and nothing in the + // drift log said so. + if (since.getTime() > 0 && artifact.timestamp <= since) { continue; } @@ -230,7 +236,10 @@ export class AntigravityScraper extends AbstractScraper { }); for (const [messageIndex, message] of sortedMessages.entries()) { - if (message.timestamp <= since) { + // See the fallback loop above: `steps.ts` falls back to `new Date(0)` + // when neither the step metadata nor the summary carries a parseable + // time, and a full sync must still emit those. + if (since.getTime() > 0 && message.timestamp <= since) { continue; } diff --git a/src/scrapers/cursor.ts b/src/scrapers/cursor.ts index 4917bab7..c8759d61 100644 --- a/src/scrapers/cursor.ts +++ b/src/scrapers/cursor.ts @@ -451,12 +451,23 @@ export class CursorScraper extends AbstractScraper { `bubbleId:${ref.composerId}:${header.bubbleId}`, ) as { value: string } | undefined; - if (!bubbleRow) continue; + // Both skips advance `messageIndex`, like the two below them. + // `scan.ts` hashes the index into the row id, so an index that counts + // only the bubbles that happened to be present shifts every later turn + // down by one the moment Cursor prunes an earlier bubble — and + // `ACCEPTED_DEGRADATIONS.prunedBubble` records that pruning as normal. + // Those turns then re-insert under new ids beside the old rows, which + // `pruneRereadSessions` does not clear on an incremental pass. + if (!bubbleRow) { + messageIndex++; + continue; + } let bubble: CursorBubbleData; try { bubble = JSON.parse(bubbleRow.value) as CursorBubbleData; } catch { + messageIndex++; continue; } diff --git a/tests/config/disconnect-preserves-unparsable.test.ts b/tests/config/disconnect-preserves-unparsable.test.ts new file mode 100644 index 00000000..453e23b7 --- /dev/null +++ b/tests/config/disconnect-preserves-unparsable.test.ts @@ -0,0 +1,75 @@ +/** + * What `disconnect` does with files it cannot safely rewrite. + * + * Both cases here share a shape: the read degrades correctly, and the + * degraded value was then used as the base for a write. Setup already + * refuses in both situations; disconnect did not, so uninstalling destroyed + * what installing had carefully left alone. + */ +import { mkdir, mkdtemp, readFile, rm, writeFile } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { disconnectProject } from "@xtctx/config/disconnect"; + +let projectRoot: string; +let homeDir: string; + +beforeEach(async () => { + projectRoot = await mkdtemp(join(tmpdir(), "xtctx-disc-keep-")); + homeDir = await mkdtemp(join(tmpdir(), "xtctx-disc-home-")); + await mkdir(join(projectRoot, ".xtctx"), { recursive: true }); +}); + +afterEach(async () => { + await rm(projectRoot, { recursive: true, force: true }); + await rm(homeDir, { recursive: true, force: true }); +}); + +describe("disconnect with an unparsable project config", () => { + // A stray tab: the exact break the read path was made tolerant of. + const BROKEN = `project: + root: . +tools: +\tcursor: + storePath: D:/elsewhere/cursor +`; + + it("leaves the file alone and says so", async () => { + const configPath = join(projectRoot, ".xtctx", "config.yaml"); + await writeFile(configPath, BROKEN, "utf-8"); + + const result = await disconnectProject({ projectPath: projectRoot, all: true, homeDir }); + + expect(await readFile(configPath, "utf-8")).toBe(BROKEN); + expect(result.warnings.join(" ")).toMatch(/could not be parsed/i); + }); +}); + +describe("disconnect with a commented TOML config", () => { + const COMMENTED = `# Codex config, hand-written. +# The comments are the point. +[mcp_servers.xtctx] +command = "npx" +args = ["-y", "xtctx"] + +[mcp_servers.other] +command = "other-server" +`; + + it("refuses to rewrite it and keeps the comments", async () => { + await writeFile( + join(projectRoot, ".xtctx", "config.yaml"), + "project:\n root: .\ntools:\n codex:\n enabled: true\n", + "utf-8", + ); + await mkdir(join(projectRoot, ".codex"), { recursive: true }); + const tomlPath = join(projectRoot, ".codex", "config.toml"); + await writeFile(tomlPath, COMMENTED, "utf-8"); + + const result = await disconnectProject({ projectPath: projectRoot, tool: "codex", homeDir }); + + expect(await readFile(tomlPath, "utf-8")).toBe(COMMENTED); + expect(result.warnings.join(" ")).toMatch(/comments/i); + }); +}); diff --git a/tests/scrapers/antigravity-epoch-timestamps.test.ts b/tests/scrapers/antigravity-epoch-timestamps.test.ts new file mode 100644 index 00000000..fd468c86 --- /dev/null +++ b/tests/scrapers/antigravity-epoch-timestamps.test.ts @@ -0,0 +1,81 @@ +/** + * A step with no readable time still has to survive a full rebuild. + * + * `steps.ts` falls back to `new Date(0)` when neither the step metadata nor + * the conversation summary carries a parseable timestamp. `fullSync()` reads + * with `since = new Date(0)`, and the filter was `timestamp <= since` with no + * guard — so `0 <= 0` dropped exactly those steps, on every path, including + * the rebuild that exists to recover them. Nothing recorded it. + * + * Every other scraper guards this. claude-code spells it out: + * "A zero `since` means full sync: emit even epoch-sentinel timestamps." + */ +import { mkdtemp, rm } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { + AntigravityScraper, + type AntigravityRuntimeClient, + type AntigravityRuntimeConversation, +} from "@xtctx/scrapers/antigravity"; +import type { AntigravityChunk } from "@xtctx/types/scraper"; + +let rootDir = ""; +let stateDir = ""; +const projectRoot = join("H:", "projects", "private", "needs-work", "xtctx"); + +function runtimeClient(conversations: AntigravityRuntimeConversation[]): AntigravityRuntimeClient { + return { + async listConversations() { + return { conversations }; + }, + } as unknown as AntigravityRuntimeClient; +} + +beforeEach(async () => { + rootDir = await mkdtemp(join(tmpdir(), "xtctx-ag-epoch-root-")); + stateDir = await mkdtemp(join(tmpdir(), "xtctx-ag-epoch-state-")); +}); + +afterEach(async () => { + await rm(rootDir, { recursive: true, force: true }); + await rm(stateDir, { recursive: true, force: true }); +}); + +describe("AntigravityScraper full sync", () => { + it("emits steps whose timestamp fell back to the epoch", async () => { + const chunks: AntigravityChunk[] = []; + const scraper = new AntigravityScraper( + rootDir, + stateDir, + projectRoot, + runtimeClient([ + { + sessionId: "cascade-epoch", + title: "Work with no usable timestamps", + createdAt: new Date(0), + workspaces: ["file:///h:/projects/private/needs-work/xtctx"], + messages: [ + { + sessionId: "cascade-epoch", + // What the parser produces when nothing readable was found. + timestamp: new Date(0), + role: "user", + content: "Please update src/cli/index.ts", + referencedFiles: ["h:/projects/private/needs-work/xtctx/src/cli/index.ts"], + metadata: {}, + }, + ], + } as unknown as AntigravityRuntimeConversation, + ]), + ); + + for await (const chunk of scraper.fullSync()) { + chunks.push(chunk); + } + + expect(chunks).toHaveLength(1); + expect(chunks[0].content).toContain("src/cli/index.ts"); + }); +}); diff --git a/tests/scrapers/cursor-pruned-bubble-index.test.ts b/tests/scrapers/cursor-pruned-bubble-index.test.ts new file mode 100644 index 00000000..23afca46 --- /dev/null +++ b/tests/scrapers/cursor-pruned-bubble-index.test.ts @@ -0,0 +1,105 @@ +/** + * A pruned bubble must not shift every later turn's index. + * + * Cursor stores a composer's turn order as a header list and each turn's text + * as a separate `bubbleId:` row, and it prunes those rows — the scraper's own + * `ACCEPTED_DEGRADATIONS.prunedBubble` records "bubble referenced by composer + * but missing from globalStorage" as normal operation. + * + * The loop skipped a missing bubble without advancing `messageIndex`, while + * the two skips below it did advance. `scan.ts` hashes `messageIndex` into + * the row id, so a bubble disappearing between scans renumbered every later + * turn and re-inserted them under new ids beside the rows already stored — + * and `pruneRereadSessions` only clears those when the re-read reached at or + * below the lowest stored index, which an incremental pass does not. + * + * The index has to describe the turn's position in the conversation, not its + * position among the rows that happen to have survived. + */ +import { mkdir, mkdtemp, rm } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import Database from "better-sqlite3"; +import { CursorScraper } from "@xtctx/scrapers/cursor"; +import type { CursorChunk } from "@xtctx/types/scraper"; + +const COMPOSER_ID = "comp-prune-0001"; +const FIRST = "bubble-first-0001"; +const PRUNED = "bubble-pruned-0002"; +const LAST = "bubble-last-0003"; + +let rootDir = ""; +let workspaceDir = ""; +let stateDir = ""; + +/** The composer lists three turns; only the first and third still have rows. */ +function createGlobalDb(path: string): void { + const db = new Database(path); + db.exec(`CREATE TABLE cursorDiskKV (key TEXT PRIMARY KEY, value TEXT NOT NULL)`); + const insert = db.prepare("INSERT INTO cursorDiskKV (key, value) VALUES (?, ?)"); + + insert.run( + `composerData:${COMPOSER_ID}`, + JSON.stringify({ + composerId: COMPOSER_ID, + fullConversationHeadersOnly: [ + { bubbleId: FIRST, type: 1 }, + { bubbleId: PRUNED, type: 2 }, + { bubbleId: LAST, type: 2 }, + ], + createdAt: new Date("2026-02-24T10:00:00Z").getTime(), + lastUpdatedAt: new Date("2026-02-24T10:00:10Z").getTime(), + unifiedMode: "agent", + }), + ); + insert.run( + `bubbleId:${COMPOSER_ID}:${FIRST}`, + JSON.stringify({ type: 1, text: "first turn", createdAt: "2026-02-24T10:00:00Z" }), + ); + // No row for PRUNED — this is the case under test. + insert.run( + `bubbleId:${COMPOSER_ID}:${LAST}`, + JSON.stringify({ type: 2, text: "third turn", createdAt: "2026-02-24T10:00:10Z" }), + ); + db.close(); +} + +function createWorkspaceDb(path: string): void { + const db = new Database(path); + db.exec(`CREATE TABLE ItemTable (key TEXT PRIMARY KEY, value TEXT NOT NULL)`); + db.prepare("INSERT INTO ItemTable (key, value) VALUES (?, ?)").run( + "composer.composerData", + JSON.stringify({ allComposers: [{ composerId: COMPOSER_ID }] }), + ); + db.close(); +} + +beforeEach(async () => { + rootDir = await mkdtemp(join(tmpdir(), "xtctx-cursor-prune-")); + stateDir = await mkdtemp(join(tmpdir(), "xtctx-cursor-prune-state-")); + workspaceDir = join(rootDir, "workspaceStorage", "abc123"); + await mkdir(workspaceDir, { recursive: true }); + await mkdir(join(rootDir, "globalStorage"), { recursive: true }); + createWorkspaceDb(join(workspaceDir, "state.vscdb")); + createGlobalDb(join(rootDir, "globalStorage", "state.vscdb")); +}); + +afterEach(async () => { + await rm(rootDir, { recursive: true, force: true }); + await rm(stateDir, { recursive: true, force: true }); +}); + +describe("CursorScraper with a pruned bubble", () => { + it("keeps the surviving turns at their original indexes", async () => { + const chunks: CursorChunk[] = []; + for await (const chunk of new CursorScraper(workspaceDir, stateDir).fullSync()) { + chunks.push(chunk); + } + + expect(chunks.map((chunk) => chunk.content)).toEqual(["first turn", "third turn"]); + // The third turn is index 2 because it is the third turn — not index 1 + // because the second one's row is gone. + expect(chunks.map((chunk) => chunk.metadata.messageIndex)).toEqual([0, 2]); + }); +}); diff --git a/tests/smoke/mcp-stdio.smoke.test.ts b/tests/smoke/mcp-stdio.smoke.test.ts index a455eb65..977278a1 100644 --- a/tests/smoke/mcp-stdio.smoke.test.ts +++ b/tests/smoke/mcp-stdio.smoke.test.ts @@ -49,9 +49,37 @@ describe("MCP server over stdio", () => { return text; } + /** + * What the spawned server wrote to stderr, and how it died if it died. + * + * Nothing read the child's stderr, so when the server failed to start the + * suite reported only "initialize did not answer within 60s" — an assertion + * about a symptom, with the cause piped into a buffer nobody drained. That + * happened once under load on 2026-09-20 (four of five cases in this file), + * could not be reproduced, and left nothing to diagnose. + * + * The likely cause is the one the comment below already names: the child + * loads a ~100MB ONNX model, and this machine was running several model + * loads at once. That is a guess, and it stays a guess precisely because + * this text was thrown away. Now it is attached to the failure. + */ + let childStderr = ""; + let childExit: string | null = null; + + function serverDiagnostics(): string { + const tail = childStderr.trim().split("\n").slice(-20).join("\n"); + return [ + childExit ? `server process ${childExit}` : "server process still running", + tail ? `stderr:\n${tail}` : "stderr: (empty)", + ].join("\n"); + } + function request(id: number, method: string, params: unknown): Promise> { return new Promise((resolvePromise, rejectPromise) => { - const timer = setTimeout(() => rejectPromise(new Error(`${method} did not answer within 60s`)), 60_000); + const timer = setTimeout( + () => rejectPromise(new Error(`${method} did not answer within 60s\n${serverDiagnostics()}`)), + 60_000, + ); pending.set(id, (message) => { clearTimeout(timer); resolvePromise(message); @@ -99,6 +127,20 @@ describe("MCP server over stdio", () => { stdio: ["pipe", "pipe", "pipe"], }) as ChildProcessWithoutNullStreams; + proc.stderr.on("data", (chunk) => { + childStderr += String(chunk); + }); + // Fail the pending call immediately rather than waiting out the timeout: + // a server that has exited is never going to answer, and sixty seconds of + // waiting adds nothing but sixty seconds. + proc.on("exit", (code, signal) => { + childExit = signal ? `killed by ${signal}` : `exited with code ${String(code)}`; + for (const [id, resolvePending] of pending) { + pending.delete(id); + resolvePending({ error: { message: `server ${childExit}\n${serverDiagnostics()}` } }); + } + }); + proc.stdout.on("data", (chunk) => { buffer += String(chunk); let newline = buffer.indexOf("\n"); From 172e640133f03950cb6f9edcdeec3fde3cee5ef0 Mon Sep 17 00:00:00 2001 From: Felix Stubner Date: Mon, 21 Sep 2026 01:26:40 +0100 Subject: [PATCH 02/34] fix: close what the second audit found in the index and MCP layers Neither area had ever been reviewed. Both had defects, which is the answer to whether the first audit had found everything. `detail_offset` pointed past the end of the session it was meant to point into. The `target` CTE had no `LIMIT 1` and the query cross-joins `messages` against it, so a session with duplicate `message_index` values multiplied the count by however many duplicates there were. Those duplicates are not hypothetical: the comment directly above the statement records 828 of them in one real session. Reproduced at six messages with three sharing an index -- offset 12 into a six-row session, so `getSessionDetail` paged past the end and returned nothing. That is the "match points somewhere unrelated" failure this statement exists to prevent, in its worst form. `literalUnreadableTools` never left `buildIndexProgress`. The field was not declared on `ProgressInputs` and the caller passes its inputs by spread, so excess-property checking could not catch it and the value was dropped in silence. The "this store cannot be read" branch in `sessions.ts` was therefore unreachable, and every unreadable store was reported as a search that stopped at its limit -- advising the caller to narrow a query, which against an unreadable store returns the same nothing forever. The comment explaining why that advice is wrong was already there; the code that acts on it could never run. `session_refs` widened silently. `uniqueStrings` answered `[]` for a non-array, and `[]` routes to the branch returning recent sessions, so a caller asking for three specific sessions got arbitrary recent ones with `missing_session_refs` empty. `validatedFilter` exists to stop exactly this and its docstring names this handler; the fix had reached `tool_filter` and `branch_filter` and missed this one. Also capped, since each ref is a synchronous sqlite `get` on the event-loop thread. Two leaks on the continuity surface, both one branch away from code that already did it right: `embedding_error` was `inlineSafe`d but not `sanitizeErrorMessage`d, though it carries an absolute path on a model load failure and `tool.last_error` two lines below gets both; and `redirected_tools` was joined raw, though its entries are arbitrary keys from a committable config file while `tool.tool` beside it -- a scraper literal -- was escaped. `loadProjectConfig` treated every read failure as a missing config. `present: false` makes every tool tell the agent to run setup, and setup rewrites config.yaml with every tool enabled, so a locked or busy file would end with the user's `enabled: false` undone. The parse-failure branch beside it distinguishes this case carefully; the read-failure branch did not. --- src/handoff/schema.ts | 12 ++ src/handoff/status.ts | 16 +++ src/mcp/tools/continuity.ts | 4 +- src/mcp/tools/manifest.ts | 37 +++++- src/runtime/services.ts | 20 +++- .../handoff/detail-offset-duplicates.test.ts | 106 ++++++++++++++++++ 6 files changed, 186 insertions(+), 9 deletions(-) create mode 100644 tests/handoff/detail-offset-duplicates.test.ts diff --git a/src/handoff/schema.ts b/src/handoff/schema.ts index 671edcad..61b4356e 100644 --- a/src/handoff/schema.ts +++ b/src/handoff/schema.ts @@ -294,9 +294,21 @@ export function prepareStatements(db: DatabaseHandle): PreparedStatements { * three changes, all three must. */ messageOffsetInSession: db.prepare( + // `LIMIT 1`, because `message_index` is not unique — the 828 duplicates + // named above are in one real session. Without it the CTE returns a row + // per duplicate and `FROM messages m, target t` cross-joins every one, + // multiplying the count by however many duplicates there are. Measured + // on six messages with `message_index = 3` on three of them: offset 12 + // for a session holding 6 rows, so `getSessionDetail` paged past the end + // and returned nothing — which is the "match points somewhere + // unrelated" failure this statement exists to prevent, in its worst + // form. The ordering picks the same first row the two statements above + // would. `WITH target AS ( SELECT timestamp, message_index, id FROM messages WHERE session_ref = ? AND message_index = ? + ORDER BY timestamp ASC, message_index ASC, id ASC + LIMIT 1 ) SELECT COUNT(*) AS count FROM messages m, target t diff --git a/src/handoff/status.ts b/src/handoff/status.ts index 2fd49263..d9d8a08a 100644 --- a/src/handoff/status.ts +++ b/src/handoff/status.ts @@ -129,6 +129,19 @@ interface ProgressInputs { vectorBacklog: number; embeddingWarming: boolean; literalSearchStoppedEarly?: boolean; + /** + * Declared, and copied below, because the caller passes it by spread. + * + * Excess-property checking does not apply to a spread, so an undeclared + * field is dropped here in silence: `literalUnreadableTools` reached this + * function and never left it, which left the unreadable-store branch in + * `mcp/tools/sessions.ts` permanently unreachable. A store that cannot be + * read was therefore always reported as a search that stopped at its limit, + * advising the caller to narrow a query — the exact wrong advice that + * branch was written to replace, since narrowing a query against an + * unreadable store returns the same nothing forever. + */ + literalUnreadableTools?: string[]; } export function buildIndexProgress(inputs: ProgressInputs): IndexProgress { @@ -142,5 +155,8 @@ export function buildIndexProgress(inputs: ProgressInputs): IndexProgress { ...(inputs.literalSearchStoppedEarly === undefined ? {} : { literalSearchStoppedEarly: inputs.literalSearchStoppedEarly }), + ...(inputs.literalUnreadableTools === undefined + ? {} + : { literalUnreadableTools: inputs.literalUnreadableTools }), }; } diff --git a/src/mcp/tools/continuity.ts b/src/mcp/tools/continuity.ts index 239def9b..190b3f3d 100644 --- a/src/mcp/tools/continuity.ts +++ b/src/mcp/tools/continuity.ts @@ -68,14 +68,14 @@ export function createContinuityStatusHandler(service: SessionService) { })(), `- Vector model: ${inlineSafe(status.vector_model)}`, ...(status.embedding_error - ? [`- Semantic search unavailable (keyword only): ${inlineSafe(status.embedding_error)}`] + ? [`- Semantic search unavailable (keyword only): ${inlineSafe(sanitizeErrorMessage(status.embedding_error))}`] : []), // Named, not pathed: `store_paths` is redacted from this surface two // branches up because it carries the machine's home-directory layout, // and a warning does not get an exception from that. ...(status.redirected_tools.length > 0 ? [ - `- WARNING: ${status.redirected_tools.join(", ")} read transcripts from a store ` + + `- WARNING: ${status.redirected_tools.map(inlineSafe).join(", ")} read transcripts from a store ` + "outside the home directory, set by .xtctx/config.yaml. That file is " + "committable, so a cloned repo can carry one — the sessions returned for " + "these tools may not be this project's.", diff --git a/src/mcp/tools/manifest.ts b/src/mcp/tools/manifest.ts index 361a1a13..e14cdd39 100644 --- a/src/mcp/tools/manifest.ts +++ b/src/mcp/tools/manifest.ts @@ -1,5 +1,5 @@ import type { SessionService, SessionSummary } from "../../handoff/types.js"; -import { indexingPayload, validatedFilter } from "./sessions.js"; +import { indexingPayload, ToolInputError, validatedFilter } from "./sessions.js"; import { inlineSafe } from "../../utils/untrusted-text.js"; interface HandoffManifestParams { @@ -22,7 +22,7 @@ const MAX_LIMIT = 100; export function createHandoffManifestHandler(service: SessionService) { return async (raw: Record = {}) => { const params = raw as unknown as HandoffManifestParams; - const requestedRefs = uniqueStrings(params.session_refs); + const requestedRefs = requestedSessionRefs(params.session_refs); // Requested refs are resolved directly by primary key — filtering a // recency-limited list would misreport older indexed sessions as missing. @@ -128,11 +128,38 @@ function formatManifestMarkdown(manifest: { return lines.join("\n").trim(); } -function uniqueStrings(value: unknown): string[] { - if (!Array.isArray(value)) { +/** + * The requested session refs, or a refusal. + * + * Previously this answered `[]` for anything that was not an array of + * strings, and `[]` routes to the branch that returns recent sessions — so a + * caller asking for three specific sessions got some arbitrary recent ones + * instead, with `missing_session_refs` empty because nothing had been asked + * for. `validatedFilter` exists to stop exactly that widening, and its + * docstring names this handler as the reason; the fix reached `tool_filter` + * and `branch_filter` here and missed `session_refs`. + * + * The cap matches `limit`'s. Each ref is one synchronous better-sqlite3 `get` + * on the event-loop thread, so an unbounded list stalls the stdio server. + */ +function requestedSessionRefs(value: unknown): string[] { + if (value === undefined || value === null) { return []; } - return [...new Set(value.filter((item): item is string => typeof item === "string" && item.length > 0))]; + + if (!Array.isArray(value)) { + throw new ToolInputError("session_refs must be an array of strings"); + } + + if (value.some((item) => typeof item !== "string" || item.trim().length === 0)) { + throw new ToolInputError("session_refs must contain only non-empty strings"); + } + + const unique = [...new Set(value as string[])]; + if (unique.length > MAX_LIMIT) { + throw new ToolInputError(`session_refs accepts at most ${MAX_LIMIT} refs`); + } + return unique; } function normalizeCorrelationId(value: unknown): string | undefined { diff --git a/src/runtime/services.ts b/src/runtime/services.ts index e6aab7ae..a9fc5b43 100644 --- a/src/runtime/services.ts +++ b/src/runtime/services.ts @@ -131,11 +131,27 @@ async function loadProjectConfig(configPath: string): Promise { let raw: string; try { raw = await readFile(configPath, "utf-8"); - } catch { + } catch (err) { // Missing config is valid — `status` still diagnoses, and the MCP server // still starts. What it must not do is behave as though the project were // configured; see `present`. - return { tools: {}, present: false }; + // + // Only ENOENT, though. A config that exists but cannot be READ is the + // same situation as one that cannot be PARSED, which the branch below + // takes care to distinguish: answering `present: false` there makes every + // tool tell the agent to run `xtctx setup`, and setup rewrites + // config.yaml with every tool `enabled: true` — so a locked or busy file + // would end with the user's `enabled: false` silently undone. The + // degraded read becoming the base for a write, again. + const code = (err as NodeJS.ErrnoException).code; + if (code === "ENOENT") { + return { tools: {}, present: false }; + } + return { + tools: {}, + present: true, + error: `could not be read (${code ?? "unknown error"}): ${err instanceof Error ? err.message : String(err)}`, + }; } try { diff --git a/tests/handoff/detail-offset-duplicates.test.ts b/tests/handoff/detail-offset-duplicates.test.ts new file mode 100644 index 00000000..7f5ad2dd --- /dev/null +++ b/tests/handoff/detail-offset-duplicates.test.ts @@ -0,0 +1,106 @@ +/** + * `detail_offset` has to survive a session with duplicate message indexes. + * + * The statement exists because a window records the `message_index` VALUES at + * its edges while detail pages by POSITION, and real transcripts make those + * two disagree — one session in a live index carries 828 duplicate message + * indexes and 862 places where index order disagrees with time order. + * + * Duplicates are exactly what broke it. The `target` CTE had no `LIMIT 1` and + * the query cross-joins `messages` against it, so N rows sharing an index + * multiplied the count by N. The offset then pointed past the end of the + * session and detail returned an empty page — the same "the match is not + * there" failure, in its worst form. + */ +import { mkdtemp, rm } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { openDatabase, prepareStatements } from "@xtctx/handoff/schema"; + +let dir = ""; + +beforeEach(async () => { + dir = await mkdtemp(join(tmpdir(), "xtctx-offset-")); +}); + +afterEach(async () => { + await rm(dir, { recursive: true, force: true }); +}); + +function seed(): ReturnType { + const db = openDatabase(join(dir, "index.db")); + db.prepare( + `INSERT INTO sessions (session_ref, tool, source_session_id, project_root, + started_at, last_activity_at, message_count, updated_at) + VALUES (?, 'claude-code', 'src', ?, ?, ?, 6, ?)`, + ).run("session-dupes", dir, "2026-01-01T00:00:00Z", "2026-01-01T00:00:05Z", "2026-01-01T00:00:05Z"); + + const insert = db.prepare( + `INSERT INTO messages (id, session_ref, tool, source_session_id, timestamp, + role, content, message_index, content_hash, metadata_json, indexed_at) + VALUES (?, ?, 'claude-code', 'src', ?, 'user', 'body', ?, 'h', '{}', ?)`, + ); + // Six messages; three of them share message_index 3. + const rows: Array<[number, string]> = [ + [0, "2026-01-01T00:00:00Z"], + [1, "2026-01-01T00:00:01Z"], + [2, "2026-01-01T00:00:02Z"], + [3, "2026-01-01T00:00:03Z"], + [3, "2026-01-01T00:00:04Z"], + [3, "2026-01-01T00:00:05Z"], + ]; + rows.forEach(([index, timestamp], position) => { + insert.run(`m${position}`, "session-dupes", timestamp, index, timestamp); + }); + return db; +} + +describe("messageOffsetInSession", () => { + it("counts each earlier message once when indexes repeat", () => { + const db = seed(); + try { + const statements = prepareStatements(db); + const row = statements.messageOffsetInSession.get( + "session-dupes", + 3, + "session-dupes", + ) as { count: number }; + + // Three messages sort before the first message_index 3, so the offset + // is 3 — not 12, which is what a cross-join against three duplicate + // target rows produced, and which pages past a six-row session. + expect(row.count).toBe(3); + expect(row.count).toBeLessThan(6); + } finally { + db.close(); + } + }); +}); + +/** + * `buildIndexProgress` builds its result field by field, and the caller passes + * its inputs by spread — so a field the interface does not declare is dropped + * without a type error. That happened to `literalUnreadableTools`, which left + * the "this store cannot be read" branch in `mcp/tools/sessions.ts` + * unreachable: every unreadable store was reported as a search that stopped + * at its limit, advising the caller to narrow a query that cannot ever match. + */ +describe("buildIndexProgress", () => { + it("passes through the tools whose store could not be read", async () => { + const { buildIndexProgress } = await import("@xtctx/handoff/status"); + + const progress = buildIndexProgress({ + scanning: false, + tools: [{ tool: "codex" }], + scannedTools: new Set(["codex"]), + vectorBacklog: 0, + embeddingWarming: false, + literalSearchStoppedEarly: true, + literalUnreadableTools: ["cursor"], + }); + + expect(progress.literalUnreadableTools).toEqual(["cursor"]); + expect(progress.literalSearchStoppedEarly).toBe(true); + }); +}); From 9a3bf33617c601797dbbf83bdcc633770a519491 Mon Sep 17 00:00:00 2001 From: Felix Stubner Date: Mon, 21 Sep 2026 01:42:35 +0100 Subject: [PATCH 03/34] fix(gates): make three release gates able to fail, and two tests able to see The scripts and the test suite had never been audited. Both were hiding failures rather than catching them, which is the worse half of the thirteen defects found earlier today: some of those were watched by tests that could not see them. `audit:production` reported a clean audit of a tree it had not read. With `--json`, an npm that cannot reach the registry prints an error object and exits 0, so the body parses, carries no `vulnerabilities`, and every check found nothing wrong. Adding `--json` is what inverted it -- bare `npm audit` exits 1 on the same failure. Reproduced with `npm_config_registry=http://127.0.0.1:9`: "audit clean", exit 0, and `verify:release` would have passed on it. A security gate has to fail closed; not knowing is not the same as knowing there is nothing. `shapeOf` could not see inside an array. The docstring said "one level of arrays" and the code did none, so `{content:[{type,text}]}` and `{content:[{kind,body}]}` produced byte-identical shapes. Claude Code's entire assistant payload is `message.content[]`, so the mechanism that exists to catch upstream format drift was blind to the place drift would appear -- in `capture:formats` and in `fixture-fidelity`, which runs in CI and in `verify:release`. Elements are now shaped under `path[]`, keeping the key redaction that stops private paths reaching a committed fingerprint. The security checklist could not see an indented unchecked control -- the matcher was anchored at column 0 while the file uses nested bullets -- and counted a citation pointing at a directory as a verified test file. Two tests could not fail, both proven by breaking the source and watching them stay green. `capSegments` swapped for `segments.slice(0, limit)`: the test asserted first element, last element and ascending order, all of which truncation satisfies. `detail_offset` hardcoded to `0`: the test compared `getSessionDetail(ref, offset, 1)` against the same call sliced at the same offset, so both sides moved together and it proved paging was self-consistent, not that the pointer was right. That is the same feature found broken earlier today by the cross-join in `messageOffsetInSession` -- the test was there and could not see it. The backlog test now asserts the backlog is zero, not only that two numbers agree; it held with one window of twenty-four embedded. Not changed, deliberately: the segment cap stays at 16. Its test only restates the constant, which is a floor against lowering it rather than the percentile property the heading claimed, and that is now written down. Raising it to 17 is a cost decision about embedding time, not a test fix. --- scripts/audit-production.mjs | 38 ++++++++++++- scripts/lib/record-shape.mjs | 39 ++++++++++++- scripts/verify-security-checklist.mjs | 19 ++++++- tests/drift/record-shape.test.ts | 67 +++++++++++++++++++++++ tests/handoff/cap-segments.test.ts | 21 +++++-- tests/handoff/match-detail-offset.test.ts | 24 ++++++-- tests/handoff/vector-backlog.test.ts | 5 ++ 7 files changed, 197 insertions(+), 16 deletions(-) create mode 100644 tests/drift/record-shape.test.ts diff --git a/scripts/audit-production.mjs b/scripts/audit-production.mjs index 578ee0c9..032acb44 100644 --- a/scripts/audit-production.mjs +++ b/scripts/audit-production.mjs @@ -76,13 +76,49 @@ function runAudit() { throw new Error(`could not run npm audit: ${result.error.message}`); } + let report; try { - return JSON.parse(result.stdout); + report = JSON.parse(result.stdout); } catch { throw new Error( `npm audit did not return JSON.\nstdout: ${result.stdout.slice(0, 2000)}\nstderr: ${result.stderr.slice(0, 2000)}`, ); } + + // Parsing is not the same as auditing. With `--json`, an npm that cannot + // reach the registry prints an error OBJECT and exits 0 — so the body parses, + // carries no `vulnerabilities`, and every check below finds nothing wrong. + // This gate then reported "audit clean" for a tree it had not looked at, and + // `verify:release` passed on it. Reproduced with + // `npm_config_registry=http://127.0.0.1:9`: clean, exit 0. Bare `npm audit` + // exits 1 on the same failure; adding `--json` is what inverted it. + // + // A security gate has to fail closed: not knowing is not the same as knowing + // there is nothing. + if (report === null || typeof report !== "object" || Array.isArray(report)) { + throw new Error(`npm audit returned ${typeof report}, not a report object.`); + } + if (report.error) { + const fields = + typeof report.error === "object" && report.error !== null + ? [report.error.summary, report.error.detail, JSON.stringify(report.error)] + : [String(report.error)]; + // npm fills some of these with an empty string rather than omitting them, + // so pick the first that carries text instead of the first that is defined. + const detail = fields.find((value) => typeof value === "string" && value.trim() !== ""); + throw new Error( + `npm audit could not complete: ${String(detail ?? "npm reported an error with no detail").slice(0, 2000)}\n` + + `stderr: ${result.stderr.slice(0, 1000)}`, + ); + } + if (typeof report.vulnerabilities !== "object" || report.vulnerabilities === null) { + throw new Error( + "npm audit returned no `vulnerabilities` section, so nothing was audited.\n" + + `stdout: ${result.stdout.slice(0, 2000)}`, + ); + } + + return report; } function main() { diff --git a/scripts/lib/record-shape.mjs b/scripts/lib/record-shape.mjs index 591fe267..e92ca459 100644 --- a/scripts/lib/record-shape.mjs +++ b/scripts/lib/record-shape.mjs @@ -37,16 +37,49 @@ export function schemaKey(key) { return key; } -/** Flatten an object into "path: type" entries, one level of arrays. */ +/** How many elements of an array are shaped. Matches `typeOf`'s own sample. */ +const ARRAY_SAMPLE = 20; + +/** + * Flatten an object into "path: type" entries, descending into arrays. + * + * Arrays used to stop at `path: array` — the docstring claimed "one + * level of arrays" and the code did none. That made the whole drift mechanism + * blind to any change inside an array, and Claude Code's entire assistant + * payload is one: `message.content[]`. Measured before the fix, + * `{content:[{type,text}]}` and `{content:[{kind,body}]}` produced byte- + * identical shape sets, so a wholesale upstream rename inside that array was + * indistinguishable from no change at all. + * + * That matters more than an ordinary gap because it is load-bearing twice: + * `capture:formats` reports drift from it, and `tests/drift/fixture-fidelity` + * asserts through it in `verify:release` and in CI. Both reported agreement + * they had not established. + * + * Elements are shaped under a `path[]` prefix, sampled like `typeOf` samples, + * and the union across elements is kept — a heterogeneous array records every + * variant it holds rather than only the first. + */ export function shapeOf(value, prefix = "", out = new Set(), depth = 0) { - if (depth > 4 || value === null || typeof value !== "object" || Array.isArray(value)) { + if (depth > 4 || value === null || typeof value !== "object") { if (prefix) out.add(`${prefix}: ${typeOf(value)}`); return out; } + + if (Array.isArray(value)) { + // The array's own type is still recorded, so an array becoming a scalar is + // still visible even when the elements shape to nothing. + if (prefix) out.add(`${prefix}: ${typeOf(value)}`); + for (const element of value.slice(0, ARRAY_SAMPLE)) { + shapeOf(element, `${prefix}[]`, out, depth + 1); + } + return out; + } + for (const [rawKey, inner] of Object.entries(value)) { const key = schemaKey(rawKey); const path = prefix ? `${prefix}.${key}` : key; - if (inner !== null && typeof inner === "object" && !Array.isArray(inner)) { + if (inner !== null && typeof inner === "object") { shapeOf(inner, path, out, depth + 1); } else { out.add(`${path}: ${typeOf(inner)}`); diff --git a/scripts/verify-security-checklist.mjs b/scripts/verify-security-checklist.mjs index b86f67c3..96c44d48 100644 --- a/scripts/verify-security-checklist.mjs +++ b/scripts/verify-security-checklist.mjs @@ -1,5 +1,5 @@ import { readFile } from "node:fs/promises"; -import { existsSync } from "node:fs"; +import { statSync } from "node:fs"; import { resolve } from "node:path"; const checklistPath = resolve(process.cwd(), "docs", "security", "owasp-asvs-lite.md"); @@ -19,7 +19,10 @@ for (const heading of requiredHeadings) { } } -const unchecked = [...content.matchAll(/^- \[ \] (.+)$/gm)].map((match) => match[1]); +// `^\s*`, not `^`: the checklist nests continuation bullets under their +// parent, so an anchored matcher could not see an unchecked control that had +// been indented — the gate passed on a checklist with open items in it. +const unchecked = [...content.matchAll(/^\s*- \[ \] (.+)$/gm)].map((match) => match[1]); if (unchecked.length > 0) { const summary = unchecked.map((item) => `- ${item}`).join("\n"); throw new Error( @@ -46,7 +49,17 @@ const cited = [...content.matchAll(//g)] const missingEvidence = []; for (const relative of cited) { - if (!existsSync(resolve(process.cwd(), relative))) { + // `isFile`, not `existsSync`: a citation left pointing at a directory after + // the named test was deleted still "existed", and the summary went on + // claiming the citation had been verified. + const cursor = resolve(process.cwd(), relative); + let isFile = false; + try { + isFile = statSync(cursor).isFile(); + } catch { + isFile = false; + } + if (!isFile) { missingEvidence.push(relative); } } diff --git a/tests/drift/record-shape.test.ts b/tests/drift/record-shape.test.ts new file mode 100644 index 00000000..9d2aaf58 --- /dev/null +++ b/tests/drift/record-shape.test.ts @@ -0,0 +1,67 @@ +/** + * The shape function the whole drift mechanism is built on. + * + * It described an array as `array` and stopped, so anything inside one + * was invisible — and Claude Code's entire assistant payload is inside one, + * `message.content[]`. Two records with wholly different element fields + * produced byte-identical shapes, which meant `capture:formats` reported "no + * drift" and `fixture-fidelity` accepted invented fields, both without having + * compared the part that changes. + * + * Both callers run inside `verify:release`, so this was a release gate + * reporting agreement it had never established. + */ +import { describe, expect, it } from "vitest"; +import { shapeOf, typeOf } from "../../scripts/lib/record-shape.mjs"; + +function sorted(value: unknown): string[] { + return [...shapeOf(value)].sort(); +} + +describe("shapeOf", () => { + it("tells apart records whose difference is inside an array", () => { + const before = sorted({ type: "assistant", message: { content: [{ type: "text", text: "hi" }] } }); + const after = sorted({ type: "assistant", message: { content: [{ kind: "prose", body: "hi" }] } }); + + expect(before).not.toEqual(after); + expect(before).toContain("message.content[].type: string"); + expect(after).toContain("message.content[].kind: string"); + }); + + it("still records the array's own type, so an array becoming a scalar shows", () => { + expect(sorted({ content: [{ a: 1 }] })).toContain("content: array"); + expect(sorted({ content: "plain" })).toContain("content: string"); + }); + + it("keeps every variant in a heterogeneous array, not just the first", () => { + const shape = sorted({ content: [{ type: "text" }, { type: "tool_use", name: "Read" }] }); + + expect(shape).toContain("content[].type: string"); + expect(shape).toContain("content[].name: string"); + }); + + it("redacts data-shaped keys inside array elements too", () => { + // The reason keys are collapsed at all: cursor stores per-file state under + // the file's own URI, and a naive walk wrote absolute paths from unrelated + // private projects into a committed fingerprint. Descending into arrays + // must not reopen that. + const shape = sorted({ items: [{ "5f2e8a1b9c3d4e6f": { seen: true } }] }); + + expect(shape).toContain("items[].*.seen: boolean"); + expect(shape.join(" ")).not.toContain("5f2e8a1b9c3d4e6f"); + }); + + it("does not descend for ever", () => { + let nested: unknown = { leaf: 1 }; + for (let depth = 0; depth < 12; depth += 1) { + nested = [{ next: nested }]; + } + + expect(() => shapeOf(nested)).not.toThrow(); + expect([...shapeOf(nested)].every((entry) => entry.split("[]").length <= 8)).toBe(true); + }); + + it("describes an empty array distinctly", () => { + expect(typeOf([])).toBe("array"); + }); +}); diff --git a/tests/handoff/cap-segments.test.ts b/tests/handoff/cap-segments.test.ts index 796af310..3c747705 100644 --- a/tests/handoff/cap-segments.test.ts +++ b/tests/handoff/cap-segments.test.ts @@ -14,12 +14,20 @@ describe("capSegments", () => { const segments = Array.from({ length: 100 }, (_, index) => `s${index}`); const capped = capSegments(segments, 4); - expect(capped).toHaveLength(4); - expect(capped[0]).toBe("s0"); - expect(capped[capped.length - 1]).not.toBe("s1"); + // The whole sample, not its endpoints. Asserting only the first and last + // element left this passing against `segments.slice(0, limit)` — proven by + // replacing the body with exactly that and watching the file stay green, + // which meant the behaviour the function exists for was undefended. + expect(capped).toEqual(["s0", "s25", "s50", "s75"]); + // Order preserved, so pooling stays deterministic. const indexes = capped.map((s) => Number(s.slice(1))); expect([...indexes].sort((a, b) => a - b)).toEqual(indexes); + + // The property the exact values above encode: every element comes from a + // different quarter of the window, which truncation cannot satisfy. + const spread = indexes[indexes.length - 1] - indexes[0]; + expect(spread).toBeGreaterThan(segments.length / 2); }); it("never returns more than the limit, at any size", () => { @@ -29,10 +37,15 @@ describe("capSegments", () => { } }); - it("keeps the default above the 95th percentile of real windows", () => { + it("pins the cap against being lowered", () => { // Measured over this project's 1,770 windows: median 4 segments, p95 17, // max 392. A cap below the bulk of the distribution would be trading // quality for speed on ordinary windows rather than trimming the tail. + // A floor, not a property: `>= 16` against a constant of 16 only restates + // the value, so this catches the cap being lowered and nothing else. It + // does NOT establish what the heading claims — p95 is 17, so 16 already + // clips part of that bucket, deliberately. Raising the cap to 17 is a cost + // decision about embedding time, not a test fix, and is not made here. expect(MAX_SEGMENTS_PER_UNIT).toBeGreaterThanOrEqual(16); }); }); diff --git a/tests/handoff/match-detail-offset.test.ts b/tests/handoff/match-detail-offset.test.ts index e6b675c7..98b4d5d1 100644 --- a/tests/handoff/match-detail-offset.test.ts +++ b/tests/handoff/match-detail-offset.test.ts @@ -117,12 +117,26 @@ describe("a match's detail offset", () => { // message. Read through the public detail path, exactly as an agent // would follow it. const [landed] = await index.getSessionDetail(ref, match.detail_offset as number, 1); - const [expected] = await index.getSessionDetail(ref, 0, 100).then((all) => - all.slice(match.detail_offset as number, (match.detail_offset as number) + 1), - ); - expect(landed).toBeDefined(); - expect(landed.content).toBe(expected.content); + + // Named from the fixture, NOT from `detail_offset`. + // + // This compared `getSessionDetail(ref, offset, 1)` against + // `getSessionDetail(ref, 0, 100).slice(offset, offset + 1)` — both sides + // derived from the same pointer, so it proved paging was self-consistent + // and nothing about whether the pointer was right. Proven by replacing + // the offset with a hardcoded `0`: both sides became element 0 and the + // whole file stayed green while every match pointed at the wrong place. + // + // The fixture numbers each message into its own content, so the window's + // `message_start_index` names exactly one message independently of any + // offset arithmetic. + const startIndex = match.message_start_index; + const expectedContent = + startIndex >= 500 + ? `late message ${startIndex - 500} about the parser fallback` + : `early message ${startIndex} about the parser fallback`; + expect(landed.content).toBe(expectedContent); } }, 60_000); diff --git a/tests/handoff/vector-backlog.test.ts b/tests/handoff/vector-backlog.test.ts index abc89f8e..58576b7c 100644 --- a/tests/handoff/vector-backlog.test.ts +++ b/tests/handoff/vector-backlog.test.ts @@ -122,6 +122,11 @@ describe("vectorBacklog", () => { const status = await index.getStatus(); expect(status.vectorized_units).toBeGreaterThan(0); + // Both halves. The subtraction alone held with one window of twenty-four + // embedded — it says the two numbers agree, not that the work finished, + // which is what the heading claims. + expect(status.vectorized_units).toBe(status.retrieval_units); + expect(index.getIndexProgress().vectorBacklog).toBe(0); expect(index.getIndexProgress().vectorBacklog).toBe( status.retrieval_units - status.vectorized_units, ); From 729d47327387fcf4450bb7e6d88f297ca22d7f60 Mon Sep 17 00:00:00 2001 From: Felix Stubner Date: Mon, 21 Sep 2026 02:23:14 +0100 Subject: [PATCH 04/34] fix(scrapers): report where drift was seen, and make six tests able to see A fourth audit, this time asking only one question of the suite: which tests would still pass if the code they cover were broken. Six could not fail, and one of them was hiding a live defect. Drift locations from an incremental scan were wrong. The counter started at zero on every pass while a resumed read starts at `startAt` BYTES -- `readJsonlLines` yields only lines past the cursor -- so a record appended as line 101 reported as `path:1`. A drift location is the only pointer anyone has when chasing an upstream format break, and every one produced after a resume pointed somewhere else. It is now the byte offset the reader already tracks, written `path@offset`, which means the same thing on a first read and a resumed one. Nothing caught that, and nothing could: no test in the suite asserted a location any scraper computed. The two that assert `firstLocation` pass the string to `recordDrift` themselves, so they pin the drift log's plumbing. Changing the format of every location in two scrapers left 700 tests green. Now pinned, including after a resume, where the old behaviour fails with "expected 1 to be greater than 100". The manifest's markdown branch had no positive assertion anywhere in the repository. Returning `""` from it passed all 697 tests: every assertion on that surface was a `not.toMatch`, and an empty string forges nothing. An entire public MCP output branch -- heading, project line, per-session block, retrieve hint, missing-refs list -- was undefended. Each scrub case now pins the value it scrubbed as well as the structure it refused to forge, because a scrub that works by rendering nothing is not a scrub. `fenceFor` was pinned by `>= 4` tildes, which a constant four-tilde fence satisfies -- and content holding four tildes then closes it from the inside. Now parameterised over three run lengths and asserting the delimiter is strictly longer than the content's longest run. The codex cursor test compared `offset` to `size` from the same written record, so recording the file's size instead of the boundary read passed it. That is the permanent silent loss the other two readers pin. Now compared against the real file, with the partial-trailing-line case that actually distinguishes them. Also: the shutdown test now asserts the callback has NOT run before stdin ends, so wiring that tears the server down at boot is distinguishable from wiring that works; and the opencode ordering test no longer sorts the very property it is named for. --- src/scrapers/claude-code.ts | 25 +++-- src/scrapers/copilot-cli.ts | 23 +++-- tests/mcp/hardening.test.ts | 26 +++++- tests/mcp/server.test.ts | 5 + tests/mcp/source-field-safety.test.ts | 32 +++++++ tests/scrapers/codex-resume.test.ts | 30 +++++- tests/scrapers/drift-location.test.ts | 127 ++++++++++++++++++++++++++ tests/scrapers/opencode.test.ts | 4 +- 8 files changed, 248 insertions(+), 24 deletions(-) create mode 100644 tests/scrapers/drift-location.test.ts diff --git a/src/scrapers/claude-code.ts b/src/scrapers/claude-code.ts index 0fdf32bc..02ab55ef 100644 --- a/src/scrapers/claude-code.ts +++ b/src/scrapers/claude-code.ts @@ -254,7 +254,14 @@ export class ClaudeCodeScraper extends AbstractScraper { const resumed = startAt > 0 ? cursor?.context : undefined; let messageIndex = resumed?.messageIndex ?? 0; - let lineNo = 0; + // A byte offset, not a line number. + // + // A resumed read starts at `startAt` bytes, so a counter starting at zero + // counts lines since the RESUME POINT: a record appended as line 101 + // reported as `path:1`, and a drift location is the only pointer anyone + // has when chasing a format break. The offset is what the reader already + // tracks and means the same thing on every pass. + let byteAt = startAt; /** * null until a record in this file names a project. * @@ -268,13 +275,13 @@ export class ClaudeCodeScraper extends AbstractScraper { let readTo = startAt; for await (const entry of readJsonlLines(filePath, { start: startAt })) { + byteAt = entry.endOffset; readTo = entry.endOffset; - lineNo++; const line = entry.line; if (line === null) { recordDrift( SCRAPER_NAME, - `${filePath}:${lineNo}`, + `${filePath}@${byteAt}`, `line exceeds ${MAX_LINE_BYTES} characters; skipped`, ); continue; @@ -291,7 +298,7 @@ export class ClaudeCodeScraper extends AbstractScraper { // mutation test can see drift instead of data silently vanishing. recordDrift( SCRAPER_NAME, - `${filePath}:${lineNo}`, + `${filePath}@${byteAt}`, `line is not valid JSON: ${(err as Error).message}`, ); continue; @@ -339,7 +346,7 @@ export class ClaudeCodeScraper extends AbstractScraper { if (!("type" in obj)) { recordDrift( SCRAPER_NAME, - `${filePath}:${lineNo}`, + `${filePath}@${byteAt}`, "record is missing required 'type' field — likely renamed", ); } else if ( @@ -351,7 +358,7 @@ export class ClaudeCodeScraper extends AbstractScraper { ) { recordDrift( SCRAPER_NAME, - `${filePath}:${lineNo}`, + `${filePath}@${byteAt}`, `unknown 'type' value ${JSON.stringify(obj.type)}`, ); } @@ -363,7 +370,7 @@ export class ClaudeCodeScraper extends AbstractScraper { ) { recordDrift( SCRAPER_NAME, - `${filePath}:${lineNo}`, + `${filePath}@${byteAt}`, `expected 'content' to be a string, got ${describeType(obj.content)}`, ); } @@ -371,13 +378,13 @@ export class ClaudeCodeScraper extends AbstractScraper { if (!("timestamp" in obj)) { recordDrift( SCRAPER_NAME, - `${filePath}:${lineNo}`, + `${filePath}@${byteAt}`, "record is missing 'timestamp' field", ); } else if (typeof obj.timestamp !== "string") { recordDrift( SCRAPER_NAME, - `${filePath}:${lineNo}`, + `${filePath}@${byteAt}`, `expected 'timestamp' string, got ${describeType(obj.timestamp)}`, ); } diff --git a/src/scrapers/copilot-cli.ts b/src/scrapers/copilot-cli.ts index 8aed650d..cc3b2077 100644 --- a/src/scrapers/copilot-cli.ts +++ b/src/scrapers/copilot-cli.ts @@ -178,7 +178,14 @@ export class CopilotCliScraper extends AbstractScraper { const resumed = startAt > 0 ? cursor?.context : undefined; let messageIndex = resumed?.messageIndex ?? 0; - let lineNo = 0; + // A byte offset, not a line number. + // + // A resumed read starts at `startAt` bytes, so a counter starting at zero + // counts lines since the RESUME POINT: a record appended as line 101 + // reported as `path:1`, and a drift location is the only pointer anyone + // has when chasing a format break. The offset is what the reader already + // tracks and means the same thing on every pass. + let byteAt = startAt; // null = no session.start context seen yet; only consulted when scoped. // Carried across a resume: it is set by the `session.start` record at the // head of the file, which a resumed read never sees again. @@ -188,8 +195,8 @@ export class CopilotCliScraper extends AbstractScraper { let readTo = startAt; for await (const entry of readJsonlLines(filePath, { start: startAt })) { + byteAt = entry.endOffset; readTo = entry.endOffset; - lineNo++; const line = entry.line; if (line === null) { warnDrift(filePath, `line exceeds the cap; skipped`); @@ -205,7 +212,7 @@ export class CopilotCliScraper extends AbstractScraper { event = JSON.parse(line) as Record; } catch (err) { warnDrift( - `${filePath}:${lineNo}`, + `${filePath}@${byteAt}`, `events.jsonl line is not valid JSON: ${(err as Error).message}`, ); continue; @@ -213,7 +220,7 @@ export class CopilotCliScraper extends AbstractScraper { if (!isRecord(event)) { warnDrift( - `${filePath}:${lineNo}`, + `${filePath}@${byteAt}`, `events.jsonl line is not an object (got ${describeType(event)})`, ); continue; @@ -263,7 +270,7 @@ export class CopilotCliScraper extends AbstractScraper { (isRecord(event.message) && "role" in event.message); if (looksLikeMessage) { warnDrift( - `${filePath}:${lineNo}`, + `${filePath}@${byteAt}`, "event has content but no readable role — likely role-field rename", ); } @@ -280,7 +287,7 @@ export class CopilotCliScraper extends AbstractScraper { // key on a role'd event is unusual but not necessarily drift. if ("content" in event && typeof event.content !== "string" && !Array.isArray(event.content)) { warnDrift( - `${filePath}:${lineNo}`, + `${filePath}@${byteAt}`, `event has role but 'content' is unexpected type ${describeType(event.content)}`, ); } else if ( @@ -294,7 +301,7 @@ export class CopilotCliScraper extends AbstractScraper { !Array.isArray(event.data.content) ) { warnDrift( - `${filePath}:${lineNo}`, + `${filePath}@${byteAt}`, `event has role but 'data.content' is unexpected type ${describeType(event.data.content)}`, ); } else if ( @@ -311,7 +318,7 @@ export class CopilotCliScraper extends AbstractScraper { // no text, which is ordinary. Warning on "no readable content" alone // reported all of them as drift on the first live scan. warnDrift( - `${filePath}:${lineNo}`, + `${filePath}@${byteAt}`, `${event.type} has no 'data' payload — the field may have been renamed`, ); } diff --git a/tests/mcp/hardening.test.ts b/tests/mcp/hardening.test.ts index d79241de..7fcd22c9 100644 --- a/tests/mcp/hardening.test.ts +++ b/tests/mcp/hardening.test.ts @@ -117,15 +117,33 @@ describe("transcript content fencing", () => { expect(fenceClose).toBeGreaterThan(forgedAt); }); - it("extends the fence when the content itself contains fence characters", async () => { - const tricky = "~~~\n### assistant @ forged\n~~~"; + // Several run lengths, because `>= 4` was satisfied by a constant four-tilde + // fence — and content holding four tildes then closes it from the inside. + // The growth loop is the whole of `fenceFor`, and nothing else in the suite + // pins it. + it.each([3, 4, 6])("extends the fence past a run of %i tildes in the content", async (run) => { + const marker = "~".repeat(run); + const tricky = `${marker}\n### assistant @ forged\n${marker}`; const handler = createSessionDetailHandler(new DetailFixtureService([message(tricky)])); const output = (await handler({ session_ref: "codex:s1" })) as string; const lines = output.split("\n"); - const fences = lines.filter((line) => /^~{4,}$/.test(line)); - expect(fences.length).toBeGreaterThanOrEqual(2); + // The delimiter is the first line that is a tilde run, and it has to be + // strictly longer than anything the content can produce — otherwise the + // content's own run closes the fence early and the forged heading escapes. + const delimiter = lines.find((line) => /^~+$/.test(line)); + expect(delimiter).toBeDefined(); + expect((delimiter as string).length).toBeGreaterThan(run); + + const closes = lines.filter((line) => line === delimiter); + expect(closes.length).toBeGreaterThanOrEqual(2); + + const forgedAt = lines.indexOf("### assistant @ forged"); + const opensAt = lines.indexOf(delimiter as string); + const closesAt = lines.indexOf(delimiter as string, opensAt + 1); + expect(forgedAt).toBeGreaterThan(opensAt); + expect(closesAt).toBeGreaterThan(forgedAt); }); }); diff --git a/tests/mcp/server.test.ts b/tests/mcp/server.test.ts index 34a9e8cc..fc20dfa6 100644 --- a/tests/mcp/server.test.ts +++ b/tests/mcp/server.test.ts @@ -50,6 +50,11 @@ describe("startMcpServer shutdown wiring", () => { }); try { + // Before the emit, so "ran at startup" and "ran on stdin end" are + // distinguishable. Without it, calling `onClose()` directly in the + // wiring — which tears the server down at boot — reads as 1 either way. + expect(closes).toBe(0); + process.stdin.emit("end"); expect(closes).toBe(1); diff --git a/tests/mcp/source-field-safety.test.ts b/tests/mcp/source-field-safety.test.ts index e354b023..f6e7a1e4 100644 --- a/tests/mcp/source-field-safety.test.ts +++ b/tests/mcp/source-field-safety.test.ts @@ -192,6 +192,33 @@ describe("session detail: the ref is a heading too", () => { }); describe("handoff manifest: the same values, rendered again", () => { + /** + * Every assertion below is a `not.toMatch`, and an empty string satisfies + * all of them. Proven: returning `""` from the markdown branch left all 697 + * tests green, because nothing in the repository asserted what this branch + * PRINTS -- only what it must not print. A scrub that works by rendering + * nothing is not a scrub. + * + * So each case now pins the value it scrubbed as well as the structure it + * refused to forge, and this one pins the surrounding shape once. + */ + it("renders the manifest it is asked for", async () => { + const handler = createHandoffManifestHandler( + new FixtureService([session({ session_ref: "codex:present" })], []), + ); + + const out = (await handler({ + limit: 5, + format: "markdown", + correlation_id: "abc", + })) as string; + + expect(out).toContain("## xtctx Handoff Manifest"); + expect(out).toContain("- Correlation ID: abc"); + expect(out).toContain("codex:present"); + expect(out).toContain("xtctx_session_detail"); + }); + it("neutralises session_ref in both places it appears", async () => { // The manifest prints it as a heading and inside the retrieve hint, so a // fix applied to the sessions tool alone leaves this surface forgeable. @@ -202,6 +229,9 @@ describe("handoff manifest: the same values, rendered again", () => { const out = (await handler({ limit: 5, format: "markdown" })) as string; expect(out).not.toMatch(/^### FORGED HEADING/m); + // Scrubbed, not dropped: the ref still has to be reported. + expect(out).toContain("codex:x"); + expect(out).toContain("FORGED HEADING"); }); it("neutralises the refs it reports as missing", async () => { @@ -216,6 +246,7 @@ describe("handoff manifest: the same values, rendered again", () => { })) as string; expect(out).not.toMatch(/^## FORGED MISSING/m); + expect(out).toMatch(/^Missing sessions: .*codex:missing/m); }); it("neutralises a caller-supplied correlation id", async () => { @@ -231,5 +262,6 @@ describe("handoff manifest: the same values, rendered again", () => { })) as string; expect(out).not.toMatch(/^## FORGED CORRELATION/m); + expect(out).toContain("- Correlation ID: abc"); }); }); diff --git a/tests/scrapers/codex-resume.test.ts b/tests/scrapers/codex-resume.test.ts index 36b93656..d66a2af2 100644 --- a/tests/scrapers/codex-resume.test.ts +++ b/tests/scrapers/codex-resume.test.ts @@ -9,7 +9,7 @@ * re-read whole, `fullSync` ignores cursors entirely, and the derived state a * resumed read cannot see is carried across. */ -import { appendFile, mkdtemp, readFile, rm, writeFile } from "node:fs/promises"; +import { appendFile, mkdtemp, readFile, rm, stat, writeFile } from "node:fs/promises"; import { tmpdir } from "node:os"; import { join } from "node:path"; import { afterEach, beforeEach, describe, expect, it } from "vitest"; @@ -62,7 +62,33 @@ describe("codex incremental resume", () => { const c = await cursor(); expect(c?.offset).toBeGreaterThan(0); - expect(c?.offset).toBe(c?.size); + // Against the FILE, not against the other half of the same written record. + // `offset === size` compares two fields the writer set together, so + // recording the file's size instead of the boundary actually read passed + // it — which is the permanent silent loss the other two readers pin with + // `offset < size` on a partial trailing line. + const { size } = await stat(file); + expect(c?.offset).toBe(size); + expect(c?.size).toBe(size); + }); + + it("stops short of a trailing line that has no newline yet", async () => { + // These files are appended to while they are read, so the last line often + // has no newline. `readJsonlLines` deliberately stops before it; recording + // the file's SIZE instead moves the next scan into the middle of that + // record and it is never yielded — a permanent loss, one per interrupted + // append. The sibling readers pin this; codex only compared the cursor to + // itself, so `offset: size` passed. + const complete = [META("abc"), MSG("first", "2026-02-24T10:00:00Z")].join("\n") + "\n"; + const partial = MSG("half-written", "2026-02-24T10:00:01Z").slice(0, 20); + await writeFile(file, complete + partial); + + expect(await scrape(make())).toContain("first"); + + const c = await cursor(); + const { size } = await stat(file); + expect(c?.offset).toBeGreaterThan(0); + expect(c?.offset).toBeLessThan(size); }); it("reads only what was appended on the next scrape", async () => { diff --git a/tests/scrapers/drift-location.test.ts b/tests/scrapers/drift-location.test.ts new file mode 100644 index 00000000..74caa9f6 --- /dev/null +++ b/tests/scrapers/drift-location.test.ts @@ -0,0 +1,127 @@ +/** + * Where a scraper says it saw something strange. + * + * Three tests drove a real scraper to a real drift log and every one of them + * asserted only the surprise TEXT. The two that did assert a `firstLocation` + * passed the string to `recordDrift` themselves, so they pinned the drift + * log's plumbing and nothing any scraper computed. Replacing the location with + * `${obj}` — the `[object Object]` already shipping in the antigravity + * client — left the suite green. + * + * That gap was hiding a live defect. The counter started at zero on every + * pass, but a resumed read starts at `startAt` BYTES: `readJsonlLines` yields + * only lines past the cursor, so a record appended as line 101 reported as + * `path:1`. Every location from an incremental scan pointed at the wrong + * place, and a drift location is the only pointer anyone has when chasing an + * upstream format break. + * + * It is now a byte offset, written `path@offset`, which means the same thing + * on a first read and a resumed one. + */ +import { appendFile, mkdir, mkdtemp, rm, writeFile } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { ClaudeCodeScraper } from "@xtctx/scrapers/claude-code"; +import { readDriftLog } from "@xtctx/scrapers/drift-log"; +import type { ClaudeCodeChunk } from "@xtctx/types/scraper"; + +const PROJECT = "H:/projects/demo"; + +const record = (text: string, ts: string): string => + JSON.stringify({ type: "human", content: text, timestamp: ts, cwd: PROJECT }); + +/** A record whose `type` the scraper does not know, which is what it reports. */ +const strange = (ts: string): string => + JSON.stringify({ type: "not-a-known-type", timestamp: ts, cwd: PROJECT }); + +let rootDir = ""; +let stateDir = ""; +let storeDir = ""; +let transcript = ""; + +async function drain(scraper: ClaudeCodeScraper, full: boolean): Promise { + const chunks: ClaudeCodeChunk[] = []; + for await (const chunk of full ? scraper.fullSync() : scraper.scrape()) { + chunks.push(chunk); + } + return chunks; +} + +function locations(log: Awaited>): string[] { + return (log?.surprises ?? []).map((entry) => entry.firstLocation); +} + +beforeEach(async () => { + rootDir = await mkdtemp(join(tmpdir(), "xtctx-driftloc-")); + stateDir = await mkdtemp(join(tmpdir(), "xtctx-driftloc-state-")); + // First argument is the projects directory itself, as the sibling tests do. + storeDir = join(rootDir, "H--projects-demo"); + await mkdir(storeDir, { recursive: true }); + transcript = join(storeDir, "session.jsonl"); +}); + +afterEach(async () => { + await rm(rootDir, { recursive: true, force: true }); + await rm(stateDir, { recursive: true, force: true }); +}); + +describe("a scraper's drift location", () => { + it("names the place in the file, not the place in the read", async () => { + await writeFile( + transcript, + [record("one", "2026-05-10T10:00:00.000Z"), strange("2026-05-10T10:00:01.000Z"), ""].join("\n"), + "utf-8", + ); + + await drain(new ClaudeCodeScraper(rootDir, stateDir, PROJECT), true); + + const reported = locations(await readDriftLog(stateDir, "claude-code")); + expect(reported.length).toBeGreaterThan(0); + // The real path, and a real position in it — not `[object Object]`, not a + // bare filename, not zero. + for (const location of reported) { + expect(location.startsWith(transcript)).toBe(true); + const at = Number(location.slice(transcript.length + 1)); + expect(Number.isFinite(at)).toBe(true); + expect(at).toBeGreaterThan(0); + } + }); + + it("does not restart its count after a resume", async () => { + // First pass: ordinary records only, so the cursor advances past them. + await writeFile( + transcript, + [ + record("one", "2026-05-10T10:00:00.000Z"), + record("two", "2026-05-10T10:00:01.000Z"), + record("three", "2026-05-10T10:00:02.000Z"), + "", + ].join("\n"), + "utf-8", + ); + + const scraper = new ClaudeCodeScraper(rootDir, stateDir, PROJECT); + await drain(scraper, false); + await scraper.saveScrapedPosition({ lastTimestamp: new Date("2026-05-10T10:00:02.000Z") }); + + const before = locations(await readDriftLog(stateDir, "claude-code")); + + // Now append the odd record, and scan again from the cursor. + await appendFile(transcript, `${strange("2026-05-10T10:00:03.000Z")}\n`, "utf-8"); + await drain(new ClaudeCodeScraper(rootDir, stateDir, PROJECT), false); + + const after = locations(await readDriftLog(stateDir, "claude-code")).filter( + (entry) => !before.includes(entry), + ); + expect(after.length).toBeGreaterThan(0); + + // The appended record sits past three complete records, so its reported + // position must too. Counting from the resume point produced a 1 here, + // which is what made every incremental drift location wrong. + for (const location of after) { + const at = Number(location.slice(transcript.length + 1)); + expect(at).toBeGreaterThan(100); + } + }); +}); diff --git a/tests/scrapers/opencode.test.ts b/tests/scrapers/opencode.test.ts index 89514ef2..bf5cce37 100644 --- a/tests/scrapers/opencode.test.ts +++ b/tests/scrapers/opencode.test.ts @@ -301,7 +301,9 @@ describe("OpenCodeScraper", () => { for await (const chunk of scraper.fullSync()) chunks.push(chunk); expect(chunks).toHaveLength(2); - expect(chunks.map((c) => c.sessionId).sort()).toEqual(["sess-A", "sess-B"]); + // No `.sort()`: sorting erases the very property this test is named for, + // so a reader returning sessions in any order passed it. + expect(chunks.map((c) => c.sessionId)).toEqual(["sess-A", "sess-B"]); }); it("respects since cursor and emits only newer chunks", async () => { From 6090fef24bfc163ece58aad7861c7400bb1e016a Mon Sep 17 00:00:00 2001 From: Felix Stubner Date: Mon, 21 Sep 2026 07:12:56 +0100 Subject: [PATCH 05/34] test: close the nine remaining findings from the third and fourth audits None was a live defect. All nine were tests that could not see the thing they were named for, which is the class the last two audits were looking for and the reason the earlier defects went unnoticed. The landing test pinned `v9.astro` -- a design draft carrying `noindex,nofollow`. The page visitors are actually sent to, `index.astro` and the components it composes, was pinned by nothing, so the product claims and the punctuation rule were enforced only on a page nobody reaches. Retargeting found two em dashes in live hero and install copy that the draft-only check had never looked at; both sentences are reworded. The test states plainly that it reads source rather than rendered output, and what that cannot catch. Dropped one assertion rather than ship it: banning the words "daemon" and "dashboard" flagged the page's own copy saying it has neither. Asserting the local-only commitment is present survives rewording; a substring ban on correct copy does not. The integration fixture's `searchSessions` took no parameters, so the query never had to reach it -- the payload echoes the handler's own local variable. Passing `""` instead of the caller's query left it green. It now records what it was asked for. The continuity disclosure test had two negative assertions over a fixture whose only path lived in `store_paths`, so both tested the same omission and returning `{sessions, messages}` passed. It now asserts the diagnostic is still a diagnostic, and a second case gives a tool a path-bearing `last_error` -- reaching both `sanitizeErrorMessage` calls on that surface, which no test had executed. The setup plan asserted only `kind` strings, so every path in the notice the user confirms could have been wrong. Paths are now pinned, and a new test compares the plan against what setup actually writes, since the two were independent lists that nothing reconciled. Also: `git_commit` now anchors on the scrubbed value, so a neutraliser is distinguishable from a deletion; the truncation test pins the real 16,000 boundary and the reported character count instead of a 20,000 bound with 4KB of slack; the Antigravity warning test asserts the global file was written and that a no-op rerun does not warn again; the opencode role-less skip pins the surviving turn's index; and the hostile-payload smoke test asserts nothing was recorded, not only that the exit code was zero. --- landing/src/data/site.ts | 4 +- src/mcp/tools/sessions.ts | 3 +- tests/config/setup.test.ts | 33 ++++++++ tests/integration/handoff-mcp.test.ts | 37 ++++++++- tests/landing/homepage-content.test.ts | 75 +++++++++++++++++++ tests/landing/v9-content.test.ts | 33 -------- tests/mcp/hardening.test.ts | 49 +++++++++++- .../opencode-scope-order-and-roles.test.ts | 6 ++ tests/smoke/hook-payload.smoke.test.ts | 6 ++ 9 files changed, 204 insertions(+), 42 deletions(-) create mode 100644 tests/landing/homepage-content.test.ts delete mode 100644 tests/landing/v9-content.test.ts diff --git a/landing/src/data/site.ts b/landing/src/data/site.ts index 8ee0fe30..8cc9922c 100644 --- a/landing/src/data/site.ts +++ b/landing/src/data/site.ts @@ -134,7 +134,7 @@ export const site: SiteData = { badge: 'Local transcript retrieval for AI coding tools', heading: 'Keep project context portable across coding tools.', subhead: - 'Move between supported coding agents without starting over. xtctx indexes the transcript files your tools already write and serves them over MCP, so the next agent can pick up recent context. Install the plugin and retrieval works — no project setup required.', + 'Move between supported coding agents without starting over. xtctx indexes the transcript files your tools already write and serves them over MCP, so the next agent can pick up recent context. Install the plugin and retrieval works, with no project setup required.', proof: ['No project setup required', 'Raw transcripts stay local', 'Five MCP tools'], quickInstall: 'claude plugin marketplace add fstubner/xtctx && claude plugin install xtctx@xtctx', installLinkLabel: 'Get started', @@ -244,7 +244,7 @@ export const site: SiteData = { label: 'Install the plugin', command: 'claude plugin marketplace add fstubner/xtctx && claude plugin install xtctx@xtctx', hint: - 'Registers the MCP server and the handoff skill, and writes nothing into your project. Retrieval works straight away — the tools resolve the project from the working directory. Codex, Copilot, Cursor and Antigravity install from the same marketplace; see the README for their commands.', + 'Registers the MCP server and the handoff skill, and writes nothing into your project. Retrieval works straight away, because the tools resolve the project from the working directory. Codex, Copilot, Cursor and Antigravity install from the same marketplace; see the README for their commands.', }, { label: 'Add project wiring', diff --git a/src/mcp/tools/sessions.ts b/src/mcp/tools/sessions.ts index 51d9c467..a007c3e4 100644 --- a/src/mcp/tools/sessions.ts +++ b/src/mcp/tools/sessions.ts @@ -30,7 +30,8 @@ export type { SessionService }; export class ToolInputError extends Error {} /** Hard cap on a single message body returned to the model. */ -const MAX_MESSAGE_CHARS = 16_000; +/** @internal Exported so the budget test can pin the real boundary. */ +export const MAX_MESSAGE_CHARS = 16_000; /** * A filter the caller got wrong is refused, not ignored. diff --git a/tests/config/setup.test.ts b/tests/config/setup.test.ts index 3daab518..f8e30894 100644 --- a/tests/config/setup.test.ts +++ b/tests/config/setup.test.ts @@ -181,6 +181,39 @@ describe("setupProject", () => { "skill:claude-code:xtctx-handoff", ]), ); + + // The PATHS, not only the kinds. This is the user's one chance to see + // which files are about to change, and asserting kinds alone let every + // path in the plan be wrong — the plan is a hand-kept literal list, never + // compared to what setup writes, which is the hazard + // `disconnect-planned-paths.test.ts` exists for on the other side. + const planned = new Map(plan.writes.map((write) => [write.kind, write.path])); + expect(planned.get("config")).toBe(join(projectRoot, ".xtctx", "config.yaml")); + expect(planned.get("mcp:claude-code")).toBe(join(projectRoot, ".mcp.json")); + expect(planned.get("mcp:cursor")).toBe(join(projectRoot, ".cursor", "mcp.json")); + expect(planned.get("mcp:copilot")).toBe(join(projectRoot, ".vscode", "mcp.json")); + expect(planned.get("mcp:codex")).toBe(join(projectRoot, ".codex", "config.toml")); + expect(planned.get("hook:claude-code")).toBe( + join(projectRoot, ".claude", "settings.json"), + ); + }); + + it("plans every file setup actually writes", async () => { + // The plan and the writes are two independent lists. Nothing compared + // them, so a file setup touches could be absent from the notice the user + // confirms — which is the whole point of showing it. + const plan = describeSetupPlan(projectRoot); + const result = await setupProject({ projectPath: projectRoot, homeDir, yes: true }); + + const planned = new Set(plan.writes.map((write) => write.path)); + const unannounced = result.writes + .map((write) => write.path) + // Global configs are only planned with `includeGlobalMcp`, and setup + // only writes them under the same flag; both are off here. + .filter((path) => path.startsWith(projectRoot)) + .filter((path) => !planned.has(path)); + + expect(unannounced).toEqual([]); }); it("writes Copilot CLI global MCP only when explicitly requested", async () => { diff --git a/tests/integration/handoff-mcp.test.ts b/tests/integration/handoff-mcp.test.ts index 0b3b8fce..5a387dc5 100644 --- a/tests/integration/handoff-mcp.test.ts +++ b/tests/integration/handoff-mcp.test.ts @@ -1,6 +1,12 @@ import { describe, expect, it } from "vitest"; import { createToolHandlers } from "@xtctx/mcp/server"; -import type { HandoffStatus, SessionMessage, SessionService, SessionSummary } from "@xtctx/handoff/types"; +import type { + HandoffStatus, + SessionMessage, + SessionSearchMode, + SessionService, + SessionSummary, +} from "@xtctx/handoff/types"; class FixtureSessionService implements SessionService { async listRecentSessions(): Promise { @@ -31,7 +37,24 @@ class FixtureSessionService implements SessionService { ]; } - async searchSessions(): Promise { + /** + * What the handler actually asked for. + * + * This took no parameters and returned the recent list, so the query never + * had to reach it: the JSON payload echoes the handler's own local variable, + * and changing `sessions.ts` to pass `""` instead of the caller's query left + * the test green. Recording the arguments is what makes the assertion about + * the code rather than about the fixture. + */ + lastSearch?: { query: string; limit: number; mode?: string }; + + async searchSessions( + query: string, + limit: number, + _toolFilter?: string[], + mode?: SessionSearchMode, + ): Promise { + this.lastSearch = { query, limit, mode }; return this.listRecentSessions(); } @@ -100,12 +123,18 @@ describe("handoff MCP integration", () => { }); }); - it("searches indexed content", async () => { + it("passes the caller's query through to the index", async () => { + const service = new FixtureSessionService(); + const map = createToolHandlers({ sessions: service }); + await expect( - handlers().get("xtctx_search_sessions")?.({ query: "setup", format: "json" }), + map.get("xtctx_search_sessions")?.({ query: "setup", limit: 3, format: "json" }), ).resolves.toMatchObject({ sessions: [{ session_ref: "codex:session-a" }], }); + + expect(service.lastSearch?.query).toBe("setup"); + expect(service.lastSearch?.limit).toBe(3); }); it("reports continuity status", async () => { diff --git a/tests/landing/homepage-content.test.ts b/tests/landing/homepage-content.test.ts new file mode 100644 index 00000000..6bacf001 --- /dev/null +++ b/tests/landing/homepage-content.test.ts @@ -0,0 +1,75 @@ +/** + * The claims on the page visitors actually see. + * + * This used to read `v9.astro`, which is a design draft carrying + * `noindex,nofollow`. The real homepage is `index.astro` — Nav, Hero, + * Workflow, Surfaces, Install, Faq, Footer, all fed from `data/site.ts` — and + * nothing pinned it, so the product claims and the punctuation rules were + * enforced only on a page nobody is sent to. Retargeting immediately found two + * em dashes in live hero and install copy that the draft-only check had never + * looked at. + * + * Source text, not rendered output, and the limit of that is worth stating: a + * claim wrapped in `{false && (…)}` would still satisfy every assertion below. + * Rendering means an astro build, which costs more in the unit suite than it + * buys here — `npm run landing:build` already runs in `verify:release` and in + * CI, so a page that cannot build is caught there. + */ +import { readFile } from "node:fs/promises"; +import { join } from "node:path"; +import { describe, expect, it } from "vitest"; + +const LANDING = join(process.cwd(), "landing", "src"); + +/** Every source file that composes the homepage, concatenated. */ +async function homepageSources(): Promise { + const files = [ + join(LANDING, "pages", "index.astro"), + join(LANDING, "data", "site.ts"), + ...["Nav", "Hero", "Workflow", "Surfaces", "Install", "Faq", "Footer", "Schematic", "IdeMock", "TerminalMock"].map( + (name) => join(LANDING, "components", `${name}.astro`), + ), + ]; + const parts = await Promise.all(files.map((file) => readFile(file, "utf-8"))); + return parts.join("\n"); +} + +describe("the homepage visitors are sent to", () => { + it("names the five MCP tools the product actually exposes", async () => { + const page = await homepageSources(); + + for (const tool of [ + "xtctx_recent_sessions", + "xtctx_session_detail", + "xtctx_search_sessions", + "xtctx_continuity_status", + "xtctx_handoff_manifest", + ]) { + expect(page, `homepage should name ${tool}`).toContain(tool); + } + }); + + it("gives the install command the README gives", async () => { + const page = await homepageSources(); + + expect(page).toContain("npx -y xtctx setup"); + }); + + it("still says the thing that makes the product what it is", async () => { + const page = await homepageSources(); + + // PRODUCT.md's first constraint. Deliberately not a ban on the words + // "daemon" or "dashboard": the page uses them to say it has neither, and a + // substring check flagged that correct copy as a false promise. Asserting + // the commitment is present is the check that survives rewording of the + // sentence around it. + expect(page.toLowerCase()).toContain("local-only"); + }); + + it("keeps em and en dashes out of the copy", async () => { + const page = await homepageSources(); + + expect(page).not.toContain("—"); + expect(page).not.toContain("–"); + }); +}); diff --git a/tests/landing/v9-content.test.ts b/tests/landing/v9-content.test.ts deleted file mode 100644 index 2928c589..00000000 --- a/tests/landing/v9-content.test.ts +++ /dev/null @@ -1,33 +0,0 @@ -import { readFile } from "node:fs/promises"; -import { join } from "node:path"; -import { describe, expect, it } from "vitest"; - -describe("v9 landing content", () => { - it("keeps onboarding, skill sync, and runtime claims aligned with the product", async () => { - const page = await readFile(join(process.cwd(), "landing", "src", "pages", "v9.astro"), "utf-8"); - - expect(page).toContain("npx -y xtctx setup"); - expect(page).toContain("Move between coding agents without starting over"); - expect(page).toContain("Skill sync"); - expect(page).toContain("Does xtctx sync skills?"); - expect(page).toContain("What are the limits?"); - expect(page).toContain("npm run demo:public"); - expect(page).toContain("Setup writes config. Tools call MCP."); - expect(page).toContain("What setup writes"); - expect(page).toContain("Selected skills stay under .xtctx/skills"); - expect(page).toContain("The next tool gets the same project instructions"); - expect(page).toContain("xtctx_recent_sessions"); - expect(page).toContain("xtctx_session_detail"); - expect(page).toContain("xtctx_search_sessions"); - expect(page).toContain("xtctx_continuity_status"); - expect(page).toContain("xtctx_handoff_manifest"); - expect(page).toContain("AntigravityMCP + GEMINI.md"); - expect(page).not.toContain("Raw transcripts"); - expect(page).not.toContain("Semantic windows"); - expect(page).not.toContain("Tool disconnect"); - expect(page).not.toContain("supported startup hooks update"); - expect(page).not.toContain("startup hooks update"); - expect(page).not.toContain("—"); - expect(page).not.toContain("–"); - }); -}); diff --git a/tests/mcp/hardening.test.ts b/tests/mcp/hardening.test.ts index 7fcd22c9..3ab8f697 100644 --- a/tests/mcp/hardening.test.ts +++ b/tests/mcp/hardening.test.ts @@ -4,6 +4,7 @@ import { createRecentSessionsHandler, createSearchSessionsHandler, createSessionDetailHandler, + MAX_MESSAGE_CHARS, } from "@xtctx/mcp/tools/sessions"; import { sanitizeErrorMessage } from "@xtctx/utils/errors"; import type { @@ -186,8 +187,13 @@ describe("response byte budgets", () => { messages: SessionMessage[]; }; - expect(result.messages[0].content.length).toBeLessThan(20_000); - expect(result.messages[0].content).toContain("truncated"); + // The real cap, not a loose bound above it. `< 20_000` against a cap of + // 16_000 left 4KB of slack, and nothing pinned the surviving prefix: a + // truncation keeping one character would have passed. + const content = result.messages[0].content; + expect(content.startsWith("a".repeat(MAX_MESSAGE_CHARS))).toBe(true); + expect(content).toContain(`truncated ${64_000 - MAX_MESSAGE_CHARS} chars`); + expect(content.length).toBeLessThan(MAX_MESSAGE_CHARS + 100); }); }); @@ -199,6 +205,45 @@ describe("status path disclosure", () => { expect(JSON.stringify(result)).not.toContain("store_paths"); expect(JSON.stringify(result)).not.toContain("/home/user"); + + // Both assertions above are negative, and the fixture's only `/home/user` + // lives inside `store_paths` — so they tested the same omission twice, and + // returning `{ sessions, messages }` and nothing else passed both. The + // diagnostic has to still be a diagnostic: this is the part that says the + // omission is a redaction rather than an empty payload. + expect(result).toMatchObject({ + vector_model: "fixture", + tools: [{ tool: "codex", detected: true }], + }); + }); + + it("redacts a path carried inside a tool's error message", async () => { + // The path that matters is not always in `store_paths`. `last_error` is a + // raw error string from a scraper, and the fixture set it to null — so + // both `sanitizeErrorMessage` calls on this surface were dead code the + // test never reached. + class FailingToolService extends DetailFixtureService { + async getStatus(): Promise { + const status = await super.getStatus(); + return { + ...status, + tools: [ + { + ...status.tools[0], + last_error: "EACCES: permission denied, open '/home/user/.codex/sessions/a.jsonl'", + }, + ], + }; + } + } + + const handlers = createToolHandlers({ sessions: new FailingToolService([]) }); + const result = await handlers.get("xtctx_continuity_status")?.({ format: "json" }); + const payload = JSON.stringify(result); + + expect(payload).not.toContain("/home/user"); + // Redacted, not dropped: the operator still has to learn the store failed. + expect(payload).toContain("EACCES"); }); }); diff --git a/tests/scrapers/opencode-scope-order-and-roles.test.ts b/tests/scrapers/opencode-scope-order-and-roles.test.ts index bdc911b4..ec9c71d4 100644 --- a/tests/scrapers/opencode-scope-order-and-roles.test.ts +++ b/tests/scrapers/opencode-scope-order-and-roles.test.ts @@ -273,5 +273,11 @@ describe("opencode skips a message whose role field has gone", () => { // `normalizeRole` returns for anything it cannot read. expect(out.map((c) => c.content)).toEqual(["still readable"]); expect(warnings.some((w) => w.includes("missing 'role' field"))).toBe(true); + + // The skipped record must not consume an index. `scan.ts` hashes the + // index into the row id, so a skip that advances it renumbers every later + // turn — the defect already fixed in the cursor reader, and the only skip + // in this loop whose index nothing pinned. + expect(out[0]?.metadata.messageIndex).toBe(0); }); }); diff --git a/tests/smoke/hook-payload.smoke.test.ts b/tests/smoke/hook-payload.smoke.test.ts index d620cc49..a4ace8e8 100644 --- a/tests/smoke/hook-payload.smoke.test.ts +++ b/tests/smoke/hook-payload.smoke.test.ts @@ -171,5 +171,11 @@ describe("session-start hook payload trust", () => { // The hook runs inside the host agent's startup. Rejecting a payload must // never become a startup error there. expect(await runHook({ cwd: 42, transcript_path: ["not", "a", "string"] })).toBe(0); + + // Exiting 0 is only half of it. A hook that ACTED on the hostile payload + // and then exited cleanly passes the line above while having recorded a + // store directory it was never given — which is the failure the sibling + // cases here exist to catch, and the one an exit code cannot see. + expect(await recordedStoreDirs()).toBeNull(); }); }); From a2fa63af08d94d77a09ba2b6cf4bc472a43574a1 Mon Sep 17 00:00:00 2001 From: Felix Stubner Date: Mon, 21 Sep 2026 09:57:34 +0100 Subject: [PATCH 06/34] fix: stop the product claiming retrieval works without setup MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The landing page's central pitch and the published plugin's skill text both said the same false thing, in five places between them: install the plugin and retrieval works anywhere, no setup required. It does not. A project with no `.xtctx/config.yaml` gets `present: false`, and `createToolHandlers` then points all five tools at `notConfigured()`. README.md has always said so in its own table -- "Retrieval in an unconfigured project: no (offers setup)" -- so the two surfaces a new user actually meets were the two that disagreed with it. The skill matters most: it is instructions to an agent, and it told them a missing config affected only instruction blocks and the retrieval tools were "worth calling anyway". An agent following that calls five tools that will not answer and reports the project has no history. The hero terminal demo showed the same impossible sequence -- install plugin, cd into a project, get sessions back -- and now runs setup in between. Tests pin both surfaces against the claim returning. Found in the same pass, all verified against the code they contradict: - The Copy button on the three feature cards copied sample OUTPUT as if it were commands, so a user pasting it ran `updated`, `configured` and `├──` in their shell. Those blocks no longer offer copy. - The cache diagram named three tables that do not exist (`retrieval_windows`, `vectors`, `fts_index`); the real ones are `retrieval_units`, `retrieval_units_fts`, `retrieval_unit_vectors`. - The IDE mock labelled an `AGENTS.md` managed block `Tool: cursor`. AGENTS.md is Codex's file; Cursor's block goes to `.cursor/rules/`. - `v9.astro` carried `noindex,nofollow` and then `index,follow` in the same head, so the draft could be indexed as a near-duplicate of the homepage. - `Schematic.astro` was imported by nothing, and was wrong where it could be read at all: "Antigravity" twice, and tool names (`mcp_recent_sessions`) that have never existed. Deleted. Also fixed, from a line-by-line read of the CLI and index: - A literal search waited out the refresh budget for a scan it never reads. That route exists to answer while the index is still filling, and on the defaults it sat four seconds behind a scan before starting its own five. It now starts the scan and moves on; measured 2025ms before, under 1500ms after. - `readStdinWithin` removed its listeners but left stdin flowing, so the hang it exists to prevent still held the process open -- measured at 6012ms against a pipe held for six seconds, with the function itself resolving at 250ms. - `CLAUDE_CONFIG_DIR` was named in a comment as the containment root and read by nothing, so with it set the real transcript path was refused and the scraper searched a `~/.claude` holding nothing. - A relative `storePath` resolved against the process's working directory, so `xtctx status -p X` from elsewhere read a different store than the server does. - `scan --embed` exited 0 while reporting it had not finished. - The release job left the admin PAT in `.git/config` for the whole of `npm ci` and `verify:release`; it now reaches git only at the push. - The upstream watcher deduped on a title with no version in it, searched across all states, so one issue ever silenced it permanently -- and a failing `gh` left the count empty, which also read as "already reported". --- .github/workflows/release.yml | 15 ++- .github/workflows/upstream-watch.yml | 36 ++++-- landing/src/components/IdeMock.astro | 2 +- landing/src/components/Schematic.astro | 94 ---------------- landing/src/components/Surfaces.astro | 6 +- landing/src/components/TerminalMock.astro | 9 +- landing/src/data/site.ts | 14 +-- landing/src/pages/v9.astro | 1 - plugin/skills/xtctx-handoff/SKILL.md | 9 +- src/cli/hook.ts | 8 ++ src/cli/scan.ts | 4 + src/config/skills.ts | 9 +- src/handoff/sqlite-index.ts | 32 +++++- src/runtime/services.ts | 16 ++- src/tools/sources.ts | 10 ++ tests/cli/hook-store-dir-containment.test.ts | 29 +++++ .../literal-search-does-not-wait.test.ts | 105 ++++++++++++++++++ tests/landing/homepage-content.test.ts | 19 +++- tests/release/plugin-package.test.ts | 23 ++++ 19 files changed, 306 insertions(+), 135 deletions(-) delete mode 100644 landing/src/components/Schematic.astro create mode 100644 tests/handoff/literal-search-does-not-wait.test.ts diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 6f7b4d59..6d5f10dc 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -95,6 +95,12 @@ jobs: # one manually dispatched workflow, while `GITHUB_TOKEN` is present # in every run of every workflow in the repository. token: ${{ secrets.RELEASE_PLEASE_TOKEN }} + # Not left in `.git/config`. With the default, the admin PAT sits on + # disk while `npm ci`, `npm --prefix landing ci` and the whole + # `verify:release` suite run — third-party install scripts and every + # test in the repository, any one of which could read it. Only the + # push at the end needs it, and it is supplied there. + persist-credentials: false - name: Setup Node uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0 @@ -185,6 +191,7 @@ jobs: env: VERSION: ${{ steps.bump.outputs.version }} TAG: ${{ steps.bump.outputs.tag }} + RELEASE_TOKEN: ${{ secrets.RELEASE_PLEASE_TOKEN }} run: | set -euo pipefail git config user.name "github-actions[bot]" @@ -195,8 +202,12 @@ jobs: git add package.json package-lock.json CHANGELOG.md plugin .claude-plugin landing/src/data/site.ts git commit -m "chore(release): ${VERSION}" git tag "$TAG" - git push origin "HEAD:${GITHUB_REF_NAME}" - git push origin "$TAG" + # The token reaches git here and nowhere else; see + # `persist-credentials: false` on the checkout. Through the + # environment rather than `${{ }}`, so it is never substituted into + # the script text. + git push "https://x-access-token:${RELEASE_TOKEN}@github.com/${GITHUB_REPOSITORY}.git" "HEAD:${GITHUB_REF_NAME}" + git push "https://x-access-token:${RELEASE_TOKEN}@github.com/${GITHUB_REPOSITORY}.git" "$TAG" - name: Create the GitHub release env: diff --git a/.github/workflows/upstream-watch.yml b/.github/workflows/upstream-watch.yml index 7cdcebf8..81029656 100644 --- a/.github/workflows/upstream-watch.yml +++ b/.github/workflows/upstream-watch.yml @@ -60,25 +60,47 @@ jobs: echo "**Run:** ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}" } > issue-body.md - # Dedupe on the exact version pair in the title, and across *all* states: - # a closed issue means the human already looked at that release, so - # reopening it nightly would recreate the noise this replaces. + # The version pair has to be IN the title for the dedupe below to mean + # anything. It was a fixed string with no versions in it, searched across + # all states — so one issue, ever, silenced this watcher permanently: + # every later upstream release matched that title and filed nothing. + - name: Build the title for this release set + id: title + if: steps.check.outputs.exit_code == '10' + run: | + set -euo pipefail + # Each MOVED line reads "MOVED : -> "; the joined + # "to" versions identify this particular set of releases. + versions=$(grep '^MOVED' upstream-report.txt | sed 's/.*-> //' | paste -sd, -) + echo "value=upstream: tracked coding tools released (${versions})" >> "$GITHUB_OUTPUT" + + # Across *all* states: a closed issue means the human already looked at + # THAT release set, so reopening it nightly would recreate the noise this + # replaces. - name: Check whether this release was already reported id: existing if: steps.check.outputs.exit_code == '10' env: GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + TITLE: ${{ steps.title.outputs.value }} run: | - title="upstream: tracked coding tool released — check transcript formats" - found=$(gh issue list --state all --search "\"$title\" in:title" --json number,title,state \ - | jq -r --arg t "$title" '[.[] | select(.title == $t)] | length') + # `pipefail`, because a failing `gh` left `count` empty — which is + # not '0', which reads as "already reported". An API error silenced + # the watcher instead of failing the run. + set -euo pipefail + found=$(gh issue list --state all --search "\"$TITLE\" in:title" --json number,title,state \ + | jq -r --arg t "$TITLE" '[.[] | select(.title == $t)] | length') + if [ -z "$found" ]; then + echo "could not determine whether this release set was already reported" >&2 + exit 1 + fi echo "count=${found}" >> "$GITHUB_OUTPUT" - name: Open an issue if: steps.check.outputs.exit_code == '10' && steps.existing.outputs.count == '0' uses: peter-evans/create-issue-from-file@fca9117c27cdc29c6c4db3b86c48e4115a786710 # v6.0.0 with: - title: "upstream: tracked coding tool released — check transcript formats" + title: ${{ steps.title.outputs.value }} content-filepath: issue-body.md labels: | drift diff --git a/landing/src/components/IdeMock.astro b/landing/src/components/IdeMock.astro index e1b19cd1..ddbd71af 100644 --- a/landing/src/components/IdeMock.astro +++ b/landing/src/components/IdeMock.astro @@ -52,7 +52,7 @@
<!-- xtctx:begin -->
 # xtctx Handoff
 
-Tool: cursor
+Tool: codex
 Project root: ~/projects/my-app
 Integration mode: instruction-only
 
diff --git a/landing/src/components/Schematic.astro b/landing/src/components/Schematic.astro
deleted file mode 100644
index af1416b2..00000000
--- a/landing/src/components/Schematic.astro
+++ /dev/null
@@ -1,94 +0,0 @@
-
-
- Fig 1 - Handoff map -
- - - xtctx handoff map - - Supported coding tools write local transcripts. xtctx indexes recent sessions - into SQLite and exposes retrieval through MCP. - - - - - - - - - - - - - - - - - - - Local tools - Codex - Claude Code - Cursor - Antigravity - Antigravity - opencode - - - - - xtctx - setup - status - syncs skills - MCP Server - - - - - SQLite cache - sessions & logs - sync-drift vectors - - - - - Next agent - mcp_recent_sessions - mcp_session_detail - - - - - - - - reads logs - indexes - 1. query - 2. handoff - - - - - - - - - - - - - - - - - - - -
- Source transcripts stay local. MCP calls pull only the session detail the - next agent needs. -
-
diff --git a/landing/src/components/Surfaces.astro b/landing/src/components/Surfaces.astro index 579dc960..c3b400a9 100644 --- a/landing/src/components/Surfaces.astro +++ b/landing/src/components/Surfaces.astro @@ -53,8 +53,12 @@ const subFeatures = surfaces.slice(1);

{f.title}

+ {/* No `has-copy`: these blocks are sample OUTPUT, not commands. + The copy handler only strips a leading `$ `, so copying one + handed the user lines like `updated .xtctx/config.yaml` and + `├── sessions` to paste into a shell. */} {f.codeHtml && ( -

+
)} diff --git a/landing/src/components/TerminalMock.astro b/landing/src/components/TerminalMock.astro index 4f91425c..f821eb64 100644 --- a/landing/src/components/TerminalMock.astro +++ b/landing/src/components/TerminalMock.astro @@ -21,7 +21,14 @@
- $ cd ~/projects/my-app && claude + $ cd ~/projects/my-app && npx -y xtctx setup +
+
+
[ok] mcp config, managed blocks, session hook
+
+ +
+ $ claude
> what was I working on?
diff --git a/landing/src/data/site.ts b/landing/src/data/site.ts index 8cc9922c..8b215508 100644 --- a/landing/src/data/site.ts +++ b/landing/src/data/site.ts @@ -134,8 +134,8 @@ export const site: SiteData = { badge: 'Local transcript retrieval for AI coding tools', heading: 'Keep project context portable across coding tools.', subhead: - 'Move between supported coding agents without starting over. xtctx indexes the transcript files your tools already write and serves them over MCP, so the next agent can pick up recent context. Install the plugin and retrieval works, with no project setup required.', - proof: ['No project setup required', 'Raw transcripts stay local', 'Five MCP tools'], + 'Move between supported coding agents without starting over. xtctx indexes the transcript files your tools already write and serves them over MCP, so the next agent can pick up recent context. Install the plugin once to reach the tools everywhere, then opt each project in with a single setup command.', + proof: ['One command per project', 'Raw transcripts stay local', 'Five MCP tools'], quickInstall: 'claude plugin marketplace add fstubner/xtctx && claude plugin install xtctx@xtctx', installLinkLabel: 'Get started', sourceUrl: REPO_URL, @@ -232,9 +232,9 @@ export const site: SiteData = { codeHtml: `cache .xtctx/state/xtctx.db ├── sessions ├── messages -├── retrieval_windows -├── vectors -└── fts_index`, +├── retrieval_units +├── retrieval_units_fts +└── retrieval_unit_vectors`, }, ], @@ -244,7 +244,7 @@ export const site: SiteData = { label: 'Install the plugin', command: 'claude plugin marketplace add fstubner/xtctx && claude plugin install xtctx@xtctx', hint: - 'Registers the MCP server and the handoff skill, and writes nothing into your project. Retrieval works straight away, because the tools resolve the project from the working directory. Codex, Copilot, Cursor and Antigravity install from the same marketplace; see the README for their commands.', + 'Registers the MCP server and the handoff skill machine-wide, and writes nothing into your project. The tools then reach every project; each one answers once it has been set up, and names the command until then. Codex, Copilot, Cursor and Antigravity install from the same marketplace; see the README for their commands.', }, { label: 'Add project wiring', @@ -281,7 +281,7 @@ export const site: SiteData = { }, { q: 'Do I need to run setup in every project?', - a: 'No. With the plugin installed, the MCP tools resolve the project from the working directory, so retrieval works in any project with no setup at all. What setup adds is delivery: managed instruction blocks put the handoff in front of the next agent whether it calls a tool or not. Plugin first, setup where you want it automatic.', + a: 'Yes, once per project you want handoff in. The plugin makes the tools reachable everywhere, but a project that has not opted in has no index to read, so every tool answers with that and names `npx -y xtctx setup`. Setup also adds delivery: managed instruction blocks put the handoff in front of the next agent whether it calls a tool or not.', }, { q: 'Does xtctx run a background service?', diff --git a/landing/src/pages/v9.astro b/landing/src/pages/v9.astro index 0a2d3221..345d816f 100644 --- a/landing/src/pages/v9.astro +++ b/landing/src/pages/v9.astro @@ -159,7 +159,6 @@ const ldJson = JSON.stringify( name="keywords" content="MCP server, Model Context Protocol, AI coding agents, local transcript retrieval, skill sync, Codex, Claude Code, Cursor, Antigravity, opencode" /> - Generated by xtctx setup. Do not edit inside this block. diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index a123cb33..e6057df6 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -57,9 +57,10 @@ there is no daemon to leave behind. **Serving a call.** A coding agent spawns `npx -y xtctx` over stdio, gets the five read-only tools, and the process exits when the agent is done with it. -On the first call that needs data, the index refreshes: every scraper reads -its own tool's store, yields only chunks attributable to this project, and the -results land in `.xtctx/state/xtctx.db`. +The server starts a scan as it starts, and refreshes again on a call whose +indexed view has gone stale: every scraper reads its own tool's store, yields +only chunks attributable to this project, and the results land in +`.xtctx/state/xtctx.db`. **How a conversation becomes searchable.** Messages are grouped into overlapping retrieval windows — eight messages, stride four — so a hit carries @@ -71,7 +72,7 @@ twice: into FTS5 for keyword search, and as one embedding vector. comes from bm25 ordering but is rescored as a linear decay, because bm25 favours short documents and a one-line mention was outranking the paragraph that decided something. Semantic matches are gated twice: a per-window floor -(0.15) and a per-query confidence floor (0.4). When nothing clears the second +(0.15) and a per-query confidence floor (0.36). When nothing clears the second one, semantic results are dropped wholesale and only keyword hits remain — whether a query found anything is a property of the query, not of each window, and no answer beats a confident wrong one. @@ -90,10 +91,11 @@ queue. **Bounded, so a tool call always returns.** Scanning gets four seconds, vectorizing six, and an indexed view is treated as current for thirty. Work -left over resumes on the next call. The embedding model loads lazily and only -for semantic search; `hybrid` deliberately answers from keyword while it is -still loading, so the first call after a cold start is fast rather than -blocked. +left over resumes on the next call. A scan also warms the embedding model and +builds vectors under the same cap, because a process spawned per agent session +would otherwise never vectorize anything; `hybrid` deliberately answers from +keyword while the model is still loading, so the first call after a cold start +is fast rather than blocked. **What comes back is raw.** Sessions, message text, and pointers — never a generated summary. A recap is the lossy artefact this exists to replace, and diff --git a/CHANGELOG.md b/CHANGELOG.md index 9e0c6a15..f4190ffa 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -133,13 +133,13 @@ Entries are written by the `release` workflow when a release is cut by hand. ### Miscellaneous -* An automated release pipeline cut 112 versions between 0.19.1 and 0.74.0 in +* An automated release pipeline cut 76 release commits between 0.18.7 and 0.74.0 in a few hours on 2026-08-29, none of which anyone asked for and none of which were published to npm. The version was reset to 0.20.0 the same day and the pipeline was replaced by a manually-triggered release; see the comments in `.github/workflows/release.yml`. The individual "release xtctx X.Y.Z" entries those runs generated are summarised here rather than listed, because - a reader scanning this file for what shipped was being shown 112 versions + a reader scanning this file for what shipped was being shown versions that never existed. diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index caf6d964..4b341009 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -42,7 +42,9 @@ Or run everything with: 1. Keep changes focused and atomic. 2. Add tests for behavior changes. 3. Update docs when CLI or MCP behavior changes. -4. Use conventional commits for release automation: +4. Use conventional commits. Nothing reads the prefix — release notes come + from GitHub's own generator over the commit range — but a reader scanning + the log does: - `feat: ...` - `fix: ...` - `docs: ...` diff --git a/PRODUCT.md b/PRODUCT.md index e3bc55e6..370436f4 100644 --- a/PRODUCT.md +++ b/PRODUCT.md @@ -36,7 +36,12 @@ Single-user, single-machine. There is no team, sync, or server component. config file that setup created may be left behind holding an empty server map. - Only the current project's sessions are ever indexed or served — content - from other projects on the machine never crosses the project boundary. + from other projects on the machine never crosses the project boundary — with + one exception the product states rather than hides: a committable + `.xtctx/config.yaml` can point a tool's `storePath` somewhere else, and that + redirect is reported by `xtctx status` and `xtctx_continuity_status` rather + than blocked. A cloned repository can carry one, which is why it is + surfaced. - The demo smoke (`npm run demo:public`) proves the loop end-to-end against synthetic data on every release. diff --git a/README.md b/README.md index c832a44c..502834ba 100644 --- a/README.md +++ b/README.md @@ -93,7 +93,15 @@ Plugins standard the package above is built against, so `setup` is the only route there. Either route registers the same MCP server (`npx -y xtctx`) and the same -handoff skill. Because the plugin writes no project config, `xtctx status` +handoff skill. + +One thing to know about the plugin specifically: it is installed from this +repository, so its skill text comes from `main`, while the server it launches +is whatever `npx -y xtctx` resolves to on npm. Those are not the same commit +whenever work has landed but not been released — which is the normal state +here — so a plugin install can describe behaviour the server it runs does not +have yet. `xtctx status` reports a skill copy that predates the built-in one; +it cannot see the server's version from the other side. Because the plugin writes no project config, `xtctx status` reports a plugin-only project as `Config missing (run xtctx setup)`, and the tools answer the same way until `setup` has been run there. @@ -132,7 +140,9 @@ blocks where that tool owns them, removes supported startup hooks, and marks the tool disabled in `.xtctx/config.yaml`. It removes generated skill adapters for that tool. It does not delete transcript sources, canonical project skills, or the local SQLite cache. Use `xtctx disconnect --all` to remove xtctx from every -supported tool. Antigravity and Copilot CLI keep one MCP config for every +supported tool — that one also deletes `.xtctx/skills`, since with nothing left +managing skills the synced source is xtctx's own scaffolding. A skill you +wrote yourself and selected at setup is kept where you wrote it. Antigravity and Copilot CLI keep one MCP config for every project on the machine, so a project disconnect leaves those two files alone; pass `--global-mcp` (as with `setup`) to remove xtctx from them as well. @@ -310,4 +320,4 @@ no `release: published` trigger. It had one once, with releases drafted so nothing published itself, and that broke outright: GitHub's `releases/latest` endpoint hides drafts, the release tooling read that endpoint to find the last release, so it saw a pre-draft version forever and proposed a release covering -the entire history. It cut 54 versions in an hour. +the entire history. It cut 76 versions over four days, 49 of them in one day. diff --git a/RELEASE.md b/RELEASE.md index 3ed32bf7..157c52a7 100644 --- a/RELEASE.md +++ b/RELEASE.md @@ -25,7 +25,8 @@ when someone runs it. To publish a version that was tagged earlier but never reached npm, dispatch `publish` on its own against that tag, typing `publish` to confirm. That is -not hypothetical: this repo once sat nine versions tagged-but-unpublished. +not hypothetical: this repo is in that state now, and has been since 0.19.0 — +fifteen tagged versions that npm has never served. A release is **not done** until `post-publish-smoke` is green. @@ -44,7 +45,8 @@ the release process outright. GitHub's `releases/latest` endpoint hides drafts, Release Please read it to find the last release, so it saw the last pre-draft version forever, proposed a release covering the entire history, auto-merge landed it, and the resulting draft was invisible again. That loop -cut 54 versions in an hour before anyone noticed. +cut 76 versions over four days, 49 of them on 2026-08-29 alone and 12 in its +busiest hour, before anyone noticed. A per-day ceiling was tried after that and was the wrong shape: capping unwanted releases still leaves them unwanted. The automatic path was removed @@ -79,7 +81,8 @@ the transcript index, which is derived data). - [ ] `post-publish-smoke` job is green on the publish run - [ ] `npx -y xtctx@latest --version` prints the new version -- [ ] `npm run demo:public` passes against the released build +- [ ] `npm run demo:public` passes against this checkout (it imports the local + `dist/`, not the published package) - [ ] Landing site footer shows the new version (synced by the `version` script; see `landing/src/data/site.ts` and `scripts/sync-version.mjs`) diff --git a/docs/embedding-performance.md b/docs/embedding-performance.md index 2210ac06..d562058b 100644 --- a/docs/embedding-performance.md +++ b/docs/embedding-performance.md @@ -8,6 +8,26 @@ guesses. Three of the four guesses were wrong, which is the main reason this file is worth keeping. +## What is reproducible here, and what is not + +Only one figure below traces to something committed: the MiniLM eval baseline +(hybrid 0.333 / 0.533 / 0.183), which is +`tests/eval/results/ranking-baseline.json` and is regenerated by +`npm run test:eval`. + +Everything else — the device table, the batch sweep, the dtype comparison, the +segment-duplication count, and the bge/gte rows — was measured in one session +on one machine with scratch scripts that are not in this repository. The method +is described precisely enough to redo, and the ratios are the durable part, but +nothing here re-runs them and nobody should treat the absolute numbers as +checkable. The bge and gte rows in particular cannot be reproduced without +changes the eval harness does not have: it runs one fixed provider and has no +way to select a model or vary a threshold. + +Stated plainly because the alternative is worse: a number that reads as +evidence while being untraceable is how the "18ms per embed" figure below +survived long enough to drive a model change that had to be reverted. + ## How these were measured Every number below comes from real segments pulled out of this project's own diff --git a/docs/embedding-providers.md b/docs/embedding-providers.md index a9da822e..b0332464 100644 --- a/docs/embedding-providers.md +++ b/docs/embedding-providers.md @@ -5,8 +5,10 @@ instead of the bundled local model. Nothing here is built yet. ## What stays true -xtctx ships local-only and stays local-only by default. The bundled MiniLM -model is what runs when nobody configures anything, and that is the behaviour +xtctx ships local-only and stays local-only by default. The default MiniLM +model — downloaded on first use, not bundled in the package, which ships +`dist` only — is what runs when nobody configures anything, and that is the +behaviour every existing claim describes. An endpoint is opt-in, per project, and never inferred — no environment @@ -94,10 +96,13 @@ local:Xenova/all-MiniLM-L6-v2 ``` Changing the endpoint or the model then invalidates vectors the same way -changing the local model already does. Dimensions do not need separate -handling: a different dimension count only ever arrives with a different +changing the local model already does. Dimensions then need no separate +handling — a different dimension count only ever arrives with a different identity string, so the old vectors are already gone by the time the new ones -are written. +are written. That is a requirement on the identity string rather than +something the schema enforces: `retrieval_unit_vectors.dimensions` is stored +and nothing reads it back, so an identity that failed to change would mix +widths silently. ## Thresholds @@ -136,9 +141,12 @@ already stored. All of them degrade to keyword search, which is the path a failed local model already takes, and all of them record the reason in `embedding_error` so -`xtctx status` and `xtctx_continuity_status` report it. None of them fail a -tool call: an agent asking for context gets keyword results and a note saying -semantic search is unavailable, rather than an error. +`xtctx status` and `xtctx_continuity_status` report it. None of them fails a +`hybrid` tool call: an agent asking for context gets keyword results and a note +saying semantic search is unavailable, rather than an error. An explicit +`vector` request still throws, as it does today — there is no other route for +it to degrade to, and answering it from keyword would be answering a different +question than the one asked. Retries are bounded and not clever — one retry on a 429 or a 5xx, then give up for that call and let the next call try again. Vectorizing is already @@ -182,7 +190,9 @@ partially. currently runs uncapped, which is right for local compute and possibly expensive against a metered API. 3. **Is a per-provider threshold sweep something xtctx can run itself?** The - eval harness does exactly this against a synthetic corpus. A + sweeps recorded in this repository were done by hand, against a temporarily + patched constant; the eval harness runs one fixed provider and has no way to + select a model or vary a threshold. A `xtctx calibrate` that sweeps against the project's own index would remove the unswept-threshold warning entirely, and is a larger piece of work than the provider itself. diff --git a/docs/testing-strategy.md b/docs/testing-strategy.md index 958c4357..f2fc2e04 100644 --- a/docs/testing-strategy.md +++ b/docs/testing-strategy.md @@ -8,7 +8,7 @@ measured rather than assumed. | Suite | Command | Runs in CI | Defends | |---|---|---|---| | unit | `npm test` | every job | Parsing, scoping, ranking mechanics, config writing. The bulk of the suite. | -| integration | `npm run test:integration` | yes | The MCP tool handlers against a real index. | +| integration | `npm run test:integration` | yes | The five MCP tool handlers end to end against a fixture service — the wiring and the payload shapes, not the index. `tests/handoff/` covers the real index. | | drift | `npm run test:drift` | yes | Each scraper against a recorded sample of the tool's real on-disk format. | | smoke | `npm run test:smoke` | yes | The built CLI, spawned as a host tool spawns it, against seeded stores. | | eval | `npm run test:eval` | `checks` job | Retrieval *quality* — MRR, top-1, recall@5 against a committed baseline. | @@ -148,7 +148,10 @@ reaches a user silently: project attribution, resume cursors, role mapping, message-index stability, timestamp handling, and the drift warnings that fire on an unrecognised shape. The existing suite killed 25 and 20 survived. Thirteen of those behaviours are now closed, each by a test verified to fail against the -exact mutation that survived it; the remaining five were left, with reasons. +exact mutation that survived it; five were left with reasons, recorded below. +Two are unaccounted for: 13 + 5 does not reach 20, and nothing in the +repository says which two they were. Treat the count as approximate rather +than as a ledger — the individual rows below are the part that was checked. The survivors clustered in three places, and all three are the same shape: a guard that only runs when a transcript says *nothing*. @@ -185,9 +188,13 @@ Left, with reasons rather than tests: | copilot-cli `session.start` mismatch returns early | The `projectMatch !== true` guard below refuses every record anyway. The only observable difference is a misleading drift message. | | opencode missing-`role` guard loses its `continue` | The type check immediately below skips the same record. Removing the guard outright, or both guards, is killed. | -The sweep also turned up one open defect, which is a bug rather than a coverage -gap and is deliberately not fixed here. **An oversized `codex` line is dropped -with no drift warning at all, and the code that was meant to warn cannot run.** +The sweep also turned up one defect, which was a bug rather than a coverage +gap. **Fixed since, in c9beb48** — `readJsonlLines` now hands back a head +sample of a discarded line so the record can still be classified, and +`tests/scrapers/codex-oversized-records.test.ts` pins the warning. The +description below is kept because the shape of the defect is the lesson: +an oversized `codex` line was dropped with no drift warning at all, and the +code that was meant to warn could not run. `readJsonlLines` already caps lines at `MAX_LINE_BYTES` and delivers anything over it as `line: null`, discarding the bytes; `codex.ts` then `continue`s on that branch in silence. The `isWithinLineLimit(line)` check below it — the one diff --git a/package.json b/package.json index 48098e74..c26f3b11 100644 --- a/package.json +++ b/package.json @@ -74,6 +74,7 @@ }, "files": [ "dist", + "src", "README.md", "LICENSE", "CHANGELOG.md" diff --git a/src/cli/status.ts b/src/cli/status.ts index bc4e8204..bef92fe7 100644 --- a/src/cli/status.ts +++ b/src/cli/status.ts @@ -172,9 +172,15 @@ export async function renderStatusBlock( lines.push("Skills:"); lines.push(` Source ${skills.sourceDir}`); for (const skill of skills.selected) { - const marker = skill.exists ? "ok" : "missing"; + const marker = skill.exists ? (skill.staleBuiltIn ? "stale" : "ok") : "missing"; const hash = skill.hash ? ` ${skill.hash.slice(0, 18)}` : ""; lines.push(` ${marker.padEnd(8)} ${skill.id}${hash}`); + if (skill.staleBuiltIn) { + lines.push( + " this project's copy predates the built-in skill shipped with " + + "this version; run `xtctx setup --yes` to refresh it", + ); + } } for (const target of skills.targets) { const skillPart = target.skillId ? ` ${target.skillId}` : ""; diff --git a/src/config/instruction-blocks.ts b/src/config/instruction-blocks.ts index 97143263..dc498254 100644 --- a/src/config/instruction-blocks.ts +++ b/src/config/instruction-blocks.ts @@ -135,7 +135,7 @@ export function renderManagedBlock(input: { "- Transport: stdio", "", "## Notes", - "- Indexing is on-demand from MCP recent, detail, and search calls.", + "- Indexing runs when the MCP server starts and on recent, detail, and search calls; `xtctx scan` does it on demand.", "- There is no xtctx daemon, API server, dashboard, durable memory, or generated brief.", "- Content outside this managed block is preserved.", MARKERS.end, diff --git a/src/config/skills.ts b/src/config/skills.ts index 0d451449..5331454a 100644 --- a/src/config/skills.ts +++ b/src/config/skills.ts @@ -43,7 +43,27 @@ interface SkillSyncResult { interface SkillStatus { sourceDir: string; - selected: Array<{ id: string; exists: boolean; hash?: string }>; + selected: Array<{ + id: string; + exists: boolean; + hash?: string; + /** + * The built-in skill's canonical copy no longer matches the one this + * version ships. + * + * Only ever set for `xtctx-handoff`. Setup copies the built-in text into + * `.xtctx/skills/` once, and everything afterwards compares the synced + * targets against THAT copy — so a project set up before the text changed + * keeps the old wording, every target agrees with it, and status reports + * `ok`. That is how a skill telling agents "no xtctx setup is required" + * survived in projects after the claim was corrected, while the server + * refuses every tool in an unconfigured project. + * + * The skill is instructions to an agent, so a stale copy is not cosmetic: + * it is an instruction to keep calling tools that will not answer. + */ + staleBuiltIn?: boolean; + }>; targets: Array<{ tool: ToolId; mode: SkillSyncMode; @@ -190,10 +210,15 @@ export async function inspectSkillStatus(projectRoot: string, configPath: string selectedIds.map(async (id) => { const path = join(sourceDir, id, "SKILL.md"); const content = await readUtf8IfExists(path); + const staleBuiltIn = + id === BUILT_IN_SKILL_ID && + content !== null && + hashContent(content) !== hashContent(builtInHandoffSkill()); return { id, exists: content !== null, hash: content ? hashContent(content) : undefined, + ...(staleBuiltIn ? { staleBuiltIn: true } : {}), }; }), ); diff --git a/src/handoff/embeddings.ts b/src/handoff/embeddings.ts index b1986a3c..87a3771d 100644 --- a/src/handoff/embeddings.ts +++ b/src/handoff/embeddings.ts @@ -7,6 +7,14 @@ * MiniLM fp32 mrr 0.598 recall@5 0.850 top1 0.450 at 0.36 * mpnet q8 mrr 0.654 recall@5 0.933 top1 0.483 at 0.40 * + * Those figures predate #318, which rebuilt the eval corpus to use realistic + * session lengths on the grounds that it "has been measuring a world that does + * not exist". The committed baseline moved with it: MiniLM hybrid reads + * 0.333 / 0.533 / 0.183 today, not 0.598 / 0.850 / 0.450. The COMPARISON above + * was measured on one corpus and stands; the absolute numbers no longer match + * anything reproducible, so do not quote them or compare a new model against + * them. `tests/eval/results/ranking-baseline.json` is the current truth. + * * It was the default for a day on the strength of that table, which measured * only half the question. What the table left out is what embedding actually * costs, and the figure used at the time — 18ms per embed — came from diff --git a/src/handoff/ranking.ts b/src/handoff/ranking.ts index 96ab4a06..cc63591c 100644 --- a/src/handoff/ranking.ts +++ b/src/handoff/ranking.ts @@ -59,6 +59,14 @@ const MIN_SEMANTIC_COSINE = 0.15; * 0.36 mrr 0.598 recall@5 0.850 top1 0.450 <- here * 0.40 mrr 0.592 recall@5 0.850 top1 0.433 * + * This sweep predates #318, which rebuilt the eval corpus to use realistic + * session lengths because the old one "has been measuring a world that does + * not exist". The committed baseline moved with it — MiniLM hybrid reads + * 0.333 / 0.533 / 0.183 today — so the SHAPE of the sweep is what survives, + * not the absolute figures. The threshold has not been re-swept against the + * current corpus; 0.36 is inherited rather than re-derived, which is worth + * knowing before treating it as measured. + * * The trap worth naming: held at 0.36 while the default was mpnet, that model * looked like it regressed false positives to 0.10. It had not — the * threshold simply belonged to the distribution it was cut from. A model diff --git a/tests/config/stale-builtin-skill.test.ts b/tests/config/stale-builtin-skill.test.ts new file mode 100644 index 00000000..7b4466b7 --- /dev/null +++ b/tests/config/stale-builtin-skill.test.ts @@ -0,0 +1,93 @@ +/** + * A project's copy of the built-in skill can go stale, and status has to say so. + * + * Setup copies `builtInHandoffSkill()` into `.xtctx/skills/` once. Everything + * after that compares the SYNCED targets against that copy — so a project set + * up before the text changed keeps the old wording, every target agrees with + * it, and `xtctx status` reports `ok`. + * + * That is not hypothetical. The skill told agents "no xtctx setup is required" + * and that the retrieval tools were "worth calling anyway" in an unconfigured + * project, while the server refuses all five tools there. The text was + * corrected at the source; every project already set up kept the old one, with + * nothing reporting it. + * + * The skill is instructions to an agent, so a stale copy is an instruction to + * keep calling tools that will not answer. + */ +import { mkdir, mkdtemp, rm, writeFile } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { BUILT_IN_SKILL_ID, builtInHandoffSkill, inspectSkillStatus } from "@xtctx/config/skills"; + +let projectRoot = ""; +let configPath = ""; + +async function writeCanonicalSkill(content: string): Promise { + const dir = join(projectRoot, ".xtctx", "skills", BUILT_IN_SKILL_ID); + await mkdir(dir, { recursive: true }); + await writeFile(join(dir, "SKILL.md"), content, "utf-8"); +} + +beforeEach(async () => { + projectRoot = await mkdtemp(join(tmpdir(), "xtctx-stale-skill-")); + configPath = join(projectRoot, ".xtctx", "config.yaml"); + await mkdir(join(projectRoot, ".xtctx"), { recursive: true }); + await writeFile( + configPath, + ["skills:", " selected:", ` ${BUILT_IN_SKILL_ID}:`, " hash: sha256:whatever", ""].join("\n"), + "utf-8", + ); +}); + +afterEach(async () => { + await rm(projectRoot, { recursive: true, force: true }); +}); + +describe("inspectSkillStatus", () => { + it("flags a built-in skill whose copy predates this version", async () => { + // The exact wording that shipped before the correction. + await writeCanonicalSkill( + builtInHandoffSkill().replace( + "# xtctx Handoff", + "# xtctx Handoff\n\nno xtctx setup is required.", + ), + ); + + const status = await inspectSkillStatus(projectRoot, configPath); + const builtIn = status.selected.find((skill) => skill.id === BUILT_IN_SKILL_ID); + + expect(builtIn?.exists).toBe(true); + expect(builtIn?.staleBuiltIn).toBe(true); + }); + + it("says nothing when the copy matches what this version ships", async () => { + await writeCanonicalSkill(builtInHandoffSkill()); + + const status = await inspectSkillStatus(projectRoot, configPath); + const builtIn = status.selected.find((skill) => skill.id === BUILT_IN_SKILL_ID); + + expect(builtIn?.exists).toBe(true); + expect(builtIn?.staleBuiltIn).toBeUndefined(); + }); + + it("does not claim staleness for a skill that is not the built-in one", async () => { + // A user's own skill is theirs; differing from the built-in text is what + // it is supposed to do. + const dir = join(projectRoot, ".xtctx", "skills", "my-own-skill"); + await mkdir(dir, { recursive: true }); + await writeFile(join(dir, "SKILL.md"), "---\nname: my-own-skill\n---\n\n# Mine\n", "utf-8"); + await writeFile( + configPath, + ["skills:", " selected:", " my-own-skill:", " hash: sha256:whatever", ""].join("\n"), + "utf-8", + ); + + const status = await inspectSkillStatus(projectRoot, configPath); + const own = status.selected.find((skill) => skill.id === "my-own-skill"); + + expect(own?.exists).toBe(true); + expect(own?.staleBuiltIn).toBeUndefined(); + }); +}); From ae9ee936398fda5651c7b62907a4e9ed36663620 Mon Sep 17 00:00:00 2001 From: Felix Stubner Date: Mon, 21 Sep 2026 13:40:00 +0100 Subject: [PATCH 09/34] feat(scripts): probe embedding devices on three operating systems MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The GPU result in docs/embedding-performance.md — ~6x, numerically identical vectors, already installed — has been blocked on one thing since it was measured: it came from one machine. DirectML is Windows-only, WebGPU is the portable candidate, and WebGPU had never run anywhere but this desk. WSL on this host has /dev/dxg but no /dev/dri, so it exercises the fallback rather than a Linux GPU. Implementing device selection on a sample of one is how a fallback path ships broken: it works for whoever wrote it and either throws or, worse, returns different vectors everywhere else. The probe asks two things per device, and only the first is about speed: does it initialise and embed at all, and are its vectors the same as the CPU's. The second decides whether a fallback can be silent — identical vectors mean a machine that falls back is running the same index as one that does not, with no re-embed and no threshold re-sweep implied. Each device runs in its own process, because two providers in one process fail intermittently with `bad allocation`. Locally (win32/x64, 48 segments of 1000 chars): cpu 22.0 ms/segment 10.6 cores baseline dml 5.1 ms/segment 0.8 cores worst cosine 1.000000 webgpu 6.4 ms/segment 0.9 cores worst cosine 0.999999 auto 23.1 ms/segment 10.1 cores worst cosine 0.999999 `auto` is what the product passes today, and it matches cpu — which confirms the GPU is currently unused rather than merely underused. The workflow runs on ubuntu, macos and windows. ubuntu is the important one: no GPU at all, so it is the machine that has to fall back cleanly, and the one this project has no evidence about. It triggers on changes to the probe because a workflow_dispatch workflow is only dispatchable from the default branch, so a manual-only probe could not be run on the PR that adds it. --- .github/workflows/embedding-device-probe.yml | 73 +++++ scripts/probe-embedding-device.mjs | 292 +++++++++++++++++++ 2 files changed, 365 insertions(+) create mode 100644 .github/workflows/embedding-device-probe.yml create mode 100644 scripts/probe-embedding-device.mjs diff --git a/.github/workflows/embedding-device-probe.yml b/.github/workflows/embedding-device-probe.yml new file mode 100644 index 00000000..2729d6c3 --- /dev/null +++ b/.github/workflows/embedding-device-probe.yml @@ -0,0 +1,73 @@ +# Which ONNX execution providers work, on operating systems nobody here owns. +# +# `docs/embedding-performance.md` has DirectML at ~6x the CPU path with +# numerically identical vectors, and stops there, because DirectML is +# Windows-only and WebGPU — the portable candidate — had only ever been run on +# the one Windows desktop this project is developed on. WSL on that host has +# `/dev/dxg` but no `/dev/dri`, so it exercises the fallback rather than a +# Linux GPU. +# +# This workflow is the second and third machine. `macos-latest` is real Apple +# Silicon with a Metal-backed WebGPU, and `ubuntu-latest` has no GPU at all, +# which makes it the more important of the two: it is the machine that has to +# fall back cleanly, and the one this project currently has no evidence about. +# +# Not on every push — it downloads a model and runs four processes per OS for +# a report a person reads, and nothing merges on its result. It runs when the +# probe itself changes, which is when its answer can have moved, and otherwise +# on request. +# +# The path filter is also what makes the report reachable before this file is +# on `main`: a `workflow_dispatch` workflow is only dispatchable from the +# default branch, so a probe that was manual-only could not be run on the pull +# request that introduces it. +name: embedding-device-probe + +on: + workflow_dispatch: + pull_request: + paths: + - scripts/probe-embedding-device.mjs + - .github/workflows/embedding-device-probe.yml + +permissions: + contents: read + +jobs: + probe: + name: probe (${{ matrix.os }}) + runs-on: ${{ matrix.os }} + # A runner where every device fails is a finding, not a broken workflow — + # the other two still have to report. + continue-on-error: true + strategy: + fail-fast: false + matrix: + os: [ubuntu-latest, macos-latest, windows-latest] + + steps: + - name: Checkout + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + + - name: Setup Node + uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0 + with: + node-version: 24 + cache: npm + cache-dependency-path: package-lock.json + + - name: Install dependencies + run: npm ci + + # Same cache and the same reason as ci.yml: four processes per OS each + # want the 86MB MiniLM, and HuggingFace answers 429 to a burst of them. + - name: Cache the embedding model + uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0 + with: + path: node_modules/@huggingface/transformers/.cache + key: hf-model-${{ runner.os }}-${{ hashFiles('src/handoff/embeddings.ts') }} + + # One run, both formats. Probing twice would re-embed 48 segments per + # device for a second copy of the same numbers. + - name: Probe + run: node scripts/probe-embedding-device.mjs --json-also diff --git a/scripts/probe-embedding-device.mjs b/scripts/probe-embedding-device.mjs new file mode 100644 index 00000000..5a795036 --- /dev/null +++ b/scripts/probe-embedding-device.mjs @@ -0,0 +1,292 @@ +#!/usr/bin/env node +/** + * Which ONNX execution providers can this machine actually embed on, and do + * they agree with the CPU? + * + * `docs/embedding-performance.md` measured DirectML at ~6x the CPU path with + * numerically identical vectors, and stopped there: DirectML is Windows-only, + * WebGPU is the portable candidate, and WebGPU had only ever been run on one + * machine. Implementing device selection on a sample of one is how a fallback + * path ships broken — it works for whoever wrote it and silently throws, or + * silently returns different vectors, everywhere else. + * + * This script is the evidence. It is deliberately not a test: it downloads a + * model, takes tens of seconds, and its output is a report to read rather than + * an assertion to pass. `.github/workflows/embedding-device-probe.yml` runs it + * on ubuntu, windows and macos runners so the report covers three operating + * systems and three GPU situations instead of this one desk. + * + * Two things are being asked, and only the first is about speed: + * + * 1. Does the device initialise and embed at all? + * 2. Are its vectors the same as the CPU's? + * + * (2) is the one that decides whether a fallback can be silent. Identical + * vectors mean a machine that falls back to CPU is running the same index as + * a machine that does not, and no re-embed or threshold re-sweep is implied. + * Vectors that drift mean device choice is part of vector identity, the way + * dtype would have to be, and that is a much larger change. + * + * Each device runs in its OWN PROCESS. Two providers in one process fail + * intermittently with `bad allocation`; the parent below spawns itself per + * device for that reason, not for isolation of timings. + * + * Usage: + * node scripts/probe-embedding-device.mjs # all devices, report + * node scripts/probe-embedding-device.mjs --json # machine-readable only + * node scripts/probe-embedding-device.mjs --json-also # report, then JSON + * node scripts/probe-embedding-device.mjs --device=webgpu --child # one + */ +import { spawnSync } from "node:child_process"; +import { mkdtempSync, readFileSync, rmSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { pathToFileURL } from "node:url"; + +/** + * Candidates, in the order a fallback chain would try them. + * + * `auto` is what @huggingface/transformers picks when `device` is not passed — + * which is what xtctx does today, so it is the baseline any change is measured + * against, not a fourth option. + */ +const DEVICES = ["cpu", "dml", "webgpu", "auto"]; + +const MODEL = "Xenova/all-MiniLM-L6-v2"; +const DTYPE = "fp32"; +/** + * Enough segments for a timing to mean something, few enough that a CI runner + * with no GPU finishes the CPU arm in a couple of minutes. + */ +const SEGMENT_COUNT = 48; +const SEGMENT_CHARS = 1000; + +/** + * Deterministic filler at the length real segments have. + * + * Not real transcript text — that is private, and this is not measuring + * retrieval quality. It IS measuring per-segment cost, where length is the + * term that matters: an earlier round of this work benchmarked strings like + * `warm query number 5`, concluded a heavier model was affordable, and shipped + * a change that had to be reverted the next day. A segment is ~1024 + * characters, the model's full sequence window, and that is what this feeds it. + */ +function buildSegments() { + const words = [ + "session", "transcript", "index", "vector", "window", "segment", "scraper", + "handoff", "retrieval", "keyword", "semantic", "threshold", "cosine", + "database", "migration", "config", "project", "message", "timestamp", "tool", + ]; + const segments = []; + let seed = 1; + for (let index = 0; index < SEGMENT_COUNT; index += 1) { + let text = ""; + while (text.length < SEGMENT_CHARS) { + // xorshift: reproducible across platforms and Node versions, unlike + // anything seeded from Math.random or from the clock. + seed ^= seed << 13; + seed ^= seed >>> 17; + seed ^= seed << 5; + seed >>>= 0; + text += `${words[seed % words.length]} `; + } + segments.push(text.slice(0, SEGMENT_CHARS)); + } + return segments; +} + +async function runOneDevice(device, vectorPath) { + const segments = buildSegments(); + const transformers = await import("@huggingface/transformers"); + + const options = { dtype: DTYPE }; + // `auto` means "pass nothing", which is today's behaviour. + if (device !== "auto") { + options.device = device; + } + + const loadStart = performance.now(); + const extractor = await transformers.pipeline("feature-extraction", MODEL, options); + const loadMs = performance.now() - loadStart; + + // One untimed batch first: the first call carries graph compilation and + // buffer allocation, which is a one-off cost and not what per-segment + // throughput means. + await extractor(segments.slice(0, 4), { pooling: "mean", normalize: true }); + + const cpuBefore = process.cpuUsage(); + const embedStart = performance.now(); + const output = await extractor(segments, { pooling: "mean", normalize: true }); + const embedMs = performance.now() - embedStart; + const cpuAfter = process.cpuUsage(cpuBefore); + + const flat = Float32Array.from(output.data); + const dimensions = flat.length / segments.length; + if (!Number.isInteger(dimensions) || dimensions <= 0) { + throw new Error(`unexpected output shape: ${flat.length} for ${segments.length} segments`); + } + if (vectorPath) { + writeFileSync(vectorPath, Buffer.from(flat.buffer, flat.byteOffset, flat.byteLength)); + } + + const cpuMs = (cpuAfter.user + cpuAfter.system) / 1000; + return { + device, + ok: true, + loadMs: Math.round(loadMs), + msPerSegment: Number((embedMs / segments.length).toFixed(1)), + coresUsed: Number((cpuMs / embedMs).toFixed(1)), + dimensions, + segments: segments.length, + }; +} + +/** + * Worst and mean cosine between two runs' vectors for the same segments. + * + * Both sides are already L2-normalized by the pipeline, so a dot product is + * the cosine. Worst pair is the number that matters — a mean of 1.000000 with + * one pair at 0.97 is still a vector space that moved. + */ +function compareVectors(pathA, pathB, dimensions) { + const a = new Float32Array(toArrayBuffer(readFileSync(pathA))); + const b = new Float32Array(toArrayBuffer(readFileSync(pathB))); + if (a.length !== b.length) { + return { comparable: false, reason: `dimension mismatch: ${a.length} vs ${b.length}` }; + } + + let total = 0; + let worst = 1; + const count = a.length / dimensions; + for (let index = 0; index < count; index += 1) { + let dot = 0; + for (let d = 0; d < dimensions; d += 1) { + dot += a[index * dimensions + d] * b[index * dimensions + d]; + } + total += dot; + worst = Math.min(worst, dot); + } + return { + comparable: true, + meanCosine: Number((total / count).toFixed(6)), + worstCosine: Number(worst.toFixed(6)), + }; +} + +function toArrayBuffer(buffer) { + return buffer.buffer.slice(buffer.byteOffset, buffer.byteOffset + buffer.byteLength); +} + +async function child(device) { + const vectorPath = process.env.XTCTX_PROBE_VECTORS ?? ""; + try { + const result = await runOneDevice(device, vectorPath); + process.stdout.write(`\n__PROBE__${JSON.stringify(result)}\n`); + } catch (error) { + // A device that cannot initialise is a RESULT, not a crash — "webgpu is + // unavailable on this runner" is exactly what the probe exists to learn. + process.stdout.write( + `\n__PROBE__${JSON.stringify({ + device, + ok: false, + error: error instanceof Error ? error.message : String(error), + })}\n`, + ); + } +} + +async function parent(asJson) { + const workDir = mkdtempSync(join(tmpdir(), "xtctx-probe-")); + const results = []; + try { + for (const device of DEVICES) { + const vectorPath = join(workDir, `${device}.f32`); + const run = spawnSync( + process.execPath, + [process.argv[1], `--device=${device}`, "--child"], + { + encoding: "utf-8", + env: { ...process.env, XTCTX_PROBE_VECTORS: vectorPath }, + // The model download is minutes on a cold CI cache. + timeout: 15 * 60 * 1000, + maxBuffer: 64 * 1024 * 1024, + }, + ); + const marker = `${run.stdout ?? ""}\n${run.stderr ?? ""}`.match(/__PROBE__(.*)/); + if (!marker) { + results.push({ + device, + ok: false, + error: `child produced no result (status ${run.status}): ${(run.stderr ?? "").trim().slice(-400)}`, + }); + continue; + } + const result = JSON.parse(marker[1]); + if (result.ok) { + result.vectorPath = vectorPath; + } + results.push(result); + } + + const cpu = results.find((result) => result.device === "cpu" && result.ok); + for (const result of results) { + if (!result.ok || !cpu || result.device === "cpu") continue; + result.vsCpu = compareVectors(cpu.vectorPath, result.vectorPath, cpu.dimensions); + } + + for (const result of results) delete result.vectorPath; + report(results, asJson); + return results; + } finally { + rmSync(workDir, { recursive: true, force: true }); + } +} + +function report(results, asJson) { + const json = () => + process.stdout.write( + `${JSON.stringify({ platform: process.platform, arch: process.arch, results }, null, 2)}\n`, + ); + if (asJson === "only") { + json(); + return; + } + + const cpu = results.find((result) => result.device === "cpu" && result.ok); + process.stdout.write(`\nEmbedding device probe — ${process.platform}/${process.arch}, node ${process.version}\n`); + process.stdout.write(`model ${MODEL} ${DTYPE}, ${SEGMENT_COUNT} segments of ${SEGMENT_CHARS} chars\n\n`); + process.stdout.write("device status ms/segment vs cpu cores load ms\n"); + for (const result of results) { + if (!result.ok) { + process.stdout.write(`${result.device.padEnd(8)} unavailable ${result.error.slice(0, 60)}\n`); + continue; + } + const speedup = cpu ? `${(cpu.msPerSegment / result.msPerSegment).toFixed(1)}x` : "—"; + const agreement = result.vsCpu + ? result.vsCpu.comparable + ? `worst cosine ${result.vsCpu.worstCosine}` + : result.vsCpu.reason + : "baseline"; + process.stdout.write( + `${result.device.padEnd(8)} ok ${String(result.msPerSegment).padEnd(11)} ${speedup.padEnd(7)} ${String(result.coresUsed).padEnd(6)} ${result.loadMs}\n`, + ); + process.stdout.write(`${" ".repeat(9)}${agreement}\n`); + } + process.stdout.write("\n"); + if (asJson === "also") json(); +} + +const isChild = process.argv.includes("--child"); +const deviceArg = process.argv.find((argument) => argument.startsWith("--device=")); + +if (import.meta.url === pathToFileURL(process.argv[1]).href) { + if (isChild) { + await child(deviceArg ? deviceArg.slice("--device=".length) : "auto"); + } else { + await parent( + process.argv.includes("--json") ? "only" : process.argv.includes("--json-also") ? "also" : "", + ); + } +} + +export { DEVICES, buildSegments, compareVectors, runOneDevice }; From 1436bc82d345cc04493e26537aa4a04adf68412a Mon Sep 17 00:00:00 2001 From: Felix Stubner Date: Mon, 21 Sep 2026 13:57:09 +0100 Subject: [PATCH 10/34] docs(perf): the GPU fallback chain does not survive three operating systems MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The probe ran on macos-latest, windows-latest and ubuntu-latest. Two results matter more than the speed column. `auto` — what the product passes today — measures identical to `cpu` on all three runners, so the GPU is unused rather than underused. WebGPU on the GPU-less Windows runner is 1784.6 ms/segment against the CPU's 35.4: fifty times slower, and it does NOT fail. It found a software adapter, initialised cleanly and returned correct vectors. Linux with no GPU threw instead, which is what a fallback chain assumes. "Try WebGPU, fall back on error" is therefore not implementable — on the one configuration where it is catastrophic there is no error to fall back from, and the symptom is a vector backlog that never drains rather than anything that looks like a failure. macOS arm64 is the case the feature is for: WebGPU 21.4 ms/segment against 125.5 on CPU, 5.9x, at 0.3 cores. Vectors agree everywhere — worst pair 0.999999 across every device on every runner — so device choice does not enter vector identity, and a machine that ends up on CPU shares an index with one that does not. This does not rule out the feature; it rules out the cheap shape of it. A device has to be chosen on evidence: a short timed probe on first use, or an explicit opt-in. --- docs/embedding-performance.md | 76 ++++++++++++++++++++++++++++++----- 1 file changed, 65 insertions(+), 11 deletions(-) diff --git a/docs/embedding-performance.md b/docs/embedding-performance.md index d562058b..e0662d36 100644 --- a/docs/embedding-performance.md +++ b/docs/embedding-performance.md @@ -15,9 +15,14 @@ Only one figure below traces to something committed: the MiniLM eval baseline `tests/eval/results/ranking-baseline.json` and is regenerated by `npm run test:eval`. -Everything else — the device table, the batch sweep, the dtype comparison, the -segment-duplication count, and the bge/gte rows — was measured in one session -on one machine with scratch scripts that are not in this repository. The method +One more is now reproducible: the three-OS device table, which +`scripts/probe-embedding-device.mjs` regenerates and +`.github/workflows/embedding-device-probe.yml` runs on the same three runners. + +Everything else — the single-machine device table, the batch sweep, the dtype +comparison, the segment-duplication count, and the bge/gte rows — was measured +in one session on one machine with scratch scripts that are not in this +repository. The method is described precisely enough to redo, and the ratios are the durable part, but nothing here re-runs them and nobody should treat the absolute numbers as checkable. The bge and gte rows in particular cannot be reproduced without @@ -71,12 +76,58 @@ about 4 minutes on the GPU. installs. Nothing needs adding to `package.json`; the `device` option is simply not passed today. -**Not yet known, and the reason this is not implemented:** DirectML is -Windows-only. WebGPU is the portable candidate and also worked here, but it -has only ever been run on this machine. WSL on this host has `/dev/dxg` but -no `/dev/dri`, so it would exercise the *fallback* path rather than a Linux -GPU. A GitHub `macos-latest` runner is real Apple Silicon and is the one -genuine second-GPU test available for free. +## Device, on three operating systems: WebGPU is not a fallback chain + +The section above was measured on one machine and left one question open — +DirectML is Windows-only, so is the portable WebGPU path safe to fall back +through? `scripts/probe-embedding-device.mjs` and +`.github/workflows/embedding-device-probe.yml` were written to answer it on +hardware nobody here owns. Run 2026-09-21, 48 segments of 1000 characters, +each device in its own process: + +| runner | device | ms/segment | vs cpu | cores | worst cosine vs cpu | +| --- | --- | --- | --- | --- | --- | +| macos-latest (arm64) | cpu | 125.5 | — | 1.0 | baseline | +| macos-latest | **webgpu** | **21.4** | **5.9x** | **0.3** | 0.999999 | +| macos-latest | dml | — | unsupported: `coreml, webgpu, cpu` | | | +| macos-latest | auto | 99.7 | 1.3x | 1.0 | 0.999999 | +| windows-latest | cpu | 35.4 | — | 2.0 | baseline | +| windows-latest | **webgpu** | **1784.6** | **0.02x** | 3.9 | 1.000000 | +| windows-latest | dml | — | `Specified display adapter handle is invalid` | | | +| windows-latest | auto | 35.9 | 1.0x | 2.0 | 0.999999 | +| ubuntu-latest | cpu | 32.8 | — | 3.8 | baseline | +| ubuntu-latest | webgpu | — | `Failed to get a WebGPU adapter: No supported adapters` | | | +| ubuntu-latest | dml | — | unsupported: `cuda, webgpu, cpu` | | | +| ubuntu-latest | auto | 32.8 | 1.0x | 3.8 | 0.999999 | + +Two results are worth more than the speed column. + +**`auto` is `cpu`, on all three.** That is what the product passes today, so +the GPU is not merely underused — it is unused, and the desktop measurement +above is not being partially realised by anyone. + +**WebGPU on a GPU-less Windows machine is fifty times SLOWER, and does not +fail.** It found a software adapter, initialised cleanly, returned correct +vectors, and took 1.78 seconds per segment against the CPU's 35ms. Linux with +no GPU threw instead, which is the behaviour a fallback chain assumes. So +"try WebGPU, fall back on error" is not implementable: on the one +configuration where it is catastrophic, there is no error to fall back from, +and the symptom is a backlog that never drains rather than anything that looks +like a failure. + +That rules out a silent chain, not the feature. What it leaves is a device +that has to be *chosen on evidence* — a short timed probe against the real +CPU path on first use, or an explicit opt-in — which is a bigger piece of +work than passing an option, and is the reason this still is not implemented. + +**Vectors agree everywhere.** Worst pair across every device that ran on every +runner is 0.999999. Whatever selects the device, it does not become part of +vector identity, and a machine that ends up on CPU shares an index with one +that does not. + +DirectML remains the best result seen anywhere (5.1ms/segment locally, 0.8 +cores) and is Windows-with-a-real-GPU only: the Windows runner has no display +adapter and DirectML said so plainly rather than degrading. ## Multi-process embedding: no headroom worth taking @@ -197,8 +248,11 @@ name. Ranked by value against effort, on the evidence above: -1. **GPU with fallback.** ~6x, identical vectors, already installed. Blocked - only on portability evidence, which `macos-latest` can supply. +1. **GPU chosen by measurement.** ~6x on a real GPU, identical vectors, + already installed. No longer blocked on portability evidence — that evidence + arrived and said a fallback *chain* is the wrong shape, because GPU-less + Windows succeeds at 50x the cost instead of failing. What it needs is a + first-use timed probe or an explicit opt-in. 2. **Batch size 16.** ~10–20%, no vector change, one constant. Blocked on a cleaner measurement. 3. **bge-small with its own thresholds.** Better retrieval at 1.8x the From b4bf8a86d7451ce1c2a97a78c0c08ada347de85f Mon Sep 17 00:00:00 2001 From: Felix Stubner Date: Mon, 21 Sep 2026 14:10:19 +0100 Subject: [PATCH 11/34] feat(embeddings): optional OpenAI-compatible endpoint, per docs/embedding-providers.md MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Implements the design merged in #381. `EmbeddingProvider` was already an interface taken by injection, so this is a second implementation rather than a new seam. Opt-in per project and never inferred: an OPENAI_API_KEY that happens to be set in the environment does not opt a project into uploading transcript text. The key is read from the env var named by `embedding.apiKeyEnv` — a literal `apiKey` in `.xtctx/config.yaml` is rejected with an error saying why, because that file is committed and would publish the credential to everyone who clones the repository. Vector identity for a remote provider is `openai:{baseUrl}:{model}`, because two services both serving `text-embedding-3-small` produce vectors in different spaces while sharing a model name, and `retrieval_unit_vectors` is keyed on that name. Two deviations from the design, both deliberate: The local identity stays the bare HuggingFace id rather than becoming `local:Xenova/all-MiniLM-L6-v2`. Every remote identity is `openai:`-prefixed and cannot collide with a HuggingFace id, so the prefix buys nothing — while renaming it makes `dropVectorsFromOtherModels` discard every vector in every existing project on the first open after upgrading. That is tens of minutes of keyword-only search bought for no gain. Remote vectors are scaled to unit length on receipt, matching the local pipeline's `normalize: true`. Not for the cosine, which divides by both norms either way — for `poolVectors`, which mean-pools a window's segments and would otherwise weight them by magnitude instead of equally. OpenAI returns unit vectors; Ollama and LM Studio do not promise to. And one defect fixed in review: vectors are placed by the `index` each response element declares, not by position in `data`. A batch endpoint is not required to answer in request order, which is why that field exists. Reading positionally pairs every window with another window's vector — search keeps returning results, ranked against the wrong text, and nothing fails anywhere. Both are covered by tests confirmed to fail against the behaviour they replace. --- src/handoff/embedding-config.ts | 131 ++++++++++++ src/handoff/embeddings.ts | 15 ++ src/handoff/openai-embeddings.ts | 209 ++++++++++++++++++++ src/handoff/ranking.ts | 16 +- src/handoff/sqlite-index.ts | 10 + src/runtime/services.ts | 33 +++- src/types/config.ts | 18 ++ tests/handoff/openai-embeddings.test.ts | 252 ++++++++++++++++++++++++ 8 files changed, 676 insertions(+), 8 deletions(-) create mode 100644 src/handoff/embedding-config.ts create mode 100644 src/handoff/openai-embeddings.ts create mode 100644 tests/handoff/openai-embeddings.test.ts diff --git a/src/handoff/embedding-config.ts b/src/handoff/embedding-config.ts new file mode 100644 index 00000000..0cbde9ab --- /dev/null +++ b/src/handoff/embedding-config.ts @@ -0,0 +1,131 @@ +import { + DEFAULT_EMBEDDING_MODEL, + TransformersEmbeddingProvider, + type EmbeddingProvider, +} from "./embeddings.js"; +import { OpenAiEmbeddingProvider } from "./openai-embeddings.js"; +import { NullEmbeddingProvider } from "./null-embeddings.js"; +import { MIN_CONFIDENT_COSINE, MIN_SEMANTIC_COSINE } from "./ranking.js"; +import type { EmbeddingConfig } from "../types/config.js"; + +const DEFAULT_BATCH_SIZE = 32; +const DEFAULT_TIMEOUT_MS = 30_000; + +export function defaultEmbeddingConfig(): EmbeddingConfig { + return { + provider: "local", + batchSize: DEFAULT_BATCH_SIZE, + timeoutMs: DEFAULT_TIMEOUT_MS, + minSemanticCosine: MIN_SEMANTIC_COSINE, + minConfidentCosine: MIN_CONFIDENT_COSINE, + }; +} + +/** + * Parse the optional `embedding:` block from `.xtctx/config.yaml`. + * + * Throws on a literal `apiKey` (the file is committable), an unknown + * provider, or an incomplete openai-compatible block. Missing block → local + * defaults; never inferred from environment alone. + */ +export function parseEmbeddingConfig(input: unknown): EmbeddingConfig { + if (input === undefined || input === null) { + return defaultEmbeddingConfig(); + } + if (!input || typeof input !== "object" || Array.isArray(input)) { + throw new Error("embedding: expected a mapping"); + } + + const raw = input as Record; + if (Object.prototype.hasOwnProperty.call(raw, "apiKey")) { + // `.xtctx/config.yaml` is committed with the project. A key written into + // it would publish the credential to everyone who clones the repository. + throw new Error( + "embedding.apiKey is not allowed: .xtctx/config.yaml is committable — " + + "set embedding.apiKeyEnv to the name of an environment variable instead", + ); + } + + const defaults = defaultEmbeddingConfig(); + const providerRaw = raw.provider === undefined ? "local" : raw.provider; + if (providerRaw !== "local" && providerRaw !== "openai-compatible") { + throw new Error( + `embedding.provider must be "local" or "openai-compatible", got ${JSON.stringify(providerRaw)}`, + ); + } + + const config: EmbeddingConfig = { + provider: providerRaw, + batchSize: readPositiveInt(raw.batchSize, defaults.batchSize, "embedding.batchSize"), + timeoutMs: readPositiveInt(raw.timeoutMs, defaults.timeoutMs, "embedding.timeoutMs"), + minSemanticCosine: readFiniteNumber( + raw.minSemanticCosine, + defaults.minSemanticCosine, + "embedding.minSemanticCosine", + ), + minConfidentCosine: readFiniteNumber( + raw.minConfidentCosine, + defaults.minConfidentCosine, + "embedding.minConfidentCosine", + ), + }; + + if (typeof raw.apiKeyEnv === "string" && raw.apiKeyEnv.trim().length > 0) { + config.apiKeyEnv = raw.apiKeyEnv.trim(); + } else if (raw.apiKeyEnv !== undefined) { + throw new Error("embedding.apiKeyEnv must be a non-empty string when set"); + } + + if (providerRaw === "openai-compatible") { + if (typeof raw.baseUrl !== "string" || raw.baseUrl.trim().length === 0) { + throw new Error('embedding.baseUrl is required when provider is "openai-compatible"'); + } + if (typeof raw.model !== "string" || raw.model.trim().length === 0) { + throw new Error('embedding.model is required when provider is "openai-compatible"'); + } + config.baseUrl = raw.baseUrl.trim(); + config.model = raw.model.trim(); + } + + return config; +} + +export function createEmbeddingProvider(config: EmbeddingConfig): EmbeddingProvider { + if (process.env.XTCTX_DISABLE_EMBEDDINGS === "1") { + return new NullEmbeddingProvider(); + } + if (config.provider === "openai-compatible") { + if (!config.baseUrl || !config.model) { + // parseEmbeddingConfig already requires these; defend the type. + throw new Error("openai-compatible embedding config is missing baseUrl or model"); + } + return new OpenAiEmbeddingProvider({ + baseUrl: config.baseUrl, + model: config.model, + apiKeyEnv: config.apiKeyEnv, + batchSize: config.batchSize, + timeoutMs: config.timeoutMs, + }); + } + return new TransformersEmbeddingProvider(DEFAULT_EMBEDDING_MODEL); +} + +function readPositiveInt(value: unknown, fallback: number, label: string): number { + if (value === undefined) { + return fallback; + } + if (typeof value !== "number" || !Number.isFinite(value) || value < 1) { + throw new Error(`${label} must be a positive number`); + } + return Math.floor(value); +} + +function readFiniteNumber(value: unknown, fallback: number, label: string): number { + if (value === undefined) { + return fallback; + } + if (typeof value !== "number" || !Number.isFinite(value)) { + throw new Error(`${label} must be a finite number`); + } + return value; +} diff --git a/src/handoff/embeddings.ts b/src/handoff/embeddings.ts index 87a3771d..2440ad4e 100644 --- a/src/handoff/embeddings.ts +++ b/src/handoff/embeddings.ts @@ -91,6 +91,7 @@ * recorded on `DEFAULT_WINDOW_SIZE`, where the decision belongs. */ export const DEFAULT_EMBEDDING_MODEL = "Xenova/all-MiniLM-L6-v2"; + /** * Weight precision to load the model at. * @@ -136,6 +137,20 @@ type PipelineFactory = ( ) => Promise; export class TransformersEmbeddingProvider implements EmbeddingProvider { + /** + * The bare HuggingFace id, which is also the vector identity. + * + * `docs/embedding-providers.md` specifies a composite identity and gives + * `local:Xenova/all-MiniLM-L6-v2` as the local form. Only the remote half of + * that is implemented, deliberately. The composite exists because two + * endpoints can both serve `text-embedding-3-small` in different vector + * spaces, and every remote identity is already `openai:…`-prefixed, so it can + * never collide with a HuggingFace id. Prefixing the local one collides with + * nothing either way — while renaming it makes + * `dropVectorsFromOtherModels` discard every vector every existing project + * has, on first open after the upgrade, for no gain. That is tens of minutes + * of keyword-only search for anyone who upgrades. + */ readonly model: string; private extractor: FeatureExtractionPipeline | null = null; private loading: Promise | null = null; diff --git a/src/handoff/openai-embeddings.ts b/src/handoff/openai-embeddings.ts new file mode 100644 index 00000000..6c9f288e --- /dev/null +++ b/src/handoff/openai-embeddings.ts @@ -0,0 +1,209 @@ +import type { EmbeddingProvider } from "./embeddings.js"; + +export interface OpenAiEmbeddingProviderOptions { + baseUrl: string; + /** Remote model name sent in the request body — not the composite identity. */ + model: string; + /** Name of the env var that holds the key; the key itself is never stored. */ + apiKeyEnv?: string; + batchSize?: number; + timeoutMs?: number; +} + +/** + * Vector identity for an OpenAI-compatible endpoint. + * + * Two services can both advertise `text-embedding-3-small` while producing + * vectors in different spaces. Keying stored vectors on the bare model name + * would leave those mixed in `retrieval_unit_vectors`; including the endpoint + * makes a change of either endpoint or model invalidate the same way a local + * model change already does. + */ +export function openAiEmbeddingIdentity(baseUrl: string, model: string): string { + return `openai:${normalizeBaseUrl(baseUrl)}:${model}`; +} + +function normalizeBaseUrl(baseUrl: string): string { + return baseUrl.replace(/\/+$/, ""); +} + +/** + * Embedding provider that POSTs to an OpenAI-compatible `/embeddings` endpoint. + * + * Opt-in only: constructed when a project names one in `.xtctx/config.yaml`. + * No SDK — one HTTP shape, global `fetch`. + */ +export class OpenAiEmbeddingProvider implements EmbeddingProvider { + readonly model: string; + private readonly baseUrl: string; + private readonly remoteModel: string; + private readonly apiKeyEnv: string | undefined; + private readonly batchSize: number; + private readonly timeoutMs: number; + + constructor(options: OpenAiEmbeddingProviderOptions) { + this.baseUrl = normalizeBaseUrl(options.baseUrl); + this.remoteModel = options.model; + this.apiKeyEnv = options.apiKeyEnv; + this.batchSize = Math.max(1, Math.floor(options.batchSize ?? 32)); + this.timeoutMs = Math.max(1, Math.floor(options.timeoutMs ?? 30_000)); + this.model = openAiEmbeddingIdentity(this.baseUrl, this.remoteModel); + } + + async embed(text: string): Promise { + const [vector] = await this.embedBatch([text]); + return vector; + } + + async embedBatch(texts: string[]): Promise { + if (texts.length === 0) { + return []; + } + + const vectors: Float32Array[] = []; + for (let start = 0; start < texts.length; start += this.batchSize) { + const batch = texts.slice(start, start + this.batchSize); + vectors.push(...(await this.embedChunk(batch))); + } + return vectors; + } + + isReady(): boolean { + // No local model to load — the endpoint is ready whenever it answers. + return true; + } + + warm(): void { + // Nothing to preload; removes the warm-budget wait rather than adding a + // second code path for remote providers. + } + + private async embedChunk(texts: string[]): Promise { + let response = await this.postEmbeddings(texts); + // One retry on transient failures, then give up for this call. Vectorizing + // is incremental and resumable, so a failed pass costs a pass, not the + // index — and clever backoff would only hide an endpoint that is down. + if (response.status === 429 || response.status >= 500) { + // Discard the first body before asking again. An unread response holds + // its connection open in undici until it is garbage collected, and a + // vectorizing pass makes this call hundreds of times. + await response.body?.cancel().catch(() => {}); + response = await this.postEmbeddings(texts); + } + if (!response.ok) { + throw new Error(safeHttpError(response.status)); + } + + let body: unknown; + try { + body = await response.json(); + } catch { + throw new Error("Embedding endpoint returned a non-JSON body"); + } + + return parseEmbeddingResponse(body, texts.length); + } + + private async postEmbeddings(texts: string[]): Promise { + const headers: Record = { + "content-type": "application/json", + }; + const apiKey = this.readApiKey(); + if (apiKey !== undefined) { + headers.authorization = `Bearer ${apiKey}`; + } + + return fetch(`${this.baseUrl}/embeddings`, { + method: "POST", + headers, + body: JSON.stringify({ model: this.remoteModel, input: texts }), + signal: AbortSignal.timeout(this.timeoutMs), + }); + } + + private readApiKey(): string | undefined { + if (!this.apiKeyEnv) { + return undefined; + } + const value = process.env[this.apiKeyEnv]; + if (value === undefined || value.length === 0) { + // Name the env var, never a value that might have been set wrong. + throw new Error(`Embedding API key environment variable ${this.apiKeyEnv} is not set`); + } + return value; + } +} + +function safeHttpError(status: number): string { + // Status only — response bodies from auth failures often echo the key or + // request id material that should not land in `embedding_error` / logs. + return `Embedding endpoint returned HTTP ${status}`; +} + +function parseEmbeddingResponse(body: unknown, expectedCount: number): Float32Array[] { + if (!body || typeof body !== "object" || Array.isArray(body)) { + throw new Error("Embedding endpoint returned a malformed body"); + } + const data = (body as { data?: unknown }).data; + if (!Array.isArray(data) || data.length !== expectedCount) { + throw new Error("Embedding endpoint returned a malformed body"); + } + + // Placed by the response's own `index`, not by position in `data`. + // + // The OpenAI embeddings shape carries `index` on every element precisely + // because the array order is not promised, and a batch endpoint that answers + // out of order is free to. Reading positionally would pair each window with + // another window's vector — search would still return results, ranked by a + // similarity computed against the wrong text, with nothing failing anywhere. + // An endpoint that omits `index` falls back to position, which is all there + // is to go on. + const vectors = new Array(expectedCount); + data.forEach((item, position) => { + if (!item || typeof item !== "object" || Array.isArray(item)) { + throw new Error("Embedding endpoint returned a malformed body"); + } + const declared = (item as { index?: unknown }).index; + const slot = typeof declared === "number" && Number.isInteger(declared) ? declared : position; + if (slot < 0 || slot >= expectedCount || vectors[slot] !== undefined) { + throw new Error(`Embedding endpoint returned an out-of-range index ${slot}`); + } + + const embedding = (item as { embedding?: unknown }).embedding; + if (!Array.isArray(embedding) || embedding.length === 0) { + throw new Error(`Embedding endpoint returned a malformed embedding at index ${slot}`); + } + if (!embedding.every((value) => typeof value === "number" && Number.isFinite(value))) { + throw new Error(`Embedding endpoint returned a malformed embedding at index ${slot}`); + } + vectors[slot] = normalized(Float32Array.from(embedding as number[])); + }); + + return vectors as Float32Array[]; +} + +/** + * Scale to unit length, matching the local pipeline's `normalize: true`. + * + * Not for the cosine — `cosineSimilarity` divides by both norms, so scoring + * would survive either way. It is for `poolVectors`, which mean-pools a + * window's segments before storing one vector. Mean-pooling vectors of + * differing magnitude weights each segment by its length rather than treating + * them equally, so a window's vector would drift toward whichever segment the + * endpoint happened to return longest. OpenAI returns unit vectors and this is + * then a no-op; Ollama and LM Studio do not promise to. + */ +function normalized(vector: Float32Array): Float32Array { + let norm = 0; + for (const value of vector) { + norm += value * value; + } + if (norm === 0) { + return vector; + } + const scale = 1 / Math.sqrt(norm); + for (let index = 0; index < vector.length; index += 1) { + vector[index] *= scale; + } + return vector; +} diff --git a/src/handoff/ranking.ts b/src/handoff/ranking.ts index cc63591c..9bd0838f 100644 --- a/src/handoff/ranking.ts +++ b/src/handoff/ranking.ts @@ -30,7 +30,7 @@ const MAX_MATCHES_PER_SESSION = 3; * Unrelated sentence-transformer pairs sit near 0; related ones are * comfortably above this. */ -const MIN_SEMANTIC_COSINE = 0.15; +export const MIN_SEMANTIC_COSINE = 0.15; /** * How similar the *best* window has to be before a query counts as having @@ -96,7 +96,7 @@ const MIN_SEMANTIC_COSINE = 0.15; * * If it needs to move, move it against the eval rather than against one query. */ -const MIN_CONFIDENT_COSINE = 0.36; +export const MIN_CONFIDENT_COSINE = 0.36; /** * Weight of the recency/continuity tie-break in the relevance modes. Small * enough that it only ever separates candidates that are otherwise equal. @@ -396,8 +396,16 @@ export function rankSearchCandidates(options: { limit: number; cosineSimilarity: (left: Float32Array, right: Float32Array) => number; deserializeVector: (buffer: Buffer, dimensions: number) => Float32Array; + /** + * Floors swept per embedding model. Defaults stay MiniLM's; a remote model + * with a higher cosine distribution needs its own or every query matches. + */ + minSemanticCosine?: number; + minConfidentCosine?: number; }): SessionSummary[] { const { rows, keywordRows, queryVector, mode, limit } = options; + const minSemanticCosine = options.minSemanticCosine ?? MIN_SEMANTIC_COSINE; + const minConfidentCosine = options.minConfidentCosine ?? MIN_CONFIDENT_COSINE; /** * Windows that matched on words but have no vector yet. @@ -449,13 +457,13 @@ export function rankSearchCandidates(options: { // matching nothing, formatted exactly like a real hit. A unit qualifies // on semantic similarity or a keyword match; "no matching sessions" is // a more useful answer than a nearest vector. - .filter((item) => item.rawCosine >= MIN_SEMANTIC_COSINE || item.keywordScore > 0); + .filter((item) => item.rawCosine >= minSemanticCosine || item.keywordScore > 0); // Nothing here is actually similar to the query — keep only what matched // on words. For a query that means nothing to this corpus that leaves // nothing at all, which is the answer. const bestCosine = candidates.reduce((best, item) => Math.max(best, item.rawCosine), 0); - const semanticallyConfident = bestCosine >= MIN_CONFIDENT_COSINE; + const semanticallyConfident = bestCosine >= minConfidentCosine; const surviving = semanticallyConfident ? candidates : candidates.filter((item) => item.keywordScore > 0); diff --git a/src/handoff/sqlite-index.ts b/src/handoff/sqlite-index.ts index 42542790..ae5943e8 100644 --- a/src/handoff/sqlite-index.ts +++ b/src/handoff/sqlite-index.ts @@ -102,6 +102,10 @@ interface SqliteHandoffIndexOptions { * embeds is a search that degrades to keyword forever. */ freezeVectors?: boolean; + /** Per-window cosine floor; see ranking.ts. */ + minSemanticCosine?: number; + /** Best-window confidence floor; see ranking.ts. */ + minConfidentCosine?: number; } /** @@ -303,6 +307,8 @@ export class SqliteHandoffIndex implements SessionService { private readonly embeddingProvider: EmbeddingProvider; private readonly windowSize: number; private readonly windowStride: number; + private readonly minSemanticCosine: number | undefined; + private readonly minConfidentCosine: number | undefined; constructor( private readonly dbPath: string, @@ -314,6 +320,8 @@ export class SqliteHandoffIndex implements SessionService { options.embeddingProvider ?? defaultEmbeddingProvider(); this.windowSize = Math.max(2, Math.floor(options.windowSize ?? DEFAULT_WINDOW_SIZE)); this.windowStride = Math.max(1, Math.floor(options.windowStride ?? DEFAULT_WINDOW_STRIDE)); + this.minSemanticCosine = options.minSemanticCosine; + this.minConfidentCosine = options.minConfidentCosine; this.refreshBudgetMs = Math.max(0, options.refreshBudgetMs ?? DEFAULT_REFRESH_BUDGET_MS); this.literalBudgetMs = Math.max(0, options.literalBudgetMs ?? DEFAULT_LITERAL_BUDGET_MS); this.embeddingWarmBudgetMs = Math.max( @@ -894,6 +902,8 @@ export class SqliteHandoffIndex implements SessionService { limit: normalizedLimit, cosineSimilarity, deserializeVector, + minSemanticCosine: this.minSemanticCosine, + minConfidentCosine: this.minConfidentCosine, }), ); } diff --git a/src/runtime/services.ts b/src/runtime/services.ts index b676c815..bbd07fce 100644 --- a/src/runtime/services.ts +++ b/src/runtime/services.ts @@ -1,12 +1,19 @@ import { readFile, realpath } from "node:fs/promises"; import { join, resolve } from "node:path"; import { parse as parseYaml } from "yaml"; +import { + createEmbeddingProvider, + defaultEmbeddingConfig, + parseEmbeddingConfig, +} from "../handoff/embedding-config.js"; import { SqliteHandoffIndex } from "../handoff/sqlite-index.js"; import type { SessionService } from "../handoff/types.js"; import { SUPPORTED_TOOLS, createDefaultScrapers } from "../tools/sources.js"; +import type { EmbeddingConfig } from "../types/config.js"; interface ProjectConfig { tools: Record; + embedding: EmbeddingConfig; /** * Whether `.xtctx/config.yaml` exists — whether anyone opted this directory * in. @@ -113,6 +120,12 @@ export async function createProjectServices( // against an in-memory database and report zeros, which is the truth. createIfMissing: options.createIfMissing ?? config.present, redirectedTools: redirectedTools(config), + // Provider comes from config, not from whatever happens to be in the + // environment — an OPENAI_API_KEY sitting around must not opt a project + // into uploading transcript text. + embeddingProvider: config.error ? undefined : createEmbeddingProvider(config.embedding), + minSemanticCosine: config.embedding.minSemanticCosine, + minConfidentCosine: config.embedding.minConfidentCosine, }, ); @@ -145,10 +158,11 @@ async function loadProjectConfig(configPath: string, projectRoot: string): Promi // degraded read becoming the base for a write, again. const code = (err as NodeJS.ErrnoException).code; if (code === "ENOENT") { - return { tools: {}, present: false }; + return { tools: {}, embedding: defaultEmbeddingConfig(), present: false }; } return { tools: {}, + embedding: defaultEmbeddingConfig(), present: true, error: `could not be read (${code ?? "unknown error"}): ${err instanceof Error ? err.message : String(err)}`, }; @@ -158,9 +172,15 @@ async function loadProjectConfig(configPath: string, projectRoot: string): Promi const parsed = parseYaml(raw); if (parsed && typeof parsed === "object" && !Array.isArray(parsed)) { const root = parsed as Record; - return { tools: normalizeTools(root.tools, projectRoot), present: true }; + const embedding = parseEmbeddingConfig(root.embedding); + return { tools: normalizeTools(root.tools, projectRoot), embedding, present: true }; } - return { tools: {}, present: true, error: "expected a mapping at the top level" }; + return { + tools: {}, + embedding: defaultEmbeddingConfig(), + present: true, + error: "expected a mapping at the top level", + }; } catch (err) { // A config that exists but will not parse is not the same as no config. // `enabled: false` is the only control a user has over which transcript @@ -170,7 +190,12 @@ async function loadProjectConfig(configPath: string, projectRoot: string): Promi // Reported rather than thrown: `status` has to keep working, since // explaining a broken config is exactly what a diagnostic is for. What // does change is that nothing is scanned until it is fixed. - return { tools: {}, present: true, error: err instanceof Error ? err.message : String(err) }; + return { + tools: {}, + embedding: defaultEmbeddingConfig(), + present: true, + error: err instanceof Error ? err.message : String(err), + }; } } diff --git a/src/types/config.ts b/src/types/config.ts index cf17d1b0..97b81383 100644 --- a/src/types/config.ts +++ b/src/types/config.ts @@ -17,6 +17,7 @@ export interface XtctxConfig { targets?: Record; }; tools?: Record; + embedding?: EmbeddingConfig; } export interface ToolConfig { @@ -24,3 +25,20 @@ export interface ToolConfig { storePath?: string; hook?: "executable" | "instruction-only" | "mcp-only"; } + +/** + * Per-project embedding settings from `.xtctx/config.yaml`. + * + * Opt-in endpoint only — never inferred from an env var that happens to be + * set. The API key itself must not appear here; `apiKeyEnv` names the env var. + */ +export interface EmbeddingConfig { + provider: "local" | "openai-compatible"; + baseUrl?: string; + model?: string; + apiKeyEnv?: string; + batchSize: number; + timeoutMs: number; + minSemanticCosine: number; + minConfidentCosine: number; +} diff --git a/tests/handoff/openai-embeddings.test.ts b/tests/handoff/openai-embeddings.test.ts new file mode 100644 index 00000000..c95f553f --- /dev/null +++ b/tests/handoff/openai-embeddings.test.ts @@ -0,0 +1,252 @@ +import { afterEach, describe, expect, it, vi } from "vitest"; +import { + OpenAiEmbeddingProvider, + openAiEmbeddingIdentity, +} from "@xtctx/handoff/openai-embeddings"; +import { DEFAULT_EMBEDDING_MODEL, TransformersEmbeddingProvider } from "@xtctx/handoff/embeddings"; +import { parseEmbeddingConfig } from "@xtctx/handoff/embedding-config"; + +describe("OpenAiEmbeddingProvider", () => { + const originalFetch = globalThis.fetch; + + afterEach(() => { + globalThis.fetch = originalFetch; + vi.restoreAllMocks(); + }); + + it("uses the composite vector identity string", () => { + const provider = new OpenAiEmbeddingProvider({ + baseUrl: "https://api.openai.com/v1/", + model: "text-embedding-3-small", + }); + + expect(provider.model).toBe("openai:https://api.openai.com/v1:text-embedding-3-small"); + expect(openAiEmbeddingIdentity("https://api.openai.com/v1", "text-embedding-3-small")).toBe( + provider.model, + ); + }); + + it("leaves the local identity as the bare HuggingFace id", () => { + // Deliberately NOT `local:Xenova/…`, which the design doc suggests. Every + // remote identity is `openai:…`-prefixed and cannot collide with a + // HuggingFace id, so prefixing the local one buys nothing — while renaming + // it makes dropVectorsFromOtherModels discard every vector in every + // existing project on the first open after upgrading. + expect(new TransformersEmbeddingProvider().model).toBe(DEFAULT_EMBEDDING_MODEL); + expect(new TransformersEmbeddingProvider().model.startsWith("local:")).toBe(false); + }); + + it("places each vector by the index the response declares, not its position", async () => { + // A batch endpoint is not required to answer in request order, which is + // why the OpenAI shape carries `index` at all. Reading positionally pairs + // every window with another window's vector: search keeps working and + // keeps ranking against the wrong text, and nothing fails. + globalThis.fetch = vi.fn(async () => + jsonResponse({ + data: [ + { index: 2, embedding: [0, 0, 1] }, + { index: 0, embedding: [1, 0, 0] }, + { index: 1, embedding: [0, 1, 0] }, + ], + }), + ) as typeof fetch; + + const provider = new OpenAiEmbeddingProvider({ + baseUrl: "http://localhost:11434/v1", + model: "nomic-embed-text", + }); + + const vectors = await provider.embedBatch(["first", "second", "third"]); + + expect([...vectors[0]]).toEqual([1, 0, 0]); + expect([...vectors[1]]).toEqual([0, 1, 0]); + expect([...vectors[2]]).toEqual([0, 0, 1]); + }); + + it("rejects a response that indexes the same slot twice", async () => { + globalThis.fetch = vi.fn(async () => + jsonResponse({ + data: [ + { index: 0, embedding: [1, 0] }, + { index: 0, embedding: [0, 1] }, + ], + }), + ) as typeof fetch; + + const provider = new OpenAiEmbeddingProvider({ + baseUrl: "http://localhost:11434/v1", + model: "nomic-embed-text", + }); + + await expect(provider.embedBatch(["a", "b"])).rejects.toThrow(/out-of-range index/); + }); + + it("scales returned vectors to unit length", async () => { + // The local pipeline passes `normalize: true`. An endpoint that does not + // would leave poolVectors mean-pooling vectors of different magnitudes, + // which weights a window's segments by length instead of equally. + globalThis.fetch = vi.fn(async () => + jsonResponse({ data: [{ index: 0, embedding: [3, 4] }] }), + ) as typeof fetch; + + const provider = new OpenAiEmbeddingProvider({ + baseUrl: "http://localhost:11434/v1", + model: "nomic-embed-text", + }); + + const [vector] = await provider.embedBatch(["a"]); + + expect(vector[0]).toBeCloseTo(0.6, 6); + expect(vector[1]).toBeCloseTo(0.8, 6); + }); + + it("batches requests to batchSize and reads data[].embedding", async () => { + const bodies: unknown[] = []; + globalThis.fetch = vi.fn(async (_url, init) => { + bodies.push(JSON.parse(String(init?.body))); + const input = (bodies.at(-1) as { input: string[] }).input; + return jsonResponse({ + data: input.map((_, index) => ({ embedding: [index + 1, 0] })), + }); + }) as typeof fetch; + + const provider = new OpenAiEmbeddingProvider({ + baseUrl: "http://localhost:11434/v1", + model: "nomic-embed-text", + batchSize: 2, + }); + + const vectors = await provider.embedBatch(["a", "b", "c"]); + + expect(bodies).toEqual([ + { model: "nomic-embed-text", input: ["a", "b"] }, + { model: "nomic-embed-text", input: ["c"] }, + ]); + expect(vectors).toHaveLength(3); + expect([...vectors[0]]).toEqual([1, 0]); + expect([...vectors[2]]).toEqual([1, 0]); + expect(provider.isReady()).toBe(true); + provider.warm(); + }); + + it("retries once on 429 then succeeds", async () => { + let calls = 0; + globalThis.fetch = vi.fn(async () => { + calls += 1; + if (calls === 1) { + return new Response("rate limited", { status: 429 }); + } + return jsonResponse({ data: [{ embedding: [1, 1] }] }); + }) as typeof fetch; + + const provider = new OpenAiEmbeddingProvider({ + baseUrl: "http://localhost:11434/v1", + model: "nomic-embed-text", + }); + + const [vector] = await provider.embedBatch(["once"]); + expect(calls).toBe(2); + // Normalized on receipt; see `normalized` in the provider. + expect(vector[0]).toBeCloseTo(Math.SQRT1_2, 6); + expect(vector[1]).toBeCloseTo(Math.SQRT1_2, 6); + }); + + it("retries once on 5xx then gives up", async () => { + let calls = 0; + globalThis.fetch = vi.fn(async () => { + calls += 1; + return new Response("unavailable", { status: 503 }); + }) as typeof fetch; + + const provider = new OpenAiEmbeddingProvider({ + baseUrl: "http://localhost:11434/v1", + model: "nomic-embed-text", + }); + + await expect(provider.embedBatch(["x"])).rejects.toThrow(/HTTP 503/); + expect(calls).toBe(2); + }); + + it("honours timeoutMs via AbortSignal", async () => { + const timeoutSpy = vi.spyOn(AbortSignal, "timeout"); + globalThis.fetch = vi.fn(async (_url, init) => { + expect(init?.signal).toBeDefined(); + return jsonResponse({ data: [{ embedding: [1] }] }); + }) as typeof fetch; + + const provider = new OpenAiEmbeddingProvider({ + baseUrl: "http://localhost:11434/v1", + model: "nomic-embed-text", + timeoutMs: 12_345, + }); + + await provider.embedBatch(["ok"]); + expect(timeoutSpy).toHaveBeenCalledWith(12_345); + }); + + it("rejects a malformed body", async () => { + globalThis.fetch = vi.fn(async () => jsonResponse({ data: "nope" })) as typeof fetch; + + const provider = new OpenAiEmbeddingProvider({ + baseUrl: "http://localhost:11434/v1", + model: "nomic-embed-text", + }); + + await expect(provider.embedBatch(["x"])).rejects.toThrow(/malformed body/); + }); + + it("rejects 401 without leaking the API key in the error", async () => { + const secret = "sk-secret-test-key-do-not-leak"; + process.env.XTCTX_TEST_EMBED_KEY = secret; + try { + globalThis.fetch = vi.fn(async () => { + return new Response(`unauthorized for ${secret}`, { status: 401 }); + }) as typeof fetch; + + const provider = new OpenAiEmbeddingProvider({ + baseUrl: "https://api.openai.com/v1", + model: "text-embedding-3-small", + apiKeyEnv: "XTCTX_TEST_EMBED_KEY", + }); + + let message = ""; + try { + await provider.embedBatch(["x"]); + } catch (error) { + message = error instanceof Error ? error.message : String(error); + } + expect(message).toMatch(/HTTP 401/); + expect(message).not.toContain(secret); + } finally { + delete process.env.XTCTX_TEST_EMBED_KEY; + } + }); +}); + +describe("parseEmbeddingConfig", () => { + it("rejects a literal apiKey because the config file is committable", () => { + expect(() => + parseEmbeddingConfig({ + provider: "openai-compatible", + baseUrl: "http://localhost:11434/v1", + model: "nomic-embed-text", + apiKey: "sk-in-the-file", + }), + ).toThrow(/committable/); + }); + + it("rejects an unknown provider instead of falling back to local", () => { + expect(() => parseEmbeddingConfig({ provider: "azure" })).toThrow(/openai-compatible/); + }); + + it("defaults to local when the block is absent", () => { + expect(parseEmbeddingConfig(undefined).provider).toBe("local"); + }); +}); + +function jsonResponse(body: unknown, status = 200): Response { + return new Response(JSON.stringify(body), { + status, + headers: { "content-type": "application/json" }, + }); +} From 1f563363c7985921b3df22b39d53d78541c81f24 Mon Sep 17 00:00:00 2001 From: Felix Stubner Date: Mon, 21 Sep 2026 15:39:27 +0100 Subject: [PATCH 12/34] feat(embeddings): pick the execution provider by timing it on this machine MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Measured end to end on this desktop, via `xtctx scan --embed` before and after: 551.9 ms/window on the CPU, 50.7 ms/window on DirectML. The remaining embed backlog for this project went from about 24 minutes to about 1.5. `xtctx calibrate` times the model on every execution provider the platform offers, each in its own process, and remembers the fastest in `~/.xtctx/device.json`. It is a command, not something that happens on its own: it costs real seconds and loads the model once per device, and nothing should spawn processes behind an agent's tool call. A machine that never runs it stays on the CPU — which is what every machine did before this existed, since passing no device at all measured identical to `cpu` on all three operating systems. It is a measurement rather than a fallback chain because a fallback chain cannot be written correctly here. On a Windows machine with no real GPU, WebGPU does not fail: it finds a software adapter, initialises cleanly, returns numerically correct vectors, and runs fifty times slower than the CPU. There is no error to fall back from and no symptom but a backlog that never drains. Device identity would not have solved it either, and this was checked rather than assumed: `onnxruntime-node` exposes no adapter information (`env.webgpu` carries only `powerPreference`, there is no `navigator.gpu`, and the verbose log names every kernel dispatch but never the device), Bun 1.4.2 has no `navigator.gpu` at all, and requiring Deno to run an npm package is not a trade worth making. More to the point, a vendor string answers the wrong question — a laptop iGPU against a 24-core CPU is a real GPU that still loses. Every ambiguous case resolves to the CPU: a GPU inside the noise, a GPU that would not initialise, and a run where the CPU baseline itself failed and there is therefore nothing to compare against. `xtctx status` reports the device, read off the provider rather than off the cache. "A verdict was written" and "the indexer is using it" are different facts, and reporting the first while meaning the second is how a wiring bug hides behind a green check. The worker is TypeScript, not plain JavaScript, because the build is `tsc` over the src tree and copies nothing: a `.js` worker would be absent from `dist` and every device would report "no result" for the published package while every test here passed. A test asserts the resolved path exists. --- src/cli/calibrate.ts | 63 +++++ src/cli/index.ts | 9 + src/cli/status.ts | 6 + src/handoff/device-worker.ts | 77 ++++++ src/handoff/device.ts | 288 +++++++++++++++++++++++ src/handoff/embedding-config.ts | 8 +- src/handoff/embeddings.ts | 22 +- src/handoff/sqlite-index.ts | 1 + src/handoff/status.ts | 5 +- src/handoff/types.ts | 9 + src/runtime/services.ts | 5 +- tests/handoff/device-calibration.test.ts | 169 +++++++++++++ 12 files changed, 657 insertions(+), 5 deletions(-) create mode 100644 src/cli/calibrate.ts create mode 100644 src/handoff/device-worker.ts create mode 100644 src/handoff/device.ts create mode 100644 tests/handoff/device-calibration.test.ts diff --git a/src/cli/calibrate.ts b/src/cli/calibrate.ts new file mode 100644 index 00000000..4d690e66 --- /dev/null +++ b/src/cli/calibrate.ts @@ -0,0 +1,63 @@ +import { calibrateEmbeddingDevice, deviceCandidates, readDeviceVerdict } from "../handoff/device.js"; + +interface CalibrateOptions { + /** Re-measure even when a verdict for this machine is already cached. */ + force?: boolean; +} + +/** + * Time the embedding model on every execution provider this machine offers, + * and remember the fastest. + * + * A command rather than something that happens on its own, because it costs + * real seconds and loads the model once per device. The MCP server only ever + * *reads* the verdict; nothing spawns processes behind an agent's tool call. + * + * Running it is optional. A machine that never does stays on CPU, which is + * what every machine did before this existed. + */ +export async function runCalibrate(options: CalibrateOptions = {}): Promise { + const existing = await readDeviceVerdict(); + if (existing && !options.force) { + process.stdout.write( + `Already calibrated on ${existing.measuredAt}: ${existing.device}\n` + + `${formatMeasurements(existing.measured)}\n` + + `Re-run with --force to measure again.\n`, + ); + return; + } + + process.stdout.write( + `Timing the embedding model on: ${deviceCandidates().join(", ")}\n` + + "Each runs in its own process and loads the model once, so this takes a minute.\n\n", + ); + + const verdict = await calibrateEmbeddingDevice({ + onProgress: (device) => process.stdout.write(` ${device}...\n`), + }); + + process.stdout.write(`\n${formatMeasurements(verdict.measured)}\n`); + if (verdict.device === "cpu") { + // Said plainly, because "no GPU was used" is a result and not a failure — + // and on a machine with no GPU it is the correct one. + process.stdout.write("Chose cpu: nothing else was decisively faster here.\n"); + return; + } + process.stdout.write(`Chose ${verdict.device}. Indexing will use it from now on.\n`); +} + +function formatMeasurements(measured: Awaited>["measured"]): string { + return measured + .map((row) => + row.msPerSegment === null + ? ` ${row.device.padEnd(7)} unavailable — ${truncate(row.error ?? "unknown")}` + : ` ${row.device.padEnd(7)} ${row.msPerSegment} ms/segment`, + ) + .join("\n"); +} + +function truncate(message: string): string { + // Device initialisation failures carry multi-line native stack text. + const firstLine = message.split("\n")[0].trim(); + return firstLine.length > 100 ? `${firstLine.slice(0, 100)}...` : firstLine; +} diff --git a/src/cli/index.ts b/src/cli/index.ts index 5988b69b..30e354f8 100644 --- a/src/cli/index.ts +++ b/src/cli/index.ts @@ -1,5 +1,6 @@ #!/usr/bin/env node import { Command } from "commander"; +import { runCalibrate } from "./calibrate.js"; import { runDisconnect } from "./disconnect.js"; import { runHook } from "./hook.js"; import { runScan } from "./scan.js"; @@ -155,6 +156,14 @@ export async function main(argv = process.argv): Promise { }); }); + program + .command("calibrate") + .option("--force", "Measure again even if this machine already has a verdict", false) + .description("Time the embedding model on this machine's GPU and CPU, and use the faster") + .action(async (options: { force: boolean }) => { + await runCalibrate({ force: options.force }); + }); + program .command("disconnect") .argument("[tool]", "Tool to stop managing for this project") diff --git a/src/cli/status.ts b/src/cli/status.ts index bef92fe7..de0c5a86 100644 --- a/src/cli/status.ts +++ b/src/cli/status.ts @@ -113,6 +113,12 @@ export async function renderStatusBlock( `${backlog.eta ? `, about ${backlog.eta} of embedding left` : ""}`, ); } + // Only when it is not the default. A machine that has never been calibrated + // is on the CPU, which is what every machine did before calibration existed, + // and does not need a line saying so on every status call. + if (status.vector_device && status.vector_device !== "cpu") { + lines.push(`Device ${status.vector_device} (from \`xtctx calibrate\`)`); + } lines.push(""); lines.push("Tools:"); diff --git a/src/handoff/device-worker.ts b/src/handoff/device-worker.ts new file mode 100644 index 00000000..c9441f07 --- /dev/null +++ b/src/handoff/device-worker.ts @@ -0,0 +1,77 @@ +/** + * Time one embedding device, in a process of its own, and print the result. + * + * Spawned by `calibrateEmbeddingDevice`. It is a separate process because two + * providers in one process fail intermittently with `bad allocation`, and a + * device that cannot initialise has to be a measurement rather than a crash in + * the caller. + * + * Spawned by path, so it has to exist wherever it is spawned from: `tsc` + * emits it to `dist/src/handoff/device-worker.js` for the published package, + * and `workerArgv` in device.ts spawns the `.ts` through tsx when running from + * source. A plain `.js` file here would be simpler to spawn and would never + * reach `dist` at all, because the build is `tsc` over the `src` tree and + * copies nothing. + */ +import { parseArgs } from "node:util"; +import { DEFAULT_EMBEDDING_DTYPE, DEFAULT_EMBEDDING_MODEL } from "./embeddings.js"; +import { calibrationSegments } from "./device.js"; + +const { values } = parseArgs({ + options: { + device: { type: "string", default: "cpu" }, + segments: { type: "string", default: "16" }, + }, + strict: false, +}); + +const device = values.device; +const segmentCount = Math.max(1, Number.parseInt(String(values.segments), 10) || 16); + +try { + const segments = calibrationSegments(segmentCount); + const transformers = (await import("@huggingface/transformers")) as unknown as { + pipeline: ( + task: "feature-extraction", + model: string, + options: Record, + ) => Promise< + ( + input: string[], + options: { pooling: "mean"; normalize: boolean }, + ) => Promise<{ data: ArrayLike }> + >; + }; + const options: Record = { dtype: DEFAULT_EMBEDDING_DTYPE, device }; + const extractor = await transformers.pipeline( + "feature-extraction", + DEFAULT_EMBEDDING_MODEL, + options, + ); + + // One untimed batch first. The first call carries graph compilation and + // buffer allocation, a one-off cost that is not what per-segment throughput + // means — and it is much larger on the GPU paths, so including it would + // bias the comparison toward the CPU. + await extractor(segments.slice(0, 2), { pooling: "mean", normalize: true }); + + const startedAt = performance.now(); + await extractor(segments, { pooling: "mean", normalize: true }); + const elapsed = performance.now() - startedAt; + + process.stdout.write( + `\n__XTCTX_DEVICE__${JSON.stringify({ + device, + msPerSegment: Number((elapsed / segments.length).toFixed(1)), + })}\n`, + ); +} catch (error) { + // A device that cannot initialise is the answer, not a failure: "webgpu is + // unavailable here" is exactly what the caller needs in order to rule it out. + process.stdout.write( + `\n__XTCTX_DEVICE__${JSON.stringify({ + device, + error: error instanceof Error ? error.message : String(error), + })}\n`, + ); +} diff --git a/src/handoff/device.ts b/src/handoff/device.ts new file mode 100644 index 00000000..d853a6a4 --- /dev/null +++ b/src/handoff/device.ts @@ -0,0 +1,288 @@ +import { spawn } from "node:child_process"; +import { existsSync } from "node:fs"; +import { readFile, mkdir, writeFile } from "node:fs/promises"; +import { cpus, homedir } from "node:os"; +import { dirname, join } from "node:path"; +import { fileURLToPath } from "node:url"; +import { DEFAULT_EMBEDDING_DTYPE, DEFAULT_EMBEDDING_MODEL } from "./embeddings.js"; + +/** + * Which ONNX execution provider the local model runs on. + * + * `cpu` is what the product used until this existed — passing no `device` at + * all measured identical to `cpu` on all three operating systems, so the GPU + * was not merely underused, it was unused. + */ +export type EmbeddingDevice = "cpu" | "dml" | "webgpu"; + +/** + * Candidates worth trying here, cheapest signal first. + * + * Platform-filtered because asking for a device the runtime does not build is + * an immediate throw rather than a measurement: `dml` answers `Unsupported + * device: "dml". Should be one of: coreml, webgpu, cpu` on macOS and + * `cuda, webgpu, cpu` on Linux. + * + * On Windows, `dml` failing is ALSO the adapter check. `onnxruntime-node` + * ships `DirectML.dll` in the package this project already installs, and on a + * machine with no real display adapter it refuses with `Specified display + * adapter handle is invalid` rather than quietly falling back — measured on a + * GitHub windows-latest runner against this maintainer's desktop, where it was + * the fastest device of the three. That matters because Windows is the one + * platform where WebGPU does NOT refuse: see `calibrate` below. + */ +export function deviceCandidates(platform = process.platform): EmbeddingDevice[] { + if (platform === "win32") return ["cpu", "dml", "webgpu"]; + return ["cpu", "webgpu"]; +} + +export interface DeviceVerdict { + device: EmbeddingDevice; + /** Every candidate that ran, so a surprising verdict can be argued with. */ + measured: Array<{ device: EmbeddingDevice; msPerSegment: number | null; error?: string }>; + fingerprint: string; + measuredAt: string; +} + +/** + * Identity of the thing that was measured. + * + * A verdict is about this machine running this model at this precision, so all + * three are in the key along with the core count — the CPU arm's speed is what + * the GPU has to beat, and a 24-core desktop and a 4-core laptop reach + * different answers from the same hardware. + */ +export function deviceFingerprint( + model = DEFAULT_EMBEDDING_MODEL, + dtype = DEFAULT_EMBEDDING_DTYPE, +): string { + return [process.platform, process.arch, `cores=${cpus().length}`, model, dtype].join("|"); +} + +function cachePath(home = homedir()): string { + // User-level, not project-level: the answer is about the hardware, and a + // second project on the same machine has already paid for it. + return join(home, ".xtctx", "device.json"); +} + +/** + * The cached verdict, or null when there is none for this exact configuration. + * + * Never throws. A machine that cannot read its own cache runs on CPU, which is + * what it did before any of this existed. + */ +export async function readDeviceVerdict(options: { + home?: string; + fingerprint?: string; +} = {}): Promise { + const fingerprint = options.fingerprint ?? deviceFingerprint(); + try { + const raw = await readFile(cachePath(options.home), "utf-8"); + const parsed = JSON.parse(raw) as DeviceVerdict; + if (parsed.fingerprint !== fingerprint) { + // Different machine, model or precision — the old answer is about + // something else, not merely stale. + return null; + } + if (!deviceCandidates().includes(parsed.device)) { + return null; + } + return parsed; + } catch { + return null; + } +} + +async function writeDeviceVerdict(verdict: DeviceVerdict, home?: string): Promise { + const path = cachePath(home); + await mkdir(dirname(path), { recursive: true }); + await writeFile(path, `${JSON.stringify(verdict, null, 2)}\n`, "utf-8"); +} + +/** Text at the length real segments have; see the note in the worker. */ +export function calibrationSegments(count: number, chars = 1000): string[] { + const words = [ + "session", "transcript", "index", "vector", "window", "segment", "scraper", + "handoff", "retrieval", "keyword", "semantic", "threshold", "cosine", + "database", "migration", "config", "project", "message", "timestamp", "tool", + ]; + const segments: string[] = []; + let seed = 1; + for (let index = 0; index < count; index += 1) { + let text = ""; + while (text.length < chars) { + seed ^= seed << 13; + seed ^= seed >>> 17; + seed ^= seed << 5; + seed >>>= 0; + text += `${words[seed % words.length]} `; + } + segments.push(text.slice(0, chars)); + } + return segments; +} + +/** + * How much faster than CPU a device has to be before it is worth switching to. + * + * Not 1.0. The calibration is one short run on a machine that is doing other + * things, and a device that wins by 5% is inside that noise — while switching + * to it is a permanent change to how every future embed runs. Measured gaps + * that are real were 3x to 6x; the ones to reject were 0.02x. Nothing observed + * so far lands anywhere near 1.3, which is the point: the threshold only has + * to separate a decisive win from noise. + */ +const MIN_SPEEDUP = 1.3; + +/** + * The fastest device that beats the CPU by more than noise, else the CPU. + * + * CPU is the default in every ambiguous case, including the one where the CPU + * arm itself failed to run: a machine whose baseline could not be measured has + * no basis for preferring anything to it, and guessing wrong there is the + * fifty-times-slower outcome this whole mechanism exists to avoid. + */ +export function chooseDevice(measured: DeviceVerdict["measured"]): EmbeddingDevice { + const cpu = measured.find((row) => row.device === "cpu")?.msPerSegment ?? null; + if (cpu === null) { + return "cpu"; + } + + let device: EmbeddingDevice = "cpu"; + let best = cpu / MIN_SPEEDUP; + for (const row of measured) { + if (row.device === "cpu" || row.msPerSegment === null) continue; + if (row.msPerSegment < best) { + best = row.msPerSegment; + device = row.device; + } + } + return device; +} + +/** + * Time each candidate device and pick the fastest, then remember the answer. + * + * This exists instead of a fallback chain because a fallback chain cannot be + * written correctly. The obvious shape — try the GPU, fall back to CPU if it + * throws — was measured on three operating systems on 2026-09-21 and fails on + * exactly one of them, silently and badly. On a Windows machine with no real + * GPU, WebGPU does not throw: it finds a software adapter, initialises + * cleanly, returns numerically correct vectors and runs at 1784.6ms per + * segment against the CPU's 35.4. Fifty times slower, with no error to fall + * back from, and the only symptom is a vector backlog that never drains. + * + * Identity would not fix it either. `onnxruntime-node` exposes no adapter + * information — `env.webgpu` carries only `powerPreference`, there is no + * `navigator.gpu`, and the verbose log names every kernel dispatch but never + * the device. And even a perfect vendor string answers the wrong question: a + * laptop iGPU against a 24-core desktop CPU is a real GPU that still loses. + * The question is "is this faster HERE", so it is measured here. + * + * Each candidate runs in its own process. Two providers in one process fail + * intermittently with `bad allocation`, which is also why this cannot simply + * be a loop inside the caller. + */ +export async function calibrateEmbeddingDevice(options: { + home?: string; + /** + * Segments to time per device. Sixteen is enough to separate a 3x win from + * noise and costs about two seconds on the slowest CPU measured. + */ + segmentCount?: number; + timeoutMs?: number; + onProgress?: (device: EmbeddingDevice) => void; +} = {}): Promise { + const segmentCount = options.segmentCount ?? 16; + const timeoutMs = options.timeoutMs ?? 5 * 60 * 1000; + const measured: DeviceVerdict["measured"] = []; + + for (const device of deviceCandidates()) { + options.onProgress?.(device); + const result = await timeDevice(device, segmentCount, timeoutMs); + measured.push({ device, ...result }); + } + + const verdict: DeviceVerdict = { + device: chooseDevice(measured), + measured, + fingerprint: deviceFingerprint(), + measuredAt: new Date().toISOString(), + }; + // Best-effort: an unwritable home directory means the calibration is paid + // again next time, not that it fails now. + await writeDeviceVerdict(verdict, options.home).catch(() => {}); + return verdict; +} + +/** + * How to spawn the worker, in both layouts this code runs in. + * + * Built, `import.meta.url` is `dist/src/handoff/device.js` and the sibling + * `device-worker.js` that `tsc` emitted is right there. From source — the test + * suite, and `npm run dev` — the sibling is `device-worker.ts`, which node + * cannot execute, so it goes through tsx the same way `drift-log.test.ts` + * spawns its workers. + * + * Returned as argv rather than a path so the caller cannot accidentally spawn + * a `.ts` file directly, which fails with a syntax error several seconds in + * and reads exactly like a device that does not work. + */ +export function workerArgv(moduleUrl = import.meta.url): string[] { + const compiled = fileURLToPath(new URL("./device-worker.js", moduleUrl)); + if (existsSync(compiled)) { + return [compiled]; + } + const source = fileURLToPath(new URL("./device-worker.ts", moduleUrl)); + return [fileURLToPath(new URL("../../node_modules/tsx/dist/cli.mjs", moduleUrl)), source]; +} + +function timeDevice( + device: EmbeddingDevice, + segmentCount: number, + timeoutMs: number, +): Promise<{ msPerSegment: number | null; error?: string }> { + return new Promise((resolve) => { + const child = spawn( + process.execPath, + [...workerArgv(), `--device=${device}`, `--segments=${segmentCount}`], + { stdio: ["ignore", "pipe", "pipe"] }, + ); + + let out = ""; + const timer = setTimeout(() => { + child.kill(); + resolve({ msPerSegment: null, error: `timed out after ${timeoutMs}ms` }); + }, timeoutMs); + + child.stdout.on("data", (chunk: Buffer) => { + out += chunk.toString(); + }); + // stderr carries the model-loading notice and ONNX warnings; the result + // only ever arrives on stdout behind its marker. + child.stderr.resume(); + + child.on("error", (error) => { + clearTimeout(timer); + resolve({ msPerSegment: null, error: error.message }); + }); + child.on("close", () => { + clearTimeout(timer); + const marker = out.match(/__XTCTX_DEVICE__(.*)/); + if (!marker) { + resolve({ msPerSegment: null, error: "no result from the calibration process" }); + return; + } + try { + const parsed = JSON.parse(marker[1]) as { msPerSegment?: number; error?: string }; + if (typeof parsed.msPerSegment === "number") { + resolve({ msPerSegment: parsed.msPerSegment }); + return; + } + resolve({ msPerSegment: null, error: parsed.error ?? "unknown calibration failure" }); + } catch { + resolve({ msPerSegment: null, error: "unreadable result from the calibration process" }); + } + }); + }); +} diff --git a/src/handoff/embedding-config.ts b/src/handoff/embedding-config.ts index 0cbde9ab..09deab0a 100644 --- a/src/handoff/embedding-config.ts +++ b/src/handoff/embedding-config.ts @@ -90,7 +90,11 @@ export function parseEmbeddingConfig(input: unknown): EmbeddingConfig { return config; } -export function createEmbeddingProvider(config: EmbeddingConfig): EmbeddingProvider { +export function createEmbeddingProvider( + config: EmbeddingConfig, + /** Calibrated execution provider, if this machine has been calibrated. */ + device?: string, +): EmbeddingProvider { if (process.env.XTCTX_DISABLE_EMBEDDINGS === "1") { return new NullEmbeddingProvider(); } @@ -107,7 +111,7 @@ export function createEmbeddingProvider(config: EmbeddingConfig): EmbeddingProvi timeoutMs: config.timeoutMs, }); } - return new TransformersEmbeddingProvider(DEFAULT_EMBEDDING_MODEL); + return new TransformersEmbeddingProvider(DEFAULT_EMBEDDING_MODEL, undefined, device); } function readPositiveInt(value: unknown, fallback: number, label: string): number { diff --git a/src/handoff/embeddings.ts b/src/handoff/embeddings.ts index 2440ad4e..fd80c863 100644 --- a/src/handoff/embeddings.ts +++ b/src/handoff/embeddings.ts @@ -99,7 +99,7 @@ export const DEFAULT_EMBEDDING_MODEL = "Xenova/all-MiniLM-L6-v2"; * little and costs accuracy. The q8 tradeoff only mattered for mpnet, where * fp32 was 416MB. */ -const DEFAULT_EMBEDDING_DTYPE = "fp32"; +export const DEFAULT_EMBEDDING_DTYPE = "fp32"; const MAX_SEQ_TOKENS = 256; /** ~4 characters per token, the budget splitTextForEmbedding segments to. */ export const MAX_SEQ_CHARS = MAX_SEQ_TOKENS * 4; @@ -107,6 +107,15 @@ const MAX_BATCH_SIZE = 32; export interface EmbeddingProvider { readonly model: string; + /** + * Execution provider this will actually load on, for `xtctx status`. + * + * Read off the provider rather than off the calibration cache on purpose. + * "A verdict was written" and "the indexer is using it" are two different + * facts, and reporting the first while meaning the second is how a wiring + * bug hides behind a green check. + */ + readonly device?: string; embed(text: string): Promise; embedBatch(texts: string[]): Promise; /** @@ -158,6 +167,16 @@ export class TransformersEmbeddingProvider implements EmbeddingProvider { constructor( model = DEFAULT_EMBEDDING_MODEL, private readonly dtype = DEFAULT_EMBEDDING_DTYPE, + /** + * Execution provider, from `xtctx calibrate`; see `device.ts`. + * + * Undefined means pass nothing, which is what this did before calibration + * existed and measured identical to `cpu` on all three operating systems. + * It is NOT a chain: a device is named here only after being timed against + * the CPU on this machine, because the one configuration where a GPU is + * catastrophic is also the one where it does not fail. + */ + readonly device?: string, ) { this.model = model; } @@ -215,6 +234,7 @@ export class TransformersEmbeddingProvider implements EmbeddingProvider { }; const extractor = await transformers.pipeline("feature-extraction", this.model, { dtype: this.dtype, + ...(this.device === undefined ? {} : { device: this.device }), }); // `model_max_length` is a getter with no setter in @huggingface/transformers, diff --git a/src/handoff/sqlite-index.ts b/src/handoff/sqlite-index.ts index ae5943e8..83254f8b 100644 --- a/src/handoff/sqlite-index.ts +++ b/src/handoff/sqlite-index.ts @@ -533,6 +533,7 @@ export class SqliteHandoffIndex implements SessionService { tools: this.tools, redirectedTools: this.redirectedTools, vectorModel: this.embeddingProvider.model, + vectorDevice: this.embeddingProvider.device ?? null, }); } diff --git a/src/handoff/status.ts b/src/handoff/status.ts index d9d8a08a..d9630c6c 100644 --- a/src/handoff/status.ts +++ b/src/handoff/status.ts @@ -32,6 +32,8 @@ interface StatusInputs { tools: StatusToolRuntime[]; redirectedTools: string[]; vectorModel: string; + /** Execution provider the indexer will load on; see HandoffStatus. */ + vectorDevice: string | null; } /** @@ -66,7 +68,7 @@ function indexedByTool(db: DatabaseHandle, scopedRoot: string): Map { - const { db, scopedRoot, projectRoot, dbPath, tools, redirectedTools, vectorModel } = inputs; + const { db, scopedRoot, projectRoot, dbPath, tools, redirectedTools, vectorModel, vectorDevice } = inputs; // Scoped like the read paths. Unscoped counts disagreed with what the // retrieval tools return, and a status saying "3 sessions" for a project // whose searches return one is the report that makes a scoping bug look @@ -115,6 +117,7 @@ export async function buildStatus(inputs: StatusInputs): Promise vector_segment_backlog: countUnvectorizedSegments(db, vectorModel), vector_ms_per_segment: numericSetting(db, "vector_ms_per_segment"), vector_model: vectorModel, + vector_device: vectorDevice, embedding_error: getSetting(db, "last_error:embeddings"), redirected_tools: redirectedTools, tools: toolStatuses, diff --git a/src/handoff/types.ts b/src/handoff/types.ts index 981ec08e..a86d072a 100644 --- a/src/handoff/types.ts +++ b/src/handoff/types.ts @@ -62,6 +62,15 @@ export interface HandoffStatus { vector_segment_backlog: number; vector_ms_per_segment: number | null; vector_model: string; + /** + * ONNX execution provider embedding actually runs on, or null when this + * machine has not been calibrated and is therefore on the CPU default. + * + * Reported because it is otherwise invisible: `xtctx calibrate` writing a + * verdict and the indexer using it are two different things, and a verdict + * that is recorded but never read looks exactly like one that works. + */ + vector_device: string | null; /** Last semantic-search failure, or null. Non-null means hybrid search is * silently answering from keyword only. */ embedding_error: string | null; diff --git a/src/runtime/services.ts b/src/runtime/services.ts index bbd07fce..05b22098 100644 --- a/src/runtime/services.ts +++ b/src/runtime/services.ts @@ -6,6 +6,7 @@ import { defaultEmbeddingConfig, parseEmbeddingConfig, } from "../handoff/embedding-config.js"; +import { readDeviceVerdict } from "../handoff/device.js"; import { SqliteHandoffIndex } from "../handoff/sqlite-index.js"; import type { SessionService } from "../handoff/types.js"; import { SUPPORTED_TOOLS, createDefaultScrapers } from "../tools/sources.js"; @@ -123,7 +124,9 @@ export async function createProjectServices( // Provider comes from config, not from whatever happens to be in the // environment — an OPENAI_API_KEY sitting around must not opt a project // into uploading transcript text. - embeddingProvider: config.error ? undefined : createEmbeddingProvider(config.embedding), + embeddingProvider: config.error + ? undefined + : createEmbeddingProvider(config.embedding, (await readDeviceVerdict())?.device), minSemanticCosine: config.embedding.minSemanticCosine, minConfidentCosine: config.embedding.minConfidentCosine, }, diff --git a/tests/handoff/device-calibration.test.ts b/tests/handoff/device-calibration.test.ts new file mode 100644 index 00000000..c38f3a37 --- /dev/null +++ b/tests/handoff/device-calibration.test.ts @@ -0,0 +1,169 @@ +/** + * Choosing an embedding device has to fail toward the CPU. + * + * The measurement that produced this code, run on three operating systems on + * 2026-09-21: on a Windows machine with no real GPU, WebGPU does not fail. It + * finds a software adapter, initialises cleanly, returns numerically correct + * vectors, and runs at 1784.6ms per segment against the CPU's 35.4 — fifty + * times slower, with no error anywhere and no symptom except a vector backlog + * that never drains. + * + * So every ambiguous case here resolves to the CPU, and the tests below are + * mostly about ambiguity rather than about picking a winner. + */ +import { mkdtemp, rm, writeFile, mkdir } from "node:fs/promises"; +import { existsSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { + chooseDevice, + deviceCandidates, + deviceFingerprint, + readDeviceVerdict, + workerArgv, +} from "@xtctx/handoff/device"; + +let home = ""; + +beforeEach(async () => { + home = await mkdtemp(join(tmpdir(), "xtctx-device-")); +}); + +afterEach(async () => { + await rm(home, { recursive: true, force: true }); +}); + +async function writeVerdict(body: unknown): Promise { + await mkdir(join(home, ".xtctx"), { recursive: true }); + await writeFile(join(home, ".xtctx", "device.json"), JSON.stringify(body), "utf-8"); +} + +describe("chooseDevice", () => { + it("takes a decisive win over the CPU", () => { + expect( + chooseDevice([ + { device: "cpu", msPerSegment: 20.1 }, + { device: "dml", msPerSegment: 6 }, + { device: "webgpu", msPerSegment: 10.1 }, + ]), + ).toBe("dml"); + }); + + it("stays on the CPU when the GPU is slower", () => { + // The windows-latest runner, exactly: WebGPU ran, returned correct + // vectors, and took fifty times as long. + expect( + chooseDevice([ + { device: "cpu", msPerSegment: 35.4 }, + { device: "webgpu", msPerSegment: 1784.6 }, + ]), + ).toBe("cpu"); + }); + + it("stays on the CPU when the win is inside the noise", () => { + // A 10% gap on one short run on a machine doing other things is not + // evidence, and switching is permanent. + expect( + chooseDevice([ + { device: "cpu", msPerSegment: 20 }, + { device: "webgpu", msPerSegment: 18.5 }, + ]), + ).toBe("cpu"); + }); + + it("stays on the CPU when no GPU could be measured at all", () => { + expect( + chooseDevice([ + { device: "cpu", msPerSegment: 32.8 }, + { device: "webgpu", msPerSegment: null, error: "No supported adapters" }, + ]), + ).toBe("cpu"); + }); + + it("stays on the CPU when the CPU baseline itself failed", () => { + // Nothing to compare against. A GPU that looks fast against no baseline + // is the same guess this mechanism exists to stop making. + expect( + chooseDevice([ + { device: "cpu", msPerSegment: null, error: "out of memory" }, + { device: "dml", msPerSegment: 6 }, + ]), + ).toBe("cpu"); + }); +}); + +describe("deviceCandidates", () => { + it("offers DirectML only on Windows", () => { + // Asking for it elsewhere throws immediately rather than measuring: + // `Unsupported device: "dml". Should be one of: coreml, webgpu, cpu`. + expect(deviceCandidates("win32")).toContain("dml"); + expect(deviceCandidates("darwin")).not.toContain("dml"); + expect(deviceCandidates("linux")).not.toContain("dml"); + }); + + it("always measures the CPU, because it is what everything is compared to", () => { + for (const platform of ["win32", "darwin", "linux"]) { + expect(deviceCandidates(platform)).toContain("cpu"); + } + }); +}); + +describe("readDeviceVerdict", () => { + it("returns nothing when this machine has never been calibrated", async () => { + expect(await readDeviceVerdict({ home })).toBeNull(); + }); + + it("ignores a verdict measured for a different configuration", async () => { + // The fingerprint carries the model and the core count: a verdict is about + // this machine embedding this model, and the CPU arm is what the GPU had + // to beat. A 4-core laptop and a 24-core desktop reach different answers + // from identical hardware. + await writeVerdict({ device: "dml", measured: [], fingerprint: "someone-else", measuredAt: "" }); + + expect(await readDeviceVerdict({ home })).toBeNull(); + }); + + it("ignores a device this platform does not offer", async () => { + await writeVerdict({ + device: "cuda", + measured: [], + fingerprint: deviceFingerprint(), + measuredAt: "", + }); + + expect(await readDeviceVerdict({ home })).toBeNull(); + }); + + it("returns the verdict when it matches this configuration", async () => { + await writeVerdict({ + device: "cpu", + measured: [], + fingerprint: deviceFingerprint(), + measuredAt: "2026-09-21T00:00:00.000Z", + }); + + expect((await readDeviceVerdict({ home }))?.device).toBe("cpu"); + }); + + it("falls back to no verdict rather than throwing on an unreadable cache", async () => { + // A machine that cannot read its own cache embeds on the CPU, which is + // what it did before any of this existed. + await mkdir(join(home, ".xtctx"), { recursive: true }); + await writeFile(join(home, ".xtctx", "device.json"), "{not json", "utf-8"); + + expect(await readDeviceVerdict({ home })).toBeNull(); + }); +}); + +describe("workerArgv", () => { + it("resolves to a file that exists", () => { + // The calibration spawns this by path. `tsc` compiles the `src` tree and + // copies nothing, so a worker written as plain `.js` would be absent from + // `dist` and every device would report "no result" for the published + // package while passing every test here. + const [entry] = workerArgv(); + + expect(existsSync(entry)).toBe(true); + }); +}); From 03983ae7bfad13c15c785885adb291ed1e2fc1b0 Mon Sep 17 00:00:00 2001 From: Felix Stubner Date: Mon, 21 Sep 2026 15:55:48 +0100 Subject: [PATCH 13/34] feat(calibrate): measure properly, and run it automatically before a long embed MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two corrections, both prompted by asking why calibration was neither automatic nor simply "pick the fastest". A single timed pass is not a measurement. The warmup batch was two segments, which is enough for the CPU and nowhere near enough for a GPU — graph compilation and buffer allocation stayed inside the timed window. Same desktop, same 16 segments, only the warmup differing: 2 segments, 1 pass cpu 20.1 dml 6.0 webgpu 10.1 -> dml 2 segments, best of 3 cpu 21.0 dml 4.6 webgpu 3.4 -> webgpu full pass, best of 3 cpu 18.6 dml 3.1 webgpu 3.4 -> dml The short warmup was not measuring a slower device, it was measuring a device still starting up, and it ranked WebGPU as the slower GPU when it is within 7% of the faster one. The last row is three consecutive runs varying by 0.0 on DirectML and 0.1 on WebGPU. The 5.1ms/segment DirectML figure in the docs was warmup-contaminated the same way; the real number on this machine is ~3.1, so the GPU win is ~6x rather than ~4x. With a measurement that repeats, the margin over the CPU drops from 1.3x to 1.1x. It was 1.3 to cover the noise in a single pass; taking the fastest of three passes removes most of that, since other load on the machine only ever makes a pass slower. The rule is "pick the fastest device", and this is how close to 1.0 that rule can honestly get. And it now runs automatically inside `xtctx scan --embed`, which is the one path where the arithmetic is not close: that command is about to spend hours, and a minute spent finding out the GPU is ten times faster is repaid inside the first one. `--no-calibrate` skips it. Still not automatic anywhere else, and for a stated reason rather than caution. The MCP server answers a tool call inside a four-second budget and must not spawn three model-loading processes behind it. `setup` would become an 86MB model download nobody asked for. --- docs/embedding-performance.md | 73 +++++++++++++++++++++++++++++------ src/cli/index.ts | 7 +++- src/cli/scan.ts | 55 ++++++++++++++++++++++++++ src/handoff/device-worker.ts | 44 ++++++++++++++++----- src/handoff/device.ts | 20 ++++++---- 5 files changed, 171 insertions(+), 28 deletions(-) diff --git a/docs/embedding-performance.md b/docs/embedding-performance.md index e0662d36..9c3b61ba 100644 --- a/docs/embedding-performance.md +++ b/docs/embedding-performance.md @@ -116,18 +116,23 @@ and the symptom is a backlog that never drains rather than anything that looks like a failure. That rules out a silent chain, not the feature. What it leaves is a device -that has to be *chosen on evidence* — a short timed probe against the real -CPU path on first use, or an explicit opt-in — which is a bigger piece of -work than passing an option, and is the reason this still is not implemented. +*chosen on evidence*: a short timed run against the real CPU path on this +machine. That is what `xtctx calibrate` does — see "Calibration, as +implemented" below. **Vectors agree everywhere.** Worst pair across every device that ran on every runner is 0.999999. Whatever selects the device, it does not become part of vector identity, and a machine that ends up on CPU shares an index with one that does not. -DirectML remains the best result seen anywhere (5.1ms/segment locally, 0.8 -cores) and is Windows-with-a-real-GPU only: the Windows runner has no display -adapter and DirectML said so plainly rather than degrading. +DirectML is Windows-with-a-real-GPU only, and that is a feature here: the +Windows runner has no display adapter and DirectML said so plainly rather than +degrading, which is exactly the check WebGPU failed to perform. + +Every ms/segment figure in this table and the one above it is warmup- +contaminated, including the 5.1 quoted for DirectML — see "Warmup" below. The +relative story survives; the absolute numbers are roughly 1.5–2x pessimistic +on the GPU rows. ## Multi-process embedding: no headroom worth taking @@ -248,11 +253,8 @@ name. Ranked by value against effort, on the evidence above: -1. **GPU chosen by measurement.** ~6x on a real GPU, identical vectors, - already installed. No longer blocked on portability evidence — that evidence - arrived and said a fallback *chain* is the wrong shape, because GPU-less - Windows succeeds at 50x the cost instead of failing. What it needs is a - first-use timed probe or an explicit opt-in. +1. **GPU chosen by measurement.** Done — `xtctx calibrate`, and automatically + inside `xtctx scan --embed`. See below. 2. **Batch size 16.** ~10–20%, no vector change, one constant. Blocked on a cleaner measurement. 3. **bge-small with its own thresholds.** Better retrieval at 1.8x the @@ -260,3 +262,52 @@ Ranked by value against effort, on the evidence above: Closed, with reasons above: quantization, segment caching, multi-process embedding, and (from `DEFAULT_EMBEDDING_MODEL`) mpnet and static models. + +## Warmup: a short one inverts the ranking + +Every per-segment figure in this file above the three-OS table was taken after +a two-segment warmup batch, which is enough for the CPU and not remotely +enough for a GPU. Graph compilation and buffer allocation are a one-off cost, +so a short warmup leaves them inside the timed window, and they are large +enough on the GPU paths to change which device looks fastest. + +Same desktop, same 16 segments, only the warmup differing: + +| warmup | cpu | dml | webgpu | winner | +| --- | --- | --- | --- | --- | +| 2 segments, 1 timed pass | 20.1 | 6.0 | 10.1 | dml | +| 2 segments, best of 3 | 21.0 | 4.6 | 3.4 | webgpu | +| **full pass, best of 3** | **18.4–21.3** | **3.1** | **3.3** | **dml** | + +The last row is three consecutive runs and varies by 0.0 on DirectML and 0.1 +on WebGPU, against a CPU arm that still moves by 3ms with machine load. That +is the measurement `xtctx calibrate` takes. + +Two things follow. The earlier DirectML figure of 5.1ms per segment was also +warmup-contaminated and the real number here is about 3.1, so the GPU win on +this machine is ~6x rather than ~4x. And a single timed pass is not a +measurement: it ranked WebGPU as the slower GPU when it is within 7% of the +faster one. + +## Calibration, as implemented + +`xtctx calibrate` times the model on every execution provider the platform +offers, each in its own process, takes the fastest of three passes per device, +and writes the verdict to `~/.xtctx/device.json` keyed by platform, arch, core +count, model and dtype. `xtctx scan --embed` runs it automatically when the +machine has no verdict yet — a minute against the hours that command is about +to spend — and `--no-calibrate` skips it. + +Nothing else calibrates. The MCP server answers a tool call inside a +four-second budget and must not spawn three model-loading processes behind it; +it reads the verdict and nothing more. `xtctx status` prints the device, read +off the provider rather than off the cache file, so the line is evidence that +the indexer is using it rather than evidence that a file exists. + +The decision rule is "fastest measured device", with a 1.1x margin over the +CPU, and every case where no comparison exists resolves to the CPU: a device +that would not initialise, and a run where the CPU arm itself failed. + +Measured end to end on this desktop, `xtctx scan --embed` before and after: +**551.9ms per window on the CPU against 50.7ms on the GPU**, taking this +project's remaining backlog from about 24 minutes to about 1.5. diff --git a/src/cli/index.ts b/src/cli/index.ts index 30e354f8..7f7b18d7 100644 --- a/src/cli/index.ts +++ b/src/cli/index.ts @@ -147,12 +147,17 @@ export async function main(argv = process.argv): Promise { "Also embed every window, so semantic search covers the whole history (slow: hours on a large one)", false, ) + .option( + "--no-calibrate", + "With --embed, skip measuring which device embeds fastest on this machine", + ) .description("Scan the enabled transcript stores into this project's index, then exit") - .action(async (options: { project?: string; embed?: boolean }) => { + .action(async (options: { project?: string; embed?: boolean; calibrate?: boolean }) => { const globalOptions = program.opts<{ project?: string }>(); await runScan({ projectPath: options.project ?? globalOptions.project, embed: options.embed, + calibrate: options.calibrate, }); }); diff --git a/src/cli/scan.ts b/src/cli/scan.ts index 0a9c958d..3a09a312 100644 --- a/src/cli/scan.ts +++ b/src/cli/scan.ts @@ -1,3 +1,4 @@ +import { calibrateEmbeddingDevice, readDeviceVerdict } from "../handoff/device.js"; import { createProjectServices } from "../runtime/services.js"; import type { SessionService } from "../handoff/types.js"; import { formatDuration } from "../utils/duration.js"; @@ -12,6 +13,57 @@ interface ScanOptions { * would start hours of embedding every time an agent opens a large project. */ embed?: boolean; + /** + * Set false to embed on whatever device is already chosen, without + * measuring first. For anyone who would rather start embedding this second + * than spend a minute finding out it could have been ten times faster. + */ + calibrate?: boolean; +} + +/** + * Measure this machine's embedding devices, unless it already has been. + * + * Automatic HERE and nowhere else, because this is the one path where the + * arithmetic is not close: `--embed` is about to spend hours, and a minute + * spent finding out the GPU is ten times faster is repaid within the first + * one. Measured on this maintainer's desktop, the same command went from + * 551.9ms per window to 50.7. + * + * Not automatic in the MCP server, which answers an agent's tool call inside a + * four-second budget and must not spawn three model-loading processes behind + * it. Not automatic in `setup` either, which would turn "wire up my project" + * into an 86MB model download the user did not ask for. + * + * Runs before the project is opened, so an unconfigured project pays for a + * calibration it then cannot use. That costs a minute once, in a path that is + * already a user error, and the verdict is cached for the next project on the + * machine — against the alternative of opening the index, checking, and + * reopening it with a different provider. + */ +async function calibrateIfNeeded(): Promise { + if (await readDeviceVerdict()) { + return; + } + + process.stdout.write( + "Measuring this machine's embedding devices first — about a minute, once.\n" + + " (skip with --no-calibrate; re-run later with `xtctx calibrate --force`)\n", + ); + try { + const verdict = await calibrateEmbeddingDevice(); + const cpu = verdict.measured.find((row) => row.device === "cpu")?.msPerSegment; + const chosen = verdict.measured.find((row) => row.device === verdict.device)?.msPerSegment; + const speedup = + cpu && chosen && verdict.device !== "cpu" ? ` (${(cpu / chosen).toFixed(1)}x faster)` : ""; + process.stdout.write(` embedding on ${verdict.device}${speedup}\n\n`); + } catch (error) { + // Never blocks the embed. Failing to work out which device is fastest is + // not a reason to refuse to use the one that has always worked. + process.stderr.write( + ` could not measure devices, embedding on the default: ${error instanceof Error ? error.message : String(error)}\n\n`, + ); + } } /** @@ -28,6 +80,9 @@ interface ScanOptions { * internal flag: "warm the index" is a reasonable thing to want to do. */ export async function runScan(options: ScanOptions = {}): Promise { + if (options.embed && options.calibrate !== false) { + await calibrateIfNeeded(); + } const services = await createProjectServices(options.projectPath); const startedAt = Date.now(); diff --git a/src/handoff/device-worker.ts b/src/handoff/device-worker.ts index c9441f07..7cbc8e83 100644 --- a/src/handoff/device-worker.ts +++ b/src/handoff/device-worker.ts @@ -25,6 +25,14 @@ const { values } = parseArgs({ strict: false, }); +/** + * Timed passes per device; the fastest is reported. + * + * Three, because the model load dominates this process's cost and two extra + * passes over sixteen segments are cheap next to it. + */ +const PASSES = 3; + const device = values.device; const segmentCount = Math.max(1, Number.parseInt(String(values.segments), 10) || 16); @@ -49,20 +57,38 @@ try { options, ); - // One untimed batch first. The first call carries graph compilation and - // buffer allocation, a one-off cost that is not what per-segment throughput - // means — and it is much larger on the GPU paths, so including it would - // bias the comparison toward the CPU. - await extractor(segments.slice(0, 2), { pooling: "mean", normalize: true }); - - const startedAt = performance.now(); + // One full untimed pass first, not a token batch of two. + // + // The first call carries graph compilation and buffer allocation, which is a + // one-off cost rather than throughput — and on the GPU paths it is large + // enough to invert the ranking. Measured on this desktop with a 2-segment + // warmup: dml 6.0ms and webgpu 10.1ms per segment, so DirectML won. With a + // full warmup: dml 4.6 and webgpu 3.4, so WebGPU wins. The short warmup was + // not measuring a slower device, it was measuring a device still starting up. await extractor(segments, { pooling: "mean", normalize: true }); - const elapsed = performance.now() - startedAt; + + // Best of several passes, not one. + // + // The noise being removed is other load on the machine: a browser, a build, + // another agent. That only ever makes a pass SLOWER, so the minimum is the + // closest this gets to the device's real throughput, and taking it is what + // lets the caller compare devices on a small margin instead of needing a + // large one to be sure the gap is not someone else's CPU time. + // + // Repeats happen here rather than by spawning again because the model is + // already loaded: a second pass costs a second pass, not another load. + let best = Number.POSITIVE_INFINITY; + for (let pass = 0; pass < PASSES; pass += 1) { + const startedAt = performance.now(); + await extractor(segments, { pooling: "mean", normalize: true }); + best = Math.min(best, performance.now() - startedAt); + } process.stdout.write( `\n__XTCTX_DEVICE__${JSON.stringify({ device, - msPerSegment: Number((elapsed / segments.length).toFixed(1)), + msPerSegment: Number((best / segments.length).toFixed(1)), + passes: PASSES, })}\n`, ); } catch (error) { diff --git a/src/handoff/device.ts b/src/handoff/device.ts index d853a6a4..f52b3c53 100644 --- a/src/handoff/device.ts +++ b/src/handoff/device.ts @@ -125,14 +125,20 @@ export function calibrationSegments(count: number, chars = 1000): string[] { /** * How much faster than CPU a device has to be before it is worth switching to. * - * Not 1.0. The calibration is one short run on a machine that is doing other - * things, and a device that wins by 5% is inside that noise — while switching - * to it is a permanent change to how every future embed runs. Measured gaps - * that are real were 3x to 6x; the ones to reject were 0.02x. Nothing observed - * so far lands anywhere near 1.3, which is the point: the threshold only has - * to separate a decisive win from noise. + * The rule is "pick the fastest", and this is how close to 1.0 that rule can + * honestly get rather than a preference for the CPU. + * + * It was 1.3 when each device got one timed pass, which is a measurement that + * cannot distinguish a 10% device difference from a browser starting up + * mid-run. The worker now takes the fastest of three passes, and since other + * load only ever makes a pass slower, the minimum is close to the device's + * real throughput — so the margin only has to cover what is left. + * + * Every gap measured so far is far outside it either way: 3x to 6x for the + * wins, 0.02x for the one to reject. A margin this small changes the answer + * only in cases nobody has actually observed. */ -const MIN_SPEEDUP = 1.3; +const MIN_SPEEDUP = 1.1; /** * The fastest device that beats the CPU by more than noise, else the CPU. From 4ddc4e5dc018f671395d158f84ade8b56c12a3ce Mon Sep 17 00:00:00 2001 From: Felix Stubner Date: Mon, 21 Sep 2026 16:04:50 +0100 Subject: [PATCH 14/34] fix: finish embedding on its own, instead of needing a command nobody knows MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Asked why `xtctx scan --embed` has to be run by hand, the answer turned out to be that nothing else ever finishes the job. Searches vectorize about sixteen windows per call. By this repository's own measurement that leaves a 9,232-window project needing on the order of 570 searches to cover its history — so semantic search was keyword-only in practice on any real project, while still paying six seconds a call for the privilege. The only way out was knowing to run a command. Three comments said the session-start hook launched a detached scan that did this. It does not and never has: the hook does a deliberate no-scan read, and nothing under `src` spawned a process except the Antigravity client. The comments are corrected rather than left describing a mechanism that was never built. The MCP server already warms the index at session start. It now also drains the vector backlog there, bounded by this machine's own measured rate: the remainder has to fit in fifteen minutes or it is left alone. The bound is the point — unconditional background embedding is what would make a large project on a CPU unusable, since that path uses nine to eleven of twenty-four cores. One threshold covers both cases because the estimate is built from the measured rate: a large history is about eight minutes on a calibrated GPU and about eighty-five on the CPU. Device calibration is what made the affordable case the common one, which is why this lands now and not before. Verified against the built CLI by starting the server the way a host does: this project went from 1,340 windows outstanding to 3,964 of 3,964 vectorized, in the background, with no command run by hand. `estimateVectorBacklog` gains `etaMs` so the budget check and the status line share one estimate rather than reimplementing it. --- src/cli/index.ts | 65 ++++++++++++++++++++++++++++++++++++ src/cli/scan.ts | 12 +++---- src/handoff/sqlite-index.ts | 10 +++--- src/utils/duration.ts | 12 ++++--- tests/utils/duration.test.ts | 12 ++++--- 5 files changed, 91 insertions(+), 20 deletions(-) diff --git a/src/cli/index.ts b/src/cli/index.ts index 7f7b18d7..3dc48465 100644 --- a/src/cli/index.ts +++ b/src/cli/index.ts @@ -10,6 +10,7 @@ import { createProjectServices } from "../runtime/services.js"; import { startMcpServer } from "../mcp/server.js"; import type { SessionService } from "../handoff/types.js"; import { readXtctxPackage } from "../utils/package-info.js"; +import { estimateVectorBacklog } from "../utils/duration.js"; const { version: CLI_VERSION } = readXtctxPackage(import.meta.url); @@ -230,11 +231,75 @@ export async function main(argv = process.argv): Promise { async function warmIndex(sessions: SessionService): Promise { try { await sessions.listRecentSessions(1); + await sessions.whenScanSettled?.(); + await drainVectorsIfAffordable(sessions); } catch { // Deliberately silent; see above. } } +/** + * Longest background embed worth starting without being asked. + * + * The server lives for the session, so the constraint is not time but how much + * of the machine this takes while an agent is working. On a calibrated GPU + * that is about 0.9 cores; on the CPU path it is nine to eleven of twenty-four, + * which is intrusive enough that it should not start behind someone's back for + * an hour. + * + * One threshold covers both, because the estimate is built from this machine's + * own measured rate: fifteen minutes is most of a large history on the GPU + * (9,232 windows at 50.7ms is about eight) and excludes the same history on + * the CPU (at 551.9ms, about eighty-five). + */ +const BACKGROUND_EMBED_BUDGET_MS = 15 * 60 * 1000; + +/** + * Work the vector backlog down in the background, when it is cheap enough. + * + * Until this existed, nothing ever finished embedding a real history. Searches + * vectorize about sixteen windows per call, which by this project's own + * measurement leaves a 9,232-window project needing on the order of 570 + * searches — so semantic search was keyword-only in practice while still + * paying six seconds a call for the privilege. The only way out was knowing to + * run `xtctx scan --embed` by hand, which is not something a user should have + * to know. + * + * Three comments in this repository claimed the session-start hook launched a + * detached scan that did this. No such code has ever existed: the hook does a + * deliberate no-scan read, and nothing in `src` spawned a process except the + * Antigravity client. + * + * Bounded rather than unconditional, because "embed everything in the + * background" is exactly what would make a large project on a CPU unusable. + * What changed is that device calibration made the affordable case the common + * one. + */ +async function drainVectorsIfAffordable(sessions: SessionService): Promise { + if (!sessions.embedBacklog) { + return; + } + + const status = await sessions.getStatus(); + const { remaining, etaMs } = estimateVectorBacklog( + status.retrieval_units, + status.vectorized_units, + status.vector_ms_per_unit, + { backlog: status.vector_segment_backlog, msPerSegment: status.vector_ms_per_segment }, + ); + if (remaining === 0) { + return; + } + // No estimate means nothing has embedded yet on this machine, so there is no + // measured rate to judge affordability by. Searches still vectorize + // incrementally, which is what produces the rate this needs. + if (etaMs === null || etaMs > BACKGROUND_EMBED_BUDGET_MS) { + return; + } + + await sessions.embedBacklog(); +} + function shouldStartMcp(argv: string[]): boolean { if (argv.length > 2) { return false; diff --git a/src/cli/scan.ts b/src/cli/scan.ts index 3a09a312..a4bdb8cf 100644 --- a/src/cli/scan.ts +++ b/src/cli/scan.ts @@ -8,9 +8,10 @@ interface ScanOptions { /** * Also embed every window the scan leaves without a vector. * - * Off by default, and the default is the important half: the session-start - * hook launches this command detached, so draining here unconditionally - * would start hours of embedding every time an agent opens a large project. + * Off by default. The MCP server now drains the backlog in the background + * at session start when this machine's measured rate says it can finish + * inside a budget, so this flag is the "do it all now, however long it + * takes" escape hatch rather than the only way it ever happens. */ embed?: boolean; /** @@ -72,9 +73,8 @@ async function calibrateIfNeeded(): Promise { * The MCP server scans on demand and answers within a budget, and the * session-start hook reads without scanning at all. Between them, a project * that has not been asked anything yet has an empty index — so the first - * session after another tool's work starts cold. This is the piece that - * fills that gap: the hook launches it detached, and it runs to completion - * with nobody waiting on it. + * session after another tool's work starts cold. This runs to completion + * instead of to a budget, which is what fills that gap. * * Also usable by hand, which is why it is a public command and not an * internal flag: "warm the index" is a reasonable thing to want to do. diff --git a/src/handoff/sqlite-index.ts b/src/handoff/sqlite-index.ts index 83254f8b..c32bfc72 100644 --- a/src/handoff/sqlite-index.ts +++ b/src/handoff/sqlite-index.ts @@ -1008,10 +1008,12 @@ export class SqliteHandoffIndex implements SessionService { * and every search pays the six seconds anyway. * * There is no daemon, so nothing works the backlog down between commands. - * This is the piece that does, and it is deliberately not what a scan does - * by default: the session-start hook launches `scan` detached, and draining - * there would start hours of embedding every time an agent opens a large - * project. + * Two things call this: `xtctx scan --embed`, which runs it to completion + * however long that takes, and the MCP server at session start, which runs + * it only when this machine's measured rate says the remainder fits in a + * budget. The bound is the whole point of the second caller — unconditional + * background embedding is what would make a large project on a CPU + * unusable. */ async embedBacklog(onProgress?: (embedded: number, total: number) => void): Promise { await this.whenScanSettled(); diff --git a/src/utils/duration.ts b/src/utils/duration.ts index d5bf12d5..e1be0619 100644 --- a/src/utils/duration.ts +++ b/src/utils/duration.ts @@ -44,21 +44,23 @@ export function estimateVectorBacklog( vectorizedUnits: number, msPerUnit: number | null | undefined, segments?: { backlog: number; msPerSegment: number | null | undefined }, -): { remaining: number; eta: string | null } { +): { remaining: number; eta: string | null; etaMs: number | null } { const remaining = Math.max(0, retrievalUnits - vectorizedUnits); if (remaining === 0) { - return { remaining, eta: null }; + return { remaining, eta: null, etaMs: null }; } const msPerSegment = segments?.msPerSegment; if (segments && msPerSegment !== null && msPerSegment !== undefined && msPerSegment > 0) { - return { remaining, eta: formatDuration(segments.backlog * msPerSegment) }; + const etaMs = segments.backlog * msPerSegment; + return { remaining, eta: formatDuration(etaMs), etaMs }; } // No segment rate yet — nothing has embedded since this was added, or the // index predates it. The window rate is a worse estimate, not no estimate. if (msPerUnit === null || msPerUnit === undefined || !(msPerUnit > 0)) { - return { remaining, eta: null }; + return { remaining, eta: null, etaMs: null }; } - return { remaining, eta: formatDuration(remaining * msPerUnit) }; + const etaMs = remaining * msPerUnit; + return { remaining, eta: formatDuration(etaMs), etaMs }; } diff --git a/tests/utils/duration.test.ts b/tests/utils/duration.test.ts index 5290dadc..1933159d 100644 --- a/tests/utils/duration.test.ts +++ b/tests/utils/duration.test.ts @@ -21,19 +21,19 @@ describe("formatDuration", () => { describe("estimateVectorBacklog", () => { it("turns a backlog into a duration", () => { - expect(estimateVectorBacklog(1770, 8, 18)).toEqual({ remaining: 1762, eta: "31.7s" }); + expect(estimateVectorBacklog(1770, 8, 18)).toEqual({ remaining: 1762, eta: "31.7s", etaMs: 31716 }); }); it("gives no estimate when nothing has been measured", () => { // An estimate invented from no measurement is worse than no estimate: the // first scan on a fresh index has no rate yet, and guessing one would put // a number in front of a user that nothing supports. - expect(estimateVectorBacklog(1770, 8, null)).toEqual({ remaining: 1762, eta: null }); - expect(estimateVectorBacklog(1770, 8, 0)).toEqual({ remaining: 1762, eta: null }); + expect(estimateVectorBacklog(1770, 8, null)).toEqual({ remaining: 1762, eta: null, etaMs: null }); + expect(estimateVectorBacklog(1770, 8, 0)).toEqual({ remaining: 1762, eta: null, etaMs: null }); }); it("reports nothing outstanding once the corpus is covered", () => { - expect(estimateVectorBacklog(1770, 1770, 18)).toEqual({ remaining: 0, eta: null }); + expect(estimateVectorBacklog(1770, 1770, 18)).toEqual({ remaining: 0, eta: null, etaMs: null }); // More vectors than windows is possible mid-rebuild; it is not a negative // backlog. expect(estimateVectorBacklog(10, 12, 18).remaining).toBe(0); @@ -54,7 +54,7 @@ describe("estimateVectorBacklog", () => { const backlog = estimateVectorBacklog(1000, 900, 10, { backlog: 800, msPerSegment: 5 }); // 800 segments x 5ms, not 100 windows x 10ms. - expect(backlog).toEqual({ remaining: 100, eta: "4.0s" }); + expect(backlog).toEqual({ remaining: 100, eta: "4.0s", etaMs: 4000 }); }); it("falls back to the window rate when no segment rate has been measured", () => { @@ -63,6 +63,7 @@ describe("estimateVectorBacklog", () => { expect(estimateVectorBacklog(1000, 900, 10, { backlog: 800, msPerSegment: null })).toEqual({ remaining: 100, eta: "1.0s", + etaMs: 1000, }); }); @@ -70,6 +71,7 @@ describe("estimateVectorBacklog", () => { expect(estimateVectorBacklog(1000, 900, null, { backlog: 800, msPerSegment: null })).toEqual({ remaining: 100, eta: null, + etaMs: null, }); }); }); From d82239e3f7108b2ef3a4e321bacdbe4efb62f1d6 Mon Sep 17 00:00:00 2001 From: Felix Stubner Date: Mon, 21 Sep 2026 18:19:57 +0100 Subject: [PATCH 15/34] feat(embeddings): bge-small at 0.62/0.64, and a bake-off that reproduces MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit docs/embedding-performance.md recorded bge-small beating MiniLM and said plainly that the rows could not be reproduced: they came from scratch scripts against a temporarily patched constant, and the eval harness "runs one fixed provider and has no way to select a model or vary a threshold". The strongest result in the file was the least checkable thing in it. scripts/embedding-bakeoff.ts is that missing harness. It reuses the eval's own corpus and scoring, and its MiniLM row at 0.15/0.36 reproduces tests/eval/results/ranking-baseline.json exactly — which is what establishes that the two measure the same thing rather than merely agreeing. On that footing it reproduced every previously-unreproducible row in the doc. Sweeping finer than the original then found a better pair than the 0.55/0.65 the doc recommended. The confidence floor, not the semantic floor, is what was destroying vector mode: at 0.55/0.65 vector scores 0.325, at 0.55/0.64 it scores 0.385, and the false-positive rate is zero at both. At 0.62/0.64, sixty queries, both models at a false-positive rate of zero: hybrid vector mrr recall@5 top1 mrr recall@5 top1 MiniLM 0.333 0.533 0.183 0.246 0.350 0.183 bge-small 0.417 0.583 0.267 0.398 0.533 0.317 Vector mode gains 62% on mrr, 52% on recall and 73% on top-1. The committed baseline regenerated to exactly those numbers from ranking.eval.test.ts, which is a second harness agreeing rather than the same one repeated. Thresholds do not transfer between models and that is the trap this closes: at MiniLM's 0.15/0.36, bge-small scores a false-positive rate of 1.00 — every deliberately unanswerable query, gibberish included, returns something. It is not worse there; it places its cosine values higher, so a floor tuned to MiniLM's distribution excludes nothing. What changed to allow this is cost. bge-small indexes the corpus in 27.5s against 15.6s, about 1.8x, and that ratio is exactly why it was rejected on 2026-09-20. Device calibration made embedding ~6x faster on a machine with a GPU and the server now drains the backlog in the background, so 1.8x of a much smaller number stopped being the deciding term. The caveat the earlier measurement carried still stands and is not claimed away: the corpus is synthetic and sixty queries. What is stronger is that the margin holds across three modes and every threshold pair swept. Both models are 384 dimensions, so the schema is unchanged. dropVectorsFromOtherModels keys on the model name, so upgrading discards every existing vector and re-embeds — paid once per project, and much cheaper than it was this morning. One test moved with the threshold: a score-reporting test used a stub cosine of 0.6, which cleared MiniLM's 0.36 floor and not bge-small's 0.64, so it failed for a reason unrelated to what it asserts. It now derives its value from MIN_CONFIDENT_COSINE. --- scripts/embedding-bakeoff.ts | 245 +++++++++++++++++++++++ src/handoff/embeddings.ts | 93 +++++---- src/handoff/ranking.ts | 4 +- tests/eval/results/ranking-baseline.json | 14 +- tests/handoff/sqlite-index.test.ts | 11 +- 5 files changed, 318 insertions(+), 49 deletions(-) create mode 100644 scripts/embedding-bakeoff.ts diff --git a/scripts/embedding-bakeoff.ts b/scripts/embedding-bakeoff.ts new file mode 100644 index 00000000..83ade73e --- /dev/null +++ b/scripts/embedding-bakeoff.ts @@ -0,0 +1,245 @@ +/** + * Score an embedding model against the eval corpus, at several thresholds. + * + * `docs/embedding-performance.md` records a bake-off that found bge-small + * beating MiniLM once each model is judged at its own thresholds, and says + * plainly that those rows cannot be reproduced: they were measured with + * scratch scripts against a temporarily patched constant, and the eval harness + * "runs one fixed provider and has no way to select a model or vary a + * threshold". That made the strongest result in the file the least checkable + * thing in it. + * + * This is that missing harness. It reuses the eval's own corpus and scoring so + * a row here is comparable with `tests/eval/results/ranking-baseline.json`, + * which is also the control: running this on the default model at the default + * thresholds must reproduce the committed baseline, or the two are measuring + * different things and nothing else here can be trusted. + * + * The thresholds sweep for free. They are applied at query time, not at embed + * time, so one indexing pass per model serves every threshold pair — which is + * what makes a sweep affordable at all, since embedding the corpus is the + * expensive part. + * + * ONE MODEL PER PROCESS. Two providers in one process fail intermittently with + * `bad allocation`, and two in separate vitest workers killed a worker + * outright (issue #101). + * + * npx tsx scripts/embedding-bakeoff.ts --model=Xenova/all-MiniLM-L6-v2 + * npx tsx scripts/embedding-bakeoff.ts --model=Xenova/bge-small-en-v1.5 + */ +import { mkdtemp, rm } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { parseArgs } from "node:util"; +import { SqliteHandoffIndex } from "../src/handoff/sqlite-index.js"; +import { TransformersEmbeddingProvider } from "../src/handoff/embeddings.js"; +import type { SessionSearchMode } from "../src/handoff/types.js"; +import type { ConversationChunk, ConversationScraper, ScraperState } from "../src/types/scraper.js"; +import { generateCorpus, type Anchor, type NegativeQuery } from "../tests/eval/corpus.js"; + +const MODES: SessionSearchMode[] = ["hybrid", "vector", "keyword"]; + +/** + * Threshold pairs to try, as (minSemanticCosine, minConfidentCosine). + * + * The spread is wide because the thing being measured is that it has to be: + * bge-small and gte-small both scored a false-positive rate of 1.00 at + * MiniLM's 0.15/0.36 — every deliberately unanswerable query, gibberish + * included, returned something. They are not worse models; they place their + * cosine values higher, so a floor tuned to one model's distribution stops + * excluding anything on another's. + */ +const SWEEP: Array<[number, number]> = [ + [0.15, 0.36], + [0.25, 0.45], + [0.35, 0.55], + [0.45, 0.55], + [0.55, 0.65], + [0.65, 0.75], + [0.75, 0.85], +]; + +/** Replays a fixed set of chunks; the corpus, not the scraper, is under test. */ +class CorpusScraper implements ConversationScraper { + constructor( + readonly tool: string, + private readonly chunks: ConversationChunk[], + ) {} + + async detect(): Promise { + return true; + } + + getStorePaths(): string[] { + return [`corpus://${this.tool}`]; + } + + async *scrape(): AsyncIterable { + yield* this.fullSync(); + } + + async *fullSync(): AsyncIterable { + for (const chunk of this.chunks) { + yield chunk; + } + } + + async getLastScrapedPosition(): Promise { + return { lastTimestamp: new Date(0) }; + } + + async saveScrapedPosition(): Promise { + return; + } +} + +interface Metrics { + mrr: number; + recallAt5: number; + top1: number; + falsePositiveRate: number; +} + +function round(value: number): number { + return Math.round(value * 1000) / 1000; +} + +/** Identical arithmetic to `score` in ranking.eval.test.ts, so rows compare. */ +function score(ranks: Array, unanswerable: number[]): Metrics { + const found = ranks.filter((rank): rank is number => rank !== null); + const sum = (values: number[]) => values.reduce((total, value) => total + value, 0); + return { + mrr: round(sum(found.map((rank) => 1 / rank)) / ranks.length), + recallAt5: round(found.filter((rank) => rank <= 5).length / ranks.length), + top1: round(found.filter((rank) => rank === 1).length / ranks.length), + falsePositiveRate: + unanswerable.length === 0 + ? 0 + : round(unanswerable.filter((count) => count > 0).length / unanswerable.length), + }; +} + +async function measure( + dbPath: string, + projectRoot: string, + tools: Array<{ tool: string; scraper: ConversationScraper }>, + provider: TransformersEmbeddingProvider, + anchors: Anchor[], + negatives: NegativeQuery[], + thresholds: [number, number], +): Promise> { + const index = new SqliteHandoffIndex(dbPath, projectRoot, tools, { + embeddingProvider: provider, + refreshBudgetMs: 600_000, + vectorBudgetMs: 600_000, + minSemanticCosine: thresholds[0], + minConfidentCosine: thresholds[1], + }); + + try { + const report: Record = {}; + for (const mode of MODES) { + const ranks: Array = []; + for (const anchor of anchors) { + const results = await index.searchSessions(anchor.query, 10, undefined, mode); + const position = results.findIndex((session) => session.session_ref === anchor.sessionRef); + ranks.push(position === -1 ? null : position + 1); + } + + // Only the genuinely unanswerable ones. The eval separates "related + // topic" queries and reports them without policing them, on the grounds + // that offering cache warming for a question about cache eviction is + // arguable rather than wrong — counting those as false positives once + // sent two rounds of work at a defect that was not there. + const unanswerable: number[] = []; + for (const negative of negatives) { + if (negative.kind === "related") continue; + const results = await index.searchSessions(negative.query, 10, undefined, mode); + unanswerable.push(results.length); + } + + report[mode] = score(ranks, unanswerable); + } + return report; + } finally { + await index.close(); + } +} + +const { values } = parseArgs({ + options: { + model: { type: "string", default: "Xenova/all-MiniLM-L6-v2" }, + /** `--sweep=0.46:0.56,0.50:0.60` to look closely at one region. */ + sweep: { type: "string" }, + }, + strict: false, +}); +const model = String(values.model); +const sweep: Array<[number, number]> = values.sweep + ? String(values.sweep) + .split(",") + .map((pair) => { + const [semantic, confident] = pair.split(":").map(Number); + if (!Number.isFinite(semantic) || !Number.isFinite(confident)) { + throw new Error(`--sweep expects pairs like 0.55:0.65, got ${JSON.stringify(pair)}`); + } + return [semantic, confident] as [number, number]; + }) + : SWEEP; + +const tempDir = await mkdtemp(join(tmpdir(), "xtctx-bakeoff-")); +try { + const corpus = generateCorpus(); + const byTool = new Map(); + for (const chunk of corpus.chunks) { + byTool.set(chunk.tool, [...(byTool.get(chunk.tool) ?? []), chunk]); + } + const tools = [...byTool.entries()].map(([tool, chunks]) => ({ + tool, + scraper: new CorpusScraper(tool, chunks) as ConversationScraper, + })); + + const provider = new TransformersEmbeddingProvider(model); + const dbPath = join(tempDir, "bakeoff.db"); + + // Index and embed once. Every threshold below reads these same vectors. + process.stderr.write(`indexing the corpus under ${model}...\n`); + const indexedAt = Date.now(); + const warm = new SqliteHandoffIndex(dbPath, tempDir, tools, { + embeddingProvider: provider, + refreshBudgetMs: 600_000, + vectorBudgetMs: 600_000, + }); + await warm.listRecentSessions(1); + // `vector` rather than the default: hybrid deliberately answers from keyword + // while the model loads, so warming through it would leave the model cold. + await warm.searchSessions("warm the embedding model", 1, undefined, "vector"); + await warm.embedBacklog?.(); + await warm.close(); + const indexMs = Date.now() - indexedAt; + + process.stdout.write(`\n${model} — corpus indexed in ${(indexMs / 1000).toFixed(1)}s\n\n`); + process.stdout.write("semantic/confident mode mrr recall@5 top1 false-pos\n"); + for (const thresholds of sweep) { + const report = await measure( + dbPath, + tempDir, + tools, + provider, + corpus.anchors, + corpus.negatives, + thresholds, + ); + for (const mode of MODES) { + const row = report[mode]; + process.stdout.write( + `${`${thresholds[0]} / ${thresholds[1]}`.padEnd(19)} ${mode.padEnd(8)} ` + + `${String(row.mrr).padEnd(6)} ${String(row.recallAt5).padEnd(9)} ` + + `${String(row.top1).padEnd(6)} ${row.falsePositiveRate}\n`, + ); + } + process.stdout.write("\n"); + } +} finally { + await rm(tempDir, { recursive: true, force: true }); +} diff --git a/src/handoff/embeddings.ts b/src/handoff/embeddings.ts index fd80c863..cb1c1cd0 100644 --- a/src/handoff/embeddings.ts +++ b/src/handoff/embeddings.ts @@ -1,49 +1,64 @@ /** - * The embedding model, chosen on indexing throughput as much as on ranking. + * The embedding model, chosen on retrieval quality once indexing cost stopped + * being the binding constraint. * - * mpnet-q8 ranks better on the sixty-query eval, each model at its own swept - * confidence threshold: + * bge-small-en-v1.5, at the thresholds on `MIN_SEMANTIC_COSINE` and + * `MIN_CONFIDENT_COSINE`, against MiniLM at the ones it replaced. Measured + * 2026-09-21 with `scripts/embedding-bakeoff.ts`, sixty queries, both at a + * false-positive rate of zero: * - * MiniLM fp32 mrr 0.598 recall@5 0.850 top1 0.450 at 0.36 - * mpnet q8 mrr 0.654 recall@5 0.933 top1 0.483 at 0.40 + * hybrid vector + * mrr recall@5 top1 mrr recall@5 top1 + * MiniLM 0.333 0.533 0.183 0.246 0.350 0.183 + * bge-small 0.417 0.583 0.267 0.398 0.533 0.317 * - * Those figures predate #318, which rebuilt the eval corpus to use realistic - * session lengths on the grounds that it "has been measuring a world that does - * not exist". The committed baseline moved with it: MiniLM hybrid reads - * 0.333 / 0.533 / 0.183 today, not 0.598 / 0.850 / 0.450. The COMPARISON above - * was measured on one corpus and stands; the absolute numbers no longer match - * anything reproducible, so do not quote them or compare a new model against - * them. `tests/eval/results/ranking-baseline.json` is the current truth. + * Reproduce either row by running that script; its MiniLM row at 0.15/0.36 + * reproduces `tests/eval/results/ranking-baseline.json` exactly, which is what + * establishes that it and `ranking.eval.test.ts` measure the same thing. * - * It was the default for a day on the strength of that table, which measured - * only half the question. What the table left out is what embedding actually - * costs, and the figure used at the time — 18ms per embed — came from - * benchmarking strings like "warm query number 5". A real segment is 1024 - * characters, the model's full sequence window, and costs far more. + * THRESHOLDS DO NOT TRANSFER BETWEEN MODELS, and this is the trap. At MiniLM's + * 0.15/0.36, bge-small scores a false-positive rate of 1.00 — every + * deliberately unanswerable query, gibberish included, returns something. It is + * not the worse model there; it places its cosine values higher, so a floor + * tuned to MiniLM's distribution excludes nothing. Any future model change has + * to re-sweep, and the sweep is the whole job. * - * Back to back over 291 real segments from this project's index, mpnet first: + * What changed to allow this is cost, not quality. bge-small indexes the eval + * corpus in 27.5s against MiniLM's 15.6s, about 1.8x, and that ratio is why + * this was rejected when first measured on 2026-09-20. Device calibration + * (`device.ts`) then made embedding roughly six times faster on a machine with + * a GPU, and the MCP server began draining the backlog in the background, so + * 1.8x of a much smaller number stopped being the deciding term. + * + * The caveat the earlier measurement carried still stands: the corpus is + * synthetic and sixty queries. What is stronger now is that the margin holds + * across three modes and every threshold pair swept, rather than resting on a + * single row. + * + * Both models are 384 dimensions, so nothing about the schema changes. + * `dropVectorsFromOtherModels` keys on the model name, so upgrading discards + * every existing vector and re-embeds — which is the cost of this change and + * is paid once per project. + * + * TWO REJECTIONS WORTH KEEPING, because the arguments for them are strong and + * the reasons they lose are not obvious. + * + * mpnet-q8 was the default for a day, on a table that measured only half the + * question. What it left out is what embedding actually costs, and the figure + * used at the time — 18ms per embed — came from benchmarking strings like + * "warm query number 5". A real segment is 1024 characters, the model's full + * sequence window, and costs far more. Back to back over 291 real segments + * from this project's index, mpnet first: * * mpnet q8 ~360ms per segment * MiniLM fp32 ~116ms per segment * - * and MiniLM alone in a fresh process, with the segment cap applied, ~65ms. * The absolute figures move with what else the process has loaded; the ratio - * is the durable part, and it favours MiniLM by three times or better. - * - * Over this project's ~12,000 segments that is tens of minutes of embedding - * either way, but roughly three times fewer of them. Vectorizing is budgeted - * per call, so the difference is not "slower indexing" in the background — it - * is how long a project's semantic search stays partly blind, tool call after - * tool call. - * - * Five queries of recall and two of top-1 on a sixty-query corpus do not buy - * fifty extra minutes of that. Measure throughput on real content before - * moving this again; a per-embed figure taken on short strings says nothing - * about it. + * is the durable part, and it favoured MiniLM by three times or better. + * Measure throughput on real content before moving this again; a per-embed + * figure taken on short strings says nothing about it. * - * A static model was measured against this on 2026-09-03 and rejected, which - * is worth writing down because the speed argument for one is overwhelming - * and the reason it loses is not obvious. + * A static model was measured on 2026-09-03 and rejected. * * Model2Vec statics (`minishlab/potion-base-8M`, 256 dimensions, 30MB) have no * transformer forward pass: they look each token's vector up and mean-pool. On @@ -90,14 +105,16 @@ * 2/1 is markedly better in `vector` mode than at the current 8/4. That is * recorded on `DEFAULT_WINDOW_SIZE`, where the decision belongs. */ -export const DEFAULT_EMBEDDING_MODEL = "Xenova/all-MiniLM-L6-v2"; +export const DEFAULT_EMBEDDING_MODEL = "Xenova/bge-small-en-v1.5"; /** * Weight precision to load the model at. * - * fp32 for MiniLM: the model is 86MB at full precision, so quantizing saves - * little and costs accuracy. The q8 tradeoff only mattered for mpnet, where - * fp32 was 416MB. + * fp32 for both MiniLM and bge-small: each is under 140MB at full precision, + * so quantizing saves little and costs accuracy. The q8 tradeoff only mattered + * for mpnet, where fp32 was 416MB. Measured on MiniLM, q8 was 16% faster while + * moving every vector (mean cosine 0.9889 against fp32) — and dtype is not part + * of the vector identity, so switching it would silently mix two spaces. */ export const DEFAULT_EMBEDDING_DTYPE = "fp32"; const MAX_SEQ_TOKENS = 256; diff --git a/src/handoff/ranking.ts b/src/handoff/ranking.ts index 9bd0838f..00389b17 100644 --- a/src/handoff/ranking.ts +++ b/src/handoff/ranking.ts @@ -30,7 +30,7 @@ const MAX_MATCHES_PER_SESSION = 3; * Unrelated sentence-transformer pairs sit near 0; related ones are * comfortably above this. */ -export const MIN_SEMANTIC_COSINE = 0.15; +export const MIN_SEMANTIC_COSINE = 0.62; /** * How similar the *best* window has to be before a query counts as having @@ -96,7 +96,7 @@ export const MIN_SEMANTIC_COSINE = 0.15; * * If it needs to move, move it against the eval rather than against one query. */ -export const MIN_CONFIDENT_COSINE = 0.36; +export const MIN_CONFIDENT_COSINE = 0.64; /** * Weight of the recency/continuity tie-break in the relevance modes. Small * enough that it only ever separates candidates that are otherwise equal. diff --git a/tests/eval/results/ranking-baseline.json b/tests/eval/results/ranking-baseline.json index 771480cc..ddeb1946 100644 --- a/tests/eval/results/ranking-baseline.json +++ b/tests/eval/results/ranking-baseline.json @@ -1,21 +1,21 @@ { "hybrid": { "queries": 60, - "mrr": 0.333, - "recallAt5": 0.533, - "top1": 0.183, + "mrr": 0.417, + "recallAt5": 0.583, + "top1": 0.267, "falsePositiveRate": 0, "gibberishFalsePositiveRate": 0, "relatedTopicHitRate": 1 }, "vector": { "queries": 60, - "mrr": 0.246, - "recallAt5": 0.35, - "top1": 0.183, + "mrr": 0.398, + "recallAt5": 0.533, + "top1": 0.317, "falsePositiveRate": 0, "gibberishFalsePositiveRate": 0, - "relatedTopicHitRate": 0.1 + "relatedTopicHitRate": 0.5 }, "keyword": { "queries": 60, diff --git a/tests/handoff/sqlite-index.test.ts b/tests/handoff/sqlite-index.test.ts index e6862df2..3ce26c26 100644 --- a/tests/handoff/sqlite-index.test.ts +++ b/tests/handoff/sqlite-index.test.ts @@ -4,6 +4,7 @@ import { join } from "node:path"; import { afterEach, beforeEach, describe, expect, it } from "vitest"; import { SqliteHandoffIndex } from "@xtctx/handoff/sqlite-index"; import { MAX_SEGMENTS_PER_UNIT } from "@xtctx/handoff/embeddings"; +import { MIN_CONFIDENT_COSINE } from "@xtctx/handoff/ranking"; import type { EmbeddingProvider } from "@xtctx/handoff/embeddings"; import type { ConversationChunk, ConversationScraper, ScraperState } from "@xtctx/types/scraper"; @@ -881,10 +882,16 @@ describe("search scores mean similarity", () => { } it("reports the similarity itself, not the best survivor rescaled to 1", async () => { - const results = await searchWith(0.6); + // Derived from the threshold rather than written as a literal. This was + // 0.6, which cleared MiniLM's 0.36 confidence floor; moving the default + // model to bge-small raised that floor to 0.64 and the single result this + // asserts on was filtered out before it could be scored, so a test about + // score REPORTING failed for a reason that had nothing to do with scoring. + const cosine = Math.min(0.99, MIN_CONFIDENT_COSINE + 0.1); + const results = await searchWith(cosine); expect(results).toHaveLength(1); - expect(results[0].score).toBeCloseTo(0.6, 2); + expect(results[0].score).toBeCloseTo(cosine, 2); }); it("finds nothing when nothing is actually similar", async () => { From 08b686228101d03a9684c09477e668a3158c19ee Mon Sep 17 00:00:00 2001 From: Felix Stubner Date: Mon, 21 Sep 2026 18:19:57 +0100 Subject: [PATCH 16/34] fix(status): say when a backlog is too large to drain on its own MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The MCP server drains the vector backlog at session start only while the estimate fits its budget. Above that nothing is working on it, and the status line said the same thing either way: a number of windows outstanding and a time estimate, identical in shape whether it finishes in four minutes or never. A large history on a CPU-only machine lands there the moment a model change invalidates every vector it had — which this session's switch to bge-small does to every existing project, since dropVectorsFromOtherModels deletes them all on the first open. Verified on this project's own index: 3,974 vectors to 0 in a single open, not gradually. So status now names the command when the backlog is past the budget, and the budget constant moved to utils/duration.ts beside the estimate it is compared against, because the server and status both need it and they must not drift. Covered at the layer it renders from: a test stubs the two rates measured on this project — 551.9ms/window on the CPU and 50.7ms on DirectML, the same 9,232 windows — and asserts the advisory appears for one and not the other. Confirmed to fail when the comparison is removed. --- src/cli/index.ts | 18 +-------- src/cli/status.ts | 20 +++++++++- src/utils/duration.ts | 15 ++++++++ tests/cli/status.test.ts | 39 +++++++++++++++++++ tests/utils/background-embed-budget.test.ts | 42 +++++++++++++++++++++ 5 files changed, 116 insertions(+), 18 deletions(-) create mode 100644 tests/utils/background-embed-budget.test.ts diff --git a/src/cli/index.ts b/src/cli/index.ts index 3dc48465..ec5c5aed 100644 --- a/src/cli/index.ts +++ b/src/cli/index.ts @@ -10,7 +10,7 @@ import { createProjectServices } from "../runtime/services.js"; import { startMcpServer } from "../mcp/server.js"; import type { SessionService } from "../handoff/types.js"; import { readXtctxPackage } from "../utils/package-info.js"; -import { estimateVectorBacklog } from "../utils/duration.js"; +import { BACKGROUND_EMBED_BUDGET_MS, estimateVectorBacklog } from "../utils/duration.js"; const { version: CLI_VERSION } = readXtctxPackage(import.meta.url); @@ -238,22 +238,6 @@ async function warmIndex(sessions: SessionService): Promise { } } -/** - * Longest background embed worth starting without being asked. - * - * The server lives for the session, so the constraint is not time but how much - * of the machine this takes while an agent is working. On a calibrated GPU - * that is about 0.9 cores; on the CPU path it is nine to eleven of twenty-four, - * which is intrusive enough that it should not start behind someone's back for - * an hour. - * - * One threshold covers both, because the estimate is built from this machine's - * own measured rate: fifteen minutes is most of a large history on the GPU - * (9,232 windows at 50.7ms is about eight) and excludes the same history on - * the CPU (at 551.9ms, about eighty-five). - */ -const BACKGROUND_EMBED_BUDGET_MS = 15 * 60 * 1000; - /** * Work the vector backlog down in the background, when it is cheap enough. * diff --git a/src/cli/status.ts b/src/cli/status.ts index de0c5a86..fa279aa1 100644 --- a/src/cli/status.ts +++ b/src/cli/status.ts @@ -4,7 +4,11 @@ import { inspectManagedFile, pathExists } from "../config/setup.js"; import { inspectMcpWiring, type McpWiringState } from "../config/mcp-config.js"; import { inspectSkillStatus } from "../config/skills.js"; import { createProjectServices, type ProjectServices } from "../runtime/services.js"; -import { estimateVectorBacklog, formatDuration } from "../utils/duration.js"; +import { + BACKGROUND_EMBED_BUDGET_MS, + estimateVectorBacklog, + formatDuration, +} from "../utils/duration.js"; import { readDriftLog, type DriftLogFile } from "../scrapers/drift-log.js"; import { SUPPORTED_TOOLS } from "../tools/sources.js"; import { readXtctxPackage } from "../utils/package-info.js"; @@ -112,6 +116,20 @@ export async function renderStatusBlock( `Embed ${backlog.remaining} windows outstanding, ${rate}` + `${backlog.eta ? `, about ${backlog.eta} of embedding left` : ""}`, ); + // Say whether anything is actually working on it. + // + // The MCP server drains the backlog in the background only while the + // estimate fits its budget, so on a slow machine with a large history + // nothing is. Until this line existed that state was invisible and + // indistinguishable from the one above it: semantic search quietly + // answering from keyword, forever, with a status report that looked like + // progress was being made. Naming the command is the point — the backlog + // does not drain by waiting. + if (backlog.etaMs !== null && backlog.etaMs > BACKGROUND_EMBED_BUDGET_MS) { + lines.push( + " too large to finish in the background — run `xtctx scan --embed`", + ); + } } // Only when it is not the default. A machine that has never been calibrated // is on the CPU, which is what every machine did before calibration existed, diff --git a/src/utils/duration.ts b/src/utils/duration.ts index e1be0619..3d16df07 100644 --- a/src/utils/duration.ts +++ b/src/utils/duration.ts @@ -39,6 +39,21 @@ export function formatDuration(ms: number | null | undefined): string | null { * `remaining` stays a window count. That is the number a person can see in * `Data`, and the estimate reading as a duration is the point of it. */ +/** + * Longest background embed the MCP server starts without being asked. + * + * Lives here, beside the estimate it is compared against, because two callers + * need it: the server deciding whether to drain, and `xtctx status` telling + * the user when it will not. A status line that stays silent about a backlog + * nothing is working on is how someone ends up with keyword-only search and no + * idea why. + * + * The constraint is not time but how much of the machine this takes while an + * agent is working — about 0.9 cores on a calibrated GPU, nine to eleven of + * twenty-four on the CPU path. + */ +export const BACKGROUND_EMBED_BUDGET_MS = 15 * 60 * 1000; + export function estimateVectorBacklog( retrievalUnits: number, vectorizedUnits: number, diff --git a/tests/cli/status.test.ts b/tests/cli/status.test.ts index 256f876c..3fd66d0a 100644 --- a/tests/cli/status.test.ts +++ b/tests/cli/status.test.ts @@ -28,6 +28,45 @@ describe("status", () => { await rm(homeDir, { recursive: true, force: true }); }); + it("says when a backlog is too large for anything to drain it on its own", async () => { + // The MCP server drains the vector backlog at session start only while the + // estimate fits its budget. Above that nothing is working on it, and + // without this line that state is indistinguishable from the one below it: + // both read as windows outstanding with a time estimate, while one + // finishes within minutes and the other never finishes at all. + // + // A large history on a CPU-only machine lands there the moment a model + // change invalidates every vector it had, which is what makes it worth a + // line rather than a footnote. + await setupProject({ projectPath: projectRoot, homeDir, yes: true }); + + const services = await createProjectServices(projectRoot); + try { + const real = await services.sessions.getStatus(); + const withBacklog = (msPerUnit: number) => ({ + ...real, + retrieval_units: 9232, + vectorized_units: 0, + vector_ms_per_unit: msPerUnit, + vector_segment_backlog: 0, + vector_ms_per_segment: null, + }); + + // 551.9ms/window, the CPU rate measured on this project. + services.sessions.getStatus = async () => withBacklog(551.9); + const slow = await renderStatusBlock(services, { homeDir }); + expect(slow).toContain("xtctx scan --embed"); + + // 50.7ms/window, the same history on DirectML. + services.sessions.getStatus = async () => withBacklog(50.7); + const fast = await renderStatusBlock(services, { homeDir }); + expect(fast).toContain("9232 windows outstanding"); + expect(fast).not.toContain("xtctx scan --embed"); + } finally { + await services.sessions.close(); + } + }); + it("does not report drift on a freshly wired project", async () => { // `managed-block` and `unsupported` are healthy skill-target states for // codex/antigravity/opencode/copilot-cli, not drift. Treating any diff --git a/tests/utils/background-embed-budget.test.ts b/tests/utils/background-embed-budget.test.ts new file mode 100644 index 00000000..99243b99 --- /dev/null +++ b/tests/utils/background-embed-budget.test.ts @@ -0,0 +1,42 @@ +/** + * A backlog nothing is working on has to say so. + * + * The MCP server drains the vector backlog at session start only while the + * estimate fits `BACKGROUND_EMBED_BUDGET_MS`. Above it, nothing is working the + * backlog down and the only thing that will is `xtctx scan --embed`. + * + * That state is invisible without a line about it, and indistinguishable from + * the one just above it in `xtctx status`: both read as a number of windows + * outstanding with a time estimate, while one finishes on its own within + * minutes and the other never finishes at all. The second is what a large + * history on a CPU-only machine gets after a model change invalidates every + * vector it had. + */ +import { describe, expect, it } from "vitest"; +import { BACKGROUND_EMBED_BUDGET_MS, estimateVectorBacklog } from "@xtctx/utils/duration"; + +describe("background embed budget", () => { + it("puts a large history on a slow machine over the budget", () => { + // 9,232 windows at the CPU rate measured on this project, 551.9ms/window. + const { etaMs } = estimateVectorBacklog(9232, 0, 551.9); + + expect(etaMs).not.toBeNull(); + expect(etaMs!).toBeGreaterThan(BACKGROUND_EMBED_BUDGET_MS); + }); + + it("keeps the same history under the budget once a GPU is chosen", () => { + // The same 9,232 windows at 50.7ms/window, measured on DirectML. This is + // the pairing that decides the threshold: one number, two devices, and the + // budget has to separate them. + const { etaMs } = estimateVectorBacklog(9232, 0, 50.7); + + expect(etaMs).not.toBeNull(); + expect(etaMs!).toBeLessThan(BACKGROUND_EMBED_BUDGET_MS); + }); + + it("reports no estimate when no rate has been measured yet", () => { + // Nothing has embedded on this machine, so there is no basis for judging + // affordability. The server leaves it alone rather than guessing. + expect(estimateVectorBacklog(1000, 0, null).etaMs).toBeNull(); + }); +}); From a8f604f49133230fd92546b88815d9580c4e79e1 Mon Sep 17 00:00:00 2001 From: Felix Stubner Date: Mon, 21 Sep 2026 18:33:50 +0100 Subject: [PATCH 17/34] fix(search): stop a model that cannot load reading as one still loading MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Hybrid search answers from keyword while the embedding model loads, because a cold cache takes minutes and nobody should wait for it holding a tool call open. That branch asked one question — `isReady()` — and a model still downloading and a model that will never download both answer false. So a failed load produced, on every call and for the life of the project: keyword results, the note "embedding model still loading, so this answer is keyword-only — ask again shortly for more", and an `xtctx status` with no semantic-unavailable line. The advice could never come true, and the failure was recorded nowhere, because the only code that writes `last_error:embeddings` is a catch that this early return skips. `warm()` swallowed the reason. The provider now keeps its last load error and the search path records it, so the failure reaches `xtctx status` and `xtctx_continuity_status` the same way a mid-search embedding exception already did. Loading is still retried every call and the error cleared on success, because the usual cause is a cold cache behind a flaky network that works on the next attempt — refusing to retry would turn a transient fault into a permanent one. Found by a delegated review of the failure-and-degradation paths, then confirmed against the code rather than taken on trust. Covered by a test using a provider that never loads, confirmed to fail when the recording is removed. --- src/handoff/embeddings.ts | 43 +++++++- src/handoff/sqlite-index.ts | 14 +++ tests/handoff/embedding-load-failure.test.ts | 108 +++++++++++++++++++ 3 files changed, 161 insertions(+), 4 deletions(-) create mode 100644 tests/handoff/embedding-load-failure.test.ts diff --git a/src/handoff/embeddings.ts b/src/handoff/embeddings.ts index cb1c1cd0..c14c88ce 100644 --- a/src/handoff/embeddings.ts +++ b/src/handoff/embeddings.ts @@ -145,6 +145,22 @@ export interface EmbeddingProvider { isReady?(): boolean; /** Begin loading the model without waiting for it. */ warm?(): void; + /** + * Why the last load attempt failed, or undefined if none has. + * + * `isReady()` answers "can I embed right now", and a model that is still + * downloading and a model that cannot download both answer false. Callers + * that only ask `isReady()` therefore tell the user to wait, forever, for + * something that is never going to finish — which is exactly what happened: + * `warm()` is best-effort and swallows its error, so a failed load left + * hybrid search answering from keyword and reporting "embedding model still + * loading, ask again shortly" on every call, indefinitely, with nothing in + * `xtctx status` to say otherwise. + * + * Cleared on a successful load, because the failure is worth retrying: a + * cold cache behind a flaky network fails once and succeeds next time. + */ + loadError?(): string | undefined; } type FeatureExtractionOutput = { @@ -180,6 +196,7 @@ export class TransformersEmbeddingProvider implements EmbeddingProvider { readonly model: string; private extractor: FeatureExtractionPipeline | null = null; private loading: Promise | null = null; + private lastLoadError: string | undefined; constructor( model = DEFAULT_EMBEDDING_MODEL, @@ -222,9 +239,15 @@ export class TransformersEmbeddingProvider implements EmbeddingProvider { return this.extractor !== null; } + loadError(): string | undefined { + return this.lastLoadError; + } + warm(): void { void this.getExtractor().catch(() => { - // Warming is best-effort; the next real embed call reports the failure. + // Still best-effort — nothing is thrown at the caller — but the reason + // is kept now. Swallowing it entirely made a permanent load failure + // indistinguishable from a slow first download, forever. }); } @@ -237,9 +260,21 @@ export class TransformersEmbeddingProvider implements EmbeddingProvider { return this.loading; } - this.loading = this.loadExtractor().finally(() => { - this.loading = null; - }); + this.loading = this.loadExtractor() + .then((extractor) => { + this.lastLoadError = undefined; + return extractor; + }) + .catch((error: unknown) => { + this.lastLoadError = error instanceof Error ? error.message : String(error); + throw error; + }) + .finally(() => { + // Cleared so the next call retries. A cold cache behind a flaky + // network fails once and succeeds next time, and refusing to try + // again would turn a transient fault into a permanent one. + this.loading = null; + }); return this.loading; } diff --git a/src/handoff/sqlite-index.ts b/src/handoff/sqlite-index.ts index c32bfc72..6bee4640 100644 --- a/src/handoff/sqlite-index.ts +++ b/src/handoff/sqlite-index.ts @@ -495,6 +495,20 @@ export class SqliteHandoffIndex implements SessionService { // and let the next search use the model. An explicit `vector` request is a // different matter: there is no other route, so that one waits. if (normalizedMode === "hybrid" && this.embeddingProvider.isReady?.() === false) { + // A load that FAILED is not a load still running, and this branch used + // to treat them the same. `warm()` swallows its error and `isReady()` + // stays false afterwards, so a model that could not be fetched at all + // returned keyword results with "still loading, ask again shortly" on + // every call for the life of the project — advice that could never come + // true — while `last_error:embeddings` stayed empty because the only + // code that writes it is the catch below, which this return skips. + const loadError = this.embeddingProvider.loadError?.(); + if (loadError !== undefined) { + setSetting(this.getDb(), "last_error:embeddings", loadError); + } + // Retried regardless: the reason is usually a cold cache behind a flaky + // network, which succeeds on a later attempt, and the setting is cleared + // when it does. this.embeddingProvider.warm?.(); return this.keywordSearch(trimmed, limit, toolFilter, branchFilter); } diff --git a/tests/handoff/embedding-load-failure.test.ts b/tests/handoff/embedding-load-failure.test.ts new file mode 100644 index 00000000..8468f854 --- /dev/null +++ b/tests/handoff/embedding-load-failure.test.ts @@ -0,0 +1,108 @@ +/** + * A model that cannot load must not look like a model that is still loading. + * + * Hybrid search answers from keyword while the embedding model loads, on the + * reasoning that a cold cache takes minutes and nobody should wait for it + * holding a tool call open. That branch asked one question — `isReady()` — and + * a model still downloading and a model that will never download both answer + * false. + * + * So a failed load produced, on every call and forever: keyword results, the + * note "embedding model still loading, so this answer is keyword-only — ask + * again shortly for more", and an `xtctx status` with no semantic-unavailable + * line, because the only code that records the error is a catch this branch + * returns before reaching. Advice that can never come true, attached to a + * failure nothing reports. + */ +import { mkdtemp, rm } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { SqliteHandoffIndex } from "@xtctx/handoff/sqlite-index"; +import type { EmbeddingProvider } from "@xtctx/handoff/embeddings"; +import type { ConversationChunk, ConversationScraper, ScraperState } from "@xtctx/types/scraper"; + +let tempDir = ""; + +class OneChunkScraper implements ConversationScraper { + readonly tool = "codex"; + async detect(): Promise { + return true; + } + getStorePaths(): string[] { + return ["memory://codex"]; + } + async *scrape(): AsyncIterable { + yield* this.fullSync(); + } + async *fullSync(): AsyncIterable { + yield { + tool: "codex", + sessionId: "one", + timestamp: new Date("2026-09-21T00:00:00.000Z"), + role: "user", + content: "the cache eviction policy we settled on", + metadata: { messageIndex: 0, tokenEstimate: 1, layer: 0 }, + }; + } + async getLastScrapedPosition(): Promise { + return { lastTimestamp: new Date(0) }; + } + async saveScrapedPosition(): Promise {} +} + +/** A provider whose model never loads, the way an offline machine's does not. */ +class UnloadableProvider implements EmbeddingProvider { + readonly model = "test/never-loads"; + warmCalls = 0; + async embed(): Promise { + throw new Error("model is not loaded"); + } + async embedBatch(): Promise { + throw new Error("model is not loaded"); + } + isReady(): boolean { + return false; + } + loadError(): string | undefined { + return "getaddrinfo ENOTFOUND huggingface.co"; + } + warm(): void { + this.warmCalls += 1; + } +} + +beforeEach(async () => { + tempDir = await mkdtemp(join(tmpdir(), "xtctx-embed-fail-")); +}); + +afterEach(async () => { + await rm(tempDir, { recursive: true, force: true }); +}); + +describe("a local model that cannot load", () => { + it("is reported in status instead of read as still loading", async () => { + const provider = new UnloadableProvider(); + const index = new SqliteHandoffIndex( + join(tempDir, "fail.db"), + tempDir, + [{ tool: "codex", scraper: new OneChunkScraper() }], + { embeddingProvider: provider, refreshBudgetMs: 30_000 }, + ); + + try { + // Hybrid still answers — degrading to keyword is correct and is not what + // this is about. + const results = await index.searchSessions("cache eviction", 5, undefined, "hybrid"); + expect(results.length).toBeGreaterThan(0); + + const status = await index.getStatus(); + expect(status.embedding_error).toBe("getaddrinfo ENOTFOUND huggingface.co"); + + // Still retried, because the usual cause is transient. + expect(provider.warmCalls).toBeGreaterThan(0); + } finally { + await index.close(); + } + }); +}); From 6dfc8166829f132fb7d0dccd8800920d87007544 Mon Sep 17 00:00:00 2001 From: Felix Stubner Date: Mon, 21 Sep 2026 18:38:38 +0100 Subject: [PATCH 18/34] fix(mcp): reject an unknown tool_filter id instead of matching nothing MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `tool_filter` reaches SQLite as `WHERE tool IN (...)`, so an id naming no tool matched nothing and the answer was "No matching sessions found." — which an agent reports to its user as "there is no Claude Code history in this project". A wrong id and an empty index were indistinguishable. The ids are not guessable either. They are `claude-code` and `antigravity`, while the natural guesses from a tool's own name are `claude` and `gemini`, and the MCP schema advertised only `items: { type: "string" }` with the description "Optional tool ids to include". Both halves are fixed: the schema now enumerates the ids from the tool registry and lists them in the description, and `validatedFilter` rejects an unrecognised one with an error naming the valid set. An agent that gets that back can correct its own call; one that got an empty result could not tell a typo from an empty history. Found by a delegated review of the agent-facing surface, then confirmed against the schema and the query path rather than taken on trust. --- src/mcp/server.ts | 28 +++++++++++++++---- src/mcp/tools/sessions.ts | 22 +++++++++++++++ tests/mcp/tool-filter-ids.test.ts | 46 +++++++++++++++++++++++++++++++ 3 files changed, 90 insertions(+), 6 deletions(-) create mode 100644 tests/mcp/tool-filter-ids.test.ts diff --git a/src/mcp/server.ts b/src/mcp/server.ts index 97284bee..e90213b2 100644 --- a/src/mcp/server.ts +++ b/src/mcp/server.ts @@ -17,6 +17,10 @@ import { import { errorMessage, sanitizeErrorMessage } from "../utils/errors.js"; import { readXtctxPackage } from "../utils/package-info.js"; import { inlineSafe } from "../utils/untrusted-text.js"; +import { SUPPORTED_TOOLS } from "../tools/sources.js"; + +/** The tool ids `tool_filter` accepts, straight from the tool registry. */ +const TOOL_IDS = SUPPORTED_TOOLS.map((tool) => tool.id); const { version: SERVER_VERSION } = readXtctxPackage(import.meta.url); @@ -52,8 +56,12 @@ export function buildToolDefinitions(): Tool[] { limit: { type: "number", description: "Max sessions. Default: 5" }, tool_filter: { type: "array", - items: { type: "string" }, - description: "Optional tool ids to include", + // Enumerated, because the ids are not guessable: `claude-code` and + // `antigravity`, against the natural guesses `claude` and `gemini`. + // An id that matches nothing used to return "No matching sessions + // found.", which an agent reports as an empty history. + items: { type: "string", enum: TOOL_IDS }, + description: `Optional tool ids to include. One or more of: ${TOOL_IDS.join(", ")}`, }, branch_filter: { type: "array", @@ -98,8 +106,12 @@ export function buildToolDefinitions(): Tool[] { limit: { type: "number", description: "Max sessions. Default: 5" }, tool_filter: { type: "array", - items: { type: "string" }, - description: "Optional tool ids to include", + // Enumerated, because the ids are not guessable: `claude-code` and + // `antigravity`, against the natural guesses `claude` and `gemini`. + // An id that matches nothing used to return "No matching sessions + // found.", which an agent reports as an empty history. + items: { type: "string", enum: TOOL_IDS }, + description: `Optional tool ids to include. One or more of: ${TOOL_IDS.join(", ")}`, }, branch_filter: { type: "array", @@ -150,8 +162,12 @@ export function buildToolDefinitions(): Tool[] { }, tool_filter: { type: "array", - items: { type: "string" }, - description: "Optional tool ids to include", + // Enumerated, because the ids are not guessable: `claude-code` and + // `antigravity`, against the natural guesses `claude` and `gemini`. + // An id that matches nothing used to return "No matching sessions + // found.", which an agent reports as an empty history. + items: { type: "string", enum: TOOL_IDS }, + description: `Optional tool ids to include. One or more of: ${TOOL_IDS.join(", ")}`, }, branch_filter: { type: "array", diff --git a/src/mcp/tools/sessions.ts b/src/mcp/tools/sessions.ts index a007c3e4..6edb0212 100644 --- a/src/mcp/tools/sessions.ts +++ b/src/mcp/tools/sessions.ts @@ -1,5 +1,6 @@ import type { SessionSearchMode, SessionService } from "../../handoff/types.js"; import { inlineSafe } from "../../utils/untrusted-text.js"; +import { SUPPORTED_TOOLS } from "../../tools/sources.js"; interface RecentSessionsParams { limit?: number; @@ -60,6 +61,27 @@ export function validatedFilter(value: unknown, field: string): string[] | undef throw new ToolInputError(`${field} must contain only non-empty strings`); } + // An id that names no tool is rejected, not filtered on. + // + // The filter reaches SQLite as `WHERE tool IN (...)`, so an unrecognised id + // matches nothing and the answer is "No matching sessions found." — which an + // agent reports to the user as "you have no Claude Code history here". The + // ids are not guessable from the schema either: they are `claude-code` and + // `antigravity`, while the obvious guesses are `claude` and `gemini`. + // + // Naming the valid ids in the error is the point. An agent that gets this + // back can fix its own call; one that gets an empty result cannot tell a + // wrong id from an empty index. + const known = new Set(SUPPORTED_TOOLS.map((tool) => tool.id)); + const unknown = (value as string[]).filter((item) => !known.has(item.trim())); + if (unknown.length > 0) { + throw new ToolInputError( + `${field} contains unknown tool id${unknown.length === 1 ? "" : "s"} ` + + `${unknown.map((item) => JSON.stringify(item)).join(", ")}. ` + + `Valid ids: ${SUPPORTED_TOOLS.map((tool) => tool.id).join(", ")}`, + ); + } + return value as string[]; } diff --git a/tests/mcp/tool-filter-ids.test.ts b/tests/mcp/tool-filter-ids.test.ts new file mode 100644 index 00000000..6266086d --- /dev/null +++ b/tests/mcp/tool-filter-ids.test.ts @@ -0,0 +1,46 @@ +/** + * A tool id that names no tool is a mistake, not a filter. + * + * `tool_filter` reaches SQLite as `WHERE tool IN (...)`, so an unrecognised id + * matches nothing and the answer is "No matching sessions found." — which an + * agent reports to the user as "there is no Claude Code history in this + * project". A wrong id and an empty index were indistinguishable. + * + * The ids are also not guessable. They are `claude-code` and `antigravity`, + * while the obvious guesses from a tool's own name are `claude` and `gemini`, + * and the MCP schema advertised only `items: { type: "string" }`. + */ +import { describe, expect, it } from "vitest"; +import { validatedFilter } from "@xtctx/mcp/tools/sessions"; +import { SUPPORTED_TOOLS } from "@xtctx/tools/sources"; + +describe("validatedFilter", () => { + it("accepts every id the tool registry defines", () => { + const ids = SUPPORTED_TOOLS.map((tool) => tool.id); + + expect(validatedFilter(ids, "tool_filter")).toEqual(ids); + }); + + it("rejects the natural wrong guesses instead of matching nothing", () => { + for (const guess of ["claude", "gemini", "vscode"]) { + expect(() => validatedFilter([guess], "tool_filter")).toThrow(/unknown tool id/); + } + }); + + it("names the valid ids in the error, so the caller can fix its own call", () => { + let message = ""; + try { + validatedFilter(["claude"], "tool_filter"); + } catch (error) { + message = error instanceof Error ? error.message : String(error); + } + + expect(message).toContain("claude-code"); + expect(message).toContain("antigravity"); + }); + + it("still allows no filter at all", () => { + expect(validatedFilter(undefined, "tool_filter")).toBeUndefined(); + expect(validatedFilter(null, "tool_filter")).toBeUndefined(); + }); +}); From 7e063fb0ee0063b36823812ca53ebdeb7360ca04 Mon Sep 17 00:00:00 2001 From: Felix Stubner Date: Mon, 21 Sep 2026 18:41:26 +0100 Subject: [PATCH 19/34] docs(readme): say that disconnect --global-mcp is not symmetric with setup MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A review flagged that `xtctx setup` followed by `xtctx disconnect --all` leaves xtctx wired into Antigravity for every project on the machine, and proposed making disconnect remove it by default. Implementing that broke three tests, and reading why they exist is the point of this commit. They record a live incident: disconnecting ONE project used to empty the machine-global Antigravity and Copilot CLI configs, taking xtctx away from every other project on the machine. Those files hold no per-project entry, so there is nothing project-scoped to remove from them — only the whole wiring for every project at once. The asymmetry is deliberate and the proposed fix would have reintroduced a known regression. What the review was right about is the documentation. The README said to pass `--global-mcp` "(as with `setup`)", which is false: `setup` writes Antigravity's config WITHOUT the flag, because Antigravity has no project-scoped MCP file. A reader who believed that sentence would assume `disconnect --all` had undone what `setup` did. So the flag is now documented as explicitly not symmetric, with the reason, and the README gains the full-removal sequence it never had — including that `.xtctx` survives on purpose and holds the indexed transcripts. --- README.md | 16 +++++++++++++++- 1 file changed, 15 insertions(+), 1 deletion(-) diff --git a/README.md b/README.md index 502834ba..6dc8d44f 100644 --- a/README.md +++ b/README.md @@ -144,7 +144,21 @@ supported tool — that one also deletes `.xtctx/skills`, since with nothing lef managing skills the synced source is xtctx's own scaffolding. A skill you wrote yourself and selected at setup is kept where you wrote it. Antigravity and Copilot CLI keep one MCP config for every project on the machine, so a project disconnect leaves those two files alone; -pass `--global-mcp` (as with `setup`) to remove xtctx from them as well. +pass `--global-mcp` to remove xtctx from them as well. + +That flag is **not** symmetric with `setup`, and the difference is worth +knowing before you assume `disconnect --all` has removed everything. `setup` +writes Antigravity's config without the flag, because Antigravity has no +project-scoped MCP file and there is nowhere else to put it; `setup +--global-mcp` additionally writes Copilot CLI's. Neither file holds a +per-project entry, so a project disconnect cannot remove "this project's" +wiring from them — it can only remove xtctx from that client for every project +at once. Doing that silently is exactly what it used to do, and it took xtctx +away from every other project on the machine, so it is an explicit step now. + +To remove xtctx from a machine entirely: `xtctx disconnect --all --global-mcp` +in each project you set up, then delete each project's `.xtctx` directory, +which holds the config and the indexed transcripts and is deliberately kept. `xtctx scan` reads every enabled transcript store into the project's index and exits. The MCP server does the same thing on its own every time it starts, so From a4bbada186e61a30cc1950a58b39404f41c38c37 Mon Sep 17 00:00:00 2001 From: Felix Stubner Date: Mon, 21 Sep 2026 18:44:07 +0100 Subject: [PATCH 20/34] fix(claims): the local-only promise, the thresholds, and a false statement of mine MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A delegated review of prose-against-code found that most of what is now wrong in the docs was made wrong by this session's own changes. The serious one is the privacy claim. README and PRODUCT.md said xtctx "never sends transcript content anywhere" and "nothing is ever sent off the machine". Those stopped being true when the opt-in OpenAI-compatible embedding endpoint shipped a few hours ago, and a claim about where a user's transcripts go is not one to leave stale. The review also caught that the qualification `docs/embedding-providers.md` specifies — "`xtctx status` reports the endpoint whenever one is configured" — is itself false, because no such line exists. Writing the qualification without the line would have replaced one false claim with another, so the line is added first: `xtctx status` now prints the endpoint in full whenever the vector identity is a remote one. "Am I uploading my transcripts, and where to" must never require opening a config file. Also corrected: ARCHITECTURE.md gave the semantic floors as 0.15/0.36 and PRODUCT.md named MiniLM, both superseded today. The thresholds now say they belong to the model and are swept per model, because that is the part that will go stale again otherwise. And a false statement of my own, from two commits ago. I wrote that no code had ever launched a detached scan from the session-start hook. It had: `launchDetachedScan` was added in #322 and removed by #323 the same day, 2026-09-02. The comment now says what is actually true — nothing launches one now, and the stale comments outlived the code by nineteen days — and records the correction rather than quietly rewriting it. --- ARCHITECTURE.md | 4 +++- PRODUCT.md | 24 +++++++++++++++--------- README.md | 6 +++++- src/cli/index.ts | 11 ++++++++--- src/cli/status.ts | 10 ++++++++++ 5 files changed, 41 insertions(+), 14 deletions(-) diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index e6057df6..6e021829 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -72,7 +72,9 @@ twice: into FTS5 for keyword search, and as one embedding vector. comes from bm25 ordering but is rescored as a linear decay, because bm25 favours short documents and a one-line mention was outranking the paragraph that decided something. Semantic matches are gated twice: a per-window floor -(0.15) and a per-query confidence floor (0.36). When nothing clears the second +(0.62) and a per-query confidence floor (0.64). Those numbers belong to the +model — they are swept per model, not carried between them, and MiniLM's were +0.15 and 0.36. When nothing clears the second one, semantic results are dropped wholesale and only keyword hits remain — whether a query found anything is a property of the query, not of each window, and no answer beats a confident wrong one. diff --git a/PRODUCT.md b/PRODUCT.md index 370436f4..3fccecef 100644 --- a/PRODUCT.md +++ b/PRODUCT.md @@ -9,9 +9,11 @@ transcript store into a per-project SQLite index and serves it back — to any of those tools — through a small read-only MCP server, so the next agent can pick up where the last one left off. -Raw local transcripts are authoritative. xtctx never summarizes, never -persists derived "memory", and never sends transcript content anywhere; the -index is derived data that can always be deleted and rebuilt. +Raw local transcripts are authoritative. xtctx never summarizes and never +persists derived "memory"; the index is derived data that can always be +deleted and rebuilt. It sends transcript content nowhere unless a project +opts in to an external embedding endpoint, which is written into +`.xtctx/config.yaml` by hand and reported by `xtctx status`. ## Users @@ -50,7 +52,7 @@ Single-user, single-machine. There is no team, sync, or server component. - Scrapers for the seven supported tools, project-scoped, incremental, and tolerant of upstream schema drift (warn, never silently drop). - One per-project SQLite index (`.xtctx/state/xtctx.db`) with keyword (FTS5) - and semantic (local MiniLM embeddings) search over chronological windows. + and semantic (local bge-small embeddings) search over chronological windows. - Five read-only MCP tools: recent sessions, session detail, search, continuity status, handoff manifest. - CLI: `setup` (wire MCP config, managed instruction blocks, skills, and the @@ -73,8 +75,12 @@ no durable memory, no write-back tools, no cloud anything. atomic, merge-preserving, and never clobber unparsable user content. - Transcript content handed to a model is untrusted data; the MCP layer fences it and never grows write capabilities. -- Everything runs local, and nothing is ever sent off the machine. Two - network dependencies exist, both narrow: the one-time embedding-model - download from Hugging Face, and loopback-only HTTPS calls to Antigravity's - local language server (127.0.0.1, exact-PID + CSRF matched; certificate - verification is off because the server is self-signed). +- Everything runs local by default. Three network dependencies exist. Two are + unavoidable and narrow: the one-time embedding-model download from Hugging + Face, and loopback-only HTTPS calls to Antigravity's local language server + (127.0.0.1, exact-PID + CSRF matched; certificate verification is off + because the server is self-signed). The third is opt-in and is the only one + that carries transcript text: an OpenAI-compatible embedding endpoint named + in `.xtctx/config.yaml`. It is never inferred from the environment, the API + key is never stored in that file, and `xtctx status` prints the endpoint + whenever one is set. diff --git a/README.md b/README.md index 6dc8d44f..eabec189 100644 --- a/README.md +++ b/README.md @@ -229,7 +229,11 @@ startup hooks; others receive MCP config plus managed instructions only. ## Limits -- xtctx is local-only. It does not upload transcripts or run telemetry. +- xtctx is local-only by default: it never uploads transcripts and runs no + telemetry. A project can opt into an external embedding endpoint by writing + one into `.xtctx/config.yaml`, in which case window text is sent there to be + vectorized — never inferred from an environment variable, and `xtctx status` + names the endpoint in full whenever one is configured. - Transcript formats belong to each upstream tool and can drift. The drift tests and format fingerprints exist to catch parser breakage, but `xtctx status` is still the source of truth for your machine. diff --git a/src/cli/index.ts b/src/cli/index.ts index ec5c5aed..ec9e7e50 100644 --- a/src/cli/index.ts +++ b/src/cli/index.ts @@ -250,9 +250,14 @@ async function warmIndex(sessions: SessionService): Promise { * to know. * * Three comments in this repository claimed the session-start hook launched a - * detached scan that did this. No such code has ever existed: the hook does a - * deliberate no-scan read, and nothing in `src` spawned a process except the - * Antigravity client. + * detached scan that did this, and the hook has not done so since #323 on + * 2026-09-02, which moved the warm-up here — to the server — on the same day + * #322 added it. The comments outlived the code they described by nineteen + * days. + * + * (An earlier version of this comment said no such code had ever existed. + * That was wrong: `launchDetachedScan` was real, in `src/cli/hook.ts`, for a + * few hours. What it was right about is that nothing launches one now.) * * Bounded rather than unconditional, because "embed everything in the * background" is exactly what would make a large project on a CPU unusable. diff --git a/src/cli/status.ts b/src/cli/status.ts index fa279aa1..5dcfae1f 100644 --- a/src/cli/status.ts +++ b/src/cli/status.ts @@ -137,6 +137,16 @@ export async function renderStatusBlock( if (status.vector_device && status.vector_device !== "cpu") { lines.push(`Device ${status.vector_device} (from \`xtctx calibrate\`)`); } + // An endpoint is the one thing that sends transcript text off this machine, + // so it is stated in full and unconditionally whenever one is configured. + // "Am I uploading my transcripts, and where to" must never require opening a + // config file to answer. The identity already carries the endpoint because + // `retrieval_unit_vectors` is keyed on it; the URL is what matters here, and + // the API key is never part of it. + if (status.vector_model.startsWith("openai:")) { + const endpoint = status.vector_model.slice("openai:".length); + lines.push(`Embedding external endpoint — window text is sent to ${endpoint}`); + } lines.push(""); lines.push("Tools:"); From 38535850a55ab3970569f676205e2eb292137e53 Mon Sep 17 00:00:00 2001 From: Felix Stubner Date: Mon, 21 Sep 2026 19:10:52 +0100 Subject: [PATCH 21/34] fix(ux): tell people what happens next, and stop advice that cannot apply MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Four findings from the UX pass, all confirmed by running the built CLI in a sandboxed first-run rather than by reading the source. `setup` ended on the last of eighteen file paths. The two things a user needs next are invisible from there: MCP clients read their config at launch, so an agent that was already open has no xtctx tools and reads as a broken install; and nothing is indexed until an agent calls a tool, so `xtctx status` — the obvious way to check setup worked — reports `Scan never` and `0 sessions`, which is the shape of a failure. Both are now stated, with the restart first. `setup` also wires all seven supported tools whether or not they are installed: measured in a clean environment with zero detected, eighteen files and eleven new top-level entries including `GEMINI.md` and `opencode.json`. The behaviour stays — detection reads a transcript store that does not exist until a tool has been used, so wiring only what is detected would silently skip whatever the user installs tomorrow, and a silently unwired tool is worse than an unwanted file because nothing reports it — but the reason is now printed, with `xtctx disconnect ` for anything unwanted. `status` had no branch for an unreadable config. It fell through to "Ask a configured agent to call xtctx_recent_sessions", which cannot work because nothing is scanned at all while the config will not parse — and with an index left from before the file broke it reported "Handoff is wired" six lines under "UNREADABLE ... No transcripts are being read until this is fixed." The last line is the one people act on, so it now names the file to fix. And literal-search advice stopped following unrelated calls. `literalSearchStoppedEarly` and `literalUnreadableTools` were set by a literal pass and never cleared, while `getIndexProgress` — which every tool calls — reports them, so one truncated literal search attached "Narrow the query or raise `limit`" to every later `xtctx_recent_sessions` and `xtctx_session_detail` answer. Those calls carry no query. An agent either follows advice that cannot apply or learns to ignore the notes, which costs the ones that matter. Both behaviour fixes have tests confirmed to fail against the old behaviour. --- src/cli/status.ts | 20 +++- src/config/setup.ts | 68 ++++++++++++++ src/handoff/sqlite-index.ts | 23 +++++ tests/cli/status.test.ts | 29 +++++- tests/handoff/literal-advice-scope.test.ts | 103 +++++++++++++++++++++ 5 files changed, 241 insertions(+), 2 deletions(-) create mode 100644 tests/handoff/literal-advice-scope.test.ts diff --git a/src/cli/status.ts b/src/cli/status.ts index 5dcfae1f..5da9d6f4 100644 --- a/src/cli/status.ts +++ b/src/cli/status.ts @@ -285,10 +285,28 @@ export async function renderStatusBlock( lines.push(""); if (!configPresent) { lines.push("Next This project is not set up yet. Run: xtctx setup"); + } else if (services.config.error) { + // Checked before everything below it, because nothing below it can be + // true while this holds. Nothing is scanned at all with an unreadable + // config, so "ask an agent to call xtctx_recent_sessions" is advice that + // cannot work, and with an index left over from before the file broke the + // old branch cheerfully reported "Handoff is wired" six lines under + // "UNREADABLE ... No transcripts are being read until this is fixed." + // The last line is the one people act on. + lines.push(`Next Fix ${services.configPath} — nothing is being read until it parses.`); } else if (needsRepair) { lines.push("Next Wiring has drifted. Run: xtctx setup --repair"); } else if (status.sessions === 0) { - lines.push("Next No sessions are indexed yet. Ask a configured agent to call xtctx_recent_sessions."); + // Worded as expected rather than as a fault. Running `status` straight + // after `setup` is the obvious way to check setup worked, and it lands + // here: nothing is indexed until an agent calls a tool, so `Scan never` + // and `0 sessions` are what a correct install looks like at this point. + lines.push( + "Next Nothing is indexed yet, which is expected until an agent calls a tool.", + ); + lines.push( + " Restart any agent that was open when setup ran, then ask it for recent context.", + ); } else { lines.push("Next Handoff is wired. Ask a configured agent to call xtctx_recent_sessions."); } diff --git a/src/config/setup.ts b/src/config/setup.ts index 2e0a4147..1cff05b1 100644 --- a/src/config/setup.ts +++ b/src/config/setup.ts @@ -295,4 +295,72 @@ function printSetupResult(result: SetupResult): void { for (const failure of result.failures) { process.stdout.write(` error ${failure}\n`); } + + if (result.failures.length === 0) { + printCoverageNote(); + printNextSteps(); + } +} + +/** + * Why a project that uses one agent just gained config for seven. + * + * Setup wires every supported tool regardless of what is installed — measured + * in a clean environment with zero tools detected, that is eighteen files and + * eleven new top-level entries in the repository, including `GEMINI.md` and + * `opencode.json` for tools the user may never have heard of. The plan is + * shown and confirmed before any of it is written, so nothing is sneaked in, + * but the *reason* was nowhere and a first-time user reads it as the tool + * making a mess. + * + * The behaviour is deliberate and stays. Detection reads a tool's transcript + * store, which does not exist until that tool has been used, so wiring only + * what is detected would silently skip a tool the user installs tomorrow — and + * a silently unwired tool is a worse failure than a file they did not want, + * because nothing reports it. The instruction files are also read by whoever + * opens the repository next, which includes a teammate on a different agent. + * + * So it is said out loud instead, with the command that undoes any of it. + */ +function printCoverageNote(): void { + process.stdout.write( + `\n All ${SUPPORTED_TOOLS.length} supported tools were wired, including any not installed here:\n` + + " a tool's config only exists once it has been used, so wiring what is\n" + + " detected today would skip whatever you install tomorrow. The instruction\n" + + " files are also read by whichever agent opens this repo next.\n" + + " Remove any you do not want with `xtctx disconnect `.\n", + ); +} + +/** + * What to do now that setup has written eighteen files. + * + * Setup used to end on the last path it wrote, and the two things a user + * needs next are both invisible from that. + * + * The first is the restart. MCP clients read their config when they launch, so + * an agent that was already open when setup ran has no xtctx tools — and the + * natural next move after running setup is to go back to the agent already + * open and ask it something. It answers that it cannot see any xtctx tools, + * which reads as a broken install rather than a stale process. + * + * The second is that there is nothing to see yet. Nothing is indexed until an + * agent calls a tool, so `xtctx status` immediately after setup reports + * `Scan never` and `0 sessions` — the shape of a failure, and the natural + * thing to run next to check whether setup worked. + */ +function printNextSteps(): void { + process.stdout.write( + [ + "", + "Next:", + " 1. Restart any agent that was already open — MCP clients read their", + " config at launch, so a running one cannot see xtctx yet.", + " 2. Ask it for recent context, or have it call `xtctx_recent_sessions`.", + "", + " Nothing is indexed until then, so `xtctx status` will report", + " `Scan never` and `0 sessions` until an agent has called a tool once.", + "", + ].join("\n"), + ); } diff --git a/src/handoff/sqlite-index.ts b/src/handoff/sqlite-index.ts index 6bee4640..c414204a 100644 --- a/src/handoff/sqlite-index.ts +++ b/src/handoff/sqlite-index.ts @@ -279,6 +279,26 @@ export class SqliteHandoffIndex implements SessionService { * store is not fixed by retrying at all. */ private lastLiteralUnreadable: string[] = []; + + /** + * Forget the last literal pass's advice, at the start of every retrieval. + * + * Both fields above were written by a literal search and never cleared, + * while `getIndexProgress` — which every tool calls — reports them. So one + * truncated literal search attached "The literal pass stopped at its limit + * or time budget. Narrow the query or raise `limit`." to every later + * `xtctx_recent_sessions` and `xtctx_session_detail` answer, calls that + * carry no query to narrow. + * + * The advice belongs to the call that produced it, not to the index. Cleared + * on entry rather than consumed on read, because a literal search sets them + * after this runs and before its own progress note is built, and nothing + * then depends on how many times that note is asked for. + */ + private clearLiteralAdvice(): void { + this.lastLiteralWasExhaustive = undefined; + this.lastLiteralUnreadable = []; + } private readonly embeddingWarmBudgetMs: number; private readonly vectorBudgetMs: number; private scanStartedMs = 0; @@ -347,6 +367,7 @@ export class SqliteHandoffIndex implements SessionService { toolFilter?: string[], branchFilter?: string[], ): Promise { + this.clearLiteralAdvice(); await this.refresh({ toolFilter }); const db = this.getDb(); const normalizedLimit = normalizeLimit(limit, DEFAULT_LIMIT); @@ -426,6 +447,7 @@ export class SqliteHandoffIndex implements SessionService { offset: number, limit: number, ): Promise { + this.clearLiteralAdvice(); await this.refresh({ sessionRef }); const db = this.getDb(); const normalizedOffset = Number.isFinite(offset) && offset > 0 ? Math.floor(offset) : 0; @@ -459,6 +481,7 @@ export class SqliteHandoffIndex implements SessionService { mode: SessionSearchMode = "hybrid", branchFilter?: string[], ): Promise { + this.clearLiteralAdvice(); const normalizedModeForRefresh = normalizeSearchMode(mode); // A literal pass reads the stores, not the index, so it starts the scan // and moves on rather than waiting out the refresh budget in front of its diff --git a/tests/cli/status.test.ts b/tests/cli/status.test.ts index 3fd66d0a..dfa16b0b 100644 --- a/tests/cli/status.test.ts +++ b/tests/cli/status.test.ts @@ -28,6 +28,28 @@ describe("status", () => { await rm(homeDir, { recursive: true, force: true }); }); + it("points at the broken config instead of at an agent that cannot help", async () => { + // With an unreadable config nothing is scanned at all, so "ask a + // configured agent to call xtctx_recent_sessions" is advice that cannot + // work — and with an index left from before the file broke, the old + // branch reported "Handoff is wired" six lines under "UNREADABLE ... No + // transcripts are being read until this is fixed." The last line is the + // one people act on, so it has to be the true one. + await setupProject({ projectPath: projectRoot, homeDir, yes: true }); + await writeFile(join(projectRoot, ".xtctx", "config.yaml"), "tools: [oops\n", "utf-8"); + + const services = await createProjectServices(projectRoot); + try { + const status = await renderStatusBlock(services, { homeDir }); + + expect(status).toContain("UNREADABLE"); + expect(status).toMatch(/Next\s+Fix .*config\.yaml/); + expect(status).not.toContain("Ask a configured agent"); + } finally { + await services.sessions.close(); + } + }); + it("says when a backlog is too large for anything to drain it on its own", async () => { // The MCP server drains the vector backlog at session start only while the // estimate fits its budget. Above that nothing is working on it, and @@ -79,7 +101,12 @@ describe("status", () => { const status = await renderStatusBlock(services, { homeDir }); expect(status).not.toContain("Wiring has drifted"); - expect(status).toContain("Ask a configured agent to call xtctx_recent_sessions"); + // A freshly wired project has nothing indexed, and the "Next" line says + // so as an expectation rather than a fault: running `status` straight + // after `setup` is the obvious way to check setup worked, and reading + // `Scan never` / `0 sessions` as a failure is what it used to invite. + expect(status).toContain("expected until an agent calls a tool"); + expect(status).toContain("Restart any agent that was open when setup ran"); } finally { await services.sessions.close(); } diff --git a/tests/handoff/literal-advice-scope.test.ts b/tests/handoff/literal-advice-scope.test.ts new file mode 100644 index 00000000..6fdd9e69 --- /dev/null +++ b/tests/handoff/literal-advice-scope.test.ts @@ -0,0 +1,103 @@ +/** + * Advice from a literal search belongs to that search, not to the index. + * + * `literalSearchStoppedEarly` and `literalUnreadableTools` were written by a + * literal pass and never cleared, while `getIndexProgress` — which every tool + * calls to build its progress note — reports them. So one truncated literal + * search attached "The literal pass stopped at its limit or time budget. + * Narrow the query or raise `limit`." to every later `xtctx_recent_sessions` + * and `xtctx_session_detail` answer for the life of the process. + * + * Those calls carry no query to narrow. An agent either follows advice that + * cannot apply, or learns to ignore these notes — which costs the ones that do + * matter, like a tool whose store could not be read. + */ +import { mkdtemp, rm } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { SqliteHandoffIndex } from "@xtctx/handoff/sqlite-index"; +import { NullEmbeddingProvider } from "@xtctx/handoff/null-embeddings"; +import type { ConversationChunk, ConversationScraper, ScraperState } from "@xtctx/types/scraper"; + +let tempDir = ""; + +/** Enough messages that a one-result literal budget has to stop early. */ +class ChattyScraper implements ConversationScraper { + readonly tool = "codex"; + async detect(): Promise { + return true; + } + getStorePaths(): string[] { + return ["memory://codex"]; + } + async *scrape(): AsyncIterable { + yield* this.fullSync(); + } + async *fullSync(): AsyncIterable { + for (let index = 0; index < 12; index += 1) { + yield { + tool: "codex", + sessionId: `session-${index}`, + timestamp: new Date(Date.UTC(2026, 8, 21, 0, index)), + role: "user", + content: `we changed the retry budget again, note ${index}`, + metadata: { messageIndex: 0, tokenEstimate: 1, layer: 0 }, + }; + } + } + async getLastScrapedPosition(): Promise { + return { lastTimestamp: new Date(0) }; + } + async saveScrapedPosition(): Promise {} +} + +beforeEach(async () => { + tempDir = await mkdtemp(join(tmpdir(), "xtctx-literal-advice-")); +}); + +afterEach(async () => { + await rm(tempDir, { recursive: true, force: true }); +}); + +function makeIndex(): SqliteHandoffIndex { + return new SqliteHandoffIndex( + join(tempDir, "advice.db"), + tempDir, + [{ tool: "codex", scraper: new ChattyScraper() }], + { embeddingProvider: new NullEmbeddingProvider(), refreshBudgetMs: 30_000 }, + ); +} + +describe("literal search advice", () => { + it("does not follow a later call that has no query to narrow", async () => { + const index = makeIndex(); + try { + // A literal pass capped low enough that it cannot finish. + await index.searchSessions("retry budget", 1, undefined, "literal"); + const afterLiteral = index.getIndexProgress?.(); + expect(afterLiteral?.literalSearchStoppedEarly).toBe(true); + + // A different question entirely. `recent_sessions` takes no query. + await index.listRecentSessions(5); + + expect(index.getIndexProgress?.().literalSearchStoppedEarly).toBeUndefined(); + } finally { + await index.close(); + } + }); + + it("is also dropped before a detail call", async () => { + const index = makeIndex(); + try { + const sessions = await index.searchSessions("retry budget", 1, undefined, "literal"); + expect(index.getIndexProgress?.().literalSearchStoppedEarly).toBe(true); + + await index.getSessionDetail(sessions[0]?.session_ref ?? "codex:session-0", 0, 5); + + expect(index.getIndexProgress?.().literalSearchStoppedEarly).toBeUndefined(); + } finally { + await index.close(); + } + }); +}); From ab22dc26d6b6bfce8300011c76db08a48de3886d Mon Sep 17 00:00:00 2001 From: Felix Stubner Date: Mon, 21 Sep 2026 19:15:19 +0100 Subject: [PATCH 22/34] fix(ux): a broken config reaches agents, and an unset-up project stops shouting MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two more from the UX pass. `.xtctx/config.yaml` records which transcript stores the user allowed to be read, so a file that will not parse yields zero scrapers rather than falling back to defaults. That is the right call and it had a consequence nobody had looked at: every MCP tool then answered "No matching sessions found.", and an agent reads that as "this project has no cross-tool history" and tells the user so. The real answer is that xtctx is reading nothing at all until one file is fixed. `xtctx status` has printed UNREADABLE for a while; agents never run the CLI, and MCP is the only surface they see. Every tool now answers with the file, the parse error, and the fact that this is an unread history rather than an absent one — returned rather than thrown, because it is a state the user can fix and an agent can pass that on. It also tells the agent not to edit the file unprompted, since rewriting it would mean widening its own read access. Kept separate from the not-configured notice on purpose: "nobody opted this directory in" and "somebody did and the file is broken" are opposite situations that had become the same empty answer. And `xtctx status` in a project that has not been set up stops printing the Skills and Managed-files sections. Every line of them reads `missing` there — about thirty, each with an absolute path — between the reader and the one sentence that matters. Someone running `status` to see what this thing does met a wall of faults describing the absence of a thing they had not asked for yet. Eighteen lines now, ending in `Run: xtctx setup`. Both covered by tests confirmed to fail against the old behaviour. --- src/cli/index.ts | 13 +++++- src/cli/status.ts | 16 +++++-- src/mcp/server.ts | 48 +++++++++++++++++++++ tests/mcp/config-error-notice.test.ts | 61 +++++++++++++++++++++++++++ 4 files changed, 134 insertions(+), 4 deletions(-) create mode 100644 tests/mcp/config-error-notice.test.ts diff --git a/src/cli/index.ts b/src/cli/index.ts index ec9e7e50..754f48b2 100644 --- a/src/cli/index.ts +++ b/src/cli/index.ts @@ -65,8 +65,19 @@ export async function main(argv = process.argv): Promise { // A tool call still in flight when stdin closes may go unanswered — the // grace window above is enough for ordinary calls, not for one waiting on // a scan. The client has closed its side by then, so nothing is listening. + // A config that exists but will not parse is not an empty project, and + // used to reach an agent as one: zero scrapers, "No matching sessions + // found.", and the agent telling the user there is no cross-tool history + // here. The CLI has said `UNREADABLE` for a while; agents never read it. + const configError = services.config.error + ? { + projectRoot: services.projectRoot, + configPath: services.configPath, + message: services.config.error, + } + : undefined; await startMcpServer( - { sessions: services.sessions, unconfiguredProjectRoot }, + { sessions: services.sessions, unconfiguredProjectRoot, configError }, () => shutdown(true), ); diff --git a/src/cli/status.ts b/src/cli/status.ts index 5da9d6f4..dbe0c199 100644 --- a/src/cli/status.ts +++ b/src/cli/status.ts @@ -202,6 +202,18 @@ export async function renderStatusBlock( } } + // Everything below reports on wiring that setup creates, so in a project + // that has not been set up every line of it reads `missing` — about thirty + // of them, each carrying an absolute path, between the reader and the one + // sentence that matters. A first-time user running `status` to see what + // this thing does met a wall of faults describing the absence of a thing + // they had not asked for yet. + if (!configPresent) { + lines.push(""); + lines.push("Next This project is not set up yet. Run: xtctx setup"); + return lines.join("\n"); + } + lines.push(""); lines.push("Skills:"); lines.push(` Source ${skills.sourceDir}`); @@ -283,9 +295,7 @@ export async function renderStatusBlock( skills.targets.some((target) => target.state === "missing" || target.state === "drift")); lines.push(""); - if (!configPresent) { - lines.push("Next This project is not set up yet. Run: xtctx setup"); - } else if (services.config.error) { + if (services.config.error) { // Checked before everything below it, because nothing below it can be // true while this holds. Nothing is scanned at all with an unreadable // config, so "ask an agent to call xtctx_recent_sessions" is advice that diff --git a/src/mcp/server.ts b/src/mcp/server.ts index e90213b2..8597cdcd 100644 --- a/src/mcp/server.ts +++ b/src/mcp/server.ts @@ -41,6 +41,18 @@ interface McpToolDependencies { * setup to the person. */ unconfiguredProjectRoot?: string; + /** + * Why `.xtctx/config.yaml` could not be read, if it could not. + * + * Separate from `unconfiguredProjectRoot`, because the two mean opposite + * things to a user: nobody opted this directory in, against somebody did and + * the file is broken. Both used to reach an agent as the same thing — an + * ordinary empty answer — because a config that will not parse yields zero + * scrapers, so every tool returned "No matching sessions found." and the + * agent told the user there was no cross-tool history. The CLI says + * `UNREADABLE`, but agents never read the CLI. + */ + configError?: { projectRoot: string; configPath: string; message: string }; } /** @internal Exported for tests only. */ @@ -205,6 +217,14 @@ export function createToolHandlers( return handlers; } + if (dependencies.configError) { + const notice = configUnreadable(dependencies.configError); + for (const name of TOOL_NAMES) { + handlers.set(name, notice); + } + return handlers; + } + if (dependencies.sessions) { handlers.set("xtctx_recent_sessions", createRecentSessionsHandler(dependencies.sessions)); handlers.set("xtctx_session_detail", createSessionDetailHandler(dependencies.sessions)); @@ -326,6 +346,34 @@ function notConfigured(projectRoot: string): ToolHandler { ].join("\n"); } +/** + * Every tool answers with the broken file, rather than with nothing found. + * + * Returned, not thrown, for the same reason `notConfigured` is: this is a + * state the user can fix, and an agent that receives it can say so. Throwing + * would make it a tool malfunction, which is a different and less useful + * message to pass on. + */ +function configUnreadable(details: { + projectRoot: string; + configPath: string; + message: string; +}): ToolHandler { + return async () => + [ + `xtctx cannot read this project's configuration: ${inlineSafe(details.configPath)}`, + "", + ` ${inlineSafe(details.message)}`, + "", + "No transcript stores are being read until that file parses, so this is", + "not an empty history — it is an unread one. Nothing has been changed.", + "", + "Tell the user to fix or delete that file. `npx -y xtctx status` prints", + "the same error. Do not edit it unprompted: it records which transcript", + "stores they allowed to be read.", + ].join("\n"); +} + function missingDependency(dependency: string): ToolHandler { return async () => { // Thrown (not returned) so the client sees isError, not a success shape. diff --git a/tests/mcp/config-error-notice.test.ts b/tests/mcp/config-error-notice.test.ts new file mode 100644 index 00000000..44b4a067 --- /dev/null +++ b/tests/mcp/config-error-notice.test.ts @@ -0,0 +1,61 @@ +/** + * A config that will not parse is not an empty project. + * + * `.xtctx/config.yaml` records which transcript stores the user allowed to be + * read, so a file that cannot be parsed yields zero scrapers rather than + * defaults — which is the right call, and it meant every MCP tool answered + * "No matching sessions found." An agent reads that as "this project has no + * cross-tool history" and tells the user so, while the real answer is "xtctx + * is reading nothing at all until you fix one file". + * + * `xtctx status` has printed `UNREADABLE` for a while. Agents never run the + * CLI, and the MCP surface is the only one they see. + */ +import { describe, expect, it } from "vitest"; +import { createToolHandlers } from "@xtctx/mcp/server"; + +const DETAILS = { + projectRoot: "/repo", + configPath: "/repo/.xtctx/config.yaml", + message: "Flow sequence must end with a ] at line 2, column 1", +}; + +describe("an unreadable config over MCP", () => { + it("answers every tool with the broken file rather than with nothing found", async () => { + const handlers = createToolHandlers({ configError: DETAILS }); + + expect(handlers.size).toBeGreaterThan(0); + for (const [name, handler] of handlers) { + const answer = String(await handler({})); + + expect(answer, name).toContain(".xtctx/config.yaml"); + expect(answer, name).toContain("Flow sequence must end with a ]"); + expect(answer, name).not.toContain("No matching sessions found"); + } + }); + + it("says the history is unread rather than absent", async () => { + const [, handler] = [...createToolHandlers({ configError: DETAILS })][0]; + + const answer = String(await handler({})); + + expect(answer).toMatch(/not an empty history/); + expect(answer).toContain("Nothing has been changed."); + }); + + it("tells the agent not to edit the file unprompted", async () => { + // It records which transcript stores the user allowed to be read, so an + // agent "helpfully" rewriting it would be widening their own access. + const [, handler] = [...createToolHandlers({ configError: DETAILS })][0]; + + expect(String(await handler({}))).toMatch(/Do not edit it unprompted/); + }); + + it("leaves the not-configured case alone, which means something different", async () => { + // Nobody opted this directory in, against somebody did and the file broke. + const handlers = createToolHandlers({ unconfiguredProjectRoot: "/repo" }); + const [, handler] = [...handlers][0]; + + expect(String(await handler({}))).toContain("not configured for xtctx"); + }); +}); From 92ad5e1c1187a4398276cadcc015a1a3ea9d1b0a Mon Sep 17 00:00:00 2001 From: Felix Stubner Date: Mon, 21 Sep 2026 19:21:08 +0100 Subject: [PATCH 23/34] docs: catch the prose up with a day that changed four things underneath it MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A delegated review checked every factual claim in the docs against the code and found most of what is wrong was made wrong by this session. docs/embedding-performance.md now says up front that it is written in two voices: everything through "Model bake-off" was written on 2026-09-20 when none of it had changed the code and every conclusion was "blocked on", and three of those shipped the next day. Sections are annotated where they are now false rather than rewritten, because how a wrong conclusion was reached is the part worth keeping — but the file said the committed baseline was MiniLM's 0.333/0.533/0.183 when that file now holds bge-small's 0.417/0.583/0.267, said the bge rows could not be reproduced when a harness for them exists, said the `device` option "is simply not passed today" when it is, and still listed bge-small as blocked on a confirmation while it was the default. It also recommended 0.55/0.65 where 0.62/0.64 shipped, and now says why. docs/embedding-providers.md opened with "Nothing here is built yet" for a design that was built hours later. It now records what was built, the two deliberate deviations (the local identity keeps its bare HuggingFace id; remote vectors are normalized on receipt) and the two parts deliberately left out (the unswept-threshold warning, and any sweep against a remote provider), and says the code is current wherever the two disagree. "Semantic search is lazy" and "vector creation is lazy" described behaviour that changed when the server started draining the backlog in the background. ARCHITECTURE.md and PRODUCT.md were missing `scan` and `calibrate` from their CLI lists entirely, and the README documented neither `--embed` nor `calibrate` — a command that makes indexing roughly six times faster was discoverable only from a performance note. Also checked and NOT changed: `xtctx_session_detail` returning nothing for a session literal search found. The empty answer already carries the progress note naming the tools not yet read, so an agent is told to ask again rather than being left with a bare miss. --- ARCHITECTURE.md | 6 +-- PRODUCT.md | 5 ++- README.md | 22 ++++++++-- docs/architecture.md | 7 ++- docs/embedding-performance.md | 80 +++++++++++++++++++++++------------ docs/embedding-providers.md | 55 ++++++++++++++++-------- 6 files changed, 121 insertions(+), 54 deletions(-) diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index 6e021829..f15d49ef 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -7,8 +7,8 @@ boundaries between them, and what each part is allowed to trust. ## Parts -- **CLI** (`src/cli/`) — `setup`, `status`, `disconnect`, and the internal - `--hook session-start` entry point. Bare `xtctx` on a non-TTY stdio pair +- **CLI** (`src/cli/`) — `setup`, `status`, `scan`, `calibrate`, + `disconnect`, and the internal `--hook session-start` entry point. Bare `xtctx` on a non-TTY stdio pair starts the MCP server. - **MCP server** (`src/mcp/`) — stdio JSON-RPC server exposing exactly five read-only tools. Spawned by coding agents via `npx -y xtctx`. @@ -19,7 +19,7 @@ boundaries between them, and what each part is allowed to trust. - **Handoff index** (`src/handoff/`) — per-project SQLite database (`.xtctx/state/xtctx.db`, WAL, schema-versioned) holding sessions, messages, retrieval windows, FTS index, and embedding vectors. Refreshed - on demand from the scrapers; fully derived, rebuilt from scratch on + on demand from the scrapers and at MCP server start; fully derived, rebuilt from scratch on corruption or schema mismatch. - **Drift log** (`src/scrapers/drift-log.ts`) — per-tool record of the places another tool's transcripts did not match what the scraper expected, diff --git a/PRODUCT.md b/PRODUCT.md index 3fccecef..d5bc82ad 100644 --- a/PRODUCT.md +++ b/PRODUCT.md @@ -56,7 +56,10 @@ Single-user, single-machine. There is no team, sync, or server component. - Five read-only MCP tools: recent sessions, session detail, search, continuity status, handoff manifest. - CLI: `setup` (wire MCP config, managed instruction blocks, skills, and the - Claude Code SessionStart hook), `status`, `disconnect`. + Claude Code SessionStart hook), `status`, `scan` (read the stores into the + index now, `--embed` to finish vectorizing too), `calibrate` (time the + embedding model on this machine's devices and use the fastest), + `disconnect`. Out of scope (deliberately, and documented everywhere the product speaks): no daemon, no API server, no dashboard, no generated summaries or briefs, diff --git a/README.md b/README.md index eabec189..d702aea1 100644 --- a/README.md +++ b/README.md @@ -168,6 +168,19 @@ per-file offset, so after the first pass it reads only what each tool has appended. The first pass over a large history is the expensive one — see the note above. +`xtctx scan --embed` additionally vectorizes every window the scan leaves +without one, running to completion however long that takes rather than to a +budget. You need it when `xtctx status` says the backlog is too large to +finish in the background — otherwise the server gets there on its own. + +`xtctx calibrate` times the embedding model on each execution provider this +machine offers and remembers the fastest in `~/.xtctx/device.json`. On a +machine with a usable GPU that has measured roughly six times faster than the +CPU; on one without, it picks the CPU and nothing changes. `scan --embed` +runs it automatically the first time, because it is about to spend far longer +than the measurement costs; `--no-calibrate` skips that. Vectors are identical +whichever device wins, so this changes speed and nothing else. + Generated MCP clients should use: ```json @@ -237,9 +250,12 @@ startup hooks; others receive MCP config plus managed instructions only. - Transcript formats belong to each upstream tool and can drift. The drift tests and format fingerprints exist to catch parser breakage, but `xtctx status` is still the source of truth for your machine. -- Semantic search is lazy. The first semantic or hybrid query may initialize - the local embedding provider and create local vectors; hybrid search falls - back to keyword search if vector generation is unavailable. +- Vectors are built incrementally, and the MCP server also works the backlog + down in the background when it starts, as long as this machine's measured + rate says the remainder fits in fifteen minutes. Above that nothing drains + it on its own and `xtctx status` says so, naming `xtctx scan --embed`. + Hybrid search falls back to keyword whenever vectors are missing or the + embedding model is unavailable, and `xtctx status` reports the reason. - Antigravity conversation `.pb` files are not parsed directly; retrieval uses the local language-server API when available, otherwise readable `brain` artifacts. diff --git a/docs/architecture.md b/docs/architecture.md index 0d7d9e6e..60180943 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -75,8 +75,11 @@ turns. Each embedded window includes the session reference, message range, turn order, message index, role, timestamp, and raw message content. Retrieval ranks semantic similarity together with keyword, recency, and continuity signals, then returns the matched message range so the agent can drill into the raw session. -Vector creation is lazy; if the embedding provider is unavailable during hybrid -search, xtctx falls back to keyword retrieval. +Vector creation is incremental: searches vectorize a slice per call, and the +MCP server drains the rest in the background at startup when the estimate fits +its budget. If the embedding provider is unavailable during hybrid search, +xtctx falls back to keyword retrieval and records the reason, which +`xtctx status` and `xtctx_continuity_status` both report. ## Storage diff --git a/docs/embedding-performance.md b/docs/embedding-performance.md index 9c3b61ba..82e87305 100644 --- a/docs/embedding-performance.md +++ b/docs/embedding-performance.md @@ -1,33 +1,43 @@ # Embedding performance -Measurements taken on 2026-09-20 while looking for a way to make indexing -faster. None of them changed the code; they exist so the next person asking -"can we speed this up" starts from evidence rather than from the same four -guesses. +Measurements taken on 2026-09-20 and 2026-09-21 while looking for a way to +make indexing faster. They exist so the next person asking "can we speed this +up" starts from evidence rather than from the same four guesses. Three of the four guesses were wrong, which is the main reason this file is worth keeping. +**Read this first, because the file is written in two voices.** Everything up +to and including "Model bake-off" was written on 2026-09-20, when none of it +had changed the code and the conclusions were all "blocked on". On 2026-09-21 +three of them shipped: device calibration, the model change to bge-small, and +a bake-off harness that makes the model rows reproducible. Sections written +before that day have been annotated where they are now false rather than +rewritten, because how a wrong conclusion was reached is the part worth +keeping. Where an annotation and a table disagree, the annotation is current. + ## What is reproducible here, and what is not -Only one figure below traces to something committed: the MiniLM eval baseline -(hybrid 0.333 / 0.533 / 0.183), which is -`tests/eval/results/ranking-baseline.json` and is regenerated by -`npm run test:eval`. +`tests/eval/results/ranking-baseline.json` is regenerated by +`npm run test:eval` and now reads bge-small's numbers — hybrid +0.417 / 0.583 / 0.267. The MiniLM figures quoted throughout this file +(0.333 / 0.533 / 0.183) were that file's contents until 2026-09-21 and are +kept as the comparison the model change was made against. -One more is now reproducible: the three-OS device table, which +Two more are now reproducible. The three-OS device table, which `scripts/probe-embedding-device.mjs` regenerates and `.github/workflows/embedding-device-probe.yml` runs on the same three runners. +And every model row below: `scripts/embedding-bakeoff.ts` takes `--model` and +`--sweep`, and its MiniLM row at 0.15/0.36 reproduces the old committed +baseline exactly, which is what establishes it measures the same thing. Everything else — the single-machine device table, the batch sweep, the dtype -comparison, the segment-duplication count, and the bge/gte rows — was measured -in one session on one machine with scratch scripts that are not in this -repository. The method +comparison, and the segment-duplication count — was measured in one session on +one machine with scratch scripts that are not in this repository. The method is described precisely enough to redo, and the ratios are the durable part, but nothing here re-runs them and nobody should treat the absolute numbers as -checkable. The bge and gte rows in particular cannot be reproduced without -changes the eval harness does not have: it runs one fixed provider and has no -way to select a model or vary a threshold. +checkable. (The bge and gte rows said the same until 2026-09-21, when +`scripts/embedding-bakeoff.ts` was written and reproduced them.) Stated plainly because the alternative is worse: a number that reads as evidence while being untraceable is how the "18ms per embed" figure below @@ -73,8 +83,9 @@ For a 21,349-segment backlog that is roughly 21–29 minutes on CPU against about 4 minutes on the GPU. `onnxruntime-node` already ships `DirectML.dll` in the package this project -installs. Nothing needs adding to `package.json`; the `device` option is -simply not passed today. +installs. Nothing needed adding to `package.json`; the `device` option was +simply not passed at the time. It is now, chosen by `xtctx calibrate` — see +"Calibration, as implemented". ## Device, on three operating systems: WebGPU is not a fallback chain @@ -102,9 +113,9 @@ each device in its own process: Two results are worth more than the speed column. -**`auto` is `cpu`, on all three.** That is what the product passes today, so -the GPU is not merely underused — it is unused, and the desktop measurement -above is not being partially realised by anyone. +**`auto` is `cpu`, on all three.** That was what the product passed at the +time, so the GPU was not merely underused but unused. It now passes the device +`xtctx calibrate` measured. **WebGPU on a GPU-less Windows machine is fifty times SLOWER, and does not fail.** It found a software adapter, initialised cleanly, returned correct @@ -244,10 +255,23 @@ The cost: indexing the eval corpus took 25.9s under bge-small against MiniLM's 14.2s, about 1.8x slower — the constraint the GPU result above loosens. **Caveat.** The eval corpus is synthetic and 60 queries. Differences this size -are suggestive, not settled, and this should be confirmed against a real index -before bge-small becomes the default. Changing model means re-embedding -everything, which `dropVectorsFromOtherModels` already handles correctly by -name. +are suggestive, not settled. Changing model means re-embedding everything, +which `dropVectorsFromOtherModels` already handles correctly by name. + +**Superseded 2026-09-21.** bge-small is the default, at **0.62 / 0.64** rather +than the 0.55 / 0.65 recommended above. `scripts/embedding-bakeoff.ts` +reproduced every row in this section and then swept finer, which found that +the *confidence* floor was what destroyed vector mode: at 0.55/0.65 vector +scores 0.325, at 0.55/0.64 it scores 0.385, both at a false-positive rate of +zero. The shipped pair scores hybrid 0.417 / 0.583 / 0.267 and vector +0.398 / 0.533 / 0.317, and is what +`tests/eval/results/ranking-baseline.json` now holds. + +The caveat above was not discharged and is not claimed to be: the corpus is +still synthetic and still sixty queries. What changed is that the margin holds +across three modes and every threshold pair swept rather than resting on one +row, and that the 1.8x indexing cost stopped being decisive once calibration +made embedding ~6x faster on a machine with a GPU. ## What this leaves @@ -255,10 +279,10 @@ Ranked by value against effort, on the evidence above: 1. **GPU chosen by measurement.** Done — `xtctx calibrate`, and automatically inside `xtctx scan --embed`. See below. -2. **Batch size 16.** ~10–20%, no vector change, one constant. Blocked on a - cleaner measurement. -3. **bge-small with its own thresholds.** Better retrieval at 1.8x the - indexing cost. Blocked on confirmation against a real index. +2. **Batch size 16.** ~10–20%, no vector change, one constant. Still blocked + on a cleaner measurement — and worth re-taking on the GPU, since every + figure in that table is a CPU one. +3. **bge-small with its own thresholds.** Done, at 0.62/0.64. Closed, with reasons above: quantization, segment caching, multi-process embedding, and (from `DEFAULT_EMBEDDING_MODEL`) mpnet and static models. diff --git a/docs/embedding-providers.md b/docs/embedding-providers.md index b0332464..f568f3d1 100644 --- a/docs/embedding-providers.md +++ b/docs/embedding-providers.md @@ -1,15 +1,25 @@ # Embedding providers Design for letting a project embed through an OpenAI-compatible endpoint -instead of the bundled local model. Nothing here is built yet. +instead of the bundled local model. + +**Built on 2026-09-21**, with two deliberate deviations and two parts left +out. Deviations: the local vector identity stays the bare HuggingFace id +rather than gaining a `local:` prefix (every remote identity is +`openai:`-prefixed and cannot collide with one, while renaming the local one +would discard every existing project's vectors for no gain), and remote +vectors are normalized on receipt to match the local pipeline. Left out: the +"thresholds unswept for this model" warning described under Status surface, +and any threshold sweep run against a remote provider. `xtctx status` does +print the endpoint. Where this document and the code disagree, the code is +current. ## What stays true -xtctx ships local-only and stays local-only by default. The default MiniLM -model — downloaded on first use, not bundled in the package, which ships -`dist` only — is what runs when nobody configures anything, and that is the -behaviour -every existing claim describes. +xtctx ships local-only and stays local-only by default. The default model — +`Xenova/bge-small-en-v1.5` since 2026-09-21, downloaded on first use rather +than bundled, since the package ships `dist` only — is what runs when nobody +configures anything. An endpoint is opt-in, per project, and never inferred — no environment variable that happens to be set, no auto-detection of a local server on a @@ -30,15 +40,20 @@ xtctx does on its own and should say so: Two cases, and the local one is the stronger of the two. **A local inference server.** Ollama and LM Studio both expose -`/v1/embeddings`, both keep everything on the machine, and both can use a GPU -that xtctx's in-process ONNX runtime is not currently using. Measured on this +`/v1/embeddings`, both keep everything on the machine, and both can use a GPU. +That last point was the stronger half of this argument until 2026-09-21, when +`xtctx calibrate` gave the in-process runtime the same GPU — so an endpoint is +now a way to reach a *different* model, not the only way to reach the +hardware. Measured on this machine, DirectML embedded the same segments about six times faster than the CPU path and produced numerically identical vectors (mean cosine 1.000000 against CPU, worst pair 0.999999). An endpoint is one way to reach that hardware without xtctx owning the GPU problem itself. -**A hosted model.** Better retrieval than a 22M-parameter model can give, for -someone who has already decided their transcripts may leave the machine. +**A hosted model.** Better retrieval than a small local model can give, for +someone who has already decided their transcripts may leave the machine. (The +"22M-parameter" figure this line used to quote was MiniLM's, which is no +longer the local default.) ## Interface @@ -70,8 +85,8 @@ embedding: apiKeyEnv: OLLAMA_API_KEY # name of an env var, never the key itself batchSize: 32 timeoutMs: 30000 - minSemanticCosine: 0.15 # see Thresholds - minConfidentCosine: 0.36 + minSemanticCosine: 0.62 # see Thresholds; defaults track the + minConfidentCosine: 0.64 # local model and moved with it ``` `apiKeyEnv` names an environment variable. The key is never written to @@ -92,9 +107,15 @@ identity therefore includes the endpoint: ``` openai:https://api.openai.com/v1:text-embedding-3-small -local:Xenova/all-MiniLM-L6-v2 +Xenova/bge-small-en-v1.5 ``` +The local form is the bare HuggingFace id, not `local:`-prefixed as an earlier +draft of this document had it. Remote identities all begin `openai:` and +cannot collide with a HuggingFace id, so the prefix buys nothing — and adding +it would make `dropVectorsFromOtherModels` discard every vector in every +existing project on the first open after the upgrade. + Changing the endpoint or the model then invalidates vectors the same way changing the local model already does. Dimensions then need no separate handling — a different dimension count only ever arrives with a different @@ -189,10 +210,10 @@ partially. 2. **Should `scan --embed` behave differently against an endpoint?** It currently runs uncapped, which is right for local compute and possibly expensive against a metered API. -3. **Is a per-provider threshold sweep something xtctx can run itself?** The - sweeps recorded in this repository were done by hand, against a temporarily - patched constant; the eval harness runs one fixed provider and has no way to - select a model or vary a threshold. A +3. **Is a per-provider threshold sweep something xtctx can run itself?** + `scripts/embedding-bakeoff.ts` now sweeps thresholds and selects a model, + but against the synthetic eval corpus rather than a project's own index, + and only for local models. A `xtctx calibrate` that sweeps against the project's own index would remove the unswept-threshold warning entirely, and is a larger piece of work than the provider itself. From d9d21f3c5b47f6d1aea6f6cc550a535950c80144 Mon Sep 17 00:00:00 2001 From: Felix Stubner Date: Mon, 21 Sep 2026 19:23:17 +0100 Subject: [PATCH 24/34] fix(cli): make --help answer a stranger's questions, not a maintainer's MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The first thing `xtctx --help` said was how to start the MCP server from a non-interactive stdio pair and which environment variable suppresses that — scripting advice, before any statement of what the tool is for or what to run first. It now says what xtctx does in two sentences and names the first command, with the stdio note kept below for the people who need it. `--hook` and `--tool` are hidden rather than removed. They are how a tool's hook re-enters this CLI, never something a person types, and listing them beside `--project` and `--version` made them read as options a newcomer is expected to understand. Every command description said what the code does instead of what the user gets. "Configure MCP, hooks, managed handoff instructions, and synced skills" requires knowing what all four are before it says anything; "Set this project up so agents can read each other's history here" says why you would run it. The same for the other four — a stranger could not previously tell from `--help` when they would ever want `scan` or `calibrate`, and `calibrate` makes indexing roughly six times faster. --- src/cli/index.ts | 29 ++++++++++++++++++++--------- 1 file changed, 20 insertions(+), 9 deletions(-) diff --git a/src/cli/index.ts b/src/cli/index.ts index 754f48b2..47293047 100644 --- a/src/cli/index.ts +++ b/src/cli/index.ts @@ -1,5 +1,5 @@ #!/usr/bin/env node -import { Command } from "commander"; +import { Command, Option } from "commander"; import { runCalibrate } from "./calibrate.js"; import { runDisconnect } from "./disconnect.js"; import { runHook } from "./hook.js"; @@ -109,7 +109,13 @@ export async function main(argv = process.argv): Promise { .name("xtctx") .description( [ - "Local cross-tool handoff for AI coding agents", + "Local cross-tool handoff for AI coding agents.", + "", + "Your coding agents already write transcripts. xtctx indexes them and", + "serves them over MCP, so the next agent you open can read what the", + "last one did in this repo.", + "", + "Start with: xtctx setup", "", "Run with no command and non-interactive stdio and xtctx starts its MCP", "server over stdio. Set XTCTX_NO_AUTO_MCP=1 to print this help instead,", @@ -126,7 +132,7 @@ export async function main(argv = process.argv): Promise { .option("-y, --yes", "Apply setup without prompting", false) .option("--repair", "Remove legacy generated xtctx config before writing current setup", false) .option("--global-mcp", "Also configure Copilot CLI global MCP (Antigravity MCP is always configured)", false) - .description("Configure MCP, hooks, managed handoff instructions, and synced skills") + .description("Set this project up so agents can read each other's history here") .action( async ( projectPath: string | undefined, @@ -145,7 +151,7 @@ export async function main(argv = process.argv): Promise { program .command("status") .option("-p, --project ", "Project root (defaults to cwd)") - .description("Diagnose xtctx handoff wiring and local transcript index") + .description("Check whether handoff is working here, and what to do if not") .action(async (options: { project?: string }) => { const globalOptions = program.opts<{ project?: string }>(); await runStatus({ projectPath: options.project ?? globalOptions.project }); @@ -163,7 +169,7 @@ export async function main(argv = process.argv): Promise { "--no-calibrate", "With --embed, skip measuring which device embeds fastest on this machine", ) - .description("Scan the enabled transcript stores into this project's index, then exit") + .description("Index this project's transcripts now instead of waiting for an agent to ask") .action(async (options: { project?: string; embed?: boolean; calibrate?: boolean }) => { const globalOptions = program.opts<{ project?: string }>(); await runScan({ @@ -176,7 +182,7 @@ export async function main(argv = process.argv): Promise { program .command("calibrate") .option("--force", "Measure again even if this machine already has a verdict", false) - .description("Time the embedding model on this machine's GPU and CPU, and use the faster") + .description("Find the fastest device on this machine for indexing, and use it") .action(async (options: { force: boolean }) => { await runCalibrate({ force: options.force }); }); @@ -192,7 +198,7 @@ export async function main(argv = process.argv): Promise { ) .option("-p, --project ", "Project root") .option("-y, --yes", "Apply disconnect without prompting", false) - .description("Remove xtctx management from a tool without deleting transcript data") + .description("Stop xtctx managing a tool here, leaving your transcripts untouched") .action( async ( tool: string | undefined, @@ -210,8 +216,13 @@ export async function main(argv = process.argv): Promise { ); program - .option("--hook ", "Internal hook event name") - .option("--tool ", "Tool invoking an internal hook") + // Hidden, not removed: these are how a tool's hook re-enters this CLI, + // never something a person types. Listed among `--project` and + // `--version`, they read as options a newcomer is expected to understand, + // and the first thing `xtctx --help` showed was two knobs for a mechanism + // that is entirely internal. + .addOption(new Option("--hook ", "Internal hook event name").hideHelp()) + .addOption(new Option("--tool ", "Tool invoking an internal hook").hideHelp()) .option("-p, --project ", "Project root"); program.action(async () => { From 82b3d27ea22128d41ad27034e4276d0a93470302 Mon Sep 17 00:00:00 2001 From: Felix Stubner Date: Mon, 21 Sep 2026 19:27:51 +0100 Subject: [PATCH 25/34] fix(tests): type-check the tests, which is where five errors were hiding `npm run verify:release` failed on five type errors that `npx tsc --noEmit` does not see. The two are not the same check: the default project compiles `src`, while `npm run typecheck` uses `tsconfig.test.json` and compiles the tests with it. I had been running the first and calling it a type check all session. Four are test stubs of `HandoffStatus` that predate `vector_device`, and one is a `deviceCandidates` call passing a bare `string` where the parameter is `NodeJS.Platform`. None affects shipped behaviour, which is exactly why they survived a suite that was green the whole time. `npm run verify:release` now exits 0. --- tests/handoff/device-calibration.test.ts | 2 +- tests/integration/handoff-mcp.test.ts | 1 + tests/mcp/hardening.test.ts | 1 + tests/mcp/manifest.test.ts | 1 + tests/mcp/source-field-safety.test.ts | 1 + 5 files changed, 5 insertions(+), 1 deletion(-) diff --git a/tests/handoff/device-calibration.test.ts b/tests/handoff/device-calibration.test.ts index c38f3a37..69b37335 100644 --- a/tests/handoff/device-calibration.test.ts +++ b/tests/handoff/device-calibration.test.ts @@ -103,7 +103,7 @@ describe("deviceCandidates", () => { }); it("always measures the CPU, because it is what everything is compared to", () => { - for (const platform of ["win32", "darwin", "linux"]) { + for (const platform of ["win32", "darwin", "linux"] as NodeJS.Platform[]) { expect(deviceCandidates(platform)).toContain("cpu"); } }); diff --git a/tests/integration/handoff-mcp.test.ts b/tests/integration/handoff-mcp.test.ts index 5a387dc5..7fa78029 100644 --- a/tests/integration/handoff-mcp.test.ts +++ b/tests/integration/handoff-mcp.test.ts @@ -77,6 +77,7 @@ class FixtureSessionService implements SessionService { vector_segment_backlog: 0, vector_ms_per_segment: null, vector_model: "fixture-embedding", + vector_device: null, tools: [ { tool: "codex", diff --git a/tests/mcp/hardening.test.ts b/tests/mcp/hardening.test.ts index 3ab8f697..b8313d7a 100644 --- a/tests/mcp/hardening.test.ts +++ b/tests/mcp/hardening.test.ts @@ -52,6 +52,7 @@ class DetailFixtureService implements SessionService { vector_segment_backlog: 0, vector_ms_per_segment: null, vector_model: "fixture", + vector_device: null, tools: [ { tool: "codex", diff --git a/tests/mcp/manifest.test.ts b/tests/mcp/manifest.test.ts index e0bd0c45..fa1e0102 100644 --- a/tests/mcp/manifest.test.ts +++ b/tests/mcp/manifest.test.ts @@ -51,6 +51,7 @@ class LimitHonoringService implements SessionService { vector_segment_backlog: 0, vector_ms_per_segment: null, vector_model: "fixture", + vector_device: null, tools: [], }; } diff --git a/tests/mcp/source-field-safety.test.ts b/tests/mcp/source-field-safety.test.ts index f6e7a1e4..df78ba39 100644 --- a/tests/mcp/source-field-safety.test.ts +++ b/tests/mcp/source-field-safety.test.ts @@ -73,6 +73,7 @@ class FixtureService implements SessionService { vector_segment_backlog: 0, vector_ms_per_segment: null, vector_model: "fixture", + vector_device: null, tools: [], }; } From 508adbc7d681f14b27347fcd81f735b6b8a6d7c6 Mon Sep 17 00:00:00 2001 From: Felix Stubner Date: Tue, 22 Sep 2026 00:07:30 +0100 Subject: [PATCH 26/34] perf(embeddings): batch 16, measured on the real path after the benchmark lied MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `MAX_BATCH_SIZE` moves from 32 to 16, worth about 13% of indexing time. The number is not the interesting part. I wrote `scripts/probe-batch-size.mjs` first. It said the GPU wanted the largest batch available — 4.0ms/segment at 128 against 5.3 at 32, a 1.37x win that reproduced cleanly across repeated runs — and that the CPU wanted the smallest. I implemented a device-dependent constant on the strength of it. Then I measured the real path: `xtctx scan --embed` over this project's own index from an empty vector table, 150 seconds per size, on DirectML. batch ms/window 8 123.4 16 107.8, 107.9 32 123.4 128 542.9 The benchmark's recommendation was a 5x regression. Its segments are all exactly 1000 characters. Real ones are not — windows hold a median of 4 segments and a 95th percentile of 17, of varying size — and a batch is padded to its longest member, so a wide batch of mixed lengths spends most of its work on padding. Uniform inputs hide the dominant cost of the real workload, which is why the benchmark was confidently, reproducibly wrong. That is the same failure as the "18ms per embed" figure that drove a model change and had to be reverted: a measurement whose inputs do not have the shape of the real ones is not weak evidence, it is evidence for a different question. So the device-dependence is gone too — it existed only because of the benchmark. One constant, 16, which is also what the independent CPU sweep in docs/embedding-performance.md found on real index segments, at the same margin. The script keeps a warning at the top saying what it is and is not evidence for. Verified end to end at 101.9 ms/window on the shipped build. `npm run verify:release` exits 0. --- docs/embedding-performance.md | 33 ++++++-- scripts/probe-batch-size.mjs | 149 ++++++++++++++++++++++++++++++++++ src/handoff/embeddings.ts | 39 ++++++++- 3 files changed, 212 insertions(+), 9 deletions(-) create mode 100644 scripts/probe-batch-size.mjs diff --git a/docs/embedding-performance.md b/docs/embedding-performance.md index 82e87305..640b40b8 100644 --- a/docs/embedding-performance.md +++ b/docs/embedding-performance.md @@ -167,12 +167,31 @@ Three paired runs on CPU, fp32, same segments each time: | 32 | 100.5 | 88.7 | 81.4 | | 64 | 99.9 | | | -16 wins every pairing against 32, by roughly 10–20%. `MAX_BATCH_SIZE` is -currently 32. +16 wins every pairing against 32, by roughly 10–20%. -The absolute numbers are noisy and the sample is three pairs, so this is a -consistent direction rather than a settled figure. Before changing the -constant it deserves more repetitions on a quiet machine, including 24. +**Settled 2026-09-21, and `MAX_BATCH_SIZE` is 16.** The direction above turned +out to be right, but nothing above it is why. Measured by running +`xtctx scan --embed` over this project's own index from an empty vector table, +150 seconds per size, on DirectML: + +| batch | ms/window | +| --- | --- | +| 8 | 123.4 | +| **16** | **107.8, 107.9** | +| 32 | 123.4 | +| 128 | 542.9 | + +`scripts/probe-batch-size.mjs` was written first and said the opposite for the +GPU — 4.0ms/segment at 128 against 5.3 at 32, a 1.37x win reproducing across +runs. Acting on that would have been a 5x regression. Its segments are all +exactly 1000 characters; real ones are not, and a batch is padded to its +longest member, so a wide batch of mixed lengths spends most of its work on +padding. Uniform inputs hide the dominant cost of the real workload. + +That is the same failure as the "18ms per embed" figure at the top of this +file. A benchmark that does not reproduce the shape of the real input is not +weak evidence, it is evidence for the wrong question — and it reproduced +cleanly three times while pointing the wrong way. ## Quantized weights: rejected @@ -279,9 +298,7 @@ Ranked by value against effort, on the evidence above: 1. **GPU chosen by measurement.** Done — `xtctx calibrate`, and automatically inside `xtctx scan --embed`. See below. -2. **Batch size 16.** ~10–20%, no vector change, one constant. Still blocked - on a cleaner measurement — and worth re-taking on the GPU, since every - figure in that table is a CPU one. +2. **Batch size 16.** Done — measured on the real path, ~13% on the GPU. 3. **bge-small with its own thresholds.** Done, at 0.62/0.64. Closed, with reasons above: quantization, segment caching, multi-process diff --git a/scripts/probe-batch-size.mjs b/scripts/probe-batch-size.mjs new file mode 100644 index 00000000..612d747c --- /dev/null +++ b/scripts/probe-batch-size.mjs @@ -0,0 +1,149 @@ +#!/usr/bin/env node +/** + * How many segments should go to the model at once, measured per device. + * + * `docs/embedding-performance.md` recorded 16 beating the current 32 by + * 10–20% and declined to act on it: "the absolute numbers are noisy and the + * sample is three pairs, so this is a consistent direction rather than a + * settled figure. Before changing the constant it deserves more repetitions + * on a quiet machine, including 24." + * + * That table is also entirely CPU. `MAX_BATCH_SIZE` now applies on whatever + * device `xtctx calibrate` picked, and batching economics are not the same on + * a GPU — the whole reason a GPU helps is that it does wide work in parallel, + * so the size that wins on eleven CPU cores has no reason to win there. + * + * Two lessons from the device probe are built in, because both changed an + * answer there: + * + * A short warmup measures a device that is still starting up. The first + * call carries graph compilation and buffer allocation, which inverted the + * dml/webgpu ranking until the warmup became a full pass. + * + * One timed pass is not a measurement. Other load on the machine only ever + * makes a pass slower, so the fastest of several is the closest this gets to + * the device's real throughput. + * + * Unlike the device probe this does NOT need a process per configuration — + * batch size is a loop parameter over one already-loaded pipeline, and it is + * two providers in one process that fail with `bad allocation`, not two batch + * sizes. + * + * node scripts/probe-batch-size.mjs --device=cpu + * node scripts/probe-batch-size.mjs --device=dml + * + * READ THIS BEFORE BELIEVING ITS OUTPUT. On 2026-09-21 it said DirectML wanted + * the largest batch available — 4.0ms/segment at 128 against 5.3 at 32, + * reproducing across runs — and acting on that was a 5x REGRESSION when + * measured through `xtctx scan --embed` on the real index. The segments here + * are all exactly the same length; real ones are not, and a batch is padded to + * its longest member, so uniform inputs hide the dominant cost of the real + * workload. This script is useful for comparing devices at a FIXED batch size. + * It is not evidence for choosing one. See `MAX_BATCH_SIZE`. + */ +import { parseArgs } from "node:util"; + +const MODEL = "Xenova/bge-small-en-v1.5"; +const DTYPE = "fp32"; +const DEFAULT_BATCH_SIZES = [8, 16, 24, 32, 64]; +/** Enough that even the largest batch runs twice, so batching is exercised. */ +const SEGMENT_COUNT = 128; +const SEGMENT_CHARS = 1000; +const PASSES = 3; + +/** + * Deterministic text at the length real segments have. + * + * Length is the term that matters: an earlier round of this work benchmarked + * strings like `warm query number 5`, concluded a heavier model was + * affordable, and shipped a change reverted the next day. + */ +function buildSegments() { + const words = [ + "session", "transcript", "index", "vector", "window", "segment", "scraper", + "handoff", "retrieval", "keyword", "semantic", "threshold", "cosine", + "database", "migration", "config", "project", "message", "timestamp", "tool", + ]; + const segments = []; + let seed = 1; + for (let index = 0; index < SEGMENT_COUNT; index += 1) { + let text = ""; + while (text.length < SEGMENT_CHARS) { + seed ^= seed << 13; + seed ^= seed >>> 17; + seed ^= seed << 5; + seed >>>= 0; + text += `${words[seed % words.length]} `; + } + segments.push(text.slice(0, SEGMENT_CHARS)); + } + return segments; +} + +async function embedAll(extractor, segments, batchSize) { + for (let start = 0; start < segments.length; start += batchSize) { + await extractor(segments.slice(start, start + batchSize), { + pooling: "mean", + normalize: true, + }); + } +} + +const { values } = parseArgs({ + options: { + device: { type: "string", default: "cpu" }, + /** `--batches=2,4,8` to look closely at one end. */ + batches: { type: "string" }, + }, + strict: false, +}); +const device = String(values.device); +const BATCH_SIZES = values.batches + ? String(values.batches).split(",").map((size) => Number.parseInt(size, 10)) + : DEFAULT_BATCH_SIZES; + +const segments = buildSegments(); +const transformers = await import("@huggingface/transformers"); +const options = { dtype: DTYPE }; +if (device !== "auto") { + options.device = device; +} + +const extractor = await transformers.pipeline("feature-extraction", MODEL, options); + +// A full untimed pass at the largest batch, so every allocation this run will +// ever need has already happened before anything is timed. +await embedAll(extractor, segments, Math.max(...BATCH_SIZES)); + +process.stdout.write( + `\n${MODEL} ${DTYPE} on ${device} — ${SEGMENT_COUNT} segments of ${SEGMENT_CHARS} chars, best of ${PASSES}\n\n`, +); +process.stdout.write("batch ms/segment vs 32 cores\n"); + +const results = []; +for (const batchSize of BATCH_SIZES) { + let best = Number.POSITIVE_INFINITY; + let bestCores = 0; + for (let pass = 0; pass < PASSES; pass += 1) { + const cpuBefore = process.cpuUsage(); + const startedAt = performance.now(); + await embedAll(extractor, segments, batchSize); + const elapsed = performance.now() - startedAt; + const cpuAfter = process.cpuUsage(cpuBefore); + if (elapsed < best) { + best = elapsed; + bestCores = (cpuAfter.user + cpuAfter.system) / 1000 / elapsed; + } + } + results.push({ batchSize, msPerSegment: best / segments.length, cores: bestCores }); +} + +const baseline = results.find((row) => row.batchSize === 32)?.msPerSegment; +for (const row of results) { + const ratio = baseline ? `${(baseline / row.msPerSegment).toFixed(2)}x` : "—"; + process.stdout.write( + `${String(row.batchSize).padEnd(6)} ${row.msPerSegment.toFixed(2).padEnd(11)} ` + + `${ratio.padEnd(7)} ${row.cores.toFixed(1)}\n`, + ); +} +process.stdout.write("\n"); diff --git a/src/handoff/embeddings.ts b/src/handoff/embeddings.ts index c14c88ce..7d21a5fe 100644 --- a/src/handoff/embeddings.ts +++ b/src/handoff/embeddings.ts @@ -120,7 +120,44 @@ export const DEFAULT_EMBEDDING_DTYPE = "fp32"; const MAX_SEQ_TOKENS = 256; /** ~4 characters per token, the budget splitTextForEmbedding segments to. */ export const MAX_SEQ_CHARS = MAX_SEQ_TOKENS * 4; -const MAX_BATCH_SIZE = 32; +/** + * Segments handed to the model in one forward pass. + * + * Sixteen, measured on the real path rather than on a benchmark — and the + * difference between those two is the whole story here. + * + * `scripts/probe-batch-size.mjs` embeds uniform 1000-character segments and + * said the GPU wanted the largest batch available: 4.0ms/segment at 128 + * against 5.3 at 32, a 1.37x win that reproduced across runs. Acting on it + * would have been a 5x REGRESSION. Measured instead by running + * `xtctx scan --embed` over this project's own index from an empty vector + * table, 150 seconds each on DirectML: + * + * batch ms/window + * 8 123.4 + * 16 107.8, 107.9 + * 32 123.4 + * 128 542.9 + * + * The benchmark's segments were all exactly the same length. Real ones are + * not — windows hold a median of 4 segments and a 95th percentile of 17, of + * varying size — and a batch is padded to its longest member, so a wide batch + * of mixed lengths spends most of its work on padding. Uniform inputs hide + * the dominant cost of the real workload entirely. + * + * This is the same mistake as the "18ms per embed" figure recorded on + * `DEFAULT_EMBEDDING_MODEL`, which was taken on strings like "warm query + * number 5" and drove a model change that had to be reverted. Measure this on + * real content, through the real path, or do not move it. + * + * One constant, not one per device. An earlier version of this change made it + * device-dependent on the strength of the benchmark above; the real-path + * measurement removed the reason. The independent CPU measurement in + * `docs/embedding-performance.md` — also taken on real segments from this + * index — put 16 ahead of 32 by 10-20%, which is the same answer and the same + * margin as the GPU rows above. + */ +const MAX_BATCH_SIZE = 16; export interface EmbeddingProvider { readonly model: string; From 1117ea8a85b056791b60b06526639d65f228f1f1 Mon Sep 17 00:00:00 2001 From: Felix Stubner Date: Tue, 22 Sep 2026 23:20:20 +0100 Subject: [PATCH 27/34] fix(calibrate): respect XTCTX_DISABLE_EMBEDDINGS, which CI caught and I did not MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `scan --embed` began calibrating automatically earlier today. Calibration loads the embedding model in a child process per device, and it did not check `XTCTX_DISABLE_EMBEDDINGS=1` — a switch whose entire meaning is that the model is never loaded. So a scan told not to touch a model spent minutes doing exactly that. It timed out two tests in `tests/cli/scan-embed.test.ts` on ubuntu-latest at 60s. They passed here every time, because this machine has a cached device verdict: with one present calibration is skipped regardless, so the missing guard was never reached locally. A fresh machine — CI, or any user — has no verdict, which is precisely the case the guard is for. `xtctx calibrate` run directly now says why it did nothing rather than silently doing nothing, since being asked for explicitly is different from being triggered. The test is fixed as well as the code. It redirected nothing, so it would have kept passing here with the guard removed; it now points HOME and USERPROFILE at an empty temp directory for the duration, and fails in 13s when the guard is taken out. Passing for the wrong reason is the same defect the product bug had. Also corrected the file's header comment, which still described a session-start hook launching a detached scan. That code was removed on 2026-09-02. --- src/cli/calibrate.ts | 9 +++++++ src/cli/scan.ts | 8 +++++++ tests/cli/scan-embed.test.ts | 46 ++++++++++++++++++++++++++++++++---- 3 files changed, 59 insertions(+), 4 deletions(-) diff --git a/src/cli/calibrate.ts b/src/cli/calibrate.ts index 4d690e66..b6b8fbb3 100644 --- a/src/cli/calibrate.ts +++ b/src/cli/calibrate.ts @@ -17,6 +17,15 @@ interface CalibrateOptions { * what every machine did before this existed. */ export async function runCalibrate(options: CalibrateOptions = {}): Promise { + // Asked for directly, so this says why nothing happened rather than + // silently doing nothing — unlike the automatic path in `scan --embed`. + if (process.env.XTCTX_DISABLE_EMBEDDINGS === "1") { + process.stdout.write( + "XTCTX_DISABLE_EMBEDDINGS=1 is set, so there is no model to time. Unset it and run this again.\n", + ); + return; + } + const existing = await readDeviceVerdict(); if (existing && !options.force) { process.stdout.write( diff --git a/src/cli/scan.ts b/src/cli/scan.ts index a4bdb8cf..ebee9dd2 100644 --- a/src/cli/scan.ts +++ b/src/cli/scan.ts @@ -43,6 +43,14 @@ interface ScanOptions { * reopening it with a different provider. */ async function calibrateIfNeeded(): Promise { + // `XTCTX_DISABLE_EMBEDDINGS=1` means the model is never loaded, and + // calibration loads it in two or three child processes — so ignoring the + // switch here made a scan that was supposed to touch no model spend minutes + // doing exactly that. It timed out two tests on a CI runner, which is the + // cheap version of the same surprise a user would get. + if (process.env.XTCTX_DISABLE_EMBEDDINGS === "1") { + return; + } if (await readDeviceVerdict()) { return; } diff --git a/tests/cli/scan-embed.test.ts b/tests/cli/scan-embed.test.ts index 9d98e601..e96fdd5f 100644 --- a/tests/cli/scan-embed.test.ts +++ b/tests/cli/scan-embed.test.ts @@ -4,10 +4,15 @@ * measured on a live 9,232-window project, covering the corpus that way * needed on the order of 570 searches. * - * The default matters more than the flag. The session-start hook launches - * `scan` detached, so a scan that drained unconditionally would start hours - * of embedding every time an agent opened a large project. That is the - * assertion below: without the flag, nothing calls the drain at all. + * The default matters more than the flag: without it, nothing calls the drain + * at all. That is the assertion below. + * + * These tests also pin that `scan --embed` touches no model when + * `XTCTX_DISABLE_EMBEDDINGS=1` is set. Auto-calibration was added to that path + * and did not check the switch, so it spawned two or three child processes + * that each loaded the real model — which timed both of these out at 60s on a + * CI runner, and would have cost a user minutes on a command they had told + * not to embed. */ import { mkdtemp, rm, writeFile } from "node:fs/promises"; import { tmpdir } from "node:os"; @@ -55,6 +60,39 @@ describe("xtctx scan and the embedding backlog", () => { expect(drain).not.toHaveBeenCalled(); }, 60_000); + it("does not calibrate when embeddings are switched off", async () => { + // `XTCTX_DISABLE_EMBEDDINGS=1` means the model is never loaded. Auto + // calibration loads it in a child process per device, so ignoring the + // switch turned "scan without touching a model" into minutes of doing + // exactly that — caught as a 60s timeout on a CI runner, where these two + // tests had always passed before. + // Home redirected at the empty temp dir, so the developer's own cached + // verdict cannot make this pass for the wrong reason: with one present, + // calibration is skipped regardless and the guard under test is never + // reached. That is exactly why these tests passed here and failed on CI. + const realHome = { HOME: process.env.HOME, USERPROFILE: process.env.USERPROFILE }; + process.env.HOME = homeDir; + process.env.USERPROFILE = homeDir; + + const written: string[] = []; + const write = vi + .spyOn(process.stdout, "write") + .mockImplementation(((chunk: unknown) => { + written.push(String(chunk)); + return true; + }) as typeof process.stdout.write); + + try { + await runScan({ projectPath: projectRoot, embed: true }); + } finally { + write.mockRestore(); + process.env.HOME = realHome.HOME; + process.env.USERPROFILE = realHome.USERPROFILE; + } + + expect(written.join("")).not.toContain("Measuring this machine's embedding devices"); + }); + it("drains it with --embed", async () => { const drain = vi.spyOn(SqliteHandoffIndex.prototype, "embedBacklog"); From fd16a0be194c644a4c9b8e37f4470dcd90cd1053 Mon Sep 17 00:00:00 2001 From: Felix Stubner Date: Tue, 22 Sep 2026 23:28:07 +0100 Subject: [PATCH 28/34] fix: calibrate from the server too, because the uncalibrated state was a trap MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Asked why `xtctx calibrate` needs to exist at all, the honest answer was that it mostly should not — it was papering over a hole in the automatic path, and the hole was self-perpetuating. The MCP server drains the vector backlog in the background only when the estimate fits a fifteen-minute budget. That estimate is computed from the rate of whatever device is in use, which is the CPU until something calibrates. So on a large history: the CPU estimate exceeds the budget, the drain is skipped, and the drain was the only thing that would have made the project fast. Nothing else calibrates except `scan --embed`, which most people never run — so the machine stays on the CPU permanently. The user who loses most is the one with the largest history, which is exactly who the mechanism is for. The server now calibrates when it finds a backlog and no verdict, alongside a background drain that already takes minutes. Not in front of a tool call, which remains the line. It takes effect NEXT session, and deliberately so. The provider was constructed with the device known at startup and keeps using it, so this session's drain still runs at the old speed and is still judged by the old estimate. Relaxing the gate on the strength of a verdict the running provider is not using would start an hour of CPU work on the promise of a GPU that is not attached yet — the same shape of mistake as trusting a benchmark over the real path. It also honours `XTCTX_DISABLE_EMBEDDINGS=1`, which is the bug fixed in the previous commit, reintroduced in a second place an hour later and caught before it shipped. Calibration loads a model per device; that switch means no model is ever loaded. `xtctx calibrate` stays as an escape hatch — `--force` after a hardware change, and a way to see the measurements — rather than something a user is expected to discover. --- src/cli/index.ts | 32 ++++++++++++++++++++++++++++++++ 1 file changed, 32 insertions(+) diff --git a/src/cli/index.ts b/src/cli/index.ts index 47293047..7c0f9f99 100644 --- a/src/cli/index.ts +++ b/src/cli/index.ts @@ -11,6 +11,7 @@ import { startMcpServer } from "../mcp/server.js"; import type { SessionService } from "../handoff/types.js"; import { readXtctxPackage } from "../utils/package-info.js"; import { BACKGROUND_EMBED_BUDGET_MS, estimateVectorBacklog } from "../utils/duration.js"; +import { calibrateEmbeddingDevice, readDeviceVerdict } from "../handoff/device.js"; const { version: CLI_VERSION } = readXtctxPackage(import.meta.url); @@ -301,6 +302,37 @@ async function drainVectorsIfAffordable(sessions: SessionService): Promise if (remaining === 0) { return; } + + // Calibrate before judging affordability, not after — otherwise the + // uncalibrated state perpetuates itself. + // + // The estimate below is computed from the rate of whatever device is in use, + // which is the CPU until something calibrates. A large history on the CPU + // estimates well past the budget, so the drain is skipped; the drain was the + // only thing that would have made the project fast; and the machine stays on + // the CPU forever. The user who loses most is the one with the largest + // history, which is the one this whole mechanism is for. + // + // Only when there is a backlog to justify it, and never in front of a tool + // call — this runs detached from the server's start, alongside a drain that + // already takes minutes. + // + // It takes effect NEXT session, not this one. The provider was constructed + // with the device that was known at startup and keeps using it, so this + // session's drain still runs at the old speed and is still judged by the old + // estimate. Deliberately not relaxed on the strength of a verdict the + // running provider is not using: that would start an hour of CPU work on the + // promise of a GPU that is not attached until next time. + // + // `XTCTX_DISABLE_EMBEDDINGS=1` means no model is ever loaded, and + // calibration loads one per device. + if (process.env.XTCTX_DISABLE_EMBEDDINGS !== "1" && !(await readDeviceVerdict())) { + await calibrateEmbeddingDevice().catch(() => { + // Best-effort. Failing to find the fastest device is not a reason to + // stop using the one that has always worked. + }); + } + // No estimate means nothing has embedded yet on this machine, so there is no // measured rate to judge affordability by. Searches still vectorize // incrementally, which is what produces the rate this needs. From f28ec4383d24d54fcbf5833517265eac5136c7ca Mon Sep 17 00:00:00 2001 From: Felix Stubner Date: Tue, 22 Sep 2026 23:48:49 +0100 Subject: [PATCH 29/34] feat(calibrate): automate it properly, so nobody has to know it exists MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Asked whether this could be automated entirely rather than left as an escape hatch, the answer was yes, and the thing blocking it was smaller than I said. I had claimed the verdict could only apply to the next session, because the provider is constructed before calibration runs. But the provider loads its model LAZILY — `extractor` and `loading` are both null until something embeds — so until then, pointing it at a different device costs nothing. `retargetDevice` does that, and refuses once the model is loaded or loading. The refusal is the careful half: swapping the device under a loaded pipeline means discarding it and paying the load again, possibly while a tool call waits on it, to speed up work already running. The caller is told it did not apply rather than left to assume it did. That closes the loop the previous commit left open. The server now measures, retargets the not-yet-loaded provider, and re-judges affordability using the per-segment rate calibration just measured on the chosen device — not the stale CPU rate, which would have skipped the drain on exactly the machines the measurement had made fast. That substitution is not a projection: calibration measures milliseconds per segment, and the estimate multiplies a per-segment rate by the segment backlog. Same arithmetic, fresher measurement of the same quantity. So `calibrate` is no longer something a user needs to discover. Both automatic paths cover it, and the command remains only for re-measuring after a hardware change and for showing the numbers. The README now leads with the automatic behaviour instead of the command, and says plainly that it is not a setup step. Still never behind an agent's tool call — that line was always about where the cost lands, not about whether it is automatic, and both automatic callers are background work that was already going to take minutes. --- README.md | 20 ++++++---- src/cli/calibrate.ts | 17 +++++--- src/cli/index.ts | 46 +++++++++++++++------ src/handoff/embeddings.ts | 35 +++++++++++++++- src/handoff/sqlite-index.ts | 6 +++ src/handoff/types.ts | 7 ++++ tests/handoff/device-retarget.test.ts | 57 +++++++++++++++++++++++++++ 7 files changed, 163 insertions(+), 25 deletions(-) create mode 100644 tests/handoff/device-retarget.test.ts diff --git a/README.md b/README.md index d702aea1..192dc467 100644 --- a/README.md +++ b/README.md @@ -173,13 +173,19 @@ without one, running to completion however long that takes rather than to a budget. You need it when `xtctx status` says the backlog is too large to finish in the background — otherwise the server gets there on its own. -`xtctx calibrate` times the embedding model on each execution provider this -machine offers and remembers the fastest in `~/.xtctx/device.json`. On a -machine with a usable GPU that has measured roughly six times faster than the -CPU; on one without, it picks the CPU and nothing changes. `scan --embed` -runs it automatically the first time, because it is about to spend far longer -than the measurement costs; `--no-calibrate` skips that. Vectors are identical -whichever device wins, so this changes speed and nothing else. +Indexing picks a device by measuring it, and **you do not have to do anything +to get that**. The first time a machine has embedding work worth doing — the +MCP server finding a backlog at session start, or `xtctx scan --embed` — it +times the embedding model on each execution provider available and remembers +the fastest in `~/.xtctx/device.json`, once per machine. On a machine with a +usable GPU that has measured roughly six times faster than the CPU; on one +without, it picks the CPU and nothing changes. Vectors are identical whichever +device wins, so this changes speed and nothing else. + +`xtctx calibrate` runs that measurement on demand and prints it. You need it +only to re-measure after the hardware changes (`--force`) or to see the +numbers — it is not a setup step. `scan --no-calibrate` skips the automatic +run for anyone who would rather start embedding immediately. Generated MCP clients should use: diff --git a/src/cli/calibrate.ts b/src/cli/calibrate.ts index b6b8fbb3..b68e0657 100644 --- a/src/cli/calibrate.ts +++ b/src/cli/calibrate.ts @@ -9,12 +9,19 @@ interface CalibrateOptions { * Time the embedding model on every execution provider this machine offers, * and remember the fastest. * - * A command rather than something that happens on its own, because it costs - * real seconds and loads the model once per device. The MCP server only ever - * *reads* the verdict; nothing spawns processes behind an agent's tool call. + * Nobody needs to run this. Both automatic paths cover it: `xtctx scan + * --embed` calibrates before a long embed, and the MCP server calibrates at + * start when it finds a backlog and no verdict, applying the result to the + * not-yet-loaded provider in the same session. * - * Running it is optional. A machine that never does stays on CPU, which is - * what every machine did before this existed. + * It stays as a command for the two things automation cannot do: `--force` + * after the hardware changes, and showing the measurements to someone who + * wants to see them. It is not a step in getting set up, and nothing should + * tell a user it is. + * + * Still never behind an agent's tool call. That line is about where the cost + * lands, not about whether it is automatic: both automatic callers are + * background work that was already going to take minutes. */ export async function runCalibrate(options: CalibrateOptions = {}): Promise { // Asked for directly, so this says why nothing happened rather than diff --git a/src/cli/index.ts b/src/cli/index.ts index 7c0f9f99..bd1902ad 100644 --- a/src/cli/index.ts +++ b/src/cli/index.ts @@ -317,26 +317,50 @@ async function drainVectorsIfAffordable(sessions: SessionService): Promise // call — this runs detached from the server's start, alongside a drain that // already takes minutes. // - // It takes effect NEXT session, not this one. The provider was constructed - // with the device that was known at startup and keeps using it, so this - // session's drain still runs at the old speed and is still judged by the old - // estimate. Deliberately not relaxed on the strength of a verdict the - // running provider is not using: that would start an hour of CPU work on the - // promise of a GPU that is not attached until next time. + // It applies to THIS session where it can. The provider loads its model + // lazily, so until something embeds, pointing it at a different device costs + // nothing — and the whole point of measuring is undone if the answer only + // arrives next time. `retargetEmbeddingDevice` returns false once the model + // is loaded or loading, which is the case where the drain below is already + // running on the old device and swapping it would mean paying the load again + // to speed up work in flight. // // `XTCTX_DISABLE_EMBEDDINGS=1` means no model is ever loaded, and // calibration loads one per device. + let applied = false; + let calibrated: Awaited> | undefined; if (process.env.XTCTX_DISABLE_EMBEDDINGS !== "1" && !(await readDeviceVerdict())) { - await calibrateEmbeddingDevice().catch(() => { - // Best-effort. Failing to find the fastest device is not a reason to - // stop using the one that has always worked. - }); + calibrated = await calibrateEmbeddingDevice().catch(() => undefined); + if (calibrated) { + applied = sessions.retargetEmbeddingDevice?.(calibrated.device) ?? false; + } + } + + // Judge affordability by the rate that will actually apply. + // + // `etaMs` above was built from `vector_ms_per_segment`, which was measured + // on the device in use before calibration — the CPU. Keeping it would skip + // the drain on exactly the machines calibration just made fast, which is the + // trap this whole branch exists to close. + // + // The substituted rate is not a projection: calibration measured + // milliseconds per segment for this model on the chosen device, and + // `estimateVectorBacklog` multiplies a per-segment rate by the segment + // backlog. Same arithmetic, fresher measurement of the same quantity. + let effectiveEtaMs = etaMs; + if (applied) { + const fresh = await sessions.getStatus(); + const chosen = calibrated?.measured.find((row) => row.device === calibrated.device); + if (chosen?.msPerSegment != null) { + effectiveEtaMs = fresh.vector_segment_backlog * chosen.msPerSegment; + } } + const current = { etaMs: effectiveEtaMs }; // No estimate means nothing has embedded yet on this machine, so there is no // measured rate to judge affordability by. Searches still vectorize // incrementally, which is what produces the rate this needs. - if (etaMs === null || etaMs > BACKGROUND_EMBED_BUDGET_MS) { + if (current.etaMs === null || current.etaMs > BACKGROUND_EMBED_BUDGET_MS) { return; } diff --git a/src/handoff/embeddings.ts b/src/handoff/embeddings.ts index 7d21a5fe..66ae8f70 100644 --- a/src/handoff/embeddings.ts +++ b/src/handoff/embeddings.ts @@ -247,9 +247,40 @@ export class TransformersEmbeddingProvider implements EmbeddingProvider { * the CPU on this machine, because the one configuration where a GPU is * catastrophic is also the one where it does not fail. */ - readonly device?: string, + device?: string, ) { this.model = model; + this.deviceName = device; + } + + private deviceName: string | undefined; + + get device(): string | undefined { + return this.deviceName; + } + + /** + * Point this provider at a device, if it is not too late to matter. + * + * The device is only read when the pipeline loads, and loading is lazy, so + * until then changing it is free. This exists so calibration can take effect + * in the session that ran it rather than the next one: the server starts, + * finds no verdict and a backlog worth draining, measures the devices in + * child processes, and points the not-yet-loaded provider at the winner + * before anything embeds. + * + * Refuses once the model is loaded or loading, and says so by returning + * false. Swapping the device under a loaded pipeline would mean discarding + * it and paying the load again, possibly while a tool call is waiting on it, + * to save time on work that is already running. The caller reports the + * verdict as taking effect next session instead. + */ + retargetDevice(device: string | undefined): boolean { + if (this.extractor !== null || this.loading !== null) { + return false; + } + this.deviceName = device; + return true; } async embed(text: string): Promise { @@ -323,7 +354,7 @@ export class TransformersEmbeddingProvider implements EmbeddingProvider { }; const extractor = await transformers.pipeline("feature-extraction", this.model, { dtype: this.dtype, - ...(this.device === undefined ? {} : { device: this.device }), + ...(this.deviceName === undefined ? {} : { device: this.deviceName }), }); // `model_max_length` is a getter with no setter in @huggingface/transformers, diff --git a/src/handoff/sqlite-index.ts b/src/handoff/sqlite-index.ts index c414204a..c57353d3 100644 --- a/src/handoff/sqlite-index.ts +++ b/src/handoff/sqlite-index.ts @@ -1052,6 +1052,12 @@ export class SqliteHandoffIndex implements SessionService { * background embedding is what would make a large project on a CPU * unusable. */ + /** See `SessionService.retargetEmbeddingDevice`. */ + retargetEmbeddingDevice(device: string | undefined): boolean { + const provider = this.embeddingProvider as { retargetDevice?: (d: string | undefined) => boolean }; + return provider.retargetDevice?.(device) ?? false; + } + async embedBacklog(onProgress?: (embedded: number, total: number) => void): Promise { await this.whenScanSettled(); // No `isReady` check and no degrading to keyword: `embedBatch` loads the diff --git a/src/handoff/types.ts b/src/handoff/types.ts index a86d072a..2b420cd1 100644 --- a/src/handoff/types.ts +++ b/src/handoff/types.ts @@ -156,6 +156,13 @@ export interface SessionService { * between commands to work the backlog down. */ embedBacklog?(onProgress?: (embedded: number, total: number) => void): Promise; + /** + * Point the embedding provider at a device, returning whether it applied. + * + * False means the model is already loaded or loading, so the choice arrives + * too late for this session and will be picked up on the next start. + */ + retargetEmbeddingDevice?(device: string | undefined): boolean; } export interface IndexProgress { diff --git a/tests/handoff/device-retarget.test.ts b/tests/handoff/device-retarget.test.ts new file mode 100644 index 00000000..19313fe4 --- /dev/null +++ b/tests/handoff/device-retarget.test.ts @@ -0,0 +1,57 @@ +/** + * Calibration has to be able to take effect in the session that ran it. + * + * The provider loads its model lazily, so until something embeds, pointing it + * at a different device costs nothing. That is what makes full automation + * possible: the server starts, finds a backlog and no verdict, measures the + * devices in child processes, and points the not-yet-loaded provider at the + * winner before anything embeds. + * + * Without this the verdict only applied to the NEXT session, and the gate that + * decides whether to drain in the background is computed from the old device's + * rate — so a large history on a CPU stayed above the budget, never drained, + * and therefore never benefited from the calibration it had just paid for. + * + * The refusal matters as much as the retarget. Once the model is loaded or + * loading, swapping the device means discarding it and paying the load again, + * possibly while a tool call is waiting on it, to speed up work already in + * flight. The caller is told it did not apply rather than left to assume it did. + */ +import { describe, expect, it } from "vitest"; +import { TransformersEmbeddingProvider } from "@xtctx/handoff/embeddings"; + +describe("retargeting an embedding provider", () => { + it("applies while the model has not been loaded", () => { + const provider = new TransformersEmbeddingProvider(); + expect(provider.device).toBeUndefined(); + + expect(provider.retargetDevice("dml")).toBe(true); + expect(provider.device).toBe("dml"); + }); + + it("replaces a device chosen earlier, so a re-measure wins", () => { + const provider = new TransformersEmbeddingProvider(undefined, undefined, "cpu"); + + expect(provider.retargetDevice("webgpu")).toBe(true); + expect(provider.device).toBe("webgpu"); + }); + + it("refuses once the model is loading, and leaves the device alone", async () => { + const provider = new TransformersEmbeddingProvider(undefined, undefined, "cpu"); + // `warm()` starts the load without waiting for it, which is exactly the + // window this guards: a tool call arriving mid-calibration. + provider.warm(); + + expect(provider.retargetDevice("dml")).toBe(false); + expect(provider.device).toBe("cpu"); + }); + + it("reports the device it will actually load on", () => { + // `xtctx status` reads this off the provider rather than off the + // calibration cache, so it has to follow a retarget. + const provider = new TransformersEmbeddingProvider(); + provider.retargetDevice("dml"); + + expect(provider.device).toBe("dml"); + }); +}); From eba3b931097a8bb6cc4312d463beab54b388464a Mon Sep 17 00:00:00 2001 From: Felix Stubner Date: Wed, 23 Sep 2026 23:16:27 +0100 Subject: [PATCH 30/34] =?UTF-8?q?fix:=20two=20merge=20blockers=20from=20th?= =?UTF-8?q?e=20audit=20=E2=80=94=20an=20invalid=20release.yml,=20and=20bra?= =?UTF-8?q?nch=5Ffilter?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit release.yml was invalid on this branch. A comment inside a `run:` block mentioned the Actions expression syntax as an empty `${{ }}` pair, and Actions evaluates expressions in `run:` strings, shell comments included: "An expression was expected". Merging would have left the only workflow that cuts releases impossible to dispatch. It was also the cause of the zero-job failed `release` run on every push to this branch, which I twice dismissed as noise. CI stayed green because CI never reads release.yml. `scripts/check-workflows.mjs` now checks every workflow parses and that every expression opened in a string value is closed and non-empty — YAML comments excluded, since Actions never evaluates those, which is exactly the distinction that was missed. Confirmed to flag the broken file at f28ec43. It runs in CI's checks job alongside `check:review-coverage`, which had also only ever run inside `verify:release`. `branch_filter` rejected every real branch name. The unknown-tool-id check added two days ago went into `validatedFilter`, which both filters share, so `branch_filter: ["main"]` failed with "unknown tool id \"main\"". The only branch test passed a bare string, which the array check rejects for its own reason, so the suite stayed green. The id check is now `validatedToolFilter`, called only for `tool_filter`, and a test passes a real branch array through the manifest handler — confirmed to fail when branch_filter is routed back through the id check. Also drops `prepack` (npm runs `prepare` before pack and publish anyway) and the explicit build before `npm pack --dry-run`, which together built the package three times per `verify:release`. --- .github/workflows/ci.yml | 9 +++ .github/workflows/release.yml | 7 +- package.json | 6 +- scripts/check-workflows.mjs | 95 +++++++++++++++++++++++++++ src/mcp/tools/manifest.ts | 4 +- src/mcp/tools/sessions.ts | 27 ++++++-- tests/mcp/manifest.test.ts | 16 +++++ tests/mcp/tool-filter-ids.test.ts | 14 ++-- tests/release/check-workflows.test.ts | 53 +++++++++++++++ 9 files changed, 213 insertions(+), 18 deletions(-) create mode 100644 scripts/check-workflows.mjs create mode 100644 tests/release/check-workflows.test.ts diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 912e18a2..85ad4ad8 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -139,6 +139,15 @@ jobs: npm run security:checklist npm run audit:production + # Both used to run only inside `verify:release`, which only the release + # workflow invokes — so a broken workflow file or an unmapped file was + # found at release time, if at all. An invalid release.yml sat on a + # branch for two days with this job green. + - name: Workflow files and review coverage + run: | + npm run check:workflows + npm run check:review-coverage + - name: Landing dependency audit # The landing site has only devDependencies (Astro is a build-time # dependency of a static site), so --omit=dev would audit nothing. diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 6d5f10dc..f0fa8735 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -204,8 +204,11 @@ jobs: git tag "$TAG" # The token reaches git here and nowhere else; see # `persist-credentials: false` on the checkout. Through the - # environment rather than `${{ }}`, so it is never substituted into - # the script text. + # environment rather than an Actions expression, so it is never + # substituted into the script text. (Do not write the expression + # syntax in this comment: Actions evaluates it inside `run:` blocks + # even in shell comments, and an empty one made this whole file + # invalid — see scripts/check-workflows.mjs.) git push "https://x-access-token:${RELEASE_TOKEN}@github.com/${GITHUB_REPOSITORY}.git" "HEAD:${GITHUB_REF_NAME}" git push "https://x-access-token:${RELEASE_TOKEN}@github.com/${GITHUB_REPOSITORY}.git" "$TAG" diff --git a/package.json b/package.json index c26f3b11..81e7e7c7 100644 --- a/package.json +++ b/package.json @@ -30,9 +30,8 @@ "landing:build": "npm --prefix landing run build", "landing:preview": "npm --prefix landing run preview", "prepare": "npm run build", - "prepack": "npm run build", "smoke:cli": "node dist/src/cli/index.js --help", - "verify:release": "npm run lint && npm run typecheck && npm test && npm run test:security && npm run security:checklist && npm run check:review-coverage && npm run audit:production && npm --prefix landing audit && npm run test:drift && npm run test:integration && npm run test:smoke && npm run test:eval && npm run build && npm pack --dry-run && npm run demo:public && npm run landing:build && npm run smoke:cli", + "verify:release": "npm run lint && npm run typecheck && npm test && npm run test:security && npm run security:checklist && npm run check:review-coverage && npm run check:workflows && npm run audit:production && npm --prefix landing audit && npm run test:drift && npm run test:integration && npm run test:smoke && npm run test:eval && npm pack --dry-run && npm run demo:public && npm run landing:build && npm run smoke:cli", "sync:version": "node scripts/sync-version.mjs", "version": "node scripts/sync-version.mjs", "test:eval": "vitest run tests/eval", @@ -40,7 +39,8 @@ "capture:formats": "node scripts/capture-format-fingerprint.mjs", "check:upstream": "node scripts/check-upstream-versions.mjs", "typecheck": "tsc --noEmit -p tsconfig.test.json", - "check:review-coverage": "node ./scripts/check-review-coverage.mjs" + "check:review-coverage": "node ./scripts/check-review-coverage.mjs", + "check:workflows": "node ./scripts/check-workflows.mjs" }, "dependencies": { "@huggingface/transformers": "^4.2.0", diff --git a/scripts/check-workflows.mjs b/scripts/check-workflows.mjs new file mode 100644 index 00000000..05e328e5 --- /dev/null +++ b/scripts/check-workflows.mjs @@ -0,0 +1,95 @@ +#!/usr/bin/env node +/** + * Fail if a GitHub Actions workflow file would be rejected by GitHub itself. + * + * Nothing in this repository checked the workflow files, and the gap was not + * theoretical. A comment inside `release.yml`'s `run:` block mentioned the + * expression syntax as an empty `${{ }}` pair. Actions evaluates expressions + * inside `run:` strings — shell comments included — so the empty one made the + * whole file invalid ("An expression was expected"). CI stayed green, because + * CI never looks at release.yml; the only symptom was a zero-job failed run on + * every push, which read as noise. Merged, it would have left the one workflow + * that cuts releases impossible to dispatch. + * + * This checks the two things that failure needed, not a full Actions schema: + * every workflow parses as YAML, and every expression opened in any string + * value is closed and non-empty. YAML comments are not checked, because they + * are not strings and Actions never sees them — only text inside a value is + * evaluated, which is exactly the distinction that was missed. + * + * node scripts/check-workflows.mjs + */ +import { readdirSync, readFileSync } from "node:fs"; +import { join } from "node:path"; +import { pathToFileURL } from "node:url"; +import { parse } from "yaml"; + +const WORKFLOW_DIR = ".github/workflows"; + +/** Problems in one workflow's text, as human-readable strings. */ +export function checkWorkflowText(text, name = "workflow") { + let document; + try { + document = parse(text); + } catch (error) { + return [`${name}: does not parse as YAML: ${error instanceof Error ? error.message : String(error)}`]; + } + + const problems = []; + const visit = (value, path) => { + if (typeof value === "string") { + for (const problem of expressionProblems(value)) { + problems.push(`${name}: ${path}: ${problem}`); + } + return; + } + if (Array.isArray(value)) { + value.forEach((item, index) => visit(item, `${path}[${index}]`)); + return; + } + if (value && typeof value === "object") { + for (const [key, item] of Object.entries(value)) { + visit(item, path ? `${path}.${key}` : key); + } + } + }; + visit(document, ""); + return problems; +} + +function expressionProblems(value) { + const problems = []; + let from = 0; + for (;;) { + const open = value.indexOf("${{", from); + if (open === -1) break; + const close = value.indexOf("}}", open + 3); + if (close === -1) { + problems.push(`unclosed expression starting ${JSON.stringify(value.slice(open, open + 30))}`); + break; + } + if (value.slice(open + 3, close).trim().length === 0) { + problems.push("empty expression — Actions evaluates this even inside a shell comment"); + } + from = close + 2; + } + return problems; +} + +export function checkWorkflowDir(dir = WORKFLOW_DIR) { + const problems = []; + for (const file of readdirSync(dir).filter((entry) => /\.ya?ml$/.test(entry))) { + problems.push(...checkWorkflowText(readFileSync(join(dir, file), "utf-8"), file)); + } + return problems; +} + +if (process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href) { + const problems = checkWorkflowDir(); + if (problems.length > 0) { + process.stderr.write(`${problems.join("\n")}\n`); + process.exitCode = 1; + } else { + process.stdout.write("Every workflow parses and every expression is closed and non-empty.\n"); + } +} diff --git a/src/mcp/tools/manifest.ts b/src/mcp/tools/manifest.ts index e14cdd39..7886429b 100644 --- a/src/mcp/tools/manifest.ts +++ b/src/mcp/tools/manifest.ts @@ -1,5 +1,5 @@ import type { SessionService, SessionSummary } from "../../handoff/types.js"; -import { indexingPayload, ToolInputError, validatedFilter } from "./sessions.js"; +import { indexingPayload, ToolInputError, validatedFilter, validatedToolFilter } from "./sessions.js"; import { inlineSafe } from "../../utils/untrusted-text.js"; interface HandoffManifestParams { @@ -38,7 +38,7 @@ export function createHandoffManifestHandler(service: SessionService) { const limit = normalizeLimit(params.limit, DEFAULT_LIMIT); selected = await service.listRecentSessions( limit, - validatedFilter(params.tool_filter, "tool_filter"), + validatedToolFilter(params.tool_filter, "tool_filter"), validatedFilter(params.branch_filter, "branch_filter"), ); missingRefs = []; diff --git a/src/mcp/tools/sessions.ts b/src/mcp/tools/sessions.ts index 6edb0212..8f503219 100644 --- a/src/mcp/tools/sessions.ts +++ b/src/mcp/tools/sessions.ts @@ -61,6 +61,25 @@ export function validatedFilter(value: unknown, field: string): string[] | undef throw new ToolInputError(`${field} must contain only non-empty strings`); } + return value as string[]; +} + +/** + * `validatedFilter`, plus: every entry must name a tool this build knows. + * + * A separate function, not a check inside the shared one. It was first written + * into `validatedFilter`, which `branch_filter` also goes through — so every + * real branch name was rejected as "unknown tool id \"main\"" and branch + * filtering became unreachable over MCP, with the whole suite green because no + * test passed a well-formed branch array. The two filters share a shape and + * nothing else. + */ +export function validatedToolFilter(value: unknown, field: string): string[] | undefined { + const filter = validatedFilter(value, field); + if (filter === undefined) { + return undefined; + } + // An id that names no tool is rejected, not filtered on. // // The filter reaches SQLite as `WHERE tool IN (...)`, so an unrecognised id @@ -73,7 +92,7 @@ export function validatedFilter(value: unknown, field: string): string[] | undef // back can fix its own call; one that gets an empty result cannot tell a // wrong id from an empty index. const known = new Set(SUPPORTED_TOOLS.map((tool) => tool.id)); - const unknown = (value as string[]).filter((item) => !known.has(item.trim())); + const unknown = filter.filter((item) => !known.has(item.trim())); if (unknown.length > 0) { throw new ToolInputError( `${field} contains unknown tool id${unknown.length === 1 ? "" : "s"} ` + @@ -82,7 +101,7 @@ export function validatedFilter(value: unknown, field: string): string[] | undef ); } - return value as string[]; + return filter; } export function createRecentSessionsHandler(service: SessionService) { @@ -92,7 +111,7 @@ export function createRecentSessionsHandler(service: SessionService) { const format = params.format ?? "markdown"; const sessions = await service.listRecentSessions( limit, - validatedFilter(params.tool_filter, "tool_filter"), + validatedToolFilter(params.tool_filter, "tool_filter"), validatedFilter(params.branch_filter, "branch_filter"), ); @@ -135,7 +154,7 @@ export function createSearchSessionsHandler(service: SessionService) { const sessions = await service.searchSessions( query, limit, - validatedFilter(params.tool_filter, "tool_filter"), + validatedToolFilter(params.tool_filter, "tool_filter"), mode, validatedFilter(params.branch_filter, "branch_filter"), ); diff --git a/tests/mcp/manifest.test.ts b/tests/mcp/manifest.test.ts index fa1e0102..73d4f68f 100644 --- a/tests/mcp/manifest.test.ts +++ b/tests/mcp/manifest.test.ts @@ -119,6 +119,7 @@ describe("xtctx_handoff_manifest", () => { /** Records what the handler actually asked the index for. */ class FilterRecordingService extends LimitHonoringService { lastToolFilter: string[] | undefined = undefined; + lastBranchFilter: string[] | undefined = undefined; called = false; constructor() { @@ -128,9 +129,11 @@ class FilterRecordingService extends LimitHonoringService { override async listRecentSessions( _limit: number, toolFilter?: string[], + branchFilter?: string[], ): Promise { this.called = true; this.lastToolFilter = toolFilter; + this.lastBranchFilter = branchFilter; return []; } } @@ -182,6 +185,19 @@ describe("manifest filter arguments that are not arrays of strings", () => { expect(service.lastToolFilter).toEqual(["codex"]); }); + + it("passes a real branch name through, rather than checking it against tool ids", async () => { + // The tool-id check was once written into the validator both filters + // share, so `branch_filter: ["main"]` failed with "unknown tool id + // \"main\"" and branch filtering was unreachable over MCP. The only + // branch test above passed a bare string, which the array check rejects + // for its own reason — so nothing noticed. + const service = new FilterRecordingService(); + + await createHandoffManifestHandler(service)({ branch_filter: ["main", "feat/x"] }); + + expect(service.lastBranchFilter).toEqual(["main", "feat/x"]); + }); }); /** Reports a scan still in flight, the way the real service does mid-index. */ diff --git a/tests/mcp/tool-filter-ids.test.ts b/tests/mcp/tool-filter-ids.test.ts index 6266086d..43ef7781 100644 --- a/tests/mcp/tool-filter-ids.test.ts +++ b/tests/mcp/tool-filter-ids.test.ts @@ -11,26 +11,26 @@ * and the MCP schema advertised only `items: { type: "string" }`. */ import { describe, expect, it } from "vitest"; -import { validatedFilter } from "@xtctx/mcp/tools/sessions"; +import { validatedToolFilter } from "@xtctx/mcp/tools/sessions"; import { SUPPORTED_TOOLS } from "@xtctx/tools/sources"; -describe("validatedFilter", () => { +describe("validatedToolFilter", () => { it("accepts every id the tool registry defines", () => { const ids = SUPPORTED_TOOLS.map((tool) => tool.id); - expect(validatedFilter(ids, "tool_filter")).toEqual(ids); + expect(validatedToolFilter(ids, "tool_filter")).toEqual(ids); }); it("rejects the natural wrong guesses instead of matching nothing", () => { for (const guess of ["claude", "gemini", "vscode"]) { - expect(() => validatedFilter([guess], "tool_filter")).toThrow(/unknown tool id/); + expect(() => validatedToolFilter([guess], "tool_filter")).toThrow(/unknown tool id/); } }); it("names the valid ids in the error, so the caller can fix its own call", () => { let message = ""; try { - validatedFilter(["claude"], "tool_filter"); + validatedToolFilter(["claude"], "tool_filter"); } catch (error) { message = error instanceof Error ? error.message : String(error); } @@ -40,7 +40,7 @@ describe("validatedFilter", () => { }); it("still allows no filter at all", () => { - expect(validatedFilter(undefined, "tool_filter")).toBeUndefined(); - expect(validatedFilter(null, "tool_filter")).toBeUndefined(); + expect(validatedToolFilter(undefined, "tool_filter")).toBeUndefined(); + expect(validatedToolFilter(null, "tool_filter")).toBeUndefined(); }); }); diff --git a/tests/release/check-workflows.test.ts b/tests/release/check-workflows.test.ts new file mode 100644 index 00000000..d761c56b --- /dev/null +++ b/tests/release/check-workflows.test.ts @@ -0,0 +1,53 @@ +/** + * The workflow files have to be ones GitHub will actually run. + * + * A comment inside `release.yml`'s `run:` block contained an empty Actions + * expression. Actions evaluates expressions in `run:` strings, comments + * included, so the file became invalid; CI never looks at release.yml, so + * everything stayed green and merging would have left releases impossible to + * dispatch. See scripts/check-workflows.mjs. + */ +import { describe, expect, it } from "vitest"; +// @ts-expect-error -- plain JS with JSDoc types, imported for its exports. +import { checkWorkflowDir, checkWorkflowText } from "../../scripts/check-workflows.mjs"; + +const check = checkWorkflowText as (text: string, name?: string) => string[]; + +describe("workflow checks", () => { + it("passes every workflow in this repository", () => { + expect((checkWorkflowDir as () => string[])()).toEqual([]); + }); + + it("rejects an empty expression inside a run block's shell comment", () => { + // The exact shape that broke release.yml. + const text = [ + "on: workflow_dispatch", + "jobs:", + " cut:", + " runs-on: ubuntu-latest", + " steps:", + " - run: |", + " # through the environment rather than `${{ }}`", + " echo hi", + "", + ].join("\n"); + + expect(check(text, "release.yml").join("\n")).toMatch(/empty expression/); + }); + + it("ignores a YAML comment, which Actions never evaluates", () => { + const text = ["# `${{ }}` in a real YAML comment is fine", "on: push", "jobs: {}", ""].join("\n"); + + expect(check(text)).toEqual([]); + }); + + it("rejects an expression that is never closed", () => { + const text = ["on: push", "jobs:", " a:", " runs-on: ${{ matrix.os", ""].join("\n"); + + expect(check(text).join("\n")).toMatch(/unclosed expression/); + }); + + it("rejects a file that does not parse", () => { + expect(check("on: [push\n").join("\n")).toMatch(/does not parse/); + }); +}); From b219f76a22c53e26eb10c4916ab2672983e0e712 Mon Sep 17 00:00:00 2001 From: Felix Stubner Date: Wed, 23 Sep 2026 23:31:01 +0100 Subject: [PATCH 31/34] fix(calibrate): apply the verdict in the session that measured it, and make it safe MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The audit found that same-session calibration — which I described as done — never applied in the MCP server. Every scan ends by calling `warm()`, so by the time calibration finished the model was already loading and `retargetDevice` refused. The verdict only ever reached the NEXT session, while the README and two docstrings said otherwise. The unit test called the provider in isolation and never exercised that ordering. The provider now defers its model load until the device is known (`deferDeviceUntil`), and the server starts calibration before it scans. A deferral cannot lose the race: whoever asks for the model first — the warm scan, a hybrid search's `warm()`, an explicit vector search — gets one load, on the chosen device. Hybrid search is unaffected in practice because it already answers from keyword while the model is not ready; only an explicit vector search waits, and only on a machine's first run. Because calibration now finishes before the scan, the scan's own vectorizing pass records its per-segment rate on the chosen device — a real measurement on the real path — and the drain budget is judged by that. The previous version substituted calibration's own rate, the fastest of three passes over uniform 1000-character segments, which is systematically quicker than real windows. Made safe for several agents starting at once on a fresh machine: - a machine-wide lock (`~/.xtctx/device.json.lock`, abandoned after ten minutes), so one server calibrates and the rest use the default for that session instead of all loading the model together and contaminating each other's timings; - the verdict written to a temp file and renamed, so a concurrent reader sees the old file or the new one, never half of one; - calibration workers killed when the server exits. It leaves by `process.exit` two seconds after its client disconnects; on Windows a child outlives its parent, and the timeout that would have stopped it lived in the parent. The background path moved out of `cli/index.ts` into `runtime/background.ts`, because the CLI runs `main()` on import and nothing in it could be tested — which is why three mutations to this logic survived the whole suite. It now has tests for the ordering, the embeddings-disabled guard, the budget in both directions, the lock, and failure reporting. A failed background drain is now logged to stderr and recorded in `embedding_error` instead of vanishing into a bare `catch`. Two more test gaps from the same sweep, one of which was hiding a bug: - `formatDuration` rounded minutes and seconds separately and printed "1m 60s" for 119.6 seconds. Swapping its rounding for flooring changed nothing any test could see. Now rounded once and split. - Hybrid mode's keyword half could be zeroed without failing anything; a ranking contract now needs it. The batch width is pinned by behaviour (sixteen per forward pass), through a pipeline-factory seam that lets tests see what the model is asked to do without downloading it. The remaining sweep survivor, the tool-level `limit` cap, is an equivalent mutant: the index caps every path at 100 independently, so no behaviour depends on it. --- src/cli/index.ts | 130 +---------------- src/handoff/device.ts | 124 +++++++++++++--- src/handoff/embeddings.ts | 70 ++++++--- src/handoff/sqlite-index.ts | 41 ++++-- src/handoff/types.ts | 9 +- src/runtime/background.ts | 136 ++++++++++++++++++ src/utils/duration.ts | 8 +- tests/handoff/device-retarget.test.ts | 103 +++++++++----- tests/handoff/embedding-batch.test.ts | 31 ++++ tests/handoff/ranking-contract.test.ts | 26 ++++ tests/runtime/background.test.ts | 189 +++++++++++++++++++++++++ tests/utils/duration.test.ts | 4 + 12 files changed, 647 insertions(+), 224 deletions(-) create mode 100644 src/runtime/background.ts create mode 100644 tests/handoff/embedding-batch.test.ts create mode 100644 tests/runtime/background.test.ts diff --git a/src/cli/index.ts b/src/cli/index.ts index bd1902ad..fef62ad9 100644 --- a/src/cli/index.ts +++ b/src/cli/index.ts @@ -8,10 +8,8 @@ import { runSetup } from "./setup.js"; import { runStatus } from "./status.js"; import { createProjectServices } from "../runtime/services.js"; import { startMcpServer } from "../mcp/server.js"; -import type { SessionService } from "../handoff/types.js"; import { readXtctxPackage } from "../utils/package-info.js"; -import { BACKGROUND_EMBED_BUDGET_MS, estimateVectorBacklog } from "../utils/duration.js"; -import { calibrateEmbeddingDevice, readDeviceVerdict } from "../handoff/device.js"; +import { runBackgroundWork } from "../runtime/background.js"; const { version: CLI_VERSION } = readXtctxPackage(import.meta.url); @@ -99,7 +97,7 @@ export async function main(argv = process.argv): Promise { // incremental scan this costs measured 9.7s in the background against a // 19GB Codex store, and the cursor design keeps it from re-reading. if (!unconfiguredProjectRoot && !services.config.error) { - void warmIndex(services.sessions); + void runBackgroundWork({ sessions: services.sessions }); } return; } @@ -243,130 +241,6 @@ export async function main(argv = process.argv): Promise { await program.parseAsync(argv); } -/** - * Start a scan and let it run. - * - * `listRecentSessions` is the read that starts a scan; its result is not - * wanted here, and the budget it waits on is short. A failure is not the - * server's problem to report at startup: the same scan runs again on the - * first call, where its error is recorded against the tool. - */ -async function warmIndex(sessions: SessionService): Promise { - try { - await sessions.listRecentSessions(1); - await sessions.whenScanSettled?.(); - await drainVectorsIfAffordable(sessions); - } catch { - // Deliberately silent; see above. - } -} - -/** - * Work the vector backlog down in the background, when it is cheap enough. - * - * Until this existed, nothing ever finished embedding a real history. Searches - * vectorize about sixteen windows per call, which by this project's own - * measurement leaves a 9,232-window project needing on the order of 570 - * searches — so semantic search was keyword-only in practice while still - * paying six seconds a call for the privilege. The only way out was knowing to - * run `xtctx scan --embed` by hand, which is not something a user should have - * to know. - * - * Three comments in this repository claimed the session-start hook launched a - * detached scan that did this, and the hook has not done so since #323 on - * 2026-09-02, which moved the warm-up here — to the server — on the same day - * #322 added it. The comments outlived the code they described by nineteen - * days. - * - * (An earlier version of this comment said no such code had ever existed. - * That was wrong: `launchDetachedScan` was real, in `src/cli/hook.ts`, for a - * few hours. What it was right about is that nothing launches one now.) - * - * Bounded rather than unconditional, because "embed everything in the - * background" is exactly what would make a large project on a CPU unusable. - * What changed is that device calibration made the affordable case the common - * one. - */ -async function drainVectorsIfAffordable(sessions: SessionService): Promise { - if (!sessions.embedBacklog) { - return; - } - - const status = await sessions.getStatus(); - const { remaining, etaMs } = estimateVectorBacklog( - status.retrieval_units, - status.vectorized_units, - status.vector_ms_per_unit, - { backlog: status.vector_segment_backlog, msPerSegment: status.vector_ms_per_segment }, - ); - if (remaining === 0) { - return; - } - - // Calibrate before judging affordability, not after — otherwise the - // uncalibrated state perpetuates itself. - // - // The estimate below is computed from the rate of whatever device is in use, - // which is the CPU until something calibrates. A large history on the CPU - // estimates well past the budget, so the drain is skipped; the drain was the - // only thing that would have made the project fast; and the machine stays on - // the CPU forever. The user who loses most is the one with the largest - // history, which is the one this whole mechanism is for. - // - // Only when there is a backlog to justify it, and never in front of a tool - // call — this runs detached from the server's start, alongside a drain that - // already takes minutes. - // - // It applies to THIS session where it can. The provider loads its model - // lazily, so until something embeds, pointing it at a different device costs - // nothing — and the whole point of measuring is undone if the answer only - // arrives next time. `retargetEmbeddingDevice` returns false once the model - // is loaded or loading, which is the case where the drain below is already - // running on the old device and swapping it would mean paying the load again - // to speed up work in flight. - // - // `XTCTX_DISABLE_EMBEDDINGS=1` means no model is ever loaded, and - // calibration loads one per device. - let applied = false; - let calibrated: Awaited> | undefined; - if (process.env.XTCTX_DISABLE_EMBEDDINGS !== "1" && !(await readDeviceVerdict())) { - calibrated = await calibrateEmbeddingDevice().catch(() => undefined); - if (calibrated) { - applied = sessions.retargetEmbeddingDevice?.(calibrated.device) ?? false; - } - } - - // Judge affordability by the rate that will actually apply. - // - // `etaMs` above was built from `vector_ms_per_segment`, which was measured - // on the device in use before calibration — the CPU. Keeping it would skip - // the drain on exactly the machines calibration just made fast, which is the - // trap this whole branch exists to close. - // - // The substituted rate is not a projection: calibration measured - // milliseconds per segment for this model on the chosen device, and - // `estimateVectorBacklog` multiplies a per-segment rate by the segment - // backlog. Same arithmetic, fresher measurement of the same quantity. - let effectiveEtaMs = etaMs; - if (applied) { - const fresh = await sessions.getStatus(); - const chosen = calibrated?.measured.find((row) => row.device === calibrated.device); - if (chosen?.msPerSegment != null) { - effectiveEtaMs = fresh.vector_segment_backlog * chosen.msPerSegment; - } - } - const current = { etaMs: effectiveEtaMs }; - - // No estimate means nothing has embedded yet on this machine, so there is no - // measured rate to judge affordability by. Searches still vectorize - // incrementally, which is what produces the rate this needs. - if (current.etaMs === null || current.etaMs > BACKGROUND_EMBED_BUDGET_MS) { - return; - } - - await sessions.embedBacklog(); -} - function shouldStartMcp(argv: string[]): boolean { if (argv.length > 2) { return false; diff --git a/src/handoff/device.ts b/src/handoff/device.ts index f52b3c53..ad5dde0d 100644 --- a/src/handoff/device.ts +++ b/src/handoff/device.ts @@ -1,6 +1,6 @@ -import { spawn } from "node:child_process"; +import { spawn, type ChildProcess } from "node:child_process"; import { existsSync } from "node:fs"; -import { readFile, mkdir, writeFile } from "node:fs/promises"; +import { mkdir, open, readFile, rename, stat, unlink, writeFile } from "node:fs/promises"; import { cpus, homedir } from "node:os"; import { dirname, join } from "node:path"; import { fileURLToPath } from "node:url"; @@ -93,10 +93,92 @@ export async function readDeviceVerdict(options: { } } +/** + * Written to a temporary file and renamed into place. + * + * Several MCP servers start at once on a fresh machine — one per agent — and a + * plain `writeFile` truncates before it writes, so a server reading the verdict + * mid-write saw half a file. `readDeviceVerdict` treats that as no verdict, + * which is safe but means calibrating again for nothing. A rename is atomic on + * one filesystem, so a reader sees the old file or the new one. + */ async function writeDeviceVerdict(verdict: DeviceVerdict, home?: string): Promise { const path = cachePath(home); await mkdir(dirname(path), { recursive: true }); - await writeFile(path, `${JSON.stringify(verdict, null, 2)}\n`, "utf-8"); + const temporary = `${path}.${process.pid}.tmp`; + await writeFile(temporary, `${JSON.stringify(verdict, null, 2)}\n`, "utf-8"); + await rename(temporary, path); +} + +/** + * How long a calibration lock is honoured before it is treated as abandoned. + * + * Longer than any calibration measured (about a minute, three devices, each + * with a model load), shorter than a user would notice as "it never + * calibrates". A lock older than this belongs to a process that was killed — + * the server exits two seconds after its client disconnects, whatever it was + * doing — and must not block every later attempt. + */ +const LOCK_STALE_MS = 10 * 60 * 1000; + +/** Thrown when another process is already calibrating this machine. */ +export class CalibrationBusyError extends Error { + constructor() { + super("another process is already calibrating this machine"); + } +} + +/** + * Take the machine-wide calibration lock, or throw `CalibrationBusyError`. + * + * One per machine, not per project, because the verdict is per machine. + * Without it, every agent that starts an MCP server on a fresh machine + * calibrates at once: three servers, each timing up to three devices, all + * loading the model together — and each one's CPU arm timed while the others + * compete for the same cores, the measurement contaminated by the act of + * measuring. + */ +async function acquireCalibrationLock(home?: string): Promise<() => Promise> { + const path = `${cachePath(home)}.lock`; + await mkdir(dirname(path), { recursive: true }); + for (let attempt = 0; attempt < 2; attempt += 1) { + try { + const handle = await open(path, "wx"); + await handle.writeFile(String(process.pid)); + await handle.close(); + return () => unlink(path).catch(() => {}); + } catch (error) { + if ((error as NodeJS.ErrnoException).code !== "EEXIST") throw error; + const modified = await stat(path).catch(() => null); + if (modified && Date.now() - modified.mtimeMs < LOCK_STALE_MS) { + throw new CalibrationBusyError(); + } + await unlink(path).catch(() => {}); + } + } + throw new CalibrationBusyError(); +} + +/** + * Calibration workers still running, so they can be killed on exit. + * + * The MCP server leaves by `process.exit` two seconds after its client goes. + * A spawned child is not killed with its parent on Windows, and its timeout + * timer lived in the parent — so a session that ended mid-calibration left a + * worker loading and timing a model with nothing to report to. + */ +const liveWorkers = new Set(); +let exitHookInstalled = false; + +function trackWorker(child: ChildProcess): void { + liveWorkers.add(child); + child.once("close", () => liveWorkers.delete(child)); + if (!exitHookInstalled) { + exitHookInstalled = true; + process.once("exit", () => { + for (const worker of liveWorkers) worker.kill(); + }); + } } /** Text at the length real segments have; see the note in the worker. */ @@ -203,22 +285,27 @@ export async function calibrateEmbeddingDevice(options: { const timeoutMs = options.timeoutMs ?? 5 * 60 * 1000; const measured: DeviceVerdict["measured"] = []; - for (const device of deviceCandidates()) { - options.onProgress?.(device); - const result = await timeDevice(device, segmentCount, timeoutMs); - measured.push({ device, ...result }); - } + const release = await acquireCalibrationLock(options.home); + try { + for (const device of deviceCandidates()) { + options.onProgress?.(device); + const result = await timeDevice(device, segmentCount, timeoutMs); + measured.push({ device, ...result }); + } - const verdict: DeviceVerdict = { - device: chooseDevice(measured), - measured, - fingerprint: deviceFingerprint(), - measuredAt: new Date().toISOString(), - }; - // Best-effort: an unwritable home directory means the calibration is paid - // again next time, not that it fails now. - await writeDeviceVerdict(verdict, options.home).catch(() => {}); - return verdict; + const verdict: DeviceVerdict = { + device: chooseDevice(measured), + measured, + fingerprint: deviceFingerprint(), + measuredAt: new Date().toISOString(), + }; + // Best-effort: an unwritable home directory means the calibration is paid + // again next time, not that it fails now. + await writeDeviceVerdict(verdict, options.home).catch(() => {}); + return verdict; + } finally { + await release(); + } } /** @@ -254,6 +341,7 @@ function timeDevice( [...workerArgv(), `--device=${device}`, `--segments=${segmentCount}`], { stdio: ["ignore", "pipe", "pipe"] }, ); + trackWorker(child); let out = ""; const timer = setTimeout(() => { diff --git a/src/handoff/embeddings.ts b/src/handoff/embeddings.ts index 66ae8f70..335e5419 100644 --- a/src/handoff/embeddings.ts +++ b/src/handoff/embeddings.ts @@ -209,7 +209,7 @@ type FeatureExtractionPipeline = ( options: { pooling: "mean"; normalize: boolean }, ) => Promise; -type PipelineFactory = ( +export type PipelineFactory = ( task: "feature-extraction", model: string, options?: Record, @@ -248,6 +248,13 @@ export class TransformersEmbeddingProvider implements EmbeddingProvider { * catastrophic is also the one where it does not fail. */ device?: string, + /** + * @internal For tests. Replaces the dynamic import of the real pipeline, + * so a test can observe which device a load was asked for without + * downloading and running a 130MB model. Which device the model actually + * loads on is exactly what went untested while calibration never applied. + */ + private readonly loadPipeline?: PipelineFactory, ) { this.model = model; this.deviceName = device; @@ -260,29 +267,36 @@ export class TransformersEmbeddingProvider implements EmbeddingProvider { } /** - * Point this provider at a device, if it is not too late to matter. + * Hold the model load until the device is known. + * + * Calibration takes about a minute and the model loads lazily, but "lazily" + * is not "late enough". An earlier version let calibration retarget the + * provider only if nothing had started loading yet — and the server's own + * warm scan always started loading first (every scan ends by calling + * `warm()`), so the retarget was refused on every run. The verdict never + * applied to the session that paid for it, while the comments and the README + * said it did. * - * The device is only read when the pipeline loads, and loading is lazy, so - * until then changing it is free. This exists so calibration can take effect - * in the session that ran it rather than the next one: the server starts, - * finds no verdict and a backlog worth draining, measures the devices in - * child processes, and points the not-yet-loaded provider at the winner - * before anything embeds. + * Deferral cannot lose that race. Whoever asks for the model first — the warm + * scan, a hybrid search's `warm()`, an explicit vector search — gets a load + * that waits for the device, then loads once, on the right one. Hybrid search + * is unaffected in practice: it answers from keyword while `isReady()` is + * false, which it already did while a model downloads. Only an explicit + * `vector` request waits, and only on the first run on a machine. * - * Refuses once the model is loaded or loading, and says so by returning - * false. Swapping the device under a loaded pipeline would mean discarding - * it and paying the load again, possibly while a tool call is waiting on it, - * to save time on work that is already running. The caller reports the - * verdict as taking effect next session instead. + * A device that resolves to undefined, or a rejected promise, leaves the + * device as it was: a failed calibration is not a reason to stop using the + * one that has always worked. */ - retargetDevice(device: string | undefined): boolean { - if (this.extractor !== null || this.loading !== null) { - return false; - } - this.deviceName = device; - return true; + deferDeviceUntil(device: Promise): void { + // Handled here rather than at the load, which may never happen: a process + // that exits before embedding anything would otherwise report a failed + // calibration as an unhandled rejection. + this.devicePending = device.catch(() => undefined); } + private devicePending: Promise | null = null; + async embed(text: string): Promise { const [vector] = await this.embedBatch([text]); return vector; @@ -347,12 +361,22 @@ export class TransformersEmbeddingProvider implements EmbeddingProvider { } private async loadExtractor(): Promise { + if (this.devicePending) { + const pending = this.devicePending; + this.devicePending = null; + const device = await pending.catch(() => undefined); + if (device !== undefined) { + this.deviceName = device; + } + } + process.stderr.write(`xtctx: Initializing local embedding provider (${this.model})...\n`); - const transformers = (await import("@huggingface/transformers")) as unknown as { - pipeline: PipelineFactory; - }; - const extractor = await transformers.pipeline("feature-extraction", this.model, { + const pipeline = + this.loadPipeline ?? + ((await import("@huggingface/transformers")) as unknown as { pipeline: PipelineFactory }) + .pipeline; + const extractor = await pipeline("feature-extraction", this.model, { dtype: this.dtype, ...(this.deviceName === undefined ? {} : { device: this.deviceName }), }); diff --git a/src/handoff/sqlite-index.ts b/src/handoff/sqlite-index.ts index c57353d3..851a84a7 100644 --- a/src/handoff/sqlite-index.ts +++ b/src/handoff/sqlite-index.ts @@ -1052,10 +1052,12 @@ export class SqliteHandoffIndex implements SessionService { * background embedding is what would make a large project on a CPU * unusable. */ - /** See `SessionService.retargetEmbeddingDevice`. */ - retargetEmbeddingDevice(device: string | undefined): boolean { - const provider = this.embeddingProvider as { retargetDevice?: (d: string | undefined) => boolean }; - return provider.retargetDevice?.(device) ?? false; + /** See `SessionService.deferEmbeddingDeviceUntil`. */ + deferEmbeddingDeviceUntil(device: Promise): void { + const provider = this.embeddingProvider as { + deferDeviceUntil?: (d: Promise) => void; + }; + provider.deferDeviceUntil?.(device); } async embedBacklog(onProgress?: (embedded: number, total: number) => void): Promise { @@ -1063,13 +1065,30 @@ export class SqliteHandoffIndex implements SessionService { // No `isReady` check and no degrading to keyword: `embedBatch` loads the // model itself and this command has nothing else it could be asking for, // so it waits however long that takes. - return ensureVectors({ - db: this.getDb(), - embeddingProvider: this.embeddingProvider, - filters: [], - vectorBudgetMs: 0, - onProgress, - }); + // + // A failure is recorded where `xtctx status` and `xtctx_continuity_status` + // already look. The MCP server runs this in the background at startup with + // nobody waiting on it, and its only caller used to swallow the error — so + // an endpoint rejecting every call, or a model that could not load, left + // the backlog frozen with no reason given anywhere. + try { + const remaining = await ensureVectors({ + db: this.getDb(), + embeddingProvider: this.embeddingProvider, + filters: [], + vectorBudgetMs: 0, + onProgress, + }); + clearSetting(this.getDb(), "last_error:embeddings"); + return remaining; + } catch (error) { + setSetting( + this.getDb(), + "last_error:embeddings", + error instanceof Error ? error.message : String(error), + ); + throw error; + } } private async ensureVectors(toolFilter?: string[]): Promise { diff --git a/src/handoff/types.ts b/src/handoff/types.ts index 2b420cd1..6c308996 100644 --- a/src/handoff/types.ts +++ b/src/handoff/types.ts @@ -157,12 +157,13 @@ export interface SessionService { */ embedBacklog?(onProgress?: (embedded: number, total: number) => void): Promise; /** - * Point the embedding provider at a device, returning whether it applied. + * Make the embedding model's first load wait for this device. * - * False means the model is already loaded or loading, so the choice arrives - * too late for this session and will be picked up on the next start. + * For calibration running alongside the index: whichever caller asks for the + * model first gets a load that waits for the verdict, so the session that + * measured the device is the one that uses it. */ - retargetEmbeddingDevice?(device: string | undefined): boolean; + deferEmbeddingDeviceUntil?(device: Promise): void; } export interface IndexProgress { diff --git a/src/runtime/background.ts b/src/runtime/background.ts new file mode 100644 index 00000000..d3933890 --- /dev/null +++ b/src/runtime/background.ts @@ -0,0 +1,136 @@ +import { + CalibrationBusyError, + calibrateEmbeddingDevice, + readDeviceVerdict, + type DeviceVerdict, +} from "../handoff/device.js"; +import type { SessionService } from "../handoff/types.js"; +import { BACKGROUND_EMBED_BUDGET_MS, estimateVectorBacklog } from "../utils/duration.js"; + +/** + * What the MCP server does in the background after it starts, with nobody + * waiting on it. + * + * Lives here rather than in `cli/index.ts` so it can be tested. That file runs + * `main()` on import, so nothing in it could be exercised by a test, and three + * mutations to this exact logic — ignoring the drain budget, calibrating with + * embeddings disabled, and the ordering that stopped calibration applying at + * all — survived the whole suite. + */ +export interface BackgroundDeps { + sessions: SessionService; + readVerdict?: () => Promise; + calibrate?: () => Promise; + env?: NodeJS.ProcessEnv; + /** Where background failures are reported; stderr in the real server. */ + log?: (line: string) => void; +} + +/** + * Calibrate if this machine never has, scan, then drain the vector backlog if + * it fits the budget. + * + * ORDER IS THE WHOLE POINT. Calibration is started before the scan and handed + * to the index as a promise the model's first load waits on. An earlier + * version calibrated after the scan and tried to retarget the provider — but + * every scan ends by starting the model load, so the retarget was refused on + * every run and the verdict never applied to the session that measured it. + * Deferring the load cannot lose that race, whoever asks for the model first. + * + * And awaiting calibration before the scan means the scan's own vectorizing + * pass runs on the chosen device and records its per-segment rate — a real + * measurement on the real path — which is then what the drain budget is judged + * by. The previous version substituted calibration's own rate, which is the + * fastest of three passes over uniform 1000-character segments and + * systematically quicker than real windows. + */ +export async function runBackgroundWork(deps: BackgroundDeps): Promise { + const env = deps.env ?? process.env; + const log = deps.log ?? ((line: string) => process.stderr.write(`${line}\n`)); + + try { + await calibrateFirstIfNeeded(deps, env, log); + await deps.sessions.listRecentSessions(1); + await deps.sessions.whenScanSettled(); + await drainIfAffordable(deps.sessions); + } catch (error) { + // Reported rather than swallowed. The drain records its own failure in + // `embedding_error`; this line is for everything else, and for anyone + // reading the server's stderr in their agent's MCP log. + log( + `xtctx: background indexing stopped: ${error instanceof Error ? error.message : String(error)}`, + ); + } +} + +/** + * Measure this machine's embedding devices, once per machine, before any + * model load. + * + * Not conditioned on a backlog, because at startup there is no way to know + * one without first scanning — and scanning first is the ordering that broke + * this. The verdict is per machine, and any configured project will embed. + */ +async function calibrateFirstIfNeeded( + deps: BackgroundDeps, + env: NodeJS.ProcessEnv, + log: (line: string) => void, +): Promise { + // No model is ever loaded with this set, and calibration loads one per + // device. This guard was missing twice, in two places, in two days. + if (env.XTCTX_DISABLE_EMBEDDINGS === "1") { + return; + } + const readVerdict = deps.readVerdict ?? (() => readDeviceVerdict()); + if (await readVerdict()) { + return; + } + + const calibrate = deps.calibrate ?? (() => calibrateEmbeddingDevice()); + const running = calibrate(); + deps.sessions.deferEmbeddingDeviceUntil?.( + running.then((verdict) => verdict.device).catch(() => undefined), + ); + + try { + await running; + } catch (error) { + // Another server holding the lock is the ordinary case on a fresh machine + // with several agents starting at once, and says nothing is wrong. Anything + // else is worth a line: this session embeds on the default device. + if (!(error instanceof CalibrationBusyError)) { + log( + `xtctx: could not measure embedding devices, using the default: ${error instanceof Error ? error.message : String(error)}`, + ); + } + } +} + +/** + * Work the vector backlog down, when this machine's measured rate says the + * remainder fits `BACKGROUND_EMBED_BUDGET_MS`. + * + * Until this existed nothing finished embedding a real history: searches + * vectorize about sixteen windows a call, which by this project's own + * measurement needs on the order of 570 searches for a 9,232-window project. + * Bounded rather than unconditional, because unbounded background embedding on + * a CPU takes nine to eleven of twenty-four cores for an hour. + */ +async function drainIfAffordable(sessions: SessionService): Promise { + if (!sessions.embedBacklog) { + return; + } + const status = await sessions.getStatus(); + const { remaining, etaMs } = estimateVectorBacklog( + status.retrieval_units, + status.vectorized_units, + status.vector_ms_per_unit, + { backlog: status.vector_segment_backlog, msPerSegment: status.vector_ms_per_segment }, + ); + // No estimate means nothing has embedded yet on this machine, so there is no + // measured rate to judge affordability by. + if (remaining === 0 || etaMs === null || etaMs > BACKGROUND_EMBED_BUDGET_MS) { + return; + } + await sessions.embedBacklog(); +} diff --git a/src/utils/duration.ts b/src/utils/duration.ts index 3d16df07..bd2f37e2 100644 --- a/src/utils/duration.ts +++ b/src/utils/duration.ts @@ -18,8 +18,12 @@ export function formatDuration(ms: number | null | undefined): string | null { if (ms < 60_000) { return `${(ms / 1_000).toFixed(1)}s`; } - const minutes = Math.floor(ms / 60_000); - const seconds = Math.round((ms % 60_000) / 1_000); + // Round once, then split. Rounding minutes and seconds separately printed + // "1m 60s" for 119.6 seconds — found by a mutation sweep, when swapping this + // file's rounding for flooring changed nothing any test could see. + const totalSeconds = Math.round(ms / 1_000); + const minutes = Math.floor(totalSeconds / 60); + const seconds = totalSeconds % 60; return `${minutes}m ${String(seconds).padStart(2, "0")}s`; } diff --git a/tests/handoff/device-retarget.test.ts b/tests/handoff/device-retarget.test.ts index 19313fe4..95bb39a3 100644 --- a/tests/handoff/device-retarget.test.ts +++ b/tests/handoff/device-retarget.test.ts @@ -1,57 +1,84 @@ /** - * Calibration has to be able to take effect in the session that ran it. + * The model loads on the device calibration chose, even when something asks + * for the model before calibration finishes. * - * The provider loads its model lazily, so until something embeds, pointing it - * at a different device costs nothing. That is what makes full automation - * possible: the server starts, finds a backlog and no verdict, measures the - * devices in child processes, and points the not-yet-loaded provider at the - * winner before anything embeds. + * The first version of same-session calibration retargeted the provider only + * if nothing had started loading yet. The MCP server's own warm scan always + * started loading first — every scan ends by calling `warm()` — so the + * retarget was refused on every run, the verdict applied only to the next + * session, and the comments and README said otherwise. The unit test for it + * called the provider in isolation and never exercised that ordering. * - * Without this the verdict only applied to the NEXT session, and the gate that - * decides whether to drain in the background is computed from the old device's - * rate — so a large history on a CPU stayed above the budget, never drained, - * and therefore never benefited from the calibration it had just paid for. - * - * The refusal matters as much as the retarget. Once the model is loaded or - * loading, swapping the device means discarding it and paying the load again, - * possibly while a tool call is waiting on it, to speed up work already in - * flight. The caller is told it did not apply rather than left to assume it did. + * These tests put the load FIRST, which is the order that actually happens. */ import { describe, expect, it } from "vitest"; -import { TransformersEmbeddingProvider } from "@xtctx/handoff/embeddings"; +import { + TransformersEmbeddingProvider, + type PipelineFactory, +} from "@xtctx/handoff/embeddings"; + +/** A pipeline that records the device it was loaded with and returns vectors. */ +function recordingPipeline(): { factory: PipelineFactory; devices: Array } { + const devices: Array = []; + const factory: PipelineFactory = async (_task, _model, options) => { + devices.push((options as { device?: string } | undefined)?.device); + return async (input) => { + const count = Array.isArray(input) ? input.length : 1; + return { data: new Float32Array(count * 2).fill(0.5) }; + }; + }; + return { factory, devices }; +} + +describe("deferring the model load until the device is known", () => { + it("loads on the calibrated device even when the load was requested first", async () => { + const { factory, devices } = recordingPipeline(); + const provider = new TransformersEmbeddingProvider(undefined, undefined, undefined, factory); + + let resolveDevice: (device: string) => void = () => {}; + provider.deferDeviceUntil(new Promise((resolve) => (resolveDevice = resolve))); + + // The warm scan asks for the model before calibration is done. + const embedding = provider.embed("some text"); + await new Promise((resolve) => setTimeout(resolve, 10)); + expect(devices).toEqual([]); -describe("retargeting an embedding provider", () => { - it("applies while the model has not been loaded", () => { - const provider = new TransformersEmbeddingProvider(); - expect(provider.device).toBeUndefined(); + resolveDevice("dml"); + await embedding; - expect(provider.retargetDevice("dml")).toBe(true); + expect(devices).toEqual(["dml"]); expect(provider.device).toBe("dml"); }); - it("replaces a device chosen earlier, so a re-measure wins", () => { - const provider = new TransformersEmbeddingProvider(undefined, undefined, "cpu"); + it("keeps the configured device when calibration fails", async () => { + // A failed measurement is not a reason to stop using the device that has + // always worked. + const { factory, devices } = recordingPipeline(); + const provider = new TransformersEmbeddingProvider(undefined, undefined, "cpu", factory); - expect(provider.retargetDevice("webgpu")).toBe(true); - expect(provider.device).toBe("webgpu"); + provider.deferDeviceUntil(Promise.reject(new Error("could not measure"))); + await provider.embed("some text"); + + expect(devices).toEqual(["cpu"]); }); - it("refuses once the model is loading, and leaves the device alone", async () => { - const provider = new TransformersEmbeddingProvider(undefined, undefined, "cpu"); - // `warm()` starts the load without waiting for it, which is exactly the - // window this guards: a tool call arriving mid-calibration. - provider.warm(); + it("keeps the configured device when calibration yields nothing", async () => { + const { factory, devices } = recordingPipeline(); + const provider = new TransformersEmbeddingProvider(undefined, undefined, undefined, factory); + + provider.deferDeviceUntil(Promise.resolve(undefined)); + await provider.embed("some text"); - expect(provider.retargetDevice("dml")).toBe(false); - expect(provider.device).toBe("cpu"); + expect(devices).toEqual([undefined]); }); - it("reports the device it will actually load on", () => { - // `xtctx status` reads this off the provider rather than off the - // calibration cache, so it has to follow a retarget. - const provider = new TransformersEmbeddingProvider(); - provider.retargetDevice("dml"); + it("loads once, however many callers were waiting", async () => { + const { factory, devices } = recordingPipeline(); + const provider = new TransformersEmbeddingProvider(undefined, undefined, undefined, factory); + provider.deferDeviceUntil(Promise.resolve("webgpu")); - expect(provider.device).toBe("dml"); + await Promise.all([provider.embed("a"), provider.embed("b"), provider.embedBatch(["c", "d"])]); + + expect(devices).toEqual(["webgpu"]); }); }); diff --git a/tests/handoff/embedding-batch.test.ts b/tests/handoff/embedding-batch.test.ts new file mode 100644 index 00000000..58334812 --- /dev/null +++ b/tests/handoff/embedding-batch.test.ts @@ -0,0 +1,31 @@ +/** + * Segments go to the model sixteen at a time. + * + * Sixteen was measured on the real path — `xtctx scan --embed` over this + * project's own index — at 107.8ms per window, against 542.9 at 128. A + * benchmark on uniform-length segments had recommended 128, reproducibly; it + * would have been a 5x regression, because a batch is padded to its longest + * member and real segments vary in length. See `MAX_BATCH_SIZE`. + * + * Changing it back survived the whole suite in a mutation sweep, so this pins + * the behaviour. It is deliberately a speed bump rather than a principle: if a + * new real-path measurement says otherwise, change both, and say why. + */ +import { describe, expect, it } from "vitest"; +import { TransformersEmbeddingProvider, type PipelineFactory } from "@xtctx/handoff/embeddings"; + +describe("embedding batch width", () => { + it("sends segments to the model sixteen at a time", async () => { + const batches: number[] = []; + const factory: PipelineFactory = async () => async (input) => { + const count = Array.isArray(input) ? input.length : 1; + batches.push(count); + return { data: new Float32Array(count * 2).fill(0.5) }; + }; + const provider = new TransformersEmbeddingProvider(undefined, undefined, undefined, factory); + + await provider.embedBatch(Array.from({ length: 40 }, (_, index) => `segment ${index}`)); + + expect(batches).toEqual([16, 16, 8]); + }); +}); diff --git a/tests/handoff/ranking-contract.test.ts b/tests/handoff/ranking-contract.test.ts index 1c2ae971..b6807588 100644 --- a/tests/handoff/ranking-contract.test.ts +++ b/tests/handoff/ranking-contract.test.ts @@ -72,6 +72,32 @@ function rank(rows: VectorUnitRow[], limit = 5) { }).map((session) => session.session_ref); } +describe("hybrid mode actually blends", () => { + it("lets a keyword match lift a session above an equally similar one", () => { + // Zeroing the keyword half of the hybrid blend survived the whole suite in + // a real mutation sweep on 2026-09-23 — hybrid silently became vector mode + // with a recency tie-break. So the keyword-matched session here is the + // OLDER one: without its keyword score, recency would put it second. + const older = unit("codex:keyword-match", "k-1", 0.8, { endedAt: "2026-05-01T10:00:00.000Z" }); + const newer = unit("codex:no-keyword", "n-1", 0.8, { endedAt: "2026-05-20T10:00:00.000Z" }); + + const ranked = rankSearchCandidates({ + rows: [older, newer], + keywordRows: [older], + queryVector: Float32Array.from([1]), + mode: "hybrid", + limit: 5, + deserializeVector: (buffer) => + Float32Array.from( + new Float64Array(buffer.buffer.slice(buffer.byteOffset, buffer.byteOffset + buffer.byteLength)), + ), + cosineSimilarity: (_query, vector) => vector[0], + }).map((session) => session.session_ref); + + expect(ranked[0]).toBe("codex:keyword-match"); + }); +}); + describe("ranking contracts the eval alone used to hold", () => { it("prefers a session corroborated by several windows to one lone window", () => { // Several windows saying the same thing is evidence the session is about diff --git a/tests/runtime/background.test.ts b/tests/runtime/background.test.ts new file mode 100644 index 00000000..84596ace --- /dev/null +++ b/tests/runtime/background.test.ts @@ -0,0 +1,189 @@ +/** + * What the MCP server does in the background after it starts. + * + * This logic lived in `cli/index.ts`, which runs `main()` on import, so no test + * could reach it — and a mutation sweep found three changes to it that the + * whole suite let through: ignoring the drain budget, calibrating with + * embeddings disabled, and (found by review rather than mutation) an ordering + * that meant calibration never applied to the session that ran it. + */ +import { describe, expect, it } from "vitest"; +import { CalibrationBusyError, type DeviceVerdict } from "@xtctx/handoff/device"; +import type { HandoffStatus, SessionService } from "@xtctx/handoff/types"; +import { runBackgroundWork } from "@xtctx/runtime/background"; + +const VERDICT: DeviceVerdict = { + device: "dml", + measured: [ + { device: "cpu", msPerSegment: 20 }, + { device: "dml", msPerSegment: 3 }, + ], + fingerprint: "test", + measuredAt: "2026-09-23T00:00:00.000Z", +}; + +/** + * Records the order things happen in. Backlog of 1,000 windows at the given + * per-window rate, so the caller decides whether it fits the budget. + */ +function fakeSessions(msPerUnit: number | null): { + sessions: SessionService; + events: string[]; + deferred: Array>; +} { + const events: string[] = []; + const deferred: Array> = []; + const status = { + retrieval_units: 1000, + vectorized_units: 0, + vector_ms_per_unit: msPerUnit, + vector_segment_backlog: 0, + vector_ms_per_segment: null, + } as unknown as HandoffStatus; + + const sessions = { + async listRecentSessions() { + events.push("scan"); + return []; + }, + async whenScanSettled() { + events.push("settled"); + }, + async getStatus() { + return status; + }, + async embedBacklog() { + events.push("drain"); + return 0; + }, + deferEmbeddingDeviceUntil(device: Promise) { + events.push("defer"); + deferred.push(device); + }, + } as unknown as SessionService; + + return { sessions, events, deferred }; +} + +describe("the server's background work", () => { + it("defers the model load to calibration BEFORE scanning", async () => { + // Scanning first is what broke this: every scan starts the model load. + const { sessions, events, deferred } = fakeSessions(50); + + await runBackgroundWork({ + sessions, + env: {}, + readVerdict: async () => null, + calibrate: async () => { + events.push("calibrate"); + return VERDICT; + }, + log: () => {}, + }); + + expect(events.indexOf("defer")).toBeLessThan(events.indexOf("scan")); + expect(events.indexOf("calibrate")).toBeLessThan(events.indexOf("scan")); + expect(await deferred[0]).toBe("dml"); + }); + + it("does not calibrate when embeddings are switched off", async () => { + // Missed twice, in two places, in two days. + const { sessions, events } = fakeSessions(50); + let calibrated = false; + + await runBackgroundWork({ + sessions, + env: { XTCTX_DISABLE_EMBEDDINGS: "1" }, + readVerdict: async () => null, + calibrate: async () => { + calibrated = true; + return VERDICT; + }, + log: () => {}, + }); + + expect(calibrated).toBe(false); + expect(events).not.toContain("defer"); + }); + + it("does not calibrate a machine that already has a verdict", async () => { + const { sessions } = fakeSessions(50); + let calibrated = false; + + await runBackgroundWork({ + sessions, + env: {}, + readVerdict: async () => VERDICT, + calibrate: async () => { + calibrated = true; + return VERDICT; + }, + log: () => {}, + }); + + expect(calibrated).toBe(false); + }); + + it("drains a backlog that fits the budget", async () => { + // 1,000 windows at 50ms is under a minute. + const { sessions, events } = fakeSessions(50); + + await runBackgroundWork({ sessions, env: {}, readVerdict: async () => VERDICT, log: () => {} }); + + expect(events).toContain("drain"); + }); + + it("leaves a backlog that does not fit the budget alone", async () => { + // 1,000 windows at 5s each is over an hour — the CPU case the budget is for. + const { sessions, events } = fakeSessions(5_000); + + await runBackgroundWork({ sessions, env: {}, readVerdict: async () => VERDICT, log: () => {} }); + + expect(events).not.toContain("drain"); + }); + + it("does not drain when no rate has ever been measured", async () => { + const { sessions, events } = fakeSessions(null); + + await runBackgroundWork({ sessions, env: {}, readVerdict: async () => VERDICT, log: () => {} }); + + expect(events).not.toContain("drain"); + }); + + it("stays quiet when another server holds the calibration lock", async () => { + // The ordinary case on a fresh machine with several agents starting at once. + const { sessions, events } = fakeSessions(50); + const lines: string[] = []; + + await runBackgroundWork({ + sessions, + env: {}, + readVerdict: async () => null, + calibrate: async () => { + throw new CalibrationBusyError(); + }, + log: (line) => lines.push(line), + }); + + expect(lines).toEqual([]); + expect(events).toContain("scan"); + }); + + it("reports a failure instead of swallowing it", async () => { + // Background failures used to vanish into a bare catch. + const { sessions } = fakeSessions(50); + (sessions as { embedBacklog: () => Promise }).embedBacklog = async () => { + throw new Error("endpoint returned HTTP 401"); + }; + const lines: string[] = []; + + await runBackgroundWork({ + sessions, + env: {}, + readVerdict: async () => VERDICT, + log: (line) => lines.push(line), + }); + + expect(lines.join("\n")).toMatch(/HTTP 401/); + }); +}); diff --git a/tests/utils/duration.test.ts b/tests/utils/duration.test.ts index 1933159d..8026cf02 100644 --- a/tests/utils/duration.test.ts +++ b/tests/utils/duration.test.ts @@ -6,6 +6,10 @@ describe("formatDuration", () => { expect(formatDuration(340)).toBe("340ms"); expect(formatDuration(1_400)).toBe("1.4s"); expect(formatDuration(125_000)).toBe("2m 05s"); + // Rounded as a whole, not per unit: this printed "1m 60s". + expect(formatDuration(119_600)).toBe("2m 00s"); + // Sub-second values round rather than truncate. + expect(formatDuration(339.6)).toBe("340ms"); }); it("returns null rather than a fabricated duration", () => { From ded22aab1e30e4071b51bff09cc7d15c09975561 Mon Sep 17 00:00:00 2001 From: Felix Stubner Date: Thu, 24 Sep 2026 00:09:04 +0100 Subject: [PATCH 32/34] =?UTF-8?q?fix:=20the=20remaining=20audit=20findings?= =?UTF-8?q?=20=E2=80=94=20claims,=20lows,=20and=20one=20tested=20lock?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The rest of the 2026-09-23 audit. Claims made false this week, all corrected: - docs/embedding-performance.md said the MCP server "reads the verdict and nothing more"; it now calibrates in the background, and the paragraph says how and records why an earlier version did not apply. - The landing page's FAQ said indexing happened "on demand" and vectors were "created lazily", and called xtctx local-only without the opt-in endpoint. Its `xtctx status` mock used an output format the command has never printed; it now shows the real labels. - This repository's own four managed blocks still said indexing was on-demand. The generator had been updated; the committed copies had not. - ux-walkthrough.md said the tools work in a project without setup. They name `xtctx setup` and do nothing else. - ranking.ts presented MiniLM's 0.36 sweep as the live threshold. - CHANGELOG now says, inside the 0.21.8 section so the release workflow's insertion point does not move it, that 0.20.0–0.21.8 were never on npm. Lows: - The OpenAI provider now releases the body of a response it gives up on, not only one it retries; and a 429 waits for Retry-After (clamped to 5s, default 1s) instead of retrying instantly into a second 429. - The segment backlog is scoped to the project, like the window count printed beside it; in a shared index the duration described more work than the count. - The unconfigured and unreadable-config notices honour `format: json` — and default to it for `xtctx_handoff_manifest`, whose documented contract is JSON. An orchestrator used to get prose it could not parse. - `scripts/*.ts` is now typechecked and linted; neither covered it. The device probe measured MiniLM for two days after the default changed. - The drift canary FAILS when it cannot run for lack of a key. It passed with a warning so as not to turn a nightly job red, and there is no nightly job: a dispatched check that checked nothing should not look like a pass. - The `v9` and `concepts` design drafts no longer deploy. The site builds one page instead of eight. And a test for the calibration lock added in the previous commit, which had none: a fresh lock refuses, an abandoned one is taken over. Verified with a mutation harness that counts a run only if vitest printed its own summary: all nine mutations — the six survivors from the audit and one per new fix — are killed, and the working tree was byte-identical before and after the sweep. `npm run verify:release` exits 0 and builds the package once rather than three times. --- .github/copilot-instructions.md | 2 +- .github/workflows/drift-canary.yml | 17 +- AGENTS.md | 2 +- CHANGELOG.md | 5 + CLAUDE.md | 2 +- GEMINI.md | 2 +- README.md | 5 +- docs/embedding-performance.md | 16 +- eslint.config.js | 2 +- landing/src/data/concepts.ts | 109 -- landing/src/data/site.ts | 16 +- landing/src/pages/concepts/[slug].astro | 961 --------------- landing/src/pages/concepts/index.astro | 124 -- landing/src/pages/v9.astro | 1200 ------------------- scripts/probe-embedding-device.mjs | 9 +- src/cli/calibrate.ts | 4 +- src/handoff/openai-embeddings.ts | 33 + src/handoff/ranking.ts | 16 +- src/handoff/status.ts | 2 +- src/handoff/vectors.ts | 18 +- src/mcp/server.ts | 48 +- tests/handoff/device-calibration.test.ts | 37 +- tests/handoff/openai-embeddings.test.ts | 51 +- tests/handoff/segment-backlog-scope.test.ts | 59 + tests/mcp/config-error-notice.test.ts | 39 +- tsconfig.test.json | 2 +- ux-walkthrough.md | 19 +- 27 files changed, 344 insertions(+), 2456 deletions(-) delete mode 100644 landing/src/data/concepts.ts delete mode 100644 landing/src/pages/concepts/[slug].astro delete mode 100644 landing/src/pages/concepts/index.astro delete mode 100644 landing/src/pages/v9.astro create mode 100644 tests/handoff/segment-backlog-scope.test.ts diff --git a/.github/copilot-instructions.md b/.github/copilot-instructions.md index a48e8ee1..56ecb729 100644 --- a/.github/copilot-instructions.md +++ b/.github/copilot-instructions.md @@ -27,7 +27,7 @@ Do not rely on this block for a generated summary; raw local transcripts are aut - Transport: stdio ## Notes -- Indexing is on-demand from MCP recent, detail, and search calls. +- Indexing runs when the MCP server starts and on recent, detail, and search calls; `xtctx scan` does it on demand. - There is no xtctx daemon, API server, dashboard, durable memory, or generated brief. - Content outside this managed block is preserved. diff --git a/.github/workflows/drift-canary.yml b/.github/workflows/drift-canary.yml index 3c104f0e..f3ae9920 100644 --- a/.github/workflows/drift-canary.yml +++ b/.github/workflows/drift-canary.yml @@ -99,14 +99,21 @@ jobs: # the run, and `upstream-watch` is what files issues now. # 78 means the canary could not run at all — no API key — which is a - # configuration gap, not upstream drift. Surfaced as a warning so it is - # visible in the run summary without turning the nightly job red and - # without filing an issue that says the scraper is broken. - - name: Note that the canary was skipped + # configuration gap, not upstream drift. It FAILS the job, with a message + # saying which of the two it is. + # + # It used to pass with a warning, on the reasoning that a missing key + # should not turn the nightly job red. There is no nightly job any more: + # this runs when a person dispatches it, to find out whether a tool's + # format drifted, and a green result that checked nothing answers that + # question wrongly. A check that could not run must not look like one + # that passed. + - name: Fail because the canary could not run if: steps.canary.outputs.exit_code == '78' run: | - echo "::warning title=drift canary skipped::${{ matrix.tool }} did not run — its API key is not configured, so this run produced no drift signal for it." + echo "::error title=drift canary did not run::${{ matrix.tool }} was not checked — its API key is not configured, so this run says nothing about whether its format drifted." cat canary-stderr.log || true + exit 1 - name: Fail job on canary failure # exit_code is empty when the canary step was skipped by a scoped dispatch. diff --git a/AGENTS.md b/AGENTS.md index 76db72ee..7a54d31e 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -77,7 +77,7 @@ Do not rely on this block for a generated summary; raw local transcripts are aut - Transport: stdio ## Notes -- Indexing is on-demand from MCP recent, detail, and search calls. +- Indexing runs when the MCP server starts and on recent, detail, and search calls; `xtctx scan` does it on demand. - There is no xtctx daemon, API server, dashboard, durable memory, or generated brief. - Content outside this managed block is preserved. diff --git a/CHANGELOG.md b/CHANGELOG.md index f4190ffa..f1b46095 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,11 @@ Entries are written by the `release` workflow when a release is cut by hand. ## [0.21.8](https://github.com/fstubner/xtctx/compare/xtctx-v0.21.7...xtctx-v0.21.8) (2026-08-31) +> **Not on npm.** 0.20.0 through 0.21.8 were tagged and given GitHub +> Releases, but none of them was ever published to npm — the last version +> published before them is 0.19.0. Everything listed from here down to 0.20.0 +> reaches `npx -y xtctx` users only in the next version that is published. + ### Bug Fixes diff --git a/CLAUDE.md b/CLAUDE.md index 0e2e3c5a..bd7b0d9e 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -27,7 +27,7 @@ Do not rely on this block for a generated summary; raw local transcripts are aut - Transport: stdio ## Notes -- Indexing is on-demand from MCP recent, detail, and search calls. +- Indexing runs when the MCP server starts and on recent, detail, and search calls; `xtctx scan` does it on demand. - There is no xtctx daemon, API server, dashboard, durable memory, or generated brief. - Content outside this managed block is preserved. diff --git a/GEMINI.md b/GEMINI.md index 1636914e..65b69c29 100644 --- a/GEMINI.md +++ b/GEMINI.md @@ -27,7 +27,7 @@ Do not rely on this block for a generated summary; raw local transcripts are aut - Transport: stdio ## Notes -- Indexing is on-demand from MCP recent, detail, and search calls. +- Indexing runs when the MCP server starts and on recent, detail, and search calls; `xtctx scan` does it on demand. - There is no xtctx daemon, API server, dashboard, durable memory, or generated brief. - Content outside this managed block is preserved. diff --git a/README.md b/README.md index 192dc467..ff0cba18 100644 --- a/README.md +++ b/README.md @@ -174,9 +174,8 @@ budget. You need it when `xtctx status` says the backlog is too large to finish in the background — otherwise the server gets there on its own. Indexing picks a device by measuring it, and **you do not have to do anything -to get that**. The first time a machine has embedding work worth doing — the -MCP server finding a backlog at session start, or `xtctx scan --embed` — it -times the embedding model on each execution provider available and remembers +to get that**. The first time the MCP server starts on a machine, or the +first `xtctx scan --embed`, it times the embedding model on each execution provider available and remembers the fastest in `~/.xtctx/device.json`, once per machine. On a machine with a usable GPU that has measured roughly six times faster than the CPU; on one without, it picks the CPU and nothing changes. Vectors are identical whichever diff --git a/docs/embedding-performance.md b/docs/embedding-performance.md index 640b40b8..5420bca3 100644 --- a/docs/embedding-performance.md +++ b/docs/embedding-performance.md @@ -339,11 +339,17 @@ count, model and dtype. `xtctx scan --embed` runs it automatically when the machine has no verdict yet — a minute against the hours that command is about to spend — and `--no-calibrate` skips it. -Nothing else calibrates. The MCP server answers a tool call inside a -four-second budget and must not spawn three model-loading processes behind it; -it reads the verdict and nothing more. `xtctx status` prints the device, read -off the provider rather than off the cache file, so the line is evidence that -the indexer is using it rather than evidence that a file exists. +The MCP server calibrates too, in the background when it starts on a machine +with no verdict — never behind a tool call, which has a four-second budget. +It starts calibration before its first scan and makes the model's first load +wait for the result, so the session that paid for the measurement is the one +that uses it. (An earlier version calibrated after the scan and tried to +retarget the provider, which every scan had already started loading; the +verdict only ever reached the next session, and this paragraph said the server +did not calibrate at all.) One server calibrates at a time, under a +machine-wide lock. `xtctx status` prints the device, read off the provider +rather than off the cache file, so the line is evidence that the indexer is +using it rather than evidence that a file exists. The decision rule is "fastest measured device", with a 1.1x margin over the CPU, and every case where no comparison exists resolves to the CPU: a device diff --git a/eslint.config.js b/eslint.config.js index 730a6d75..9345f078 100644 --- a/eslint.config.js +++ b/eslint.config.js @@ -7,7 +7,7 @@ export default tseslint.config( }, // TypeScript sources and tests. { - files: ["src/**/*.ts", "tests/**/*.ts"], + files: ["src/**/*.ts", "tests/**/*.ts", "scripts/**/*.ts"], extends: [...tseslint.configs.recommended], rules: { "@typescript-eslint/no-explicit-any": "error", diff --git a/landing/src/data/concepts.ts b/landing/src/data/concepts.ts deleted file mode 100644 index 671b761c..00000000 --- a/landing/src/data/concepts.ts +++ /dev/null @@ -1,109 +0,0 @@ -export interface Concept { - slug: string; - name: string; - shortName: string; - description: string; - headline: string; - subhead: string; - approach: string; -} - -export const concepts: Concept[] = [ - { - slug: 'boxed-product', - name: 'Boxed blueprint product', - shortName: 'Boxed', - description: 'A contained product page with setup selection and blueprint proof.', - headline: 'Move between coding agents without starting over', - subhead: - 'Run setup in a repo, choose the project skills to sync, and xtctx writes the local MCP config and managed instructions the next tool needs.', - approach: 'Best when the page should feel deliberate and productized while keeping the blueprint feel from the current root.', - }, - { - slug: 'v9-less-diagram', - name: 'v9, less diagram', - shortName: 'v9 Lite', - description: 'The current v9 rhythm, but with a smaller diagram and tighter proof.', - headline: 'Local context handoff for coding agents', - subhead: - 'xtctx keeps the surface small: setup, status, disconnect, and five MCP tools for recent local transcript sessions.', - approach: 'Best if the v9 page shape is right but the middle section should do less.', - }, - { - slug: 'cli-first', - name: 'CLI-first landing', - shortName: 'CLI First', - description: 'A developer-native page led by the command and its output.', - headline: 'Set up local transcript retrieval from the CLI', - subhead: - 'Install nothing global. Run one command in the repo, then let MCP clients read recent local sessions on demand.', - approach: 'Best if xtctx should feel like a serious open-source CLI rather than a product site.', - }, - { - slug: 'two-column-proof', - name: 'Two-column proof', - shortName: 'Proof', - description: 'A split proof page showing what setup writes and what agents call.', - headline: 'Project wiring for local agent handoff', - subhead: - 'xtctx writes local project files and exposes raw transcript retrieval through a small MCP surface.', - approach: 'Best if the page should make the product contract obvious in the first scroll.', - }, - { - slug: 'docs-hybrid', - name: 'Docs-landing hybrid', - shortName: 'Docs', - description: 'An understated docs-like landing page with command-first sections.', - headline: 'Local MCP setup for recent coding sessions', - subhead: - 'A small CLI and MCP server for reading recent local transcript sessions from supported AI coding tools.', - approach: 'Best if the public page should feel open-source, plain, and trustworthy.', - }, -]; - -export const conceptNav = [ - { href: '#how', label: 'How it works' }, - { href: '#features', label: 'Features' }, - { href: '#setup', label: 'Setup' }, - { href: '#faqs', label: 'FAQs' }, - { href: 'https://github.com/fstubner/xtctx', label: 'GitHub' }, -]; - -export const conceptFeatures = [ - { - label: 'Setup', - title: 'Writes local config', - body: '.xtctx/config.yaml, managed instruction blocks, MCP config, and selected skill targets.', - }, - { - label: 'MCP', - title: 'Five retrieval tools', - body: 'recent sessions, session detail, transcript search, continuity status, and handoff manifest.', - }, - { - label: 'Storage', - title: 'Raw transcripts stay local', - body: 'SQLite is rebuildable cache state. Transcript files remain authoritative.', - }, - { - label: 'Limits', - title: 'No background service', - body: 'No daemon, dashboard, API server, durable memory, or generated summary layer.', - }, -]; - -export const setupLines = [ - 'updated .xtctx/config.yaml', - 'updated .xtctx/skills/xtctx-handoff/SKILL.md', - 'updated AGENTS.md', - 'updated .codex/config.toml', - 'ready MCP config', -]; - -export const statusLines = [ - 'ready mcp command npx -y xtctx', - 'ready managed instructions repaired', - 'ready 3 selected skills synced', - 'cache indexes on MCP retrieval', - 'data raw transcripts stay authoritative', -]; diff --git a/landing/src/data/site.ts b/landing/src/data/site.ts index 8b215508..9d8863a2 100644 --- a/landing/src/data/site.ts +++ b/landing/src/data/site.ts @@ -135,7 +135,7 @@ export const site: SiteData = { heading: 'Keep project context portable across coding tools.', subhead: 'Move between supported coding agents without starting over. xtctx indexes the transcript files your tools already write and serves them over MCP, so the next agent can pick up recent context. Install the plugin once to reach the tools everywhere, then opt each project in with a single setup command.', - proof: ['One command per project', 'Raw transcripts stay local', 'Five MCP tools'], + proof: ['One command per project', 'Local by default', 'Five MCP tools'], quickInstall: 'claude plugin marketplace add fstubner/xtctx && claude plugin install xtctx@xtctx', installLinkLabel: 'Get started', sourceUrl: REPO_URL, @@ -218,11 +218,11 @@ export const site: SiteData = { body: 'Status reports configured tools, transcript freshness, selected skills, managed blocks, and unsupported targets.', codeHtml: `$ npx -y xtctx status -configured -mcp command npx -y xtctx -cache 12 sessions -codex instruction only -claude-code executable hook`, +MCP npx -y xtctx +Data 12 sessions, 1840 messages, 460 retrieval windows, 460 vectorized +Tools: + + codex detected; 7 sessions; hook: instruction-only + + claude-code detected; 5 sessions; hook: executable`, }, { title: 'Search stays local', @@ -285,7 +285,7 @@ export const site: SiteData = { }, { q: 'Does xtctx run a background service?', - a: 'No. xtctx has no daemon, API server, dashboard, watcher, or web service. MCP retrieval calls update the local cache on demand.', + a: 'No. xtctx has no daemon, API server, dashboard, watcher, or web service. Each agent starts its own xtctx MCP server, which indexes when it starts and when it is called, and stops when the agent does.', }, { q: 'Does xtctx sync skills?', @@ -297,7 +297,7 @@ export const site: SiteData = { }, { q: 'What are the limits?', - a: 'xtctx is local-only. Transcript formats can change upstream, semantic vectors are created lazily, and keyword fallback is expected when local vector generation is unavailable.', + a: 'xtctx is local-only by default; sending window text to an external embedding endpoint is something a project has to opt into by hand. Transcript formats can change upstream, semantic vectors are built incrementally in the background, and search falls back to keyword while vectors are missing or the local model is unavailable.', }, { q: 'Can I test it without private transcripts?', diff --git a/landing/src/pages/concepts/[slug].astro b/landing/src/pages/concepts/[slug].astro deleted file mode 100644 index 02cad64e..00000000 --- a/landing/src/pages/concepts/[slug].astro +++ /dev/null @@ -1,961 +0,0 @@ ---- -import IdeMock from '../../components/IdeMock.astro'; -import TerminalMock from '../../components/TerminalMock.astro'; -import { - conceptFeatures, - conceptNav, - concepts, - setupLines, - statusLines, - type Concept, -} from '../../data/concepts'; - -export function getStaticPaths() { - return concepts.map((concept) => ({ - params: { slug: concept.slug }, - props: { concept }, - })); -} - -const { concept } = Astro.props as { concept: Concept }; -const currentIndex = concepts.findIndex((item) => item.slug === concept.slug); -const setupOutputHtml = [ - '$ npx -y xtctx setup', - ...setupLines.map((line) => { - const [status, ...rest] = line.split(' '); - return `${status} ${rest.join(' ')}`; - }), -].join('\n'); - -const blueprintNodes = [ - { label: 'setup', title: 'Detect tools', body: 'MCP config, instructions, hooks' }, - { label: 'select', title: 'Choose skills', body: 'Keep the project skill set explicit' }, - { label: 'handoff', title: 'Call MCP', body: 'recent, detail, search, status' }, -]; ---- - - - - - - - {concept.name} - xtctx landing concept - - - -
- -
- -
-
-
-

{concept.shortName} concept

-

{concept.headline}

-

{concept.subhead}

-
    -
  • Setup writes local config
  • -
  • Raw transcripts stay local
  • -
  • Five MCP tools
  • -
-
$npx -y xtctx setup
- -
-
- {concept.slug === 'boxed-product' ? ( -
-
- - bash - -
-
-

$ npx -y xtctx setup

-

xtctx setup for ~/projects/my-app

-

Detected Claude Code, Codex, Cursor, and Antigravity.

-

Select skills to sync

-

[x]xtctx-handoffrequired

-

[x]review-notesClaude Code

-

[x]release-checks.xtctx/skills

-

[ ]scratchpadskip

-

Apply these changes? Yes

-
-

ok.xtctx/config.yaml

-

okmanaged instructions written

-

okMCP server registered

-

Ready for the next coding agent.

-
-
- ) : concept.slug === 'cli-first' || concept.slug === 'docs-hybrid' ? ( -
- ) : ( -
-
setup commandrun from project root
-

-            
- )} -
-
- -
-
-
- -

- {concept.slug === 'cli-first' - ? 'Run a command. MCP reads the local index.' - : concept.slug === 'docs-hybrid' - ? 'Setup writes files. Clients call MCP.' - : 'Setup writes config. Tools call MCP.'} -

-

{concept.approach}

-
-
- {concept.slug === 'boxed-product' ? ( - <> -
handoff blueprintno background service
-
- {blueprintNodes.map((node) => ( -
- {node.label} -
- {node.title} -

{node.body}

-
-
- ))} -
-

- Raw transcript files stay authoritative. The SQLite index is rebuildable and updated on MCP retrieval. -

- - ) : concept.slug === 'two-column-proof' ? ( - <> -
- -

Project files

-
    -
  • .xtctx/config.yaml
  • -
  • AGENTS.md managed block
  • -
  • .codex/config.toml
  • -
  • .xtctx/skills
  • -
-
-
- -

MCP tools

-
    -
  • xtctx_recent_sessions
  • -
  • xtctx_session_detail
  • -
  • xtctx_search_sessions
  • -
  • xtctx_continuity_status
  • -
-
- - ) : ( - <> -
MCP toolsNo generated summary required
-
-
-

Tool transcripts

-
    -
  • Codex sessions
  • -
  • Claude Code projects
  • -
  • Antigravity conversations
  • -
  • Antigravity conversations
  • -
-
-
-

MCP

-
xtctx_recent_sessions
-
xtctx_session_detail
-
xtctx_search_sessions
-
xtctx_continuity_status
-
-
-

Client result

-
    -
  • Find recent work
  • -
  • Open raw messages
  • -
  • Search ordered windows
  • -
  • Check wiring status
  • -
-
-
- - )} -
-
- {concept.slug !== 'cli-first' && ( -
- )} -
- -
- -

- {concept.slug === 'docs-hybrid' - ? 'The complete product surface' - : 'Config, instructions, skills, and cache'} -

-

- xtctx does not run a web app, API server, daemon, or durable memory - layer. It writes local files and serves MCP retrieval on demand. -

-
- {conceptFeatures.map((feature) => ( -
- {feature.label} -

{feature.title}

-

{feature.body}

-
- ))} -
-
- -
-
-
- -

Install, then check

-

- The plugin gets you the MCP tools without touching the repo. Setup - writes the generated files this page describes. Status reports - configured tools, transcript freshness, selected skills, and - unsupported targets. -

-
-
-
- Install the plugin - claude plugin install xtctx@xtctx -

- Registers the MCP server and handoff skill, writes nothing into the - repo. Retrieval works without any of the files below. Add the - marketplace first; see the README for other tools. -

-
-
- Add project wiring - npx -y xtctx setup -

Detect tools, choose skills, write MCP config, and repair managed blocks.

-
-
- Check configured surfaces - npx -y xtctx status -

Inspect wiring, transcript freshness, skill drift, and unsupported targets.

-
-
- Remove one tool - npx -y xtctx disconnect antigravity -

Remove generated xtctx management without deleting transcripts.

-
-
-
-
- -
-
-
- -

Questions before using it in a repo

-

Short answers about storage, setup, and limits.

-
-
-
- Does xtctx run in the background? -

No. MCP retrieval updates the local cache on demand.

-
-
- Does it summarize sessions? -

No. Raw transcript messages remain the source of truth.

-
-
- Can I test it without private transcripts? -

Yes. The public demo uses synthetic Claude Code and Codex transcript stores.

-
-
-
-
-
- -
-
- Concept {currentIndex + 1} of {concepts.length}: {concept.name} - - {concepts.map((item, index) => ( - <> - {index + 1} - {index < concepts.length - 1 && ' / '} - - ))} - -
-
- - diff --git a/landing/src/pages/concepts/index.astro b/landing/src/pages/concepts/index.astro deleted file mode 100644 index e06336d5..00000000 --- a/landing/src/pages/concepts/index.astro +++ /dev/null @@ -1,124 +0,0 @@ ---- -import { concepts } from '../../data/concepts'; ---- - - - - - - - xtctx landing concepts - - - -
- -
-
-
-

Landing concepts

-

Five directions using the same palette

-

- These are implementation previews, not proposals in prose. Open them - side by side and pick the structure worth refining. -

-
-
- {concepts.map((concept, index) => ( - -
- {String(index + 1).padStart(2, '0')} - {concept.name} - {concept.description} -
- Open -
- ))} -
-
- - diff --git a/landing/src/pages/v9.astro b/landing/src/pages/v9.astro deleted file mode 100644 index 345d816f..00000000 --- a/landing/src/pages/v9.astro +++ /dev/null @@ -1,1200 +0,0 @@ ---- -import { site } from '../data/site'; -import IdeMock from '../components/IdeMock.astro'; - -const repo = 'fstubner/xtctx'; -const repoUrl = `https://github.com/${repo}`; - -const faqs = [ - { - q: 'What does xtctx do?', - a: 'xtctx lets a configured AI coding tool read recent local transcript sessions from the current repo through MCP. It lists sessions, opens raw messages, searches chronological windows, and reports continuity status.', - }, - { - q: 'Does xtctx run in the background?', - a: 'No. xtctx does not run a daemon, dashboard, watcher, or API server. Retrieval happens when an agent calls the MCP tools, and the SQLite cache can be rebuilt from local transcript files.', - }, - { - q: 'Where does the context come from?', - a: 'Recent context comes from local transcript files written by each tool. xtctx keeps source pointers, message rows, search windows, and vectors in SQLite so retrieval can stay local and ordered.', - }, - { - q: 'What are the limits?', - a: 'xtctx is local only. Transcript formats can change upstream, semantic vectors are created lazily, and keyword fallback is expected when local vector generation is unavailable.', - }, - { - q: 'Can I test it without private transcripts?', - a: 'Yes. Run npm run demo:public after building the package. The demo uses synthetic Claude Code and Codex transcript stores in a temporary project.', - }, - { - q: 'Does xtctx sync skills?', - a: 'Yes. Setup creates the built in xtctx handoff skill, inventories compatible project skills, and syncs selected skills into verified tool targets. Unsupported tools are reported as unsupported.', - }, - { - q: 'Can I remove one tool later?', - a: 'Yes. Run xtctx disconnect with the tool name. It removes generated MCP wiring, managed instructions, hooks, and generated skill targets for that tool without deleting transcripts or canonical project skills.', - }, - { - q: 'What if a tool hangs while starting MCP?', - a: 'Disconnect that tool from the project, then fix the package path or command before reconnecting it. Tools that eagerly start MCP servers need a command that starts stdio cleanly.', - }, - { - q: 'Which tools are supported?', - a: 'xtctx supports Codex, Claude Code, Cursor, GitHub Copilot, GitHub Copilot CLI, Google Antigravity, and opencode with different integration modes depending on the real surfaces each tool exposes.', - }, -]; - -const surfaces = [ - { - label: 'MCP', - title: 'Five MCP tools', - body: 'Tools can list recent sessions, open session detail, search transcript windows, and check continuity status.', - }, - { - label: 'Instructions', - title: 'Managed instructions', - body: 'Setup writes stable project guidance so tools know how to call xtctx.', - }, - { - label: 'Skills', - title: 'Selected skills', - body: 'Canonical skills stay under .xtctx/skills and sync into verified tool targets.', - }, - { - label: 'Cache', - title: 'SQLite cache', - body: 'Raw transcripts remain authoritative while SQLite stores rebuildable retrieval data.', - }, -]; - -const installCommands = [ - { - label: 'Set up this project', - command: 'npx -y xtctx setup', - detail: 'Detect tools, choose skills, write MCP config, and repair managed blocks.', - }, - { - label: 'Check the handoff', - command: 'npx -y xtctx status', - detail: 'Inspect wiring, transcript freshness, skill drift, and unsupported targets.', - }, - { - label: 'Remove one tool', - command: 'npx -y xtctx disconnect antigravity', - detail: 'Remove generated xtctx management for a tool without deleting transcripts.', - }, -]; - -const ldJson = JSON.stringify( - { - '@context': 'https://schema.org', - '@graph': [ - { - '@type': 'SoftwareApplication', - name: 'xtctx', - url: 'https://xtctx.com/v9/', - codeRepository: repoUrl, - applicationCategory: 'DeveloperApplication', - operatingSystem: 'Windows, macOS, Linux', - softwareVersion: site.version, - programmingLanguage: 'TypeScript', - license: 'https://opensource.org/licenses/MIT', - description: - 'Local MCP handoff for AI coding agents that need recent raw transcript context when switching tools.', - offers: { '@type': 'Offer', price: '0', priceCurrency: 'USD' }, - }, - { - '@type': 'FAQPage', - mainEntity: faqs.map((faq) => ({ - '@type': 'Question', - name: faq.q, - acceptedAnswer: { '@type': 'Answer', text: faq.a }, - })), - }, - { - '@type': 'HowTo', - name: 'Set up xtctx for local AI coding handoff', - step: [ - { - '@type': 'HowToStep', - position: 1, - name: 'Run setup', - text: 'Run npx -y xtctx setup in the project root.', - }, - { - '@type': 'HowToStep', - position: 2, - name: 'Review status', - text: 'Run npx -y xtctx status to verify MCP wiring, skill sync, and transcript sources.', - }, - { - '@type': 'HowToStep', - position: 3, - name: 'Open another tool', - text: 'The configured agent can call xtctx MCP tools to retrieve recent local transcript detail.', - }, - ], - }, - ], - }, - null, - 2, -); ---- - - - - - - - - xtctx v9 - Move between coding agents without starting over - - - - - - - - - - - - - diff --git a/scripts/probe-embedding-device.mjs b/scripts/probe-embedding-device.mjs index 5a795036..651f1736 100644 --- a/scripts/probe-embedding-device.mjs +++ b/scripts/probe-embedding-device.mjs @@ -47,12 +47,15 @@ import { pathToFileURL } from "node:url"; * Candidates, in the order a fallback chain would try them. * * `auto` is what @huggingface/transformers picks when `device` is not passed — - * which is what xtctx does today, so it is the baseline any change is measured - * against, not a fourth option. + * which is what xtctx did before calibration existed, so it is the baseline any + * change is measured against, not a fourth option. */ const DEVICES = ["cpu", "dml", "webgpu", "auto"]; -const MODEL = "Xenova/all-MiniLM-L6-v2"; +// Kept in step with DEFAULT_EMBEDDING_MODEL in src/handoff/embeddings.ts by +// hand: this runs without a build, so it cannot import the TypeScript. It was +// still measuring MiniLM for two days after the default moved to bge-small. +const MODEL = "Xenova/bge-small-en-v1.5"; const DTYPE = "fp32"; /** * Enough segments for a timing to mean something, few enough that a CI runner diff --git a/src/cli/calibrate.ts b/src/cli/calibrate.ts index b68e0657..376c6d69 100644 --- a/src/cli/calibrate.ts +++ b/src/cli/calibrate.ts @@ -11,8 +11,8 @@ interface CalibrateOptions { * * Nobody needs to run this. Both automatic paths cover it: `xtctx scan * --embed` calibrates before a long embed, and the MCP server calibrates at - * start when it finds a backlog and no verdict, applying the result to the - * not-yet-loaded provider in the same session. + * start on a machine with no verdict, deferring the model's first load until + * the result is in so it applies in the same session (`runtime/background.ts`). * * It stays as a command for the two things automation cannot do: `--force` * after the hardware changes, and showing the measurements to someone who diff --git a/src/handoff/openai-embeddings.ts b/src/handoff/openai-embeddings.ts index 6c9f288e..fcf3355b 100644 --- a/src/handoff/openai-embeddings.ts +++ b/src/handoff/openai-embeddings.ts @@ -88,9 +88,20 @@ export class OpenAiEmbeddingProvider implements EmbeddingProvider { // its connection open in undici until it is garbage collected, and a // vectorizing pass makes this call hundreds of times. await response.body?.cancel().catch(() => {}); + // A 429 is the server asking to be asked later, so the retry waits — + // for `Retry-After` when it says, bounded so one call cannot stall a + // pass. Retrying a rate limit instantly almost always earns a second 429. + // A 5xx retries at once: that one is not a request to slow down. + if (response.status === 429) { + await sleep(retryDelayMs(response.headers.get("retry-after"))); + } response = await this.postEmbeddings(texts); } if (!response.ok) { + // Same reason as above, on the path that gives up: a misconfigured + // endpoint answers every call with an error, and each unread body held a + // connection until GC — one per chunk across a whole pass. + await response.body?.cancel().catch(() => {}); throw new Error(safeHttpError(response.status)); } @@ -134,6 +145,28 @@ export class OpenAiEmbeddingProvider implements EmbeddingProvider { } } +/** Longest a single 429 retry waits, whatever `Retry-After` asks for. */ +const MAX_RETRY_DELAY_MS = 5_000; +/** Wait when a 429 gives no `Retry-After`. */ +const DEFAULT_RETRY_DELAY_MS = 1_000; + +/** `Retry-After` in seconds or as an HTTP date, clamped; see `embedChunk`. */ +export function retryDelayMs(header: string | null, now = Date.now()): number { + if (header === null) { + return DEFAULT_RETRY_DELAY_MS; + } + const seconds = Number(header); + const delay = Number.isFinite(seconds) ? seconds * 1_000 : Date.parse(header) - now; + if (!Number.isFinite(delay) || delay < 0) { + return DEFAULT_RETRY_DELAY_MS; + } + return Math.min(delay, MAX_RETRY_DELAY_MS); +} + +function sleep(ms: number): Promise { + return new Promise((resolve) => setTimeout(resolve, ms)); +} + function safeHttpError(status: number): string { // Status only — response bodies from auth failures often echo the key or // request id material that should not land in `embedding_error` / logs. diff --git a/src/handoff/ranking.ts b/src/handoff/ranking.ts index 00389b17..0210d4dc 100644 --- a/src/handoff/ranking.ts +++ b/src/handoff/ranking.ts @@ -56,16 +56,16 @@ export const MIN_SEMANTIC_COSINE = 0.62; * * 0.28 mrr 0.571 recall@5 0.783 top1 0.433 (false positives 0.05) * 0.32 mrr 0.581 recall@5 0.800 top1 0.433 - * 0.36 mrr 0.598 recall@5 0.850 top1 0.450 <- here + * 0.36 mrr 0.598 recall@5 0.850 top1 0.450 <- was, for MiniLM * 0.40 mrr 0.592 recall@5 0.850 top1 0.433 * - * This sweep predates #318, which rebuilt the eval corpus to use realistic - * session lengths because the old one "has been measuring a world that does - * not exist". The committed baseline moved with it — MiniLM hybrid reads - * 0.333 / 0.533 / 0.183 today — so the SHAPE of the sweep is what survives, - * not the absolute figures. The threshold has not been re-swept against the - * current corpus; 0.36 is inherited rather than re-derived, which is worth - * knowing before treating it as measured. + * That table is MiniLM's, and HISTORICAL: the default model is bge-small now, + * and its floor is 0.64, swept on 2026-09-21 with `scripts/embedding-bakeoff.ts` + * against the current corpus (see `DEFAULT_EMBEDDING_MODEL` for the figures). + * It is kept because it is the clearest record of why the number belongs to + * the model. It also predates #318, which rebuilt the eval corpus to use + * realistic session lengths — so even for MiniLM its absolute figures are not + * comparable with today's baseline. * * The trap worth naming: held at 0.36 while the default was mpnet, that model * looked like it regressed false positives to 0.10. It had not — the diff --git a/src/handoff/status.ts b/src/handoff/status.ts index d9630c6c..d0f2cb0a 100644 --- a/src/handoff/status.ts +++ b/src/handoff/status.ts @@ -114,7 +114,7 @@ export async function buildStatus(inputs: StatusInputs): Promise retrieval_units: retrievalUnitCount, vectorized_units: vectorizedUnitCount, vector_ms_per_unit: numericSetting(db, "vector_ms_per_unit"), - vector_segment_backlog: countUnvectorizedSegments(db, vectorModel), + vector_segment_backlog: countUnvectorizedSegments(db, vectorModel, scopedRoot), vector_ms_per_segment: numericSetting(db, "vector_ms_per_segment"), vector_model: vectorModel, vector_device: vectorDevice, diff --git a/src/handoff/vectors.ts b/src/handoff/vectors.ts index 4dfc2e55..c3970160 100644 --- a/src/handoff/vectors.ts +++ b/src/handoff/vectors.ts @@ -9,6 +9,7 @@ import { } from "./embeddings.js"; import { type CountRow, placeholders, setSetting } from "./schema.js"; import { serializeVector } from "./vector.js"; +import { PROJECT_ROOT_SQL } from "./queries.js"; /** Poll interval while waiting for the model; see `waitUntilEmbeddingReady`. */ const EMBEDDING_WARM_POLL_MS = 100; @@ -77,7 +78,17 @@ export function dropVectorsFromOtherModels(db: DatabaseHandle, model: string): v * Computed in SQL rather than by reading content out: the whole point is to * cost the backlog without loading it. */ -export function countUnvectorizedSegments(db: DatabaseHandle, model: string): number { +export function countUnvectorizedSegments( + db: DatabaseHandle, + model: string, + /** + * Count only this project's windows. `xtctx status` shows a per-project + * window count beside this, and without the scope the two disagreed in any + * index that holds another project's sessions (a copied `.xtctx/`, a renamed + * root): "N windows outstanding" with a duration computed from more than N. + */ + scopedRoot?: string, +): number { const row = db .prepare( // Integer ceiling, then the same cap `capSegments` applies. At least one @@ -88,9 +99,10 @@ export function countUnvectorizedSegments(db: DatabaseHandle, model: string): nu ON v.unit_id = u.id AND v.model = ? AND v.content_hash = u.content_hash - WHERE v.unit_id IS NULL`, + WHERE v.unit_id IS NULL + ${scopedRoot === undefined ? "" : `AND u.session_ref IN (SELECT session_ref FROM sessions WHERE ${PROJECT_ROOT_SQL} = ?)`}`, ) - .get(model) as CountRow | undefined; + .get(...(scopedRoot === undefined ? [model] : [model, scopedRoot])) as CountRow | undefined; return row?.count ?? 0; } diff --git a/src/mcp/server.ts b/src/mcp/server.ts index 8597cdcd..638d4296 100644 --- a/src/mcp/server.ts +++ b/src/mcp/server.ts @@ -210,17 +210,23 @@ export function createToolHandlers( const handlers = new Map(); if (dependencies.unconfiguredProjectRoot) { - const notice = notConfigured(dependencies.unconfiguredProjectRoot); for (const name of TOOL_NAMES) { - handlers.set(name, notice); + handlers.set(name, asRequestedFormat(name, notConfigured(dependencies.unconfiguredProjectRoot), { + status: "not_configured", + project_root: dependencies.unconfiguredProjectRoot, + setup_command: "npx -y xtctx setup", + })); } return handlers; } if (dependencies.configError) { - const notice = configUnreadable(dependencies.configError); for (const name of TOOL_NAMES) { - handlers.set(name, notice); + handlers.set(name, asRequestedFormat(name, configUnreadable(dependencies.configError), { + status: "config_unreadable", + config_path: dependencies.configError.configPath, + error: dependencies.configError.message, + })); } return handlers; } @@ -374,6 +380,40 @@ function configUnreadable(details: { ].join("\n"); } +/** + * Answer a notice in the format the caller asked for. + * + * Both notices used to be one prose string for every tool. `xtctx_handoff_manifest` + * defaults to JSON and documents a versioned contract for orchestrators, so a + * program calling it in an unconfigured directory got English it could not + * parse and no `structuredContent` — while an agent reading markdown was fine. + * The prose stays in the payload, because the words are what tell a person + * what to do. + */ +function asRequestedFormat( + toolName: string, + prose: ToolHandler, + fields: Record, +): ToolHandler { + return async (params: ToolParams) => { + const text = await prose(params); + const format = (params as { format?: unknown } | undefined)?.format; + const wantsJson = + format === "json" || (format === undefined && toolName === "xtctx_handoff_manifest"); + if (!wantsJson) { + return text; + } + // Paths go through the same untrusted-text handling as the prose copy. + const safeFields = Object.fromEntries( + Object.entries(fields).map(([key, value]) => [ + key, + typeof value === "string" ? inlineSafe(value) : value, + ]), + ); + return { ...safeFields, message: text }; + }; +} + function missingDependency(dependency: string): ToolHandler { return async () => { // Thrown (not returned) so the client sees isError, not a success shape. diff --git a/tests/handoff/device-calibration.test.ts b/tests/handoff/device-calibration.test.ts index 69b37335..458d0536 100644 --- a/tests/handoff/device-calibration.test.ts +++ b/tests/handoff/device-calibration.test.ts @@ -11,12 +11,14 @@ * So every ambiguous case here resolves to the CPU, and the tests below are * mostly about ambiguity rather than about picking a winner. */ -import { mkdtemp, rm, writeFile, mkdir } from "node:fs/promises"; +import { mkdtemp, rm, utimes, writeFile, mkdir } from "node:fs/promises"; import { existsSync } from "node:fs"; import { tmpdir } from "node:os"; import { join } from "node:path"; import { afterEach, beforeEach, describe, expect, it } from "vitest"; import { + CalibrationBusyError, + calibrateEmbeddingDevice, chooseDevice, deviceCandidates, deviceFingerprint, @@ -167,3 +169,36 @@ describe("workerArgv", () => { expect(existsSync(entry)).toBe(true); }); }); + +describe("the machine-wide calibration lock", () => { + it("refuses to calibrate while another process holds a fresh lock", async () => { + // Several agents start MCP servers at once on a fresh machine. Without the + // lock each calibrates, all loading the model together — and each CPU arm + // is timed while the others compete for the same cores. + await mkdir(join(home, ".xtctx"), { recursive: true }); + await writeFile(join(home, ".xtctx", "device.json.lock"), "12345", "utf-8"); + + await expect( + // A 1ms worker timeout, so that if the lock were ignored the run ends + // quickly instead of loading a model. + calibrateEmbeddingDevice({ home, segmentCount: 1, timeoutMs: 1 }), + ).rejects.toBeInstanceOf(CalibrationBusyError); + }); + + it("takes over a lock abandoned by a process that was killed", async () => { + // The server exits two seconds after its client disconnects, whatever it + // is doing, so a lock can outlive its owner. It must not block for ever. + const lock = join(home, ".xtctx", "device.json.lock"); + await mkdir(join(home, ".xtctx"), { recursive: true }); + await writeFile(lock, "12345", "utf-8"); + const longAgo = new Date(Date.now() - 60 * 60 * 1000); + await utimes(lock, longAgo, longAgo); + + const verdict = await calibrateEmbeddingDevice({ home, segmentCount: 1, timeoutMs: 1 }); + + // Every worker timed out at 1ms, so nothing was measured — and the choice + // falls back to the CPU, as it must when there is no comparison. + expect(verdict.device).toBe("cpu"); + expect(existsSync(lock)).toBe(false); + }); +}); diff --git a/tests/handoff/openai-embeddings.test.ts b/tests/handoff/openai-embeddings.test.ts index c95f553f..7c56fbe1 100644 --- a/tests/handoff/openai-embeddings.test.ts +++ b/tests/handoff/openai-embeddings.test.ts @@ -2,6 +2,7 @@ import { afterEach, describe, expect, it, vi } from "vitest"; import { OpenAiEmbeddingProvider, openAiEmbeddingIdentity, + retryDelayMs, } from "@xtctx/handoff/openai-embeddings"; import { DEFAULT_EMBEDDING_MODEL, TransformersEmbeddingProvider } from "@xtctx/handoff/embeddings"; import { parseEmbeddingConfig } from "@xtctx/handoff/embedding-config"; @@ -134,7 +135,7 @@ describe("OpenAiEmbeddingProvider", () => { globalThis.fetch = vi.fn(async () => { calls += 1; if (calls === 1) { - return new Response("rate limited", { status: 429 }); + return new Response("rate limited", { status: 429, headers: { "retry-after": "0" } }); } return jsonResponse({ data: [{ embedding: [1, 1] }] }); }) as typeof fetch; @@ -223,6 +224,54 @@ describe("OpenAiEmbeddingProvider", () => { }); }); +describe("retrying a rate limit", () => { + it("waits for what Retry-After asks, in seconds or as a date", () => { + expect(retryDelayMs("2")).toBe(2_000); + const now = Date.parse("2026-09-23T10:00:00.000Z"); + expect(retryDelayMs("Wed, 23 Sep 2026 10:00:03 GMT", now)).toBe(3_000); + }); + + it("never waits longer than the cap, however long it is asked to", () => { + // One stalled call must not stall a whole vectorizing pass. + expect(retryDelayMs("3600")).toBe(5_000); + }); + + it("waits a default second when the header is absent or unreadable", () => { + expect(retryDelayMs(null)).toBe(1_000); + expect(retryDelayMs("soon")).toBe(1_000); + expect(retryDelayMs("-5")).toBe(1_000); + }); +}); + +describe("error responses", () => { + const originalFetch = globalThis.fetch; + afterEach(() => { + globalThis.fetch = originalFetch; + }); + + it("releases the body of a response it gives up on", async () => { + // A misconfigured endpoint answers every chunk with an error, and each + // unread body held its connection until garbage collection. + let cancelled = false; + globalThis.fetch = vi.fn(async () => { + const body = new ReadableStream({ + cancel() { + cancelled = true; + }, + }); + return new Response(body, { status: 404 }); + }) as typeof fetch; + + const provider = new OpenAiEmbeddingProvider({ + baseUrl: "http://localhost:11434/v1", + model: "nomic-embed-text", + }); + + await expect(provider.embedBatch(["x"])).rejects.toThrow(/HTTP 404/); + expect(cancelled).toBe(true); + }); +}); + describe("parseEmbeddingConfig", () => { it("rejects a literal apiKey because the config file is committable", () => { expect(() => diff --git a/tests/handoff/segment-backlog-scope.test.ts b/tests/handoff/segment-backlog-scope.test.ts new file mode 100644 index 00000000..96e9288c --- /dev/null +++ b/tests/handoff/segment-backlog-scope.test.ts @@ -0,0 +1,59 @@ +/** + * The segment backlog counts this project's windows, like the window count + * beside it. + * + * `xtctx status` prints "N windows outstanding, about X left", where N is + * scoped to the project and X was computed from every unvectorized window in + * the database. One index can legitimately hold another project's sessions — + * a copied `.xtctx/`, a renamed root — and then the duration described more + * work than the count it was printed next to. + */ +import { mkdtemp, rm } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import Database from "better-sqlite3"; +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { openDatabase } from "@xtctx/handoff/schema"; +import { countUnvectorizedSegments } from "@xtctx/handoff/vectors"; +import { normalizeRootForCompare } from "@xtctx/handoff/queries"; + +let dir = ""; + +beforeEach(async () => { + dir = await mkdtemp(join(tmpdir(), "xtctx-segment-scope-")); +}); + +afterEach(async () => { + await rm(dir, { recursive: true, force: true }); +}); + +function addWindow(db: Database.Database, projectRoot: string, ref: string, chars: number): void { + const now = "2026-09-23T00:00:00.000Z"; + db.prepare( + `INSERT INTO sessions (session_ref, tool, source_session_id, project_root, started_at, last_activity_at, updated_at) + VALUES (?, 'codex', ?, ?, ?, ?, ?)`, + ).run(ref, ref, projectRoot, now, now, now); + db.prepare( + `INSERT INTO retrieval_units (id, session_ref, tool, message_start_index, message_end_index, started_at, ended_at, content, content_hash, updated_at) + VALUES (?, ?, 'codex', 0, 8, ?, ?, ?, ?, ?)`, + ).run(`${ref}-u`, ref, now, now, "x".repeat(chars), `${ref}-h`, now); +} + +describe("segment backlog", () => { + it("counts only this project's windows when asked to", () => { + const db = openDatabase(join(dir, "index.db")) as unknown as Database.Database; + try { + // 3 segments here, 5 in a project that happens to share the database. + addWindow(db, "/repo/mine", "codex:mine", 3_000); + addWindow(db, "/repo/theirs", "codex:theirs", 5_000); + + const all = countUnvectorizedSegments(db, "test-model"); + const mine = countUnvectorizedSegments(db, "test-model", normalizeRootForCompare("/repo/mine")); + + expect(all).toBe(8); + expect(mine).toBe(3); + } finally { + db.close(); + } + }); +}); diff --git a/tests/mcp/config-error-notice.test.ts b/tests/mcp/config-error-notice.test.ts index 44b4a067..9bbdd22c 100644 --- a/tests/mcp/config-error-notice.test.ts +++ b/tests/mcp/config-error-notice.test.ts @@ -14,6 +14,11 @@ import { describe, expect, it } from "vitest"; import { createToolHandlers } from "@xtctx/mcp/server"; +/** The prose, whether the notice came back as text or as a JSON payload. */ +function prose(answer: unknown): string { + return typeof answer === "string" ? answer : String((answer as { message?: unknown }).message); +} + const DETAILS = { projectRoot: "/repo", configPath: "/repo/.xtctx/config.yaml", @@ -26,7 +31,7 @@ describe("an unreadable config over MCP", () => { expect(handlers.size).toBeGreaterThan(0); for (const [name, handler] of handlers) { - const answer = String(await handler({})); + const answer = prose(await handler({})); expect(answer, name).toContain(".xtctx/config.yaml"); expect(answer, name).toContain("Flow sequence must end with a ]"); @@ -37,7 +42,7 @@ describe("an unreadable config over MCP", () => { it("says the history is unread rather than absent", async () => { const [, handler] = [...createToolHandlers({ configError: DETAILS })][0]; - const answer = String(await handler({})); + const answer = prose(await handler({})); expect(answer).toMatch(/not an empty history/); expect(answer).toContain("Nothing has been changed."); @@ -48,7 +53,33 @@ describe("an unreadable config over MCP", () => { // agent "helpfully" rewriting it would be widening their own access. const [, handler] = [...createToolHandlers({ configError: DETAILS })][0]; - expect(String(await handler({}))).toMatch(/Do not edit it unprompted/); + expect(prose(await handler({}))).toMatch(/Do not edit it unprompted/); + }); + + it("answers the manifest in JSON, which is what an orchestrator parses", async () => { + // The manifest defaults to JSON and documents a versioned contract. These + // notices used to be one prose string for every tool, so an orchestrator + // calling it in a broken project got English it could not parse. + const handler = createToolHandlers({ configError: DETAILS }).get("xtctx_handoff_manifest")!; + + const answer = (await handler({})) as Record; + + expect(answer.status).toBe("config_unreadable"); + expect(answer.config_path).toBe("/repo/.xtctx/config.yaml"); + expect(String(answer.message)).toContain("Do not edit it unprompted"); + }); + + it("answers any tool in JSON when asked for JSON", async () => { + const handler = createToolHandlers({ unconfiguredProjectRoot: "/repo" }).get( + "xtctx_recent_sessions", + )!; + + const answer = (await handler({ format: "json" })) as Record; + + expect(answer.status).toBe("not_configured"); + expect(answer.setup_command).toBe("npx -y xtctx setup"); + // And stays prose when not asked. + expect(typeof (await handler({}))).toBe("string"); }); it("leaves the not-configured case alone, which means something different", async () => { @@ -56,6 +87,6 @@ describe("an unreadable config over MCP", () => { const handlers = createToolHandlers({ unconfiguredProjectRoot: "/repo" }); const [, handler] = [...handlers][0]; - expect(String(await handler({}))).toContain("not configured for xtctx"); + expect(prose(await handler({}))).toContain("not configured for xtctx"); }); }); diff --git a/tsconfig.test.json b/tsconfig.test.json index f9f137cc..43840be7 100644 --- a/tsconfig.test.json +++ b/tsconfig.test.json @@ -14,6 +14,6 @@ // against the same contracts the product is. "types": ["node"] }, - "include": ["src/**/*", "tests/**/*"], + "include": ["src/**/*", "tests/**/*", "scripts/**/*.ts"], "exclude": ["node_modules", "dist", "web", "landing"] } diff --git a/ux-walkthrough.md b/ux-walkthrough.md index 8354a85d..bf9974ac 100644 --- a/ux-walkthrough.md +++ b/ux-walkthrough.md @@ -24,18 +24,21 @@ content through xtctx without any manual export. 2. **Check the wiring.** `xtctx status` shows config path, index counts, per-tool detection, any last scrape error, skill-sync drift, and a final `Next` line. On a plugin-only project `Config` reads - `missing (run xtctx setup)` while the tools still work — that line - describes the managed blocks, not the MCP surface. -3. **Work normally.** Nothing appears in the process list and `.xtctx/state/` - timestamps do not move: no daemon runs, and nothing happens until an agent - asks. + `missing (run xtctx setup)`, and that is the whole story there: the MCP + tools answer every call by naming `xtctx setup` until it has been run, + because a project nobody opted in has no index to read. +3. **Work normally.** No daemon runs. Each agent starts its own xtctx MCP + server, which scans when it starts — and, the first time on a machine, + measures which device embeds fastest — then works through any vector + backlog that fits a fifteen-minute budget, and stops with the agent. 4. **Hand off.** Ask the next tool what you were working on and it returns the - other tool's work rather than asking you. With setup run the agent answers - straight from the managed block; otherwise it calls + other tool's work rather than asking you. The managed block tells the agent + the tools exist, and Claude Code's session-start hook also names the most + recent session; the agent then calls `xtctx_recent_sessions`, which returns the *other* tool's sessions with timestamps and branches. `xtctx_session_detail` on one of those `session_ref`s returns the raw messages; `xtctx_search_sessions` returns - keyword or semantic matches. Indexing happens lazily inside these calls, so + keyword or semantic matches. Indexing also happens inside these calls, so the first on a large history returns partial results and later ones return more — a thin first answer means the index is still filling, not that the history is empty. Orchestrators use `xtctx_handoff_manifest` for stable From 58570939fae73a808d8ae922a13fc78e384c759f Mon Sep 17 00:00:00 2001 From: Felix Stubner Date: Thu, 24 Sep 2026 01:03:25 +0100 Subject: [PATCH 33/34] fix: an urgent embed stops waiting for calibration, and lock tests spawn no workers CI on #383 failed two ways. On ubuntu the first explicit vector search waited for a device measurement it did not need and hit the 60s limit; embedBatch now abandons the deferral, so only the background warm-up waits. On windows the calibration-lock tests timed real devices through tsx, and killing tsx orphans its ONNX child, starving the 2-core runner until time-budgeted scans in other tests came back empty. The lock tests now pass devices: []. --- src/handoff/device.ts | 9 +++- src/handoff/embeddings.ts | 31 ++++++++++-- tests/handoff/device-calibration.test.ts | 12 ++--- tests/handoff/device-retarget.test.ts | 60 +++++++++++++++--------- 4 files changed, 78 insertions(+), 34 deletions(-) diff --git a/src/handoff/device.ts b/src/handoff/device.ts index ad5dde0d..be84086e 100644 --- a/src/handoff/device.ts +++ b/src/handoff/device.ts @@ -280,6 +280,13 @@ export async function calibrateEmbeddingDevice(options: { segmentCount?: number; timeoutMs?: number; onProgress?: (device: EmbeddingDevice) => void; + /** + * Devices to time; this platform's candidates by default. Tests of the lock + * pass none: timing real devices spawned ONNX workers through tsx, and on + * Windows killing tsx orphans its child, which starved a 2-core CI runner + * until unrelated time-budgeted scans came back empty. + */ + devices?: EmbeddingDevice[]; } = {}): Promise { const segmentCount = options.segmentCount ?? 16; const timeoutMs = options.timeoutMs ?? 5 * 60 * 1000; @@ -287,7 +294,7 @@ export async function calibrateEmbeddingDevice(options: { const release = await acquireCalibrationLock(options.home); try { - for (const device of deviceCandidates()) { + for (const device of options.devices ?? deviceCandidates()) { options.onProgress?.(device); const result = await timeDevice(device, segmentCount, timeoutMs); measured.push({ device, ...result }); diff --git a/src/handoff/embeddings.ts b/src/handoff/embeddings.ts index 335e5419..90333010 100644 --- a/src/handoff/embeddings.ts +++ b/src/handoff/embeddings.ts @@ -289,14 +289,34 @@ export class TransformersEmbeddingProvider implements EmbeddingProvider { * one that has always worked. */ deferDeviceUntil(device: Promise): void { - // Handled here rather than at the load, which may never happen: a process - // that exits before embedding anything would otherwise report a failed - // calibration as an unhandled rejection. - this.devicePending = device.catch(() => undefined); + let abandon: () => void = () => {}; + const abandoned = new Promise((resolve) => { + abandon = () => resolve(undefined); + }); + this.abandonDeferral = abandon; + // Rejections handled here rather than at the load, which may never happen: + // a process that exits before embedding anything would otherwise report a + // failed calibration as an unhandled rejection. + this.devicePending = Promise.race([device.catch(() => undefined), abandoned]); } private devicePending: Promise | null = null; + /** + * Stop waiting for calibration and load on the device already configured. + * + * Called by anything that needs a vector NOW. The deferral exists so the + * background warm-up loads the model on the measured device, and nobody is + * waiting on that. An explicit `vector` search is different: an agent is + * holding a tool call open, calibration takes about a minute on a fresh + * machine, and many MCP hosts give up on a call at sixty seconds. Measured on + * a GitHub ubuntu runner, the first vector search did exactly that — "did not + * answer within 60s" — because it was waiting for a measurement it did not + * need. A caller that is waiting wins; the verdict is still written and + * applies from the next session. + */ + private abandonDeferral: (() => void) | null = null; + async embed(text: string): Promise { const [vector] = await this.embedBatch([text]); return vector; @@ -307,6 +327,9 @@ export class TransformersEmbeddingProvider implements EmbeddingProvider { return []; } + // Someone needs vectors now; see `abandonDeferral`. A no-op once the + // device is known, which is every call after the first load. + this.abandonDeferral?.(); const extractor = await this.getExtractor(); const vectors: Float32Array[] = []; for (let start = 0; start < texts.length; start += MAX_BATCH_SIZE) { diff --git a/tests/handoff/device-calibration.test.ts b/tests/handoff/device-calibration.test.ts index 458d0536..5c51e035 100644 --- a/tests/handoff/device-calibration.test.ts +++ b/tests/handoff/device-calibration.test.ts @@ -179,9 +179,9 @@ describe("the machine-wide calibration lock", () => { await writeFile(join(home, ".xtctx", "device.json.lock"), "12345", "utf-8"); await expect( - // A 1ms worker timeout, so that if the lock were ignored the run ends - // quickly instead of loading a model. - calibrateEmbeddingDevice({ home, segmentCount: 1, timeoutMs: 1 }), + // No devices, so that if the lock were ignored nothing is spawned: real + // workers orphaned on Windows starved the CI runner. + calibrateEmbeddingDevice({ home, devices: [] }), ).rejects.toBeInstanceOf(CalibrationBusyError); }); @@ -194,10 +194,10 @@ describe("the machine-wide calibration lock", () => { const longAgo = new Date(Date.now() - 60 * 60 * 1000); await utimes(lock, longAgo, longAgo); - const verdict = await calibrateEmbeddingDevice({ home, segmentCount: 1, timeoutMs: 1 }); + const verdict = await calibrateEmbeddingDevice({ home, devices: [] }); - // Every worker timed out at 1ms, so nothing was measured — and the choice - // falls back to the CPU, as it must when there is no comparison. + // Nothing was measured, so the choice falls back to the CPU, as it must + // when there is no comparison. expect(verdict.device).toBe("cpu"); expect(existsSync(lock)).toBe(false); }); diff --git a/tests/handoff/device-retarget.test.ts b/tests/handoff/device-retarget.test.ts index 95bb39a3..a3772fce 100644 --- a/tests/handoff/device-retarget.test.ts +++ b/tests/handoff/device-retarget.test.ts @@ -1,15 +1,18 @@ /** - * The model loads on the device calibration chose, even when something asks - * for the model before calibration finishes. + * The model loads on the device calibration chose — unless someone is + * waiting for a vector, in which case it loads now on the device it has. * * The first version of same-session calibration retargeted the provider only * if nothing had started loading yet. The MCP server's own warm scan always * started loading first — every scan ends by calling `warm()` — so the - * retarget was refused on every run, the verdict applied only to the next - * session, and the comments and README said otherwise. The unit test for it - * called the provider in isolation and never exercised that ordering. + * retarget was refused on every run, and the verdict applied only to the next + * session while the comments and README said otherwise. The test for it called + * the provider in isolation and never exercised that ordering. * - * These tests put the load FIRST, which is the order that actually happens. + * The fix that followed, deferring every load to calibration, overcorrected: + * an explicit `vector` search then waited for a measurement it did not need, + * and on a fresh GitHub ubuntu runner it "did not answer within 60s". So the + * background warm-up waits and a caller holding a tool call open does not. */ import { describe, expect, it } from "vitest"; import { @@ -30,53 +33,64 @@ function recordingPipeline(): { factory: PipelineFactory; devices: Array new Promise((resolve) => setTimeout(resolve, 10)); + describe("deferring the model load until the device is known", () => { - it("loads on the calibrated device even when the load was requested first", async () => { + it("makes the background warm-up wait, then load on the calibrated device", async () => { + // The order that actually happens: the warm scan asks for the model while + // calibration is still running. const { factory, devices } = recordingPipeline(); const provider = new TransformersEmbeddingProvider(undefined, undefined, undefined, factory); let resolveDevice: (device: string) => void = () => {}; provider.deferDeviceUntil(new Promise((resolve) => (resolveDevice = resolve))); - // The warm scan asks for the model before calibration is done. - const embedding = provider.embed("some text"); - await new Promise((resolve) => setTimeout(resolve, 10)); + provider.warm(); + await tick(); expect(devices).toEqual([]); resolveDevice("dml"); - await embedding; + await tick(); expect(devices).toEqual(["dml"]); expect(provider.device).toBe("dml"); }); - it("keeps the configured device when calibration fails", async () => { - // A failed measurement is not a reason to stop using the device that has - // always worked. + it("does not make a caller that needs a vector now wait for calibration", async () => { + // A calibration that never finishes, standing in for one that takes a + // minute on a fresh machine behind a 60-second tool-call limit. const { factory, devices } = recordingPipeline(); const provider = new TransformersEmbeddingProvider(undefined, undefined, "cpu", factory); + provider.deferDeviceUntil(new Promise(() => {})); - provider.deferDeviceUntil(Promise.reject(new Error("could not measure"))); - await provider.embed("some text"); + await provider.embed("an explicit vector search"); expect(devices).toEqual(["cpu"]); }); - it("keeps the configured device when calibration yields nothing", async () => { + it("keeps the configured device when calibration fails", async () => { + // A failed measurement is not a reason to stop using the device that has + // always worked. const { factory, devices } = recordingPipeline(); - const provider = new TransformersEmbeddingProvider(undefined, undefined, undefined, factory); + const provider = new TransformersEmbeddingProvider(undefined, undefined, "cpu", factory); - provider.deferDeviceUntil(Promise.resolve(undefined)); - await provider.embed("some text"); + provider.deferDeviceUntil(Promise.reject(new Error("could not measure"))); + provider.warm(); + await tick(); - expect(devices).toEqual([undefined]); + expect(devices).toEqual(["cpu"]); }); - it("loads once, however many callers were waiting", async () => { + it("loads once, however many callers there are", async () => { const { factory, devices } = recordingPipeline(); const provider = new TransformersEmbeddingProvider(undefined, undefined, undefined, factory); - provider.deferDeviceUntil(Promise.resolve("webgpu")); + let resolveDevice: (device: string) => void = () => {}; + provider.deferDeviceUntil(new Promise((resolve) => (resolveDevice = resolve))); + provider.warm(); + provider.warm(); + resolveDevice("webgpu"); + await tick(); await Promise.all([provider.embed("a"), provider.embed("b"), provider.embedBatch(["c", "d"])]); expect(devices).toEqual(["webgpu"]); From 71feb0aed2a7459df35a82e7b2e8523386872d6e Mon Sep 17 00:00:00 2001 From: Felix Stubner Date: Thu, 24 Sep 2026 01:17:52 +0100 Subject: [PATCH 34/34] fix(demo): retry the temp-dir removal, and give the server a temp home On Windows the killed server's handles on the index outlive its exit by a moment, so the first rmdir failed with EBUSY on GitHub's windows runner and locally. The server also ran with the real home, so it indexed and calibrated there; it now uses the demo's temp home. --- scripts/public-demo-smoke.mjs | 16 +++++++++++++++- 1 file changed, 15 insertions(+), 1 deletion(-) diff --git a/scripts/public-demo-smoke.mjs b/scripts/public-demo-smoke.mjs index 6dc64d10..a89e9ee7 100644 --- a/scripts/public-demo-smoke.mjs +++ b/scripts/public-demo-smoke.mjs @@ -40,6 +40,16 @@ try { command: process.execPath, args: [cliPath], cwd: projectRoot, + // The server's home is the temp one too, so that it indexes and calibrates + // there rather than in the home of whoever runs this — which also makes the + // run below exercise a first start on a fresh machine every time. + env: { + ...process.env, + HOME: homeDir, + USERPROFILE: homeDir, + APPDATA: join(homeDir, "AppData", "Roaming"), + LOCALAPPDATA: join(homeDir, "AppData", "Local"), + }, }); const client = new Client( { name: "xtctx-public-demo", version: "0.0.0" }, @@ -122,7 +132,11 @@ try { if (keepTemp) { console.log(`kept temp project: ${projectRoot}`); } else { - await rm(tempRoot, { recursive: true, force: true }); + // Retried because on Windows the killed server's handles on the index + // outlive its exit by a moment, and the first rmdir fails with EBUSY. It + // did on GitHub's windows runner and here; the directory was removable + // again by the time a separate process tried. + await rm(tempRoot, { recursive: true, force: true, maxRetries: 10, retryDelay: 200 }); } }