From aa5d42afb525d88f7f9ca581ff7175770bf14785 Mon Sep 17 00:00:00 2001 From: 0thernet Date: Sat, 26 Sep 2026 22:05:45 -0400 Subject: [PATCH] Show Oh's full 500-question LongMemEval-S study on benchmarks Vendor Oh's memory-longmemeval-s-500-v1.json (hraness/oh 21c500c) byte for byte, parse it into a typed BenchmarkStudy, and lead the Oh section of /benchmarks with it: Oh semantic 88.87% vs BM25 86.13% (mean of three runs), with the pre-registered 2-of-3 measure (+2.8 points, 95% interval 0.0 to 5.6) stated as not ruling out a tie. The in-sample lab pipeline figure is framed as neither Oh's nor Wordcell's score. The 60-question pilot stays as the only matched Oh-vs-Supermemory run. README, changelog, evidence doc, compare page, blog post, and llms.txt updated to match. Co-Authored-By: Claude Opus 5.5 --- CHANGELOG.md | 6 +- README.md | 7 +- .../oh/memory-longmemeval-s-500-v1.json | 1316 +++++++++++++++++ docs/evaluations/oh/sources.json | 10 + docs/evidence.md | 5 +- kb/plans/supermemory-competitive-launch.md | 13 + site/app/benchmarks/page.tsx | 61 +- site/app/blog/articles.ts | 3 +- site/app/blog/blog.generated.ts | 2 +- site/app/compare/supermemory/page.tsx | 6 +- site/app/docs/docs.generated.ts | 4 +- site/app/readme.generated.ts | 2 +- site/content/blog/free-local-agent-memory.md | 4 +- site/public/llms.txt | 6 +- site/scripts/sync-blog.ts | 23 +- site/tests/blog.test.tsx | 7 + site/tests/launch-pages.test.tsx | 119 +- site/wordcell/format.ts | 6 +- site/wordcell/oh-evidence.ts | 294 +++- 19 files changed, 1865 insertions(+), 29 deletions(-) create mode 100644 docs/evaluations/oh/memory-longmemeval-s-500-v1.json diff --git a/CHANGELOG.md b/CHANGELOG.md index 6b83244..ad90f4e 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -43,8 +43,10 @@ commands name the next step. Supermemory, Mem0, and Zep. - The [benchmarks page](https://wordcell.io/benchmarks) on wordcell.io shows Wordcell's payload and SciFact results beside the published results of Oh, - the embedded memory framework, each with its source data and limits. New - pages compare Wordcell with + the embedded memory framework, each with its source data and limits. It + leads with Oh's 500-question LongMemEval-S study, where Oh semantic + retrieval scored 88.87% and BM25 86.13%; on the measure Oh named before the + run, its interval does not rule out a tie. New pages compare Wordcell with [Supermemory](https://wordcell.io/compare/supermemory), [Basic Memory](https://wordcell.io/compare/basic-memory), and [Mem0](https://wordcell.io/compare/mem0), and the diff --git a/README.md b/README.md index 321fa54..d4db6b1 100644 --- a/README.md +++ b/README.md @@ -278,8 +278,11 @@ workflow that meets your needs. The [benchmarks page](https://wordcell.io/benchmarks) shows the payload and SciFact results beside the published results of Oh, the embedded memory -framework, each with its source data and limits. Oh's scores measure its own -memory-retrieval path, not a Wordcell vault. +framework, each with its source data and limits. On all 500 LongMemEval-S +questions, Oh semantic retrieval scored 88.87% and BM25 86.13% with the same +reader and budget; on the measure Oh named before the run, its interval does +not rule out a tie. Oh's scores measure its own memory-retrieval path, not a +Wordcell vault. ## How the files fit together diff --git a/docs/evaluations/oh/memory-longmemeval-s-500-v1.json b/docs/evaluations/oh/memory-longmemeval-s-500-v1.json new file mode 100644 index 0000000..696226d --- /dev/null +++ b/docs/evaluations/oh/memory-longmemeval-s-500-v1.json @@ -0,0 +1,1316 @@ +{ + "protocol": "oh.longmemeval-s-500-public-result.v1", + "completed": "2026-09-26", + "report": "benchmarks/LONGMEMEVAL_S_500_RESULT_V1.md", + "dataset": { + "name": "LongMemEval-S", + "paper": "https://arxiv.org/abs/2410.10813", + "file": "longmemeval_s_cleaned.json", + "repository": "xiaowu0162/longmemeval-cleaned", + "revision": "98d7416c24c778c2fee6e6f3006e7a073259d48f", + "sha256": "d6f21ea9d60a0d56f34a05b609c79c88a451d2ae03597821ea3d5a9678c3a442", + "bytes": 277383467, + "questions": 500, + "questionTypes": { + "knowledge-update": 78, + "multi-session": 133, + "single-session-assistant": 56, + "single-session-preference": 30, + "single-session-user": 70, + "temporal-reasoning": 133 + }, + "abstentionQuestions": 30, + "historiesWithEverySessionOnOneDay": 94, + "histories": { + "meanSessions": 47.71, + "meanUserMessages": 244.83, + "meanUserBytes": 61536, + "meanAssistantBytes": 428022, + "assistantShareOfBytes": 0.8743 + }, + "evidenceSpeakers": { + "userOnly": 425, + "assistantOnly": 51, + "both": 3, + "noneMarked": 21 + } + }, + "exposure": { + "priorStudies": "earlier Oh studies scored all 500 questions and read some of them one by one", + "allQuestionsOpened": "2026-09-25", + "split": { + "seed": 17, + "developmentQuestions": 100, + "otherQuestions": 400, + "use": "order of work only; every idea ran on the development questions first" + }, + "inSample": true + }, + "reader": { + "model": "openai/gpt-5-mini", + "transport": "Vercel AI Gateway", + "identity": "gateway-alias", + "snapshotPinned": false, + "reasoningEffort": "medium", + "maxOutputTokens": 8192, + "profile": "gpt5-mini-explicit-abstention-composition-v1-reader", + "profileSha256": "ad71aadff7f96ef5c206f2ecdc40a3030ecf6adf0a91df0afc8800223c0b279a" + }, + "judge": { + "model": "openai/gpt-4o", + "transport": "Vercel AI Gateway", + "identity": "gateway-alias", + "snapshotPinned": false, + "temperature": 0, + "maxOutputTokens": 16, + "profile": "gpt4o-gateway-native-rubric-16-judge-v1", + "profileSha256": "b7ca5dbb946b3555ae590b91c3c53d5567f7e7f6dc067fb11bb3f0848d3e2dfd", + "prompts": "benchmarks/profiles/longmemeval-judge-v1.json", + "officialScriptCommit": "9e0b455f", + "parsing": "exact yes/no", + "paperJudge": "gpt-4o-2024-08-06" + }, + "runs": { + "repeatsPerQuestion": 3, + "attempts": "first attempt only; a missing or failed answer counts as incorrect", + "callStates": { + "pipelineFirstPassReader": { + "completed": 1500 + }, + "pipelineReReadReader": { + "completed": 501 + }, + "ohSemanticReader": { + "completed": 1500 + }, + "bm25Reader": { + "completed": 1500 + }, + "judge": "every judge call completed on its first attempt" + }, + "emptyAnswers": 0, + "answersCutOffByOutputLimit": 0 + }, + "scoring": { + "headline": "mean of three runs: correct answers out of 1,500", + "secondary": "questions answered correctly in at least two of three runs", + "freezePrimary": "questions answered correctly in at least two of three runs", + "reason": "the protocol card scores the per-question mean, and for the pipeline it is the lower of the two measures" + }, + "systems": [ + { + "id": "oh-reading-pipeline", + "label": "Oh reading pipeline", + "protocol": "user log and fused retrieval, then four rule-chosen re-reads", + "runs": "confirmation run: a fresh run of the frozen design on all 500 questions, 2026-09-26", + "answers": 1500, + "correctAnswers": 1396, + "percent": 93.07, + "questionsCorrectInTwoOrThreeRuns": 474, + "correctPerRun": [ + 459, + 469, + 468 + ], + "typeAveragePercent": 93.69, + "byType": [ + { + "questionType": "knowledge-update", + "questions": 78, + "answers": 234, + "correctAnswers": 218, + "percent": 93.16, + "questionsCorrectInTwoOrThreeRuns": 73 + }, + { + "questionType": "multi-session", + "questions": 133, + "answers": 399, + "correctAnswers": 346, + "percent": 86.72, + "questionsCorrectInTwoOrThreeRuns": 118 + }, + { + "questionType": "single-session-assistant", + "questions": 56, + "answers": 168, + "correctAnswers": 164, + "percent": 97.62, + "questionsCorrectInTwoOrThreeRuns": 55 + }, + { + "questionType": "single-session-preference", + "questions": 30, + "answers": 90, + "correctAnswers": 83, + "percent": 92.22, + "questionsCorrectInTwoOrThreeRuns": 30 + }, + { + "questionType": "single-session-user", + "questions": 70, + "answers": 210, + "correctAnswers": 203, + "percent": 96.67, + "questionsCorrectInTwoOrThreeRuns": 69 + }, + { + "questionType": "temporal-reasoning", + "questions": 133, + "answers": 399, + "correctAnswers": 382, + "percent": 95.74, + "questionsCorrectInTwoOrThreeRuns": 129 + } + ], + "abstention": { + "questions": 30, + "answers": 90, + "correctAnswers": 80, + "percent": 88.89, + "questionsCorrectInTwoOrThreeRuns": 27 + }, + "developmentQuestions": { + "questions": 100, + "answers": 300, + "correctAnswers": 283, + "percent": 94.33, + "questionsCorrectInTwoOrThreeRuns": 98 + }, + "otherQuestions": { + "questions": 400, + "answers": 1200, + "correctAnswers": 1113, + "percent": 92.75, + "questionsCorrectInTwoOrThreeRuns": 376 + }, + "memoryBytes": { + "firstPassMean": 168265, + "firstPassMedian": 168794.5, + "firstPassMax": 179997, + "firstPassMin": 108816, + "firstPassBudget": 180000, + "reReadViewBudget": 100000 + } + }, + { + "id": "first-pass-only", + "label": "First pass only", + "protocol": "user log and fused retrieval, one reader call", + "runs": "the confirmation run before any re-read", + "answers": 1500, + "correctAnswers": 1368, + "percent": 91.2, + "questionsCorrectInTwoOrThreeRuns": 458, + "correctPerRun": [ + 450, + 459, + 459 + ], + "typeAveragePercent": 89.35, + "byType": [ + { + "questionType": "knowledge-update", + "questions": 78, + "answers": 234, + "correctAnswers": 217, + "percent": 92.74, + "questionsCorrectInTwoOrThreeRuns": 72 + }, + { + "questionType": "multi-session", + "questions": 133, + "answers": 399, + "correctAnswers": 342, + "percent": 85.71, + "questionsCorrectInTwoOrThreeRuns": 114 + }, + { + "questionType": "single-session-assistant", + "questions": 56, + "answers": 168, + "correctAnswers": 160, + "percent": 95.24, + "questionsCorrectInTwoOrThreeRuns": 53 + }, + { + "questionType": "single-session-preference", + "questions": 30, + "answers": 90, + "correctAnswers": 63, + "percent": 70.0, + "questionsCorrectInTwoOrThreeRuns": 21 + }, + { + "questionType": "single-session-user", + "questions": 70, + "answers": 210, + "correctAnswers": 202, + "percent": 96.19, + "questionsCorrectInTwoOrThreeRuns": 69 + }, + { + "questionType": "temporal-reasoning", + "questions": 133, + "answers": 399, + "correctAnswers": 384, + "percent": 96.24, + "questionsCorrectInTwoOrThreeRuns": 129 + } + ], + "abstention": { + "questions": 30, + "answers": 90, + "correctAnswers": 81, + "percent": 90.0, + "questionsCorrectInTwoOrThreeRuns": 27 + }, + "developmentQuestions": { + "questions": 100, + "answers": 300, + "correctAnswers": 276, + "percent": 92.0, + "questionsCorrectInTwoOrThreeRuns": 94 + }, + "otherQuestions": { + "questions": 400, + "answers": 1200, + "correctAnswers": 1092, + "percent": 91.0, + "questionsCorrectInTwoOrThreeRuns": 364 + }, + "memoryBytes": { + "mean": 168265, + "median": 168794.5, + "max": 179997, + "min": 108816, + "budget": 180000 + } + }, + { + "id": "oh-semantic-96k", + "label": "Oh semantic retrieval", + "protocol": "protocol card: top 100 turns within 96,000 bytes, one reader call", + "runs": "development questions from the 2026-09-25 baseline run; the other 400 from a run frozen at 2026-09-26T08:28:30Z", + "answers": 1500, + "correctAnswers": 1333, + "percent": 88.87, + "questionsCorrectInTwoOrThreeRuns": 445, + "correctPerRun": [ + 445, + 442, + 446 + ], + "typeAveragePercent": 86.84, + "byType": [ + { + "questionType": "knowledge-update", + "questions": 78, + "answers": 234, + "correctAnswers": 213, + "percent": 91.03, + "questionsCorrectInTwoOrThreeRuns": 71 + }, + { + "questionType": "multi-session", + "questions": 133, + "answers": 399, + "correctAnswers": 336, + "percent": 84.21, + "questionsCorrectInTwoOrThreeRuns": 112 + }, + { + "questionType": "single-session-assistant", + "questions": 56, + "answers": 168, + "correctAnswers": 160, + "percent": 95.24, + "questionsCorrectInTwoOrThreeRuns": 53 + }, + { + "questionType": "single-session-preference", + "questions": 30, + "answers": 90, + "correctAnswers": 57, + "percent": 63.33, + "questionsCorrectInTwoOrThreeRuns": 19 + }, + { + "questionType": "single-session-user", + "questions": 70, + "answers": 210, + "correctAnswers": 200, + "percent": 95.24, + "questionsCorrectInTwoOrThreeRuns": 67 + }, + { + "questionType": "temporal-reasoning", + "questions": 133, + "answers": 399, + "correctAnswers": 367, + "percent": 91.98, + "questionsCorrectInTwoOrThreeRuns": 123 + } + ], + "abstention": { + "questions": 30, + "answers": 90, + "correctAnswers": 82, + "percent": 91.11, + "questionsCorrectInTwoOrThreeRuns": 27 + }, + "developmentQuestions": { + "questions": 100, + "answers": 300, + "correctAnswers": 270, + "percent": 90.0, + "questionsCorrectInTwoOrThreeRuns": 91 + }, + "otherQuestions": { + "questions": 400, + "answers": 1200, + "correctAnswers": 1063, + "percent": 88.58, + "questionsCorrectInTwoOrThreeRuns": 354 + }, + "memoryBytes": { + "mean": 83058, + "median": 88520, + "max": 96000, + "min": 32980, + "budget": 96000 + } + }, + { + "id": "bm25-96k", + "label": "BM25 retrieval", + "protocol": "protocol card: top 100 turns within 96,000 bytes, one reader call", + "runs": "full-500 baseline run", + "answers": 1500, + "correctAnswers": 1292, + "percent": 86.13, + "questionsCorrectInTwoOrThreeRuns": 431, + "correctPerRun": [ + 429, + 433, + 430 + ], + "typeAveragePercent": 85.66, + "byType": [ + { + "questionType": "knowledge-update", + "questions": 78, + "answers": 234, + "correctAnswers": 217, + "percent": 92.74, + "questionsCorrectInTwoOrThreeRuns": 72 + }, + { + "questionType": "multi-session", + "questions": 133, + "answers": 399, + "correctAnswers": 299, + "percent": 74.94, + "questionsCorrectInTwoOrThreeRuns": 101 + }, + { + "questionType": "single-session-assistant", + "questions": 56, + "answers": 168, + "correctAnswers": 159, + "percent": 94.64, + "questionsCorrectInTwoOrThreeRuns": 53 + }, + { + "questionType": "single-session-preference", + "questions": 30, + "answers": 90, + "correctAnswers": 59, + "percent": 65.56, + "questionsCorrectInTwoOrThreeRuns": 20 + }, + { + "questionType": "single-session-user", + "questions": 70, + "answers": 210, + "correctAnswers": 205, + "percent": 97.62, + "questionsCorrectInTwoOrThreeRuns": 68 + }, + { + "questionType": "temporal-reasoning", + "questions": 133, + "answers": 399, + "correctAnswers": 353, + "percent": 88.47, + "questionsCorrectInTwoOrThreeRuns": 117 + } + ], + "abstention": { + "questions": 30, + "answers": 90, + "correctAnswers": 81, + "percent": 90.0, + "questionsCorrectInTwoOrThreeRuns": 27 + }, + "developmentQuestions": { + "questions": 100, + "answers": 300, + "correctAnswers": 255, + "percent": 85.0, + "questionsCorrectInTwoOrThreeRuns": 85 + }, + "otherQuestions": { + "questions": 400, + "answers": 1200, + "correctAnswers": 1037, + "percent": 86.42, + "questionsCorrectInTwoOrThreeRuns": 346 + }, + "memoryBytes": { + "mean": 92042, + "median": 95965, + "max": 96000, + "min": 8844, + "budget": 96000 + } + } + ], + "reReads": { + "order": [ + "advice", + "first-decline", + "second-decline", + "recall" + ], + "answersReReadAtLeastOnce": 378, + "answersReReadOnce": 261, + "answersReReadTwice": 111, + "answersReReadThreeTimes": 6, + "questionsWithAReRead": 136, + "maximumReReadsPerAnswer": 3, + "designMaximumReReadsPerAnswer": 4, + "failedReReadKeepsEarlierAnswer": true, + "rulesSee": "question text or current answer text only; never the question type or reference answer", + "wholeSessionView": { + "budgetBytes": 100000, + "messageCharacterCap": 12000, + "steps": [ + "up to four sessions dated within three days of a date the question refers to (skipped when every session is dated the same calendar day)", + "the three sessions whose messages contain the most question content words, scaled for session length", + "sessions in fused retrieval order until the budget is full" + ] + } + }, + "stages": [ + { + "step": "first-pass", + "memory": "first-pass memory", + "correctAnswers": 1368, + "percent": 91.2, + "questionsCorrectInTwoOrThreeRuns": 458 + }, + { + "step": "advice", + "memory": "first-pass memory", + "correctAnswers": 1388, + "percent": 92.53, + "questionsCorrectInTwoOrThreeRuns": 467, + "questionsChosen": 30, + "answersReRead": 90, + "answersFixed": 24, + "answersBroken": 4 + }, + { + "step": "first-decline", + "memory": "whole sessions", + "correctAnswers": 1393, + "percent": 92.87, + "questionsCorrectInTwoOrThreeRuns": 469, + "questionsChosen": 53, + "answersReRead": 128, + "answersFixed": 11, + "answersBroken": 6, + "viewMeanBytes": 98879 + }, + { + "step": "second-decline", + "memory": "whole sessions", + "correctAnswers": 1393, + "percent": 92.87, + "questionsCorrectInTwoOrThreeRuns": 473, + "questionsChosen": 45, + "answersReRead": 115, + "answersFixed": 6, + "answersBroken": 6, + "viewMeanBytes": 98854 + }, + { + "step": "recall", + "memory": "whole sessions", + "correctAnswers": 1396, + "percent": 93.07, + "questionsCorrectInTwoOrThreeRuns": 474, + "questionsChosen": 56, + "answersReRead": 168, + "answersFixed": 3, + "answersBroken": 0, + "viewMeanBytes": 98898 + } + ], + "comparisons": [ + { + "left": "oh-reading-pipeline", + "right": "bm25-96k", + "pairedQuestions": 500, + "meanOfThreeRuns": { + "differencePoints": 6.93, + "interval95": [ + 4.6, + 9.4 + ], + "questionsWithMoreCorrectRuns": 70, + "questionsWithFewerCorrectRuns": 26, + "questionsEqual": 404 + }, + "correctInTwoOrThreeRuns": { + "differencePoints": 8.6, + "interval95": [ + 5.8, + 11.6 + ], + "questionsGained": 51, + "questionsLost": 8, + "questionsEqual": 441 + } + }, + { + "left": "oh-semantic-96k", + "right": "bm25-96k", + "pairedQuestions": 500, + "meanOfThreeRuns": { + "differencePoints": 2.73, + "interval95": [ + 0.53, + 5.07 + ], + "questionsWithMoreCorrectRuns": 47, + "questionsWithFewerCorrectRuns": 37, + "questionsEqual": 416 + }, + "correctInTwoOrThreeRuns": { + "differencePoints": 2.8, + "interval95": [ + 0.0, + 5.6 + ], + "questionsGained": 31, + "questionsLost": 17, + "questionsEqual": 452 + } + }, + { + "left": "oh-reading-pipeline", + "right": "first-pass-only", + "pairedQuestions": 500, + "meanOfThreeRuns": { + "differencePoints": 1.87, + "interval95": [ + 0.87, + 3.0 + ], + "questionsWithMoreCorrectRuns": 21, + "questionsWithFewerCorrectRuns": 7, + "questionsEqual": 472 + }, + "correctInTwoOrThreeRuns": { + "differencePoints": 3.2, + "interval95": [ + 1.8, + 4.8 + ], + "questionsGained": 16, + "questionsLost": 0, + "questionsEqual": 484 + } + }, + { + "left": "first-pass-only", + "right": "bm25-96k", + "pairedQuestions": 500, + "meanOfThreeRuns": { + "differencePoints": 5.07, + "interval95": [ + 2.8, + 7.4 + ], + "questionsWithMoreCorrectRuns": 58, + "questionsWithFewerCorrectRuns": 29, + "questionsEqual": 413 + }, + "correctInTwoOrThreeRuns": { + "differencePoints": 5.4, + "interval95": [ + 2.6, + 8.2 + ], + "questionsGained": 39, + "questionsLost": 12, + "questionsEqual": 449 + } + }, + { + "left": "oh-reading-pipeline", + "right": "oh-semantic-96k", + "pairedQuestions": 500, + "meanOfThreeRuns": { + "differencePoints": 4.2, + "interval95": [ + 2.47, + 6.0 + ], + "questionsWithMoreCorrectRuns": 49, + "questionsWithFewerCorrectRuns": 20, + "questionsEqual": 431 + }, + "correctInTwoOrThreeRuns": { + "differencePoints": 5.8, + "interval95": [ + 3.6, + 8.0 + ], + "questionsGained": 33, + "questionsLost": 4, + "questionsEqual": 463 + } + }, + { + "left": "first-pass-only", + "right": "oh-semantic-96k", + "pairedQuestions": 500, + "meanOfThreeRuns": { + "differencePoints": 2.33, + "interval95": [ + 0.8, + 3.93 + ], + "questionsWithMoreCorrectRuns": 42, + "questionsWithFewerCorrectRuns": 21, + "questionsEqual": 437 + }, + "correctInTwoOrThreeRuns": { + "differencePoints": 2.6, + "interval95": [ + 0.6, + 4.6 + ], + "questionsGained": 20, + "questionsLost": 7, + "questionsEqual": 473 + } + } + ], + "comparisonMethod": "paired by question; 95% intervals from 10,000 bootstrap resamples of questions within each type, seed 20260926", + "firstPassMemory": { + "userLogBytes": { + "mean": 74836, + "median": 73510.5, + "max": 130657, + "min": 56086 + }, + "meanUserMessages": 244.83, + "meanSessions": 47.71, + "meanResolvedDateLines": 24.09, + "retrievedReplies": { + "meanBytes": 93241, + "meanReplies": 50.48, + "capBytes": 96000 + }, + "fusion": "reciprocal rank fusion of BM25 and Oh semantic top-100 lists, k = 60; user messages dropped" + }, + "questionRuleFit": { + "advice": { + "questionsChosen": 30, + "byType": { + "single-session-preference": 30 + } + }, + "recall": { + "questionsChosen": 56, + "byType": { + "single-session-assistant": 55, + "temporal-reasoning": 1 + } + }, + "note": "rules read only question text but were written after reading all 500 questions" + }, + "instructions": { + "base": { + "id": "explicit-abstention-composition-v1", + "source": "scripts/benchmarks/evolution-reader-contracts.ts", + "sha256": "77fdcac530792b977f41e7e436ad7f7a2953f542cfa57e49790f5a9583dbe0b0", + "characters": 1794 + }, + "join": "base instruction, then each listed paragraph, separated by single spaces", + "digest": "SHA-256 of the instruction encoded as a JSON string", + "paragraphs": { + "userLog": "The memory has two parts. First, the complete log of the user's own messages from every session of the conversation history, verbatim and in chronological order, grouped by session with the session date and its distance from the question date; lines beginning with \"Resolved dates\" give absolute dates computed by the system for the relative time expressions in the preceding message (exact unless marked ≈; an ambiguous expression lists both readings). Second, retrieved assistant replies that best match the question, verbatim, only a subset of the history; each names the user message it answered. Enumerate and count across sessions using the user log, determine current state from the latest dated message, and use the retrieved assistant replies for what the assistant said, suggested or explained. Treat a question as unanswerable only when neither part contains the asked-about fact or event.", + "calibration": "Report a computed result as the direct value (for example \"$315\"), not as an inequality or a hedge, even when an input was approximate; at most add a brief note after the value. Interpret a time window in the question loosely: when the only matching events fall within a few days of the literal window, use them and mention their dates rather than declining. When the memory covers the subject of the question but a qualifier in the question (such as first, last, or a period) is not stated explicitly, answer under the most natural reading of the evidence and add that assumption in a few words. Decline only when no observation or turn addresses the subject of the question at all. For a recommendation, propose options consistent with the remembered preferences and avoid options the user said they dislike or want to move beyond.", + "answerFormat": "Begin the response with the answer itself: never open with a heading, a list of facts, or a restatement of the evidence. When the question presupposes an event (for example the first order from a service) and the memory records exactly one event matching the description, treat that event as the one meant and answer from it. When the memory notes that session timestamps are unreliable, date and order events by the dates stated inside the messages and by their tense, and count an event described as already having happened as having happened even if its stated date is later than the session timestamp. When an answer depends on an unstated detail such as a birthday or an exact day, give the single most likely value rather than a range.", + "advice": "This question asks for advice or recommendations for the user. Never decline and never say the conversation lacks information: the answer is advice, not a remembered fact. First, search the memory for everything about the user that bears on this request: what they own or already use, what they have done or tried, their plans, constraints, stated likes and dislikes, and what they said they want to try next or move beyond. Then answer concisely with tailored suggestions that explicitly build on those specifics (for example \"since you already have …\" or \"building on your …\"), put first the options that match what they said they want, and avoid anything they said they dislike, already have, or want to move away from. Do not pad the answer with generic advice or long lists copied from earlier assistant replies; a few well-targeted suggestions are better.", + "wholeSessions": "The memory contains the complete text, user and assistant messages verbatim, of the sessions of the conversation history most relevant to the question, including sessions dated near any time the question refers to; other sessions are not shown. Read every message carefully, including details the user mentions in passing inside longer messages, and the assistant's own earlier answers. If these sessions contain the asked-about fact or event, answer the question directly from them. Before answering, check that the question's premise matches the memory: if the question names a role, item, person, place, date or event that differs from what was actually said (for example a different job title, product, or relative), the conversation does not contain the answer, so reply with the decline sentence.", + "recall": "This question asks what the assistant said, listed, or created earlier. The full text of the relevant sessions is in the memory, including the assistant's own messages verbatim. Find the assistant's content and answer with it directly — reproduce names, lists, progressions or details exactly as the assistant gave them — never decline just because the answer is long, unusual, or inside an earlier assistant reply." + }, + "steps": [ + { + "step": "first-pass", + "id": "eac-userlog-v2n", + "paragraphs": [ + "userLog", + "calibration", + "answerFormat" + ], + "sha256": "551f0c0de5df5642a732ecbee7c5fef9359bb6ce4cfa67526476b20c1719dd84", + "characters": 4271 + }, + { + "step": "advice", + "id": "eac-userlog-advice-v1", + "paragraphs": [ + "userLog", + "answerFormat", + "advice" + ], + "sha256": "4c1404fa181fad5e1ae281aa462a87b9c5e1099ad6349a2d0cda289ed60d9616", + "characters": 4299 + }, + { + "step": "decline", + "id": "eac-zoom-v1", + "paragraphs": [ + "wholeSessions", + "answerFormat" + ], + "sha256": "5058ea6c013da4dac2d933a6ba397ac7b6e36ab5c5cf78259d24c47528335bba", + "characters": 3339 + }, + { + "step": "recall", + "id": "eac-zoom-recall-v1", + "paragraphs": [ + "wholeSessions", + "answerFormat", + "recall" + ], + "sha256": "6f055965d7594ef9bd7a4ac345b8b42d23335c371ba76578d0565a237b9c49de", + "characters": 3755 + } + ], + "developmentFirstPass": { + "id": "eac-userlog-v2", + "sha256": "217105e1e917eed7fd01077406293d321fcce0f37ab3bc211e6e580ade639850", + "differsBy": "calibration example was a dollar amount that is the reference answer to one development question, in place of $315" + } + }, + "rules": { + "flags": "case-insensitive", + "advice": { + "matches": "\\b(suggest|suggestions?|recommend|recommendations?|advice|tips|ideas|what do you think|do you think|should i|could there be a reason|what should i|which one to choose|not sure which|any thoughts)\\b", + "unless": "\\b(remind me|you (?:suggested|recommended|mentioned|told|said|gave|listed|provided)|did you|previous (?:conversation|chat)|last time|we (?:talked|discussed|spoke)|earlier conversation)\\b", + "reads": "question" + }, + "decline": { + "matches": "does not contain enough information|not enough information|insufficient information|cannot determine|can.t determine|not specified in|not mentioned|no information about|does not (?:include|mention|contain|record)", + "reads": "current answer" + }, + "recall": { + "matches": "\\b(remind me|you (?:suggested|recommended|mentioned|told|said|gave|listed|provided|created|wrote|made)|did you|previous (?:conversation|chat)|last time|we (?:talked|discussed|spoke)|earlier conversation|looking back at our previous)\\b", + "reads": "question" + } + }, + "development": { + "developmentQuestionsCorrectInTwoOrThreeRuns": [ + { + "memory": "BM25 retrieval, 96 KB", + "correct": 87 + }, + { + "memory": "Oh semantic retrieval, 96 KB", + "correct": 91 + }, + { + "memory": "BM25 retrieval with resolved-date lines", + "correct": 87 + }, + { + "memory": "notes extracted from the user’s messages by GPT-5 nano, with BM25 replies", + "correct": [ + 90, + 91 + ] + }, + { + "memory": "complete user log with BM25 replies", + "correct": 94 + }, + { + "memory": "complete user log with fused replies and the answer-format paragraph", + "correct": 95 + } + ], + "secondBm25RunOnDevelopmentQuestions": 85, + "fullEvaluations": [ + { + "design": 1, + "description": "user log with BM25 replies, one pass", + "correctPerRun": [ + 458, + 454, + 454 + ], + "questionsCorrectInTwoOrThreeRuns": 459, + "kept": false + }, + { + "design": 2, + "description": "user log with fused replies, one pass", + "correctPerRun": [ + 455, + 456, + 458 + ], + "questionsCorrectInTwoOrThreeRuns": 460, + "kept": true + }, + { + "design": 3, + "description": "2 plus decline re-read", + "correctPerRun": [ + 457, + 456, + 458 + ], + "questionsCorrectInTwoOrThreeRuns": 459, + "kept": false + }, + { + "design": 4, + "description": "2 plus advice re-read", + "correctPerRun": [ + 466, + 465, + 467 + ], + "questionsCorrectInTwoOrThreeRuns": 471, + "kept": false + }, + { + "design": 5, + "description": "2 plus advice and decline re-reads", + "correctPerRun": [ + 470, + 464, + 471 + ], + "questionsCorrectInTwoOrThreeRuns": 474, + "kept": true + }, + { + "design": 6, + "description": "5 plus a counting re-read", + "correctPerRun": [ + 469, + 465, + 468 + ], + "questionsCorrectInTwoOrThreeRuns": 471, + "kept": false + }, + { + "design": 7, + "description": "5 plus an arbiter for questions whose runs disagreed", + "correctPerRun": [ + 455, + 454, + 454 + ], + "questionsCorrectInTwoOrThreeRuns": 454, + "kept": false + }, + { + "design": 8, + "description": "instructions routed by question rule, one pass", + "correctPerRun": [ + 458, + 466, + 471 + ], + "questionsCorrectInTwoOrThreeRuns": 470, + "kept": false + }, + { + "design": 9, + "description": "8 plus decline re-read", + "correctPerRun": [ + 457, + 465, + 467 + ], + "questionsCorrectInTwoOrThreeRuns": 468, + "kept": false + }, + { + "design": 10, + "description": "5 plus a second decline re-read", + "correctPerRun": [ + 470, + 465, + 472 + ], + "questionsCorrectInTwoOrThreeRuns": 475, + "kept": true + }, + { + "design": 11, + "description": "10 plus a current-state re-read", + "correctPerRun": [ + 468, + 462, + 468 + ], + "questionsCorrectInTwoOrThreeRuns": 472, + "kept": false + }, + { + "design": 12, + "description": "10 plus the recall re-read", + "correctPerRun": [ + 472, + 467, + 472 + ], + "questionsCorrectInTwoOrThreeRuns": 476, + "kept": true + }, + { + "design": 13, + "description": "12 plus a model-built ledger in place of the memory, for counting questions whose runs disagreed", + "correctPerRun": [ + 436, + 431, + 439 + ], + "questionsCorrectInTwoOrThreeRuns": 437, + "kept": false + }, + { + "design": 14, + "description": "12 plus a current-state re-read on the first-pass memory", + "correctPerRun": [ + 472, + 466, + 470 + ], + "questionsCorrectInTwoOrThreeRuns": 474, + "kept": false + }, + { + "design": 15, + "description": "12 plus the ledger appended to the user log, for the same questions as 13", + "correctPerRun": [ + 471, + 466, + 470 + ], + "questionsCorrectInTwoOrThreeRuns": 474, + "kept": false + } + ], + "developmentOnlyDesigns": 10, + "champion": 12 + }, + "referenceAnswerScan": { + "when": "after the last development run", + "scope": "every instruction and rule", + "matching": "each reference answer as a whole word or phrase in the instruction and rule text", + "findings": [ + { + "value": "a dollar amount that is the reference answer to one development question", + "where": "calibration paragraph example", + "questionsAffected": 1, + "split": "development", + "correctRunsWithIt": 3, + "correctRunsWithout": 3, + "action": "replaced by $315 in the confirmation run" + }, + { + "value": "a place name", + "where": "an instruction used by designs 8 and 9, the second stage of design 14, and one development-only design", + "questionsAffected": 1, + "designs": [ + 8, + 9, + 14 + ], + "developmentOnlyDesigns": 1, + "action": "all four designs dropped" + }, + { + "value": "a number word that is the entire reference answer", + "where": "user-log paragraph, carried by the reported first pass and advice step", + "questionsAffected": 2, + "split": "test", + "effect": "BM25, Oh semantic retrieval and the pipeline answered both correctly in all three runs; comparisons unaffected" + }, + { + "value": "a common two-word time phrase", + "where": "counting paragraph, used only by dropped design 6", + "questionsAffected": 3, + "split": "test", + "designs": [ + 6 + ], + "effect": "design 6 re-read none of the three questions" + } + ] + }, + "confirmation": { + "design": 12, + "frozenAt": "2026-09-26T08:10:52Z", + "commitment": "report whatever the run scores", + "freshCalls": "new reader and judge calls for every step", + "differencesFromDevelopmentRun": [ + "calibration example, a dollar amount that is the reference answer to one development question, replaced by $315", + "the first decline re-read uses the whole-session view with the content-word step, which the development run did not have yet" + ], + "developmentRun": { + "correctAnswers": 1411, + "percent": 94.07, + "questionsCorrectInTwoOrThreeRuns": 476, + "correctPerRun": [ + 472, + 467, + 472 + ], + "typeAveragePercent": 95.09 + }, + "confirmationRun": { + "correctAnswers": 1396, + "percent": 93.07, + "questionsCorrectInTwoOrThreeRuns": 474 + }, + "labSourceSha256": { + "chain-confirm.sh": "3366f9822ca9808cc21d14c2566dd741ce29a4ae3037d2d7a4096a88fe560113", + "contracts.ts": "dfd07097bc73a51901741bbfd5df2c9990d2afb1c3ea68e42f16400c1f0fed4d", + "programs.ts": "2266acd48ce205814a5a4775ed0387fd513440351c85d611de914eee133b0c0f", + "zoom.ts": "dd016c46ad83eb87d11ad024583c8e0a6acb1801f582aeb7f7d23ac0295b9c26", + "cascade.ts": "16d8f3dbf9c0b73ff574d2e02102e9bed55b207697b5a897eb89a784e9295945", + "reader.ts": "65ebec55c4ecd38798b4eb5a1d6f65f93c8896451e0bb8c430585eb7f200f364", + "judge.ts": "8504d9b926a38359b0f0d2e09d40c06ce316a848f3008551c58a843783a8bfda", + "lib.ts": "e8f8f42c5c2b0998840600112a6d209a510d1bf3842131248279560b01882cf9", + "combine.ts": "c6385bf936457e702dc995cd46ce84efaabef9eeddf8d7cd0ab431bfeb4baed7", + "compose-userlog.ts": "a0c501447f849bf760f2479ad4cebaf658ad6640574e47dcbf054e8cf3c1f38b", + "dates.ts": "a2c8f178ac0821143721db1f01d5e1e14176c75a2d3437c6d081a1f76dd738db", + "contexts.ts": "f93bb9cbc004cdd3e5676c9cd8ec47cd7ecf2fb97a0ad1c39e43505d94396d02" + }, + "semanticRunFrozenAt": "2026-09-26T08:28:30Z" + }, + "cost": { + "unit": "USD micros, as reported by Vercel AI Gateway per call", + "systems": [ + { + "system": "oh-reading-pipeline", + "readerCalls": 2001, + "readerMicros": 16229249, + "firstPassReaderMicros": 13182312, + "reReadReaderMicros": 3046937, + "reReadCalls": 501, + "judgeCalls": 2001, + "judgeMicros": 915081, + "readerMicrosPerAnswer": 10819 + }, + { + "system": "oh-semantic-96k", + "readerCalls": 1500, + "readerMicros": 7306445, + "judgeCalls": 1500, + "judgeMicros": 644549, + "readerMicrosPerAnswer": 4871 + }, + { + "system": "bm25-96k", + "readerCalls": 1500, + "readerMicros": 7606804, + "judgeCalls": 1500, + "judgeMicros": 646032, + "readerMicrosPerAnswer": 5071 + } + ], + "tokensPerReaderCall": { + "pipelineFirstPass": { + "meanInputTokens": 40944, + "meanCachedInputTokens": 14042, + "meanOutputTokens": 856, + "meanReasoningTokens": 804 + }, + "pipelineReRead": { + "meanInputTokens": 26222, + "meanCachedInputTokens": 10234, + "meanOutputTokens": 914, + "meanReasoningTokens": 822 + }, + "ohSemantic": { + "meanInputTokens": 20054, + "meanCachedInputTokens": 7077, + "meanOutputTokens": 725, + "meanReasoningTokens": 675 + }, + "bm25": { + "meanInputTokens": 21837, + "meanCachedInputTokens": 8037, + "meanOutputTokens": 710, + "meanReasoningTokens": 660 + } + }, + "studyTotal": { + "micros": 121829009, + "settledMicros": 121740715, + "heldMicros": 88294, + "byDay": { + "2026-09-25": 84510260, + "2026-09-26": 37318749 + }, + "scope": "every model call made for the study: development designs, the Supermemory attempts and the baselines" + } + }, + "otherPublishedResults": { + "checked": "2026-09-26", + "note": "each chose its own answering model, judge, prompts and configuration, so those scores do not compare directly with these; not a ranking", + "results": [ + { + "system": "Supermemory", + "reported": "97% with GPT-4o answering; 84.6% with GPT-5; 85.2% with Gemini 3 Pro", + "correctOf500": [ + 485, + 423, + 426 + ], + "reader": "GPT-4o; GPT-5; Gemini 3 Pro", + "judge": "GPT-4o, LongMemEval prompts", + "source": "https://supermemory.ai/research/longmembench/", + "label": "the page also calls 97% its overall Recall@20 with aggregation, the retrieval configuration it names" + }, + { + "system": "Mastra observational memory", + "reported": "94.87%, average of the six question types", + "correctOf500": 468, + "reader": "GPT-5 mini", + "judge": "GPT-4o, LongMemEval prompts", + "runBasis": "latest run; the only source here that states its run basis", + "source": "https://mastra.ai/research/observational-memory" + }, + { + "system": "Mem0 platform", + "reported": "94.4% (top 200), 94.8% (top 50)", + "correctOf500": [ + 472, + 474 + ], + "reader": "not stated", + "judge": "not stated", + "source": "https://github.com/mem0ai/memory-benchmarks" + }, + { + "system": "Hindsight 0.4.19", + "reported": "94.6%", + "correctOf500": 473, + "reader": "Gemini 3.1 Pro preview", + "judge": "Gemini 2.5 Flash Lite", + "source": "https://github.com/vectorize-io/agent-memory-benchmark" + }, + { + "system": "Chronos", + "reported": "94.20%; also reports 95.60%", + "correctOf500": [ + 471, + 478 + ], + "reader": "GPT-5 mini", + "judge": "LongMemEval prompts; judge model not named", + "source": "https://arxiv.org/abs/2603.16862" + }, + { + "system": "MemMachine", + "reported": "93.0%", + "correctOf500": 465, + "reader": "GPT-5 mini", + "judge": "GPT-4o mini", + "source": "https://arxiv.org/abs/2604.04853", + "configurationChoice": "about 12 variants scored on the same 500 questions" + }, + { + "system": "Honcho", + "reported": "90.4%", + "correctOf500": 452, + "reader": "Claude Haiku 4.5", + "judge": "GPT-4o, LongMemEval prompts", + "source": "https://plasticlabs.ai/blog/research/Benchmarking-Honcho" + } + ] + }, + "supermemory": { + "pilot": { + "report": "benchmarks/FRAMEWORK_PILOT_RESULT_V1.md", + "questions": 60, + "percent": { + "supermemory": 75.0, + "oh": 71.67, + "bm25": 68.33 + } + }, + "fullRun": "scored during this study and discarded; not reported", + "discardedRuns": [ + { + "id": "m125-smarm-v1", + "questions": 125, + "createdAt": "2026-09-26T00:49Z" + }, + { + "questions": 500 + } + ], + "discardedRunCostsUsd": [ + 2.66, + 2.76 + ], + "discardedFor": "both runs used the same harness with the three departures below", + "departuresFromPilot": [ + "stored each turn as its own document instead of one document per session", + "searched while Supermemory was still generating memories from those documents", + "kept only search results holding one whole stored turn, dropping every generated memory" + ], + "matchedFullRun": false + }, + "packageBoundary": { + "inOh": [ + "Oh semantic retrieval (EmbeddingGemma through the optional local QMD runtime)", + "BM25 over SQLite FTS5 in scripts/benchmarks/" + ], + "labOnly": [ + "user-log builder", + "date rules", + "re-read rules", + "added instructions" + ], + "labPublished": false + }, + "limitations": [ + "In-sample: the added instructions and question rules were written after studying all 500 questions, and two rules match single question types on this benchmark.", + "The reader and judge are gateway aliases; the models behind them can change.", + "The user log fits because LongMemEval-S histories are short; longer histories, such as the 500-session M variant, are untested.", + "The pipeline reads up to 180,000 bytes and makes extra calls; the matched baselines read at most 96,000 bytes once.", + "There is no GPT-5 mini run with the full history as memory.", + "The lab harness is not published; the report gives its instructions and rules.", + "AI agents ran the study and wrote the report; no person or outside group has audited it." + ] +} diff --git a/docs/evaluations/oh/sources.json b/docs/evaluations/oh/sources.json index c5a8d4a..4fbf035 100644 --- a/docs/evaluations/oh/sources.json +++ b/docs/evaluations/oh/sources.json @@ -20,6 +20,16 @@ "gitBlobSha1": "49ca4a9a97125e824b66d524f757aa931bc65483", "sha256": "87cc50aa692e5a97bd949d8bf9d6c2c058eca0fbaa486726c98f64c8e32a0a97", "fetchedOn": "2026-09-26" + }, + { + "file": "memory-longmemeval-s-500-v1.json", + "repository": "hraness/oh", + "commit": "21c500cf38928ab610c15c438557fbed5227ca4b", + "path": "benchmarks/results/memory-longmemeval-s-500-v1.json", + "bytes": 43551, + "gitBlobSha1": "696226da0134c9c97a5a7a53e301177a84bfd858", + "sha256": "c954d955a0b1c06f8e2b07a0206ad9c8a0ddab046ecd48913547b6bae6e1c44f", + "fetchedOn": "2026-09-26" } ] } diff --git a/docs/evidence.md b/docs/evidence.md index 766bbf6..5dd03ea 100644 --- a/docs/evidence.md +++ b/docs/evidence.md @@ -124,7 +124,10 @@ Jev reranking follow separate retrieval paths. Oh's conversation-memory benchmarks evaluate its own memory-retrieval API, reader models, and evaluation protocols. Those scores do not transfer to a -Wordcell vault merely because it uses the same library. Wordcell's +Wordcell vault merely because it uses the same library. The +[benchmarks page](https://wordcell.io/benchmarks) reports Oh's 500-question +LongMemEval-S study, its LoCoMo run, and its smaller pilot with Supermemory as +Oh's results, each with its limits. Wordcell's [SciFact study](reranking.md#evidence-and-limits) compares exact search with hosted Jev reranking on the same public queries and candidate windows. It measures source ranking, without generating answers; this page measures only diff --git a/kb/plans/supermemory-competitive-launch.md b/kb/plans/supermemory-competitive-launch.md index 5fd0e9c..52f6016 100644 --- a/kb/plans/supermemory-competitive-launch.md +++ b/kb/plans/supermemory-competitive-launch.md @@ -1047,6 +1047,19 @@ do not guess them. rebuilt; `bun run check` passed 1,928/0; `cd site && bun run check` passed 145/0 plus runtime 3/0; and the kb percolate, refresh, and check commands passed. +- 2026-09-26 — After close-out, Oh published its 500-question LongMemEval-S + study (hraness/oh `21c500c`). Its result file is vendored under + docs/evaluations/oh/ and registered in sources.json, and + site/wordcell/oh-evidence.ts derives a same-run study from it that leads the + Oh section of /benchmarks, before LoCoMo. The matched arms, Oh semantic + retrieval and BM25 at one byte budget, are the only charted rows. The + frozen two-of-three measure's interval reaches zero, so every surface says + the study does not rule out a tie. The lab pipeline's figure is in-sample + and outside the Oh package: it appears once, on /benchmarks, inside the + sentence that says so, and nowhere else. The earlier pilot stays the only + matched Oh-and-Supermemory run. The README, changelog, docs/evidence.md, + llms.txt, /compare/supermemory, and the launch post (through new + `oh-longmemeval.*` evidence keys) follow. ## Result diff --git a/site/app/benchmarks/page.tsx b/site/app/benchmarks/page.tsx index 478b8a4..2f32bba 100644 --- a/site/app/benchmarks/page.tsx +++ b/site/app/benchmarks/page.tsx @@ -6,6 +6,13 @@ import { scifactDetails, scifactStudy } from "../../wordcell/benchmark-evidence" import { grouped, longDate, prose, signed } from "../../wordcell/format"; import { formatBytes, handoffEvidence } from "../../wordcell/handoff-evidence"; import { + longMemEvalArms, + longMemEvalComparison, + longMemEvalFacts, + longMemEvalLabPipeline, + longMemEvalLimitQuotes, + longMemEvalStudy, + longMemEvalTypes, locomoArms, locomoCategories, locomoFacts, @@ -14,6 +21,7 @@ import { locomoStudy, ohAttribution, ohLinks, + ohLongMemEvalPost, ohSources, pilotInterval, pilotLimitQuotes, @@ -26,7 +34,7 @@ import { publishedRelease } from "../publication"; const pageTitle = "Wordcell and Oh benchmarks, with their limits"; const pageDescription = - "See Wordcell’s own payload and reranking measurements and the Oh memory kernel’s published results, each with its source data and limits."; + `Wordcell’s payload and reranking measurements, and the Oh kernel’s ${longMemEvalFacts.questions}-question LongMemEval-S and LoCoMo results, each with source data and limits.`; export const metadata: Metadata = { title: pageTitle, @@ -53,6 +61,10 @@ const status = publishedRelease === null const handoffShare = Math.round((handoffEvidence.packedBytes / handoffEvidence.fullNoteBytes) * 100); const [miniPaired, nanoPaired] = locomoPaired; const [locomoLimitSaturation, locomoLimitHarness, locomoLimitUnseen] = locomoLimitQuotes; +const [semanticArm, bm25Arm] = longMemEvalArms; +const primary = longMemEvalComparison.primary; +const meanDifference = longMemEvalComparison.mean; + const [pilotLimitSample, pilotLimitGranularity, pilotLimitProfile] = pilotLimitQuotes; function intervalSentence(interval: typeof pilotInterval): string { @@ -61,6 +73,7 @@ function intervalSentence(interval: typeof pilotInterval): string { export default function Benchmarks() { if (miniPaired === undefined || nanoPaired === undefined) throw new TypeError("LoCoMo paired results need both readers."); + if (semanticArm === undefined || bm25Arm === undefined) throw new TypeError("LongMemEval-S results need both matched arms."); return ( + +

On the measure Oh named before the run, questions answered correctly in at least two of {prose(longMemEvalFacts.runsPerQuestion)} runs, Oh semantic retrieval got {grouped(semanticArm.majorityCorrect)} of {grouped(longMemEvalFacts.questions)} and BM25 got {grouped(bm25Arm.majorityCorrect)}. That is {signed(primary.difference, 1)} percentage points, with a {primary.level} interval from {signed(primary.lower, 1)} to {signed(primary.upper, 1)}.{longMemEvalComparison.tieNotRuledOut ? " The interval reaches zero, so this result does not rule out a tie." : ""} Oh semantic retrieval gained {grouped(longMemEvalComparison.gained)} questions and lost {grouped(longMemEvalComparison.lost)}.

+

On the mean of {prose(longMemEvalFacts.runsPerQuestion)} runs shown in the chart, the difference is {signed(meanDifference.difference)} points, with a {meanDifference.level} interval from {signed(meanDifference.lower)} to {signed(meanDifference.upper)}. Oh’s intervals come from resampling questions within each question type.

+

This is Oh’s retrieval measured on its own. It is not a measurement of Wordcell search, and it includes no matched run of Supermemory or any other memory framework.

+
+ +
+ + + + + + + {longMemEvalArms.map((arm) => )} + + + + {longMemEvalTypes.map((type) => ( + + + + {type.percents.map((percent, index) => { + const arm = longMemEvalArms[index]; + return ; + })} + + ))} + +
LongMemEval-S answers judged correct by question type, mean of {prose(longMemEvalFacts.runsPerQuestion)} runs, in percent
Question typeQuestions{arm.system}
{type.name}{grouped(type.questions)}{percent}%
+
+ +

Oh’s report also describes a lab reading pipeline that scored {longMemEvalLabPipeline.percent}% on the mean of {prose(longMemEvalFacts.runsPerQuestion)} runs and answered {grouped(longMemEvalLabPipeline.majorityCorrect)} of {grouped(longMemEvalLabPipeline.questions)} questions correctly in at least two. It is not charted here and is not Oh’s or Wordcell’s score: its instructions and rules were written after studying all {grouped(longMemEvalLabPipeline.questions)} questions, so the figure is in-sample, and its {longMemEvalLabPipeline.labOnlyText} are not part of the Oh package. Oh’s report gives the details.

+ +

Limits

+
    +
  • Oh reports that “{longMemEvalLimitQuotes.exposure},” so none of the {grouped(longMemEvalFacts.questions)} questions is unseen.
  • +
  • Oh’s limits say: “{longMemEvalLimitQuotes.aliases}”
  • +
  • On the lab pipeline, Oh’s limits say: “{longMemEvalLimitQuotes.inSample}” They add: “{longMemEvalLimitQuotes.budget}”
  • +
  • Oh adds: “{longMemEvalLimitQuotes.audit}”
  • +
+

Paired by question with GPT-5 mini, Oh semantic retrieval was right where the BM25 window was wrong on {grouped(miniPaired.better)} questions, wrong where it was right on {grouped(miniPaired.worse)}, and matched it on {grouped(miniPaired.tied)}. With GPT-5 nano the counts were {grouped(nanoPaired.better)}, {grouped(nanoPaired.worse)}, and {grouped(nanoPaired.tied)}.

@@ -147,7 +201,7 @@ export default function Benchmarks() { heading="Matched and published comparisons" headingId="comparisons-title" id="comparisons" - summary="Oh ran one small pilot of Supermemory, Oh, and BM25 under one protocol; it is Oh’s result, not Wordcell’s. Figures that other memory systems publish use their own protocols, so they appear in a table, not a chart." + summary="Oh ran one small pilot of Supermemory, Oh, and BM25 under one protocol; it is Oh’s result, not Wordcell’s. It is smaller and earlier than the study above, it was a development pilot on previously seen questions, and it remains the only matched run of Oh against Supermemory. Figures that other memory systems publish use their own protocols, so they appear in a table, not a chart." >

{intervalSentence(pilotInterval)} {pilotInterval.crossesZero ? `The interval includes zero, so the pilot does not separate ${pilotInterval.left} from ${pilotInterval.right}.` : ""}

@@ -200,6 +254,7 @@ export default function Benchmarks() {
  • Context handoff: method, raw results, and reproduction
  • Reranking study: evidence and limits
  • +
  • Oh’s LongMemEval-S result on all {grouped(longMemEvalFacts.questions)} questions and Oh’s write-up of it
  • Oh’s LoCoMo result and Oh’s LongMemEval pilot result
  • Oh’s benchmark guide
  • {ohSources.map((source) => ( diff --git a/site/app/blog/articles.ts b/site/app/blog/articles.ts index a4a1cf4..9430c4a 100644 --- a/site/app/blog/articles.ts +++ b/site/app/blog/articles.ts @@ -63,6 +63,7 @@ const launchSources = [ { title: "Oh LoCoMo result file, copy in the Wordcell repository", href: launchSource("docs/evaluations/oh/memory-evolution-locomo-sealed-1540-v1.json") }, { title: "Oh framework pilot result file, copy in the Wordcell repository", href: launchSource("docs/evaluations/oh/memory-framework-pilot-v1.json") }, { title: "Query the derived graph", href: launchSource("docs/graph-authority.md") }, + { title: "LongMemEval-S result V1, all 500 questions", publisher: "Oh", href: ohLinks.longMemEvalResult }, { title: "Matched descriptive comparison on LoCoMo", publisher: "Oh", href: ohLinks.locomoResult }, { title: "Framework pilot result V1", publisher: "Oh", href: ohLinks.pilotResult }, { title: "Self-hosting overview", publisher: "Supermemory", href: "https://supermemory.ai/docs/self-hosting/overview" }, @@ -253,7 +254,7 @@ export const blogArticles = [ review: { reviewer: REVIEWER, reviewerType: "ai", reviewedOn: "2026-09-26" }, humanReview: null, reassessOn: "2026-11-07", - harmIfWrong: "A reader could cite Oh's LoCoMo figures as Wordcell's results, or switch from Supermemory expecting hosted extraction and connectors.", + harmIfWrong: "A reader could cite Oh's LoCoMo or LongMemEval-S figures as Wordcell's results, read Oh's lead over BM25 as settled when its interval reaches zero, or switch from Supermemory expecting hosted extraction and connectors.", refreshTriggers: [ "A release that includes wordcell mcp, wordcell import supermemory, or the skill's session-memory reference", "A change to a vendored evidence file under docs/ or to the evidence modules under site/wordcell/", diff --git a/site/app/blog/blog.generated.ts b/site/app/blog/blog.generated.ts index 10519df..6b5ac9e 100644 --- a/site/app/blog/blog.generated.ts +++ b/site/app/blog/blog.generated.ts @@ -2,7 +2,7 @@ export const blogHtml: Readonly> = { "introducing-wordcell": "

    Wordcell keeps decisions, plans, and sources as Markdown files beside your code. Coding agents find them by exact words, by meaning with an optional local model, or from the file they are about to change. Wordcell builds a graph from the links you wrote and gives every result a path back to the note it came from. The files stay where they are, in a format Obsidian, Git, and any text editor can read, and every index can be deleted and rebuilt from them.

    \n

    A rule like "parser retries stop after three attempts" tends to get decided once and then lost. It lives in a chat that closes, a commit message nobody searches, or a note on someone's laptop. The next coding agent to touch the parser starts from the code and never sees it. With Wordcell, the rule is a note the agent can find and cite.

    \n

    Your Markdown stays the record

    \n

    Wordcell reads a folder of Markdown and never moves your notes into a database of its own, as some agent memory tools do. Everything it builds on top is a view it can rebuild: exact search, optional search by meaning with a local model, backlinks, graph queries, and published sites. None of these views writes back into your notes, so losing one costs you a way to look things up and leaves the notes untouched.

    \n

    You can check any Wordcell answer against the file. An exact search result names the note and the line that matched. A graph result names the note that wrote a link, the note it points to, and the line where the link appears, plus a proof that ties the row to the exact version of the file it was read from. When an agent cites a Wordcell result, you open the file and read the sentence yourself.

    \n

    Wordcell was called KB until version 0.20.0. Releases up to 0.19.6 keep the package name @hraness/kb: npm carries them through 0.19.2, and the later 0.19.x versions exist as GitHub Release archives, so older install notes and lockfiles may show that name. Newer releases use @hraness/wordcell, and the vault format keeps its kb names, so an existing vault needs no migration.

    \n

    Who it suits

    \n

    Wordcell is for people who keep decisions, sources, and plans in Markdown or an Obsidian vault and work with coding agents such as Claude Code, Codex, Cursor, or GitHub Copilot. It helps most when the notes explain code: why a module rejects a tempting shortcut, which plan introduced a constraint, which source backed a decision.

    \n

    Some people need less. A small set of notes may be fine with plain Markdown and a text search. If you only want local document retrieval, QMD is a good fit on its own; Wordcell uses it for its optional search by meaning. And if you want a service that silently records everything an agent does, Wordcell is the wrong tool: only what you save becomes part of the record.

    \n

    From one saved rule to a cited answer

    \n

    Wordcell runs as a command-line tool and TypeScript SDK on Bun, with Git. After installing it from the release on GitHub or its npm mirror, one saved rule and one search look like this:

    \n
    wordcell init kb\nwordcell note create notes/parser-contract \\\n  --title "Parser contract" --type concept \\\n  --body "Parser retries stop after three attempts." --root kb\nwordcell search "parser retries" --root kb --mode exact\n
    \n

    The result points to notes/parser-contract, which is an ordinary Markdown file you can open and edit. Exact search needs no model, account, or network request. You can also point the same search at a vault you already have, without initializing or converting it.

    \n

    Links you write become the graph. A plan that mentions [[notes/parser-contract]] shows up when you ask what depends on the rule:

    \n
    wordcell graph query --program backlinks --note notes/parser-contract --root kb --json\n
    \n

    Each row names the source note, the target, and the line of the link. Wordcell computes these queries with Oh, an embedded engine that needs no separate account or service, and by default the whole graph lives in memory for one query and then closes. How Wordcell uses Oh covers what the proofs contain and what they leave out.

    \n

    To tie a note to code, add the paths it explains to its frontmatter:

    \n
    repository_scopes:\n  - packages/parser\n
    \n

    Then an agent about to edit a file in that package can ask for the notes and repository rules that apply to it:

    \n
    wordcell context packages/parser/src/index.ts --root kb --repo .\n
    \n

    The result groups current notes and plans apart from finished or superseded ones, and includes the AGENTS.md files that govern the path. A public Agent Skill teaches compatible agents these commands. Installing it adds instructions only; it does not create a vault, change your notes, or give an agent access to other accounts. xcb can bind one vault you choose and give its workers read-only exact search over it with citations, and saving a note back is always a separate step, as How xcb uses Wordcell explains.

    \n

    Clipped pages and PDFs land in the same folder

    \n

    Sources you read become files in the vault too. wordcell clip saves a web page as a Markdown bundle with its images and a record of how the page was fetched, and wordcell pdf keeps the original PDF beside the extracted text. Search and the graph read them alongside your notes, and the vault's own rules keep captured text as source material rather than your conclusions.

    \n

    Some of what people read sits behind their own logins: a newsletter they subscribe to, a member article, a page already open in their browser. Wordcell can save those pages for the person who is signed in. It can read the tab you have open without navigating it, open a page with a browser profile you select (a profile given by its folder path runs from a temporary copy, so that profile is unchanged), or use your browser's cookies for that site when the page needs nothing else. Capture only reads. It does not post, like, follow, send, delete, or submit anything. When a site answers with a login wall, paywall, or CAPTCHA, Wordcell stops rather than looking for an archived copy elsewhere. For a public page that no direct route can read, it may make one read-only lookup of that exact URL on Archive.today, which tells that service the URL. Screenshots can include private notifications, so review a bundle before sharing it.

    \n

    What stays fixed

    \n

    The project's guidelines keep Markdown and Git as the only record. Search indexes, backlinks, and the graph are views rebuilt from the files, and none of them writes back. Material from other tools follows the same rule. Wordcell's SDK can turn a verified set of Oh records into a Markdown review candidate, and that step never opens the vault, writes a note, or marks anything as reviewed. A person reads the candidate and writes the note. The agent workflow asks agents to start from wordcell context for the path they are changing, and the notes a person wrote about it, before running a broad search.

    \n

    Status and limits

    \n

    Latest release: v0.22.5. Install the versioned archive from GitHub Releases or the npm mirror with Bun 1.3.14 or newer and Git; the documentation has the commands.

    \n

    Wordcell does not write answers. It returns notes, snippets, and graph rows, and your agent writes the answer from them. A graph proof shows that a file said something at a given version, not that the note is correct. Graph queries accept vaults of up to 4,000 notes and mark a result as truncated rather than presenting a partial answer as complete. Search by meaning downloads a local model the first time you use it. An optional reranking step sends query and note snippets to a paid hosted provider and is off by default.

    \n", "how-wordcell-uses-oh": "

    Every Wordcell graph answer comes with a proof you can check against your notes. Wordcell can list the notes that link to a decision, the plans that reach it through a chain of links, and the notes that share its tags. When Wordcell says a retry plan depends on your parser rule, you want to see the line in the plan that makes the link, and you want to know that line still reads the way it did when the answer was computed.

    \n

    Wordcell reads your Markdown and hands the graph work to Oh, which works out the answer and keeps a proof for every row, naming the file that supports it, a fingerprint of that file's contents, and the rule that joined the pieces. The Markdown files stay the record. The graph is a copy Wordcell can throw away and build again from them.

    \n

    What Oh does, for someone who has not used it

    \n

    Oh is a memory framework that applications embed as a library. It stores typed records, derives new facts from rules, and returns each derived answer with the chain of facts and rules that produced it. Wordcell uses the part that stores records and answers graph questions. There is no Oh account to create and no service to run. Wordcell pins one released version of Oh and upgrades only by changing that pin.

    \n

    Oh writes every record in one exact text form, which it calls canonical JSON, and names the record by the SHA-256 fingerprint of that text. Two programs holding the same record produce the same bytes and the same fingerprint, whatever order they assembled its fields in. So a fingerprint in a proof names exactly one record.

    \n

    From a folder of notes to an answer you can check

    \n

    When you run a graph query, Wordcell reads the whole vault as it is at that moment:

    \n
      \n
    1. It fingerprints the text of each Markdown file.
    2. \n
    3. It collects what you wrote: links, typed relationships, tags, and the code paths a note declares under repository_scopes. Each becomes a fact tied to the note it came from, and links keep their line.
    4. \n
    5. It stores each note's facts as an Oh record in canonical JSON, and fingerprints the whole snapshot as one revision.
    6. \n
    7. It turns the named query you asked for into a small set of Oh rules, and Oh evaluates them over that snapshot.
    8. \n
    \n

    By default all of this lives in memory for one query and then closes. Nothing is written into the vault and no cache file appears. Here is the smallest case, a note that links to a rule and a query for what points at the rule:

    \n
    wordcell note create notes/retry-review --title "Retry review" --type concept \\\n  --body "Use [[notes/parser-contract]] when changing retry behavior." --root kb\nwordcell graph query --program backlinks --note notes/parser-contract --root kb --json\n
    \n

    The row names notes/retry-review as the source, notes/parser-contract as the target, and the line where the link was written. Its proof names the source note, the fingerprint of that note's contents, and the fingerprint of the Oh record built from it. A longer answer, such as everything reachable within three links, adds the rule applied at each step and the facts it used. Six named queries are available: backlinks, reachability, relation-closure, scope-route, shared-tags, and shared-concepts.

    \n

    To keep a graph on disk, wordcell graph rebuild writes one to a .wordcell/oh.sqlite file that Git ignores, checks it by replaying it, and only then replaces the previous file. Deleting that file loses nothing, because Wordcell can rebuild it from the notes.

    \n

    Three rules that give a proof its meaning

    \n

    Same files, same answer. Equal snapshots of your notes produce equal rows and equal source proofs. An in-memory graph uses a fixed logical start time so the result can be reproduced in a later session; that time never claims anything about when a note was written.

    \n

    An edit retires the old proof. A proof carries the fingerprint of each file it relies on, taken over the file's exact text. Change the file, even its spacing, and the fingerprint changes, so the old proof no longer matches. A Wordcell session holds one snapshot for its whole life, so reopen it after editing. From code, you can re-check any result against its session:

    \n
    import { openKnowledgeBase } from "@hraness/wordcell";\n\nconst kb = await openKnowledgeBase({ root: "kb" });\ntry {\n  const result = await kb.graphQuery({ program: "backlinks", note: "notes/parser-contract" });\n  console.log(await kb.graphVerifyResult(result)); // false for modified, foreign, or stale evidence\n} finally {\n  await kb.close();\n}\n
    \n

    Answers never write back. A query never adds a link or an inferred relationship to a note. With wordcell percolate --proofs, Wordcell can show shared tags or shared concepts as evidence beside a suggested connection, but whether two notes should link stays your decision, made by editing the Markdown.

    \n

    Two engines held to the same bytes

    \n

    Oh ships its canonical encoder and its query engine twice: a TypeScript reference and a Rust version compiled to WebAssembly. Wordcell's graph queries prefer the Rust engine when it loads and fall back to TypeScript when it does not, with the same source revision and the same limits either way. Wordcell also uses Oh's Rust canonical encoder to fingerprint the review drafts described in the last section, again with a TypeScript fallback.

    \n

    Two encoders are only safe if they agree on every input, because one differing character changes a fingerprint and breaks every proof that cites it. The rule they share is short. Object keys are sorted, array order is kept, there is no extra whitespace, and numbers are written the way JavaScript's JSON writer writes them:

    \n
    import { canonicalJson } from "@hraness/oh";\n\ncanonicalJson({ b: 1, a: [2, 1] }); // '{"a":[2,1],"b":1}'\n
    \n

    Wordcell's own tests hold the Rust encoder it loads from Oh to that rule. They confirm that the WebAssembly bytes Wordcell loads match the SHA-256 recorded in the Oh package, then generate random JSON values, including nested arrays and objects and very large and very small numbers, and require the Rust and TypeScript encoders to return the same text and the same fingerprint for each. Fixed cases such as 1e21, 5e-324, an empty key, and an emoji key are checked on every run. Oh runs its own version of this test; How Oh keeps its TypeScript and Rust encoders byte for byte identical covers that side.

    \n

    What a Wordcell user can check

    \n

    Every row of a graph answer names a file you can open and, for links, the line to read. When an agent cites a Wordcell graph result, you can open the cited note and read the line yourself instead of taking the agent's word for it. A result you saved can be verified against the session it came from, and a stale one fails that check. Your knowledge stays in the notes. If you delete the graph or the Rust engine is unavailable, the notes are unchanged and give the same answers.

    \n

    Where the proofs stop

    \n

    A proof shows that a file said something at a given version. It does not show that the note is right. Wordcell runs only its six reviewed queries; there is no free-form query language. Absences, orphan notes, and counts are computed by Wordcell from the complete snapshot, not proved by Oh. A vault can hold up to 4,000 notes, 100,000 facts, and 64 MiB of text for graph queries, and a query that runs out of work fails instead of returning a partial answer as complete. A truncated result says so in its JSON and exits with code 4.

    \n

    Wordcell's search does not use Oh. Exact search and optional local search by meaning are Wordcell's own, so Oh's memory benchmarks say nothing about Wordcell's search or answers. Wordcell also does not keep an agent's memory in Oh. Wordcell's SDK can turn selected records from an Oh memory store into a Markdown review draft that lists the source records and their fingerprints and fingerprints the draft itself; preparing that draft never opens a vault or writes a note, and whether any of it becomes a note is a separate, reviewed decision. A fresh rebuild of the on-disk graph returns the same rows and source proofs, though the stored graph's own history identifiers can differ.

    \n

    Latest release: v0.22.5. For the full query reference, see Query the derived graph in the Wordcell documentation. Oh is the library behind the graph.

    \n", - "free-local-agent-memory": "

    On Monday a coding agent works out why the release script pins an older compiler. On Tuesday a new session opens in the same repository and starts without that reason, unless someone wrote it where the agent looks. Wordcell keeps that kind of record as Markdown notes in a folder you control. This launch adds three things: a Model Context Protocol (MCP) server so agents read and write those notes through their own tools, an importer for Supermemory exports, and a skill workflow that saves what a session decided as a note.

    \n

    Latest release: v0.22.5. The MCP server, the Supermemory importer, and the skill’s session-memory workflow are available from source until the next release. The migration page starts with the source install, and bunx skills add hraness/wordcell --skill wordcell adds the skill from the main branch.

    \n

    What the launch adds

    \n

    wordcell mcp --root <vault> serves a vault over standard input and output to local MCP clients such as Claude Code, Claude Desktop, Cursor, and Codex. Agents search, list, and read notes, follow links and backlinks, create notes, replace a note body at the revision they read, and add relations between notes. With --repo, the server also returns context for a repository. Every write goes through the same checks as the command line, and --read-only leaves the write tools out. The MCP server reference lists each tool and shows how to connect a client.

    \n

    wordcell import supermemory <export.json> turns documents and memory entries saved from the Supermemory API into notes. Each version of a memory becomes its own note, linked newest to oldest by supersedes relations. Running the import again updates notes you have not edited, skips unchanged ones, and reports notes you changed yourself as conflicts without touching them. The importer reads export files only and makes no network calls. Import from Supermemory lists the fields and where each kind of item lands.

    \n

    The wordcell skill includes a session-memory workflow. When you ask, the agent saves what the conversation decided as a dated session note, links it to the notes it changed, and keeps a profile note with a Stable section and a Recent section. Wordcell extracts nothing on its own: the agent writes the note, and you can read it before anything depends on it. The steps are in the session-memory reference.

    \n

    Two guides cover the rest of a move. Migrate from Supermemory exports your data, imports it, and lists what does not transfer. Sync a vault with Git keeps one vault current on several machines through a private repository.

    \n

    What Wordcell has measured

    \n

    Wordcell’s own measurements are narrow. Across four queries on a seven-note public vault, packed snippets used 79.98% fewer UTF-8 bytes than the same notes in full: 12,126 bytes against 60,584 bytes, measured with Wordcell 0.21.3. That is payload size only. It does not measure tokens, answer quality, speed, or an advantage over another search tool.

    \n

    On 300 BEIR SciFact queries over scientific abstracts, exact search put a relevant abstract first for 33.7% of queries. Optional Jev reranking, which sends each query and candidate snippets to a paid provider, raised that to 53.7% in a study run on September 19, 2026. The study does not establish answer quality. The benchmarks page has both studies with raw results.

    \n

    What Oh measured on its own

    \n

    Wordcell builds its graph with Oh, memory for agents that stores each fact with its sources and history. Oh also has its own memory-retrieval API and publishes conversation-memory results for it. LoCoMo is a benchmark of questions about long conversations held over many sessions. In Oh’s LoCoMo run, published September 10, 2026, Oh’s semantic retrieval had 84.4% of 1,540 answers judged correct with GPT-5 mini as the reader and 81.0% with GPT-5 nano. BM25, a keyword-search baseline over the same conversations, scored 81.6% and 78.1% with the same two readers. A GPT-4o mini judge graded every answer. Oh’s conversation-memory benchmarks evaluate its own memory-retrieval API, reader models, and evaluation protocols. Those scores do not transfer to a Wordcell vault merely because it uses the same library.

    \n

    The run covered 10 conversations once, with no confidence interval, and every question had been seen before: 1,226 in earlier evaluations and 314 during development. The result file records no run date, so the date above is when Oh published it. Oh’s result file says the figures “do not establish fresh confirmation, statistical superiority or benchmark saturation”.

    \n

    Oh also ran a small pilot of its API against Supermemory, dated September 24, 2026. It used 60 questions from LongMemEval-S, a benchmark of questions about long chat histories, and Oh had seen those questions during development. Supermemory answered 75.00% of the 60 questions correctly, Oh 71.67%, and BM25 68.33%. Oh and BM25 each did not finish three of the questions, and those count as misses. GPT-4o wrote and judged the answers, called through a gateway name that is not pinned to one model version. Oh minus Supermemory came to −3.33 percentage points, with a 95% interval from −13.33 to +6.67. The interval includes zero, and Oh’s pilot report says “the paired primary comparison does not separate Oh from Supermemory”. Supermemory ran with one fixed profile and was indexed per session, while Oh and BM25 were indexed per turn, so the pilot says nothing about Supermemory’s defaults or best configuration. The benchmarks page sets out both studies, and its comparison section lists figures other memory systems publish.

    \n

    Why the agent writes the memory

    \n

    Supermemory builds user profiles automatically through ingestion: a model reads your content for facts about you and adds, updates, or removes them. Its graph memory goes further and “infers a fact you never stated in one place, from patterns across memories” (graph memory and user profiles, checked September 26, 2026). That suits an application that wants memory built for it. It also means a stored fact can come from a step you never saw.

    \n

    In Wordcell the agent writes the memory as Markdown, and the file is the memory. You can read a note, diff it, and revert it with Git. update_note_body applies an edit only at the revision the agent read, so an edit based on an older copy is refused instead of overwriting a newer one. A supersedes relation keeps the older claim readable beside the newer one. Supermemory marks the latest fact for retrieval, and its documentation says the history “can remain for audit” (graph memory, checked September 26, 2026). Graph answers carry a proof that names each source note and a digest of its content, so an edited note no longer matches the proof (Query the derived graph). Search by meaning uses an embedding model that runs on your machine. When a memory is wrong, it is a line in a file you can find and fix. Markdown memory for coding agents covers the approach.

    \n

    Limits

    \n

    Wordcell itself has not been measured against Supermemory; the pilot above tested Oh’s API. Wordcell is a command-line tool over a folder, not a hosted memory service for your product’s users. Choose Supermemory when you want extraction and connectors run for you: its documentation points to the hosted platform for “connectors, MCP, and the best-tuned extraction pipeline” (self-hosting overview, checked September 26, 2026). When Supermemory fits better lists more cases. The payload measurement used a seven-note vault, and the SciFact study searched 5,183 public abstracts with exact search, without the graph or Git history. Nothing here measures a large vault of notes your agents wrote and linked, or the graph, search by meaning, or the MCP server at that size.

    \n

    Free, open source, and yours

    \n

    Memory for your agents should be free, open source, and superb. That is a position, not a measurement. Free and open source are facts about Wordcell: it is MIT licensed, and the commands above run on your machine without an account. Superb is the goal, and the figures above do not establish it. The same self-hosting overview says Supermemory’s self-hosted edition is free and open source, so price does not separate the two.

    \n

    To move an existing Supermemory account, start with Migrate from Supermemory.

    \n", + "free-local-agent-memory": "

    On Monday a coding agent works out why the release script pins an older compiler. On Tuesday a new session opens in the same repository and starts without that reason, unless someone wrote it where the agent looks. Wordcell keeps that kind of record as Markdown notes in a folder you control. This launch adds three things: a Model Context Protocol (MCP) server so agents read and write those notes through their own tools, an importer for Supermemory exports, and a skill workflow that saves what a session decided as a note.

    \n

    Latest release: v0.22.5. The MCP server, the Supermemory importer, and the skill’s session-memory workflow are available from source until the next release. The migration page starts with the source install, and bunx skills add hraness/wordcell --skill wordcell adds the skill from the main branch.

    \n

    What the launch adds

    \n

    wordcell mcp --root <vault> serves a vault over standard input and output to local MCP clients such as Claude Code, Claude Desktop, Cursor, and Codex. Agents search, list, and read notes, follow links and backlinks, create notes, replace a note body at the revision they read, and add relations between notes. With --repo, the server also returns context for a repository. Every write goes through the same checks as the command line, and --read-only leaves the write tools out. The MCP server reference lists each tool and shows how to connect a client.

    \n

    wordcell import supermemory <export.json> turns documents and memory entries saved from the Supermemory API into notes. Each version of a memory becomes its own note, linked newest to oldest by supersedes relations. Running the import again updates notes you have not edited, skips unchanged ones, and reports notes you changed yourself as conflicts without touching them. The importer reads export files only and makes no network calls. Import from Supermemory lists the fields and where each kind of item lands.

    \n

    The wordcell skill includes a session-memory workflow. When you ask, the agent saves what the conversation decided as a dated session note, links it to the notes it changed, and keeps a profile note with a Stable section and a Recent section. Wordcell extracts nothing on its own: the agent writes the note, and you can read it before anything depends on it. The steps are in the session-memory reference.

    \n

    Two guides cover the rest of a move. Migrate from Supermemory exports your data, imports it, and lists what does not transfer. Sync a vault with Git keeps one vault current on several machines through a private repository.

    \n

    What Wordcell has measured

    \n

    Wordcell’s own measurements are narrow. Across four queries on a seven-note public vault, packed snippets used 79.98% fewer UTF-8 bytes than the same notes in full: 12,126 bytes against 60,584 bytes, measured with Wordcell 0.21.3. That is payload size only. It does not measure tokens, answer quality, speed, or an advantage over another search tool.

    \n

    On 300 BEIR SciFact queries over scientific abstracts, exact search put a relevant abstract first for 33.7% of queries. Optional Jev reranking, which sends each query and candidate snippets to a paid provider, raised that to 53.7% in a study run on September 19, 2026. The study does not establish answer quality. The benchmarks page has both studies with raw results.

    \n

    What Oh measured on its own

    \n

    Wordcell builds its graph with Oh, memory for agents that stores each fact with its sources and history. Oh also has its own memory-retrieval API and publishes conversation-memory results for it. LoCoMo is a benchmark of questions about long conversations held over many sessions. In Oh’s LoCoMo run, published September 10, 2026, Oh’s semantic retrieval had 84.4% of 1,540 answers judged correct with GPT-5 mini as the reader and 81.0% with GPT-5 nano. BM25, a keyword-search baseline over the same conversations, scored 81.6% and 78.1% with the same two readers. A GPT-4o mini judge graded every answer. Oh’s conversation-memory benchmarks evaluate its own memory-retrieval API, reader models, and evaluation protocols. Those scores do not transfer to a Wordcell vault merely because it uses the same library.

    \n

    The run covered 10 conversations once, with no confidence interval, and every question had been seen before: 1,226 in earlier evaluations and 314 during development. The result file records no run date, so the date above is when Oh published it. Oh’s result file says the figures “do not establish fresh confirmation, statistical superiority or benchmark saturation”.

    \n

    LongMemEval-S is a benchmark of questions about long chat histories. Oh’s study of all 500 of its questions, completed September 26, 2026, gave Oh’s semantic retrieval and BM25 the same byte budget and the same reader, GPT-5 mini, which answered every question three times. A GPT-4o judge graded the answers with LongMemEval’s own prompts. Oh’s semantic retrieval had 88.87% of answers judged correct and BM25 86.13%. On the measure Oh chose before the run, questions answered correctly in at least two of the three runs, Oh minus BM25 came to +2.8 percentage points, with a 95% interval from 0.0 to +5.6. That interval reaches zero, so the study does not rule out a tie. Earlier Oh studies had scored all of these questions, and this one reported no Supermemory result.

    \n

    Before that study, Oh ran a smaller pilot of its API against Supermemory, dated September 24, 2026. It used 60 questions from LongMemEval-S, and Oh had seen those questions during development. Supermemory answered 75.00% of the 60 questions correctly, Oh 71.67%, and BM25 68.33%. Oh and BM25 each did not finish three of the questions, and those count as misses. GPT-4o wrote and judged the answers, called through a gateway name that is not pinned to one model version. Oh minus Supermemory came to −3.33 percentage points, with a 95% interval from −13.33 to +6.67. The interval includes zero, and Oh’s pilot report says “the paired primary comparison does not separate Oh from Supermemory”. Supermemory ran with one fixed profile and was indexed per session, while Oh and BM25 were indexed per turn, so the pilot says nothing about Supermemory’s defaults or best configuration. The benchmarks page sets out all three studies, and its comparison section lists figures other memory systems publish.

    \n

    Why the agent writes the memory

    \n

    Supermemory builds user profiles automatically through ingestion: a model reads your content for facts about you and adds, updates, or removes them. Its graph memory goes further and “infers a fact you never stated in one place, from patterns across memories” (graph memory and user profiles, checked September 26, 2026). That suits an application that wants memory built for it. It also means a stored fact can come from a step you never saw.

    \n

    In Wordcell the agent writes the memory as Markdown, and the file is the memory. You can read a note, diff it, and revert it with Git. update_note_body applies an edit only at the revision the agent read, so an edit based on an older copy is refused instead of overwriting a newer one. A supersedes relation keeps the older claim readable beside the newer one. Supermemory marks the latest fact for retrieval, and its documentation says the history “can remain for audit” (graph memory, checked September 26, 2026). Graph answers carry a proof that names each source note and a digest of its content, so an edited note no longer matches the proof (Query the derived graph). Search by meaning uses an embedding model that runs on your machine. When a memory is wrong, it is a line in a file you can find and fix. Markdown memory for coding agents covers the approach.

    \n

    Limits

    \n

    Wordcell itself has not been measured against Supermemory; the pilot above tested Oh’s API. Wordcell is a command-line tool over a folder, not a hosted memory service for your product’s users. Choose Supermemory when you want extraction and connectors run for you: its documentation points to the hosted platform for “connectors, MCP, and the best-tuned extraction pipeline” (self-hosting overview, checked September 26, 2026). When Supermemory fits better lists more cases. The payload measurement used a seven-note vault, and the SciFact study searched 5,183 public abstracts with exact search, without the graph or Git history. Nothing here measures a large vault of notes your agents wrote and linked, or the graph, search by meaning, or the MCP server at that size.

    \n

    Free, open source, and yours

    \n

    Memory for your agents should be free, open source, and superb. That is a position, not a measurement. Free and open source are facts about Wordcell: it is MIT licensed, and the commands above run on your machine without an account. Superb is the goal, and the figures above do not establish it. The same self-hosting overview says Supermemory’s self-hosted edition is free and open source, so price does not separate the two.

    \n

    To move an existing Supermemory account, start with Migrate from Supermemory.

    \n", }; export const blogContents: Readonly> = { diff --git a/site/app/compare/supermemory/page.tsx b/site/app/compare/supermemory/page.tsx index 9d60edc..4ffc284 100644 --- a/site/app/compare/supermemory/page.tsx +++ b/site/app/compare/supermemory/page.tsx @@ -2,8 +2,8 @@ import type { Metadata } from "next"; import type { ReactNode } from "react"; import { MarketingSection, ProductHero } from "@hraness/design-kit/react/server"; -import { longDate, prose } from "../../../wordcell/format"; -import { pilotInterval } from "../../../wordcell/oh-evidence"; +import { grouped, longDate, prose } from "../../../wordcell/format"; +import { longMemEvalFacts, pilotInterval } from "../../../wordcell/oh-evidence"; import { WordcellPageChrome } from "../../../wordcell/page-chrome"; import { formatPlanCredits, @@ -205,7 +205,7 @@ export default function CompareSupermemory() { id="evidence" summary="Wordcell has published no head-to-head comparison with Supermemory of retrieval quality or speed." > -

    The Oh memory kernel that Wordcell embeds published a {prose(pilotInterval.pairedQuestions)}-question pilot that includes Supermemory. It is Oh’s result, not Wordcell’s, and {pilotInterval.crossesZero ? `it does not separate ${pilotInterval.left} from ${pilotInterval.right}` : "it is a small pilot"}. The benchmarks page shows it with its limits.

    +

    The Oh memory kernel that Wordcell embeds published a {prose(pilotInterval.pairedQuestions)}-question pilot that includes Supermemory. It is Oh’s result, not Wordcell’s, and {pilotInterval.crossesZero ? `it does not separate ${pilotInterval.left} from ${pilotInterval.right}` : "it is a small pilot"}.{longMemEvalFacts.matchedSupermemoryRun ? "" : ` Oh’s later ${grouped(longMemEvalFacts.questions)}-question LongMemEval-S study compares Oh with BM25 only, so this pilot remains the only matched comparison with Supermemory.`} The benchmarks page shows both with their limits.

    • Benchmarks: matched and published comparisons
    • Move from Supermemory to Wordcell
    • diff --git a/site/app/docs/docs.generated.ts b/site/app/docs/docs.generated.ts index 5a738a1..84c08c0 100644 --- a/site/app/docs/docs.generated.ts +++ b/site/app/docs/docs.generated.ts @@ -16,7 +16,7 @@ export const docHtml: Readonly> = { "design": "

      Design

      \n

      hraness/wordcell treats a knowledge base as durable Markdown plus replaceable views.\nA vault must remain useful when the CLI is absent, and a capture must remain\ninspectable when the original page changes or disappears. Exact graph and\nmetadata views are deterministic; semantic search is optional derived state\nthat can be deleted and rebuilt.

      \n

      Storage is the interface

      \n

      The vault is an ordinary directory of Obsidian-compatible Markdown, suitable for a text editor, Git, and standard filesystem tools. Frontmatter, headings, prose, and wikilinks are owned content. Refresh, check, graph navigation, metadata queries, and capture require no hosted account or model. A local QMD index is an optional cache for semantic recall, never the authoritative copy of a note.

      \n

      wordcell init creates a small set of authority boundaries:

      \n
        \n
      • articles/ contains captured sources and their local artifacts.
      • \n
      • notes/ contains maintained concepts, entities, comparisons, and syntheses.
      • \n
      • plans/ contains proposals, decisions, execution state, and verification.
      • \n
      • riffs/ contains cleaned first-person thought from dictated or stream-of-consciousness material.
      • \n
      • scopes/ contains optional pull-based context for selected repository directories.
      • \n
      • index.md is the front door. It can be a short authored page or contain one marked, tool-managed catalog block.
      • \n
      \n

      The boundaries separate what a source said from what the vault currently concludes. They are conventions expressed in Markdown and agent guides, not proprietary file formats.

      \n

      Setup is an approved instruction workflow

      \n

      The packaged Agent Skill routes setup and evolution requests before it prepares\na runtime. It can inspect an explicitly proposed location, interview the user,\nand present exact read and write surfaces without installing Wordcell, creating a\nvault, building a QMD index, or accessing an ambient account. Only the approved\nproposal may scaffold files. A changed path, skill, repository, account, or\nintegration requires renewed approval.

      \n

      A vault may add zero to three companion skills for recurring rituals whose\ninputs, authority, durable output, and failure behavior need a distinct\ncontract. These are inert instruction files. They do not form a runtime plugin\nregistry, execute vault metadata, inherit account authority, or couple the\nvault to application code. An exact repeated scaffold is a no-op; divergent\ncontent, path escape, symbolic links, partial failure, and unapproved external\nsurfaces stop the workflow.

      \n

      The repository models these transitions with fake capabilities and verifies\npreapproval zero mutation, exact approved writes, no-op repeats, renewed\napproval, divergence, confinement, symbolic links, partial failure, and\nexternal-surface rejection. This is a tested contract example, not proof that\nevery agent or host integration complies.

      \n

      This interview-first setup and bounded extension model builds on Frank Chen's\npublic notes about designing a personal knowledge base with an\nagent\nand extending it with skills.

      \n

      Wordcell ships no kb_role field, lifecycle resolver or API, lifecycle CLI,\ncompatibility diagnostic, or metadata migration. A frozen value gate must show\nthat those surfaces improve deterministic agent decisions before they are\nintroduced. Current and historical plan routing remains derived from existing\nnote type, path, and status.

      \n

      Repository instructions and context have different authority

      \n

      An AGENTS.md file is normative, path-scoped, and always loaded before an\nagent edits within its directory. It owns the information that must be present\nat edit time: directory ownership, required commands, prohibitions, invariants,\nand release or verification gates.

      \n

      A scope hub is optional and pull-based. It explains why a rule exists and\ncarries the history, examples, evidence, rejected alternatives, and links that\nwould make an always-loaded guide too large. A hub cannot override a guide, and\nit cannot be the only home of a rule whose omission could make an edit unsafe\nor invalid.

      \n

      Each hub maps to one exact repository-relative directory:

      \n
      ---\ntitle: Source context\nsummary: Design history and evidence for work under src.\ntype: agent-context\nscope: src\n---\n\n# Source context\n
      \n

      The corresponding file is scopes/src--25a6634263c1.md. The identity consists\nof a lowercase ASCII slug made from the full scope, bounded to 48 characters,\nfollowed by the first 12 lowercase hexadecimal characters of the SHA-256 digest\nof the full NFC-normalized scope. The root scope is . and has the reserved\nidentity scopes/repository--cdb4ee2aea69. The full exact scope remains in\nfrontmatter; the readable filename is not a substitute for it. Moving a\ndirectory changes its scope and therefore its hub identity.

      \n

      Derive the exact tuple without writing files:

      \n
      wordcell agents identity src --json\n
      \n

      The command returns the normalized scope, extensionless note ID, Markdown path,\nowning guide path, and reciprocal marker. Use its output rather than\nreimplementing slug or hash logic.

      \n

      The guide points back to the extensionless note ID with one exact marker before\nits headings:

      \n
      <!-- kb:context scopes/src--25a6634263c1 -->\n# Contents\n\n- ...\n\n# Guidelines\n\n- ...\n
      \n

      Mappings are reciprocal: a hub requires the marker in the AGENTS.md at its\nexact scope, and a marker requires that canonical hub. A guide without a marker\nis valid and remains fully normative.

      \n

      wordcell context <repository-path> --root <vault> --repo <repository> returns the\napplicable guides from root to nearest, verified hubs from nearest to root, and\nthe authored memory records whose repository_scopes contain the target. The\ntext view includes summaries, not bodies. It keeps maintained knowledge, active\nplans, dated research, reports, and terminal plans in separate bounded groups,\nreports the exact declaration that matched, and prefers the deepest matching\nscope. Open only the useful record, then use wordcell links, wordcell backlinks, kb list, or wordcell search for a bounded expansion. --kind auto uses filesystem\nstate and a conservative path hint; --kind file or --kind directory makes\nthe target interpretation explicit.

      \n

      repository_scopes is an optional array of exact, canonical,\nrepository-relative paths. Matching is case-sensitive and lexical. A directory\nscope matches itself and descendants; a file scope matches only that file.\nScopes can name future or retired paths, so existence is reported separately\nfrom validity. Active plans and maintained notes with missing scopes produce an\nadvisory; terminal plans may retain a retired path as historical evidence. The\ntool never follows Git renames or writes inferred scopes back into Markdown.

      \n

      The dated-research group is deliberately narrower than an arbitrary note with\na date. A record must live under projects/<domain>/market/, declare type: market-research, status: snapshot, a valid as_of date, and at least one\nrepository scope. Reports analogously declare type: report, a valid\ngenerated date, and a repository scope. Records outside those contracts stay\navailable to ordinary metadata, text, and graph queries without being labeled\ncurrent path memory.

      \n

      wordcell agents check verifies canonical IDs, type and scope metadata,\nduplicate, case-fold, and Unicode-normalization collisions, repository\nconfinement, real scope directories and regular guide files, exact reciprocal\nmarkers, and the required guide shape. wordcell agents audit runs the same gate\nand adds deterministic measurements for every guide and section, inherited\nchains, long guideline bullets, and exact duplicate rules. Those measurements\nidentify review candidates. Length is not a correctness test, and moving a\nload-bearing rule out of an AGENTS.md file to satisfy a budget makes the\nsystem worse.

      \n

      Guide discovery skips common version-control, dependency, cache, coverage, and\nbuild-output directories. It never follows symbolic-link directories, and a\nmapped or discovered AGENTS.md symbolic link is invalid. These constraints\nkeep the inherited chain reproducible and confined to the selected repository.

      \n

      The graph is explicit

      \n

      The graph is built from wikilinks and typed relationships in authored Markdown.\nA scan parses note identity, title, aliases, tags, typed metadata, readable\ntext, outgoing links, and outbound assertions. It resolves each target and\nreports broken or ambiguous references rather than choosing a convenient\nmatch.

      \n

      Reusable concepts are ordinary notes:

      \n
      ---\ntype: concept\ntitle: Durable agent memory\naliases:\n  - persistent agent memory\n---\n\n# Durable agent memory\n
      \n

      A note owns each typed assertion it makes:

      \n
      relations:\n  supports:\n    - notes/durable-agent-memory\n  contrasts-with:\n    - notes/conversation-history\n
      \n

      Predicates use lower-kebab-case and targets use exact vault-root note IDs\nwithout .md. The source note is the implicit subject. Different agents can\ntherefore edit relationships on different notes without contending on a central\nontology or edge file.

      \n

      The recommended vocabulary covers common Wordcell claims: synthesizes,\nevidenced-by, informed-by, supersedes, and contradicts. It is advisory,\nnot a closed ontology. A vault may author another canonical predicate when its\nprose and evidence define the claim. Note type, directory, chronology, shared\ntags, and semantic similarity do not choose a predicate.

      \n

      Four rules keep the result honest:

      \n
        \n
      1. Backlinks and inverse relationships are derived, never written into source\nnotes.
      2. \n
      3. The catalog or authored front door is navigation, so links to or from its\nnote (index.md by default) do not count as contextual edges.
      4. \n
      5. A title, alias, recurring tag, shared neighborhood, or semantic match is a\ncandidate. It becomes an edge only after an agent or person reviews the\nevidence and authors the assertion.
      6. \n
      7. Reciprocal, inverse, transitive, and similarity-derived relationships\nremain query results. They never silently become Markdown facts. External or\nunclassified material remains unresolved until evidence supports an authored\nassertion.
      8. \n
      \n

      This makes inbound and outbound counts, backlinks, relationships, and orphans\nreproducible. It also prevents reciprocal sections and generated catalogs from\nmaking a disconnected vault appear healthy.

      \n

      Fenced code, inline code, frontmatter, and HTML comments are excluded from\nmention analysis. Line breaks are preserved during masking so diagnostics\ncontinue to point at the authored source.

      \n

      Focused graph views are rebuilt from Markdown

      \n

      Every structural command scans the current notes and resolves canonical note\nidentities, contextual wikilinks, and source-owned typed relationships. The\npackage does not maintain a second graph database or generated fact file.\nwordcell graph returns the whole resolved graph and its diagnostics;\nwordcell backlinks and wordcell relation list answer focused inbound or typed-edge\nquestions; and wordcell links performs cycle-safe traversal with explicit depth and\nresult limits.

      \n

      wordcell percolate runs named, read-only analyses that surface repeated tags\nwithout concept notes, unconnected shared-concept neighborhoods, exact\nunlinked mentions, and relationship-hygiene findings. The output cites the\nauthored evidence that caused each candidate. A person or agent decides whether\nto run wordcell note create or wordcell relation add.

      \n

      Percolation Result V2 emits missing relationships as unordered endpoint pairs\nwith a required predicate. It does not present either endpoint as the source,\ndraw a directional edge, or suggest related-to. The reviewer reads both\nnotes, then authors a directed assertion only when the evidence determines its\nowner, target, and predicate. Explicit V1 parsers preserve historical\nunversioned results for one deprecation cycle; the default parser accepts V2\nonly, and no compatibility path guesses a semantic upgrade. V1 remains\navailable throughout 0.19.x and is not removed before 0.20.0.

      \n

      This named-command surface is deliberate. Common graph questions receive a\nsmall typed contract, deterministic ordering, and an operation-specific bound\ninstead of requiring every agent to construct an ad hoc query program. A\none-off whole-vault question can inspect wordcell graph --json; a recurring question\nearns a focused command and regression tests when real use demonstrates the\nneed.

      \n

      Commands that only need links and typed relationships skip quadratic\nprose-mention discovery. A scoped wordcell percolate <note> considers only mention\npairs touching the resolved note; vault-wide percolation and ordinary graph\nmaintenance use explicit pair and result budgets. Scans reject more than 10,000\nnotes before parsing, then bound each note at 16 MiB of valid UTF-8 and the\nvault at 256 MiB. These are package ceilings; callers may select lower\noperation-specific limits.

      \n

      Rebuilding views from Markdown keeps Git history on assertions people can read\nand avoids a repository-wide merge hotspot. A future cache may live outside the\nvault only if measurements justify it; it must be content-addressed by source\nand analysis version and rebuild on any mismatch.

      \n

      Oh adoption stops at a review candidate

      \n

      createOhAdoptionPreparerV1 captures, in trusted host code, one exact Oh\nworking-authority binding and head plus the destination purpose and proposed\nnotes/ path, a purpose-matched rights decision, the required review route,\nand the conflict assessment. Its returned facade accepts only an Oh dependency\nclosure and explicit transformation or redaction disclosures. A model cannot\nreplace the source authority, destination, rights, review, or conflict policy\ninside a preparation call.

      \n

      Wordcell delegates contract, binding, head, record, and exact dependency-closure\nverification to its pinned immutable @hraness/oh store API. Wordcell keeps lower\nlocal byte, record, root, depth, and node ceilings and rejects accessors,\nsymbols, cycles, canonical-authority bindings, tampered or incomplete records,\nover-complete closures, wrong bindings or heads, and derived-only roots. The\nreview artifact records the source authority and binding digest, full head,\nclosure roots, and record digests without copying source realm or space IDs.

      \n

      The returned status is always prepared. The function does not open a vault,\nwrite a note, invoke Git, import an operation chain or database, retain a\nprojection, or write to canonical Oh. A reviewer must inspect the candidate and\nauthor destination Markdown through Wordcell's existing revision-checked write path;\nthe source's proposed assertion is never relabeled as reviewed knowledge.

      \n

      Catalog ownership is explicit

      \n

      A managed vault gives one marked region in index.md to the tool. wordcell refresh\nrenders a sorted catalog and atomically replaces only that region. Text outside\nthe markers belongs to the author. Malformed or duplicate markers fail closed.

      \n

      An authored vault declares kb_catalog: authored in index.md. Refresh and\ncheck leave the complete file untouched, while wordcell catalog renders the same\nexhaustive inventory on demand. This removes a repository-wide generated-file\nhotspot without weakening the scan, graph, metadata, attachment, plan, research,\nor context checks.

      \n

      wordcell check computes the expected managed catalog when one exists and applies\nthe remaining vault policy in either mode. wordcell check --no-catalog skips only\ncatalog freshness, which lets independent lanes validate their notes before\nintegration. A managed vault still performs one final refresh after lanes join;\nan authored vault has no shared generated Markdown write.

      \n

      wordcell graph exposes the scan as a human-readable or structured report.\nwordcell backlinks and wordcell relation list use the same identities to retrieve\nincoming links and typed assertions. wordcell links traverses both kinds of authored\nedge to a bounded depth and node count, reporting when a high-degree\nneighborhood reaches the cap. There is no second graph state to synchronize.

      \n

      Single-note authoring commands confine paths to the vault, reject symbolic and\nhard-linked targets, serialize local same-note writers, compare an optimistic\nsource revision, and atomically replace the source file. Different-note writes\ndo not share a lock or graph file. Git remains the cross-worktree review and\nmerge mechanism.

      \n

      The single-note transaction has one internal Effect owner behind the Promise\nAPI. It retains the lock through every admitted native write and directory\nsync, including a sibling sync still pending after another fails. A visible\nreplacement and its directory durability are separate milestones: failure\nafter installation cannot authorize overwriting that replacement with older\nbytes. Existing revision, inode and no-clobber recovery checks decide which\nrecovery operation is safe. Temporary cleanup and outer lock release keep\ntheir established error precedence.

      \n

      Exact metadata is authored

      \n

      Frontmatter is parsed as typed, nested data rather than flattened strings. Scalars retain their string, number, boolean, or null type; arrays and objects retain their structure. Tags from frontmatter are normalized for matching while the original metadata remains available in structured output.

      \n

      wordcell list filters that authored state by nested dotted paths, field existence,\ntags, or repeated exact repository scopes, then sorts by title, path, graph\ncounts, or nested metadata. wordcell search and the SDK expose the same\ncase-sensitive scope constraint. Repeated filters are conjunctive; repeated\nscope values form one exact allowlist. Missing sort values are placed last and\nties are stable, so the same vault and query produce the same order.

      \n

      Metadata is useful for exact questions such as “which implementation plans are in progress?” It is not inferred from prose and the tool does not invent tags to improve retrieval. Authors and agents can evolve conventions in the vault's scoped AGENTS.md files without migrating to a package-owned schema.

      \n

      Hybrid retrieval keeps its evidence visible

      \n

      wordcell search starts with the current Markdown. Its exact lane scans note identity,\ntitle, aliases, path, tags, typed metadata, and prose. Exact title and alias\nidentities remain visible in the result evidence and stay ahead of broader\nmatches.

      \n

      Hybrid mode is the default. It runs the exact lane alongside QMD.\nWordcell requests QMD's direct local full-text and vector rankings at the declared\ncandidate bound, then fuses them without query expansion or reranking models.\nThis avoids QMD 2.5.3's smaller fixed structured-hybrid pool. Both the inner\nQMD lists and the outer exact/QMD lists receive neutral equal weights in\nreciprocal-rank fusion. Exact title, alias, and path identities are\npinned separately, while agreement between lanes outranks single-lane evidence.\nEach result reports the lane ranks and contributions that produced its final\nposition. --mode exact stays model-free,\n--mode keyword uses QMD's full-text index, and --mode semantic selects its\nvector lane.

      \n

      Reranking is opt-in through --rerank typesafe. It uses jev-1.13.0 and sends\nthe query plus each candidate's identifier, title, vault-relative path, and at\nmost 512 UTF-8 bytes of snippet text to TypeSafe. Enable it only for vaults\napproved for external processing and provider input-token charges. Use\n--rerank-limit 25 to bound the window (2–25 candidates); four requests run\nconcurrently under one eight-second deadline with no retries. Exact identities\nremain first. Results retain their original scores and retrieval evidence.

      \n

      The CLI reads TYPESAFE_API_KEY, an explicit TYPESAFE_API_KEY_FILE, or the\nowner-only file ~/.config/wordcell/typesafe-api-key (honoring\nXDG_CONFIG_HOME). Keep credentials outside the repository. A missing key or\nprovider failure retains baseline ordering and marks the rerank lane\nunavailable or degraded. Inspect its structured rerank receipt for attempted\nrequests, known usage, elapsed time, and whether usage is complete; an unknown\ncharge is never reported as zero. A successful exit alone does not establish\nthat reranking occurred. Omit --rerank for local-only retrieval.

      \n

      For a repository's ordinary KB searches, prefer its approved, pinned\nkb:search script when present. That script declares the vault and processing\nchoice. Read returned notes before acting; model probabilities are ranking\nsignals, not proof of truth. Explicit priority rules still run last.

      \n

      Wordcell pins QMD 2.5.3 and one full upstream revision of its compact\nEmbeddingGemma model for local vector retrieval. The revision prevents branch\ndrift and gives the model a revision-specific cache identity. Without an\nexplicit local source, the first hybrid or semantic query downloads that\nrevision; later runs reuse the local cache and incrementally update changed\nMarkdown.

      \n

      An explicit model file is accepted only when its SHA-256 matches the pinned\nartifact. Wordcell gives QMD that file as the per-store load source while retaining\nthe stable model URI and digest as derived-index identity. QMD 2.5.3's public\nvector method falls back to a process-global model for query embeddings, so Wordcell\nuses QMD's exposed per-store vector boundary for both query and document\ninference. That QMD release also asks its process-global model to tokenize fresh\ndocument chunks and legacy fingerprint samples. Wordcell pins an immutable public\nHraness QMD compatibility commit\nthat routes those two internal calls through QMD's existing store-local model\nwithout changing its public chunking API. The\nfork includes compiled distribution files so standalone Git installs do not\ndepend on consumer-relative patches or installation-time compilation. The local model-file path does not\nenter reports, SDK results, or generation identity, and moving identical model\nbytes does not require a new logical index.

      \n

      Each vault gets a path-derived SQLite cache under the user's cache directory unless --database selects another file outside the vault. Wordcell refuses a database symlink or multiply linked database file and claims its adjacent snapshot directory with a versioned ownership record before cleanup. It scans and bounds the live Markdown first, then atomically refreshes a disposable validated source projection beside the database. QMD indexes that projection, so it cannot read a note that bypassed Wordcell's per-note or aggregate vault limits or recursively ingest its own cache. Cached files are checked against the manifest before reuse. An older snapshot directory without the ownership record is never removed automatically; delete the explicitly named disposable .snapshot directory and retry.

      \n

      A database-scoped process lease serializes projection installation, store updates, and embedding writes across agents. The generation identity includes the immutable note bytes and the QMD version, embedding model, collection configuration, and projection contract that interpret the shared SQLite state. Sessions with the same identity may read concurrently; an identity change waits for older readers to close before mutating the database. QMD operations within one open session remain serialized. index.md and every AGENTS.md are excluded because they are navigation and always-loaded instructions rather than knowledge records. Scope hubs remain ordinary Markdown, so QMD indexes their rationale and evidence like any other note. The database and source projection may be removed at any time and recreated with wordcell index.

      \n

      Search results are joined back to the live session snapshot, so each hit carries\ncurrent typed metadata and tags. Files outside the requested vault and stale\nindexed identities are discarded. Metadata and tag constraints are authoritative\nat that join boundary. QMD 2.5.3 cannot rank against a path allowlist, so a\nfiltered search uses a bounded global candidate window. Selective searches use\nthe largest supported window by default. Any observed QMD rows discarded by\nlive reconciliation or filters leave an underfilled request explicitly degraded,\neven when QMD's chunk-level retrieval returned fewer rows than requested. QMD failure\ndoes not erase exact results; the response marks a failed lane unavailable or\nan incomplete embedding pass degraded, and reports that the result is partial. A\nretrieval score is a discovery aid, not a graph edge, a citation, or evidence\nthat the result is true.

      \n

      Immediate explicit links and typed relationships can be returned with search,\nalong with a bounded neighborhood around the strongest results. These graph\nneighbors remain a separate context collection. They do not enter primary text\nrank or become authored edges. When explicitly requested, bounded Git history\ncan likewise explain when a note changed and which paths changed with it. kb history <note> retrieves one note's provenance directly, and wordcell history search <query-or-path> searches commit subjects, note paths, and co-change\npaths without running text retrieval. --history, --require-history, or SDK\nhistory options enable that separate lane on search. Omitted history performs\nno Git indexing, and an explicit request with no\nprimary results has nothing to enrich. Query, note, and detail bounds are\nvalidated before the Git index opens. A commit that exceeds the per-commit\nchanged-path detail limit retains its\nhash, subject, time, and vault-local note associations while its co-change set\nis marked incomplete. Later commits continue indexing. Automatic history\nreturns the usable provenance with a degraded diagnostic only when the selected\nnotes are affected. Optional Git failure returns an explicit unavailable\ndiagnostic. Both cases mark the search partial. history: "required" or an\noptions object with policy: "required" rejects unavailable or incomplete\nselected-note provenance instead. Aggregate process time, output, commit, and\npath-observation limits remain hard failures. Git evidence is provenance and\nhistorical recall, not a recency boost.

      \n

      Code mode shares one bounded snapshot

      \n

      Agents that need several retrieval operations can use the SDK without spawning\none CLI process per question:

      \n
      import { openKnowledgeBase, packUntrustedSearchContext } from "@hraness/wordcell/sdk";\n\nconst kb = await openKnowledgeBase({ root: "kb", repository: "." });\ntry {\n  const result = await kb.search({\n    query: "why captures preserve incomplete threads",\n    tags: ["capture"],\n    graph: { depth: 1 },\n    history: "auto",\n  });\n  console.log(packUntrustedSearchContext(result).content);\n} finally {\n  await kb.close();\n}\n
      \n

      Opening a session performs one confined vault scan. grep, list, read,\nlinks, backlinks, search, history, and searchHistory reuse that\nsnapshot. QMD and Git are opened lazily. The session is intentionally read-only\nand does not watch the filesystem. Close it and open a new session after any\nMarkdown write so later work cannot mistake an old snapshot for current state.

      \n

      packUntrustedSearchContext accepts ordinary plain objects and arrays, such as\nvalues produced by JSON parsing or Wordcell itself. Do not pass same-realm Proxy\nobjects: proxy inspection can execute user code and is outside a data-only\nprojection boundary. Isolate or serialize foreign executable objects before\npacking them.

      \n

      Code-mode DAGs use defineWorkflow and runWorkflow. The staged\ndefineWorkflow<Input>("id").node(...).output(...) builder infers each node's\nresult and the final output while exposing only declared dependencies. A\ndefinition has at most 64 nodes, must be acyclic, and names one output node.\nReady nodes run in declaration order with a default global concurrency of four\nand a maximum of eight. QMD work is always serialized; Git permits at most four\nnodes, bounded again by the global limit. The runner applies an aggregate\nstructured-output byte limit. Failure or abort stops dependent nodes from\nstarting and waits for already-running siblings to settle. The packaged\nworkflows are ordinary imports, accept explicit inputs, and return structured\nresults without writing the vault.

      \n

      One internal Effect program owns that scheduling lifetime. Native callbacks\nstill enter through the original Promise microtask and settle through the\nnative Promise race, preserving observable ordering and raw failure reasons.\nAn interrupted fiber does not stand in for a settled callback. The public\nrunner continues to return a Promise, so callers need no Effect runtime or\nservice configuration.

      \n
      import { openKnowledgeBase } from "@hraness/wordcell/sdk";\nimport { runWorkflow } from "@hraness/wordcell/workflow";\nimport { explainChangeWorkflow } from "@hraness/wordcell/workflows";\n\nconst kb = await openKnowledgeBase({ root: "kb", repository: "." });\ntry {\n  const explanation = await runWorkflow(explainChangeWorkflow, {\n    kb,\n    input: { query: "why the capture path changed" },\n  });\n  console.log(explanation.output);\n} finally {\n  await kb.close();\n}\n
      \n

      decisionContextWorkflow assembles ranked rationale and note provenance,\nexplainChangeWorkflow searches authored rationale and Git evolution in\nparallel, and planRadarWorkflow joins exact plan state with retrieval and\nhistory.

      \n

      Only decisionContextWorkflow returns a bounded untrusted context envelope.\nexplainChangeWorkflow and planRadarWorkflow intentionally return raw\nsource-derived Wordcell and Git structures for trusted application code to inspect.\nTreat every string field in those results as untrusted data: do not execute it\nor place it in a model instruction channel, and project or pack the selected\nfields through the untrusted-content boundary before an agent handoff.

      \n

      Changes to retrieval ranking use the exported deterministic metric helpers for\nrecall at k, reciprocal rank, and nDCG. The six-case synthetic rank-fusion\nfixture supplies already-ranked IDs for identity, conceptual, and mixed\nexamples. It checks metric and fusion arithmetic only.

      \n

      The real-corpus evaluator accepts a versioned manifest with query text,\nindependently authored relevance judgments, query classes, and structured lane\ninputs. wordcell evaluate fails before retrieval unless the checkout's exact HEAD,\nthe HEAD:<vault-root> tree, and the clean vault match the frozen manifest. It\nthen runs built-in exact, keyword, semantic, hybrid, graph, metadata,\npath-context, and Git adapters through one immutable session. Human query prose\nis never parsed into tool arguments; each adapter receives only its explicit\ninput object.

      \n

      The report retains raw rankings and evidence, unavailable and failed lanes,\nbackend and wall timings, bounded resource counters, aggregate and per-class\nquality, no-answer accuracy, and deterministic paired bootstrap intervals.\nMachine-local home and temporary roots are redacted from persisted hit evidence,\ndiagnostics, and failures while relative document identities and the surrounding\nevidence remain intact.\nSemantic or hybrid runs require --model-file. The evaluator verifies those\nbytes against the pinned model digest before retrieval, gives that file to QMD,\nand verifies it again before reporting. Reports retain the stable model URI,\nrevision, and digest without persisting the machine path. Cache state and\nhardware remain explicit environment evidence. The evaluator does not turn a\nlocal fixture into an industry claim: model download, cold and warm runs,\nscale, concurrency, and agent-task outcomes still need measured protocols of\ntheir own.

      \n

      Local artifacts remain inspectable

      \n

      Graph validation also checks local Markdown and Obsidian attachments. Relative\nimage, PDF, and tldraw targets must resolve to one regular confined file with\nmatching case. Symlinks, hard links, ambiguous case-fold matches, missing files,\nand paths outside the vault fail. External URLs and fragment-only links remain\noutside this local integrity lane.

      \n

      wordcell inbox is a bounded advisory view over recent captured sources that have no\nmaintained-note disposition. Source-to-source and catalog links do not count as\nsynthesis. A capture may intentionally remain a leaf, so the inbox never writes\nlinks, creates notes, or fails the vault merely because an item is present.

      \n

      Capture preserves an audit trail

      \n

      Web capture is a bounded selection process rather than a promise to reproduce an unlimited website. Given a URL and requested scope, the capture pipeline can try:

      \n
        \n
      1. A platform-specific public structured adapter when one can make a stronger completeness claim.
      2. \n
      3. Bounded HTTP acquisition and article extraction.
      4. \n
      5. Optional browser rendering for client-side or authenticated pages.
      6. \n
      7. Explicit saved-HTML input when the user already has a saved representation.
      8. \n
      \n

      Candidates retain their attempt results. The selected representation becomes readable Markdown, while capture.json records the routes attempted, extractor, scope, status, counts, warnings, limits reached, asset hashes, and requested artifact outcomes. A failed lane does not erase useful output from another lane, and an uncertain fallback does not promote a conversation to complete.

      \n

      A bundle is installed atomically:

      \n
      <slug>/\n  <slug>.md\n  capture.json\n  assets/\n  evidence/\n
      \n

      The capture body is source material. Later synthesis belongs in a maintained note so recapture and interpretation do not silently overwrite each other.

      \n

      Completeness is a data property

      \n

      Capture status distinguishes complete, partial, auth-required, blocked, and unsupported. The status describes the selected bounded representation, not the importance or quality of its prose.

      \n

      Counts use scope-specific semantics. Page counts cover primary entries; thread and comment counts cover replies or comments rather than roots, quotes, or pagination markers. Generic rendered prose does not prove a trustworthy item tree, so it may remain partial with a zero structured-item count even when the Markdown is useful.

      \n

      Safety is part of acquisition

      \n

      URLs, redirects, DNS answers, response bodies, browser pages, cookies, subprocess output, and filesystem paths are foreign input. The controlled acquisition lanes therefore share several invariants:

      \n
        \n
      • Only HTTP and HTTPS source URLs are accepted, with embedded credentials rejected.
      • \n
      • Private, reserved, and locally assigned network targets are denied by default.
      • \n
      • DNS answers are validated and accepted addresses are pinned across requests and redirects.
      • \n
      • Time, HTML bytes, asset bytes, total bytes, item counts, depth, browser actions, and process output are bounded.
      • \n
      • Cookies are read only from an explicitly selected source, filtered to matching targets, and kept out of persisted artifacts.
      • \n
      • Active source evidence is converted to inert HTML with credential-shaped values redacted.
      • \n
      • Bundle paths are owned, staged beside the target, and installed by atomic rename; forced replacement requires a compatible manifest and rollback.
      • \n
      \n

      Live or CDP browser attachment keeps the browser's existing network stack and signed-in state. wordcell clip current reads the active tab without navigating or interacting with it and leaves the browser open. URL-based attached capture may navigate that tab and scroll within the configured bounds, taking bounded observations as content is rendered. Screenshots are also different from sanitized source evidence because private content can remain visible in pixels.

      \n

      These boundaries are not entitlement mechanisms. Capture does not bypass authentication, access controls, paywalls, CAPTCHAs, rate limits, DRM, or platform policy.

      \n

      Dependencies follow capabilities

      \n

      Bun is the required runtime.\nYAML parses typed frontmatter,\nand QMD supplies the optional local keyword and\nembedding index. QMD is loaded only by index and search commands, so\ndeterministic graph and metadata commands do not initialize its native runtime\nor model.

      \n

      Defuddle performs article extraction. agent-browser provides optional rendered acquisition. The pinned Sweet Cookie 0.4.3 supports explicit browser-cookie import while retaining host-only scope and rejecting partitioned or container-scoped state that the capture lanes cannot replay faithfully.

      \n

      yt-dlp and FFmpeg remain optional because only full audio or video localization needs them. wordcell doctor reports what is installed without probing cookie stores, and wordcell adapters reports the installed platform claims. A missing optional capability narrows the available route; it does not change the storage or graph model.

      \n

      Extension boundaries

      \n

      New platform adapters should improve the strength of a capture claim, not merely add another scraper. Each adapter declares the scopes, acquisition modes, authentication requirements, item semantics, and media behavior it can support. It must remain bounded and must downgrade honestly when pagination, hidden branches, virtualized content, or access controls prevent completeness.

      \n

      New graph policy should remain a pure function of vault content and explicit configuration. Derived reports may guide an agent or person, but the tool should not silently mutate authored prose. This keeps automation reviewable and lets users replace any analysis layer without migrating their notes.

      \n

      Repository context follows the same separation. The CLI reads the repository\nand vault as development inputs, but no application needs to import Wordcell or\nread a scope hub at runtime.

      \n", "agent-memory": "

      Markdown memory for coding agents

      \n

      Back to Wordcell · Practical agent workflow

      \n

      A knowledge base for your coding agents

      \n
      \n

      Give coding agents durable, searchable memory beside the repository with plain Markdown, Git history, and replaceable local search.

      \n
      \n

      Coding agents lose useful context when a session ends. The next agent can search the code again, but it cannot recover a source that was never saved, a decision that stayed in chat, or the relationship between two notes that nobody recorded. Repeating that work costs time and produces inconsistent answers.

      \n

      Search alone cannot preserve agent memory. The system also needs a write path into inspectable files under version control: evidence can be captured, current understanding can be revised, plans can accumulate outcomes, and mandatory edit rules can move onto the instruction path. Search indexes, graph views, and embeddings used for meaning-based similarity should remain derived and replaceable.

      \n

      Wordcell implements that split as repository-adjacent Markdown and Git. Exact lookup, metadata filters, local search, explicit links, and Git provenance help an agent find and inspect the files without making application code depend on the knowledge system.

      \n

      The pattern converged across agent tools

      \n

      Devin's 2024 release history records Knowledge that could be recalled across future sessions and Repo Knowledge produced by scanning repositories. Its 2025 release history records DeepWiki in April, codebase intelligence inside Devin in May, and a DeepWiki Model Context Protocol server later that month.

      \n

      In April 2026, Andrej Karpathy published an LLM Wiki proposal with immutable raw sources, an agent-maintained interlinked Markdown wiki, and an instruction schema. Its operations are ingest, query, and lint, with QMD as an optional search layer when a simple index stops being enough. These systems converged on durable agent-readable knowledge. The sequence does not establish direct lineage between them or Wordcell.

      \n

      Separate rules from explanations

      \n

      A repository needs two kinds of memory. Rules that must govern an edit belong in a scoped AGENTS.md file on the path to the code. Rationale, history, examples, evidence, plans, and neighboring decisions belong in a knowledge base that an agent pulls only when the task needs them. This keeps mandatory instructions short without throwing away the context behind them.

      \n

      A root guide carries repository-wide policy, and nested guides add constraints owned by a package or product. A nearby knowledge note can explain why a parser rejects a tempting shortcut, preserve the source behind the decision, and link the plan that introduced it. If the note and the applicable guide disagree, the guide controls the edit and the note needs repair.

      \n

      The result has two concrete parts: scoped instruction files govern edits, while an ordinary Markdown vault stores supporting context. Application code imports neither the vault nor its search indexes:

      \n

      Repository rules beside durable knowledge

      \n
      repository/\n├── AGENTS.md                         # inherited root rules\n├── packages/parser/\n│   ├── AGENTS.md                     # scoped rules and checks\n│   └── src/\n└── kb/\n    ├── articles/<slug>/              # captured evidence and assets\n    ├── notes/                         # maintained explanations\n    ├── plans/                         # decisions and outcomes\n    └── index.md                       # short authored front door\n
      \n

      Keep the implementation small and the files authoritative

      \n

      Wordcell packages the pattern as a small file contract. A useful vault can begin with Markdown, Git, index.md, and standard file search. Source capture, metadata queries, repository-path context, QMD, typed relationships, graph traversal, and TypeScript sessions are layers to add when the simpler setup stops answering the repository's questions. Application code need not import Wordcell, and no hosted service or graph database owns its records.

      \n

      Captured sources preserve evidence, notes hold current explanations, and plans retain decisions and outcomes. YAML frontmatter adds queryable metadata without requiring one domain schema for every vault. A code-related record may declare a few exact repository-relative repository_scopes so an agent can recover it from the path it is about. The declaration stays in the record instead of a central project database, which lets parallel agents update unrelated memory without sharing a generated file.

      \n

      The Markdown files are authoritative. The catalog, QMD database, backlink view, path-context view, graph traversal, and bounded Git index are derived and replaceable. A vault can keep a managed catalog or an authored front door and render the complete inventory on demand. Deleting one of those views removes a way to retrieve knowledge, not the knowledge itself.

      \n

      Route current memory from the code path

      \n

      A broad semantic search over years of completed plans can rank a detailed historical record above the short explanation that owns the code today. wordcell context packages/parser/src/index.ts --root kb --repo . starts from a stronger signal: the path being changed. It returns the inherited guides that govern the edit, curated scope hubs, and bounded records whose declared scope is that path or one of its ancestors.

      \n

      The records stay grouped by role. Maintained notes, proposed through blocked plans, dated market research, and generated reports form current memory. Completed, superseded, and cancelled plans remain available in a separate historical group. Every result states the declaration that matched and whether the target currently exists. A plan can therefore describe a future path, while a retired path remains honest historical evidence instead of being silently rewritten after a rename.

      \n

      Path context, exact scope filtering, and Git memory

      \n
      wordcell context packages/parser/src/index.ts --root kb --repo .\nwordcell list --root kb --scope packages/parser --where type=plan --json\nwordcell history search packages/parser/src/index.ts --root kb --repo . --json\n
      \n

      Preserve evidence and plans as working records

      \n

      Durable reasoning needs inspectable evidence. wordcell clip can read a public URL,\nsaved HTML, rendered page, a page already open in an authenticated browser, or\nan existing exact Archive.today snapshot after the direct routes fail. Archive\nfallback is read-only and always partial. The\ncapture documentation defines the supported routes. A capture\nwrites readable Markdown beside localized assets and capture.json, whose\nmanifest records where the material came from, how it was extracted, what was\nsaved, and any warnings. “Complete” describes the selected page surface, not\nevery hidden branch or future version of the site.

      \n

      Capture a web source or local PDF

      \n
      wordcell clip "https://example.com/article" --output articles\nwordcell pdf "/absolute/path/to/document.pdf" --output articles\n
      \n

      The resulting bundle is evidence, not final interpretation. A maintained note can cite several captures, record disagreement, and change when later evidence warrants it. The sources stay available for audit. This prevents an agent from silently replacing what a page said with what it now believes the page meant.

      \n

      The wordcell Agent Skill routes vault planning requests to a focused durable-plan workflow. It creates a normal Markdown file under kb/plans/ with an outcome, status, area, repository scopes, assumptions, dependencies, decisions, and verification method. The file grows during execution as agents record deviations, review findings, and reproducible evidence. Closeout adds a compact result and durable-memory disposition: each reusable conclusion links to the maintained note, guide, code contract, or runbook that now owns it, or says that no promotion was needed. Completed plans remain in Git as the history of the work. When a finding becomes a rule whose omission would make a future edit wrong, move that rule into the applicable AGENTS.md and retain the plan as its rationale.

      \n

      Session memory and profiles

      \n

      A conversation can settle a decision that no note records. On request, or at a session close that you or the vault instructions ask for, the session-memory reference of the wordcell Agent Skill has the agent save what the session settled as a dated type: session note whose repository_scopes name the paths it touched. The agent links that note to the notes it changed with authored relations and keeps one type: profile note with Stable and Recent sections. Wordcell extracts no facts and writes no note on its own.

      \n

      On recall and resume requests, the skill reads the profile and the latest session notes first. wordcell context does not list session or profile notes, so list recent sessions directly with wordcell list --root kb --where type=session --sort date --order desc --limit 5.

      \n

      An MCP client connected through wordcell mcp follows the same steps with the search, list_notes, get_note, create_note, update_note_body, and add_relation tools. create_note writes only the title, type, tags, and a generated document_id in the frontmatter and creates no directories, so a session note written only through MCP has no date or repository_scopes until someone edits its frontmatter. The MCP server and the session-memory workflow are available from source until the next release.

      \n

      Search and connect with bounded signals

      \n

      An identifier, title, alias, path, tag, or quoted phrase should not depend on an embedding. Exact mode reads the live Markdown. The default hybrid mode combines those results with keyword and vector result orders from QMD, a local search engine for Markdown, while keeping exact identity matches first. Graph context and Git provenance remain separate evidence, so neither silently changes the primary text rank.

      \n

      Path context, exact and hybrid search, and direct history

      \n
      wordcell context packages/parser/src/index.ts --root kb --repo .\nwordcell search "parser-v2" --root kb --mode exact\nwordcell search "why does the parser reject this input?" --root kb \\\n  --tag architecture --where status=active --json\nwordcell history "notes/parser-design" --root kb --repo . --json\n
      \n

      --mode keyword uses QMD's local full-text index without loading an embedding model. Hybrid and semantic modes use a pinned local embedding model. Wordcell reconciles every QMD hit with current Markdown before returning it, and applies metadata and tag filters to those live notes. Search modes remain explicit through --mode exact, --mode keyword, --mode semantic, and --mode hybrid.

      \n

      Retrieval is bounded. The high-level wordcell search and KnowledgeBaseSession.search surfaces return at most 100 primary results and request at most 500 candidates from each QMD retrieval lane. Selective filters can discard stale or ineligible rows from that window. When those discards prevent Wordcell from filling the requested eligible result set, Wordcell marks the QMD lane degraded and the overall result partial instead of presenting the bounded approximation as complete. Scores are local ranking signals, not probabilities, and cannot be compared across modes.

      \n

      Each note owns its outbound typed relationships in frontmatter. Wordcell derives backlinks, inverse edges, and bounded traversal at read time, so parallel agents do not contend on one generated fact file. wordcell percolate <note> reports recurring concepts and missing-link candidates with inspectable support but writes nothing. An agent reads the cited notes before creating a reusable concept or relationship. Semantic similarity never creates an edge automatically.

      \n

      Percolation Result V2 presents a missing relationship as an unordered pair of\nnotes with a required predicate. It does not choose the source, direction, or a\nrelated-to fallback. Recommended authored predicates include synthesizes,\nevidenced-by, informed-by, supersedes, and contradicts; they are an\nadvisory vocabulary, so a vault can use another canonical predicate when its\nprose and evidence define the claim. Wordcell never infers reciprocal, inverse,\ntransitive, or similarity-derived relationships.

      \n

      Git provenance is opt-in. A search without --history performs no Git indexing. --history requests best-effort provenance, while --require-history rejects unavailable history or incomplete provenance for the selected notes. If one commit exceeds the 2,000-path detail limit, Wordcell retains its identity and vault-local note associations, marks its co-change detail incomplete, and continues through later commits. Best-effort search reports that requested lane as partial.

      \n

      Local attachment checks cover Markdown and Obsidian references to images, PDFs, and editable tldraw sources. They reject missing or escaping files while leaving external URLs alone. A source-inbox view separately lists recent captures that have no inbound disposition from maintained knowledge. It is an advisory, not an automatic backlink requirement: a saved source may intentionally remain a leaf.

      \n

      Measure retrieval on a frozen corpus

      \n

      The August 2, 2026 pilot froze one repository snapshot and 18 questions whose graded relevance judgments were written before the rankings were inspected. The evaluator scanned 156 Markdown records and projected 155 searchable notes into QMD after excluding the authored vault index and agent guides. Nine questions formed the development set, and nine were held out for the test. The test covered exact identity, conceptual recall, active plans, current decisions, code-path context, source evidence, historical rationale, stale-versus-current conflicts, and one no-answer case.

      \n

      At a cutoff of 10 results, exact search recorded Recall@10 of 0.833333, MRR@10 of 0.892857, and nDCG@10 of 0.790377. Hybrid search recorded 0.833333, 0.937500, and 0.833884, respectively. Recall measures how much of the judged relevant set appeared; mean reciprocal rank rewards an earlier first relevant result; normalized discounted cumulative gain also accounts for graded relevance and position.

      \n

      Eight test questions had an answer. A 10,000-resample paired bootstrap, which repeatedly samples those same questions to estimate the stability of the difference, measured hybrid minus exact. The Recall@10 difference was 0 with a 95% confidence interval of [0, 0]; the MRR@10 difference was +0.044643 with [0, 0.133929]; and the nDCG@10 difference was +0.043508 with [-0.012752, 0.111832]. Both retrievers returned a result for the one no-answer question instead of abstaining, so their no-answer accuracy was 0.

      \n

      The same mixed-cache, single-run test recorded p95 latencies of 44.345 milliseconds for exact, 62.834 for hybrid, 821.370 for keyword, and 41,000.524 for semantic retrieval. The semantic figure includes the first in-process model load. The run used QMD 2.5.3 at Hraness compatibility commit aa993dc and a locally verified EmbeddingGemma 300M Q8 model on Bun 1.3.14 and Node 24.3.0 under arm64 Darwin 25.5.0, with an Apple M4 Max, 16 logical CPUs, and 128 GiB of memory. Each p95 summarizes only nine queries with mixed cold and warm state, so these are local diagnostics, not speed claims. The corpus is too small to establish that hybrid is generally superior to exact search or to compare Wordcell with industry retrieval systems.

      \n

      Search finds candidates. Similarity does not establish that a passage is current, correct, or supported by its sources. The Markdown, cited captures, explicit relationships, and requested Git history supply the material a reader must inspect.

      \n

      Customize through an approved proposal

      \n

      The Agent Skill routes setup and evolution requests before it prepares a\nruntime. It inspects the proposed location without mutation, interviews the\nuser about the memory questions the Wordcell should answer, and presents exact read\nand write targets. Only the approved targets may be scaffolded. A changed path,\nrepository, account, integration, or companion skill requires renewed\napproval.

      \n

      The standard router may be enough. A recurring ritual can instead receive a\ncompanion skill with explicit inputs, authority, durable outputs, idempotence,\nfailure behavior, and verification. These skills are inert instructions. They\ndo not create a plugin runtime, execute vault metadata, inherit ambient account\naccess, or couple application code to the Wordcell. An exact repeat is a no-op;\ndivergence, path escape, symbolic links, partial writes, and unapproved\nexternal surfaces stop the workflow.

      \n

      The repository's fake-capability suite exercises those transitions. It is a\ntested contract example, not proof that every agent or host integration\ncomplies.

      \n

      This workflow builds on Frank Chen's public notes about designing a personal\nknowledge base with an\nagent\nand extending it with\nskills.

      \n

      Wordcell ships no kb_role metadata, lifecycle resolver or API, lifecycle CLI,\ncompatibility diagnostic, or metadata migration. A frozen Phase 0 value gate\nmust show that those surfaces improve deterministic agent decisions before they\nare introduced. Current and historical plan routing remains derived from\nexisting type, path, and status conventions.

      \n

      Adopt the smallest useful split

      \n

      Start with a short inherited AGENTS.md path for rules whose omission would make an edit wrong. A small knowledge base may need only Markdown, Git, an index page, and ordinary file search. Add source capture when evidence keeps disappearing. Add repository scopes when agents need to recover current memory from code paths. Add metadata or hybrid search when file search stops answering the repository's questions. Add links and graph views only when the relationships themselves help people make decisions.

      \n

      Treat the knowledge base as repository-adjacent durable memory. Authored Markdown and Git are the record; catalogs, indexes, embeddings, and graph views are replaceable ways to find and inspect it. Checks can validate structure, captures can preserve a selected surface, and similarity can suggest candidates. None of those mechanisms proves that a source is trustworthy or an explanation is still true. People and agents must revise the knowledge as the repository changes.

      \n", "comparisons": "

      Choose a Markdown knowledge or agent memory tool

      \n

      Wordcell is for people who want their coding agent to reuse decisions, follow\nauthored links, recover Git context, and publish selected notes from a local\nMarkdown knowledge base. It combines those operations in a CLI and TypeScript\nSDK. You can keep your text editor, Git workflow, and existing Markdown files.

      \n

      Local storage is a shared strength of this category. QMD, Basic Memory's local\nmode, and Obsidian also work with files on your machine. Choose by the workflow\nyou need, rather than by a claim that only one of these projects is local.

      \n

      Supermemory, Mem0, and Zep are memory services for AI applications that derive\nfacts from the conversations and content you send them. Supermemory and Mem0\nalso publish versions you run yourself\n(Supermemory local and\nMem0 open source).

      \n

      This comparison was checked against the linked primary documentation on\nSeptember 19, 2026. It describes the documented products and standard workflows;\nplugins or custom scripts can add other behavior. The paragraph above and the\nSupermemory, Mem0, and Zep rows and sections were checked against their linked\nprimary documentation on September 26, 2026. The selection advice below is our\nassessment of those documented capabilities.

      \n

      At a glance

      \n\n\n\n\n\n\n\n\n\n\n\n\n\n\n
      ApproachA good fit when you want…What Wordcell adds or changes
      Plain Markdown with your editor, Git, and rgPortable files and a small toolset you already know.Consistent metadata queries, derived backlinks and typed relations, repository-scope context, bounded agent handoffs, and a publication workflow.
      QMDLocal search over document collections, with keyword, vector, and hybrid retrieval.Uses QMD as an optional retrieval layer, then joins results to current authored metadata, explicit graph context, and bounded Git provenance.
      Basic MemoryA Markdown knowledge graph that AI assistants can read and update through MCP.A headless CLI/SDK workflow centered on repository scopes, code-mode composition, explicit Git evidence, and static publication. wordcell mcp serves a vault to local MCP clients and is available from source until the next release.
      ObsidianAn interactive editor for a local vault, with links, graph navigation, and optional publishing.Agent-oriented operations over compatible Markdown; it can accompany your editor. Wordcell does not supply a desktop note editor.
      QuartzA customizable website or digital garden built from Markdown.Publishes selected slices directly from the same agent knowledge workflow, with paths, metadata, tags, or graph neighborhoods as selectors.
      SupermemoryA memory API for your own product, with per-user container tags, managed connectors, and a hosted MCP server.Memory as Markdown notes that you edit and commit, an importer for Supermemory exports, and a local MCP server that needs no account. The importer and the MCP server are available from source until the next release.
      Mem0Memories that a model extracts per user, agent, or run inside your application, on a managed platform or self-hosted.Notes that people and agents write, with exact search, links, and Git history that need no model.
      ZepA temporal context graph built from each user’s chat and business data.Authored relations and Git history in place of facts that Zep derives from your data.
      \n

      Start with Markdown when it already solves the problem

      \n

      Markdown is a plain-text document format. Its readability and ordinary files\nare part of Wordcell's foundation, not a limitation to replace. The format\nitself does not define a search service, repository-context router, or publishing\npolicy. CommonMark's specification\ndescribes the document syntax.

      \n

      A small set of notes and rg may be enough. Wordcell becomes useful when you\nrepeatedly reconstruct backlinks, filter frontmatter, connect a decision to a\nrepository path, or prepare the same context for another agent session. It adds\nthose operations while keeping Markdown authoritative. There is no evidence\nhere that Wordcell beats a carefully scripted Markdown workflow on speed or\nanswer quality.

      \n

      Use QMD for retrieval, or use it inside Wordcell

      \n

      QMD provides on-device BM25 keyword search,\nvector search, query expansion, and reranking. Its CLI, SDK, and MCP interfaces\nreturn structured results; it supports collection context, metadata filtering,\nsnippets, and bounded document retrieval. Those are substantial agent features.

      \n

      Wordcell's optional semantic search uses a pinned QMD implementation. Its\nadditional workflow combines current Markdown metadata with authored\nrelationships, repository scopes, graph neighbors, and Git history. Search\nevidence keeps exact matches and QMD matches inspectable; graph neighbors and\nhistory remain context instead of silently changing relevance scores.\nSee the design and agent workflow.

      \n

      If all you need is local retrieval, standalone QMD may be the shorter path.\nChoose Wordcell when maintaining and reusing the knowledge around retrieval\nmatters too. We have not published a head-to-head retrieval-quality or latency\nbenchmark. The payload demonstration compares full-note and\nsnippet handoffs within Wordcell.

      \n

      Consider Basic Memory for an MCP-centered knowledge graph

      \n

      Basic Memory's knowledge format\nuses Markdown, observations, and relations to derive a connected knowledge\ngraph. Its local MCP tools\ninclude note reading, writing, editing, search, and graph context. Its\ncurrent tool reference\nalso documents text, vector, and hybrid search plus structured filters and\nbounded graph traversal. Local mode and a cloud service are available.

      \n

      Both projects preserve editable Markdown and expose connected knowledge to\nagents. Wordcell's emphasis is repository work: exact authored path scopes,\nAGENTS.md context routing, Git provenance, deterministic graph maintenance,\nand composable read-only SDK sessions. Basic Memory is worth considering when\nyou prefer its MCP tools and observation/relation format. Review both formats\nbefore sharing the same authored vault between them; Markdown compatibility\ndoes not imply identical metadata conventions.

      \n

      Keep Obsidian as an editor

      \n

      Obsidian is a local note-taking application with linked\nnotes and a plugin ecosystem. Wordcell is headless. You can edit an\nObsidian-compatible Markdown vault in Obsidian while using Wordcell for agent\nqueries and repository context; Wordcell does not replace Obsidian's editor.

      \n

      Obsidian Publish lets you select\ncontent for a hosted site. Its\nheadless publishing commands also\nsupport automation and require a Publish subscription. Wordcell instead emits\nordinary static files that you can host with a provider you choose. The hosting\nprovider may still charge you.

      \n

      Consider Quartz when the website is the main product

      \n

      Quartz is a Markdown static-site generator with\nsearch, backlinks, graph views, and customizable layouts. It supports\nprivate-page filtering,\nincluding an explicit publish: true workflow. Selective publishing is not\nunique to Wordcell.

      \n

      Wordcell is useful when the published site is one output of an agent-maintained\nknowledge base. Its publish command selects note paths, tags,\nmetadata, repository scopes, or bounded link neighborhoods, previews the\nselection, and emits a self-contained reader. Quartz may be a better fit if\ncustom website layouts and its publishing ecosystem are your main concern.\nWordcell's built-in reader deliberately has a smaller customization surface.

      \n

      Consider Supermemory for a memory API inside your product

      \n

      Supermemory stores\nmemories for the users of your application and isolates them with container\ntags, and API keys can be limited to chosen tags. Its\nconnectors sync content from\nsources such as Google Drive, Gmail, Notion, and GitHub, and its\nhosted MCP server gives\nassistants tools to search and add memories after an OAuth sign-in. Its\nsecurity page lists SOC 2\nType II compliance and a HIPAA business associate agreement on some plans.\nSupermemory local runs the\nsame engine on your machine with a model you provide.

      \n

      Wordcell keeps one person’s or one team’s memory as Markdown files in Git,\nwhich you and your agent read, edit, and review like code. It does not store\nmemory for the users of your product. wordcell import supermemory converts\nsaved Supermemory exports into notes, and wordcell mcp serves a vault to\nlocal MCP clients. Both commands are available from source until the next\nrelease. Migrate from Supermemory covers the\nexport, the import, and what does not transfer.

      \n

      Supermemory may be a better fit if you build a product that stores memory for\nmany users, or if you want connectors and extraction to run for you. We have\nnot published a head-to-head comparison of retrieval quality or speed.

      \n

      Consider Mem0 for extracted memories in your application

      \n

      Mem0 uses a model to extract\nmemories from the messages your application sends and scopes them by\nuser_id, agent_id, or run_id. The\nMem0 Platform runs the vector\nstore, language model, and embedder for you; with the open-source version you\nprovision each one. Mem0 also offers a\nhosted MCP server, and its\nsource code is licensed under Apache-2.0.

      \n

      Wordcell does not extract memories. A note holds what a person or agent chose\nto write, and exact search, links, and Git history work without a model. Mem0\nmay be a better fit if your application should remember facts about each of\nits users without anyone writing notes.

      \n

      Consider Zep for temporal context over conversations

      \n

      Zep describes itself as “the unified\ncontext layer for enterprise data.” It builds a graph for each user of your\napplication from chat messages and business data. When new data invalidates\na fact, Zep stores the time the fact became invalid on that fact. Zep uses\nGraphiti to derive\nits graph. Graphiti is an open-source framework that you can run yourself on\nNeo4j, FalkorDB, or Amazon Neptune.

      \n

      Wordcell records change by hand: a supersedes relation points from a newer\nnote to the one it replaces, and Git keeps every earlier version. Zep may be a\nbetter fit if your application needs context assembled automatically from\nmany users’ conversations.

      \n

      What “local” means in Wordcell

      \n

      Authored Markdown and Git remain yours. Exact queries, graph operations, and\nstatic site generation run locally without a hosted model or account. Optional\nQMD search runs local models after setup. Package installation and model\ndownloads require network access. Opt-in --rerank typesafe sends bounded\ncandidate context to the external Jev provider; capturing a URL contacts that\nsource, and publishing files to a host makes the selected content available\nthere. A cloud agent can receive whatever context you send it. Local storage\ndoes not change the agent host's data handling. See\nsecurity and data boundaries.

      \n", - "evidence": "

      Measure a smaller context handoff

      \n

      Wordcell can pass selected snippets to an agent before the agent opens full notes.\nIn the four-query example below, those handoffs contained 79.98% fewer UTF-8\nbytes than the same matching notes in full: 12,126 bytes instead of 60,584.\nThis is a small, reproducible payload-size demonstration on Wordcell's public\nseven-note knowledge base. It does not measure tokens, answer quality, speed,\nor an advantage over another search tool.

      \n

      What was measured

      \n

      Each query uses the real openKnowledgeBase().search() API in exact mode,\nwith up to five results and graph and Git context disabled. The results pass\nthrough packUntrustedSearchContext() with a 12,000-byte ceiling. The measured\noutput includes its source paths, snippets, metadata, and explicit\nuntrusted-content envelope. No result was dropped by the packer's byte ceiling\nin this run. The search result limit still applies, and snippets omit most of\neach note's body.

      \n

      The baseline is the sum of the complete, original UTF-8 Markdown files selected\nby that same query, including their frontmatter. It adds no artificial padding\nor tool wrappers. A note selected by two queries counts twice, once for each\nseparate handoff.

      \n\n\n\n\n\n\n\n\n\n\n\n
      Fixed querySelected notesFull selected notes, bytesPacked handoff, bytesReduction
      selective publishing219,3032,32287.97%
      repository memory529,7035,09782.84%
      typed relationships22,9492,35420.18%
      capture28,6292,35372.73%
      All four handoffs11 selections60,58412,12679.98%
      \n

      The aggregate is 100 × (1 − 12,126 / 60,584), rounded to two decimal places.\nIt weights by bytes; it is not the arithmetic mean of the percentages.\nThe small-note case saves 20.18%, which illustrates why one headline number\ncannot predict the savings on another vault.

      \n

      One read-only session scanned the seven-note vault once for all four queries.\nThe measurement opened zero semantic search sessions and used no model.\nWordcell still reads the corpus locally to build that snapshot. These numbers\ndescribe what crosses the context boundary, not reduced disk I/O. Passing every\nnote for every query would total 130,456 bytes; the raw report also records that\nseparate, less selective baseline.

      \n

      Reproduce the result

      \n

      The raw report contains every selected path, each source\nfile's byte count and SHA-256, the full packed output, output hashes, the exact\noptions, and aggregate calculations. The\nmeasurement script runs against the source SDK;\nits tests cover UTF-8 counting, repeatable\nresults, corpus changes, empty matches, and negative savings.

      \n

      Recorded on September 19, 2026 using the source SDK at the snapshot below and\nBun 1.3.14. Its package manifest declared version 0.21.3; this is a source\nmeasurement, not verification of the immutable 0.21.3 release artifact. The\npublic corpus is kb/ at commit\n454bfeca3f09680a0a9dc3a8f85d417684818ebc:\nseven scanned Markdown notes, including the index, totaling 32,614 bytes.\nAGENTS.md guides are excluded by the vault scanner. The corpus identity is\nSHA-256 over the compact JSON array of {path, bytes, sha256} records sorted by\npath, with no trailing newline:

      \n
      19854ac834111f66cd4b1bbc0cfc8610f5b92312403d11d72dfee8fef8a95c9b\n
      \n

      From a source checkout with its dependencies installed, measure its current KB:

      \n
      bun scripts/product-evidence.ts\n
      \n

      To restore the frozen corpus into a temporary directory without changing your\nworking files, run these commands from the repository root:

      \n
      evidence_dir="$(mktemp -d)"\ngit archive 454bfeca3f09680a0a9dc3a8f85d417684818ebc kb | tar -x -C "$evidence_dir"\nbun scripts/product-evidence.ts "$evidence_dir/kb" > "$evidence_dir/result.json"\n
      \n

      The script prints JSON and does not write the vault or a search index. Compare\ncorpus, configuration, cases, and aggregate with the recorded report.\nThe tool field reports the installed source version, so a later release may\nhave a different version field even when its measured output is identical.\nHardware is not specified because this report makes no timing measurement.

      \n

      Use the measurement correctly

      \n

      This demonstrates the cost of the first context handoff. It leaves the original\nnotes available for follow-up reading. If an agent then reads every selected\nnote, its total context can exceed the full-note baseline. Model token counts,\nprompt caching, tool wrappers, follow-up calls, and billing depend on the host;\nno fixed byte-to-token conversion is assumed.

      \n

      These four queries were chosen to exercise product topics. They are not an\nindependent or representative relevance test set. The small corpus consists of\nWordcell's own engineering notes and plans. A shorter payload does not establish\nthat it contains enough information to answer a question correctly.

      \n

      QMD and other tools also return snippets and bounded results. This report\ncompares two Wordcell handoff strategies, not Wordcell against QMD, Basic\nMemory, or a well-tuned rg workflow. See the comparison guide\nfor differences in workflow and features.

      \n

      Retrieval quality needs separate evidence

      \n

      The six-case synthetic rank-fusion fixture in src/benchmark.ts checks\ndeterministic behavior. It is not a retrieval-quality or performance benchmark\nand must not be cited as one. Wordcell provides an\nevaluation-builder API for frozen corpora and relevance\njudgments. A competitive quality claim needs a published corpus, independently\njudged queries, pinned tool versions and settings, the raw rankings, and named\nhardware for any timing claims. No competitive quality or latency result is\nclaimed here.

      \n

      Distinguish the graph engine from search quality

      \n

      Oh backs Wordcell's named graph queries and source proofs.\nWordcell keeps Markdown and Git authoritative and builds a disposable projection\nfor those queries. Its exact search, optional QMD search, and optional hosted\nJev reranking follow separate retrieval paths.

      \n

      Oh's conversation-memory benchmarks evaluate its own memory-retrieval API,\nreader models, and evaluation protocols. Those scores do not transfer to a\nWordcell vault merely because it uses the same library. Wordcell's\nSciFact study compares exact search with\nhosted Jev reranking on the same public queries and candidate windows. It\nmeasures source ranking, without generating answers; this page measures only\nthe size of the first context handoff. The graph guide\nexplains the integration.

      \n", + "evidence": "

      Measure a smaller context handoff

      \n

      Wordcell can pass selected snippets to an agent before the agent opens full notes.\nIn the four-query example below, those handoffs contained 79.98% fewer UTF-8\nbytes than the same matching notes in full: 12,126 bytes instead of 60,584.\nThis is a small, reproducible payload-size demonstration on Wordcell's public\nseven-note knowledge base. It does not measure tokens, answer quality, speed,\nor an advantage over another search tool.

      \n

      What was measured

      \n

      Each query uses the real openKnowledgeBase().search() API in exact mode,\nwith up to five results and graph and Git context disabled. The results pass\nthrough packUntrustedSearchContext() with a 12,000-byte ceiling. The measured\noutput includes its source paths, snippets, metadata, and explicit\nuntrusted-content envelope. No result was dropped by the packer's byte ceiling\nin this run. The search result limit still applies, and snippets omit most of\neach note's body.

      \n

      The baseline is the sum of the complete, original UTF-8 Markdown files selected\nby that same query, including their frontmatter. It adds no artificial padding\nor tool wrappers. A note selected by two queries counts twice, once for each\nseparate handoff.

      \n\n\n\n\n\n\n\n\n\n\n\n
      Fixed querySelected notesFull selected notes, bytesPacked handoff, bytesReduction
      selective publishing219,3032,32287.97%
      repository memory529,7035,09782.84%
      typed relationships22,9492,35420.18%
      capture28,6292,35372.73%
      All four handoffs11 selections60,58412,12679.98%
      \n

      The aggregate is 100 × (1 − 12,126 / 60,584), rounded to two decimal places.\nIt weights by bytes; it is not the arithmetic mean of the percentages.\nThe small-note case saves 20.18%, which illustrates why one headline number\ncannot predict the savings on another vault.

      \n

      One read-only session scanned the seven-note vault once for all four queries.\nThe measurement opened zero semantic search sessions and used no model.\nWordcell still reads the corpus locally to build that snapshot. These numbers\ndescribe what crosses the context boundary, not reduced disk I/O. Passing every\nnote for every query would total 130,456 bytes; the raw report also records that\nseparate, less selective baseline.

      \n

      Reproduce the result

      \n

      The raw report contains every selected path, each source\nfile's byte count and SHA-256, the full packed output, output hashes, the exact\noptions, and aggregate calculations. The\nmeasurement script runs against the source SDK;\nits tests cover UTF-8 counting, repeatable\nresults, corpus changes, empty matches, and negative savings.

      \n

      Recorded on September 19, 2026 using the source SDK at the snapshot below and\nBun 1.3.14. Its package manifest declared version 0.21.3; this is a source\nmeasurement, not verification of the immutable 0.21.3 release artifact. The\npublic corpus is kb/ at commit\n454bfeca3f09680a0a9dc3a8f85d417684818ebc:\nseven scanned Markdown notes, including the index, totaling 32,614 bytes.\nAGENTS.md guides are excluded by the vault scanner. The corpus identity is\nSHA-256 over the compact JSON array of {path, bytes, sha256} records sorted by\npath, with no trailing newline:

      \n
      19854ac834111f66cd4b1bbc0cfc8610f5b92312403d11d72dfee8fef8a95c9b\n
      \n

      From a source checkout with its dependencies installed, measure its current KB:

      \n
      bun scripts/product-evidence.ts\n
      \n

      To restore the frozen corpus into a temporary directory without changing your\nworking files, run these commands from the repository root:

      \n
      evidence_dir="$(mktemp -d)"\ngit archive 454bfeca3f09680a0a9dc3a8f85d417684818ebc kb | tar -x -C "$evidence_dir"\nbun scripts/product-evidence.ts "$evidence_dir/kb" > "$evidence_dir/result.json"\n
      \n

      The script prints JSON and does not write the vault or a search index. Compare\ncorpus, configuration, cases, and aggregate with the recorded report.\nThe tool field reports the installed source version, so a later release may\nhave a different version field even when its measured output is identical.\nHardware is not specified because this report makes no timing measurement.

      \n

      Use the measurement correctly

      \n

      This demonstrates the cost of the first context handoff. It leaves the original\nnotes available for follow-up reading. If an agent then reads every selected\nnote, its total context can exceed the full-note baseline. Model token counts,\nprompt caching, tool wrappers, follow-up calls, and billing depend on the host;\nno fixed byte-to-token conversion is assumed.

      \n

      These four queries were chosen to exercise product topics. They are not an\nindependent or representative relevance test set. The small corpus consists of\nWordcell's own engineering notes and plans. A shorter payload does not establish\nthat it contains enough information to answer a question correctly.

      \n

      QMD and other tools also return snippets and bounded results. This report\ncompares two Wordcell handoff strategies, not Wordcell against QMD, Basic\nMemory, or a well-tuned rg workflow. See the comparison guide\nfor differences in workflow and features.

      \n

      Retrieval quality needs separate evidence

      \n

      The six-case synthetic rank-fusion fixture in src/benchmark.ts checks\ndeterministic behavior. It is not a retrieval-quality or performance benchmark\nand must not be cited as one. Wordcell provides an\nevaluation-builder API for frozen corpora and relevance\njudgments. A competitive quality claim needs a published corpus, independently\njudged queries, pinned tool versions and settings, the raw rankings, and named\nhardware for any timing claims. No competitive quality or latency result is\nclaimed here.

      \n

      Distinguish the graph engine from search quality

      \n

      Oh backs Wordcell's named graph queries and source proofs.\nWordcell keeps Markdown and Git authoritative and builds a disposable projection\nfor those queries. Its exact search, optional QMD search, and optional hosted\nJev reranking follow separate retrieval paths.

      \n

      Oh's conversation-memory benchmarks evaluate its own memory-retrieval API,\nreader models, and evaluation protocols. Those scores do not transfer to a\nWordcell vault merely because it uses the same library. The\nbenchmarks page reports Oh's 500-question\nLongMemEval-S study, its LoCoMo run, and its smaller pilot with Supermemory as\nOh's results, each with its limits. Wordcell's\nSciFact study compares exact search with\nhosted Jev reranking on the same public queries and candidate windows. It\nmeasures source ranking, without generating answers; this page measures only\nthe size of the first context handoff. The graph guide\nexplains the integration.

      \n", "graph-authority": "

      Query the derived graph

      \n

      Wordcell's named graph programs use its pinned immutable Oh release.\nMarkdown is authoritative. Queries never add links or inferred\nrelationships to notes, and the existing wordcell graph --json report keeps\nits format.

      \n

      Wordcell 0.22.4 pins Oh 0.12.0. Earlier Wordcell 0.22.1 used Oh 0.11.0.\nThe graph contracts stay compatible and existing vaults need no migration.

      \n

      How Wordcell and Oh fit together

      \n

      Wordcell owns the Markdown knowledge base: authoring, capture, search,\nrepository context, and publishing. Oh is the embedded\nmemory framework that evaluates Wordcell's named graph queries and carries\ntheir source proofs. No Oh account or separate service is required.

      \n\n\n\n\n\n\n\n\n\n
      LayerRoleAuthority
      Markdown and GitNote text, frontmatter, authored relationships, and history.The record you own and edit.
      WordcellReads a bounded vault snapshot, resolves links, and prepares named queries.Current files determine the graph facts.
      OhEvaluates the derived graph and returns bounded proofs.A replaceable projection of those facts.
      \n

      For example, a backlinks query returns the note that authored a link, the linked\nnote, and the source line. Its proof identifies that source note's content\ndigest and the exact projection revision. Editing a note changes the snapshot;\nopen a new session to query the new record. A query never writes the relationship\nback into a note.

      \n

      After the quick start has created notes/parser-contract,\nsave a second note with one authored link and query its backlink:

      \n
      wordcell note create notes/retry-review --title "Retry review" --type concept --body "Use [[notes/parser-contract]] when changing retry behavior." --root kb\nwordcell graph query --program backlinks --note notes/parser-contract --root kb --json\n
      \n

      The returned source is notes/retry-review and the target is\nnotes/parser-contract. Their proof follows the link you wrote. This uses an\nin-memory projection and creates no .wordcell/oh.sqlite file.

      \n

      Wordcell search uses its own exact matching and optional QMD local retrieval.\nOptional Jev reranking reorders a bounded candidate window through a hosted\nprovider. These search paths do not invoke Oh's conversation-memory retrieval\nAPI. Oh's memory benchmarks therefore do not measure Wordcell search or answer\nquality. See Wordcell's retrieval study and\ncontext-payload measurement for the paths evaluated here.

      \n

      The graph engine prefers the bundled Rust implementation automatically when\navailable and retains the TypeScript fallback. Choosing an engine requires no\nexperimental mode. The source revision, proof limits, and disposable-cache\nboundary are the same in either case.

      \n

      Query without creating a cache

      \n

      Use an exact extensionless note ID. The default query creates an in-memory\nprojection of a complete bounded Markdown snapshot and closes it afterward.

      \n
      wordcell graph query --program backlinks --note notes/design --root kb --json\nwordcell graph query --program reachability --note notes/design --depth 3 --root kb --json\nwordcell graph query --program relation-closure --note notes/design --predicate depends-on --depth 3 --root kb --json\nwordcell graph query --program scope-route --scope src --root kb --json\n
      \n\n\n\n\n\n\n\n\n\n\n\n\n
      ProgramReturned columnsMeaning
      backlinkssource, target, line, kind, predicateIncoming resolved contextual links and local typed relationships.
      reachabilitysource, target, depthPositive contextual and typed paths of lengths 1 through the requested depth.
      relation-closuresource, target, depthPositive local relationship paths using one exact predicate.
      scope-routenote, scopeExact authored repository_scopes declarations.
      shared-tagsnote, other, tagPositive tag evidence shared with another note.
      shared-conceptsnote, other, conceptPositive evidence that two ordinary notes connect to the same concept, in either direction.
      \n

      Depth defaults to 3 and is bounded at 8. A cyclic path may return its starting\nnote; the same target at different lengths is a different row. External\nkb:// relationship targets remain external facts and never enter local path\nclosure. Broken and ambiguous links remain diagnostics in the existing graph\nreport. The configured catalog is excluded from graph relationships.

      \n

      There is no arbitrary Datalog text evaluator. These reviewed named programs\nuse positive rules. Wordcell computes absence, orphans and counts from a\ncomplete snapshot; an Oh proof does not prove that an authored claim is true.

      \n

      Rebuild and verify local state

      \n
      wordcell graph rebuild --root kb --json\nwordcell graph verify --root kb --json\nwordcell graph query --program backlinks --note notes/design --root kb --persisted --json\n
      \n

      Only rebuild writes .wordcell/oh.sqlite and its self-ignoring cache directory.\nIt serializes local rebuilds, applies changed records and deletions in a private\nstaging database, verifies replay, rechecks the Markdown revision, and atomically\ninstalls the result. A failed build leaves the previous database in place.\nThe operation never edits Markdown or initializes a vault.

      \n

      Verification and --persisted queries read a bounded copy into memory. They\nreject a missing, stale, foreign or corrupt cache; they do not repair it or\ncreate files. SQLite sidecars and symbolic or hard links are refused. Close any\nforeign writer before retrying. To rebuild a damaged disposable cache from\nMarkdown, explicitly request a fresh database:

      \n
      wordcell graph rebuild --fresh --root kb --json\n
      \n

      The previous file stays in place until its replacement passes verification.\nA fresh rebuild also removes accumulated derived history. Oh supports atomic\nrecord updates, but its query engine evaluates the complete projection after\nan input change. Equal source snapshots produce equal rows and source proofs;\nhistory-bound projection identities can differ after a fresh rebuild.\nIn-memory projections use a fixed synthetic genesis time so identical snapshots\ncan reproduce and verify their proof results across sessions. That logical time\nis never evidence of when a note was authored or an operation occurred.

      \n

      Inspect proof and limit information

      \n

      Each result identifies the vault, exact source revision, named request,\nprojection digest, evaluated limits, rows, and supporting proof trees. Fact\nproofs name the source note, its content digest and its exact Oh record digest.\nDerived proofs name the applied rule and its premises. A declared document_id\nretains record identity across renames; notes without one use an explicit path\nidentity. Wordcell never invents or writes IDs. Invalid or duplicate stable IDs\nand invalid scope metadata stop extraction.

      \n

      The graph boundary accepts at most 4,000 Markdown notes, 100,000 facts, and\n64 MiB of source text; individual query atoms are bounded at 16 KiB. Queries\nalso have explicit row, work, derived-tuple, round, result-byte and proof\nbudgets. --limit sets the row limit up to 1,000. SDK callers can reduce each\nbudget. Work exhaustion fails instead of claiming a complete answer. Row or\nproof truncation is retained in JSON and yields CLI exit code 4. Inspect\ntruncated, proofsTruncated, and proof truncation nodes before using a result\nas complete evidence.

      \n
      import { openKnowledgeBase } from "@hraness/wordcell";\nconst kb = await openKnowledgeBase({ root: "kb" });\ntry {\n  const result = await kb.graphQuery({\n    program: "reachability", note: "notes/design", depth: 3,\n    limits: { rows: 100, workUnits: 500_000 },\n  });\n  console.log(result.revision, result.rows);\n  console.log(await kb.graphVerifyResult(result));\n} finally {\n  await kb.close();\n}\n
      \n

      A session intentionally retains one read-only snapshot. Reopen the session\nwhen Markdown changes. graphVerifyResult re-evaluates a bounded result\nagainst that session and rejects modified, foreign or stale evidence; it does\nnot authorize writing any fact into Markdown. The lower-level graph API is\navailable at @hraness/wordcell/graph-authority.

      \n

      Add positive proofs to percolation

      \n
      wordcell percolate notes/design --proofs --root kb --json\n
      \n

      This opt-in result has kind wordcell.graph-percolation. suggestions retains\nthe existing percolation V2 candidates. positiveSupport contains separate\nshared-tag and shared-concept query results, each with its own limits and\nproofs. Missing concepts, absent edges, support counts, and predicate selection\nremain Wordcell or author decisions. A positive shared tag does not establish\nthat two notes need a relationship. Read the cited Markdown before editing.

      \n

      SDK callers use kb.percolateWithProofs({ note: "notes/design" }) or\npercolateWithGraph(snapshot, options) from\n@hraness/wordcell/graph-percolation. No proof operation writes notes.

      \n", - "overview": "

      Wordcell

      \n

      Install the Agent Skill

      \n

      Wordcell keeps decisions, plans, and sources as Markdown files beside your\ncode. Coding agents find them by exact words, by meaning with an optional\nlocal model, or from the file they are about to change.

      \n

      A new coding-agent session can read your code, but not the decisions that\nstayed in the last session's chat. Wordcell keeps those decisions as Markdown\nfiles beside the repository, with the plans that depend on them and the web\npages and PDFs that informed them. Tie a note to the paths it explains, and an\nagent about to change that code runs one command to get the notes and plans\nfor that path. Exact search, backlinks, and Git history run on your machine\nwith no account or model, and every index rebuilds from files you can read in\nany editor. Wordcell is free and open source.

      \n

      Web capture, optional hosted reranking, and your agent's provider reach other\nservices; Privacy and boundaries says what each one\nsends.

      \n

      Documentation · Comparisons · Measured evidence · Changelog

      \n

      Why Wordcell

      \n
        \n
      • Keep what you learn in files you own. Write decisions, sources, and plans\nin Markdown. Obsidian, Git, and Wordcell read the same record, and every\nindex rebuilds from the files.
      • \n
      • Find it by words or by meaning. Exact search needs no model or account.\nOptional local semantic search joins each match to current metadata, links,\nand history instead of returning isolated text.
      • \n
      • Author the connections. Wikilinks and typed relationships turn notes into\na graph you can query. Backlinks and graph queries use only what you wrote,\nand wordcell percolate suggests missing links for you to review without\nwriting them into your notes.
      • \n
      • Give coding agents the reasons behind the code. Tie notes to repository\npaths, list the commits behind a note, and let the next session start from\nthe notes for the file it is changing. Only context you save becomes part of\nthe record.
      • \n
      \n

      Plain Markdown may be enough for a small set of notes. QMD is a good fit for\nlocal document retrieval and also supplies Wordcell's optional semantic search.\nWordcell adds a connected workflow for repository context, authored relationships,\nGit evidence, and selective publishing. Compare the tradeoffs.

      \n

      Wordcell keeps the record in Markdown files you own, rebuilds every index from those files, and lets the next session start from the notes for the file it is changing: the design every Hraness project shares. The thread through hraness follows that design across the projects, and the ALGAL vision states the bet behind it.

      \n

      Install

      \n

      Bun 1.3.14 or newer and Git are required.\nThe CLI and TypeScript SDK run with Bun. Install the versioned GitHub archive:

      \n
      bun add --global --ignore-scripts https://github.com/hraness/wordcell/releases/download/v0.22.5/hraness-wordcell-0.22.5.tgz\nwordcell --help\n
      \n

      Prefer npm? The same release is mirrored there:

      \n
      npm install --global --ignore-scripts @hraness/wordcell@0.22.5\nwordcell --help\n
      \n

      Keep Bun in PATH for either installation. If your shell cannot find wordcell,\nadd your package manager's global executable directory to PATH and reopen the\nterminal. Optional semantic search, browser capture, and PDF tools have\nadditional prerequisites.

      \n

      Keep one decision available to the next session

      \n

      Run this from a directory where you want a new kb/ folder. It creates one\nMarkdown note and finds it without downloading a model or contacting a service:

      \n
      wordcell init kb\nwordcell note create notes/parser-contract \\\n  --title "Parser contract" --type concept --tag architecture \\\n  --body "Parser retries stop after three attempts." --root kb\nwordcell search "parser retries" --root kb --mode exact\n
      \n

      The result includes notes/parser-contract and the saved retry constraint.\nOpen kb/notes/parser-contract.md to see the ordinary Markdown file. Wordcell\nadds a stable document_id in its frontmatter; you can edit the prose normally.

      \n

      Already have Markdown or an Obsidian vault? In v0.22.0 or newer, search\nthe existing folder without initialization or an index.md file:

      \n
      wordcell search "a phrase from your notes" --root /path/to/your/vault --mode exact\n
      \n

      Replace the path and phrase with your own. You don't need to initialize,\nconvert, or move the existing files to search them.

      \n

      Use with a coding agent

      \n

      After trying the CLI, install the public Agent Skill into a compatible agent,\nsuch as Claude Code, Codex, Cursor, or GitHub Copilot:

      \n
      bunx skills add hraness/wordcell#v0.22.5 --skill wordcell\n
      \n

      Then ask:

      \n
      Use Wordcell to search for "parser retries" in ./kb using exact mode.\nRead the matching note and explain the saved constraint.\n
      \n

      The skill installs instructions, not a service. Installation does not create a\nvault, modify your notes, or grant an agent permission to access other accounts.\nThe agent still follows its own provider and data-handling settings.\nInspect the skill.

      \n

      Recover the stopped session

      \n

      Link a plan to the decision you saved:

      \n
      wordcell note create plans/parser-v2 \\\n  --title "Parser v2" --type plan \\\n  --body "The plan implements [[notes/parser-contract|the parser contract]]." \\\n  --root kb\nwordcell backlinks notes/parser-contract --root kb\n
      \n

      The backlink result includes plans/parser-v2, so a later session can find the\nwork that depends on the constraint.

      \n

      For code-path lookup, add this field inside the existing frontmatter of\nkb/notes/parser-contract.md:

      \n
      repository_scopes:\n  - packages/parser\n
      \n

      From your repository root, use an actual path under that scope and inspect the\nreturned notes and inherited AGENTS.md guides:

      \n
      wordcell context packages/parser/src/index.ts --root kb --repo .\n
      \n

      Commit the vault with your repository to preserve its history, then use\nwordcell history notes/parser-contract --root kb --repo . to inspect the\ncommits behind the note. History is optional and requires recorded Git commits.

      \n

      These views recover saved decisions, related work, rules, and provenance.\nThey do not reconstruct private chat or prove that the note is still correct.\nOpen the returned Markdown and guides before acting on them.

      \n

      Rerank a search window

      \n
      wordcell search "why releases use immutable archives" --root kb --mode exact \\\n  --rerank typesafe --rerank-limit 25 --limit 5 --json\n
      \n

      This optional hosted lane uses TypeSafe's pinned jev-1.13.0 model. It sends\nbounded query and note snippets to the provider, needs a private local\ncredential, and incurs provider charges. Exact identities remain first; a\nprovider failure retains the baseline order with a diagnostic. See the\nsetup, SDK examples, measured results, and limits.

      \n

      What you can do

      \n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n
      TaskCommandEvidence and effects
      Find a saved decisionwordcell search "parser retries" --root kb --mode exactReads current Markdown; no model or network request.
      Recover context for codewordcell context packages/parser/src/index.ts --root kb --repo .Returns scoped notes, plans, and inherited AGENTS.md rules.
      Inspect explicit connectionswordcell backlinks notes/parser-contract --root kbReturns notes that link to the decision.
      Search by meaningwordcell search "retry policy" --root kb --mode hybridAdds optional local QMD keyword and vector retrieval; model setup is required.
      Query with graph proofswordcell graph query --program scope-route --scope packages/parser --root kbReturns bounded results tied to the source revision. Queries do not write a cache.
      Capture a sourcewordcell clip https://example.com/article --output kb/articlesReads the selected URL and writes a Markdown bundle with a capture receipt.
      Capture a PDFwordcell pdf /absolute/path/to/document.pdf --output kb/articlesPreserves the original PDF and extracted evidence; Poppler is required.
      Check the vaultwordcell check --root kbReports structural and attachment problems without editing files.
      Publish selected noteswordcell publish --root kb --out site/ --include notes/parser-contract --dry-run --jsonPreviews a static site selection locally; remove --dry-run to build it.
      Preview a sitewordcell serve --root site --port 8080Serves a published site on a loopback static file server with the emitted 404.html fallback.
      Connect an MCP clientwordcell mcp --root kbServes this vault to a local MCP client over standard input and output; --read-only removes the write tools. Available from source until the next release.
      Import from Supermemorywordcell import supermemory documents.json --root kbTurns saved Supermemory API responses into Markdown notes and reports local edits as conflicts on later imports. Available from source until the next release.
      \n

      Use --json for structured output and wordcell --help for the complete command\nsurface. Full command reference.

      \n

      Publish a selected part of your vault

      \n

      Preview the decision and plan from the example before writing an output folder:

      \n
      wordcell publish --root kb --out site \\\n  --include notes/parser-contract --include plans/parser-v2 --dry-run --json\n
      \n

      Review the selection, then build and preview it:

      \n
      wordcell publish --root kb --out site \\\n  --include notes/parser-contract --include plans/parser-v2\nwordcell serve --root site --port 8080\n
      \n

      Open http://127.0.0.1:8080. You get readable pages, linked notes, and search\nthat runs in the browser. Upload the site/ folder to your chosen static host\nwhen you want to share it; publish itself never uploads anything.

      \n

      For a repeatable slice, use path patterns or combine folders, tags, metadata,\ncode scopes, and linked neighborhoods:

      \n
      wordcell publish --root kb --out site-notes \\\n  --include-glob 'notes/**/*.md' --exclude-glob '**/draft-*' --dry-run --json\n
      \n

      Preview reports show up to 20 selected IDs by default, with a total count and\nselection digest. They keep note bodies out of the agent's context. Use\n--list-limit to adjust that preview without changing what gets published. publish: false excludes a note, but selected\nprose and attachments still need review before sharing: selection is not secret\nredaction. Selection recipes and hosting guide.

      \n

      Use a vault as agent memory

      \n

      The MCP server, the Supermemory importer, and the skill's session-memory\nworkflow described here are available from source until the next release. To\nuse them before that release, install the CLI from a checkout as the\ninstallation reference\nshows, and add the skill from the main branch with\nbunx skills add hraness/wordcell --skill wordcell.

      \n

      wordcell mcp --root kb serves a vault to a local MCP client, such as Claude\nCode, Claude Desktop, Cursor, or Codex, over standard input and output. The\nclient can search, list, and read notes, follow links, create notes, update a\nnote body at a known revision, and add typed relations. --read-only removes\nthe write tools. Connect a client\nshows the setup for each host.

      \n

      With the Agent Skill, ask your agent to save the session. It writes a dated\nsession note, links it to the notes it changed, and keeps a profile note for\nyou or your project. Wordcell extracts no facts and writes no note on its own.\nSession memory and profiles.

      \n

      To move from Supermemory, wordcell import supermemory turns saved Supermemory\nAPI responses into Markdown notes. The migration guide\ncovers export, import, and what does not transfer, and the\nSupermemory comparison lists when to\nchoose Supermemory instead. Sync with Git\nkeeps one vault current on several machines.

      \n

      Evidence and comparisons

      \n

      In a four-query example over a seven-note public vault, packed search snippets\nused 80% fewer UTF-8 bytes than passing the same matching notes in full:\n12,126 versus 60,584 bytes. This measures context payload size, not tokenizer\ncounts, answer quality, latency, or a win over another search tool.

      \n

      In a separate public SciFact study, optional hosted Jev reranking placed a\njudged relevant result first for 161 of 300 queries, versus 101 with\nWordcell exact search alone. It sends bounded context to a paid provider;\nthis is evidence on scientific abstracts, not a comparison with QMD or a\nguarantee for repository notes. Results and limits.

      \n

      Wordcell's benefit is selecting relevant context and keeping its sources\ninspectable. Local ownership is also available in other tools, and Wordcell\ndoes not claim to beat QMD's retrieval quality or every Markdown workflow.

      \n

      Measured evidence\nshows a reproducible public-vault example, with the inputs, output sizes, and\nlimits beside each result. The comparison guide\ncovers Markdown alone, QMD, Basic Memory, Obsidian, static publishing tools,\nSupermemory, Mem0, and Zep using their own documentation. Choose the smallest\nworkflow that meets your needs.

      \n

      The benchmarks page shows the payload and\nSciFact results beside the published results of Oh, the embedded memory\nframework, each with its source data and limits. Oh's scores measure its own\nmemory-retrieval path, not a Wordcell vault.

      \n

      How the files fit together

      \n
      repository/\n├── AGENTS.md                # rules that govern edits\n├── packages/parser/\n│   └── AGENTS.md            # rules scoped to this code\n└── kb/\n    ├── index.md             # authored or managed front door\n    ├── articles/            # captured sources and assets\n    ├── notes/               # maintained explanations\n    └── plans/               # decisions and outcomes\n
      \n

      Markdown, YAML frontmatter, explicit wikilinks, and Git hold the record. Open\nthe same files in Obsidian, a text editor, or ordinary file-search tools.\nApplication code does not need to import Wordcell or its vault.

      \n

      Wordcell is the Markdown knowledge base. Oh is the\nembedded memory framework that backs its named graph queries and source proofs.\nMarkdown and Git remain authoritative. Query the graph immediately without an\nOh account, service, or persisted database:

      \n
      wordcell graph query --program backlinks --note notes/parser-contract --root kb --json\n
      \n

      The result traces each returned link to its source note and revision. Only an explicit\nwordcell graph rebuild --root kb writes .wordcell/oh.sqlite, the file stays\nignored and rebuildable, and nothing flows from the projection back into notes.\nBacklinks and typed relationships come from authored links. Percolation\nsuggests connections for review and does not add inferred edges to notes.

      \n

      Wordcell search combines its own exact matching with optional QMD local search\nand optional hosted Jev reranking. Oh also offers memory retrieval for applications;\nits conversation-memory benchmark scores measure that separate path. They do not\nestablish Wordcell's retrieval or answer quality. How the integration works.

      \n

      Graph proofs explain a supported derivation from a specific source revision.\nThey do not prove that a note is true or that a missing relationship cannot\nexist. Graph queries and proof limits.

      \n

      Build with the TypeScript SDK

      \n

      Add the same immutable release to a Bun project:

      \n
      bun add --exact --ignore-scripts https://github.com/hraness/wordcell/releases/download/v0.22.5/hraness-wordcell-0.22.5.tgz\n
      \n

      The SDK provides read-only vault sessions, metadata queries, search, graph\nproofs, Git context, and composable workflows. A session owns one snapshot;\nreopen it after Markdown changes. SDK and workflow examples\nshow the public imports and lifecycle.

      \n

      Privacy and boundaries

      \n
        \n
      • Structural queries and exact search read local files. Optional semantic\nsearch downloads its model on first use and runs locally. Optional Jev\nreranking sends the query and bounded candidate context to a remote provider;\nit is off by default.
      • \n
      • URL capture contacts the requested source. Signed-in capture uses only\nexplicitly selected browser state. Review the security policy\nbefore using it with private sources.
      • \n
      • An agent that reads the vault follows its own provider and data-handling\nsettings. Keep private records out of public repositories and outputs.
      • \n
      • Git history is opt-in. Saved notes preserve recorded context; Wordcell does\nnot reconstruct unsaved conversations or silently record every agent action.
      • \n
      \n

      Documentation

      \n

      The documentation follows the Diataxis split: a tutorial to learn the loop,\nhow-to guides for tasks, reference for exact interfaces, and explanation for\nthe design. Browse it on the documentation index.

      \n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n
      Read nextPurpose
      Get startedLearn the full loop on a first vault: save, find, connect, and publish one note.
      Agent workflowSet up, query, maintain, and revise repository memory.
      Local MCP serverConnect Claude Code, Claude Desktop, Cursor, or Codex to a vault and review the write tools. Available from source until the next release.
      Installation and command referenceExact interfaces, SDK imports, optional adapters, and troubleshooting prerequisites.
      Web capture and PDF captureSave sources with provenance, assets, and explicit completeness limits.
      Publish selected notesPreview a slice, build a static site, and choose how to host it.
      Graph guideNamed queries, proofs, revisions, resource limits, and cache recovery.
      Portfolio federationSearch only selected, authorized vaults.
      Sync with GitKeep one vault current on several machines with a private repository.
      Migrate from SupermemoryExport documents and memory entries, import them as notes, and replace connectors. The importer is available from source until the next release.
      Design and memory rationaleFile contracts, design choices, and evaluation context.
      Release verificationVerify archive identity, signatures, and provenance.
      \n

      Installation reference

      \n

      Alternative installs and optional capabilities.

      \n

      The kb vault format

      \n

      Existing kb/, index.md, kb:// identifiers, and Markdown formats remain\ncompatible. Vault format reference.

      \n

      Command surface

      \n

      Complete CLI reference.

      \n

      Capture reference

      \n

      Capture commands and dependencies.

      \n

      Graph reference

      \n

      Graph and SDK interfaces.

      \n

      Agent skills

      \n

      Installation and runtime contract.\nThe same skill is included at node_modules/@hraness/wordcell/skills/wordcell/.

      \n

      Release notes

      \n

      Version 0.22 adds selective static publishing and local preview. Choose notes,\nfolders, path patterns, metadata, or linked neighborhoods, then inspect a\nbounded selection report before building. Existing vaults need no migration.\nThe deprecated kb command was removed in 0.21.0; use wordcell.\nAll release notes and upgrade instructions.

      \n

      Contributing

      \n

      Run bun run check for source, architecture, package installation, types, and\nnative-helper validation. The site has its own bun run check under site/.\nSee Contributing, Security, and the MIT License.

      \n

      Install Wordcell from GitHub Releases ·\nWordcell source on GitHub ·\nWordcell overview

      \n

      Optional development support

      \n

      wordcell support shows optional development support for Wordcell. Features stay\nfree and no account is required. Accounts presents current terms; the person\nconfirms any payment in their browser. Wordcell has no product newsletter and\ndoes not look up an email address for support.

      \n

      After useful standalone work, agents can receive a discovery record on stderr.\nRead wordcell support protocol --json for the shared closeout procedure.\nHelp, diagnostics, evaluation, SDK calls and nested commands stay quiet.\nHRANESS_SUPPORT=off or HRANESS_SUPPORT_AUDIENCE=off suppresses incidental\ninvitations. Explicit human terminal rendering requires\nHRANESS_SUPPORT_AUDIENCE=human; the default audience is an agent, including in\na pseudo-terminal.

      \n

      wordcell support dismiss disables invitations across participating tools on\nthis machine. snooze pauses them for thirty days, enable restores them, and\nstatus --json shows their separate local preferences. These commands do not\nchange a vault, sign up or pay. Acknowledged invitations share a seven-day\ncooldown; discovery itself does not consume it.

      \n", + "overview": "

      Wordcell

      \n

      Install the Agent Skill

      \n

      Wordcell keeps decisions, plans, and sources as Markdown files beside your\ncode. Coding agents find them by exact words, by meaning with an optional\nlocal model, or from the file they are about to change.

      \n

      A new coding-agent session can read your code, but not the decisions that\nstayed in the last session's chat. Wordcell keeps those decisions as Markdown\nfiles beside the repository, with the plans that depend on them and the web\npages and PDFs that informed them. Tie a note to the paths it explains, and an\nagent about to change that code runs one command to get the notes and plans\nfor that path. Exact search, backlinks, and Git history run on your machine\nwith no account or model, and every index rebuilds from files you can read in\nany editor. Wordcell is free and open source.

      \n

      Web capture, optional hosted reranking, and your agent's provider reach other\nservices; Privacy and boundaries says what each one\nsends.

      \n

      Documentation · Comparisons · Measured evidence · Changelog

      \n

      Why Wordcell

      \n
        \n
      • Keep what you learn in files you own. Write decisions, sources, and plans\nin Markdown. Obsidian, Git, and Wordcell read the same record, and every\nindex rebuilds from the files.
      • \n
      • Find it by words or by meaning. Exact search needs no model or account.\nOptional local semantic search joins each match to current metadata, links,\nand history instead of returning isolated text.
      • \n
      • Author the connections. Wikilinks and typed relationships turn notes into\na graph you can query. Backlinks and graph queries use only what you wrote,\nand wordcell percolate suggests missing links for you to review without\nwriting them into your notes.
      • \n
      • Give coding agents the reasons behind the code. Tie notes to repository\npaths, list the commits behind a note, and let the next session start from\nthe notes for the file it is changing. Only context you save becomes part of\nthe record.
      • \n
      \n

      Plain Markdown may be enough for a small set of notes. QMD is a good fit for\nlocal document retrieval and also supplies Wordcell's optional semantic search.\nWordcell adds a connected workflow for repository context, authored relationships,\nGit evidence, and selective publishing. Compare the tradeoffs.

      \n

      Wordcell keeps the record in Markdown files you own, rebuilds every index from those files, and lets the next session start from the notes for the file it is changing: the design every Hraness project shares. The thread through hraness follows that design across the projects, and the ALGAL vision states the bet behind it.

      \n

      Install

      \n

      Bun 1.3.14 or newer and Git are required.\nThe CLI and TypeScript SDK run with Bun. Install the versioned GitHub archive:

      \n
      bun add --global --ignore-scripts https://github.com/hraness/wordcell/releases/download/v0.22.5/hraness-wordcell-0.22.5.tgz\nwordcell --help\n
      \n

      Prefer npm? The same release is mirrored there:

      \n
      npm install --global --ignore-scripts @hraness/wordcell@0.22.5\nwordcell --help\n
      \n

      Keep Bun in PATH for either installation. If your shell cannot find wordcell,\nadd your package manager's global executable directory to PATH and reopen the\nterminal. Optional semantic search, browser capture, and PDF tools have\nadditional prerequisites.

      \n

      Keep one decision available to the next session

      \n

      Run this from a directory where you want a new kb/ folder. It creates one\nMarkdown note and finds it without downloading a model or contacting a service:

      \n
      wordcell init kb\nwordcell note create notes/parser-contract \\\n  --title "Parser contract" --type concept --tag architecture \\\n  --body "Parser retries stop after three attempts." --root kb\nwordcell search "parser retries" --root kb --mode exact\n
      \n

      The result includes notes/parser-contract and the saved retry constraint.\nOpen kb/notes/parser-contract.md to see the ordinary Markdown file. Wordcell\nadds a stable document_id in its frontmatter; you can edit the prose normally.

      \n

      Already have Markdown or an Obsidian vault? In v0.22.0 or newer, search\nthe existing folder without initialization or an index.md file:

      \n
      wordcell search "a phrase from your notes" --root /path/to/your/vault --mode exact\n
      \n

      Replace the path and phrase with your own. You don't need to initialize,\nconvert, or move the existing files to search them.

      \n

      Use with a coding agent

      \n

      After trying the CLI, install the public Agent Skill into a compatible agent,\nsuch as Claude Code, Codex, Cursor, or GitHub Copilot:

      \n
      bunx skills add hraness/wordcell#v0.22.5 --skill wordcell\n
      \n

      Then ask:

      \n
      Use Wordcell to search for "parser retries" in ./kb using exact mode.\nRead the matching note and explain the saved constraint.\n
      \n

      The skill installs instructions, not a service. Installation does not create a\nvault, modify your notes, or grant an agent permission to access other accounts.\nThe agent still follows its own provider and data-handling settings.\nInspect the skill.

      \n

      Recover the stopped session

      \n

      Link a plan to the decision you saved:

      \n
      wordcell note create plans/parser-v2 \\\n  --title "Parser v2" --type plan \\\n  --body "The plan implements [[notes/parser-contract|the parser contract]]." \\\n  --root kb\nwordcell backlinks notes/parser-contract --root kb\n
      \n

      The backlink result includes plans/parser-v2, so a later session can find the\nwork that depends on the constraint.

      \n

      For code-path lookup, add this field inside the existing frontmatter of\nkb/notes/parser-contract.md:

      \n
      repository_scopes:\n  - packages/parser\n
      \n

      From your repository root, use an actual path under that scope and inspect the\nreturned notes and inherited AGENTS.md guides:

      \n
      wordcell context packages/parser/src/index.ts --root kb --repo .\n
      \n

      Commit the vault with your repository to preserve its history, then use\nwordcell history notes/parser-contract --root kb --repo . to inspect the\ncommits behind the note. History is optional and requires recorded Git commits.

      \n

      These views recover saved decisions, related work, rules, and provenance.\nThey do not reconstruct private chat or prove that the note is still correct.\nOpen the returned Markdown and guides before acting on them.

      \n

      Rerank a search window

      \n
      wordcell search "why releases use immutable archives" --root kb --mode exact \\\n  --rerank typesafe --rerank-limit 25 --limit 5 --json\n
      \n

      This optional hosted lane uses TypeSafe's pinned jev-1.13.0 model. It sends\nbounded query and note snippets to the provider, needs a private local\ncredential, and incurs provider charges. Exact identities remain first; a\nprovider failure retains the baseline order with a diagnostic. See the\nsetup, SDK examples, measured results, and limits.

      \n

      What you can do

      \n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n
      TaskCommandEvidence and effects
      Find a saved decisionwordcell search "parser retries" --root kb --mode exactReads current Markdown; no model or network request.
      Recover context for codewordcell context packages/parser/src/index.ts --root kb --repo .Returns scoped notes, plans, and inherited AGENTS.md rules.
      Inspect explicit connectionswordcell backlinks notes/parser-contract --root kbReturns notes that link to the decision.
      Search by meaningwordcell search "retry policy" --root kb --mode hybridAdds optional local QMD keyword and vector retrieval; model setup is required.
      Query with graph proofswordcell graph query --program scope-route --scope packages/parser --root kbReturns bounded results tied to the source revision. Queries do not write a cache.
      Capture a sourcewordcell clip https://example.com/article --output kb/articlesReads the selected URL and writes a Markdown bundle with a capture receipt.
      Capture a PDFwordcell pdf /absolute/path/to/document.pdf --output kb/articlesPreserves the original PDF and extracted evidence; Poppler is required.
      Check the vaultwordcell check --root kbReports structural and attachment problems without editing files.
      Publish selected noteswordcell publish --root kb --out site/ --include notes/parser-contract --dry-run --jsonPreviews a static site selection locally; remove --dry-run to build it.
      Preview a sitewordcell serve --root site --port 8080Serves a published site on a loopback static file server with the emitted 404.html fallback.
      Connect an MCP clientwordcell mcp --root kbServes this vault to a local MCP client over standard input and output; --read-only removes the write tools. Available from source until the next release.
      Import from Supermemorywordcell import supermemory documents.json --root kbTurns saved Supermemory API responses into Markdown notes and reports local edits as conflicts on later imports. Available from source until the next release.
      \n

      Use --json for structured output and wordcell --help for the complete command\nsurface. Full command reference.

      \n

      Publish a selected part of your vault

      \n

      Preview the decision and plan from the example before writing an output folder:

      \n
      wordcell publish --root kb --out site \\\n  --include notes/parser-contract --include plans/parser-v2 --dry-run --json\n
      \n

      Review the selection, then build and preview it:

      \n
      wordcell publish --root kb --out site \\\n  --include notes/parser-contract --include plans/parser-v2\nwordcell serve --root site --port 8080\n
      \n

      Open http://127.0.0.1:8080. You get readable pages, linked notes, and search\nthat runs in the browser. Upload the site/ folder to your chosen static host\nwhen you want to share it; publish itself never uploads anything.

      \n

      For a repeatable slice, use path patterns or combine folders, tags, metadata,\ncode scopes, and linked neighborhoods:

      \n
      wordcell publish --root kb --out site-notes \\\n  --include-glob 'notes/**/*.md' --exclude-glob '**/draft-*' --dry-run --json\n
      \n

      Preview reports show up to 20 selected IDs by default, with a total count and\nselection digest. They keep note bodies out of the agent's context. Use\n--list-limit to adjust that preview without changing what gets published. publish: false excludes a note, but selected\nprose and attachments still need review before sharing: selection is not secret\nredaction. Selection recipes and hosting guide.

      \n

      Use a vault as agent memory

      \n

      The MCP server, the Supermemory importer, and the skill's session-memory\nworkflow described here are available from source until the next release. To\nuse them before that release, install the CLI from a checkout as the\ninstallation reference\nshows, and add the skill from the main branch with\nbunx skills add hraness/wordcell --skill wordcell.

      \n

      wordcell mcp --root kb serves a vault to a local MCP client, such as Claude\nCode, Claude Desktop, Cursor, or Codex, over standard input and output. The\nclient can search, list, and read notes, follow links, create notes, update a\nnote body at a known revision, and add typed relations. --read-only removes\nthe write tools. Connect a client\nshows the setup for each host.

      \n

      With the Agent Skill, ask your agent to save the session. It writes a dated\nsession note, links it to the notes it changed, and keeps a profile note for\nyou or your project. Wordcell extracts no facts and writes no note on its own.\nSession memory and profiles.

      \n

      To move from Supermemory, wordcell import supermemory turns saved Supermemory\nAPI responses into Markdown notes. The migration guide\ncovers export, import, and what does not transfer, and the\nSupermemory comparison lists when to\nchoose Supermemory instead. Sync with Git\nkeeps one vault current on several machines.

      \n

      Evidence and comparisons

      \n

      In a four-query example over a seven-note public vault, packed search snippets\nused 80% fewer UTF-8 bytes than passing the same matching notes in full:\n12,126 versus 60,584 bytes. This measures context payload size, not tokenizer\ncounts, answer quality, latency, or a win over another search tool.

      \n

      In a separate public SciFact study, optional hosted Jev reranking placed a\njudged relevant result first for 161 of 300 queries, versus 101 with\nWordcell exact search alone. It sends bounded context to a paid provider;\nthis is evidence on scientific abstracts, not a comparison with QMD or a\nguarantee for repository notes. Results and limits.

      \n

      Wordcell's benefit is selecting relevant context and keeping its sources\ninspectable. Local ownership is also available in other tools, and Wordcell\ndoes not claim to beat QMD's retrieval quality or every Markdown workflow.

      \n

      Measured evidence\nshows a reproducible public-vault example, with the inputs, output sizes, and\nlimits beside each result. The comparison guide\ncovers Markdown alone, QMD, Basic Memory, Obsidian, static publishing tools,\nSupermemory, Mem0, and Zep using their own documentation. Choose the smallest\nworkflow that meets your needs.

      \n

      The benchmarks page shows the payload and\nSciFact results beside the published results of Oh, the embedded memory\nframework, each with its source data and limits. On all 500 LongMemEval-S\nquestions, Oh semantic retrieval scored 88.87% and BM25 86.13% with the same\nreader and budget; on the measure Oh named before the run, its interval does\nnot rule out a tie. Oh's scores measure its own memory-retrieval path, not a\nWordcell vault.

      \n

      How the files fit together

      \n
      repository/\n├── AGENTS.md                # rules that govern edits\n├── packages/parser/\n│   └── AGENTS.md            # rules scoped to this code\n└── kb/\n    ├── index.md             # authored or managed front door\n    ├── articles/            # captured sources and assets\n    ├── notes/               # maintained explanations\n    └── plans/               # decisions and outcomes\n
      \n

      Markdown, YAML frontmatter, explicit wikilinks, and Git hold the record. Open\nthe same files in Obsidian, a text editor, or ordinary file-search tools.\nApplication code does not need to import Wordcell or its vault.

      \n

      Wordcell is the Markdown knowledge base. Oh is the\nembedded memory framework that backs its named graph queries and source proofs.\nMarkdown and Git remain authoritative. Query the graph immediately without an\nOh account, service, or persisted database:

      \n
      wordcell graph query --program backlinks --note notes/parser-contract --root kb --json\n
      \n

      The result traces each returned link to its source note and revision. Only an explicit\nwordcell graph rebuild --root kb writes .wordcell/oh.sqlite, the file stays\nignored and rebuildable, and nothing flows from the projection back into notes.\nBacklinks and typed relationships come from authored links. Percolation\nsuggests connections for review and does not add inferred edges to notes.

      \n

      Wordcell search combines its own exact matching with optional QMD local search\nand optional hosted Jev reranking. Oh also offers memory retrieval for applications;\nits conversation-memory benchmark scores measure that separate path. They do not\nestablish Wordcell's retrieval or answer quality. How the integration works.

      \n

      Graph proofs explain a supported derivation from a specific source revision.\nThey do not prove that a note is true or that a missing relationship cannot\nexist. Graph queries and proof limits.

      \n

      Build with the TypeScript SDK

      \n

      Add the same immutable release to a Bun project:

      \n
      bun add --exact --ignore-scripts https://github.com/hraness/wordcell/releases/download/v0.22.5/hraness-wordcell-0.22.5.tgz\n
      \n

      The SDK provides read-only vault sessions, metadata queries, search, graph\nproofs, Git context, and composable workflows. A session owns one snapshot;\nreopen it after Markdown changes. SDK and workflow examples\nshow the public imports and lifecycle.

      \n

      Privacy and boundaries

      \n
        \n
      • Structural queries and exact search read local files. Optional semantic\nsearch downloads its model on first use and runs locally. Optional Jev\nreranking sends the query and bounded candidate context to a remote provider;\nit is off by default.
      • \n
      • URL capture contacts the requested source. Signed-in capture uses only\nexplicitly selected browser state. Review the security policy\nbefore using it with private sources.
      • \n
      • An agent that reads the vault follows its own provider and data-handling\nsettings. Keep private records out of public repositories and outputs.
      • \n
      • Git history is opt-in. Saved notes preserve recorded context; Wordcell does\nnot reconstruct unsaved conversations or silently record every agent action.
      • \n
      \n

      Documentation

      \n

      The documentation follows the Diataxis split: a tutorial to learn the loop,\nhow-to guides for tasks, reference for exact interfaces, and explanation for\nthe design. Browse it on the documentation index.

      \n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n
      Read nextPurpose
      Get startedLearn the full loop on a first vault: save, find, connect, and publish one note.
      Agent workflowSet up, query, maintain, and revise repository memory.
      Local MCP serverConnect Claude Code, Claude Desktop, Cursor, or Codex to a vault and review the write tools. Available from source until the next release.
      Installation and command referenceExact interfaces, SDK imports, optional adapters, and troubleshooting prerequisites.
      Web capture and PDF captureSave sources with provenance, assets, and explicit completeness limits.
      Publish selected notesPreview a slice, build a static site, and choose how to host it.
      Graph guideNamed queries, proofs, revisions, resource limits, and cache recovery.
      Portfolio federationSearch only selected, authorized vaults.
      Sync with GitKeep one vault current on several machines with a private repository.
      Migrate from SupermemoryExport documents and memory entries, import them as notes, and replace connectors. The importer is available from source until the next release.
      Design and memory rationaleFile contracts, design choices, and evaluation context.
      Release verificationVerify archive identity, signatures, and provenance.
      \n

      Installation reference

      \n

      Alternative installs and optional capabilities.

      \n

      The kb vault format

      \n

      Existing kb/, index.md, kb:// identifiers, and Markdown formats remain\ncompatible. Vault format reference.

      \n

      Command surface

      \n

      Complete CLI reference.

      \n

      Capture reference

      \n

      Capture commands and dependencies.

      \n

      Graph reference

      \n

      Graph and SDK interfaces.

      \n

      Agent skills

      \n

      Installation and runtime contract.\nThe same skill is included at node_modules/@hraness/wordcell/skills/wordcell/.

      \n

      Release notes

      \n

      Version 0.22 adds selective static publishing and local preview. Choose notes,\nfolders, path patterns, metadata, or linked neighborhoods, then inspect a\nbounded selection report before building. Existing vaults need no migration.\nThe deprecated kb command was removed in 0.21.0; use wordcell.\nAll release notes and upgrade instructions.

      \n

      Contributing

      \n

      Run bun run check for source, architecture, package installation, types, and\nnative-helper validation. The site has its own bun run check under site/.\nSee Contributing, Security, and the MIT License.

      \n

      Install Wordcell from GitHub Releases ·\nWordcell source on GitHub ·\nWordcell overview

      \n

      Optional development support

      \n

      wordcell support shows optional development support for Wordcell. Features stay\nfree and no account is required. Accounts presents current terms; the person\nconfirms any payment in their browser. Wordcell has no product newsletter and\ndoes not look up an email address for support.

      \n

      After useful standalone work, agents can receive a discovery record on stderr.\nRead wordcell support protocol --json for the shared closeout procedure.\nHelp, diagnostics, evaluation, SDK calls and nested commands stay quiet.\nHRANESS_SUPPORT=off or HRANESS_SUPPORT_AUDIENCE=off suppresses incidental\ninvitations. Explicit human terminal rendering requires\nHRANESS_SUPPORT_AUDIENCE=human; the default audience is an agent, including in\na pseudo-terminal.

      \n

      wordcell support dismiss disables invitations across participating tools on\nthis machine. snooze pauses them for thirty days, enable restores them, and\nstatus --json shows their separate local preferences. These commands do not\nchange a vault, sign up or pay. Acknowledged invitations share a seven-day\ncooldown; discovery itself does not consume it.

      \n", }; diff --git a/site/app/readme.generated.ts b/site/app/readme.generated.ts index ada23b1..d047408 100644 --- a/site/app/readme.generated.ts +++ b/site/app/readme.generated.ts @@ -2,4 +2,4 @@ export const readmeVersion = "0.22.5"; export const readmeTitle = "Wordcell"; export const readmeLead = "Wordcell keeps decisions, plans, and sources as Markdown files beside your code. Coding agents find them by exact words, by meaning with an optional local model, or from the file they are about to change."; -export const readmeHtml = "

      Wordcell

      \n

      Install the Agent Skill

      \n

      Wordcell keeps decisions, plans, and sources as Markdown files beside your\ncode. Coding agents find them by exact words, by meaning with an optional\nlocal model, or from the file they are about to change.

      \n

      A new coding-agent session can read your code, but not the decisions that\nstayed in the last session's chat. Wordcell keeps those decisions as Markdown\nfiles beside the repository, with the plans that depend on them and the web\npages and PDFs that informed them. Tie a note to the paths it explains, and an\nagent about to change that code runs one command to get the notes and plans\nfor that path. Exact search, backlinks, and Git history run on your machine\nwith no account or model, and every index rebuilds from files you can read in\nany editor. Wordcell is free and open source.

      \n

      Web capture, optional hosted reranking, and your agent's provider reach other\nservices; Privacy and boundaries says what each one\nsends.

      \n

      Documentation · Comparisons · Measured evidence · Changelog

      \n

      Why Wordcell

      \n
        \n
      • Keep what you learn in files you own. Write decisions, sources, and plans\nin Markdown. Obsidian, Git, and Wordcell read the same record, and every\nindex rebuilds from the files.
      • \n
      • Find it by words or by meaning. Exact search needs no model or account.\nOptional local semantic search joins each match to current metadata, links,\nand history instead of returning isolated text.
      • \n
      • Author the connections. Wikilinks and typed relationships turn notes into\na graph you can query. Backlinks and graph queries use only what you wrote,\nand wordcell percolate suggests missing links for you to review without\nwriting them into your notes.
      • \n
      • Give coding agents the reasons behind the code. Tie notes to repository\npaths, list the commits behind a note, and let the next session start from\nthe notes for the file it is changing. Only context you save becomes part of\nthe record.
      • \n
      \n

      Plain Markdown may be enough for a small set of notes. QMD is a good fit for\nlocal document retrieval and also supplies Wordcell's optional semantic search.\nWordcell adds a connected workflow for repository context, authored relationships,\nGit evidence, and selective publishing. Compare the tradeoffs.

      \n

      Wordcell keeps the record in Markdown files you own, rebuilds every index from those files, and lets the next session start from the notes for the file it is changing: the design every Hraness project shares. The thread through hraness follows that design across the projects, and the ALGAL vision states the bet behind it.

      \n

      Install

      \n

      Bun 1.3.14 or newer and Git are required.\nThe CLI and TypeScript SDK run with Bun. Install the versioned GitHub archive:

      \n
      bun add --global --ignore-scripts https://github.com/hraness/wordcell/releases/download/v0.22.5/hraness-wordcell-0.22.5.tgz\nwordcell --help\n
      \n

      Prefer npm? The same release is mirrored there:

      \n
      npm install --global --ignore-scripts @hraness/wordcell@0.22.5\nwordcell --help\n
      \n

      Keep Bun in PATH for either installation. If your shell cannot find wordcell,\nadd your package manager's global executable directory to PATH and reopen the\nterminal. Optional semantic search, browser capture, and PDF tools have\nadditional prerequisites.

      \n

      Keep one decision available to the next session

      \n

      Run this from a directory where you want a new kb/ folder. It creates one\nMarkdown note and finds it without downloading a model or contacting a service:

      \n
      wordcell init kb\nwordcell note create notes/parser-contract \\\n  --title "Parser contract" --type concept --tag architecture \\\n  --body "Parser retries stop after three attempts." --root kb\nwordcell search "parser retries" --root kb --mode exact\n
      \n

      The result includes notes/parser-contract and the saved retry constraint.\nOpen kb/notes/parser-contract.md to see the ordinary Markdown file. Wordcell\nadds a stable document_id in its frontmatter; you can edit the prose normally.

      \n

      Already have Markdown or an Obsidian vault? In v0.22.0 or newer, search\nthe existing folder without initialization or an index.md file:

      \n
      wordcell search "a phrase from your notes" --root /path/to/your/vault --mode exact\n
      \n

      Replace the path and phrase with your own. You don't need to initialize,\nconvert, or move the existing files to search them.

      \n

      Use with a coding agent

      \n

      After trying the CLI, install the public Agent Skill into a compatible agent,\nsuch as Claude Code, Codex, Cursor, or GitHub Copilot:

      \n
      bunx skills add hraness/wordcell#v0.22.5 --skill wordcell\n
      \n

      Then ask:

      \n
      Use Wordcell to search for "parser retries" in ./kb using exact mode.\nRead the matching note and explain the saved constraint.\n
      \n

      The skill installs instructions, not a service. Installation does not create a\nvault, modify your notes, or grant an agent permission to access other accounts.\nThe agent still follows its own provider and data-handling settings.\nInspect the skill.

      \n

      Recover the stopped session

      \n

      Link a plan to the decision you saved:

      \n
      wordcell note create plans/parser-v2 \\\n  --title "Parser v2" --type plan \\\n  --body "The plan implements [[notes/parser-contract|the parser contract]]." \\\n  --root kb\nwordcell backlinks notes/parser-contract --root kb\n
      \n

      The backlink result includes plans/parser-v2, so a later session can find the\nwork that depends on the constraint.

      \n

      For code-path lookup, add this field inside the existing frontmatter of\nkb/notes/parser-contract.md:

      \n
      repository_scopes:\n  - packages/parser\n
      \n

      From your repository root, use an actual path under that scope and inspect the\nreturned notes and inherited AGENTS.md guides:

      \n
      wordcell context packages/parser/src/index.ts --root kb --repo .\n
      \n

      Commit the vault with your repository to preserve its history, then use\nwordcell history notes/parser-contract --root kb --repo . to inspect the\ncommits behind the note. History is optional and requires recorded Git commits.

      \n

      These views recover saved decisions, related work, rules, and provenance.\nThey do not reconstruct private chat or prove that the note is still correct.\nOpen the returned Markdown and guides before acting on them.

      \n

      Rerank a search window

      \n
      wordcell search "why releases use immutable archives" --root kb --mode exact \\\n  --rerank typesafe --rerank-limit 25 --limit 5 --json\n
      \n

      This optional hosted lane uses TypeSafe's pinned jev-1.13.0 model. It sends\nbounded query and note snippets to the provider, needs a private local\ncredential, and incurs provider charges. Exact identities remain first; a\nprovider failure retains the baseline order with a diagnostic. See the\nsetup, SDK examples, measured results, and limits.

      \n

      What you can do

      \n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n
      TaskCommandEvidence and effects
      Find a saved decisionwordcell search "parser retries" --root kb --mode exactReads current Markdown; no model or network request.
      Recover context for codewordcell context packages/parser/src/index.ts --root kb --repo .Returns scoped notes, plans, and inherited AGENTS.md rules.
      Inspect explicit connectionswordcell backlinks notes/parser-contract --root kbReturns notes that link to the decision.
      Search by meaningwordcell search "retry policy" --root kb --mode hybridAdds optional local QMD keyword and vector retrieval; model setup is required.
      Query with graph proofswordcell graph query --program scope-route --scope packages/parser --root kbReturns bounded results tied to the source revision. Queries do not write a cache.
      Capture a sourcewordcell clip https://example.com/article --output kb/articlesReads the selected URL and writes a Markdown bundle with a capture receipt.
      Capture a PDFwordcell pdf /absolute/path/to/document.pdf --output kb/articlesPreserves the original PDF and extracted evidence; Poppler is required.
      Check the vaultwordcell check --root kbReports structural and attachment problems without editing files.
      Publish selected noteswordcell publish --root kb --out site/ --include notes/parser-contract --dry-run --jsonPreviews a static site selection locally; remove --dry-run to build it.
      Preview a sitewordcell serve --root site --port 8080Serves a published site on a loopback static file server with the emitted 404.html fallback.
      Connect an MCP clientwordcell mcp --root kbServes this vault to a local MCP client over standard input and output; --read-only removes the write tools. Available from source until the next release.
      Import from Supermemorywordcell import supermemory documents.json --root kbTurns saved Supermemory API responses into Markdown notes and reports local edits as conflicts on later imports. Available from source until the next release.
      \n

      Use --json for structured output and wordcell --help for the complete command\nsurface. Full command reference.

      \n

      Publish a selected part of your vault

      \n

      Preview the decision and plan from the example before writing an output folder:

      \n
      wordcell publish --root kb --out site \\\n  --include notes/parser-contract --include plans/parser-v2 --dry-run --json\n
      \n

      Review the selection, then build and preview it:

      \n
      wordcell publish --root kb --out site \\\n  --include notes/parser-contract --include plans/parser-v2\nwordcell serve --root site --port 8080\n
      \n

      Open http://127.0.0.1:8080. You get readable pages, linked notes, and search\nthat runs in the browser. Upload the site/ folder to your chosen static host\nwhen you want to share it; publish itself never uploads anything.

      \n

      For a repeatable slice, use path patterns or combine folders, tags, metadata,\ncode scopes, and linked neighborhoods:

      \n
      wordcell publish --root kb --out site-notes \\\n  --include-glob 'notes/**/*.md' --exclude-glob '**/draft-*' --dry-run --json\n
      \n

      Preview reports show up to 20 selected IDs by default, with a total count and\nselection digest. They keep note bodies out of the agent's context. Use\n--list-limit to adjust that preview without changing what gets published. publish: false excludes a note, but selected\nprose and attachments still need review before sharing: selection is not secret\nredaction. Selection recipes and hosting guide.

      \n

      Use a vault as agent memory

      \n

      The MCP server, the Supermemory importer, and the skill's session-memory\nworkflow described here are available from source until the next release. To\nuse them before that release, install the CLI from a checkout as the\ninstallation reference\nshows, and add the skill from the main branch with\nbunx skills add hraness/wordcell --skill wordcell.

      \n

      wordcell mcp --root kb serves a vault to a local MCP client, such as Claude\nCode, Claude Desktop, Cursor, or Codex, over standard input and output. The\nclient can search, list, and read notes, follow links, create notes, update a\nnote body at a known revision, and add typed relations. --read-only removes\nthe write tools. Connect a client\nshows the setup for each host.

      \n

      With the Agent Skill, ask your agent to save the session. It writes a dated\nsession note, links it to the notes it changed, and keeps a profile note for\nyou or your project. Wordcell extracts no facts and writes no note on its own.\nSession memory and profiles.

      \n

      To move from Supermemory, wordcell import supermemory turns saved Supermemory\nAPI responses into Markdown notes. The migration guide\ncovers export, import, and what does not transfer, and the\nSupermemory comparison lists when to\nchoose Supermemory instead. Sync with Git\nkeeps one vault current on several machines.

      \n

      Evidence and comparisons

      \n

      In a four-query example over a seven-note public vault, packed search snippets\nused 80% fewer UTF-8 bytes than passing the same matching notes in full:\n12,126 versus 60,584 bytes. This measures context payload size, not tokenizer\ncounts, answer quality, latency, or a win over another search tool.

      \n

      In a separate public SciFact study, optional hosted Jev reranking placed a\njudged relevant result first for 161 of 300 queries, versus 101 with\nWordcell exact search alone. It sends bounded context to a paid provider;\nthis is evidence on scientific abstracts, not a comparison with QMD or a\nguarantee for repository notes. Results and limits.

      \n

      Wordcell's benefit is selecting relevant context and keeping its sources\ninspectable. Local ownership is also available in other tools, and Wordcell\ndoes not claim to beat QMD's retrieval quality or every Markdown workflow.

      \n

      Measured evidence\nshows a reproducible public-vault example, with the inputs, output sizes, and\nlimits beside each result. The comparison guide\ncovers Markdown alone, QMD, Basic Memory, Obsidian, static publishing tools,\nSupermemory, Mem0, and Zep using their own documentation. Choose the smallest\nworkflow that meets your needs.

      \n

      The benchmarks page shows the payload and\nSciFact results beside the published results of Oh, the embedded memory\nframework, each with its source data and limits. Oh's scores measure its own\nmemory-retrieval path, not a Wordcell vault.

      \n

      How the files fit together

      \n
      repository/\n├── AGENTS.md                # rules that govern edits\n├── packages/parser/\n│   └── AGENTS.md            # rules scoped to this code\n└── kb/\n    ├── index.md             # authored or managed front door\n    ├── articles/            # captured sources and assets\n    ├── notes/               # maintained explanations\n    └── plans/               # decisions and outcomes\n
      \n

      Markdown, YAML frontmatter, explicit wikilinks, and Git hold the record. Open\nthe same files in Obsidian, a text editor, or ordinary file-search tools.\nApplication code does not need to import Wordcell or its vault.

      \n

      Wordcell is the Markdown knowledge base. Oh is the\nembedded memory framework that backs its named graph queries and source proofs.\nMarkdown and Git remain authoritative. Query the graph immediately without an\nOh account, service, or persisted database:

      \n
      wordcell graph query --program backlinks --note notes/parser-contract --root kb --json\n
      \n

      The result traces each returned link to its source note and revision. Only an explicit\nwordcell graph rebuild --root kb writes .wordcell/oh.sqlite, the file stays\nignored and rebuildable, and nothing flows from the projection back into notes.\nBacklinks and typed relationships come from authored links. Percolation\nsuggests connections for review and does not add inferred edges to notes.

      \n

      Wordcell search combines its own exact matching with optional QMD local search\nand optional hosted Jev reranking. Oh also offers memory retrieval for applications;\nits conversation-memory benchmark scores measure that separate path. They do not\nestablish Wordcell's retrieval or answer quality. How the integration works.

      \n

      Graph proofs explain a supported derivation from a specific source revision.\nThey do not prove that a note is true or that a missing relationship cannot\nexist. Graph queries and proof limits.

      \n

      Build with the TypeScript SDK

      \n

      Add the same immutable release to a Bun project:

      \n
      bun add --exact --ignore-scripts https://github.com/hraness/wordcell/releases/download/v0.22.5/hraness-wordcell-0.22.5.tgz\n
      \n

      The SDK provides read-only vault sessions, metadata queries, search, graph\nproofs, Git context, and composable workflows. A session owns one snapshot;\nreopen it after Markdown changes. SDK and workflow examples\nshow the public imports and lifecycle.

      \n

      Privacy and boundaries

      \n
        \n
      • Structural queries and exact search read local files. Optional semantic\nsearch downloads its model on first use and runs locally. Optional Jev\nreranking sends the query and bounded candidate context to a remote provider;\nit is off by default.
      • \n
      • URL capture contacts the requested source. Signed-in capture uses only\nexplicitly selected browser state. Review the security policy\nbefore using it with private sources.
      • \n
      • An agent that reads the vault follows its own provider and data-handling\nsettings. Keep private records out of public repositories and outputs.
      • \n
      • Git history is opt-in. Saved notes preserve recorded context; Wordcell does\nnot reconstruct unsaved conversations or silently record every agent action.
      • \n
      \n

      Documentation

      \n

      The documentation follows the Diataxis split: a tutorial to learn the loop,\nhow-to guides for tasks, reference for exact interfaces, and explanation for\nthe design. Browse it on the documentation index.

      \n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n
      Read nextPurpose
      Get startedLearn the full loop on a first vault: save, find, connect, and publish one note.
      Agent workflowSet up, query, maintain, and revise repository memory.
      Local MCP serverConnect Claude Code, Claude Desktop, Cursor, or Codex to a vault and review the write tools. Available from source until the next release.
      Installation and command referenceExact interfaces, SDK imports, optional adapters, and troubleshooting prerequisites.
      Web capture and PDF captureSave sources with provenance, assets, and explicit completeness limits.
      Publish selected notesPreview a slice, build a static site, and choose how to host it.
      Graph guideNamed queries, proofs, revisions, resource limits, and cache recovery.
      Portfolio federationSearch only selected, authorized vaults.
      Sync with GitKeep one vault current on several machines with a private repository.
      Migrate from SupermemoryExport documents and memory entries, import them as notes, and replace connectors. The importer is available from source until the next release.
      Design and memory rationaleFile contracts, design choices, and evaluation context.
      Release verificationVerify archive identity, signatures, and provenance.
      \n

      Installation reference

      \n

      Alternative installs and optional capabilities.

      \n

      The kb vault format

      \n

      Existing kb/, index.md, kb:// identifiers, and Markdown formats remain\ncompatible. Vault format reference.

      \n

      Command surface

      \n

      Complete CLI reference.

      \n

      Capture reference

      \n

      Capture commands and dependencies.

      \n

      Graph reference

      \n

      Graph and SDK interfaces.

      \n

      Agent skills

      \n

      Installation and runtime contract.\nThe same skill is included at node_modules/@hraness/wordcell/skills/wordcell/.

      \n

      Release notes

      \n

      Version 0.22 adds selective static publishing and local preview. Choose notes,\nfolders, path patterns, metadata, or linked neighborhoods, then inspect a\nbounded selection report before building. Existing vaults need no migration.\nThe deprecated kb command was removed in 0.21.0; use wordcell.\nAll release notes and upgrade instructions.

      \n

      Contributing

      \n

      Run bun run check for source, architecture, package installation, types, and\nnative-helper validation. The site has its own bun run check under site/.\nSee Contributing, Security, and the MIT License.

      \n

      Install Wordcell from GitHub Releases ·\nWordcell source on GitHub ·\nWordcell overview

      \n

      Optional development support

      \n

      wordcell support shows optional development support for Wordcell. Features stay\nfree and no account is required. Accounts presents current terms; the person\nconfirms any payment in their browser. Wordcell has no product newsletter and\ndoes not look up an email address for support.

      \n

      After useful standalone work, agents can receive a discovery record on stderr.\nRead wordcell support protocol --json for the shared closeout procedure.\nHelp, diagnostics, evaluation, SDK calls and nested commands stay quiet.\nHRANESS_SUPPORT=off or HRANESS_SUPPORT_AUDIENCE=off suppresses incidental\ninvitations. Explicit human terminal rendering requires\nHRANESS_SUPPORT_AUDIENCE=human; the default audience is an agent, including in\na pseudo-terminal.

      \n

      wordcell support dismiss disables invitations across participating tools on\nthis machine. snooze pauses them for thirty days, enable restores them, and\nstatus --json shows their separate local preferences. These commands do not\nchange a vault, sign up or pay. Acknowledged invitations share a seven-day\ncooldown; discovery itself does not consume it.

      \n"; +export const readmeHtml = "

      Wordcell

      \n

      Install the Agent Skill

      \n

      Wordcell keeps decisions, plans, and sources as Markdown files beside your\ncode. Coding agents find them by exact words, by meaning with an optional\nlocal model, or from the file they are about to change.

      \n

      A new coding-agent session can read your code, but not the decisions that\nstayed in the last session's chat. Wordcell keeps those decisions as Markdown\nfiles beside the repository, with the plans that depend on them and the web\npages and PDFs that informed them. Tie a note to the paths it explains, and an\nagent about to change that code runs one command to get the notes and plans\nfor that path. Exact search, backlinks, and Git history run on your machine\nwith no account or model, and every index rebuilds from files you can read in\nany editor. Wordcell is free and open source.

      \n

      Web capture, optional hosted reranking, and your agent's provider reach other\nservices; Privacy and boundaries says what each one\nsends.

      \n

      Documentation · Comparisons · Measured evidence · Changelog

      \n

      Why Wordcell

      \n
        \n
      • Keep what you learn in files you own. Write decisions, sources, and plans\nin Markdown. Obsidian, Git, and Wordcell read the same record, and every\nindex rebuilds from the files.
      • \n
      • Find it by words or by meaning. Exact search needs no model or account.\nOptional local semantic search joins each match to current metadata, links,\nand history instead of returning isolated text.
      • \n
      • Author the connections. Wikilinks and typed relationships turn notes into\na graph you can query. Backlinks and graph queries use only what you wrote,\nand wordcell percolate suggests missing links for you to review without\nwriting them into your notes.
      • \n
      • Give coding agents the reasons behind the code. Tie notes to repository\npaths, list the commits behind a note, and let the next session start from\nthe notes for the file it is changing. Only context you save becomes part of\nthe record.
      • \n
      \n

      Plain Markdown may be enough for a small set of notes. QMD is a good fit for\nlocal document retrieval and also supplies Wordcell's optional semantic search.\nWordcell adds a connected workflow for repository context, authored relationships,\nGit evidence, and selective publishing. Compare the tradeoffs.

      \n

      Wordcell keeps the record in Markdown files you own, rebuilds every index from those files, and lets the next session start from the notes for the file it is changing: the design every Hraness project shares. The thread through hraness follows that design across the projects, and the ALGAL vision states the bet behind it.

      \n

      Install

      \n

      Bun 1.3.14 or newer and Git are required.\nThe CLI and TypeScript SDK run with Bun. Install the versioned GitHub archive:

      \n
      bun add --global --ignore-scripts https://github.com/hraness/wordcell/releases/download/v0.22.5/hraness-wordcell-0.22.5.tgz\nwordcell --help\n
      \n

      Prefer npm? The same release is mirrored there:

      \n
      npm install --global --ignore-scripts @hraness/wordcell@0.22.5\nwordcell --help\n
      \n

      Keep Bun in PATH for either installation. If your shell cannot find wordcell,\nadd your package manager's global executable directory to PATH and reopen the\nterminal. Optional semantic search, browser capture, and PDF tools have\nadditional prerequisites.

      \n

      Keep one decision available to the next session

      \n

      Run this from a directory where you want a new kb/ folder. It creates one\nMarkdown note and finds it without downloading a model or contacting a service:

      \n
      wordcell init kb\nwordcell note create notes/parser-contract \\\n  --title "Parser contract" --type concept --tag architecture \\\n  --body "Parser retries stop after three attempts." --root kb\nwordcell search "parser retries" --root kb --mode exact\n
      \n

      The result includes notes/parser-contract and the saved retry constraint.\nOpen kb/notes/parser-contract.md to see the ordinary Markdown file. Wordcell\nadds a stable document_id in its frontmatter; you can edit the prose normally.

      \n

      Already have Markdown or an Obsidian vault? In v0.22.0 or newer, search\nthe existing folder without initialization or an index.md file:

      \n
      wordcell search "a phrase from your notes" --root /path/to/your/vault --mode exact\n
      \n

      Replace the path and phrase with your own. You don't need to initialize,\nconvert, or move the existing files to search them.

      \n

      Use with a coding agent

      \n

      After trying the CLI, install the public Agent Skill into a compatible agent,\nsuch as Claude Code, Codex, Cursor, or GitHub Copilot:

      \n
      bunx skills add hraness/wordcell#v0.22.5 --skill wordcell\n
      \n

      Then ask:

      \n
      Use Wordcell to search for "parser retries" in ./kb using exact mode.\nRead the matching note and explain the saved constraint.\n
      \n

      The skill installs instructions, not a service. Installation does not create a\nvault, modify your notes, or grant an agent permission to access other accounts.\nThe agent still follows its own provider and data-handling settings.\nInspect the skill.

      \n

      Recover the stopped session

      \n

      Link a plan to the decision you saved:

      \n
      wordcell note create plans/parser-v2 \\\n  --title "Parser v2" --type plan \\\n  --body "The plan implements [[notes/parser-contract|the parser contract]]." \\\n  --root kb\nwordcell backlinks notes/parser-contract --root kb\n
      \n

      The backlink result includes plans/parser-v2, so a later session can find the\nwork that depends on the constraint.

      \n

      For code-path lookup, add this field inside the existing frontmatter of\nkb/notes/parser-contract.md:

      \n
      repository_scopes:\n  - packages/parser\n
      \n

      From your repository root, use an actual path under that scope and inspect the\nreturned notes and inherited AGENTS.md guides:

      \n
      wordcell context packages/parser/src/index.ts --root kb --repo .\n
      \n

      Commit the vault with your repository to preserve its history, then use\nwordcell history notes/parser-contract --root kb --repo . to inspect the\ncommits behind the note. History is optional and requires recorded Git commits.

      \n

      These views recover saved decisions, related work, rules, and provenance.\nThey do not reconstruct private chat or prove that the note is still correct.\nOpen the returned Markdown and guides before acting on them.

      \n

      Rerank a search window

      \n
      wordcell search "why releases use immutable archives" --root kb --mode exact \\\n  --rerank typesafe --rerank-limit 25 --limit 5 --json\n
      \n

      This optional hosted lane uses TypeSafe's pinned jev-1.13.0 model. It sends\nbounded query and note snippets to the provider, needs a private local\ncredential, and incurs provider charges. Exact identities remain first; a\nprovider failure retains the baseline order with a diagnostic. See the\nsetup, SDK examples, measured results, and limits.

      \n

      What you can do

      \n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n
      TaskCommandEvidence and effects
      Find a saved decisionwordcell search "parser retries" --root kb --mode exactReads current Markdown; no model or network request.
      Recover context for codewordcell context packages/parser/src/index.ts --root kb --repo .Returns scoped notes, plans, and inherited AGENTS.md rules.
      Inspect explicit connectionswordcell backlinks notes/parser-contract --root kbReturns notes that link to the decision.
      Search by meaningwordcell search "retry policy" --root kb --mode hybridAdds optional local QMD keyword and vector retrieval; model setup is required.
      Query with graph proofswordcell graph query --program scope-route --scope packages/parser --root kbReturns bounded results tied to the source revision. Queries do not write a cache.
      Capture a sourcewordcell clip https://example.com/article --output kb/articlesReads the selected URL and writes a Markdown bundle with a capture receipt.
      Capture a PDFwordcell pdf /absolute/path/to/document.pdf --output kb/articlesPreserves the original PDF and extracted evidence; Poppler is required.
      Check the vaultwordcell check --root kbReports structural and attachment problems without editing files.
      Publish selected noteswordcell publish --root kb --out site/ --include notes/parser-contract --dry-run --jsonPreviews a static site selection locally; remove --dry-run to build it.
      Preview a sitewordcell serve --root site --port 8080Serves a published site on a loopback static file server with the emitted 404.html fallback.
      Connect an MCP clientwordcell mcp --root kbServes this vault to a local MCP client over standard input and output; --read-only removes the write tools. Available from source until the next release.
      Import from Supermemorywordcell import supermemory documents.json --root kbTurns saved Supermemory API responses into Markdown notes and reports local edits as conflicts on later imports. Available from source until the next release.
      \n

      Use --json for structured output and wordcell --help for the complete command\nsurface. Full command reference.

      \n

      Publish a selected part of your vault

      \n

      Preview the decision and plan from the example before writing an output folder:

      \n
      wordcell publish --root kb --out site \\\n  --include notes/parser-contract --include plans/parser-v2 --dry-run --json\n
      \n

      Review the selection, then build and preview it:

      \n
      wordcell publish --root kb --out site \\\n  --include notes/parser-contract --include plans/parser-v2\nwordcell serve --root site --port 8080\n
      \n

      Open http://127.0.0.1:8080. You get readable pages, linked notes, and search\nthat runs in the browser. Upload the site/ folder to your chosen static host\nwhen you want to share it; publish itself never uploads anything.

      \n

      For a repeatable slice, use path patterns or combine folders, tags, metadata,\ncode scopes, and linked neighborhoods:

      \n
      wordcell publish --root kb --out site-notes \\\n  --include-glob 'notes/**/*.md' --exclude-glob '**/draft-*' --dry-run --json\n
      \n

      Preview reports show up to 20 selected IDs by default, with a total count and\nselection digest. They keep note bodies out of the agent's context. Use\n--list-limit to adjust that preview without changing what gets published. publish: false excludes a note, but selected\nprose and attachments still need review before sharing: selection is not secret\nredaction. Selection recipes and hosting guide.

      \n

      Use a vault as agent memory

      \n

      The MCP server, the Supermemory importer, and the skill's session-memory\nworkflow described here are available from source until the next release. To\nuse them before that release, install the CLI from a checkout as the\ninstallation reference\nshows, and add the skill from the main branch with\nbunx skills add hraness/wordcell --skill wordcell.

      \n

      wordcell mcp --root kb serves a vault to a local MCP client, such as Claude\nCode, Claude Desktop, Cursor, or Codex, over standard input and output. The\nclient can search, list, and read notes, follow links, create notes, update a\nnote body at a known revision, and add typed relations. --read-only removes\nthe write tools. Connect a client\nshows the setup for each host.

      \n

      With the Agent Skill, ask your agent to save the session. It writes a dated\nsession note, links it to the notes it changed, and keeps a profile note for\nyou or your project. Wordcell extracts no facts and writes no note on its own.\nSession memory and profiles.

      \n

      To move from Supermemory, wordcell import supermemory turns saved Supermemory\nAPI responses into Markdown notes. The migration guide\ncovers export, import, and what does not transfer, and the\nSupermemory comparison lists when to\nchoose Supermemory instead. Sync with Git\nkeeps one vault current on several machines.

      \n

      Evidence and comparisons

      \n

      In a four-query example over a seven-note public vault, packed search snippets\nused 80% fewer UTF-8 bytes than passing the same matching notes in full:\n12,126 versus 60,584 bytes. This measures context payload size, not tokenizer\ncounts, answer quality, latency, or a win over another search tool.

      \n

      In a separate public SciFact study, optional hosted Jev reranking placed a\njudged relevant result first for 161 of 300 queries, versus 101 with\nWordcell exact search alone. It sends bounded context to a paid provider;\nthis is evidence on scientific abstracts, not a comparison with QMD or a\nguarantee for repository notes. Results and limits.

      \n

      Wordcell's benefit is selecting relevant context and keeping its sources\ninspectable. Local ownership is also available in other tools, and Wordcell\ndoes not claim to beat QMD's retrieval quality or every Markdown workflow.

      \n

      Measured evidence\nshows a reproducible public-vault example, with the inputs, output sizes, and\nlimits beside each result. The comparison guide\ncovers Markdown alone, QMD, Basic Memory, Obsidian, static publishing tools,\nSupermemory, Mem0, and Zep using their own documentation. Choose the smallest\nworkflow that meets your needs.

      \n

      The benchmarks page shows the payload and\nSciFact results beside the published results of Oh, the embedded memory\nframework, each with its source data and limits. On all 500 LongMemEval-S\nquestions, Oh semantic retrieval scored 88.87% and BM25 86.13% with the same\nreader and budget; on the measure Oh named before the run, its interval does\nnot rule out a tie. Oh's scores measure its own memory-retrieval path, not a\nWordcell vault.

      \n

      How the files fit together

      \n
      repository/\n├── AGENTS.md                # rules that govern edits\n├── packages/parser/\n│   └── AGENTS.md            # rules scoped to this code\n└── kb/\n    ├── index.md             # authored or managed front door\n    ├── articles/            # captured sources and assets\n    ├── notes/               # maintained explanations\n    └── plans/               # decisions and outcomes\n
      \n

      Markdown, YAML frontmatter, explicit wikilinks, and Git hold the record. Open\nthe same files in Obsidian, a text editor, or ordinary file-search tools.\nApplication code does not need to import Wordcell or its vault.

      \n

      Wordcell is the Markdown knowledge base. Oh is the\nembedded memory framework that backs its named graph queries and source proofs.\nMarkdown and Git remain authoritative. Query the graph immediately without an\nOh account, service, or persisted database:

      \n
      wordcell graph query --program backlinks --note notes/parser-contract --root kb --json\n
      \n

      The result traces each returned link to its source note and revision. Only an explicit\nwordcell graph rebuild --root kb writes .wordcell/oh.sqlite, the file stays\nignored and rebuildable, and nothing flows from the projection back into notes.\nBacklinks and typed relationships come from authored links. Percolation\nsuggests connections for review and does not add inferred edges to notes.

      \n

      Wordcell search combines its own exact matching with optional QMD local search\nand optional hosted Jev reranking. Oh also offers memory retrieval for applications;\nits conversation-memory benchmark scores measure that separate path. They do not\nestablish Wordcell's retrieval or answer quality. How the integration works.

      \n

      Graph proofs explain a supported derivation from a specific source revision.\nThey do not prove that a note is true or that a missing relationship cannot\nexist. Graph queries and proof limits.

      \n

      Build with the TypeScript SDK

      \n

      Add the same immutable release to a Bun project:

      \n
      bun add --exact --ignore-scripts https://github.com/hraness/wordcell/releases/download/v0.22.5/hraness-wordcell-0.22.5.tgz\n
      \n

      The SDK provides read-only vault sessions, metadata queries, search, graph\nproofs, Git context, and composable workflows. A session owns one snapshot;\nreopen it after Markdown changes. SDK and workflow examples\nshow the public imports and lifecycle.

      \n

      Privacy and boundaries

      \n
        \n
      • Structural queries and exact search read local files. Optional semantic\nsearch downloads its model on first use and runs locally. Optional Jev\nreranking sends the query and bounded candidate context to a remote provider;\nit is off by default.
      • \n
      • URL capture contacts the requested source. Signed-in capture uses only\nexplicitly selected browser state. Review the security policy\nbefore using it with private sources.
      • \n
      • An agent that reads the vault follows its own provider and data-handling\nsettings. Keep private records out of public repositories and outputs.
      • \n
      • Git history is opt-in. Saved notes preserve recorded context; Wordcell does\nnot reconstruct unsaved conversations or silently record every agent action.
      • \n
      \n

      Documentation

      \n

      The documentation follows the Diataxis split: a tutorial to learn the loop,\nhow-to guides for tasks, reference for exact interfaces, and explanation for\nthe design. Browse it on the documentation index.

      \n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n
      Read nextPurpose
      Get startedLearn the full loop on a first vault: save, find, connect, and publish one note.
      Agent workflowSet up, query, maintain, and revise repository memory.
      Local MCP serverConnect Claude Code, Claude Desktop, Cursor, or Codex to a vault and review the write tools. Available from source until the next release.
      Installation and command referenceExact interfaces, SDK imports, optional adapters, and troubleshooting prerequisites.
      Web capture and PDF captureSave sources with provenance, assets, and explicit completeness limits.
      Publish selected notesPreview a slice, build a static site, and choose how to host it.
      Graph guideNamed queries, proofs, revisions, resource limits, and cache recovery.
      Portfolio federationSearch only selected, authorized vaults.
      Sync with GitKeep one vault current on several machines with a private repository.
      Migrate from SupermemoryExport documents and memory entries, import them as notes, and replace connectors. The importer is available from source until the next release.
      Design and memory rationaleFile contracts, design choices, and evaluation context.
      Release verificationVerify archive identity, signatures, and provenance.
      \n

      Installation reference

      \n

      Alternative installs and optional capabilities.

      \n

      The kb vault format

      \n

      Existing kb/, index.md, kb:// identifiers, and Markdown formats remain\ncompatible. Vault format reference.

      \n

      Command surface

      \n

      Complete CLI reference.

      \n

      Capture reference

      \n

      Capture commands and dependencies.

      \n

      Graph reference

      \n

      Graph and SDK interfaces.

      \n

      Agent skills

      \n

      Installation and runtime contract.\nThe same skill is included at node_modules/@hraness/wordcell/skills/wordcell/.

      \n

      Release notes

      \n

      Version 0.22 adds selective static publishing and local preview. Choose notes,\nfolders, path patterns, metadata, or linked neighborhoods, then inspect a\nbounded selection report before building. Existing vaults need no migration.\nThe deprecated kb command was removed in 0.21.0; use wordcell.\nAll release notes and upgrade instructions.

      \n

      Contributing

      \n

      Run bun run check for source, architecture, package installation, types, and\nnative-helper validation. The site has its own bun run check under site/.\nSee Contributing, Security, and the MIT License.

      \n

      Install Wordcell from GitHub Releases ·\nWordcell source on GitHub ·\nWordcell overview

      \n

      Optional development support

      \n

      wordcell support shows optional development support for Wordcell. Features stay\nfree and no account is required. Accounts presents current terms; the person\nconfirms any payment in their browser. Wordcell has no product newsletter and\ndoes not look up an email address for support.

      \n

      After useful standalone work, agents can receive a discovery record on stderr.\nRead wordcell support protocol --json for the shared closeout procedure.\nHelp, diagnostics, evaluation, SDK calls and nested commands stay quiet.\nHRANESS_SUPPORT=off or HRANESS_SUPPORT_AUDIENCE=off suppresses incidental\ninvitations. Explicit human terminal rendering requires\nHRANESS_SUPPORT_AUDIENCE=human; the default audience is an agent, including in\na pseudo-terminal.

      \n

      wordcell support dismiss disables invitations across participating tools on\nthis machine. snooze pauses them for thirty days, enable restores them, and\nstatus --json shows their separate local preferences. These commands do not\nchange a vault, sign up or pay. Acknowledged invitations share a seven-day\ncooldown; discovery itself does not consume it.

      \n"; diff --git a/site/content/blog/free-local-agent-memory.md b/site/content/blog/free-local-agent-memory.md index 12618ff..2251c97 100644 --- a/site/content/blog/free-local-agent-memory.md +++ b/site/content/blog/free-local-agent-memory.md @@ -24,7 +24,9 @@ Wordcell builds its graph with Oh, memory for agents that stores each fact with The run covered {{evidence.oh-locomo.conversations}} conversations once, with no confidence interval, and every question had been seen before: {{evidence.oh-locomo.exposed}} in earlier evaluations and {{evidence.oh-locomo.development}} during development. The result file records no run date, so the date above is when Oh published it. [Oh’s result file](https://github.com/hraness/wordcell/blob/d87d4ecdd0a0b1351bc2b0d0f3cdf8de30c047dc/docs/evaluations/oh/memory-evolution-locomo-sealed-1540-v1.json) says the figures “{{evidence.oh-locomo.limit}}”. -Oh also ran a small pilot of its API against Supermemory, dated {{evidence.oh-pilot.date}}. It used {{evidence.oh-pilot.questions}} questions from LongMemEval-S, a benchmark of questions about long chat histories, and Oh had seen those questions during development. Supermemory answered {{evidence.oh-pilot.supermemory}} of the {{evidence.oh-pilot.questions}} questions correctly, Oh {{evidence.oh-pilot.oh}}, and BM25 {{evidence.oh-pilot.bm25}}. Oh and BM25 each did not finish {{evidence.oh-pilot.incomplete}} of the questions, and those count as misses. GPT-4o wrote and judged the answers, called through a gateway name that is not pinned to one model version. Oh minus Supermemory came to {{evidence.oh-pilot.estimate}} percentage points, with a {{evidence.oh-pilot.level}} interval from {{evidence.oh-pilot.lower}} to {{evidence.oh-pilot.upper}}. The interval includes zero, and [Oh’s pilot report]({{evidence.oh-pilot.url}}) says “the paired primary comparison does not separate Oh from Supermemory”. Supermemory ran with one fixed profile and was indexed per session, while Oh and BM25 were indexed per turn, so the pilot says nothing about Supermemory’s defaults or best configuration. The [benchmarks page](/benchmarks#oh) sets out both studies, and its [comparison section](/benchmarks#comparisons) lists figures other memory systems publish. +LongMemEval-S is a benchmark of questions about long chat histories. [Oh’s study of all {{evidence.oh-longmemeval.questions}} of its questions]({{evidence.oh-longmemeval.url}}), completed {{evidence.oh-longmemeval.completed}}, gave Oh’s semantic retrieval and BM25 the same byte budget and the same reader, GPT-5 mini, which answered every question three times. A GPT-4o judge graded the answers with LongMemEval’s own prompts. Oh’s semantic retrieval had {{evidence.oh-longmemeval.semantic}} of answers judged correct and BM25 {{evidence.oh-longmemeval.bm25}}. On the measure Oh chose before the run, questions answered correctly in at least two of the three runs, Oh minus BM25 came to {{evidence.oh-longmemeval.estimate}} percentage points, with a {{evidence.oh-longmemeval.level}} interval from {{evidence.oh-longmemeval.lower}} to {{evidence.oh-longmemeval.upper}}. That interval reaches zero, so the study does not rule out a tie. Earlier Oh studies had scored all of these questions, and this one reported no Supermemory result. + +Before that study, Oh ran a smaller pilot of its API against Supermemory, dated {{evidence.oh-pilot.date}}. It used {{evidence.oh-pilot.questions}} questions from LongMemEval-S, and Oh had seen those questions during development. Supermemory answered {{evidence.oh-pilot.supermemory}} of the {{evidence.oh-pilot.questions}} questions correctly, Oh {{evidence.oh-pilot.oh}}, and BM25 {{evidence.oh-pilot.bm25}}. Oh and BM25 each did not finish {{evidence.oh-pilot.incomplete}} of the questions, and those count as misses. GPT-4o wrote and judged the answers, called through a gateway name that is not pinned to one model version. Oh minus Supermemory came to {{evidence.oh-pilot.estimate}} percentage points, with a {{evidence.oh-pilot.level}} interval from {{evidence.oh-pilot.lower}} to {{evidence.oh-pilot.upper}}. The interval includes zero, and [Oh’s pilot report]({{evidence.oh-pilot.url}}) says “the paired primary comparison does not separate Oh from Supermemory”. Supermemory ran with one fixed profile and was indexed per session, while Oh and BM25 were indexed per turn, so the pilot says nothing about Supermemory’s defaults or best configuration. The [benchmarks page](/benchmarks#oh) sets out all three studies, and its [comparison section](/benchmarks#comparisons) lists figures other memory systems publish. ## Why the agent writes the memory diff --git a/site/public/llms.txt b/site/public/llms.txt index 35fcbcc..d9b4e5f 100644 --- a/site/public/llms.txt +++ b/site/public/llms.txt @@ -30,8 +30,10 @@ When to use this site: - [For developers](https://wordcell.io/developers): The coding-agent workflow: code-path context, scoped rules, note history, and the Agent Skill. - [Benchmarks](https://wordcell.io/benchmarks): Wordcell's context-size and - SciFact reranking measurements with their limits, the Oh kernel's LoCoMo run - and its 60-question pilot with Supermemory and BM25, and figures that other + SciFact reranking measurements with their limits; the Oh kernel's + 500-question LongMemEval-S study of Oh semantic retrieval against BM25, its + LoCoMo run, and its earlier 60-question pilot with Supermemory and BM25, all + measured by Oh rather than through a Wordcell vault; and figures that other memory systems publish, each with its source. - [Wordcell and Basic Memory compared](https://wordcell.io/compare/basic-memory): Where notes live, how notes form, note structure, search, repository diff --git a/site/scripts/sync-blog.ts b/site/scripts/sync-blog.ts index 7f33b6e..079759a 100644 --- a/site/scripts/sync-blog.ts +++ b/site/scripts/sync-blog.ts @@ -8,7 +8,19 @@ import { scifactStudy } from "../wordcell/benchmark-evidence.ts"; import { grouped, longDate, prose, signed } from "../wordcell/format.ts"; import { formatBytes, handoffEvidence } from "../wordcell/handoff-evidence.ts"; import { isLaunchRoute } from "../wordcell/launch-routes.ts"; -import { locomoArms, locomoFacts, locomoLimitQuotes, ohAttribution, ohLinks, pilotArms, pilotInterval, pilotStudy } from "../wordcell/oh-evidence.ts"; +import { + locomoArms, + locomoFacts, + locomoLimitQuotes, + longMemEvalArms, + longMemEvalComparison, + longMemEvalFacts, + ohAttribution, + ohLinks, + pilotArms, + pilotInterval, + pilotStudy, +} from "../wordcell/oh-evidence.ts"; import { renderMarkdownHtml, type RelativeTargetResolver } from "./readme-html.ts"; import pilotJson from "../../docs/evaluations/oh/memory-framework-pilot-v1.json"; import scifactJson from "../../docs/evaluations/wordcell-scifact-20260919.json"; @@ -97,6 +109,15 @@ export const evidenceFigures = { "oh-locomo.attribution": ohAttribution, "oh-locomo.limit": locomoLimitQuotes[0], "oh-locomo.url": ohLinks.locomoResult, + "oh-longmemeval.questions": grouped(longMemEvalFacts.questions), + "oh-longmemeval.completed": longDate(longMemEvalFacts.completed), + "oh-longmemeval.semantic": percentText(pick(longMemEvalArms, "oh-semantic-96k").percent, 2), + "oh-longmemeval.bm25": percentText(pick(longMemEvalArms, "bm25-96k").percent, 2), + "oh-longmemeval.estimate": signed(longMemEvalComparison.primary.difference, 1), + "oh-longmemeval.lower": signed(longMemEvalComparison.primary.lower, 1), + "oh-longmemeval.upper": signed(longMemEvalComparison.primary.upper, 1), + "oh-longmemeval.level": longMemEvalComparison.primary.level, + "oh-longmemeval.url": ohLinks.longMemEvalResult, "oh-pilot.questions": grouped(pilotStudy.sampleSize), "oh-pilot.date": longDate(pilotStudy.measuredAt), "oh-pilot.supermemory": percentText(pick(pilotArms, "supermemory").percent, 2), diff --git a/site/tests/blog.test.tsx b/site/tests/blog.test.tsx index fcd6541..d4ced0b 100644 --- a/site/tests/blog.test.tsx +++ b/site/tests/blog.test.tsx @@ -156,6 +156,9 @@ describe("Wordcell blog", () => { expect(evidenceFigures["oh-locomo.attribution"]).toBe(ohAttribution); expect(evidenceFigures["oh-locomo.url"]).toBe(ohLinks.locomoResult); expect(evidenceFigures["oh-pilot.url"]).toBe(ohLinks.pilotResult); + expect(evidenceFigures["oh-longmemeval.url"]).toBe(ohLinks.longMemEvalResult); + expect([evidenceFigures["oh-longmemeval.semantic"], evidenceFigures["oh-longmemeval.bm25"]]).toEqual(["88.87%", "86.13%"]); + expect([evidenceFigures["oh-longmemeval.estimate"], evidenceFigures["oh-longmemeval.lower"], evidenceFigures["oh-longmemeval.upper"], evidenceFigures["oh-longmemeval.level"]]).toEqual(["+2.8", "0.0", "+5.6", "95%"]); for (const [key, value] of Object.entries(evidenceFigures)) { expect(value, key).not.toBe(""); expect(value, key).not.toContain("{{"); @@ -222,6 +225,9 @@ describe("Wordcell blog", () => { test("the launch post keeps Oh's results attributed, states its status once, and makes no ranking claim", async () => { const source = await read(`content/blog/${launchSlug}.md`); expect(blogHtml[launchSlug]).toContain(ohAttribution); + // The in-sample lab pipeline is never cited in the post, so its figure cannot read as a product score. + expect(blogHtml[launchSlug]).not.toContain("93.07"); + expect(blogHtml[launchSlug]).toContain("does not rule out a tie"); expect(source.match(/Latest release: /gu)?.length).toBe(1); expect(source.match(/available from source until the next release/gu)?.length).toBe(1); for (const banned of [/\bSOTA\b/iu, /state of the art/iu, /CLONEMEM/iu, /\bwe\b/iu, /\bour\b/iu, /honest/iu, /\bthe first\b/iu, /\bthe only\b/iu]) { @@ -242,6 +248,7 @@ describe("Wordcell blog", () => { const ohHrefs = hrefs.filter((href) => href.startsWith("https://github.com/hraness/oh/")); expect(ohHrefs).toContain(ohLinks.locomoResult); expect(ohHrefs).toContain(ohLinks.pilotResult); + expect(ohHrefs).toContain(ohLinks.longMemEvalResult); const ohValues: string[] = Object.values(ohLinks); for (const href of ohHrefs) expect(ohValues).toContain(href); const sourceHrefs: string[] = article.sources.map((source) => source.href); diff --git a/site/tests/launch-pages.test.tsx b/site/tests/launch-pages.test.tsx index dd0030a..e77497d 100644 --- a/site/tests/launch-pages.test.tsx +++ b/site/tests/launch-pages.test.tsx @@ -8,6 +8,13 @@ import { renderToStaticMarkup } from "react-dom/server"; import { BenchmarkComparison, type BenchmarkStudy } from "../wordcell/benchmark-comparison"; import { scifactStudy } from "../wordcell/benchmark-evidence"; import { + longMemEvalArms, + longMemEvalComparison, + longMemEvalFacts, + longMemEvalLabPipeline, + longMemEvalLimitQuotes, + longMemEvalStudy, + longMemEvalTypes, locomoArms, locomoCategories, locomoFacts, @@ -16,6 +23,7 @@ import { locomoStudy, ohAttribution, ohLinks, + ohLongMemEvalPost, ohSources, pilotArms, pilotInterval, @@ -76,7 +84,7 @@ describe("vendored Oh artifacts", () => { expect(manifest.schema).toBe("wordcell.vendored-sources.v1"); const artifacts = manifest.artifacts; if (!Array.isArray(artifacts)) throw new TypeError("sources.json artifacts must be an array."); - expect(artifacts.length).toBe(2); + expect(artifacts.length).toBe(3); for (const [index, entry] of artifacts.entries()) { const artifact = record(entry, `artifacts[${index}]`); const file = text(artifact.file, "file"); @@ -183,6 +191,70 @@ describe("Oh LoCoMo evidence", () => { }); }); +describe("Oh LongMemEval-S 500 evidence", () => { + test("matched arms, the frozen interval, and question types match the vendored result", async () => { + const raw = await vendoredJson("memory-longmemeval-s-500-v1.json"); + const systems = raw.systems as readonly Readonly>[]; + const system = (id: string) => record(systems.find((entry) => entry.id === id), id); + expect(longMemEvalArms.map((arm) => [arm.system, arm.percent, arm.correctAnswers, arm.answers, arm.majorityCorrect])).toEqual([ + ["Oh semantic retrieval", "88.87", 1333, 1500, 445], + ["BM25 retrieval", "86.13", 1292, 1500, 431], + ]); + for (const arm of longMemEvalArms) { + const entry = system(arm.id); + expect(arm.percent).toBe((entry.percent as number).toFixed(2)); + expect(arm.correctAnswers).toBe(entry.correctAnswers as number); + } + expect(longMemEvalStudy.rows.map((row) => row.detail)).toEqual(["1,333 of 1,500 answers", "1,292 of 1,500 answers"]); + expect(longMemEvalStudy.rows.map((row) => row.id)).not.toContain("oh-reading-pipeline"); + expect(longMemEvalStudy.comparability).toBe("same-run"); + expect(longMemEvalStudy.sampleSize).toBe(500); + expect(longMemEvalStudy.measuredAt).toBe(raw.completed as string); + expect(longMemEvalComparison.primary).toEqual({ difference: 2.8, lower: 0, upper: 5.6, level: "95%" }); + expect(longMemEvalComparison.mean).toEqual({ difference: 2.73, lower: 0.53, upper: 5.07, level: "95%" }); + const comparison = record((raw.comparisons as readonly unknown[]).find((entry) => record(entry, "comparison").left === "oh-semantic-96k"), "comparison"); + expect(record(comparison.correctInTwoOrThreeRuns, "primary").interval95).toEqual([longMemEvalComparison.primary.lower, longMemEvalComparison.primary.upper]); + expect(record(comparison.meanOfThreeRuns, "mean").interval95).toEqual([longMemEvalComparison.mean.lower, longMemEvalComparison.mean.upper]); + expect(longMemEvalComparison.tieNotRuledOut).toBe(true); + expect([longMemEvalComparison.gained, longMemEvalComparison.lost]).toEqual([31, 17]); + expect(longMemEvalTypes.map((type) => [type.id, type.questions, ...type.percents])).toEqual([ + ["knowledge-update", 78, "91.03", "92.74"], + ["multi-session", 133, "84.21", "74.94"], + ["single-session-assistant", 56, "95.24", "94.64"], + ["single-session-preference", 30, "63.33", "65.56"], + ["single-session-user", 70, "95.24", "97.62"], + ["temporal-reasoning", 133, "91.98", "88.47"], + ["abstention", 30, "91.11", "90.00"], + ]); + expect(longMemEvalFacts.matchedSupermemoryRun).toBe(false); + expect(record(raw.supermemory, "supermemory").matchedFullRun).toBe(false); + }); + + test("the lab pipeline stays framed as in-sample and outside the Oh package", async () => { + const raw = await vendoredJson("memory-longmemeval-s-500-v1.json"); + expect(longMemEvalLabPipeline).toMatchObject({ percent: "93.07", majorityCorrect: 474, inSample: true, inOhPackage: false }); + expect(record(raw.packageBoundary, "packageBoundary").labPublished).toBe(false); + expect(record(raw.exposure, "exposure").inSample).toBe(true); + const markup = renderToStaticMarkup(); + const text = pageText(markup); + // The in-sample figure appears once, inside the sentence that says it is not Oh's or Wordcell's score. + expect(text.split("93.07").length - 1).toBe(1); + expect(text).toMatch(/lab reading pipeline that scored 93\.07% on the mean of three runs and answered 474 of 500 questions correctly in at least two\. It is not charted here and is not Oh’s or Wordcell’s score/); + expect(text).toContain("is not Oh’s or Wordcell’s score"); + expect(text).toContain("the figure is in-sample"); + expect(text).toContain("are not part of the Oh package"); + }); + + test("quoted limits are verbatim", async () => { + const raw = await vendoredJson("memory-longmemeval-s-500-v1.json"); + const statements = [...(raw.limitations as readonly string[]), text(record(raw.exposure, "exposure").priorStudies, "priorStudies")]; + for (const quote of Object.values(longMemEvalLimitQuotes)) { + expect(statements.some((statement) => statement.includes(quote))).toBe(true); + expect(quote).not.toMatch(notSota); + } + }); +}); + describe("Oh LongMemEval pilot evidence", () => { test("arm rates and the interval match the vendored pilot result", async () => { const raw = await vendoredJson("memory-framework-pilot-v1.json"); @@ -235,10 +307,17 @@ describe("Oh evidence provenance and rendering", () => { expect(ohSources.map((source) => source.href)).toEqual([ "https://github.com/hraness/oh/blob/3add170ca8d931603e68dee07f3cbdcf9c08c706/benchmarks/results/memory-evolution-locomo-sealed-1540-v1.json", "https://github.com/hraness/oh/blob/9edd9f1bc18d0f4c15b040add10caad26e782275/benchmarks/results/memory-framework-pilot-v1.json", + "https://github.com/hraness/oh/blob/21c500cf38928ab610c15c438557fbed5227ca4b/benchmarks/results/memory-longmemeval-s-500-v1.json", ]); + const pinned = [ + "https://github.com/hraness/oh/blob/9edd9f1bc18d0f4c15b040add10caad26e782275/benchmarks/", + "https://github.com/hraness/oh/blob/21c500cf38928ab610c15c438557fbed5227ca4b/benchmarks/", + ]; for (const href of Object.values(ohLinks)) { - expect(href).toStartWith("https://github.com/hraness/oh/blob/9edd9f1bc18d0f4c15b040add10caad26e782275/benchmarks/"); + expect(pinned.some((prefix) => href.startsWith(prefix)), href).toBe(true); } + expect(ohLinks.longMemEvalResult).toStartWith(pinned[1] ?? ""); + expect(ohLongMemEvalPost).toBe("https://oh.computer/blog/longmemeval-s-user-log"); }); test("both studies render as same-run charts with their derived figures and no SOTA claim", () => { @@ -416,6 +495,18 @@ describe("/benchmarks", () => { for (const value of [paired.better, paired.worse, paired.tied]) expect(text).toContain(grouped(value)); } for (const arm of pilotArms) expect(text).toContain(`${arm.percent}%`); + for (const arm of longMemEvalArms) expect(text).toContain(`${arm.percent}%`); + for (const type of longMemEvalTypes) { + expect(text).toContain(`${type.name} ${grouped(type.questions)} ${type.percents.map((percent) => `${percent}%`).join(" ")}`); + } + expect(text).toContain("That is +2.8 percentage points, with a 95% interval from 0.0 to +5.6."); + expect(text).toContain("does not rule out a tie"); + expect(text).toContain("it includes no matched run of Supermemory or any other memory framework"); + expect(text).toContain("the only matched run of Oh against Supermemory"); + for (const quote of Object.values(longMemEvalLimitQuotes)) expect(text).toContain(quote); + expect(markup).toContain(`href="${ohLongMemEvalPost}"`); + expect(text.indexOf("all 500 LongMemEval-S questions")).toBeLessThan(text.indexOf("Answers judged correct on LoCoMo")); + expect(text.indexOf("Answers judged correct on LoCoMo")).toBeLessThan(text.indexOf("smaller, earlier LongMemEval pilot")); for (const interval of [pilotInterval, pilotSecondaryInterval]) { expect(text).toContain(`${signed(interval.estimate)} percentage points, with a 95% interval from ${signed(interval.lower)} to ${signed(interval.upper)}`); } @@ -444,6 +535,25 @@ describe("/benchmarks", () => { expect(text).not.toContain("\u2014"); }); + test("prose surfaces cite the LongMemEval-S figures derived from the vendored result, never the in-sample pipeline", async () => { + const [semantic, bm25] = longMemEvalArms; + const read = async (...path: string[]) => (await readFile(join(repository, ...path), "utf8")).replace(/\s+/g, " "); + for (const path of [["README.md"], ["CHANGELOG.md"]]) { + const prose = await read(...path); + expect(prose, path.join("/")).toContain(`${semantic?.percent}% and BM25 ${bm25?.percent}%`); + expect(prose, path.join("/")).toContain("on the measure Oh named before the run, its interval does not rule out a tie"); + } + for (const path of [["docs", "evidence.md"], ["site", "public", "llms.txt"]]) { + expect(await read(...path), path.join("/")).toContain(`${longMemEvalFacts.questions}-question LongMemEval-S`); + } + for (const path of [["README.md"], ["CHANGELOG.md"], ["docs", "evidence.md"], ["docs", "comparisons.md"], ["site", "public", "llms.txt"]]) { + const prose = await read(...path); + expect(prose, path.join("/")).not.toContain(longMemEvalLabPipeline.percent); + // Oh's older, unmatched GPT-5 mini figure; the launch plan's claims check keeps it off every surface. + expect(prose, path.join("/")).not.toContain("89.8"); + } + }); + test("metadata has a canonical path and a description of 110 to 160 characters", () => { expect(benchmarksMetadata.alternates?.canonical).toBe("/benchmarks"); expect(benchmarksMetadata.title).toBe("Wordcell and Oh benchmarks, with their limits"); @@ -522,6 +632,9 @@ describe("/compare/supermemory", () => { expect(text).toContain("It is Oh’s result, not Wordcell’s"); expect(text).toContain("Wordcell has published no head-to-head comparison with Supermemory"); if (pilotInterval.crossesZero) expect(text).toContain("it does not separate Oh from Supermemory"); + expect(text).toContain("Oh’s later 500-question LongMemEval-S study compares Oh with BM25 only, so this pilot remains the only matched comparison with Supermemory."); + expect(text).not.toContain("88.87"); + expect(text).not.toContain(longMemEvalLabPipeline.percent); expect(markup).toContain('href="/migrate/supermemory"'); expect(markup).toContain('href="/benchmarks#comparisons"'); expect(await expectDocLinksResolve(markup)).toBeGreaterThanOrEqual(4); @@ -717,7 +830,7 @@ describe("discovery files", () => { describe("stacked tables", () => { const pages = [ - { name: "/benchmarks", markup: () => renderToStaticMarkup(), tables: 2 }, + { name: "/benchmarks", markup: () => renderToStaticMarkup(), tables: 3 }, { name: "/compare/basic-memory", markup: () => renderToStaticMarkup(), tables: 1 }, { name: "/compare/mem0", markup: () => renderToStaticMarkup(), tables: 1 }, { name: "/compare/supermemory", markup: () => renderToStaticMarkup(), tables: 3 }, diff --git a/site/wordcell/format.ts b/site/wordcell/format.ts index 036b71f..e416751 100644 --- a/site/wordcell/format.ts +++ b/site/wordcell/format.ts @@ -17,8 +17,8 @@ export function longDate(isoDate: string): string { return dates.format(new Date(`${isoDate}T00:00:00Z`)); } -/** Signed to two decimals with a real minus sign: "−3.33", "+6.67". */ -export function signed(value: number): string { - const magnitude = Math.abs(value).toFixed(2); +/** Signed with a real minus sign, to two decimals unless told otherwise: "−3.33", "+6.67", "+2.8". */ +export function signed(value: number, digits: 1 | 2 = 2): string { + const magnitude = Math.abs(value).toFixed(digits); return value < 0 ? `−${magnitude}` : value > 0 ? `+${magnitude}` : magnitude; } diff --git a/site/wordcell/oh-evidence.ts b/site/wordcell/oh-evidence.ts index c6319a2..9dc5c1d 100644 --- a/site/wordcell/oh-evidence.ts +++ b/site/wordcell/oh-evidence.ts @@ -1,7 +1,9 @@ import locomoJson from "../../docs/evaluations/oh/memory-evolution-locomo-sealed-1540-v1.json"; +import longMemEvalJson from "../../docs/evaluations/oh/memory-longmemeval-s-500-v1.json"; import pilotJson from "../../docs/evaluations/oh/memory-framework-pilot-v1.json"; import sourcesJson from "../../docs/evaluations/oh/sources.json"; import type { BenchmarkStudy } from "./benchmark-comparison"; +import { signed } from "./format"; // The vendored files are Oh's published bytes. They are read as `unknown` // through small readers that throw, so a changed shape fails the build instead @@ -72,15 +74,22 @@ export function quoteFrom(statements: readonly string[], needle: string): string return needle; } -// Pinned upstream pages, at the commit that also published the pilot result. +// Pinned upstream pages: the LoCoMo and pilot pages at the commit that published +// the pilot result, and the 500-question study at the commit that published it. const ohDocs = "https://github.com/hraness/oh/blob/9edd9f1bc18d0f4c15b040add10caad26e782275/benchmarks"; +export const ohLongMemEvalCommit = "21c500cf38928ab610c15c438557fbed5227ca4b"; +const ohLongMemEvalDocs = `https://github.com/hraness/oh/blob/${ohLongMemEvalCommit}/benchmarks`; export const ohLinks = { + longMemEvalResult: `${ohLongMemEvalDocs}/LONGMEMEVAL_S_500_RESULT_V1.md`, locomoResult: `${ohDocs}/EVOLUTION_RELEASE_RESULTS.md#matched-descriptive-comparison-on-locomo`, pilotResult: `${ohDocs}/FRAMEWORK_PILOT_RESULT_V1.md`, - benchmarks: `${ohDocs}/README.md`, + benchmarks: `${ohLongMemEvalDocs}/README.md`, } as const; +/** Oh's own write-up of the 500-question study, for readers rather than reviewers. */ +export const ohLongMemEvalPost = "https://oh.computer/blog/longmemeval-s-user-log"; + /** docs/evidence.md carries the same two sentences with straight apostrophes; a test pins them together. */ export const ohAttribution = "Oh’s conversation-memory benchmarks evaluate its own memory-retrieval API, reader models, and evaluation protocols. Those scores do not transfer to a Wordcell vault merely because it uses the same library."; @@ -442,7 +451,7 @@ const incompleteSentence = export const pilotStudy = { id: "oh-pilot", - title: "Answers judged correct in a small LongMemEval pilot", + title: "Answers judged correct in a smaller, earlier LongMemEval pilot", dataset: pilotDataset, metric: "Answers judged correct", unit: "percent", @@ -471,3 +480,282 @@ export const pilotStudy = { color: pilotArmColors[arm.id], })), } as const satisfies BenchmarkStudy; + +// LongMemEval-S, all 500 questions: Oh semantic retrieval against BM25 under +// one protocol card, measured by Oh. The lab pipeline in the same file is +// in-sample and outside the Oh package; it is described, never charted. + +const longMemEval = record(longMemEvalJson as unknown, "LongMemEval-S 500 result"); +if (longMemEval.protocol !== "oh.longmemeval-s-500-public-result.v1") throw new TypeError("Unexpected LongMemEval-S 500 result protocol."); +const longMemEvalCompleted = text(longMemEval.completed, "completed"); +if (!/^\d{4}-\d{2}-\d{2}$/.test(longMemEvalCompleted)) throw new TypeError("The LongMemEval-S 500 completion date must be an ISO date."); + +const longMemEvalDataset = record(longMemEval.dataset, "dataset"); +if (text(longMemEvalDataset.name, "dataset.name") !== "LongMemEval-S") throw new TypeError("Unexpected LongMemEval-S dataset name."); +const longMemEvalQuestions = count(longMemEvalDataset.questions, "dataset.questions"); +const longMemEvalTypeCounts = record(longMemEvalDataset.questionTypes, "dataset.questionTypes"); +if (Object.values(longMemEvalTypeCounts).reduce((sum, value) => sum + count(value, "question type count"), 0) !== longMemEvalQuestions) { + throw new TypeError("LongMemEval-S question types must cover every question."); +} + +const longMemEvalExposure = record(longMemEval.exposure, "exposure"); +if (longMemEvalExposure.inSample !== true) throw new TypeError("The LongMemEval-S 500 exposure record changed; update the page's exposure line."); +const priorStudies = text(longMemEvalExposure.priorStudies, "exposure.priorStudies"); + +const longMemEvalReader = record(longMemEval.reader, "reader"); +const longMemEvalJudge = record(longMemEval.judge, "judge"); +if (text(longMemEvalReader.model, "reader.model") !== "openai/gpt-5-mini") throw new TypeError("Unexpected LongMemEval-S reader."); +if (text(longMemEvalJudge.model, "judge.model") !== "openai/gpt-4o") throw new TypeError("Unexpected LongMemEval-S judge."); +if (longMemEvalReader.snapshotPinned !== false || longMemEvalJudge.snapshotPinned !== false) { + throw new TypeError("The LongMemEval-S reader or judge is now pinned; update the page's model lines."); +} +const paperJudge = text(longMemEvalJudge.paperJudge, "judge.paperJudge"); + +const longMemEvalRuns = record(longMemEval.runs, "runs"); +const runsPerQuestion = count(longMemEvalRuns.repeatsPerQuestion, "runs.repeatsPerQuestion"); +if (runsPerQuestion !== 3) throw new TypeError("The two-of-three measure assumes three runs per question."); +const longMemEvalScoring = record(longMemEval.scoring, "scoring"); +const frozenPrimary = text(longMemEvalScoring.freezePrimary, "scoring.freezePrimary"); +if (frozenPrimary !== "questions answered correctly in at least two of three runs") { + throw new TypeError("The pre-registered LongMemEval-S measure changed; update the page."); +} + +function bounded(value: number, limit: number, label: string): number { + if (value > limit) throw new TypeError(`${label} counts more correct results than it has.`); + return value; +} + +type TypeResult = Readonly<{ questionType: string; questions: number; correctAnswers: number; answers: number; percent: number }>; +type LongMemEvalSystem = Readonly<{ + id: string; + label: string; + protocol: string; + runs: string; + answers: number; + correctAnswers: number; + percent: number; + majorityCorrect: number; + budget: number; + byType: readonly TypeResult[]; + abstention: TypeResult; +}>; + +function typeResult(value: unknown, label: string, questionType: string): TypeResult { + const entry = record(value, label); + const questions = count(entry.questions, `${label}.questions`); + const answers = count(entry.answers, `${label}.answers`); + const correctAnswers = bounded(count(entry.correctAnswers, `${label}.correctAnswers`), answers, label); + if (answers !== questions * runsPerQuestion) throw new TypeError(`${label} must answer every question ${runsPerQuestion} times.`); + const stated = finite(entry.percent, `${label}.percent`); + if (Math.abs(stated - hundredths((100 * correctAnswers) / answers)) > 1e-9) throw new TypeError(`${label}.percent must equal its fraction.`); + return { questionType, questions, answers, correctAnswers, percent: stated }; +} + +const longMemEvalSystems: readonly LongMemEvalSystem[] = list(longMemEval.systems, "systems").map((entry, index) => { + const label = `systems[${index}]`; + const system = record(entry, label); + const answers = count(system.answers, `${label}.answers`); + if (answers !== longMemEvalQuestions * runsPerQuestion) throw new TypeError(`${label} must answer every question ${runsPerQuestion} times.`); + const correctAnswers = bounded(count(system.correctAnswers, `${label}.correctAnswers`), answers, label); + const perRun = list(system.correctPerRun, `${label}.correctPerRun`).map((value, run) => count(value, `${label}.correctPerRun[${run}]`)); + if (perRun.length !== runsPerQuestion || perRun.reduce((sum, value) => sum + value, 0) !== correctAnswers) { + throw new TypeError(`${label} runs must add up to its correct answers.`); + } + const percentValue = finite(system.percent, `${label}.percent`); + if (Math.abs(percentValue - hundredths((100 * correctAnswers) / answers)) > 1e-9) throw new TypeError(`${label}.percent must equal its fraction.`); + const memory = record(system.memoryBytes, `${label}.memoryBytes`); + const byType = list(system.byType, `${label}.byType`).map((value, position) => { + const typeRecord = record(value, `${label}.byType[${position}]`); + const questionType = text(typeRecord.questionType, `${label}.byType[${position}].questionType`); + const result = typeResult(typeRecord, `${label}.byType[${position}]`, questionType); + if (result.questions !== count(longMemEvalTypeCounts[questionType], `dataset.questionTypes.${questionType}`)) { + throw new TypeError(`${label} ${questionType} must cover every question of that type.`); + } + return result; + }); + if (new Set(byType.map((result) => result.questionType)).size !== Object.keys(longMemEvalTypeCounts).length || byType.length !== Object.keys(longMemEvalTypeCounts).length) { + throw new TypeError(`${label} must report each question type exactly once.`); + } + const abstention = typeResult(system.abstention, `${label}.abstention`, "abstention"); + if (abstention.questions !== count(longMemEvalDataset.abstentionQuestions, "dataset.abstentionQuestions")) { + throw new TypeError(`${label} abstention must cover every abstention question.`); + } + return { + id: text(system.id, `${label}.id`), + label: text(system.label, `${label}.label`), + protocol: text(system.protocol, `${label}.protocol`), + runs: text(system.runs, `${label}.runs`), + answers, + correctAnswers, + percent: percentValue, + majorityCorrect: bounded(count(system.questionsCorrectInTwoOrThreeRuns, `${label}.questionsCorrectInTwoOrThreeRuns`), longMemEvalQuestions, label), + budget: count(memory.budget ?? memory.firstPassBudget, `${label}.memoryBytes budget`), + byType, + abstention, + }; +}); + +const semanticSystem = one(longMemEvalSystems, (system) => system.id === "oh-semantic-96k", "Oh semantic retrieval system"); +const bm25System = one(longMemEvalSystems, (system) => system.id === "bm25-96k", "BM25 retrieval system"); +const pipelineSystem = one(longMemEvalSystems, (system) => system.id === "oh-reading-pipeline", "lab pipeline system"); +if (semanticSystem.protocol !== bm25System.protocol || semanticSystem.budget !== bm25System.budget) { + throw new TypeError("The matched LongMemEval-S baselines must share one protocol card and budget."); +} +const matchedBudget = semanticSystem.budget; +const matchedTopTurns = /\btop (\d+) turns\b/u.exec(semanticSystem.protocol)?.[1]; +if (matchedTopTurns === undefined) throw new TypeError("The matched protocol card must name how many turns it packs."); + +type Interval = Readonly<{ difference: number; lower: number; upper: number; level: string }>; +function pairedInterval(value: unknown, label: string): Interval { + const entry = record(value, label); + // The confidence level comes from the artifact's own field name, such as `interval95`. + const keys = Object.keys(entry).filter((key) => /^interval\d{2}$/u.test(key)); + const [key] = keys; + if (keys.length !== 1 || key === undefined) throw new TypeError(`${label} must record exactly one interval.`); + const bounds = list(entry[key], `${label}.${key}`).map((bound, index) => finite(bound, `${label}.${key}[${index}]`)); + const [lower, upper] = bounds; + if (bounds.length !== 2 || lower === undefined || upper === undefined || lower > upper) throw new TypeError(`${label} must have a two-sided interval.`); + const difference = finite(entry.differencePoints, `${label}.differencePoints`); + if (difference < lower || difference > upper) throw new TypeError(`${label} must lie inside its interval.`); + return { difference, lower, upper, level: `${key.slice("interval".length)}%` }; +} + +const semanticOverBm25 = one( + list(longMemEval.comparisons, "comparisons").map((entry, index) => record(entry, `comparisons[${index}]`)), + (entry) => entry.left === semanticSystem.id && entry.right === bm25System.id, + "Oh semantic over BM25 comparison", +); +if (count(semanticOverBm25.pairedQuestions, "pairedQuestions") !== longMemEvalQuestions) throw new TypeError("The comparison must pair every question."); +const majorityComparison = record(semanticOverBm25.correctInTwoOrThreeRuns, "correctInTwoOrThreeRuns"); +const primaryInterval = pairedInterval(majorityComparison, "correctInTwoOrThreeRuns"); + +/** Oh semantic retrieval minus BM25, in percentage points, on both of Oh's measures, with Oh's 95% bootstrap intervals. */ +export const longMemEvalComparison = { + left: semanticSystem.label, + right: bm25System.label, + pairedQuestions: longMemEvalQuestions, + method: text(longMemEval.comparisonMethod, "comparisonMethod"), + /** The measure Oh named before the run: questions correct in at least two of three runs. */ + primary: primaryInterval, + gained: count(majorityComparison.questionsGained, "questionsGained"), + lost: count(majorityComparison.questionsLost, "questionsLost"), + mean: pairedInterval(semanticOverBm25.meanOfThreeRuns, "meanOfThreeRuns"), + /** True when the pre-registered interval reaches zero, so a tie is not ruled out. */ + tieNotRuledOut: primaryInterval.lower <= 0 && primaryInterval.upper >= 0, +} as const; + +export type LongMemEvalArm = Readonly<{ id: string; system: string; percent: string; correctAnswers: number; answers: number; majorityCorrect: number }>; + +const matchedSystems = [ + { system: semanticSystem, color: "var(--primary)" }, + { system: bm25System, color: "var(--muted)" }, +] as const; + +/** The two matched arms, Oh semantic retrieval first. */ +export const longMemEvalArms: readonly LongMemEvalArm[] = matchedSystems.map(({ system }) => ({ + id: system.id, + system: system.label, + percent: system.percent.toFixed(2), + correctAnswers: system.correctAnswers, + answers: system.answers, + majorityCorrect: system.majorityCorrect, +})); + +const typeNames: Readonly> = { + "knowledge-update": "Knowledge update", + "multi-session": "Multi-session", + "single-session-assistant": "Single-session assistant", + "single-session-preference": "Single-session preference", + "single-session-user": "Single-session user", + "temporal-reasoning": "Temporal reasoning", + abstention: "Abstention, across types", +}; + +export type LongMemEvalType = Readonly<{ id: string; name: string; questions: number; percents: readonly string[] }>; + +/** Accuracy by question type, mean of three runs; `percents` follows `longMemEvalArms` order. */ +export const longMemEvalTypes: readonly LongMemEvalType[] = [...Object.keys(longMemEvalTypeCounts), "abstention"].map((id) => { + const cells = matchedSystems.map(({ system }) => + id === "abstention" ? system.abstention : one(system.byType, (result) => result.questionType === id, `${system.id} ${id} result`), + ); + const questions = cells[0]?.questions ?? 0; + if (cells.some((cell) => cell.questions !== questions)) throw new TypeError(`${id} must have one question count across arms.`); + const name = typeNames[id]; + if (name === undefined) throw new TypeError(`Unknown LongMemEval-S question type ${id}.`); + return { id, name, questions, percents: cells.map((cell) => cell.percent.toFixed(2)) }; +}); + +const longMemEvalLimitations = list(longMemEval.limitations, "limitations").map((entry, index) => text(entry, `limitations[${index}]`)); + +/** Verbatim sentences from Oh's own limits for the 500-question study. */ +export const longMemEvalLimitQuotes = { + aliases: quoteFrom(longMemEvalLimitations, "The reader and judge are gateway aliases; the models behind them can change."), + audit: quoteFrom(longMemEvalLimitations, "AI agents ran the study and wrote the report; no person or outside group has audited it."), + inSample: quoteFrom(longMemEvalLimitations, "In-sample: the added instructions and question rules were written after studying all 500 questions, and two rules match single question types on this benchmark."), + budget: quoteFrom(longMemEvalLimitations, "The pipeline reads up to 180,000 bytes and makes extra calls; the matched baselines read at most 96,000 bytes once."), + exposure: quoteFrom([priorStudies], "earlier Oh studies scored all 500 questions and read some of them one by one"), +} as const; + +const packageBoundary = record(longMemEval.packageBoundary, "packageBoundary"); +const labOnly = list(packageBoundary.labOnly, "packageBoundary.labOnly").map((entry, index) => text(entry, `labOnly[${index}]`)); +if (labOnly.length === 0 || packageBoundary.labPublished !== false) throw new TypeError("The lab pipeline's package boundary changed; update the page."); +const supermemoryRecord = record(longMemEval.supermemory, "supermemory"); +if (typeof supermemoryRecord.matchedFullRun !== "boolean") throw new TypeError("supermemory.matchedFullRun must be a boolean."); +if (supermemoryRecord.matchedFullRun) throw new TypeError("Oh now reports a matched full Supermemory run; update the page."); +const matchedSupermemoryRun: boolean = supermemoryRecord.matchedFullRun; + +/** + * The lab pipeline's result, for a clearly framed sentence only. It is + * in-sample and not part of the Oh package, so it never becomes a chart row or + * a headline figure. + */ +export const longMemEvalLabPipeline = { + label: pipelineSystem.label, + percent: pipelineSystem.percent.toFixed(2), + majorityCorrect: pipelineSystem.majorityCorrect, + questions: longMemEvalQuestions, + firstPassBudget: pipelineSystem.budget, + labOnly, + labOnlyText: sentenceList(labOnly), + inSample: true, + inOhPackage: false, +} as const; + +export const longMemEvalFacts = { + completed: longMemEvalCompleted, + questions: longMemEvalQuestions, + runsPerQuestion, + matchedBudget, + semanticRuns: semanticSystem.runs, + paperJudge, + matchedSupermemoryRun, + sourceCommit: ohLongMemEvalCommit, +} as const; + +export const longMemEvalStudy = { + id: "oh-longmemeval-500", + title: `Answers judged correct on all ${longMemEvalQuestions} LongMemEval-S questions`, + dataset: "LongMemEval-S", + metric: `Answers judged correct, mean of ${spelled(runsPerQuestion)} runs`, + unit: "percent", + sampleSize: longMemEvalQuestions, + sampleNoun: "questions", + scope: `Oh measured on its own, not through a Wordcell vault. On the measure Oh named before the run, Oh semantic retrieval minus BM25 is ${signed(primaryInterval.difference, 1)} points with a ${primaryInterval.level} interval from ${signed(primaryInterval.lower, 1)} to ${signed(primaryInterval.upper, 1)}${longMemEvalComparison.tieNotRuledOut ? ", which does not rule out a tie" : ""}.`, + measuredAt: longMemEvalCompleted, + dateLabel: "Completed", + valueDigits: 2, + model: `Oh semantic retrieval and BM25 keyword retrieval, each packing its top ${matchedTopTurns} turns into ${matchedBudget.toLocaleString("en-US")} bytes per question.`, + reader: `GPT-5 mini, answering every question ${spelled(runsPerQuestion)} times through an unpinned Vercel AI Gateway alias.`, + evaluator: `A GPT-4o judge with LongMemEval’s own grading prompts, through an unpinned gateway alias; the paper’s judge is the pinned ${paperJudge}. Each system answered ${(longMemEvalQuestions * runsPerQuestion).toLocaleString("en-US")} times in all, and every call completed.`, + contextBudget: `The same ${matchedBudget.toLocaleString("en-US")}-byte cap and one reader call per answer for both systems.`, + exposure: `${capitalized(priorStudies)}, so none is unseen. The Oh semantic row combines two runs: ${semanticSystem.runs}.`, + comparability: "same-run", + source: { label: "Oh’s published LongMemEval-S result", href: ohLinks.longMemEvalResult }, + rows: matchedSystems.map(({ system, color }) => ({ + id: system.id, + label: system.label, + value: (100 * system.correctAnswers) / system.answers, + detail: `${system.correctAnswers.toLocaleString("en-US")} of ${system.answers.toLocaleString("en-US")} answers`, + color, + })), +} as const satisfies BenchmarkStudy;