diff --git a/.agents/skills/augustus/SKILL.md b/.agents/skills/augustus/SKILL.md index 3987ff67..e13bf7eb 100644 --- a/.agents/skills/augustus/SKILL.md +++ b/.agents/skills/augustus/SKILL.md @@ -54,7 +54,7 @@ classical method you already trust, substitute it, classify the win "paraphrase brittleness", "allowlist then judge", "TOCTOU-of-Noul", "Jev inside the database / sqlite-jev", "Jev picks bitrate / join order / the model", "wait for Archer", "lint the request / missing - other", "training confronts Choice other / none-of-the-above", "soft AGENTS.md rules vs the linter", "screenshot Choice / omni System One", "extractive quotes / pointer not generator", "compaction summarize vs pointer", "encoder vs Jev compaction backend", "shadow-mode compaction rollout", "CI flaky-vs-real merge gate", "fail-open VOI wake/resume", "claim vs session evidence", "S1 indexer escalate-S2", "Harbor on/off routing", "fail-open vs fail-closed wake vs CI gate", "encoder vs Jev computer-use backend", "hybrid local decide + remote fill", "DONE vs verified success", "stdout prune vs session compaction", "OpenCode jev-pruner vs Claude jev-pruner", "zen-chat vs jev-zen Noul", "hard envelope then Noul prune", "Cua-S1 vs TypeSafe Jev", "plan vs execute dry-run", "specialist computer-use vs general agent", "local drop-in vs stub scorer", "route vs memory", "when does it hold / extractable from state", "decision model vs constrained LLM", "dual-process S1/S2", "combinatorial grid vs extractive", "uncalibrated local likelihoods", "decision-native RAG", "classify-first / read selectively", "living applied-mappings atlas / class patterns", "silence as safer / draft-gate heartbeat", "robotics text-state vs pixels", "verbatim ledger vs summary", "judgment as language primitive", "Stagehand extract pick-and-copy", "harness observe-score-act vs demo loop", "public judgment wall / six parallel questions", "meaning-search without embeddings", "attention ≠ correctness", "skills→oxlint / AST prove ∩ remainder", "session-sticky first-prompt routing", "measured RAG rerank vs generative rerank", "capability kernel / secrets never in the agent", "Jev is SENSOR not policy", "type-safe ≠ correct", "typed control plane around DSPy", "native vs verbalized confidence", "engine owns truth / Jev owns judgment", "human-confirmed kill gate", "train specialist vs few-shot hosted", "decide→policy→LLM leftover", "Noul 0.5 cannot-tell never rounded", "calibration ≠ sortable / ORDER BY", "pairwise inversion / Score ordinality / two-decimal ties", "wire-compat GLiFormer /v1/systemone", "class-backend economics", "loopback gateway hosted + local", "do not distill Jev as teacher", "active-learning triage", "evidence-packet explorer", "meaning-grep AND/OR/NOT", "closed-vote-only / no planner LLM", "Jev vs PCD Harbor", "PCD O(1) ≠ Noul", "host-owned handlers × System One", "OMP/pi fail-open gate", "permission vs probability / operator owns thresholds", "judgment ≠ permission / Jev never grants access", "eval integrity / instrument not score", "constrained optimizer + S1 features / never sole hot-path gate", "privilege ≠ verdict / effect contracts not tokens", "attention filter / VOI for human review / never blocks / never green unless sure", "measurement owns endorsement / evidence-gated question packs", "Jev supplies evidence / code owns authority", "ranking ≠ calibration / never hard-threshold raw p as frequency", "hot-click CU / indexed element table", "Jev judges relevance / code decides structure", "local rules first then remainder / never auto-train on own hides", "combinators / System One as control plane", "receipts not leaderboard / type-safe ≠ correct jaggedness", "VOI over skill library / skillranker abstention", "OOD calibration / AUC ≠ ECE", "Jev vs thinking-budget small models", "turnstile / replayable evidence≠authority", "MLX one-pass schema→JSON / Apple Silicon replica economics", "memory leases ended by new evidence", "never confidently wrong / TLA+ compose / escalate instead of hard-gate", "no seal no advance / coverage ledger / mint ≠ product brain", "skill-broker sibling / judgment ≠ permission", "sureness bands / max_prob is generous", "JevBench / calibration not in Main Score", "CI typed gate before expensive review", "Codex MCP host adapter", "judgment as attention redirect / jev-preflight", "compress-before-first-send / dizk jev-lens", "tools≠use / SessionStart over hoping", "observational memory / pi-om keep-kind", "open-Jev class / openvons / JevPick", "physical-world System One / HA-Jev / not for locks", "judgment outside the store / jevql", "landed-script trust / headless≠auto-approve", "digital-design combinators / extended five", "VOI cache admission / same-intent skip LLM", "BM25 vs Jev skill routing Harbor harness", "zeroshot vs BERT / contamination DiD", "typed escalate continue abort baton / inverted loop", "worth-your-attention VOI / ThinkyMiner Winnow", "Jev WHETHER Python HOW LLM WHAT", "conflict vs ignorance / named Choice escape", "Playwright executes Jev chooses", "OpenJev /v1/decide not drop-in", "SemIf wire-compat runoff; SemIf rename densify / MLX backend / 5.21× systems≠semantic / Softmax ≠ Noul (`notes.md` §117)", "decision-as-memory flywheel", "record/replay CI / jevassert", "failure-finding arena / jevarena ≠ jev-arena", "BBQ not a bias cert", "decider≠executor", "sentence-as-rule lint / jevlint", "sentence-as-rule lint / jev-lint is jevlint rename", "VOI hunk prune", "whole-repo intent VERIFIED/VIOLATION/UNKNOWN", "GLiNER2 spec ≠ replica", "open replica substrates / grande / laya-jolt / JEV-CPU", "ONNX local-jev not equivalent", "persist constraints across compaction / pi-heed", "calibration+cost first-class gates", "Harbor-shaped Jev vs SGR LLM-as-judge / jev-judge-bench ≠ jevarena ≠ jevbench", "hand no-text steps / jev-use / Vercel drops confidence", "Pi System-One control plane / pi-jev-control", "never free-generates / jev-gpt tree of Choices", "OpenRouter recipe atlas / samples not benches", "personal history feed / jevfeed / no social graph", "competing NAR claims / dual-channel ECE / openJev-verdict ≠ OpenJev", "empty compaction-proxy skip / IPECTER", "throughput ≠ latency / like-for-like ECE", "1-token logprob endpoint ≠ Noul / coverage ≠ correctness", "open replica engine / jevinf", "unofficial Elixir SDK ≠ OTP peer", "jevex n=16 files-to-read VOI", "commit pre-review attention≠verdict / middle band", "Hermes plugin is Agnes not TypeSafe", "pi-jev-compact ≠ pi-jev-compaction", "empty Codex-proxy skip / IPECTER runway", "decision-native inbox / mailordinal", "unofficial jev-cli not ready / ≠ jevql", "laya-multilingual / English checkpoint confident-wrong OOD", "schema-scorer peaked ranking ≠ calibration", "HF 401 / GitHub 404 Hub-only", "productized System One HTTP / classifier.dev", "escalate-under-threshold / smart tier / multi-label ignores", "silent FALLBACK / granite 0.546 vs advertised 0.800", "vs_jev tracked JSON / read eval/README", "choxos/jev-reviewer ≠ egma-ai / systematic-review pointer", "two-pass Choice+Noul evidence extraction", "not-found is an answer", "human check as productized judgment", "githubnext/localjev ≠ kunchenguid/local-jev", "wire-compat ≠ logit-equiv / prompted JSON ≠ structured read", "self-reported probs / entropy confidence", "GitHub Next local /v1/systemone", "LM Studio runner gap / structured-read primitives", "NandhaKishorM/laya packaging ≠ Hub-only / Router script-before-p", "post-T ECE ≠ raw ECE / Banking77 token-budget", "0.85 still soft / not TypeSafe drop-in", "external census ≠ scored bake-off", "GLiNER2+routers class-boundary", "incomplete openjev census vs watch", "Harbor honesty watch / silent fallback", "JevBench v1.2 geometric mean / cal ON rank / weight sensitivity", "option-order 72→21 / instruction models in the class table", "self-host latency ×2 assumption / est. costs", "Laya absent is a gap not a named exclusion", "Qwen3.8 27B ≠ Archer", "hourly already-folded watch / apply-the-five / skip thin noise", "hard-gate Noul as PR/quality gate is soundness theater", "S1 never stalls waiting / S2 one-use advisory", "Local controller ≠ githubnext/localjev", "purple telemetry = consumed not arrived", "seed = geometry not async replay", "20% starting gate still soft", "no pixels to either provider", "OCR+AX observe-score-act / typesafe-computer-use", "never send screenshot to frontier for the decision", "overlapping CU options = false low confidence", "split kind/item/site", "155× one-screenshot ≠ Harbor taskset", "decision ≠ answer-reader capture", "ASR observe-score-act / jev-voice-browser", "partial-speech VOI / free-text waits", "spoken confirm ≠ hard auth", "numbered overlay without another model", "wrap-as-execution / AgentGhost ALLOW ASK DENY", "rules first then Jev remainder / ASK throws / fail-closed", "reddpy/AgentGhost ≠ jwen5419807/agentghost ≠ vventirozos", "JP genre atlas / studio_yebisu / stars ephemeral ≠ eval", "Jev Clearly Explained / akshay_pachaar / LLM hammer", "schema-safe ≠ correct / 200× 400× TypeSafe ceiling", "questions-as-code / shadow first / not a TypeSafe how-to", "proposition ≠ embedding / contrast-set", "boolean composition of soft Nouls / AND OR NOT", "uehaj/jev-semgrep ≠ semgrep.dev", "meaning-grep dedicated fold / not a gate", "decision-validated UI / Jev never authors text", "decision-as-assert / jevtest ambiguous band", "typed decisions drive UI / jev2ui", "hybrid S1 closed verb menu / anima3", "pointer-not-generator search / JevFind", "jev-frontier-bench ≠ frontier-100", "product bakeoff ≠ architecture duel / GLiClass", "four engines same questions / majority floor", "authorship named escape / not evidence", "ha-switchboard HA remains execution", "n8n classify/route/score / Low Confidence", "fast-jev-compaction-pi ≠ pi-jev-compact ≠ pi-jev-compaction", "jevloop full-distribution optimizer / no LLM in the loop", "laya-vision SmolVLM / score untrained", "Cerebellum-2B /v1/decide ≠ TypeSafe / wire-compat vs agent-routing", "laya-grounded not drop-in / Platt not temperature", "GestaltLabs/Jeff-1 ≠ logan-markewich/jeff / acc vs ECE n=9730", "stanley-code empty findings ≠ approval / human promote", "findme ≠ JevFind / NL memory beam-search FS", "jevsubrouter price workers not conversation / counts ≠ dollars", "feelings .feels() default 0.5 is Noul-0.5-never-rounded / ≠ hunch ≠ Probably", "apa-agent-harness ≠ AntonioCoppe/jev-harness / unpublished npm", "grok-bot-jev skill cannot force a bot that ignores it / A/B proxies not tokens", "Essentiel-Jev never authority / human every action", "enzo-mcp independently falsifiable claims / ≠ jev-sift", "pigeonhole OTHER skip / decision-as-filing", "jev-reliability Nothing about accuracy", "clduab11/jev-test ≠ realZachi/jevtest / Nothing runs yet", "jev-rag-benchmark Jev wins is not an assumption", "dairui1/jev-lab ≠ BrendanH18/jev-lab", "jevmail gmail.readonly / mailjay archive/trash", "ZHUBoer/ego-jev reserved __none__", "runWorkflow completed ≠ success", "jsort scores are relative", "Noul not Choice for scale", "groundedness-judge-bench native vs schema-guided", "implicit_true included in yes", "jev_playground 0 promotions", "routing-backtest 0.0447%", "yuyang2230/jev-agent-skill jev-1.13-free", "jev-techstack-classifier stack_config.json", "s1_ruby collapse late", "undecided? abstain", "2389-research/judgement license null", "confidence ≠ winner p", "typesafeai-sdk-community not a new species", "tpellet/hunch exit 3", "never-execute list", "jev-file-search scores not calibrated accuracy", "jev-linkmap Jev never sees S2 prose", "muhammedilyasy/jev-mail metadata only", "tidy none-of-folders stay", "tab-bouncer pinned/audio/current never closed", "lkclean Show fail-open", "jev-yt-time-saver Show anyway", "ORIGIN pause-if-no-Jev", "validResponse sums-to-1", "jev-crawlers risk bands never raw boolean", "jevbrain AUTO_ACT is not a Noul", "judgekit YAML classify/score/route/verify", "typed-judge-kit verdict-in-code", "alsoleg89/decide packing VOI", "0.8 ≠ 80% accuracy", "Jev-Calibration Platt ECE 0.117→0.052", "jev-calibration-arena never acts", "ctmx/openrouter-jev-mcp Decision-as-Plugin", "FrancoisChastel/jev-code ≠ npm jev-code", "claudecode-jev-marketplace fail-open not hot path", "pedroknigge/mcp_jev packs not ask_jev", "cyrusasco/typesafe-mcp noul deadband 0.35–0.65", "codaaiteam/jev-skill jevtypesafeai.com ≠ TypeSafe", "hermes-switchyard ≠ hermes-jev-router ≠ hermes-plugin-jev", "nanoprune 2.8MB ECE 2.58%", "smartdio/jev-browser-agent ≠ ZHUBoer/ego-jev", "Dakai/omp-jev-web DONE ≠ proof", "hari007sh/jev ≠ dannote/jev", "0thernet/system-one-skills deterministic verify", "typed-gate band [0.40,0.60] is refusal", "pi-jev-gate fail-closed; choice is the verdict", "Foq ~25ms/2.2GB local", "rev prefill-only + HF jev-0.5b", "robfrase/jev planning memo", "typesafe_agent_gates 27/27 / 31/31", "EpicEric/safe-sh static remainder", "pastepilot Confirm before act", "Jev-Reranker live Jev not yet measured", "sessionwise opt-in relevance", "jev-search pointer sieve", "400ms Salesforce WebMCP", "typesafe-scheduler-diagnostics advisory", "droidjev screenshot-free", "Tewoto1 jevcu planner still writes", "ha-conversation-jev Jev→Grok", "dsh-jev can only gate", "jev-classification-benchmark specified not run", "jev-luna-pagerduty p≥0.50", "meldltd/meldecision laya-go ONNX", "laya-doom never pixels", "logixism/laya-api empty README", "akpsahan/laya ≠ Archer", "choxos/jevchess engine owns truth", "jev-drive sim not AV", "story-arc Jev never authors", "jev-hs-assistant HS6", "golergka/jev-plays-starcraft-2 UI-verified ≠ API Victory", "awesome-jev-use-cases catalog", "Nibir1/typesafe-go ≠ official", "fingerprint after redact", "recall vs decide", "publish fingerprints+answers", "CI replay as Harbor cousin", "Cache hit ≠ correctness", "hyperspaceai/jevcache ≠ kushals256/jevcache", "human labels only", "score never auto-accepts", "production capture flywheel", "sutro-sh/jev-align ≠ caiovicentino/jev-align", "guidance ≠ hook", "catalysts ≠ summaries", "compile-time System One", "unofficial ≠ TypeSafe", "format_version modernbert-jev/1", "Argos1111/jev_local ≠ us/jev-local ≠ kunchenguid/local-jev", "LFM default ≠ ModernBERT backend", "Nemotron ≠ TypeSafe Jev", "not a calibrated replacement", "djev-dev complements djev-spark", "images as Choice options", "Laya essay numbers *theirs*", "Router/OOD confidence", "hosted bootstrap ≠ silent TypeSafe", "difficulty + policy thresholds + JSONL trace", "jev-codex-pilot model + reasoning depth", "keep/shadow/hybrid/reject", "quarry evidence projection", "Frank-ZY-Dou/awesome-jev robotics/3D/control", "one-dollar-tahoe TypeSafe Jev defense eval", "jevguard calibrator/cache/escape", "jev-ci-selector CI shadow mode", "llama-jev llama.cpp replica", "petercr/jev-orchestrator ≠ FleeexCorp/jev-orchestrator", "seb4ez/jevguard ≠ AseemPrasad/JevGuard ≠ pablozr/JevGuard", "webNeat/llama-jev ≠ WiktorB2004/llama-index-jev", "OpenCode jev-pruner context sieve", "observe→score-candidates→prune", "jev-zen / jev-1.13-free", "zen-chat ≠ Noul", "fail-open original", "keepScore >0.1 floor", "host port of tamaratran/jev-pruner", "indiejoseph/opencode-jev-pruner ≠ nrdz-labs/fast-jev-opencode", "jev-webagent-bench empty stub", "Kiln-AI/jev_jsonschema noul_threshold 0.5", "NSStudent/JevSwiftSDK unofficial", "GLiNER2 native Apple path", "unofficial Swift/Core ML GLiNER 2.5-small", "entity spans + confidence", "not Choice/Score/Noul", "not TypeSafe", "label descriptions as schema", "on-device ANE economics", "honesty locks", "shershah1024/gliner-native-runtime ≠ Fastino", "≠ gliner25-compaction ≠ gliner2-ultrafast ≠ Eran-BA/Jev_from_GLiNER2 ≠ NSStudent/JevSwiftSDK ≠ jevmlx", "default threshold 0.1 still soft", "soft Noul ≠ hard safety", "Decision Graph Protocol frame→assess→commit", "app retains permissions/effects", "Jev-first assessor-neutral", "guarded commit / receipt/next frame", "assessment batching", "hard-gating DGP as safety theater", "numerous-com/dgp ≠ TypeSafe official", "jegrep calibrated path+range Nouls", "no embeddings/index/daemon", "~$0.01–0.03 typical", "agent --json", "can1357/jegrep ≠ Bentlybro/jevgrep ≠ uehaj/jev-semgrep", "Archer-arch fidelity", "kev family OOD 0.76–0.77 vs Jev 0.86", "block-causal isolation", "pointer/readout CE-trained", "/v1/systemone drop-in", "replica honesty", "cost-sensitive decision theory × System One probabilities → control flow", "thresholds derived from costs not hard-coded", "YES / NO / UNSURE from cost_false_yes / cost_false_no / cost_human", "auto-batching same-object questions", "Kungie/gut ≠ tpellet/hunch ≠ carldaws/hunch", "judgment vs generation", "deterministic execution after probabilistic judgment", "exactly one app-owned callback", "explicit uncertain branch", "Illusion47586/judge ≠ lexingtonhibiki/judgekit ≠ Ascurse/typed-judge-kit", "variable-N option scoring as the trainable object", "dynamic candidate bags not fixed label sets", "zwliJay/jev-forge ≠ NanoJev", "open replica economics / latency vs closed Jev", "NAR local drop-in", "wfzyx/von late-catch HIGH", "competing NAR claims / replica honesty", "typed judgments vs chat judges on guardrailing", "ishaannk/llm-vs-jev cross-note only", "deeper integrity fold is rh-guard", "nothing wins outright", "can be argued out of guarding"", "Jev IS the if-statement", "judgments/probabilities drive branches", "text model only writes prose", "interpreter owns variables/loops/budgets/replay", "otherwise maybe / confidence gate", "chaos samples after the gate", "southpolesteve/probably ≠ carldaws/hunch ≠ feelings ≠ Kungie/gut ≠ Illusion47586/judge ≠ tidymodels/probably", "133★ / forks 10 live", "build calibrated classifiers from human feedback", "retrieve by relevance not resemblance", "one calibrated yes/no per memory in one request", "pointer mode 17/18 19/20 *theirs*", "embedding resemblance misses the allergy", "samdotmak/jev-recall ≠ jev-search ≠ jev-sift ≠ carryforward ≠ chopratejas/invalidate", "memory leases ended by new evidence", "six Nouls then fixed rules in code", "0 of 157 false invalidations", "questions/plans/directives are not evidence", "unsure → review queue", "host keeps the store", "name↔body / comment truth / test-claims", "mizchi/jev-lint is mizchi/jevlint rename", "no shipped rule has severity error", "~1 in 5 findings wrong *theirs*", "mizchi/jev-lint ≠ huntedman/JevLint ≠ MichitoSugawara/jev-lint", "JSON Schema → typed JSON via Jev", "noul_threshold 0.5 decoder not a proof", "IncompatibleSchemaError lists every bad property", "on-device Laya CoreML ANE", "~5 ms P50 short decisions", "189/189 FP16 checkpoint parity", "10× not achieved", "mizorewww/laya-coreml ≠ gliner-native-runtime ≠ jevmlx ≠ NandhaKishorM/laya", "softmax over allowed tokens ≠ Noul", "question-first cache", "Micha0827/snapjudge ≠ githubnext/localjev ≠ jevmlx ≠ cendress/SnapJudge", "Jev-first Pi agent loop", "slow-LLM fallback", "explicit action menu / CandidateSource unimplemented", "62 tests wiring not quality", "direwolfiy/JevPi ≠ standardagents/jevpilot ≠ pi-jev-control", "resume-screening bias audit methodology", "name×resume factorial independent Nouls", "callback determined by resume quality", "mean-probability name gaps operationally negligible", "natemoo-re/bias-bench ≠ BBQ", "Plan/PRD panel → code-owned pass|review|block", "cheerleading out of scope", "austindixson/planalyzer ≠ single-goodness Noul", "cost-aware multi-model routing/escalation", "decide vs do", "successful-task cost", "cannacre8ive/switchboard-ai ≠ ha-switchboard ≠ hermes-switchyard", "frozen-protocol zero-shot bench", "TypeSafe Jev vs PrismNLI vs Laya", "contamination caveat", "elcronos/jev-vs-open-decision-models ≠ JevBench ≠ DMB", "context-window admission control", "VOI gate which tokens are worth the expensive model", "fail polarity per lens", "on small inputs lenses lose money", "cvsgireesh/jevusher ≠ jev-sift ≠ winnow", "typed decision control plane", "receipt ≠ authorization", "historical-v0 zero retained cases", "MokiMeow/jev-fabric ≠ jev-forge ≠ dgp", "live 15-dim typed rubric re-score per pause", "scoring economics exemplar", "OpenJev/Codiv ≠ TypeSafe hosted", "jose-troche/live-rubric ~$0.000004 desc / ~$0.000006 README", "adversarial pre-registered Jev eval", "28 predictions before data", "123,805 requests", "confidence does not track ignorance", "polite injection 65% / crude 0%", "willkelly/jev-evaluation ≠ jevals ≠ jev-baselines-eval", "provider-neutral Elixir/BEAM Noul/Choice/Score SDK", "class infrastructure", "nshkrdotcom/system_one_sdk ≠ typesafe_sdk ≠ dannote/jev", "question-linting of Jev questions themselves", "nine jaggedness rules, no API key, no labelled data", "static lint ≠ measured separation", "yodablocks/jevq ≠ tenbin ≠ JevLint ≠ commitjev", "open-weights Laya as class exemplar (binding)", "Nx/Bumblebee runtime", "host chooses backend", "ChristianAlexander/laya_ex ≠ system_one_sdk ≠ dannote/jev ≠ NandhaKishorM/laya", "on-chain/edge Laya deploy", "parity_verified stays false", "model output never grants Tx", "humandebri/IC-Laya ≠ laya_ex", "auditable weekend replica", "Jev outputs never used for training", "soft human-vote distributions", "unpaired 0.577 vs 0.727", "agilabs-ai/jev48 ≠ JevBench ≠ Mapika/decider", "adversarial dual-judge / framing attack surface", "comparative framing is the usable judgment", "prior injection crowds out evidence", "copyleftdev/ember ≠ ember.js", "Laya specialist fine-tune pipeline", "training still GPU-pending", "PIXELZX0/XERON ≠ convaiinnovations/laya", "Hub Laya replica drop", "daliborsb/laya ≠ convaiinnovations/laya ≠ NandhaKishorM/laya", "System One student distillation corpus", "gold is programmatic", "teacher is closed-API clone", "do not distill Jev as teacher of record", "MagaBitmex/jev-4b-distill-data ≠ missing student checkpoint", "non-LLM VIN System One", "planning depth not chat", "lewislululu/jevon ≠ douglance/jevon", "source-bound evidence checks", "local quote mismatch needs no API", "exit 0 ≠ claim truth", "WaynezProg/jev-kit ≠ jonathanavis96/jev-kit (Airlock) ≠ jev-use ≠ jev-mcp", "independent System One evidence catalog", "scores not one leaderboard", "no external record currently reproduced", "TokenTrim no-Jev matched hybrid 62.4%", "reachjalil/system-one-bench ≠ mallahyari/system-one-benchmark", "21 tasks · 134 items · 208 questions", "scenes from public GitHub contracts, not production logs", "SivletLabs/jev-eval ≠ willkelly/jev-evaluation ≠ 4esv/jev-eval ≠ xxkuboxx/jev-eval ≠ onlyoneaman/jev-eval ≠ dayhaysoos/jevals", "option isolation (sibling-blind)", "permutation-equivariant", "Hub OWNER not published", "nafisazizir/hev ≠ jaredpalmer/kev", "frozen local LLM logits, no trained decision head", "residual-head 9,222-param decreased 73/96→67/96", "confidence = 1−normalized entropy, not P(correct)", "yuki-oshio/mini-jev ≠ r-ms/mini-jev", "Jev classifier as autoregressive next-token predictor", "ChatJev-style soundness theater", "erik-dunteman/ChatJev ≠ dannote/jev ≠ jev-gpt", "calibrated decision head × AlphaProof value head", "implementation-layer isomorphism, semantic difference", "timeout = censoring", "do not launder Noul as proof", "parallel rank-prediction vs serial selection", "independent questions can conflict", "zzzzzec/jevsort ≠ keltokhy/jsort", "curated open System One ecosystem catalog", "rupeshpoojary9/awesome-open-system-one ≠ AnotiaWang/awesome-jev", "arXiv paper radar with Jev relevance scoring", "ranking ≠ calibration / 0.5 still soft", "fail-open failed evals not marked seen", "train calibrated ~27M from scratch", "typed Q→prob dist / one forward pass / no LLM decode", "hyusi2003/MiniSystemOne ≠ Colvin0315/MiniSystemOne", "description-only stub / size 5", "ESCI hard probe fails four of six", "jev_bool ECE 0.242 inversion 0.255", "do not re-fold §60 six-gates as new", "jobbyjev one-request-per-company from batch-size result", "find/design/evaluate TypeSafe Jev decision loops", "karanb192/jev-architect ≠ samtay32/jev-system-architect", "Jairik/jev-distiller size 1", "distill-Jev UI stub / do not distill Jev as teacher of record", "post-launch scored use-case map / Jev self-scores then human curation", "licensedsaucer9-web/jev-opportunities", "Jev-inize a use case into classifier/router", "gavinHuang/jevinize → simple-jev not TypeSafe", "featherless-ai/simple-jev", "compare saved decisions / same label can still change the branch", "VihaanAgarwal/jev-diff ≠ Saik0s/diffusiongemma-jev-macos", "not tested with a live Jev API key", "constrained logprob + temp/Platt ≠ Noul", "OpenJevPro pastes openjev-sglang JevBench as own", "zhangcy122/OpenJevPro ≠ IamBusy/OpenJev ≠ ekzhang/openjev-sglang", "PolyForm Noncommercial", "SmolLM-135M / sub-70ms / 0 output tokens", "demo P(True) 0.5052 / Choice conf 0.2872 / Score conf 0.0055", "README claims MIT / GitHub license null / no LICENSE file", "patelvishwa112/jev-system-one-rlcd ≠ arnabgho/rlcd-lite ≠ blackwood-rlcd", "source-backed Awesome Jev radar / 306+ commit-pinned", "logicrw/awesome-jev-projects ≠ AnotiaWang/awesome-jev ≠ yibie/awesome-jev ≠ cobanov/awesome-jev ≠ rupeshpoojary9/awesome-open-system-one", "auto GitHub sync / Issue-only submissions", "hashed n-gram encoder / rival-aware attention", "olanotolu/jevbetter vs jevlike starter", "synthetic hard menus top-1 0.916 vs 0.873 / ECE 0.0182 vs 0.0367 / 40 vs 4608 menus/sec", "shuffled-context control 0.335", "Turn any open LLM into System-One Jev", "uspraveen/Jevify ≠ Mintzs/jevify ≠ gulagala001/jevify", "Jevify-any-LLM architecture probe", "description-only stub / size 0", "Train encoder-only calibrated decision models from a task sentence", "Exu is a toolkit, not a method", "strictly proper scoring rule", "Pre-alpha", "Ruivalim/exu-base", "scratch-trained calibrated decision model", "typed Q → probability dists", "Colvin0315/MiniSystemOne ≠ hyusi2003/MiniSystemOne", "no published weights download URL", "90.5 seconds / 29.2% pipeline evidence", "p_i/p_j independent of other candidates", "Recipe for calibrated decision models — small model out", "init → synth → train → eval → serve", "91.1 % / ECE 0.022 *theirs*", "Jev zero-shot 75.1", "scienthoon/luce", "Put Jev's three headline claims on trial", "0.5B local GPU", "46x speedup / accuracy identical", "ECE 0.624 sentiment catastrophe", "bigger model worse calibration", "RichardoMrMu/jev-mini ≠ yuki-oshio/mini-jev ≠ r-ms/mini-jev", "System-1 decision engine for local LLMs", "structured choices only", "JSON parse of generated text ≠ Noul", "TypefAI JEV / Journal Entry Voucher", "tapsin/jev-local ≠ us/jev-local ≠ Argos1111/jev_local", "Jev 1.13 reward-model eval across 8 benchmark tracks", "40,940 examples / 0 API errors", "RewardBench v1 92.58%", "Precise IF 50.63%", "goya4140/jev-reward-model-evaluation", "Scaffolding in progress", "Jev vs LLM support-ticket routing", "static + live decision bench", "TypeSafe's own published benchmark", "illustrative simulations, not live API calls", "JevBench v1 — smart/cheap/fast/reliable", "I/C/S/K 25% geometric mean", "classifier.dev fast tier 84.8 is Jev behind its own API", "do not re-fold §78 v1.2 board as new", "Laya (421M) 70.1 now on board", "Zero-shot/few-shot LLM routing", "hard budget filter before Jev", "Jev never asked to perform budget arithmetic", "Jev judges the next state, XState enforces transitions", "simulation uses synthetic keyword fixtures", "catalog gravity", "v-modal/awesome-jev-tools", "★339 live REST", "curation is not endorsement", "crawler-maintained directory", "Daily GitHub + npm sweep, human-merged", "RadRebelSam/awesome-jev ≠ AnotiaWang ≠ yibie ≠ cobanov ≠ logicrw ≠ v-modal", "HF peft SPLADE/BGE reranker", "rdxtremity/jev-reranking ≠ carlaiau/jev-reranking", "query-side encoders, not a Jev replica", "ONNX System One Qwen3.5-4B scorer", "source:pngwn/system-one-qwen3.5-4b-scorer", "CC-BY-NC-4.0", "temperature 1.75", "transformers.js AutoModel cannot load this graph", "Consistency benchmark Space", "This Space contains no benchmark result yet", "12-case plumbing fixture", "Benchmark-driven Jev router and judge", "cheap alone is not success", "Jev does not write, sum prices, or claim accuracy %", "Sol 94.2 / Luna 83.9 / Jev path 89.7", "19.2% Sol / 62.3% cost save / 4.5pp miss of 2pp non-inferiority", "p50 latency worse than Sol due to routing overhead", "erendikmenn/jev-llm-router-benchmark ≠ jev-rag-benchmark ≠ ryantsai/jev-llm-router", "Express + node:sqlite", "mock and Jev decision engines", "previous_ticket_count >= 3 is code", "MIN_CONFIDENCE 0.6 still soft", "substring false positives", "aesaganda/jev-ticket-router ≠ SarathChandraBellam/jev-vs-llm-ticket-router", "Universal Figure & Diagram Router", "confidence ≥ 0.85 hard-gate is theater", "generative AI banned from scientific plots", "six visual branches", "hoangngochuong24947-gif/jev-figure-router", "human-labeled (state, question, label)", "166,054 rows / 22 configs", "soft_label for human uncertainty", "Praveenrajus/jev-bench ≠ fstandhartinger/jevbench", "ternary bonsai System One GGUF", "openjev's mechanism, Bonsai's weights", "Hub does not ship weights", "100/100 easy T/F is not Harbor", "label_mass ≠ correctness", "stock llama.cpp Q2_0 silently gibberish", "NicolaiMTLassen/open-bonzi-jev ≠ NicolaiLassen", "transformers.js DeBERTa ONNX", "source:com-kotobalabs/open-jev-deberta-v3-large", "temperature 1.05", "AutoModel from_pretrained works", "onnx-community/open-jev-deberta-v3-large-ONNX ≠ system-one-qwen3.5-4b-scorer-ONNX", "107★ densify", "GH 151M vs README 149.6M", "PR #1 now closed unmerged", "do not re-fold §71 claim-audit as a beat", "typed decisions, RLCD, confidence-gated routing", "structured ≠ correct", "mock not live API", "26 tests", "wjdjdakf17/jev-study ≠ baekenough/jev-study", "bonzi-27b-v2 / ternary-8b / 27b-v1 GGUF family densify", "WANLI-256 74.6% / 65.2% / 71.1% *theirs*", "Bonsai 1 27B Q1_0 runs on stock llama.cpp", "ternary still needs PrismML fork", "hf:heman10x/openJev-verdict-2.0 twin tokenizer-only", "OpenJev Vision image classification + uncertainty", "CLEVR-4 held-out joint 0%", "hfdataset:IamBusy/OpenJev-Vision-Research-v0.1 12,832", "294,912 derived targets not independent samples", "Laya multilingual ONNX WebGPU typed-decisions port", "63/63 selected answers / 5.1e-4 CPU / 1.2e-2 WebGPU", "UpHash-Network/mini-jev is yuki-oshio transfer", "jev-injection-bench 11,900 labelled prompts", "Jev best ranking / Haiku better ECE 0.021 vs 0.058", "0.5–0.9 band is where Jev's numbers do not mean what they say", "Prompt wording moves panic 28%", "manojlds/jev-dspy-bench ≠ dspachos/jev-dspy ≠ jmanhype/jev-dspy-lab", "Jev agreement is similarity, never ground truth", "no aggregate quality grade or merge gate", "AbstentionBench-on-Jev rank 1 of 20 vs 2025 field", "question-asymmetry", "forward-looking 0.465 never extreme", "openkev calibration layer not a runtime", "ECE vs coverage independent", "select_threshold returns inf", "escalation catches uncertainty not ignorance", "misakaikato/openkev ≠ jaredpalmer/kev", "pdf-race Docling→Jev vs Gemini", "parser owns the wall clock", "12/12 tie is a tie", "titles selected not generated", "flopcheck 16 calibrated tweet judgments", "mechanical tells in code", "ZeroX-01/jev-atlas ≠ Zaious/jev-capability-atlas ≠ gorock007/jev-atlas", "Laya calibration lab Gradio MCP", "T never changes argmax", "confidence ≠ top-label p", "easy probe set refused", "40–48 rows too small to ship T", "Gemma-4 26B-A4B jevify classification+calibration", "LoRA adapter twin not independent eval", "Gemma-4 E4B jevify", "E4B LoRA stub card", "kushalpatil/jevify-gemma4 ≠ Mintzs/jevify ≠ gulagala001/jevify ≠ uspraveen/Jevify", "GH kushalpatil07/jevify 404", "PAWS 0.580/ece 0.288 is the weak cell", "smaller E4B slightly better OOD ECE than 26B-A4B", "Hub jevify merged LoRA ships weights", "bonzi Bonsai-8B v1 GGUF densify", "Bonsai-1.7B v1", "Bonsai-4B v1", "WANLI-256 64.5% / 60.2% / 52.0% *theirs*", "rank #4 / #5 / #6 of 6", "JulesHuisman/jev-eval scaffolding / README SHA c356a584 (was empty e69de29b)", "JulesHuisman/jev-eval ≠ SivletLabs/jev-eval ≠ willkelly/jev-evaluation ≠ 4esv ≠ xxkuboxx ≠ onlyoneaman ≠ dayhaysoos/jevals", "7 bands 6/10 vs 40 bands 0/10", "source receipts + confidence slider re-policy without re-inference", "32/32 synthetic is smoke not production", "classify HF datasets across typed semantic dimensions", "roadus2 watch misspelling; lock roadius2/ultra_laya", "ultra_laya REVIEW defects", "default branch claude/laya-jev-review-gg5ppo", "XNLI EN 88.3% ECE 0.032 → RU 77.3% ECE 0.096", "Δ −11.0 pp [−14.2,−7.8]; ECE +0.063", "MASSIVE no detectable difference at n=600", "confidence is function of p_max (r=1.000)", "pointer-not-generator 400 human-authored responses", "proposed ≠ authorized", "FewRel 160: Jev 85.0% vs lexical 13.125%", "gated 100% (95/95) coverage 59.375%", "J++ composable semantic computation language", "judge-jev 0.5 still soft", "947 repos scored; A 273 / B 302 / C 372", "LLM rubric ≠ benches", "No benchmark winner is claimed", "phishing: naive 62.6% vs regex 91.8%; 5-atomic + LR 95.0% *theirs*", "AITuber tension ±15", "README npm global; repo is Rust", "git-confess code owns counting/blame/ratio", "httpx exhibit 11% (13/119) *theirs*", "90d trend +12.40% vs random +12.75% vs BH +41.71%", "5m win rate 25%", "Awesomejev 656 entries / 38,160 stars", "tracker likes 64 (+4) lastModified UNCHANGED", "Laya present; Blackwood ABSENT; Archer still promised_not_landed", "Blackwood tracker ABSENT; likes 2 gated manual", "r = c - p_a", "ECE 0.021; acc 0.807 vs warmup 0.746", "Independent primitive", "11.57s vs 54.10s · 4.67× · 120/128 *theirs*", "default path is pretrained Gemma probs not trained RLCD head", "GH Meanblock 404; lock leesk212/JEV-CPU", "softmax over letter slots ≠ Noul", "WANLI 0.741 vs openjev v2 0.77 *theirs*", "3-way NLI ≠ Noul", "priority 0.464 = majority floor", "banking77 contaminated", "raw margins not probabilities", "do not distill Jev as teacher of record (they distilled Haiku)", "“0.9 is not one number”", "ranking ≠ calibration", "banking77 0.8–0.9 stated 0.86 actual 0.73 over-confident *theirs*", "≠ Praveenrajus/jev-bench ≠ fstandhartinger/jevbench", "$0.0000153–$0.0000226 vs circulating $0.0004 (~20×)", "Score is 0..n-1 expectation not 0–1", "Noul has no confidence field", "TCP floor 198.8 ms", "type reliability is not a reason to choose Jev (json_schema 5/5)", "gateway tax not one number", "Function-only 5/8 vs hybrid 8/8", "4/8 without Jev", "8 designed cases not conversion lift", "200-row pilot Jev 86.5% 173/200 vs Gemini Flash-Lite 86.0% 172/200 vs Pro 87.0% 174/200 *theirs*", "not a ranking", "情緒測謊器", "8-example Jev vs GPT-5.6 Sol ~64× cost 5.4× latency *theirs*", "synthetic; no inference", "≠ JevBench v1.2 §78", "Judged 3317 / listed 2560", "Jev judges, code applies policy", "APA “microsecond policy / zero hallucination” overclaim", "Client-side quiz; pointer from held docs; scanned-PDF warn", "Jev judges / agent reasons / user decides", "selecting an option is not permission to implement", "pattern exact, judgement must clear floor", "no matching pattern → no model call", "not a correctness oracle", "Spec vs artifact remainder", "treating 0.85 as 85% / minProbability hard-gate as Harbor", "VERIFY acquires discriminating evidence, never same-pool confidence-only rescoring", "fast/full/max are ceilings not sizes", "Solar writes, Jev chooses NEXT ACTION", "do not reopen or amend PR #23 or #24 or #25 or #26 or #27", , "Calibration is not alpha", "NO CURRENT ALPHA CANDIDATE", "ΔR² approximately +0.00084", "Brier 0.2131387", "ECE 0.0421875", "Adding Jev probability to deterministic volatility improved Brier by only 1.4058e-05", "default 0.5 keeps zero non pinned", "keepResult median 0.14 to 0.17", "keepCall median 0.28 to 0.35", "usable range is about 0.10 to 0.25", "7.8% to 57.9%", "judges results it never sees", "task-finish eval not built yet", "$0.002 per compaction", "slavadubrov/sgr-judge-bench ≠ slavadubrov/jev-judge-bench", "Jev 108/120 $0.083 0.34 s", "Luna SGR 114/120", "paired Jev accuracy-difference intervals include zero", "not evidence of equivalence", "GLM SGR 26/120 93 format failures", "Terra-planned Jev hybrid 55/120", "rule-based by default, optionally Jev-backed", "empty README", "missing key cannot break the experience", "prefill plus exactly one decode", "softmax over A/B/C ≠ Noul", "BBQ 9,053/10,000 (90.53%)", "ECE 0.0890", "Mean confidence 0.9943", "overconfident", "score and noul not implemented", "DGUI 12 rows (was 6)", "INSTRUCT 119 rows likes 2", "encode the state once, decide everything in parallel", "0.740 accuracy against a 0.508 majority", "ECE 0.047", "fine-tune's advantage ends where its 384-token training data does", "jasonkneen/open-jev ≠ pngwn/open-jev", "same sha d41dc3cd", "Space does not call Jev", "recomputes routing from saved probabilities", "200-case Jev 97.0% / 100.0% / 95.0% / MAE 9.22", "synthetic repository benchmark", "Jev evaluations are advisory", "YehuiTang0316/jev-nlgrep ≠ Bentlybro/jevgrep ≠ can1357/jegrep ≠ uehaj/jev-semgrep", "default threshold 0.8 still soft", "40-line windows cannot prove whole function", "token-native sequential start/end Choice", "Gemini/Haiku stubs not configured yet", "handful of hand-written examples, not a benchmark", "Jev judged exactly what it was given", "laguagu/jev-skills ≠ laguagu/jev-evidence-lab ≠ Pleo2/awesome-jev-agent-skills", "contract_passed is not a claim of guaranteed factual truth", "Wilson lower bound 0.85 floor", "fixture mode no savings claim", "SemIf 2207★ (+21 vs §110 2186)", "jevlike 1043★ (+5 vs 1038)", "TypeAR 15★ (+1 vs 14)", "AnotiaWang 97★ (+1 vs 96)", "yibie/awesome-jev 506★ (+16 vs 490)", "Laya likes 822 (was 802)", "tracker likes 64 flat, lastModified UNCHANGED", "do not reopen or amend PR #23/#24/#25/#26/#27/#28", "Heman10x-NGU/Verdict-open-jev ≠ Heman10x-NGU/openJev-verdict-2.0", "TF-IDF + LogReg ECE 0.0207 vs Jev 0.1440", "Verdict-open-jev 48.07% vs Jev 90.80%", "abstention combined recall 10.00%", "p50 35.58 ms", "K=25 (maximum capacity) 72.00%", "0.85 coverage 84.60% selective risk 1.18%", "26.1× faster than standard Qwen JSON generation", "Jevify 90.0% / 167 ms CUDA graphs disabled", "Finding 1: Brier on stated confidence alone is a trap", "grpo_rlcr 0.78 / ECE 0.084", "reliability 0.007 but resolution 0.000", "27 900 schema-driven decisions", "13 600 / 13 600 questions", "candidate mass min 0.99999624", "22 configs · 166,054 rows · 4 calibration-gold", "sha a39eba3f", "Student B MAE 0.148 / Pearson 0.836 / 86.0%", "pngwn/open-jev-laya-bench README 404", "sha 9f69c742 likes 2", "HDFS 0.9933 (745/750) / retain 0.0084", "BGL ERROR/FATAL protection 1.0000", "2,479 / 2,500 HDFS uncertain", "cache hit 0.9648 (2412/2500)", "$0.153936 estimated", "E2 recomputes from saved probabilities", "Space sha eda59e0a", "MASSIVE English 0.783 / Khmer 0.033 / Hindi 0.133", "40–48 rows too small to ship T", "T never changes argmax", "siren2345/jev-single-decode-transformers ≠ siren2345/jev-single-decode", "Split Transformers experiment from llama.cpp runtime", "tanayvasishtha/jev-lab ≠ dairui1/jev-lab ≠ BrendanH18/jev-lab ≠ yibie/laya-jev-lab", "Four experiments stress-testing TypeSafe's Jev: calibration, bundle bias, label bias, and ensembling", "second pass must be $0.00 from cache", "The pages never call Jev", "Gemma 4 31B 77.0% / Jev 1.13.0 61.4% / Laya 322M 0.0%", "restriction state 95.0% against 84.4%", "None of the systems are particularly good at knowing when to stop and ask", "They skip the question and call a tool directly", "100% schema pass", "six-field joint 48.8% vs 72.8%", "ywchiu/jev_benchmark ≠ Running-Dolphins/jev-bench ≠ Praveenrajus/jev-bench", "ACT / REVIEW / FALLBACK", "A provider failure, timeout, malformed output, or missing answer is **not** a policy outcome", "confidence is descriptive provider output, not a substitute for probability", "Quality denominators include only valid scored answers", "an exact halfway tie chooses the lower level", "aiwithenoch/Jev-Skill ≠ simplosophy/jev-skill ≠ laguagu/jev-skills", "The local path does not claim to turn a smaller checkpoint into Jev", "Low support becomes decision: \"review\"", "MIT-0 SPDX NOASSERTION", "current-llm", "结构兼容,不是 Jev 模型能力", "altryne/jevify ≠ Mintzs/jevify ≠ gulagala001/jevify ≠ uspraveen/Jevify", "Find where Jev belongs. Design the questions. Measure the difference", "TypeAR-AI/TypeAR 301 → TypeLLM/TypeLLM", "TypeLLM/TypeLLM 16★", "SemIf 2241★ (+34 vs §111 2207)", "jevlike 1051★ (+8 vs 1043)", "AnotiaWang 98★ (+1 vs 97)", "yibie/awesome-jev 525★ (+19 vs 506)", "Laya likes 864 (was 822)", "tracker likes 67 (+3 vs 64)", "lastModified UNCHANGED `2026-09-20T04:29:16.000Z`", "do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#32", "hysteresis enter/exit / replay policy without inference", "calibration does not compose / hop-ECE permutation-invariant", "equal-width vs quantile ECE / ranking ≠ calibration", "Qwen2.5 ≠ Archer / Qwen 3.8 sparring ≠ Archer / Qwen/Qwen3.8-27B ≠ Archer", "Deferred Crispification / TCE / AMS", "g0runmezadam/what-is-jev IS tunahansahin897/what-is-jev", "pd.cut equal-width vs jeval quantile", "A hunch is a probability with a policy attached", "soundness theater / measurement theater / hourly 0843", , "Jev Capability Resolver / NiazMorshed2007/jcr", "one tool nested capability tree / returns context / does not execute", "skills vs capabilities / workflow+judgment vs operations", "format independent of Jev / proposed open standard", "JCR_BAND_RATIO 0.6 is application policy / soft scores ≠ hard gates", "routing ≠ permission / docs ≠ authority to run", "sol-vs-opus5-20 lookup+explain / n=1 / Not Harbor task-execution", "wall-time mixed / Sol slower with JCR in 19/20", "NiazMorshed2007/jcr ≠ skill-broker ≠ skillranker ≠ jev-sift ≠ jev-lens ≠ jevusher ≠ jev_select_capability", "do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34", "notes.md §116", "copy the SemIf/MLX installer?", "quote 5.21× as beating Jev?", "treat 0.845 as a TypeSafe replica?", "collapse SemIf into kw2828/zhihz/semif-rs/semif-serve", "softmax over options as a Noul", "llm prompt to jev primitives", "conversion assistant not equivalent behavior", "heuristic conversion ≠ calibrated Noul", "alexwestco/llm-to-jev ≠ altryne/jevify", "user-provided 0940 / notes.md §118", "judge ≠ actuator", "candidate_mass", "softmax over A–H ≠ Noul", "hourly 0947 / notes.md §119", "ggmlc GGUF is not llama.cpp", "serving substrate ≠ calibrated replica", "Qwen3.5-9B ≠ Archer", "planner writes JEV selects", "hourly 1049 / notes.md §120", "open recreation ≠ calibrated replica", "semantic lint is a sensor not a proof", "cutoff 0.8 still soft", "paired bootstrap CIs *theirs*", "Same accuracy, 35x faster *theirs*", "hourly 1143 / notes.md §121", or "cascade sign-flip / calibration theater": read `references/faq.md`, + other", "training confronts Choice other / none-of-the-above", "soft AGENTS.md rules vs the linter", "screenshot Choice / omni System One", "extractive quotes / pointer not generator", "compaction summarize vs pointer", "encoder vs Jev compaction backend", "shadow-mode compaction rollout", "CI flaky-vs-real merge gate", "fail-open VOI wake/resume", "claim vs session evidence", "S1 indexer escalate-S2", "Harbor on/off routing", "fail-open vs fail-closed wake vs CI gate", "encoder vs Jev computer-use backend", "hybrid local decide + remote fill", "DONE vs verified success", "stdout prune vs session compaction", "OpenCode jev-pruner vs Claude jev-pruner", "zen-chat vs jev-zen Noul", "hard envelope then Noul prune", "Cua-S1 vs TypeSafe Jev", "plan vs execute dry-run", "specialist computer-use vs general agent", "local drop-in vs stub scorer", "route vs memory", "when does it hold / extractable from state", "decision model vs constrained LLM", "dual-process S1/S2", "combinatorial grid vs extractive", "uncalibrated local likelihoods", "decision-native RAG", "classify-first / read selectively", "living applied-mappings atlas / class patterns", "silence as safer / draft-gate heartbeat", "robotics text-state vs pixels", "verbatim ledger vs summary", "judgment as language primitive", "Stagehand extract pick-and-copy", "harness observe-score-act vs demo loop", "public judgment wall / six parallel questions", "meaning-search without embeddings", "attention ≠ correctness", "skills→oxlint / AST prove ∩ remainder", "session-sticky first-prompt routing", "measured RAG rerank vs generative rerank", "capability kernel / secrets never in the agent", "Jev is SENSOR not policy", "type-safe ≠ correct", "typed control plane around DSPy", "native vs verbalized confidence", "engine owns truth / Jev owns judgment", "human-confirmed kill gate", "train specialist vs few-shot hosted", "decide→policy→LLM leftover", "Noul 0.5 cannot-tell never rounded", "calibration ≠ sortable / ORDER BY", "pairwise inversion / Score ordinality / two-decimal ties", "wire-compat GLiFormer /v1/systemone", "class-backend economics", "loopback gateway hosted + local", "do not distill Jev as teacher", "active-learning triage", "evidence-packet explorer", "meaning-grep AND/OR/NOT", "closed-vote-only / no planner LLM", "Jev vs PCD Harbor", "PCD O(1) ≠ Noul", "host-owned handlers × System One", "OMP/pi fail-open gate", "permission vs probability / operator owns thresholds", "judgment ≠ permission / Jev never grants access", "eval integrity / instrument not score", "constrained optimizer + S1 features / never sole hot-path gate", "privilege ≠ verdict / effect contracts not tokens", "attention filter / VOI for human review / never blocks / never green unless sure", "measurement owns endorsement / evidence-gated question packs", "Jev supplies evidence / code owns authority", "ranking ≠ calibration / never hard-threshold raw p as frequency", "hot-click CU / indexed element table", "Jev judges relevance / code decides structure", "local rules first then remainder / never auto-train on own hides", "combinators / System One as control plane", "receipts not leaderboard / type-safe ≠ correct jaggedness", "VOI over skill library / skillranker abstention", "OOD calibration / AUC ≠ ECE", "Jev vs thinking-budget small models", "turnstile / replayable evidence≠authority", "MLX one-pass schema→JSON / Apple Silicon replica economics", "memory leases ended by new evidence", "never confidently wrong / TLA+ compose / escalate instead of hard-gate", "no seal no advance / coverage ledger / mint ≠ product brain", "skill-broker sibling / judgment ≠ permission", "sureness bands / max_prob is generous", "JevBench / calibration not in Main Score", "CI typed gate before expensive review", "Codex MCP host adapter", "judgment as attention redirect / jev-preflight", "compress-before-first-send / dizk jev-lens", "tools≠use / SessionStart over hoping", "observational memory / pi-om keep-kind", "open-Jev class / openvons / JevPick", "physical-world System One / HA-Jev / not for locks", "judgment outside the store / jevql", "landed-script trust / headless≠auto-approve", "digital-design combinators / extended five", "VOI cache admission / same-intent skip LLM", "BM25 vs Jev skill routing Harbor harness", "zeroshot vs BERT / contamination DiD", "typed escalate continue abort baton / inverted loop", "worth-your-attention VOI / ThinkyMiner Winnow", "Jev WHETHER Python HOW LLM WHAT", "conflict vs ignorance / named Choice escape", "Playwright executes Jev chooses", "OpenJev /v1/decide not drop-in", "SemIf wire-compat runoff; SemIf rename densify / MLX backend / 5.21× systems≠semantic / Softmax ≠ Noul (`notes.md` §117)", "decision-as-memory flywheel", "record/replay CI / jevassert", "failure-finding arena / jevarena ≠ jev-arena", "BBQ not a bias cert", "decider≠executor", "sentence-as-rule lint / jevlint", "sentence-as-rule lint / jev-lint is jevlint rename", "VOI hunk prune", "whole-repo intent VERIFIED/VIOLATION/UNKNOWN", "GLiNER2 spec ≠ replica", "open replica substrates / grande / laya-jolt / JEV-CPU", "ONNX local-jev not equivalent", "persist constraints across compaction / pi-heed", "calibration+cost first-class gates", "Harbor-shaped Jev vs SGR LLM-as-judge / jev-judge-bench ≠ jevarena ≠ jevbench", "hand no-text steps / jev-use / Vercel drops confidence", "Pi System-One control plane / pi-jev-control", "never free-generates / jev-gpt tree of Choices", "OpenRouter recipe atlas / samples not benches", "personal history feed / jevfeed / no social graph", "competing NAR claims / dual-channel ECE / openJev-verdict ≠ OpenJev", "empty compaction-proxy skip / IPECTER", "throughput ≠ latency / like-for-like ECE", "1-token logprob endpoint ≠ Noul / coverage ≠ correctness", "open replica engine / jevinf", "unofficial Elixir SDK ≠ OTP peer", "jevex n=16 files-to-read VOI", "commit pre-review attention≠verdict / middle band", "Hermes plugin is Agnes not TypeSafe", "pi-jev-compact ≠ pi-jev-compaction", "empty Codex-proxy skip / IPECTER runway", "decision-native inbox / mailordinal", "unofficial jev-cli not ready / ≠ jevql", "laya-multilingual / English checkpoint confident-wrong OOD", "schema-scorer peaked ranking ≠ calibration", "HF 401 / GitHub 404 Hub-only", "productized System One HTTP / classifier.dev", "escalate-under-threshold / smart tier / multi-label ignores", "silent FALLBACK / granite 0.546 vs advertised 0.800", "vs_jev tracked JSON / read eval/README", "choxos/jev-reviewer ≠ egma-ai / systematic-review pointer", "two-pass Choice+Noul evidence extraction", "not-found is an answer", "human check as productized judgment", "githubnext/localjev ≠ kunchenguid/local-jev", "wire-compat ≠ logit-equiv / prompted JSON ≠ structured read", "self-reported probs / entropy confidence", "GitHub Next local /v1/systemone", "LM Studio runner gap / structured-read primitives", "NandhaKishorM/laya packaging ≠ Hub-only / Router script-before-p", "post-T ECE ≠ raw ECE / Banking77 token-budget", "0.85 still soft / not TypeSafe drop-in", "external census ≠ scored bake-off", "GLiNER2+routers class-boundary", "incomplete openjev census vs watch", "Harbor honesty watch / silent fallback", "JevBench v1.2 geometric mean / cal ON rank / weight sensitivity", "option-order 72→21 / instruction models in the class table", "self-host latency ×2 assumption / est. costs", "Laya absent is a gap not a named exclusion", "Qwen3.8 27B ≠ Archer", "hourly already-folded watch / apply-the-five / skip thin noise", "hard-gate Noul as PR/quality gate is soundness theater", "S1 never stalls waiting / S2 one-use advisory", "Local controller ≠ githubnext/localjev", "purple telemetry = consumed not arrived", "seed = geometry not async replay", "20% starting gate still soft", "no pixels to either provider", "OCR+AX observe-score-act / typesafe-computer-use", "never send screenshot to frontier for the decision", "overlapping CU options = false low confidence", "split kind/item/site", "155× one-screenshot ≠ Harbor taskset", "decision ≠ answer-reader capture", "ASR observe-score-act / jev-voice-browser", "partial-speech VOI / free-text waits", "spoken confirm ≠ hard auth", "numbered overlay without another model", "wrap-as-execution / AgentGhost ALLOW ASK DENY", "rules first then Jev remainder / ASK throws / fail-closed", "reddpy/AgentGhost ≠ jwen5419807/agentghost ≠ vventirozos", "JP genre atlas / studio_yebisu / stars ephemeral ≠ eval", "Jev Clearly Explained / akshay_pachaar / LLM hammer", "schema-safe ≠ correct / 200× 400× TypeSafe ceiling", "questions-as-code / shadow first / not a TypeSafe how-to", "proposition ≠ embedding / contrast-set", "boolean composition of soft Nouls / AND OR NOT", "uehaj/jev-semgrep ≠ semgrep.dev", "meaning-grep dedicated fold / not a gate", "decision-validated UI / Jev never authors text", "decision-as-assert / jevtest ambiguous band", "typed decisions drive UI / jev2ui", "hybrid S1 closed verb menu / anima3", "pointer-not-generator search / JevFind", "jev-frontier-bench ≠ frontier-100", "product bakeoff ≠ architecture duel / GLiClass", "four engines same questions / majority floor", "authorship named escape / not evidence", "ha-switchboard HA remains execution", "n8n classify/route/score / Low Confidence", "fast-jev-compaction-pi ≠ pi-jev-compact ≠ pi-jev-compaction", "jevloop full-distribution optimizer / no LLM in the loop", "laya-vision SmolVLM / score untrained", "Cerebellum-2B /v1/decide ≠ TypeSafe / wire-compat vs agent-routing", "laya-grounded not drop-in / Platt not temperature", "GestaltLabs/Jeff-1 ≠ logan-markewich/jeff / acc vs ECE n=9730", "stanley-code empty findings ≠ approval / human promote", "findme ≠ JevFind / NL memory beam-search FS", "jevsubrouter price workers not conversation / counts ≠ dollars", "feelings .feels() default 0.5 is Noul-0.5-never-rounded / ≠ hunch ≠ Probably", "apa-agent-harness ≠ AntonioCoppe/jev-harness / unpublished npm", "grok-bot-jev skill cannot force a bot that ignores it / A/B proxies not tokens", "Essentiel-Jev never authority / human every action", "enzo-mcp independently falsifiable claims / ≠ jev-sift", "pigeonhole OTHER skip / decision-as-filing", "jev-reliability Nothing about accuracy", "clduab11/jev-test ≠ realZachi/jevtest / Nothing runs yet", "jev-rag-benchmark Jev wins is not an assumption", "dairui1/jev-lab ≠ BrendanH18/jev-lab", "jevmail gmail.readonly / mailjay archive/trash", "ZHUBoer/ego-jev reserved __none__", "runWorkflow completed ≠ success", "jsort scores are relative", "Noul not Choice for scale", "groundedness-judge-bench native vs schema-guided", "implicit_true included in yes", "jev_playground 0 promotions", "routing-backtest 0.0447%", "yuyang2230/jev-agent-skill jev-1.13-free", "jev-techstack-classifier stack_config.json", "s1_ruby collapse late", "undecided? abstain", "2389-research/judgement license null", "confidence ≠ winner p", "typesafeai-sdk-community not a new species", "tpellet/hunch exit 3", "never-execute list", "jev-file-search scores not calibrated accuracy", "jev-linkmap Jev never sees S2 prose", "muhammedilyasy/jev-mail metadata only", "tidy none-of-folders stay", "tab-bouncer pinned/audio/current never closed", "lkclean Show fail-open", "jev-yt-time-saver Show anyway", "ORIGIN pause-if-no-Jev", "validResponse sums-to-1", "jev-crawlers risk bands never raw boolean", "jevbrain AUTO_ACT is not a Noul", "judgekit YAML classify/score/route/verify", "typed-judge-kit verdict-in-code", "alsoleg89/decide packing VOI", "0.8 ≠ 80% accuracy", "Jev-Calibration Platt ECE 0.117→0.052", "jev-calibration-arena never acts", "ctmx/openrouter-jev-mcp Decision-as-Plugin", "FrancoisChastel/jev-code ≠ npm jev-code", "claudecode-jev-marketplace fail-open not hot path", "pedroknigge/mcp_jev packs not ask_jev", "cyrusasco/typesafe-mcp noul deadband 0.35–0.65", "codaaiteam/jev-skill jevtypesafeai.com ≠ TypeSafe", "hermes-switchyard ≠ hermes-jev-router ≠ hermes-plugin-jev", "nanoprune 2.8MB ECE 2.58%", "smartdio/jev-browser-agent ≠ ZHUBoer/ego-jev", "Dakai/omp-jev-web DONE ≠ proof", "hari007sh/jev ≠ dannote/jev", "0thernet/system-one-skills deterministic verify", "typed-gate band [0.40,0.60] is refusal", "pi-jev-gate fail-closed; choice is the verdict", "Foq ~25ms/2.2GB local", "rev prefill-only + HF jev-0.5b", "robfrase/jev planning memo", "typesafe_agent_gates 27/27 / 31/31", "EpicEric/safe-sh static remainder", "pastepilot Confirm before act", "Jev-Reranker live Jev not yet measured", "sessionwise opt-in relevance", "jev-search pointer sieve", "400ms Salesforce WebMCP", "typesafe-scheduler-diagnostics advisory", "droidjev screenshot-free", "Tewoto1 jevcu planner still writes", "ha-conversation-jev Jev→Grok", "dsh-jev can only gate", "jev-classification-benchmark specified not run", "jev-luna-pagerduty p≥0.50", "meldltd/meldecision laya-go ONNX", "laya-doom never pixels", "logixism/laya-api empty README", "akpsahan/laya ≠ Archer", "choxos/jevchess engine owns truth", "jev-drive sim not AV", "story-arc Jev never authors", "jev-hs-assistant HS6", "golergka/jev-plays-starcraft-2 UI-verified ≠ API Victory", "awesome-jev-use-cases catalog", "Nibir1/typesafe-go ≠ official", "fingerprint after redact", "recall vs decide", "publish fingerprints+answers", "CI replay as Harbor cousin", "Cache hit ≠ correctness", "hyperspaceai/jevcache ≠ kushals256/jevcache", "human labels only", "score never auto-accepts", "production capture flywheel", "sutro-sh/jev-align ≠ caiovicentino/jev-align", "guidance ≠ hook", "catalysts ≠ summaries", "compile-time System One", "unofficial ≠ TypeSafe", "format_version modernbert-jev/1", "Argos1111/jev_local ≠ us/jev-local ≠ kunchenguid/local-jev", "LFM default ≠ ModernBERT backend", "Nemotron ≠ TypeSafe Jev", "not a calibrated replacement", "djev-dev complements djev-spark", "images as Choice options", "Laya essay numbers *theirs*", "Router/OOD confidence", "hosted bootstrap ≠ silent TypeSafe", "difficulty + policy thresholds + JSONL trace", "jev-codex-pilot model + reasoning depth", "keep/shadow/hybrid/reject", "quarry evidence projection", "Frank-ZY-Dou/awesome-jev robotics/3D/control", "one-dollar-tahoe TypeSafe Jev defense eval", "jevguard calibrator/cache/escape", "jev-ci-selector CI shadow mode", "llama-jev llama.cpp replica", "petercr/jev-orchestrator ≠ FleeexCorp/jev-orchestrator", "seb4ez/jevguard ≠ AseemPrasad/JevGuard ≠ pablozr/JevGuard", "webNeat/llama-jev ≠ WiktorB2004/llama-index-jev", "OpenCode jev-pruner context sieve", "observe→score-candidates→prune", "jev-zen / jev-1.13-free", "zen-chat ≠ Noul", "fail-open original", "keepScore >0.1 floor", "host port of tamaratran/jev-pruner", "indiejoseph/opencode-jev-pruner ≠ nrdz-labs/fast-jev-opencode", "jev-webagent-bench empty stub", "Kiln-AI/jev_jsonschema noul_threshold 0.5", "NSStudent/JevSwiftSDK unofficial", "GLiNER2 native Apple path", "unofficial Swift/Core ML GLiNER 2.5-small", "entity spans + confidence", "not Choice/Score/Noul", "not TypeSafe", "label descriptions as schema", "on-device ANE economics", "honesty locks", "shershah1024/gliner-native-runtime ≠ Fastino", "≠ gliner25-compaction ≠ gliner2-ultrafast ≠ Eran-BA/Jev_from_GLiNER2 ≠ NSStudent/JevSwiftSDK ≠ jevmlx", "default threshold 0.1 still soft", "soft Noul ≠ hard safety", "Decision Graph Protocol frame→assess→commit", "app retains permissions/effects", "Jev-first assessor-neutral", "guarded commit / receipt/next frame", "assessment batching", "hard-gating DGP as safety theater", "numerous-com/dgp ≠ TypeSafe official", "jegrep calibrated path+range Nouls", "no embeddings/index/daemon", "~$0.01–0.03 typical", "agent --json", "can1357/jegrep ≠ Bentlybro/jevgrep ≠ uehaj/jev-semgrep", "Archer-arch fidelity", "kev family OOD 0.76–0.77 vs Jev 0.86", "block-causal isolation", "pointer/readout CE-trained", "/v1/systemone drop-in", "replica honesty", "cost-sensitive decision theory × System One probabilities → control flow", "thresholds derived from costs not hard-coded", "YES / NO / UNSURE from cost_false_yes / cost_false_no / cost_human", "auto-batching same-object questions", "Kungie/gut ≠ tpellet/hunch ≠ carldaws/hunch", "judgment vs generation", "deterministic execution after probabilistic judgment", "exactly one app-owned callback", "explicit uncertain branch", "Illusion47586/judge ≠ lexingtonhibiki/judgekit ≠ Ascurse/typed-judge-kit", "variable-N option scoring as the trainable object", "dynamic candidate bags not fixed label sets", "zwliJay/jev-forge ≠ NanoJev", "open replica economics / latency vs closed Jev", "NAR local drop-in", "wfzyx/von late-catch HIGH", "competing NAR claims / replica honesty", "typed judgments vs chat judges on guardrailing", "ishaannk/llm-vs-jev cross-note only", "deeper integrity fold is rh-guard", "nothing wins outright", "can be argued out of guarding"", "Jev IS the if-statement", "judgments/probabilities drive branches", "text model only writes prose", "interpreter owns variables/loops/budgets/replay", "otherwise maybe / confidence gate", "chaos samples after the gate", "southpolesteve/probably ≠ carldaws/hunch ≠ feelings ≠ Kungie/gut ≠ Illusion47586/judge ≠ tidymodels/probably", "133★ / forks 10 live", "build calibrated classifiers from human feedback", "retrieve by relevance not resemblance", "one calibrated yes/no per memory in one request", "pointer mode 17/18 19/20 *theirs*", "embedding resemblance misses the allergy", "samdotmak/jev-recall ≠ jev-search ≠ jev-sift ≠ carryforward ≠ chopratejas/invalidate", "memory leases ended by new evidence", "six Nouls then fixed rules in code", "0 of 157 false invalidations", "questions/plans/directives are not evidence", "unsure → review queue", "host keeps the store", "name↔body / comment truth / test-claims", "mizchi/jev-lint is mizchi/jevlint rename", "no shipped rule has severity error", "~1 in 5 findings wrong *theirs*", "mizchi/jev-lint ≠ huntedman/JevLint ≠ MichitoSugawara/jev-lint", "JSON Schema → typed JSON via Jev", "noul_threshold 0.5 decoder not a proof", "IncompatibleSchemaError lists every bad property", "on-device Laya CoreML ANE", "~5 ms P50 short decisions", "189/189 FP16 checkpoint parity", "10× not achieved", "mizorewww/laya-coreml ≠ gliner-native-runtime ≠ jevmlx ≠ NandhaKishorM/laya", "softmax over allowed tokens ≠ Noul", "question-first cache", "Micha0827/snapjudge ≠ githubnext/localjev ≠ jevmlx ≠ cendress/SnapJudge", "Jev-first Pi agent loop", "slow-LLM fallback", "explicit action menu / CandidateSource unimplemented", "62 tests wiring not quality", "direwolfiy/JevPi ≠ standardagents/jevpilot ≠ pi-jev-control", "resume-screening bias audit methodology", "name×resume factorial independent Nouls", "callback determined by resume quality", "mean-probability name gaps operationally negligible", "natemoo-re/bias-bench ≠ BBQ", "Plan/PRD panel → code-owned pass|review|block", "cheerleading out of scope", "austindixson/planalyzer ≠ single-goodness Noul", "cost-aware multi-model routing/escalation", "decide vs do", "successful-task cost", "cannacre8ive/switchboard-ai ≠ ha-switchboard ≠ hermes-switchyard", "frozen-protocol zero-shot bench", "TypeSafe Jev vs PrismNLI vs Laya", "contamination caveat", "elcronos/jev-vs-open-decision-models ≠ JevBench ≠ DMB", "context-window admission control", "VOI gate which tokens are worth the expensive model", "fail polarity per lens", "on small inputs lenses lose money", "cvsgireesh/jevusher ≠ jev-sift ≠ winnow", "typed decision control plane", "receipt ≠ authorization", "historical-v0 zero retained cases", "MokiMeow/jev-fabric ≠ jev-forge ≠ dgp", "live 15-dim typed rubric re-score per pause", "scoring economics exemplar", "OpenJev/Codiv ≠ TypeSafe hosted", "jose-troche/live-rubric ~$0.000004 desc / ~$0.000006 README", "adversarial pre-registered Jev eval", "28 predictions before data", "123,805 requests", "confidence does not track ignorance", "polite injection 65% / crude 0%", "willkelly/jev-evaluation ≠ jevals ≠ jev-baselines-eval", "provider-neutral Elixir/BEAM Noul/Choice/Score SDK", "class infrastructure", "nshkrdotcom/system_one_sdk ≠ typesafe_sdk ≠ dannote/jev", "question-linting of Jev questions themselves", "nine jaggedness rules, no API key, no labelled data", "static lint ≠ measured separation", "yodablocks/jevq ≠ tenbin ≠ JevLint ≠ commitjev", "open-weights Laya as class exemplar (binding)", "Nx/Bumblebee runtime", "host chooses backend", "ChristianAlexander/laya_ex ≠ system_one_sdk ≠ dannote/jev ≠ NandhaKishorM/laya", "on-chain/edge Laya deploy", "parity_verified stays false", "model output never grants Tx", "humandebri/IC-Laya ≠ laya_ex", "auditable weekend replica", "Jev outputs never used for training", "soft human-vote distributions", "unpaired 0.577 vs 0.727", "agilabs-ai/jev48 ≠ JevBench ≠ Mapika/decider", "adversarial dual-judge / framing attack surface", "comparative framing is the usable judgment", "prior injection crowds out evidence", "copyleftdev/ember ≠ ember.js", "Laya specialist fine-tune pipeline", "training still GPU-pending", "PIXELZX0/XERON ≠ convaiinnovations/laya", "Hub Laya replica drop", "daliborsb/laya ≠ convaiinnovations/laya ≠ NandhaKishorM/laya", "System One student distillation corpus", "gold is programmatic", "teacher is closed-API clone", "do not distill Jev as teacher of record", "MagaBitmex/jev-4b-distill-data ≠ missing student checkpoint", "non-LLM VIN System One", "planning depth not chat", "lewislululu/jevon ≠ douglance/jevon", "source-bound evidence checks", "local quote mismatch needs no API", "exit 0 ≠ claim truth", "WaynezProg/jev-kit ≠ jonathanavis96/jev-kit (Airlock) ≠ jev-use ≠ jev-mcp", "independent System One evidence catalog", "scores not one leaderboard", "no external record currently reproduced", "TokenTrim no-Jev matched hybrid 62.4%", "reachjalil/system-one-bench ≠ mallahyari/system-one-benchmark", "21 tasks · 134 items · 208 questions", "scenes from public GitHub contracts, not production logs", "SivletLabs/jev-eval ≠ willkelly/jev-evaluation ≠ 4esv/jev-eval ≠ xxkuboxx/jev-eval ≠ onlyoneaman/jev-eval ≠ dayhaysoos/jevals", "option isolation (sibling-blind)", "permutation-equivariant", "Hub OWNER not published", "nafisazizir/hev ≠ jaredpalmer/kev", "frozen local LLM logits, no trained decision head", "residual-head 9,222-param decreased 73/96→67/96", "confidence = 1−normalized entropy, not P(correct)", "yuki-oshio/mini-jev ≠ r-ms/mini-jev", "Jev classifier as autoregressive next-token predictor", "ChatJev-style soundness theater", "erik-dunteman/ChatJev ≠ dannote/jev ≠ jev-gpt", "calibrated decision head × AlphaProof value head", "implementation-layer isomorphism, semantic difference", "timeout = censoring", "do not launder Noul as proof", "parallel rank-prediction vs serial selection", "independent questions can conflict", "zzzzzec/jevsort ≠ keltokhy/jsort", "curated open System One ecosystem catalog", "rupeshpoojary9/awesome-open-system-one ≠ AnotiaWang/awesome-jev", "arXiv paper radar with Jev relevance scoring", "ranking ≠ calibration / 0.5 still soft", "fail-open failed evals not marked seen", "train calibrated ~27M from scratch", "typed Q→prob dist / one forward pass / no LLM decode", "hyusi2003/MiniSystemOne ≠ Colvin0315/MiniSystemOne", "description-only stub / size 5", "ESCI hard probe fails four of six", "jev_bool ECE 0.242 inversion 0.255", "do not re-fold §60 six-gates as new", "jobbyjev one-request-per-company from batch-size result", "find/design/evaluate TypeSafe Jev decision loops", "karanb192/jev-architect ≠ samtay32/jev-system-architect", "Jairik/jev-distiller size 1", "distill-Jev UI stub / do not distill Jev as teacher of record", "post-launch scored use-case map / Jev self-scores then human curation", "licensedsaucer9-web/jev-opportunities", "Jev-inize a use case into classifier/router", "gavinHuang/jevinize → simple-jev not TypeSafe", "featherless-ai/simple-jev", "compare saved decisions / same label can still change the branch", "VihaanAgarwal/jev-diff ≠ Saik0s/diffusiongemma-jev-macos", "not tested with a live Jev API key", "constrained logprob + temp/Platt ≠ Noul", "OpenJevPro pastes openjev-sglang JevBench as own", "zhangcy122/OpenJevPro ≠ IamBusy/OpenJev ≠ ekzhang/openjev-sglang", "PolyForm Noncommercial", "SmolLM-135M / sub-70ms / 0 output tokens", "demo P(True) 0.5052 / Choice conf 0.2872 / Score conf 0.0055", "README claims MIT / GitHub license null / no LICENSE file", "patelvishwa112/jev-system-one-rlcd ≠ arnabgho/rlcd-lite ≠ blackwood-rlcd", "source-backed Awesome Jev radar / 306+ commit-pinned", "logicrw/awesome-jev-projects ≠ AnotiaWang/awesome-jev ≠ yibie/awesome-jev ≠ cobanov/awesome-jev ≠ rupeshpoojary9/awesome-open-system-one", "auto GitHub sync / Issue-only submissions", "hashed n-gram encoder / rival-aware attention", "olanotolu/jevbetter vs jevlike starter", "synthetic hard menus top-1 0.916 vs 0.873 / ECE 0.0182 vs 0.0367 / 40 vs 4608 menus/sec", "shuffled-context control 0.335", "Turn any open LLM into System-One Jev", "uspraveen/Jevify ≠ Mintzs/jevify ≠ gulagala001/jevify", "Jevify-any-LLM architecture probe", "description-only stub / size 0", "Train encoder-only calibrated decision models from a task sentence", "Exu is a toolkit, not a method", "strictly proper scoring rule", "Pre-alpha", "Ruivalim/exu-base", "scratch-trained calibrated decision model", "typed Q → probability dists", "Colvin0315/MiniSystemOne ≠ hyusi2003/MiniSystemOne", "no published weights download URL", "90.5 seconds / 29.2% pipeline evidence", "p_i/p_j independent of other candidates", "Recipe for calibrated decision models — small model out", "init → synth → train → eval → serve", "91.1 % / ECE 0.022 *theirs*", "Jev zero-shot 75.1", "scienthoon/luce", "Put Jev's three headline claims on trial", "0.5B local GPU", "46x speedup / accuracy identical", "ECE 0.624 sentiment catastrophe", "bigger model worse calibration", "RichardoMrMu/jev-mini ≠ yuki-oshio/mini-jev ≠ r-ms/mini-jev", "System-1 decision engine for local LLMs", "structured choices only", "JSON parse of generated text ≠ Noul", "TypefAI JEV / Journal Entry Voucher", "tapsin/jev-local ≠ us/jev-local ≠ Argos1111/jev_local", "Jev 1.13 reward-model eval across 8 benchmark tracks", "40,940 examples / 0 API errors", "RewardBench v1 92.58%", "Precise IF 50.63%", "goya4140/jev-reward-model-evaluation", "Scaffolding in progress", "Jev vs LLM support-ticket routing", "static + live decision bench", "TypeSafe's own published benchmark", "illustrative simulations, not live API calls", "JevBench v1 — smart/cheap/fast/reliable", "I/C/S/K 25% geometric mean", "classifier.dev fast tier 84.8 is Jev behind its own API", "do not re-fold §78 v1.2 board as new", "Laya (421M) 70.1 now on board", "Zero-shot/few-shot LLM routing", "hard budget filter before Jev", "Jev never asked to perform budget arithmetic", "Jev judges the next state, XState enforces transitions", "simulation uses synthetic keyword fixtures", "catalog gravity", "v-modal/awesome-jev-tools", "★339 live REST", "curation is not endorsement", "crawler-maintained directory", "Daily GitHub + npm sweep, human-merged", "RadRebelSam/awesome-jev ≠ AnotiaWang ≠ yibie ≠ cobanov ≠ logicrw ≠ v-modal", "HF peft SPLADE/BGE reranker", "rdxtremity/jev-reranking ≠ carlaiau/jev-reranking", "query-side encoders, not a Jev replica", "ONNX System One Qwen3.5-4B scorer", "source:pngwn/system-one-qwen3.5-4b-scorer", "CC-BY-NC-4.0", "temperature 1.75", "transformers.js AutoModel cannot load this graph", "Consistency benchmark Space", "This Space contains no benchmark result yet", "12-case plumbing fixture", "Benchmark-driven Jev router and judge", "cheap alone is not success", "Jev does not write, sum prices, or claim accuracy %", "Sol 94.2 / Luna 83.9 / Jev path 89.7", "19.2% Sol / 62.3% cost save / 4.5pp miss of 2pp non-inferiority", "p50 latency worse than Sol due to routing overhead", "erendikmenn/jev-llm-router-benchmark ≠ jev-rag-benchmark ≠ ryantsai/jev-llm-router", "Express + node:sqlite", "mock and Jev decision engines", "previous_ticket_count >= 3 is code", "MIN_CONFIDENCE 0.6 still soft", "substring false positives", "aesaganda/jev-ticket-router ≠ SarathChandraBellam/jev-vs-llm-ticket-router", "Universal Figure & Diagram Router", "confidence ≥ 0.85 hard-gate is theater", "generative AI banned from scientific plots", "six visual branches", "hoangngochuong24947-gif/jev-figure-router", "human-labeled (state, question, label)", "166,054 rows / 22 configs", "soft_label for human uncertainty", "Praveenrajus/jev-bench ≠ fstandhartinger/jevbench", "ternary bonsai System One GGUF", "openjev's mechanism, Bonsai's weights", "Hub does not ship weights", "100/100 easy T/F is not Harbor", "label_mass ≠ correctness", "stock llama.cpp Q2_0 silently gibberish", "NicolaiMTLassen/open-bonzi-jev ≠ NicolaiLassen", "transformers.js DeBERTa ONNX", "source:com-kotobalabs/open-jev-deberta-v3-large", "temperature 1.05", "AutoModel from_pretrained works", "onnx-community/open-jev-deberta-v3-large-ONNX ≠ system-one-qwen3.5-4b-scorer-ONNX", "107★ densify", "GH 151M vs README 149.6M", "PR #1 now closed unmerged", "do not re-fold §71 claim-audit as a beat", "typed decisions, RLCD, confidence-gated routing", "structured ≠ correct", "mock not live API", "26 tests", "wjdjdakf17/jev-study ≠ baekenough/jev-study", "bonzi-27b-v2 / ternary-8b / 27b-v1 GGUF family densify", "WANLI-256 74.6% / 65.2% / 71.1% *theirs*", "Bonsai 1 27B Q1_0 runs on stock llama.cpp", "ternary still needs PrismML fork", "hf:heman10x/openJev-verdict-2.0 twin tokenizer-only", "OpenJev Vision image classification + uncertainty", "CLEVR-4 held-out joint 0%", "hfdataset:IamBusy/OpenJev-Vision-Research-v0.1 12,832", "294,912 derived targets not independent samples", "Laya multilingual ONNX WebGPU typed-decisions port", "63/63 selected answers / 5.1e-4 CPU / 1.2e-2 WebGPU", "UpHash-Network/mini-jev is yuki-oshio transfer", "jev-injection-bench 11,900 labelled prompts", "Jev best ranking / Haiku better ECE 0.021 vs 0.058", "0.5–0.9 band is where Jev's numbers do not mean what they say", "Prompt wording moves panic 28%", "manojlds/jev-dspy-bench ≠ dspachos/jev-dspy ≠ jmanhype/jev-dspy-lab", "Jev agreement is similarity, never ground truth", "no aggregate quality grade or merge gate", "AbstentionBench-on-Jev rank 1 of 20 vs 2025 field", "question-asymmetry", "forward-looking 0.465 never extreme", "openkev calibration layer not a runtime", "ECE vs coverage independent", "select_threshold returns inf", "escalation catches uncertainty not ignorance", "misakaikato/openkev ≠ jaredpalmer/kev", "pdf-race Docling→Jev vs Gemini", "parser owns the wall clock", "12/12 tie is a tie", "titles selected not generated", "flopcheck 16 calibrated tweet judgments", "mechanical tells in code", "ZeroX-01/jev-atlas ≠ Zaious/jev-capability-atlas ≠ gorock007/jev-atlas", "Laya calibration lab Gradio MCP", "T never changes argmax", "confidence ≠ top-label p", "easy probe set refused", "40–48 rows too small to ship T", "Gemma-4 26B-A4B jevify classification+calibration", "LoRA adapter twin not independent eval", "Gemma-4 E4B jevify", "E4B LoRA stub card", "kushalpatil/jevify-gemma4 ≠ Mintzs/jevify ≠ gulagala001/jevify ≠ uspraveen/Jevify", "GH kushalpatil07/jevify 404", "PAWS 0.580/ece 0.288 is the weak cell", "smaller E4B slightly better OOD ECE than 26B-A4B", "Hub jevify merged LoRA ships weights", "bonzi Bonsai-8B v1 GGUF densify", "Bonsai-1.7B v1", "Bonsai-4B v1", "WANLI-256 64.5% / 60.2% / 52.0% *theirs*", "rank #4 / #5 / #6 of 6", "JulesHuisman/jev-eval scaffolding / README SHA c356a584 (was empty e69de29b)", "JulesHuisman/jev-eval ≠ SivletLabs/jev-eval ≠ willkelly/jev-evaluation ≠ 4esv ≠ xxkuboxx ≠ onlyoneaman ≠ dayhaysoos/jevals", "7 bands 6/10 vs 40 bands 0/10", "source receipts + confidence slider re-policy without re-inference", "32/32 synthetic is smoke not production", "classify HF datasets across typed semantic dimensions", "roadus2 watch misspelling; lock roadius2/ultra_laya", "ultra_laya REVIEW defects", "default branch claude/laya-jev-review-gg5ppo", "XNLI EN 88.3% ECE 0.032 → RU 77.3% ECE 0.096", "Δ −11.0 pp [−14.2,−7.8]; ECE +0.063", "MASSIVE no detectable difference at n=600", "confidence is function of p_max (r=1.000)", "pointer-not-generator 400 human-authored responses", "proposed ≠ authorized", "FewRel 160: Jev 85.0% vs lexical 13.125%", "gated 100% (95/95) coverage 59.375%", "J++ composable semantic computation language", "judge-jev 0.5 still soft", "947 repos scored; A 273 / B 302 / C 372", "LLM rubric ≠ benches", "No benchmark winner is claimed", "phishing: naive 62.6% vs regex 91.8%; 5-atomic + LR 95.0% *theirs*", "AITuber tension ±15", "README npm global; repo is Rust", "git-confess code owns counting/blame/ratio", "httpx exhibit 11% (13/119) *theirs*", "90d trend +12.40% vs random +12.75% vs BH +41.71%", "5m win rate 25%", "Awesomejev 656 entries / 38,160 stars", "tracker likes 64 (+4) lastModified UNCHANGED", "Laya present; Blackwood ABSENT; Archer still promised_not_landed", "Blackwood tracker ABSENT; likes 2 gated manual", "r = c - p_a", "ECE 0.021; acc 0.807 vs warmup 0.746", "Independent primitive", "11.57s vs 54.10s · 4.67× · 120/128 *theirs*", "default path is pretrained Gemma probs not trained RLCD head", "GH Meanblock 404; lock leesk212/JEV-CPU", "softmax over letter slots ≠ Noul", "WANLI 0.741 vs openjev v2 0.77 *theirs*", "3-way NLI ≠ Noul", "priority 0.464 = majority floor", "banking77 contaminated", "raw margins not probabilities", "do not distill Jev as teacher of record (they distilled Haiku)", "“0.9 is not one number”", "ranking ≠ calibration", "banking77 0.8–0.9 stated 0.86 actual 0.73 over-confident *theirs*", "≠ Praveenrajus/jev-bench ≠ fstandhartinger/jevbench", "$0.0000153–$0.0000226 vs circulating $0.0004 (~20×)", "Score is 0..n-1 expectation not 0–1", "Noul has no confidence field", "TCP floor 198.8 ms", "type reliability is not a reason to choose Jev (json_schema 5/5)", "gateway tax not one number", "Function-only 5/8 vs hybrid 8/8", "4/8 without Jev", "8 designed cases not conversion lift", "200-row pilot Jev 86.5% 173/200 vs Gemini Flash-Lite 86.0% 172/200 vs Pro 87.0% 174/200 *theirs*", "not a ranking", "情緒測謊器", "8-example Jev vs GPT-5.6 Sol ~64× cost 5.4× latency *theirs*", "synthetic; no inference", "≠ JevBench v1.2 §78", "Judged 3317 / listed 2560", "Jev judges, code applies policy", "APA “microsecond policy / zero hallucination” overclaim", "Client-side quiz; pointer from held docs; scanned-PDF warn", "Jev judges / agent reasons / user decides", "selecting an option is not permission to implement", "pattern exact, judgement must clear floor", "no matching pattern → no model call", "not a correctness oracle", "Spec vs artifact remainder", "treating 0.85 as 85% / minProbability hard-gate as Harbor", "VERIFY acquires discriminating evidence, never same-pool confidence-only rescoring", "fast/full/max are ceilings not sizes", "Solar writes, Jev chooses NEXT ACTION", "do not reopen or amend PR #23 or #24 or #25 or #26 or #27", , "Calibration is not alpha", "NO CURRENT ALPHA CANDIDATE", "ΔR² approximately +0.00084", "Brier 0.2131387", "ECE 0.0421875", "Adding Jev probability to deterministic volatility improved Brier by only 1.4058e-05", "default 0.5 keeps zero non pinned", "keepResult median 0.14 to 0.17", "keepCall median 0.28 to 0.35", "usable range is about 0.10 to 0.25", "7.8% to 57.9%", "judges results it never sees", "task-finish eval not built yet", "$0.002 per compaction", "slavadubrov/sgr-judge-bench ≠ slavadubrov/jev-judge-bench", "Jev 108/120 $0.083 0.34 s", "Luna SGR 114/120", "paired Jev accuracy-difference intervals include zero", "not evidence of equivalence", "GLM SGR 26/120 93 format failures", "Terra-planned Jev hybrid 55/120", "rule-based by default, optionally Jev-backed", "empty README", "missing key cannot break the experience", "prefill plus exactly one decode", "softmax over A/B/C ≠ Noul", "BBQ 9,053/10,000 (90.53%)", "ECE 0.0890", "Mean confidence 0.9943", "overconfident", "score and noul not implemented", "DGUI 12 rows (was 6)", "INSTRUCT 119 rows likes 2", "encode the state once, decide everything in parallel", "0.740 accuracy against a 0.508 majority", "ECE 0.047", "fine-tune's advantage ends where its 384-token training data does", "jasonkneen/open-jev ≠ pngwn/open-jev", "same sha d41dc3cd", "Space does not call Jev", "recomputes routing from saved probabilities", "200-case Jev 97.0% / 100.0% / 95.0% / MAE 9.22", "synthetic repository benchmark", "Jev evaluations are advisory", "YehuiTang0316/jev-nlgrep ≠ Bentlybro/jevgrep ≠ can1357/jegrep ≠ uehaj/jev-semgrep", "default threshold 0.8 still soft", "40-line windows cannot prove whole function", "token-native sequential start/end Choice", "Gemini/Haiku stubs not configured yet", "handful of hand-written examples, not a benchmark", "Jev judged exactly what it was given", "laguagu/jev-skills ≠ laguagu/jev-evidence-lab ≠ Pleo2/awesome-jev-agent-skills", "contract_passed is not a claim of guaranteed factual truth", "Wilson lower bound 0.85 floor", "fixture mode no savings claim", "SemIf 2207★ (+21 vs §110 2186)", "jevlike 1043★ (+5 vs 1038)", "TypeAR 15★ (+1 vs 14)", "AnotiaWang 97★ (+1 vs 96)", "yibie/awesome-jev 506★ (+16 vs 490)", "Laya likes 822 (was 802)", "tracker likes 64 flat, lastModified UNCHANGED", "do not reopen or amend PR #23/#24/#25/#26/#27/#28", "Heman10x-NGU/Verdict-open-jev ≠ Heman10x-NGU/openJev-verdict-2.0", "TF-IDF + LogReg ECE 0.0207 vs Jev 0.1440", "Verdict-open-jev 48.07% vs Jev 90.80%", "abstention combined recall 10.00%", "p50 35.58 ms", "K=25 (maximum capacity) 72.00%", "0.85 coverage 84.60% selective risk 1.18%", "26.1× faster than standard Qwen JSON generation", "Jevify 90.0% / 167 ms CUDA graphs disabled", "Finding 1: Brier on stated confidence alone is a trap", "grpo_rlcr 0.78 / ECE 0.084", "reliability 0.007 but resolution 0.000", "27 900 schema-driven decisions", "13 600 / 13 600 questions", "candidate mass min 0.99999624", "22 configs · 166,054 rows · 4 calibration-gold", "sha a39eba3f", "Student B MAE 0.148 / Pearson 0.836 / 86.0%", "pngwn/open-jev-laya-bench README 404", "sha 9f69c742 likes 2", "HDFS 0.9933 (745/750) / retain 0.0084", "BGL ERROR/FATAL protection 1.0000", "2,479 / 2,500 HDFS uncertain", "cache hit 0.9648 (2412/2500)", "$0.153936 estimated", "E2 recomputes from saved probabilities", "Space sha eda59e0a", "MASSIVE English 0.783 / Khmer 0.033 / Hindi 0.133", "40–48 rows too small to ship T", "T never changes argmax", "siren2345/jev-single-decode-transformers ≠ siren2345/jev-single-decode", "Split Transformers experiment from llama.cpp runtime", "tanayvasishtha/jev-lab ≠ dairui1/jev-lab ≠ BrendanH18/jev-lab ≠ yibie/laya-jev-lab", "Four experiments stress-testing TypeSafe's Jev: calibration, bundle bias, label bias, and ensembling", "second pass must be $0.00 from cache", "The pages never call Jev", "Gemma 4 31B 77.0% / Jev 1.13.0 61.4% / Laya 322M 0.0%", "restriction state 95.0% against 84.4%", "None of the systems are particularly good at knowing when to stop and ask", "They skip the question and call a tool directly", "100% schema pass", "six-field joint 48.8% vs 72.8%", "ywchiu/jev_benchmark ≠ Running-Dolphins/jev-bench ≠ Praveenrajus/jev-bench", "ACT / REVIEW / FALLBACK", "A provider failure, timeout, malformed output, or missing answer is **not** a policy outcome", "confidence is descriptive provider output, not a substitute for probability", "Quality denominators include only valid scored answers", "an exact halfway tie chooses the lower level", "aiwithenoch/Jev-Skill ≠ simplosophy/jev-skill ≠ laguagu/jev-skills", "The local path does not claim to turn a smaller checkpoint into Jev", "Low support becomes decision: \"review\"", "MIT-0 SPDX NOASSERTION", "current-llm", "结构兼容,不是 Jev 模型能力", "altryne/jevify ≠ Mintzs/jevify ≠ gulagala001/jevify ≠ uspraveen/Jevify", "Find where Jev belongs. Design the questions. Measure the difference", "TypeAR-AI/TypeAR 301 → TypeLLM/TypeLLM", "TypeLLM/TypeLLM 16★", "SemIf 2241★ (+34 vs §111 2207)", "jevlike 1051★ (+8 vs 1043)", "AnotiaWang 98★ (+1 vs 97)", "yibie/awesome-jev 525★ (+19 vs 506)", "Laya likes 864 (was 822)", "tracker likes 67 (+3 vs 64)", "lastModified UNCHANGED `2026-09-20T04:29:16.000Z`", "do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#32", "hysteresis enter/exit / replay policy without inference", "calibration does not compose / hop-ECE permutation-invariant", "equal-width vs quantile ECE / ranking ≠ calibration", "Qwen2.5 ≠ Archer / Qwen 3.8 sparring ≠ Archer / Qwen/Qwen3.8-27B ≠ Archer", "Deferred Crispification / TCE / AMS", "g0runmezadam/what-is-jev IS tunahansahin897/what-is-jev", "pd.cut equal-width vs jeval quantile", "A hunch is a probability with a policy attached", "soundness theater / measurement theater / hourly 0843", , "Jev Capability Resolver / NiazMorshed2007/jcr", "one tool nested capability tree / returns context / does not execute", "skills vs capabilities / workflow+judgment vs operations", "format independent of Jev / proposed open standard", "JCR_BAND_RATIO 0.6 is application policy / soft scores ≠ hard gates", "routing ≠ permission / docs ≠ authority to run", "sol-vs-opus5-20 lookup+explain / n=1 / Not Harbor task-execution", "wall-time mixed / Sol slower with JCR in 19/20", "NiazMorshed2007/jcr ≠ skill-broker ≠ skillranker ≠ jev-sift ≠ jev-lens ≠ jevusher ≠ jev_select_capability", "do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34", "notes.md §116", "copy the SemIf/MLX installer?", "quote 5.21× as beating Jev?", "treat 0.845 as a TypeSafe replica?", "collapse SemIf into kw2828/zhihz/semif-rs/semif-serve", "softmax over options as a Noul", "llm prompt to jev primitives", "conversion assistant not equivalent behavior", "heuristic conversion ≠ calibrated Noul", "alexwestco/llm-to-jev ≠ altryne/jevify", "user-provided 0940 / notes.md §118", "judge ≠ actuator", "candidate_mass", "softmax over A–H ≠ Noul", "hourly 0947 / notes.md §119", "ggmlc GGUF is not llama.cpp", "serving substrate ≠ calibrated replica", "Qwen3.5-9B ≠ Archer", "planner writes JEV selects", "hourly 1049 / notes.md §120", "open recreation ≠ calibrated replica", "semantic lint is a sensor not a proof", "cutoff 0.8 still soft", "paired bootstrap CIs *theirs*", "Same accuracy, 35x faster *theirs*", "hourly 1143 / notes.md §121", "revisit HIGH / since-last-look", "catalogued repo changed", "star-noise vs material change", "densify prior notes without inventing equivalence", or "cascade sign-flip / calibration theater": read `references/faq.md`, then `references/mental-models.md`, then `references/mixed-architecture.md`, then `references/judgment-class.md` before any mapping. Proof, @@ -110,6 +110,43 @@ classical method you already trust, substitute it, classify the win rubric, candidate-source, and policy versions. Keep questions, criteria, and thresholds in one reviewable module; store raw judgments separately from derived actions. +8. Already-catalogued sources are not done. When fingerprints move, + densify the prior card. Treat **revisit HIGH** like **novel HIGH**. + Star-noise is not a fold. See **REVISIT / since last look**. + +## REVISIT / since last look + +Basit standing order (2026-09-20): Augustus revisits already-catalogued +repos when they change since last look, not only first sightings. +Things change a lot. + +Hourly fold prompts must treat **revisit HIGH** like **novel HIGH** +for Augustus. Skip is for star-noise and name collisions, not for "we +already have a card." + +Revisit / since-last-look lock: catalogued repos are not done; store fingerprints default_sha, pushed_at, description_hash, release_tag; material change is README/API/release/calibration claim/serving port/bench rewrite; star-noise is stars/likes/forks alone; densify the prior notes section, do not mint a sibling first sighting; do not invent equivalence; SHA move is not a replica; treat revisit HIGH like novel HIGH for Augustus; notes.md §122 + +**Fingerprints to store** (hourly diffs these): `default_sha` +(full HEAD, not a 12-char prefix), `pushed_at` (GitHub push +clock), `description_hash` (sha256[:12] of the GitHub / Space +description, or null if unquoted; not README SHA), +`release_tag`. Helper: +`research/revisit_fingerprints.py`. Checklist: +`research/revisit-checklist.md`. + +**Material change** (revisit HIGH): README rewrite, API / primitive +surface, release tag, calibration claim, serving port or bottle, +bench rewrite. **Star-noise** (pulse only): stars, watchers, forks, +Hub likes, likes-only jitter with `lastModified UNCHANGED`. + +**Densify, do not invent a second census.** Keep the original +`notes.md` section id. Append a dated since-last-look card. Quote the +new README *theirs*. Keep the prior quotes. Do not mint a sibling +first-sighting section. Do not invent equivalence: SHA move is not a +replica; wire-compat is not a calibrated Noul; a new bench number is +not Harbor. Namesake locks stay. Prior uniqueness locks stay one +substring. Do not reopen or amend PR #23–#44. Does not bump 0.5.0. +Merged #44 owns §121. This protocol is §122. ## Mapping index diff --git a/.agents/skills/augustus/scripts/uniqueness_gate.py b/.agents/skills/augustus/scripts/uniqueness_gate.py index 75e210ba..06d92df6 100644 --- a/.agents/skills/augustus/scripts/uniqueness_gate.py +++ b/.agents/skills/augustus/scripts/uniqueness_gate.py @@ -7,7 +7,11 @@ Each lock must appear as one consecutive substring in every listed overlay. Fragments scattered across files do not count. -Also: YAML-parse SKILL.md frontmatter; notes.md owns §114–§121; +Revisit / since-last-look protocol (`notes.md` §122) is a consecutive +substring in the skill + research files (not a 21-overlay dump wall). +Hourly must treat revisit HIGH like novel HIGH. Star-noise is not a fold. + +Also: YAML-parse SKILL.md frontmatter; notes.md owns §114–§122; composition items 289–316, 322–329, 330–336, 337–352, 353–368, and 369–384 exist; findings batches #97–#104 exist. Items 317–321 stay unused. CHANGELOG.md must not hold uniqueness dump walls (dumps live in @@ -18,6 +22,7 @@ """ from pathlib import Path +import subprocess import sys import yaml @@ -129,6 +134,24 @@ 'Hourly 1143 uniqueness lock: open recreation ≠ calibrated replica; Qwen3.5-4B ≠ Archer; It is an open re-creation of Jev; less calibrated; perch 164★ MIT HEAD ba775a9940b6 README SHA 7ad0403b; semantic lint is a sensor not a proof; oxlint-plugin-jev cutoff 0.8 still soft; nassim-arifette/jevgrep ≠ Bentlybro/jevgrep ≠ can1357/jegrep ≠ uehaj/jev-semgrep; patdown fuzzy linter; PanAchy/jevvy ≠ Atominac/jevvy; No orders, no advice; SmartMoney-Cub 25★ HEAD d93cf493853d; paired bootstrap CIs *theirs*; emretheus/jev-rag-benchmark ≠ erendikmenn/jev-rag-benchmark; +0.82 pts XQuAD-EN 95% CI +0.35 to +1.31; +7.62 pts SciFact CI +4.88 to +10.38; Same accuracy, 35x faster *theirs*; systems comparison ≠ semantic equivalence; BANKING77 500 Jev 81.0% GPT-OSS 82.8% Mercury 73.2% Gemini 85.4% *theirs*; frozen cascade missed its evaluation accuracy target 430/500 vs GPT-5 432/500; This is not demonstrated equal-quality savings; 24 invented tickets; Routing errors caught by the gate 0 of 3; sample too small to establish calibration; This is not TypeSafe Jev; No real API requests were made; wire-compat ≠ replica; KonghaYao/laya-jev 按官方接口写的客户端只改一个 base URL; gqgs/laya-onnx densify 496.8 MiB; tozp/laya-onnx ≠ Mattepiu/laya-onnx ≠ gqgs/laya-onnx; serving substrate ≠ calibrated replica; BeatAPI/awesome-jev ≠ 99hansling/awesome-jev ≠ Vishnurr2k01/awesome-jev ≠ robokrunch/awesome-jev ≠ rudy2steiner/awesome-jev-hub; All 125 projects; catalog ≠ endorsement; Pasblinn/jev-lab ≠ tanayvasishtha/jev-lab ≠ dairui1/jev-lab ≠ BrendanH18/jev-lab ≠ yibie/laya-jev-lab ≠ q93304989-bit/jev-lab; Independent project. Not affiliated with TypeSafe; Kevthetech143/super-jev densify experimental V0.2.0; permission ≠ confidence; allay-team/openjev ≠ piyush-infocusp/openjev ≠ TheoLeeCJ/openjev; 2022 Mineflayer Jevalent collision; kushalpatil/jevify-gemma4-e4b GGUF densify; static quants; This dataset and model are independent research artifacts, not reproductions of Jev or RLCD; pngwn demo accuracy 0.705 ECE 0.046 ~112 ms *theirs*; cutoff 0.8 still soft; soft scores ≠ hard gates; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43; notes.md §121' ) +REVISIT_LOCK = ( + "Revisit / since-last-look lock: catalogued repos are not done; " + "store fingerprints default_sha, pushed_at, description_hash, release_tag; " + "material change is README/API/release/calibration claim/serving port/bench rewrite; " + "star-noise is stars/likes/forks alone; densify the prior notes section, " + "do not mint a sibling first sighting; do not invent equivalence; " + "SHA move is not a replica; treat revisit HIGH like novel HIGH for Augustus; " + "notes.md §122" +) + +REVISIT_OVERLAYS = [ + ".agents/skills/augustus/SKILL.md", + "research/notes.md", + "research/README.md", + "research/revisit-checklist.md", + "CONTRIBUTING.md", +] + OVERLAYS = [ "research/notes.md", "research/changelog-hourly.md", @@ -190,6 +213,14 @@ def main() -> int: failed.append(f"1049 lock missing as one substring: {rel}") if UNIQ_1143 not in body: failed.append(f"1143 lock missing as one substring: {rel}") + for rel in REVISIT_OVERLAYS: + path = ROOT / rel + if not path.is_file(): + failed.append(f"missing revisit overlay {rel}") + continue + body = path.read_text(encoding="utf-8") + if REVISIT_LOCK not in body: + failed.append(f"revisit lock missing as one substring: {rel}") notes = (ROOT / "research/notes.md").read_text(encoding="utf-8") if "## 114. Hourly 0843 HIGH" not in notes: failed.append("notes.md missing §114 heading") @@ -207,6 +238,8 @@ def main() -> int: failed.append("notes.md missing §120 heading") if "## 121. Hourly 1143 HIGH" not in notes: failed.append("notes.md missing §121 heading") + if "## 122. Revisit / since-last-look" not in notes: + failed.append("notes.md missing §122 heading") algebra = (ROOT / ".agents/skills/augustus/references/composition-algebra.md").read_text( encoding="utf-8" ) @@ -316,6 +349,10 @@ def main() -> int: "paired bootstrap CIs *theirs*", "Same accuracy, 35x faster *theirs*", "hourly 1143 / notes.md §121", + "revisit HIGH / since-last-look", + "catalogued repo changed", + "star-noise vs material change", + "densify prior notes without inventing equivalence", ): if frag not in proto_line: failed.append(f"SKILL.md protocol missing {frag!r}") @@ -350,6 +387,42 @@ def main() -> int: eco = (ROOT / "docs/ecosystem.md").read_text(encoding="utf-8") if "decision-circuits tutorial. `notes.md` §114." not in eco: failed.append("docs/ecosystem.md 0843 blurb must cite notes.md §114") + if REVISIT_LOCK in changelog: + failed.append( + "CHANGELOG.md holds revisit lock dump " + "(protocol lock lives in SKILL.md + research/)" + ) + helper = ROOT / "research/revisit_fingerprints.py" + store = ROOT / "research/revisit_fingerprints.json" + if not helper.is_file(): + failed.append("missing research/revisit_fingerprints.py") + if not store.is_file(): + failed.append("missing research/revisit_fingerprints.json") + else: + try: + r = subprocess.run( + [sys.executable, str(helper), "--self-test"], + cwd=str(ROOT), + capture_output=True, + text=True, + check=False, + ) + except OSError as e: + failed.append(f"revisit_fingerprints.py could not run: {e}") + else: + if r.returncode != 0: + failed.append( + "revisit_fingerprints.py --self-test failed: " + + (r.stdout + r.stderr).strip() + ) + skill_headings = (ROOT / ".agents/skills/augustus/SKILL.md").read_text( + encoding="utf-8" + ) + if "## REVISIT / since last look" not in skill_headings: + failed.append("SKILL.md missing REVISIT / since last look heading") + contributing = (ROOT / "CONTRIBUTING.md").read_text(encoding="utf-8") + if "revisit HIGH like novel HIGH" not in contributing: + failed.append("CONTRIBUTING.md missing revisit HIGH like novel HIGH") if failed: print("uniqueness-gate FAIL") for line in failed: @@ -362,7 +435,9 @@ def main() -> int: f"0940 chars={len(UNIQ_0940)} 0947 chars={len(UNIQ_0947)} " f"1049 chars={len(UNIQ_1049)} " f"1143 chars={len(UNIQ_1143)} " - f"overlays={len(OVERLAYS)}" + f"revisit chars={len(REVISIT_LOCK)} " + f"overlays={len(OVERLAYS)} " + f"revisit_overlays={len(REVISIT_OVERLAYS)}" ) return 0 diff --git a/CHANGELOG.md b/CHANGELOG.md index 05f889ec..97132908 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -16,6 +16,28 @@ folds: `research/notes.md`. ## [Unreleased] +Revisit / since-last-look protocol (`research/notes.md` §122). +Does **not** bump the 0.5.0 pin. Catalogued repos get a densify +card when fingerprints move. Star-noise is not a fold. Treat +revisit HIGH like novel HIGH. uniqueness_gate.py checks the +protocol substring in the skill and research files (not a +21-overlay dump). Do not reopen or amend PR #23–#44. Merged #44 +owns §121. + +### Added + +- **Revisit / since-last-look protocol (`notes.md` §122).** Store + fingerprints `default_sha`, `pushed_at`, `description_hash`, + `release_tag` so hourly can diff. Material change is README / + API / release / calibration claim / serving port / bench + rewrite. Star-noise is stars / likes / forks alone. Densify + the prior notes section; do not mint a sibling first + sighting; do not invent equivalence; SHA move is not a + replica. Treat revisit HIGH like novel HIGH for Augustus. + Helper: `research/revisit_fingerprints.py`. Checklist: + `research/revisit-checklist.md`. **HARD RULE:** do not reopen + or amend PR #23–#44. Does **not** bump 0.5.0. + Hourly 1143 HIGH (`research/notes.md` §121 / composition items 369–384 / findings batch #104). Does **not** bump the 0.5.0 pin. Uniqueness dumps live in diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index fa62c5c2..47116582 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -28,6 +28,7 @@ Run what you can locally: ```bash python3 .agents/skills/augustus/scripts/evaluate_decisions.py --self-test python3 .agents/skills/augustus/scripts/uniqueness_gate.py +python3 research/revisit_fingerprints.py --self-test ``` The Pages workflow must stay green. After #34 it greps `_site/index.html` @@ -43,10 +44,21 @@ re-opened as "new." Before folding: - Read `research/notes.md` and the uniqueness fragments in `.agents/skills/augustus/SKILL.md` - Do not re-fold an already-landed section as a new beat -- Do not reopen or amend a merged fold PR (#23–#43) -- uniqueness_gate.py checks 0843 + 0915 + jcr + 0922 + 0940 + 0947 + 1049 + 1143 consecutive locks. +- Do not reopen or amend a merged fold PR (#23–#44) +- uniqueness_gate.py checks 0843 + 0915 + jcr + 0922 + 0940 + 0947 + 1049 + 1143 consecutive locks, plus the revisit / since-last-look protocol substring in the skill and research files. - Hourly uniqueness dump: `research/changelog-hourly.md` (archive, not release notes) +- Treat **revisit HIGH like novel HIGH**. Catalogued repos are not + done. If fingerprints moved (`default_sha`, `pushed_at`, + `description_hash`, `release_tag`) or the README / API / release / + calibration claim / serving port / bench rewrote, densify the prior + notes card. Stars / likes / forks alone is star-noise, not a fold. + Do not mint a sibling first sighting. Do not invent equivalence. + SHA move is not a replica. Checklist: + `research/revisit-checklist.md`. Helper: + `research/revisit_fingerprints.py`. `notes.md` §122. + +Revisit / since-last-look lock: catalogued repos are not done; store fingerprints default_sha, pushed_at, description_hash, release_tag; material change is README/API/release/calibration claim/serving port/bench rewrite; star-noise is stars/likes/forks alone; densify the prior notes section, do not mint a sibling first sighting; do not invent equivalence; SHA move is not a replica; treat revisit HIGH like novel HIGH for Augustus; notes.md §122 ## Secrets diff --git a/README.md b/README.md index b8409fd1..ebc20ff4 100644 --- a/README.md +++ b/README.md @@ -74,8 +74,9 @@ One line per file. The living catalog is in the reference cards and - `.agents/skills/augustus/references/optimizer-integration.md`: Jev inside Ax/DSPy optimizer loops - `.agents/skills/augustus/references/agent-self-assessment.md`: agent self-supervision gates (pre-action, done, stuck) - `.agents/skills/augustus/scripts/evaluate_decisions.py`: offline Brier / reliability / cost-threshold evaluator -- `research/notes.md`: living hourly catalog (dense); §118 llm-to-jev conversion assistant (heuristic on-ramp, not a replica) +- `research/notes.md`: living hourly catalog (dense); §118 llm-to-jev conversion assistant (heuristic on-ramp, not a replica); §122 revisit / since-last-look protocol - `research/README.md`: evidence archive index (sources, refresh log, hourly dumps) +- `research/revisit-checklist.md`: revisit already-catalogued repos when fingerprints move (revisit HIGH like novel HIGH) - `research/changelog-hourly.md`: hourly uniqueness dumps after v0.3.0 - Hourly 0743 uniqueness dump: [`research/changelog-hourly.md`](research/changelog-hourly.md) (`notes.md` §113 / batch #96). - Hourly 0843 uniqueness lock: A hunch is a probability with a policy attached; { enter: 0.8, exit: 0.6 } is hysteresis; replay a policy change without inference; Decision models are providers, not the product; huncho ≠ Kungie/gut ≠ carldaws/hunch ≠ tpellet/hunch; pretrained Qwen2.5 base ECE 0.030 (0.5B) / 0.040 (7B); instruct 0.302 / 0.269; 70.9% → 70.0% mean conf 74.1% → 96.7%; temperature scaling still matches it in-distribution; No Jev API was called; Qwen2.5 ≠ Archer; Qwen/Qwen3.8-27B ≠ Archer; 学習済みモデル v0.1 は準備中です; bool AUROC 0.523; 先頭だと0件、末尾だと250件; 温度を渡さない場合、確率は較正されていません; このリポジトリには Jev を呼ぶコードが存在しません; g0runmezadam/what-is-jev IS tunahansahin897/what-is-jev (same GitHub id 1378007307); 947 repos scored; A 273 · B 302 · C 372; LLM rubric ≠ benches; Data as of 2026-09-20; HEAD 895b9498; README SHA 3ae98c56; 13 focused checks and one mutually exclusive outcome; Probabilities are advisory, not calibrated guarantees; omni-/ask-jev ≠ pedroknigge/mcp_jev; pd.cut bins by equal width while jeval bins by quantile; ECE 0.113 and ECE 0.076; jeval drift is not implemented yet; rlaope/jeval ≠ dayhaysoos/jevals; calibration does not compose; ECE has exactly zero statistical power to detect the failure mode that kills trajectories; 25–60× headline withdrawn; P(all-correct): 0.0071 vs 0.0001; TCE / AMS; Qwen 3.8 sparring ≠ Archer; Deferred Crispification; light_cutoff_applied_to_combination 0; recorded run, kinematic animation; BANKING77 Accuracy BERT-Base 93.02 Jev 79.90; Analyse jev calibration (NLL, ECE) backlog; BERT figures are published supervised references, not zero-shot; 档位措辞效应 分数极差中位 0.50、最大 1.32; 修好后对照组是 0.01; 不是 benchmark; 概率没做 calibration; ~1,430 API calls, about $0.15; xiaohuaxi/jev-study ≠ wjdjdakf17/jev-study ≠ baekenough/jev-study; AND: product (independence assumed and recorded in the trace); chat model's stated confidence is not calibrated; circuit-vl-4b ≠ Archer; Bring your own API key; vamsikrishna2421/jev-usecases ≠ whyashthakker/awesome-jev-use-cases; catalog ≠ endorsement; SemIf 2237★ (+30 vs §111 2207); jevlike 1049★ (+6 vs 1043); TypeLLM/TypeLLM 16★; AnotiaWang 98★; yibie/awesome-jev 520★ (+14 vs 506); Laya likes 861 (was 822); tracker likes 66 (+2 vs 64) lastModified UNCHANGED; Blackwood likes 2 gated manual; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35; notes.md §114 diff --git a/research/README.md b/research/README.md index 04563f00..4ffa5cc1 100644 --- a/research/README.md +++ b/research/README.md @@ -13,6 +13,13 @@ refreshes diff against a known baseline instead of re-discovering the world. - `archive/curriculum/` — attached research briefs folded into the skill (formal-methods × System One, mental-models across domains, source list). Provenance; the skill cards are the doctrine. +- `revisit-checklist.md` — since-last-look protocol. Treat revisit HIGH + like novel HIGH. Material change vs star-noise. How densify cards + update prior notes without inventing equivalence. +- `revisit_fingerprints.json` — last-look snapshots (`default_sha`, + `pushed_at`, `description_hash`, `release_tag`) so hourly can diff. +- `revisit_fingerprints.py` — offline classifier. `--self-test` does + not fetch the network. SHA move is not a replica. Method (2026-09-18, UTC): built-in web search + direct docs reads (docs.typesafe.ai via llms.txt, GitHub READMEs, launch coverage, X posts via @@ -34,3 +41,14 @@ result pages, sort by updated/created, never by stars alone. Hourly dumps (2026-09-18T14): when a live X+GH scan is provided, archive it verbatim under `archive/hourly/` before distilling into notes.md. Do not treat X ECE/latency claims as Contract until reproduced. + +Revisit / since last look (2026-09-20 standing order): already-catalogued +repos are not done. Hourly diffs fingerprints against the last look. +Material change (README / API / release / calibration claim / serving +port / bench rewrite) is revisit HIGH: densify the prior `notes.md` +section, do not mint a sibling first sighting, do not invent +equivalence. Star-noise (stars / likes / forks alone) is a pulse, not +a fold. Hourly fold prompts must treat revisit HIGH like novel HIGH +for Augustus. `notes.md` §122. + +Revisit / since-last-look lock: catalogued repos are not done; store fingerprints default_sha, pushed_at, description_hash, release_tag; material change is README/API/release/calibration claim/serving port/bench rewrite; star-noise is stars/likes/forks alone; densify the prior notes section, do not mint a sibling first sighting; do not invent equivalence; SHA move is not a replica; treat revisit HIGH like novel HIGH for Augustus; notes.md §122 diff --git a/research/changelog-hourly.md b/research/changelog-hourly.md index 5696d6c5..7538c8f5 100644 --- a/research/changelog-hourly.md +++ b/research/changelog-hourly.md @@ -13,6 +13,23 @@ This is the uniqueness-lock archive after hourly folds (#2–#40 / notes --- +## Revisit / since-last-look protocol (notes.md §122) + +- Fresh PR off latest `main` after merged #44 (1143 / §121). **HARD RULE:** + do not reopen or amend PR #23–#44. Does not bump 0.5.0. Skip Archer. + Quote *theirs*. No wrappers. `invented_signal: false`. +- Treat revisit HIGH like novel HIGH. Fingerprints: `default_sha`, + `pushed_at`, `description_hash`, `release_tag`. Material change is + README / API / release / calibration claim / serving port / bench + rewrite. Star-noise is stars / likes / forks alone. Densify the prior + notes section. Do not mint a sibling first sighting. SHA move is not + a replica. +- Helper: `research/revisit_fingerprints.py`. Checklist: + `research/revisit-checklist.md`. uniqueness_gate checks the protocol + substring in the skill and research files (not a 21-overlay dump). + +--- + ## Hourly 1143 HIGH (notes.md §121 / items 369–384 / batch #104) - Fresh PR off latest `main` after merged #43 (1049 / §120). **HARD RULE:** diff --git a/research/notes.md b/research/notes.md index 4143a6e8..eb6795eb 100644 --- a/research/notes.md +++ b/research/notes.md @@ -30299,3 +30299,51 @@ Hooks for the reviewer: + 1143). No live Jev key. No wrappers. Hourly 1143 uniqueness lock: open recreation ≠ calibrated replica; Qwen3.5-4B ≠ Archer; It is an open re-creation of Jev; less calibrated; perch 164★ MIT HEAD ba775a9940b6 README SHA 7ad0403b; semantic lint is a sensor not a proof; oxlint-plugin-jev cutoff 0.8 still soft; nassim-arifette/jevgrep ≠ Bentlybro/jevgrep ≠ can1357/jegrep ≠ uehaj/jev-semgrep; patdown fuzzy linter; PanAchy/jevvy ≠ Atominac/jevvy; No orders, no advice; SmartMoney-Cub 25★ HEAD d93cf493853d; paired bootstrap CIs *theirs*; emretheus/jev-rag-benchmark ≠ erendikmenn/jev-rag-benchmark; +0.82 pts XQuAD-EN 95% CI +0.35 to +1.31; +7.62 pts SciFact CI +4.88 to +10.38; Same accuracy, 35x faster *theirs*; systems comparison ≠ semantic equivalence; BANKING77 500 Jev 81.0% GPT-OSS 82.8% Mercury 73.2% Gemini 85.4% *theirs*; frozen cascade missed its evaluation accuracy target 430/500 vs GPT-5 432/500; This is not demonstrated equal-quality savings; 24 invented tickets; Routing errors caught by the gate 0 of 3; sample too small to establish calibration; This is not TypeSafe Jev; No real API requests were made; wire-compat ≠ replica; KonghaYao/laya-jev 按官方接口写的客户端只改一个 base URL; gqgs/laya-onnx densify 496.8 MiB; tozp/laya-onnx ≠ Mattepiu/laya-onnx ≠ gqgs/laya-onnx; serving substrate ≠ calibrated replica; BeatAPI/awesome-jev ≠ 99hansling/awesome-jev ≠ Vishnurr2k01/awesome-jev ≠ robokrunch/awesome-jev ≠ rudy2steiner/awesome-jev-hub; All 125 projects; catalog ≠ endorsement; Pasblinn/jev-lab ≠ tanayvasishtha/jev-lab ≠ dairui1/jev-lab ≠ BrendanH18/jev-lab ≠ yibie/laya-jev-lab ≠ q93304989-bit/jev-lab; Independent project. Not affiliated with TypeSafe; Kevthetech143/super-jev densify experimental V0.2.0; permission ≠ confidence; allay-team/openjev ≠ piyush-infocusp/openjev ≠ TheoLeeCJ/openjev; 2022 Mineflayer Jevalent collision; kushalpatil/jevify-gemma4-e4b GGUF densify; static quants; This dataset and model are independent research artifacts, not reproductions of Jev or RLCD; pngwn demo accuracy 0.705 ECE 0.046 ~112 ms *theirs*; cutoff 0.8 still soft; soft scores ≠ hard gates; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43; notes.md §121 + +## 122. Revisit / since-last-look protocol (2026-09-20 standing order) + +Basit standing order: Augustus revisits already-catalogued repos when +they change since last look, not only first sightings. Things change +a lot. Protocol on a **fresh PR off latest `main`** after merged #44 +(hourly 1143, `notes.md` §121 / items 369–384 / batch #104). **HARD +RULE:** do not reopen or amend PR #23–#44. Do **not** bump 0.5.0. +`invented_signal: false`. + +Hourly fold prompts must treat **revisit HIGH** like **novel HIGH** +for Augustus. Skip is for star-noise and name collisions, not for "we +already have a card." + +Revisit / since-last-look lock: catalogued repos are not done; store fingerprints default_sha, pushed_at, description_hash, release_tag; material change is README/API/release/calibration claim/serving port/bench rewrite; star-noise is stars/likes/forks alone; densify the prior notes section, do not mint a sibling first sighting; do not invent equivalence; SHA move is not a replica; treat revisit HIGH like novel HIGH for Augustus; notes.md §122 + +### Fingerprints (hourly diffs these) + +Store in `research/revisit_fingerprints.json`. Helper +`research/revisit_fingerprints.py --self-test` (offline, no network). + +- `default_sha`: full default-branch HEAD (not a 12-char prefix) +- `pushed_at`: GitHub push clock (not the commit timestamp) +- `description_hash`: sha256[:12] of the repo / Space description, or + null when notes do not quote it. README SHA is an optional extra, + not a substitute. +- `release_tag`: latest release tag, or null + +### Material vs star-noise + +Material (revisit HIGH): README rewrite, API / primitive surface, +release tag, calibration claim, serving port or bottle, bench rewrite. +Star-noise (pulse only): stars, watchers, forks, Hub likes, likes-only +jitter with `lastModified UNCHANGED`. A star jump on a thin README is +still star-noise. A SHA move on a 0-star repo is still material. + +### Densify cards update prior notes + +1. Keep the original section id. Append a dated since-last-look card. +2. Do not mint a sibling first-sighting section for the same source. +3. Quote the new README / API / release *theirs*. Keep the prior quotes. +4. Do not invent equivalence. SHA move is not a replica. Wire-compat is + not a calibrated Noul. A new bench number is not Harbor. +5. Namesake locks stay. Prior uniqueness locks stay one substring. +6. Class-relevant revisit HIGH gets the same overlay care as novel HIGH. + +Checklist: `research/revisit-checklist.md`. Skill heading: **REVISIT / +since last look**. Merged #44 owns §121. This protocol is §122. diff --git a/research/refresh-log.md b/research/refresh-log.md index 85e4241e..53566fc9 100644 --- a/research/refresh-log.md +++ b/research/refresh-log.md @@ -1,4 +1,20 @@ +## 2026-09-20 ~18:17 UTC / ~12:17 Boise - Revisit / since-last-look protocol +- Fresh PR off latest `main` after merged #44 (1143 / `notes.md` §121 / + items 369–384 / batch #104). This is `notes.md` §122, a protocol, not + a census fold. **HARD RULE:** do not reopen or amend PR #23–#44. + Does not bump 0.5.0. Skip Archer. Quote *theirs*. `invented_signal: false`. +- PRIMARY: catalogued repos are not done. Treat revisit HIGH like novel HIGH. + Fingerprints: `default_sha`, `pushed_at`, `description_hash`, `release_tag`. + Material change is README / API / release / calibration claim / serving + port / bench rewrite. Star-noise is stars / likes / forks alone. + Densify the prior notes section. Do not mint a sibling first sighting. + SHA move is not a replica. +- uniqueness_gate 0843+0915+jcr+0922+0940+0947+1049+1143 stays. Revisit lock + is a 5-file protocol substring (skill + research), not a 21-overlay dump. +- Helper: `research/revisit_fingerprints.py --self-test`. Checklist: + `research/revisit-checklist.md`. + ## 2026-09-20 ~17:43 UTC / ~11:43 Boise - Hourly 1143 HIGH - Fresh PR off latest `main` after merged #43 (1049 / `notes.md` §120 / items 353–368 / batch #103). Next free IDs: diff --git a/research/revisit-checklist.md b/research/revisit-checklist.md new file mode 100644 index 00000000..b2ed53c0 --- /dev/null +++ b/research/revisit-checklist.md @@ -0,0 +1,84 @@ +# Revisit / since last look + +Catalogued repos are not done. Things change a lot. Hourly must +diff fingerprints against the last look and densify when the change +is material. + +Revisit / since-last-look lock: catalogued repos are not done; store fingerprints default_sha, pushed_at, description_hash, release_tag; material change is README/API/release/calibration claim/serving port/bench rewrite; star-noise is stars/likes/forks alone; densify the prior notes section, do not mint a sibling first sighting; do not invent equivalence; SHA move is not a replica; treat revisit HIGH like novel HIGH for Augustus; notes.md §122 + +## Fingerprints to store + +Default set, so hourly can diff without re-discovering the world: + +| Field | What it catches | +| --- | --- | +| `default_sha` | Full default-branch HEAD moved (not a 12-char prefix) | +| `pushed_at` | GitHub push clock moved (not the commit timestamp) | +| `description_hash` | sha256[:12] of the GitHub / Space description. Null if notes do not quote it. | +| `release_tag` | Latest release tag moved or appeared | + +Optional extras after a fingerprint hit (not substitutes for the four): +README SHA, OpenAPI / `/v1/systemone` surface, serving port or base URL, +calibration claim, bench rewrite. README SHA is not `description_hash`. + +Store lives in [`revisit_fingerprints.json`](revisit_fingerprints.json). +Helper: `python3 research/revisit_fingerprints.py --self-test`. + +## Material change vs star-noise + +**Material (revisit HIGH, treat like novel HIGH):** + +- README rewrite (claims, API, install surface, serving port) +- API / primitive / `/v1/systemone` contract change +- New or moved release tag +- Calibration claim appeared, retracted, or restated +- Serving port, base URL, or bottle (GGUF / ONNX / MLX / ggmlc) +- Bench rewrite (new n, new metric, retracted headline) + +**Star-noise (pulse only, not a fold):** + +- Stars, watchers, forks, Hub likes +- `lastModified UNCHANGED` with likes-only jitter +- Issue-count flap, traffic graphs + +A star jump on a thin README is still star-noise. A SHA move on a +0-star repo is still material. + +## How densify cards update prior notes + +1. Keep the original `notes.md` section id. Append a dated **since last + look** densify card under that section. +2. Do not mint a sibling first-sighting section for the same source. + Rename densify is still densify (SemIf §117), not a second census. +3. Quote the new README / API / release *theirs*. Keep the prior quotes + so the card shows what changed. +4. Do not invent equivalence. SHA move is not a replica. Wire-compat is + not a calibrated Noul. A new bench number is not Harbor. Argmax + agree is not semantic equivalence. +5. Namesake locks stay (`a/x ≠ b/x`). Prior uniqueness locks stay one + consecutive substring. Do not mutate them. +6. Class-relevant revisit HIGH gets the same overlay care as novel HIGH + (notes, skill mapping if the class table moved, uniqueness if this + hour is a fold). Skip is for star-noise and collisions, not for "we + already have a card." + +## Paste into hourly fold prompts + +Treat **revisit HIGH** like **novel HIGH** for Augustus. + +If fingerprints moved on a catalogued repo (`default_sha`, +`pushed_at`, `description_hash`, `release_tag`) or the README / API / +release / calibration claim / serving port / bench rewrote, densify +the prior notes card. Do not skip because it was already catalogued. +Stars / likes / forks alone is star-noise, not a fold. Do not invent +equivalence. SHA move is not a replica. Soft Noul ≠ hard gate. + +## Offline check + +```bash +python3 research/revisit_fingerprints.py --self-test +python3 .agents/skills/augustus/scripts/uniqueness_gate.py +``` + +Does not bump 0.5.0. Does not fetch the network. Merged #44 owns +§121. This protocol is §122. diff --git a/research/revisit_fingerprints.json b/research/revisit_fingerprints.json new file mode 100644 index 00000000..c3edf43f --- /dev/null +++ b/research/revisit_fingerprints.json @@ -0,0 +1,50 @@ +{ + "schema_version": 1, + "stored_fields": [ + "default_sha", + "pushed_at", + "description_hash", + "release_tag" + ], + "material_signals": [ + "readme", + "api", + "release", + "calibration_claim", + "serving_port", + "bench_rewrite" + ], + "star_noise_fields": [ + "stargazers_count", + "watchers_count", + "forks_count", + "likes" + ], + "note": "Last-look snapshots for already-catalogued sources. Hourly diffs these four fingerprints. Star-noise is not a fold. SHA move is not a replica. Seeded from merged notes (SemIf §117, NanoJev §115), not a live census this pass. default_sha is the full HEAD from those cards. pushed_at is the GitHub push clock from §117 (not the commit timestamp). description_hash is sha256[:12] of the GitHub/Space description, or null when notes do not quote it. README SHA is an optional extra, not a substitute for description_hash.", + "looks": [ + { + "id": "github:TheoLeeCJ/SemIf", + "notes_section": "117", + "last_look": "2026-09-20T16:10Z", + "fingerprints": { + "default_sha": "ca3ba65f142967030ecb453346e94d6f476a69df", + "pushed_at": "2026-09-19T04:46:36Z", + "description_hash": null, + "release_tag": null + }, + "readme_sha": "74ab7f7ffa3d492c2dc905c1e75410dca3e0677e" + }, + { + "id": "github:TianyuCodings/NanoJev", + "notes_section": "115", + "last_look": "2026-09-20T15:15Z", + "fingerprints": { + "default_sha": "618cea6d906d54e128360786d12f703fff2b1245", + "pushed_at": null, + "description_hash": null, + "release_tag": "unified-games-v1" + }, + "readme_sha": "4190093c64ee75b26e9726daa3b00cbcf6d3157a" + } + ] +} diff --git a/research/revisit_fingerprints.py b/research/revisit_fingerprints.py new file mode 100644 index 00000000..e1ee5fe6 --- /dev/null +++ b/research/revisit_fingerprints.py @@ -0,0 +1,312 @@ +#!/usr/bin/env python3 +"""Since-last-look fingerprint helper. + +Hourly diffs catalogued sources against the last look. A fingerprint +move is revisit HIGH. Star-noise is not a fold. This script does not +fetch the network. It does not treat a SHA move as a replica. It does +not invent equivalence. + +Stored fingerprints (default): + default_sha, pushed_at, description_hash, release_tag + +Material change (inspect when fingerprints move, or when hourly already +saw the rewrite): + README, API, release, calibration claim, serving port, bench rewrite + +Star-noise (not a fold): + stargazers_count, watchers_count, forks_count, likes + +Usage: + python3 research/revisit_fingerprints.py --self-test + python3 research/revisit_fingerprints.py --classify stored.json observed.json +""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import sys +from pathlib import Path +from typing import Any, Mapping + +ROOT = Path(__file__).resolve().parents[1] +STORE = ROOT / "research" / "revisit_fingerprints.json" + +STORED_FIELDS = ( + "default_sha", + "pushed_at", + "description_hash", + "release_tag", +) + +MATERIAL_SIGNALS = ( + "readme", + "api", + "release", + "calibration_claim", + "serving_port", + "bench_rewrite", +) + +STAR_NOISE_FIELDS = ( + "stargazers_count", + "watchers_count", + "forks_count", + "likes", +) + +GRADES = ("unchanged", "star_noise", "material") + + +def description_hash(text: str) -> str: + return hashlib.sha256(text.encode("utf-8")).hexdigest()[:12] + + +def _fp(record: Mapping[str, Any]) -> dict[str, Any]: + src = record.get("fingerprints", record) + return {k: src.get(k) for k in STORED_FIELDS} + + +def classify( + stored: Mapping[str, Any], + observed: Mapping[str, Any], + *, + material_signals: tuple[str, ...] = (), +) -> dict[str, Any]: + """Diff last-look fingerprints against this hour. + + Returns grade, reasons, and the densify action. Unknown material + signal names fail closed (raise). Unknown grades cannot be returned. + """ + reasons: list[str] = [] + before = _fp(stored) + after = _fp(observed) + for field in STORED_FIELDS: + if before.get(field) != after.get(field): + reasons.append(field) + for signal in material_signals: + if signal not in MATERIAL_SIGNALS: + raise AssertionError(f"unknown material signal: {signal}") + tag = f"material:{signal}" + if tag not in reasons: + reasons.append(tag) + + noise_moved: list[str] = [] + for field in STAR_NOISE_FIELDS: + if field in stored or field in observed: + if stored.get(field) != observed.get(field): + noise_moved.append(field) + + if reasons: + grade = "material" + action = "revisit_high_like_novel_high" + elif noise_moved: + grade = "star_noise" + action = "pulse_only" + reasons = list(noise_moved) + else: + grade = "unchanged" + action = "skip" + + if grade not in GRADES: + raise AssertionError(f"unclassified grade {grade}") + return { + "grade": grade, + "reasons": reasons, + "action": action, + "invents_equivalence": False, + } + + +def densify_card_rules() -> tuple[str, ...]: + return ( + "Keep the original notes.md section id. Append a dated densify card.", + "Do not mint a sibling first-sighting section for the same source.", + "Quote the new README/API/release *theirs*. Keep the prior quotes.", + "Do not invent equivalence. SHA move is not a replica.", + "Wire-compat is not a calibrated Noul. A new bench number is not Harbor.", + "Namesake locks stay. Prior uniqueness locks stay one substring.", + "Treat revisit HIGH like novel HIGH for Augustus.", + ) + + +def _hex12(value: Any) -> bool: + return isinstance(value, str) and len(value) == 12 and all( + c in "0123456789abcdef" for c in value + ) + + +def load_store(path: Path | None = None) -> dict[str, Any]: + p = path or STORE + data = json.loads(p.read_text(encoding="utf-8")) + if data.get("stored_fields") != list(STORED_FIELDS): + raise AssertionError("store stored_fields mismatch") + looks = data.get("looks") + if not isinstance(looks, list): + raise AssertionError("store looks must be a list") + for look in looks: + fp = look.get("fingerprints") or {} + look_id = look.get("id") + for field in STORED_FIELDS: + if field not in fp: + raise AssertionError(f"look {look_id} missing {field}") + digest = fp.get("description_hash") + if digest is not None: + if not _hex12(digest): + raise AssertionError( + f"look {look_id} description_hash must be null or sha256[:12]" + ) + readme = look.get("readme_sha") + if isinstance(readme, str) and digest == readme[:12]: + raise AssertionError( + f"look {look_id} description_hash is README SHA prefix; " + "README SHA is an optional extra, not a substitute" + ) + sha = fp.get("default_sha") + if isinstance(sha, str) and len(sha) == 12: + raise AssertionError( + f"look {look_id} default_sha is a 12-char prefix; store the full HEAD" + ) + return data + + +def self_test() -> None: + stored = { + "id": "github:TheoLeeCJ/SemIf", + "fingerprints": { + "default_sha": "ca3ba65f142967030ecb453346e94d6f476a69df", + "pushed_at": "2026-09-19T04:46:36Z", + "description_hash": description_hash( + "interface pattern reproduction with open models" + ), + "release_tag": None, + }, + "stargazers_count": 2282, + "likes": None, + } + same = classify(stored, dict(stored)) + assert same["grade"] == "unchanged" and same["action"] == "skip", same + assert same["invents_equivalence"] is False + + stars = dict(stored) + stars["stargazers_count"] = 2300 + noise = classify(stored, stars) + assert noise["grade"] == "star_noise" and noise["action"] == "pulse_only", noise + assert "stargazers_count" in noise["reasons"] + + likes = dict(stored) + likes["likes"] = 58 + like_noise = classify(stored, likes) + assert like_noise["grade"] == "star_noise", like_noise + assert "likes" in like_noise["reasons"] + + forks = dict(stored) + forks["forks_count"] = 141 + fork_noise = classify(stored, forks) + assert fork_noise["grade"] == "star_noise", fork_noise + assert "forks_count" in fork_noise["reasons"] + + sha = { + "fingerprints": dict(stored["fingerprints"]), + "stargazers_count": 2300, + } + sha["fingerprints"]["default_sha"] = "deadbeef0001" + moved = classify(stored, sha) + assert moved["grade"] == "material", moved + assert moved["action"] == "revisit_high_like_novel_high", moved + assert "default_sha" in moved["reasons"] + assert moved["invents_equivalence"] is False + + trunc = {"fingerprints": dict(stored["fingerprints"])} + trunc["fingerprints"]["default_sha"] = "ca3ba65f1429" + assert classify(stored, trunc)["grade"] == "material" + + pushed = {"fingerprints": dict(stored["fingerprints"])} + pushed["fingerprints"]["pushed_at"] = "2026-09-20T18:00:00Z" + assert classify(stored, pushed)["grade"] == "material" + + desc = {"fingerprints": dict(stored["fingerprints"])} + desc["fingerprints"]["description_hash"] = description_hash("new blurb") + assert classify(stored, desc)["grade"] == "material" + + unknown_desc = { + "fingerprints": { + "default_sha": "ca3ba65f142967030ecb453346e94d6f476a69df", + "pushed_at": "2026-09-19T04:46:36Z", + "description_hash": None, + "release_tag": None, + } + } + saw_desc = {"fingerprints": dict(unknown_desc["fingerprints"])} + saw_desc["fingerprints"]["description_hash"] = description_hash("live blurb") + assert classify(unknown_desc, saw_desc)["grade"] == "material" + assert classify(unknown_desc, dict(unknown_desc))["grade"] == "unchanged" + + tag = {"fingerprints": dict(stored["fingerprints"])} + tag["fingerprints"]["release_tag"] = "v2.0.0" + assert classify(stored, tag)["grade"] == "material" + + readme = classify(stored, dict(stored), material_signals=("readme", "bench_rewrite")) + assert readme["grade"] == "material" + assert "material:readme" in readme["reasons"] + assert "material:bench_rewrite" in readme["reasons"] + + try: + classify(stored, dict(stored), material_signals=("vibes",)) + except AssertionError: + pass + else: + raise AssertionError("unknown material signal must fail closed") + + store = load_store() + assert store["schema_version"] == 1 + by_id = {look["id"]: look for look in store["looks"]} + assert "github:TheoLeeCJ/SemIf" in by_id + assert "github:TianyuCodings/NanoJev" in by_id + semif = by_id["github:TheoLeeCJ/SemIf"]["fingerprints"] + assert semif["default_sha"] == "ca3ba65f142967030ecb453346e94d6f476a69df" + assert semif["pushed_at"] == "2026-09-19T04:46:36Z" + assert semif["description_hash"] is None + assert semif["release_tag"] is None + nano = by_id["github:TianyuCodings/NanoJev"]["fingerprints"] + assert nano["default_sha"] == "618cea6d906d54e128360786d12f703fff2b1245" + assert nano["pushed_at"] is None + assert nano["description_hash"] is None + assert nano["release_tag"] == "unified-games-v1" + nano_readme = by_id["github:TianyuCodings/NanoJev"].get("readme_sha") + assert isinstance(nano_readme, str) and nano_readme.startswith("4190093c64ee") + semif_readme = by_id["github:TheoLeeCJ/SemIf"].get("readme_sha") + assert isinstance(semif_readme, str) and semif_readme.startswith("74ab7f7f") + rules = densify_card_rules() + assert any("sibling first-sighting" in r for r in rules) + assert any("revisit HIGH like novel HIGH" in r for r in rules) + print("revisit-fingerprints self-test ok") + + +def main() -> int: + ap = argparse.ArgumentParser() + ap.add_argument("--self-test", action="store_true") + ap.add_argument("--classify", nargs=2, metavar=("STORED", "OBSERVED")) + ap.add_argument( + "--material", + default="", + help="comma-separated material signals (readme,api,release," + "calibration_claim,serving_port,bench_rewrite)", + ) + args = ap.parse_args() + if args.self_test: + self_test() + return 0 + if args.classify: + stored = json.loads(Path(args.classify[0]).read_text(encoding="utf-8")) + observed = json.loads(Path(args.classify[1]).read_text(encoding="utf-8")) + signals = tuple(s for s in args.material.split(",") if s) + print(json.dumps(classify(stored, observed, material_signals=signals), indent=2)) + return 0 + ap.error("need --self-test or --classify stored.json observed.json") + return 2 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/hourly-refresh.sh b/scripts/hourly-refresh.sh index eb742b84..94af4425 100755 --- a/scripts/hourly-refresh.sh +++ b/scripts/hourly-refresh.sh @@ -2,9 +2,16 @@ # Hourly refresh during US business hours (Mon-Fri 13:00-21:00 UTC = 9am-5pm ET). # Runs the deterministic checks, re-sweeps GitHub for NEW jev repos not yet in the # archive, clones them, appends refresh-log, and pushes any repo changes to main. +# +# Revisit / since last look (notes.md §122): already-catalogued clones are +# not done. Star-noise is not a fold. Fingerprint diffs +# (default_sha, pushed_at, description_hash, release_tag) belong on the +# fold path as revisit HIGH, treated like novel HIGH. See +# research/revisit-checklist.md. Do not invent equivalence. set -u cd "$(dirname "$0")/.." scripts/refresh-jev-research.sh +python3 research/revisit_fingerprints.py --self-test || exit 1 # GitHub re-sweep: repos mentioning jev created in last 3 days, not yet archived. ARCHIVE=/home/user/workspace/jev-archive diff --git a/scripts/refresh-jev-research.sh b/scripts/refresh-jev-research.sh index 02dd266e..f217c719 100755 --- a/scripts/refresh-jev-research.sh +++ b/scripts/refresh-jev-research.sh @@ -16,5 +16,6 @@ TS=$(date -u +"%Y-%m-%d %H:%M UTC") echo "- $u -> HTTP $code" done echo "- action: diff index/cookbook list vs research/sources.json; update notes.md + log." + echo "- revisit: diff research/revisit_fingerprints.json (default_sha, pushed_at, description_hash, release_tag). Material change is revisit HIGH. Star-noise is not a fold. See research/revisit-checklist.md / notes.md §122." } >> "$LOG" echo "logged to $LOG"