diff --git a/.agents/skills/augustus/SKILL.md b/.agents/skills/augustus/SKILL.md index a549f60..9c5ea9b 100644 --- a/.agents/skills/augustus/SKILL.md +++ b/.agents/skills/augustus/SKILL.md @@ -54,7 +54,7 @@ classical method you already trust, substitute it, classify the win "paraphrase brittleness", "allowlist then judge", "TOCTOU-of-Noul", "Jev inside the database / sqlite-jev", "Jev picks bitrate / join order / the model", "wait for Archer", "lint the request / missing - other", "training confronts Choice other / none-of-the-above", "soft AGENTS.md rules vs the linter", "screenshot Choice / omni System One", "extractive quotes / pointer not generator", "compaction summarize vs pointer", "encoder vs Jev compaction backend", "shadow-mode compaction rollout", "CI flaky-vs-real merge gate", "fail-open VOI wake/resume", "claim vs session evidence", "S1 indexer escalate-S2", "Harbor on/off routing", "fail-open vs fail-closed wake vs CI gate", "encoder vs Jev computer-use backend", "hybrid local decide + remote fill", "DONE vs verified success", "stdout prune vs session compaction", "OpenCode jev-pruner vs Claude jev-pruner", "zen-chat vs jev-zen Noul", "hard envelope then Noul prune", "Cua-S1 vs TypeSafe Jev", "plan vs execute dry-run", "specialist computer-use vs general agent", "local drop-in vs stub scorer", "route vs memory", "when does it hold / extractable from state", "decision model vs constrained LLM", "dual-process S1/S2", "combinatorial grid vs extractive", "uncalibrated local likelihoods", "decision-native RAG", "classify-first / read selectively", "living applied-mappings atlas / class patterns", "silence as safer / draft-gate heartbeat", "robotics text-state vs pixels", "verbatim ledger vs summary", "judgment as language primitive", "Stagehand extract pick-and-copy", "harness observe-score-act vs demo loop", "public judgment wall / six parallel questions", "meaning-search without embeddings", "attention ≠ correctness", "skills→oxlint / AST prove ∩ remainder", "session-sticky first-prompt routing", "measured RAG rerank vs generative rerank", "capability kernel / secrets never in the agent", "Jev is SENSOR not policy", "type-safe ≠ correct", "typed control plane around DSPy", "native vs verbalized confidence", "engine owns truth / Jev owns judgment", "human-confirmed kill gate", "train specialist vs few-shot hosted", "decide→policy→LLM leftover", "Noul 0.5 cannot-tell never rounded", "calibration ≠ sortable / ORDER BY", "pairwise inversion / Score ordinality / two-decimal ties", "wire-compat GLiFormer /v1/systemone", "class-backend economics", "loopback gateway hosted + local", "do not distill Jev as teacher", "active-learning triage", "evidence-packet explorer", "meaning-grep AND/OR/NOT", "closed-vote-only / no planner LLM", "Jev vs PCD Harbor", "PCD O(1) ≠ Noul", "host-owned handlers × System One", "OMP/pi fail-open gate", "permission vs probability / operator owns thresholds", "judgment ≠ permission / Jev never grants access", "eval integrity / instrument not score", "constrained optimizer + S1 features / never sole hot-path gate", "privilege ≠ verdict / effect contracts not tokens", "attention filter / VOI for human review / never blocks / never green unless sure", "measurement owns endorsement / evidence-gated question packs", "Jev supplies evidence / code owns authority", "ranking ≠ calibration / never hard-threshold raw p as frequency", "hot-click CU / indexed element table", "Jev judges relevance / code decides structure", "local rules first then remainder / never auto-train on own hides", "combinators / System One as control plane", "receipts not leaderboard / type-safe ≠ correct jaggedness", "VOI over skill library / skillranker abstention", "OOD calibration / AUC ≠ ECE", "Jev vs thinking-budget small models", "turnstile / replayable evidence≠authority", "MLX one-pass schema→JSON / Apple Silicon replica economics", "memory leases ended by new evidence", "never confidently wrong / TLA+ compose / escalate instead of hard-gate", "no seal no advance / coverage ledger / mint ≠ product brain", "skill-broker sibling / judgment ≠ permission", "sureness bands / max_prob is generous", "JevBench / calibration not in Main Score", "CI typed gate before expensive review", "Codex MCP host adapter", "judgment as attention redirect / jev-preflight", "compress-before-first-send / dizk jev-lens", "tools≠use / SessionStart over hoping", "observational memory / pi-om keep-kind", "open-Jev class / openvons / JevPick", "physical-world System One / HA-Jev / not for locks", "judgment outside the store / jevql", "landed-script trust / headless≠auto-approve", "digital-design combinators / extended five", "VOI cache admission / same-intent skip LLM", "BM25 vs Jev skill routing Harbor harness", "zeroshot vs BERT / contamination DiD", "typed escalate continue abort baton / inverted loop", "worth-your-attention VOI / ThinkyMiner Winnow", "Jev WHETHER Python HOW LLM WHAT", "conflict vs ignorance / named Choice escape", "Playwright executes Jev chooses", "OpenJev /v1/decide not drop-in", "SemIf wire-compat runoff; SemIf rename densify / MLX backend / 5.21× systems≠semantic / Softmax ≠ Noul (`notes.md` §117)", "decision-as-memory flywheel", "record/replay CI / jevassert", "failure-finding arena / jevarena ≠ jev-arena", "BBQ not a bias cert", "decider≠executor", "sentence-as-rule lint / jevlint", "sentence-as-rule lint / jev-lint is jevlint rename", "VOI hunk prune", "whole-repo intent VERIFIED/VIOLATION/UNKNOWN", "GLiNER2 spec ≠ replica", "open replica substrates / grande / laya-jolt / JEV-CPU", "ONNX local-jev not equivalent", "persist constraints across compaction / pi-heed", "calibration+cost first-class gates", "Harbor-shaped Jev vs SGR LLM-as-judge / jev-judge-bench ≠ jevarena ≠ jevbench", "hand no-text steps / jev-use / Vercel drops confidence", "Pi System-One control plane / pi-jev-control", "never free-generates / jev-gpt tree of Choices", "OpenRouter recipe atlas / samples not benches", "personal history feed / jevfeed / no social graph", "competing NAR claims / dual-channel ECE / openJev-verdict ≠ OpenJev", "empty compaction-proxy skip / IPECTER", "throughput ≠ latency / like-for-like ECE", "1-token logprob endpoint ≠ Noul / coverage ≠ correctness", "open replica engine / jevinf", "unofficial Elixir SDK ≠ OTP peer", "jevex n=16 files-to-read VOI", "commit pre-review attention≠verdict / middle band", "Hermes plugin is Agnes not TypeSafe", "pi-jev-compact ≠ pi-jev-compaction", "empty Codex-proxy skip / IPECTER runway", "decision-native inbox / mailordinal", "unofficial jev-cli not ready / ≠ jevql", "laya-multilingual / English checkpoint confident-wrong OOD", "schema-scorer peaked ranking ≠ calibration", "HF 401 / GitHub 404 Hub-only", "productized System One HTTP / classifier.dev", "escalate-under-threshold / smart tier / multi-label ignores", "silent FALLBACK / granite 0.546 vs advertised 0.800", "vs_jev tracked JSON / read eval/README", "choxos/jev-reviewer ≠ egma-ai / systematic-review pointer", "two-pass Choice+Noul evidence extraction", "not-found is an answer", "human check as productized judgment", "githubnext/localjev ≠ kunchenguid/local-jev", "wire-compat ≠ logit-equiv / prompted JSON ≠ structured read", "self-reported probs / entropy confidence", "GitHub Next local /v1/systemone", "LM Studio runner gap / structured-read primitives", "NandhaKishorM/laya packaging ≠ Hub-only / Router script-before-p", "post-T ECE ≠ raw ECE / Banking77 token-budget", "0.85 still soft / not TypeSafe drop-in", "external census ≠ scored bake-off", "GLiNER2+routers class-boundary", "incomplete openjev census vs watch", "Harbor honesty watch / silent fallback", "JevBench v1.2 geometric mean / cal ON rank / weight sensitivity", "option-order 72→21 / instruction models in the class table", "self-host latency ×2 assumption / est. costs", "Laya absent is a gap not a named exclusion", "Qwen3.8 27B ≠ Archer", "hourly already-folded watch / apply-the-five / skip thin noise", "hard-gate Noul as PR/quality gate is soundness theater", "S1 never stalls waiting / S2 one-use advisory", "Local controller ≠ githubnext/localjev", "purple telemetry = consumed not arrived", "seed = geometry not async replay", "20% starting gate still soft", "no pixels to either provider", "OCR+AX observe-score-act / typesafe-computer-use", "never send screenshot to frontier for the decision", "overlapping CU options = false low confidence", "split kind/item/site", "155× one-screenshot ≠ Harbor taskset", "decision ≠ answer-reader capture", "ASR observe-score-act / jev-voice-browser", "partial-speech VOI / free-text waits", "spoken confirm ≠ hard auth", "numbered overlay without another model", "wrap-as-execution / AgentGhost ALLOW ASK DENY", "rules first then Jev remainder / ASK throws / fail-closed", "reddpy/AgentGhost ≠ jwen5419807/agentghost ≠ vventirozos", "JP genre atlas / studio_yebisu / stars ephemeral ≠ eval", "Jev Clearly Explained / akshay_pachaar / LLM hammer", "schema-safe ≠ correct / 200× 400× TypeSafe ceiling", "questions-as-code / shadow first / not a TypeSafe how-to", "proposition ≠ embedding / contrast-set", "boolean composition of soft Nouls / AND OR NOT", "uehaj/jev-semgrep ≠ semgrep.dev", "meaning-grep dedicated fold / not a gate", "decision-validated UI / Jev never authors text", "decision-as-assert / jevtest ambiguous band", "typed decisions drive UI / jev2ui", "hybrid S1 closed verb menu / anima3", "pointer-not-generator search / JevFind", "jev-frontier-bench ≠ frontier-100", "product bakeoff ≠ architecture duel / GLiClass", "four engines same questions / majority floor", "authorship named escape / not evidence", "ha-switchboard HA remains execution", "n8n classify/route/score / Low Confidence", "fast-jev-compaction-pi ≠ pi-jev-compact ≠ pi-jev-compaction", "jevloop full-distribution optimizer / no LLM in the loop", "laya-vision SmolVLM / score untrained", "Cerebellum-2B /v1/decide ≠ TypeSafe / wire-compat vs agent-routing", "laya-grounded not drop-in / Platt not temperature", "GestaltLabs/Jeff-1 ≠ logan-markewich/jeff / acc vs ECE n=9730", "stanley-code empty findings ≠ approval / human promote", "findme ≠ JevFind / NL memory beam-search FS", "jevsubrouter price workers not conversation / counts ≠ dollars", "feelings .feels() default 0.5 is Noul-0.5-never-rounded / ≠ hunch ≠ Probably", "apa-agent-harness ≠ AntonioCoppe/jev-harness / unpublished npm", "grok-bot-jev skill cannot force a bot that ignores it / A/B proxies not tokens", "Essentiel-Jev never authority / human every action", "enzo-mcp independently falsifiable claims / ≠ jev-sift", "pigeonhole OTHER skip / decision-as-filing", "jev-reliability Nothing about accuracy", "clduab11/jev-test ≠ realZachi/jevtest / Nothing runs yet", "jev-rag-benchmark Jev wins is not an assumption", "dairui1/jev-lab ≠ BrendanH18/jev-lab", "jevmail gmail.readonly / mailjay archive/trash", "ZHUBoer/ego-jev reserved __none__", "runWorkflow completed ≠ success", "jsort scores are relative", "Noul not Choice for scale", "groundedness-judge-bench native vs schema-guided", "implicit_true included in yes", "jev_playground 0 promotions", "routing-backtest 0.0447%", "yuyang2230/jev-agent-skill jev-1.13-free", "jev-techstack-classifier stack_config.json", "s1_ruby collapse late", "undecided? abstain", "2389-research/judgement license null", "confidence ≠ winner p", "typesafeai-sdk-community not a new species", "tpellet/hunch exit 3", "never-execute list", "jev-file-search scores not calibrated accuracy", "jev-linkmap Jev never sees S2 prose", "muhammedilyasy/jev-mail metadata only", "tidy none-of-folders stay", "tab-bouncer pinned/audio/current never closed", "lkclean Show fail-open", "jev-yt-time-saver Show anyway", "ORIGIN pause-if-no-Jev", "validResponse sums-to-1", "jev-crawlers risk bands never raw boolean", "jevbrain AUTO_ACT is not a Noul", "judgekit YAML classify/score/route/verify", "typed-judge-kit verdict-in-code", "alsoleg89/decide packing VOI", "0.8 ≠ 80% accuracy", "Jev-Calibration Platt ECE 0.117→0.052", "jev-calibration-arena never acts", "ctmx/openrouter-jev-mcp Decision-as-Plugin", "FrancoisChastel/jev-code ≠ npm jev-code", "claudecode-jev-marketplace fail-open not hot path", "pedroknigge/mcp_jev packs not ask_jev", "cyrusasco/typesafe-mcp noul deadband 0.35–0.65", "codaaiteam/jev-skill jevtypesafeai.com ≠ TypeSafe", "hermes-switchyard ≠ hermes-jev-router ≠ hermes-plugin-jev", "nanoprune 2.8MB ECE 2.58%", "smartdio/jev-browser-agent ≠ ZHUBoer/ego-jev", "Dakai/omp-jev-web DONE ≠ proof", "hari007sh/jev ≠ dannote/jev", "0thernet/system-one-skills deterministic verify", "typed-gate band [0.40,0.60] is refusal", "pi-jev-gate fail-closed; choice is the verdict", "Foq ~25ms/2.2GB local", "rev prefill-only + HF jev-0.5b", "robfrase/jev planning memo", "typesafe_agent_gates 27/27 / 31/31", "EpicEric/safe-sh static remainder", "pastepilot Confirm before act", "Jev-Reranker live Jev not yet measured", "sessionwise opt-in relevance", "jev-search pointer sieve", "400ms Salesforce WebMCP", "typesafe-scheduler-diagnostics advisory", "droidjev screenshot-free", "Tewoto1 jevcu planner still writes", "ha-conversation-jev Jev→Grok", "dsh-jev can only gate", "jev-classification-benchmark specified not run", "jev-luna-pagerduty p≥0.50", "meldltd/meldecision laya-go ONNX", "laya-doom never pixels", "logixism/laya-api empty README", "akpsahan/laya ≠ Archer", "choxos/jevchess engine owns truth", "jev-drive sim not AV", "story-arc Jev never authors", "jev-hs-assistant HS6", "golergka/jev-plays-starcraft-2 UI-verified ≠ API Victory", "awesome-jev-use-cases catalog", "Nibir1/typesafe-go ≠ official", "fingerprint after redact", "recall vs decide", "publish fingerprints+answers", "CI replay as Harbor cousin", "Cache hit ≠ correctness", "hyperspaceai/jevcache ≠ kushals256/jevcache", "human labels only", "score never auto-accepts", "production capture flywheel", "sutro-sh/jev-align ≠ caiovicentino/jev-align", "guidance ≠ hook", "catalysts ≠ summaries", "compile-time System One", "unofficial ≠ TypeSafe", "format_version modernbert-jev/1", "Argos1111/jev_local ≠ us/jev-local ≠ kunchenguid/local-jev", "LFM default ≠ ModernBERT backend", "Nemotron ≠ TypeSafe Jev", "not a calibrated replacement", "djev-dev complements djev-spark", "images as Choice options", "Laya essay numbers *theirs*", "Router/OOD confidence", "hosted bootstrap ≠ silent TypeSafe", "difficulty + policy thresholds + JSONL trace", "jev-codex-pilot model + reasoning depth", "keep/shadow/hybrid/reject", "quarry evidence projection", "Frank-ZY-Dou/awesome-jev robotics/3D/control", "one-dollar-tahoe TypeSafe Jev defense eval", "jevguard calibrator/cache/escape", "jev-ci-selector CI shadow mode", "llama-jev llama.cpp replica", "petercr/jev-orchestrator ≠ FleeexCorp/jev-orchestrator", "seb4ez/jevguard ≠ AseemPrasad/JevGuard ≠ pablozr/JevGuard", "webNeat/llama-jev ≠ WiktorB2004/llama-index-jev", "OpenCode jev-pruner context sieve", "observe→score-candidates→prune", "jev-zen / jev-1.13-free", "zen-chat ≠ Noul", "fail-open original", "keepScore >0.1 floor", "host port of tamaratran/jev-pruner", "indiejoseph/opencode-jev-pruner ≠ nrdz-labs/fast-jev-opencode", "jev-webagent-bench empty stub", "Kiln-AI/jev_jsonschema noul_threshold 0.5", "NSStudent/JevSwiftSDK unofficial", "GLiNER2 native Apple path", "unofficial Swift/Core ML GLiNER 2.5-small", "entity spans + confidence", "not Choice/Score/Noul", "not TypeSafe", "label descriptions as schema", "on-device ANE economics", "honesty locks", "shershah1024/gliner-native-runtime ≠ Fastino", "≠ gliner25-compaction ≠ gliner2-ultrafast ≠ Eran-BA/Jev_from_GLiNER2 ≠ NSStudent/JevSwiftSDK ≠ jevmlx", "default threshold 0.1 still soft", "soft Noul ≠ hard safety", "Decision Graph Protocol frame→assess→commit", "app retains permissions/effects", "Jev-first assessor-neutral", "guarded commit / receipt/next frame", "assessment batching", "hard-gating DGP as safety theater", "numerous-com/dgp ≠ TypeSafe official", "jegrep calibrated path+range Nouls", "no embeddings/index/daemon", "~$0.01–0.03 typical", "agent --json", "can1357/jegrep ≠ Bentlybro/jevgrep ≠ uehaj/jev-semgrep", "Archer-arch fidelity", "kev family OOD 0.76–0.77 vs Jev 0.86", "block-causal isolation", "pointer/readout CE-trained", "/v1/systemone drop-in", "replica honesty", "cost-sensitive decision theory × System One probabilities → control flow", "thresholds derived from costs not hard-coded", "YES / NO / UNSURE from cost_false_yes / cost_false_no / cost_human", "auto-batching same-object questions", "Kungie/gut ≠ tpellet/hunch ≠ carldaws/hunch", "judgment vs generation", "deterministic execution after probabilistic judgment", "exactly one app-owned callback", "explicit uncertain branch", "Illusion47586/judge ≠ lexingtonhibiki/judgekit ≠ Ascurse/typed-judge-kit", "variable-N option scoring as the trainable object", "dynamic candidate bags not fixed label sets", "zwliJay/jev-forge ≠ NanoJev", "open replica economics / latency vs closed Jev", "NAR local drop-in", "wfzyx/von late-catch HIGH", "competing NAR claims / replica honesty", "typed judgments vs chat judges on guardrailing", "ishaannk/llm-vs-jev cross-note only", "deeper integrity fold is rh-guard", "nothing wins outright", "can be argued out of guarding"", "Jev IS the if-statement", "judgments/probabilities drive branches", "text model only writes prose", "interpreter owns variables/loops/budgets/replay", "otherwise maybe / confidence gate", "chaos samples after the gate", "southpolesteve/probably ≠ carldaws/hunch ≠ feelings ≠ Kungie/gut ≠ Illusion47586/judge ≠ tidymodels/probably", "133★ / forks 10 live", "build calibrated classifiers from human feedback", "retrieve by relevance not resemblance", "one calibrated yes/no per memory in one request", "pointer mode 17/18 19/20 *theirs*", "embedding resemblance misses the allergy", "samdotmak/jev-recall ≠ jev-search ≠ jev-sift ≠ carryforward ≠ chopratejas/invalidate", "memory leases ended by new evidence", "six Nouls then fixed rules in code", "0 of 157 false invalidations", "questions/plans/directives are not evidence", "unsure → review queue", "host keeps the store", "name↔body / comment truth / test-claims", "mizchi/jev-lint is mizchi/jevlint rename", "no shipped rule has severity error", "~1 in 5 findings wrong *theirs*", "mizchi/jev-lint ≠ huntedman/JevLint ≠ MichitoSugawara/jev-lint", "JSON Schema → typed JSON via Jev", "noul_threshold 0.5 decoder not a proof", "IncompatibleSchemaError lists every bad property", "on-device Laya CoreML ANE", "~5 ms P50 short decisions", "189/189 FP16 checkpoint parity", "10× not achieved", "mizorewww/laya-coreml ≠ gliner-native-runtime ≠ jevmlx ≠ NandhaKishorM/laya", "softmax over allowed tokens ≠ Noul", "question-first cache", "Micha0827/snapjudge ≠ githubnext/localjev ≠ jevmlx ≠ cendress/SnapJudge", "Jev-first Pi agent loop", "slow-LLM fallback", "explicit action menu / CandidateSource unimplemented", "62 tests wiring not quality", "direwolfiy/JevPi ≠ standardagents/jevpilot ≠ pi-jev-control", "resume-screening bias audit methodology", "name×resume factorial independent Nouls", "callback determined by resume quality", "mean-probability name gaps operationally negligible", "natemoo-re/bias-bench ≠ BBQ", "Plan/PRD panel → code-owned pass|review|block", "cheerleading out of scope", "austindixson/planalyzer ≠ single-goodness Noul", "cost-aware multi-model routing/escalation", "decide vs do", "successful-task cost", "cannacre8ive/switchboard-ai ≠ ha-switchboard ≠ hermes-switchyard", "frozen-protocol zero-shot bench", "TypeSafe Jev vs PrismNLI vs Laya", "contamination caveat", "elcronos/jev-vs-open-decision-models ≠ JevBench ≠ DMB", "context-window admission control", "VOI gate which tokens are worth the expensive model", "fail polarity per lens", "on small inputs lenses lose money", "cvsgireesh/jevusher ≠ jev-sift ≠ winnow", "typed decision control plane", "receipt ≠ authorization", "historical-v0 zero retained cases", "MokiMeow/jev-fabric ≠ jev-forge ≠ dgp", "live 15-dim typed rubric re-score per pause", "scoring economics exemplar", "OpenJev/Codiv ≠ TypeSafe hosted", "jose-troche/live-rubric ~$0.000004 desc / ~$0.000006 README", "adversarial pre-registered Jev eval", "28 predictions before data", "123,805 requests", "confidence does not track ignorance", "polite injection 65% / crude 0%", "willkelly/jev-evaluation ≠ jevals ≠ jev-baselines-eval", "provider-neutral Elixir/BEAM Noul/Choice/Score SDK", "class infrastructure", "nshkrdotcom/system_one_sdk ≠ typesafe_sdk ≠ dannote/jev", "question-linting of Jev questions themselves", "nine jaggedness rules, no API key, no labelled data", "static lint ≠ measured separation", "yodablocks/jevq ≠ tenbin ≠ JevLint ≠ commitjev", "open-weights Laya as class exemplar (binding)", "Nx/Bumblebee runtime", "host chooses backend", "ChristianAlexander/laya_ex ≠ system_one_sdk ≠ dannote/jev ≠ NandhaKishorM/laya", "on-chain/edge Laya deploy", "parity_verified stays false", "model output never grants Tx", "humandebri/IC-Laya ≠ laya_ex", "auditable weekend replica", "Jev outputs never used for training", "soft human-vote distributions", "unpaired 0.577 vs 0.727", "agilabs-ai/jev48 ≠ JevBench ≠ Mapika/decider", "adversarial dual-judge / framing attack surface", "comparative framing is the usable judgment", "prior injection crowds out evidence", "copyleftdev/ember ≠ ember.js", "Laya specialist fine-tune pipeline", "training still GPU-pending", "PIXELZX0/XERON ≠ convaiinnovations/laya", "Hub Laya replica drop", "daliborsb/laya ≠ convaiinnovations/laya ≠ NandhaKishorM/laya", "System One student distillation corpus", "gold is programmatic", "teacher is closed-API clone", "do not distill Jev as teacher of record", "MagaBitmex/jev-4b-distill-data ≠ missing student checkpoint", "non-LLM VIN System One", "planning depth not chat", "lewislululu/jevon ≠ douglance/jevon", "source-bound evidence checks", "local quote mismatch needs no API", "exit 0 ≠ claim truth", "WaynezProg/jev-kit ≠ jonathanavis96/jev-kit (Airlock) ≠ jev-use ≠ jev-mcp", "independent System One evidence catalog", "scores not one leaderboard", "no external record currently reproduced", "TokenTrim no-Jev matched hybrid 62.4%", "reachjalil/system-one-bench ≠ mallahyari/system-one-benchmark", "21 tasks · 134 items · 208 questions", "scenes from public GitHub contracts, not production logs", "SivletLabs/jev-eval ≠ willkelly/jev-evaluation ≠ 4esv/jev-eval ≠ xxkuboxx/jev-eval ≠ onlyoneaman/jev-eval ≠ dayhaysoos/jevals", "option isolation (sibling-blind)", "permutation-equivariant", "Hub OWNER not published", "nafisazizir/hev ≠ jaredpalmer/kev", "frozen local LLM logits, no trained decision head", "residual-head 9,222-param decreased 73/96→67/96", "confidence = 1−normalized entropy, not P(correct)", "yuki-oshio/mini-jev ≠ r-ms/mini-jev", "Jev classifier as autoregressive next-token predictor", "ChatJev-style soundness theater", "erik-dunteman/ChatJev ≠ dannote/jev ≠ jev-gpt", "calibrated decision head × AlphaProof value head", "implementation-layer isomorphism, semantic difference", "timeout = censoring", "do not launder Noul as proof", "parallel rank-prediction vs serial selection", "independent questions can conflict", "zzzzzec/jevsort ≠ keltokhy/jsort", "curated open System One ecosystem catalog", "rupeshpoojary9/awesome-open-system-one ≠ AnotiaWang/awesome-jev", "arXiv paper radar with Jev relevance scoring", "ranking ≠ calibration / 0.5 still soft", "fail-open failed evals not marked seen", "train calibrated ~27M from scratch", "typed Q→prob dist / one forward pass / no LLM decode", "hyusi2003/MiniSystemOne ≠ Colvin0315/MiniSystemOne", "description-only stub / size 5", "ESCI hard probe fails four of six", "jev_bool ECE 0.242 inversion 0.255", "do not re-fold §60 six-gates as new", "jobbyjev one-request-per-company from batch-size result", "find/design/evaluate TypeSafe Jev decision loops", "karanb192/jev-architect ≠ samtay32/jev-system-architect", "Jairik/jev-distiller size 1", "distill-Jev UI stub / do not distill Jev as teacher of record", "post-launch scored use-case map / Jev self-scores then human curation", "licensedsaucer9-web/jev-opportunities", "Jev-inize a use case into classifier/router", "gavinHuang/jevinize → simple-jev not TypeSafe", "featherless-ai/simple-jev", "compare saved decisions / same label can still change the branch", "VihaanAgarwal/jev-diff ≠ Saik0s/diffusiongemma-jev-macos", "not tested with a live Jev API key", "constrained logprob + temp/Platt ≠ Noul", "OpenJevPro pastes openjev-sglang JevBench as own", "zhangcy122/OpenJevPro ≠ IamBusy/OpenJev ≠ ekzhang/openjev-sglang", "PolyForm Noncommercial", "SmolLM-135M / sub-70ms / 0 output tokens", "demo P(True) 0.5052 / Choice conf 0.2872 / Score conf 0.0055", "README claims MIT / GitHub license null / no LICENSE file", "patelvishwa112/jev-system-one-rlcd ≠ arnabgho/rlcd-lite ≠ blackwood-rlcd", "source-backed Awesome Jev radar / 306+ commit-pinned", "logicrw/awesome-jev-projects ≠ AnotiaWang/awesome-jev ≠ yibie/awesome-jev ≠ cobanov/awesome-jev ≠ rupeshpoojary9/awesome-open-system-one", "auto GitHub sync / Issue-only submissions", "hashed n-gram encoder / rival-aware attention", "olanotolu/jevbetter vs jevlike starter", "synthetic hard menus top-1 0.916 vs 0.873 / ECE 0.0182 vs 0.0367 / 40 vs 4608 menus/sec", "shuffled-context control 0.335", "Turn any open LLM into System-One Jev", "uspraveen/Jevify ≠ Mintzs/jevify ≠ gulagala001/jevify", "Jevify-any-LLM architecture probe", "description-only stub / size 0", "Train encoder-only calibrated decision models from a task sentence", "Exu is a toolkit, not a method", "strictly proper scoring rule", "Pre-alpha", "Ruivalim/exu-base", "scratch-trained calibrated decision model", "typed Q → probability dists", "Colvin0315/MiniSystemOne ≠ hyusi2003/MiniSystemOne", "no published weights download URL", "90.5 seconds / 29.2% pipeline evidence", "p_i/p_j independent of other candidates", "Recipe for calibrated decision models — small model out", "init → synth → train → eval → serve", "91.1 % / ECE 0.022 *theirs*", "Jev zero-shot 75.1", "scienthoon/luce", "Put Jev's three headline claims on trial", "0.5B local GPU", "46x speedup / accuracy identical", "ECE 0.624 sentiment catastrophe", "bigger model worse calibration", "RichardoMrMu/jev-mini ≠ yuki-oshio/mini-jev ≠ r-ms/mini-jev", "System-1 decision engine for local LLMs", "structured choices only", "JSON parse of generated text ≠ Noul", "TypefAI JEV / Journal Entry Voucher", "tapsin/jev-local ≠ us/jev-local ≠ Argos1111/jev_local", "Jev 1.13 reward-model eval across 8 benchmark tracks", "40,940 examples / 0 API errors", "RewardBench v1 92.58%", "Precise IF 50.63%", "goya4140/jev-reward-model-evaluation", "Scaffolding in progress", "Jev vs LLM support-ticket routing", "static + live decision bench", "TypeSafe's own published benchmark", "illustrative simulations, not live API calls", "JevBench v1 — smart/cheap/fast/reliable", "I/C/S/K 25% geometric mean", "classifier.dev fast tier 84.8 is Jev behind its own API", "do not re-fold §78 v1.2 board as new", "Laya (421M) 70.1 now on board", "Zero-shot/few-shot LLM routing", "hard budget filter before Jev", "Jev never asked to perform budget arithmetic", "Jev judges the next state, XState enforces transitions", "simulation uses synthetic keyword fixtures", "catalog gravity", "v-modal/awesome-jev-tools", "★339 live REST", "curation is not endorsement", "crawler-maintained directory", "Daily GitHub + npm sweep, human-merged", "RadRebelSam/awesome-jev ≠ AnotiaWang ≠ yibie ≠ cobanov ≠ logicrw ≠ v-modal", "HF peft SPLADE/BGE reranker", "rdxtremity/jev-reranking ≠ carlaiau/jev-reranking", "query-side encoders, not a Jev replica", "ONNX System One Qwen3.5-4B scorer", "source:pngwn/system-one-qwen3.5-4b-scorer", "CC-BY-NC-4.0", "temperature 1.75", "transformers.js AutoModel cannot load this graph", "Consistency benchmark Space", "This Space contains no benchmark result yet", "12-case plumbing fixture", "Benchmark-driven Jev router and judge", "cheap alone is not success", "Jev does not write, sum prices, or claim accuracy %", "Sol 94.2 / Luna 83.9 / Jev path 89.7", "19.2% Sol / 62.3% cost save / 4.5pp miss of 2pp non-inferiority", "p50 latency worse than Sol due to routing overhead", "erendikmenn/jev-llm-router-benchmark ≠ jev-rag-benchmark ≠ ryantsai/jev-llm-router", "Express + node:sqlite", "mock and Jev decision engines", "previous_ticket_count >= 3 is code", "MIN_CONFIDENCE 0.6 still soft", "substring false positives", "aesaganda/jev-ticket-router ≠ SarathChandraBellam/jev-vs-llm-ticket-router", "Universal Figure & Diagram Router", "confidence ≥ 0.85 hard-gate is theater", "generative AI banned from scientific plots", "six visual branches", "hoangngochuong24947-gif/jev-figure-router", "human-labeled (state, question, label)", "166,054 rows / 22 configs", "soft_label for human uncertainty", "Praveenrajus/jev-bench ≠ fstandhartinger/jevbench", "ternary bonsai System One GGUF", "openjev's mechanism, Bonsai's weights", "Hub does not ship weights", "100/100 easy T/F is not Harbor", "label_mass ≠ correctness", "stock llama.cpp Q2_0 silently gibberish", "NicolaiMTLassen/open-bonzi-jev ≠ NicolaiLassen", "transformers.js DeBERTa ONNX", "source:com-kotobalabs/open-jev-deberta-v3-large", "temperature 1.05", "AutoModel from_pretrained works", "onnx-community/open-jev-deberta-v3-large-ONNX ≠ system-one-qwen3.5-4b-scorer-ONNX", "107★ densify", "GH 151M vs README 149.6M", "PR #1 now closed unmerged", "do not re-fold §71 claim-audit as a beat", "typed decisions, RLCD, confidence-gated routing", "structured ≠ correct", "mock not live API", "26 tests", "wjdjdakf17/jev-study ≠ baekenough/jev-study", "bonzi-27b-v2 / ternary-8b / 27b-v1 GGUF family densify", "WANLI-256 74.6% / 65.2% / 71.1% *theirs*", "Bonsai 1 27B Q1_0 runs on stock llama.cpp", "ternary still needs PrismML fork", "hf:heman10x/openJev-verdict-2.0 twin tokenizer-only", "OpenJev Vision image classification + uncertainty", "CLEVR-4 held-out joint 0%", "hfdataset:IamBusy/OpenJev-Vision-Research-v0.1 12,832", "294,912 derived targets not independent samples", "Laya multilingual ONNX WebGPU typed-decisions port", "63/63 selected answers / 5.1e-4 CPU / 1.2e-2 WebGPU", "UpHash-Network/mini-jev is yuki-oshio transfer", "jev-injection-bench 11,900 labelled prompts", "Jev best ranking / Haiku better ECE 0.021 vs 0.058", "0.5–0.9 band is where Jev's numbers do not mean what they say", "Prompt wording moves panic 28%", "manojlds/jev-dspy-bench ≠ dspachos/jev-dspy ≠ jmanhype/jev-dspy-lab", "Jev agreement is similarity, never ground truth", "no aggregate quality grade or merge gate", "AbstentionBench-on-Jev rank 1 of 20 vs 2025 field", "question-asymmetry", "forward-looking 0.465 never extreme", "openkev calibration layer not a runtime", "ECE vs coverage independent", "select_threshold returns inf", "escalation catches uncertainty not ignorance", "misakaikato/openkev ≠ jaredpalmer/kev", "pdf-race Docling→Jev vs Gemini", "parser owns the wall clock", "12/12 tie is a tie", "titles selected not generated", "flopcheck 16 calibrated tweet judgments", "mechanical tells in code", "ZeroX-01/jev-atlas ≠ Zaious/jev-capability-atlas ≠ gorock007/jev-atlas", "Laya calibration lab Gradio MCP", "T never changes argmax", "confidence ≠ top-label p", "easy probe set refused", "40–48 rows too small to ship T", "Gemma-4 26B-A4B jevify classification+calibration", "LoRA adapter twin not independent eval", "Gemma-4 E4B jevify", "E4B LoRA stub card", "kushalpatil/jevify-gemma4 ≠ Mintzs/jevify ≠ gulagala001/jevify ≠ uspraveen/Jevify", "GH kushalpatil07/jevify 404", "PAWS 0.580/ece 0.288 is the weak cell", "smaller E4B slightly better OOD ECE than 26B-A4B", "Hub jevify merged LoRA ships weights", "bonzi Bonsai-8B v1 GGUF densify", "Bonsai-1.7B v1", "Bonsai-4B v1", "WANLI-256 64.5% / 60.2% / 52.0% *theirs*", "rank #4 / #5 / #6 of 6", "JulesHuisman/jev-eval scaffolding / README SHA c356a584 (was empty e69de29b)", "JulesHuisman/jev-eval ≠ SivletLabs/jev-eval ≠ willkelly/jev-evaluation ≠ 4esv ≠ xxkuboxx ≠ onlyoneaman ≠ dayhaysoos/jevals", "7 bands 6/10 vs 40 bands 0/10", "source receipts + confidence slider re-policy without re-inference", "32/32 synthetic is smoke not production", "classify HF datasets across typed semantic dimensions", "roadus2 watch misspelling; lock roadius2/ultra_laya", "ultra_laya REVIEW defects", "default branch claude/laya-jev-review-gg5ppo", "XNLI EN 88.3% ECE 0.032 → RU 77.3% ECE 0.096", "Δ −11.0 pp [−14.2,−7.8]; ECE +0.063", "MASSIVE no detectable difference at n=600", "confidence is function of p_max (r=1.000)", "pointer-not-generator 400 human-authored responses", "proposed ≠ authorized", "FewRel 160: Jev 85.0% vs lexical 13.125%", "gated 100% (95/95) coverage 59.375%", "J++ composable semantic computation language", "judge-jev 0.5 still soft", "947 repos scored; A 273 / B 302 / C 372", "LLM rubric ≠ benches", "No benchmark winner is claimed", "phishing: naive 62.6% vs regex 91.8%; 5-atomic + LR 95.0% *theirs*", "AITuber tension ±15", "README npm global; repo is Rust", "git-confess code owns counting/blame/ratio", "httpx exhibit 11% (13/119) *theirs*", "90d trend +12.40% vs random +12.75% vs BH +41.71%", "5m win rate 25%", "Awesomejev 656 entries / 38,160 stars", "tracker likes 64 (+4) lastModified UNCHANGED", "Laya present; Blackwood ABSENT; Archer still promised_not_landed", "Blackwood tracker ABSENT; likes 2 gated manual", "r = c - p_a", "ECE 0.021; acc 0.807 vs warmup 0.746", "Independent primitive", "11.57s vs 54.10s · 4.67× · 120/128 *theirs*", "default path is pretrained Gemma probs not trained RLCD head", "GH Meanblock 404; lock leesk212/JEV-CPU", "softmax over letter slots ≠ Noul", "WANLI 0.741 vs openjev v2 0.77 *theirs*", "3-way NLI ≠ Noul", "priority 0.464 = majority floor", "banking77 contaminated", "raw margins not probabilities", "do not distill Jev as teacher of record (they distilled Haiku)", "“0.9 is not one number”", "ranking ≠ calibration", "banking77 0.8–0.9 stated 0.86 actual 0.73 over-confident *theirs*", "≠ Praveenrajus/jev-bench ≠ fstandhartinger/jevbench", "$0.0000153–$0.0000226 vs circulating $0.0004 (~20×)", "Score is 0..n-1 expectation not 0–1", "Noul has no confidence field", "TCP floor 198.8 ms", "type reliability is not a reason to choose Jev (json_schema 5/5)", "gateway tax not one number", "Function-only 5/8 vs hybrid 8/8", "4/8 without Jev", "8 designed cases not conversion lift", "200-row pilot Jev 86.5% 173/200 vs Gemini Flash-Lite 86.0% 172/200 vs Pro 87.0% 174/200 *theirs*", "not a ranking", "情緒測謊器", "8-example Jev vs GPT-5.6 Sol ~64× cost 5.4× latency *theirs*", "synthetic; no inference", "≠ JevBench v1.2 §78", "Judged 3317 / listed 2560", "Jev judges, code applies policy", "APA “microsecond policy / zero hallucination” overclaim", "Client-side quiz; pointer from held docs; scanned-PDF warn", "Jev judges / agent reasons / user decides", "selecting an option is not permission to implement", "pattern exact, judgement must clear floor", "no matching pattern → no model call", "not a correctness oracle", "Spec vs artifact remainder", "treating 0.85 as 85% / minProbability hard-gate as Harbor", "VERIFY acquires discriminating evidence, never same-pool confidence-only rescoring", "fast/full/max are ceilings not sizes", "Solar writes, Jev chooses NEXT ACTION", "do not reopen or amend PR #23 or #24 or #25 or #26 or #27", , "Calibration is not alpha", "NO CURRENT ALPHA CANDIDATE", "ΔR² approximately +0.00084", "Brier 0.2131387", "ECE 0.0421875", "Adding Jev probability to deterministic volatility improved Brier by only 1.4058e-05", "default 0.5 keeps zero non pinned", "keepResult median 0.14 to 0.17", "keepCall median 0.28 to 0.35", "usable range is about 0.10 to 0.25", "7.8% to 57.9%", "judges results it never sees", "task-finish eval not built yet", "$0.002 per compaction", "slavadubrov/sgr-judge-bench ≠ slavadubrov/jev-judge-bench", "Jev 108/120 $0.083 0.34 s", "Luna SGR 114/120", "paired Jev accuracy-difference intervals include zero", "not evidence of equivalence", "GLM SGR 26/120 93 format failures", "Terra-planned Jev hybrid 55/120", "rule-based by default, optionally Jev-backed", "empty README", "missing key cannot break the experience", "prefill plus exactly one decode", "softmax over A/B/C ≠ Noul", "BBQ 9,053/10,000 (90.53%)", "ECE 0.0890", "Mean confidence 0.9943", "overconfident", "score and noul not implemented", "DGUI 12 rows (was 6)", "INSTRUCT 119 rows likes 2", "encode the state once, decide everything in parallel", "0.740 accuracy against a 0.508 majority", "ECE 0.047", "fine-tune's advantage ends where its 384-token training data does", "jasonkneen/open-jev ≠ pngwn/open-jev", "same sha d41dc3cd", "Space does not call Jev", "recomputes routing from saved probabilities", "200-case Jev 97.0% / 100.0% / 95.0% / MAE 9.22", "synthetic repository benchmark", "Jev evaluations are advisory", "YehuiTang0316/jev-nlgrep ≠ Bentlybro/jevgrep ≠ can1357/jegrep ≠ uehaj/jev-semgrep", "default threshold 0.8 still soft", "40-line windows cannot prove whole function", "token-native sequential start/end Choice", "Gemini/Haiku stubs not configured yet", "handful of hand-written examples, not a benchmark", "Jev judged exactly what it was given", "laguagu/jev-skills ≠ laguagu/jev-evidence-lab ≠ Pleo2/awesome-jev-agent-skills", "contract_passed is not a claim of guaranteed factual truth", "Wilson lower bound 0.85 floor", "fixture mode no savings claim", "SemIf 2207★ (+21 vs §110 2186)", "jevlike 1043★ (+5 vs 1038)", "TypeAR 15★ (+1 vs 14)", "AnotiaWang 97★ (+1 vs 96)", "yibie/awesome-jev 506★ (+16 vs 490)", "Laya likes 822 (was 802)", "tracker likes 64 flat, lastModified UNCHANGED", "do not reopen or amend PR #23/#24/#25/#26/#27/#28", "Heman10x-NGU/Verdict-open-jev ≠ Heman10x-NGU/openJev-verdict-2.0", "TF-IDF + LogReg ECE 0.0207 vs Jev 0.1440", "Verdict-open-jev 48.07% vs Jev 90.80%", "abstention combined recall 10.00%", "p50 35.58 ms", "K=25 (maximum capacity) 72.00%", "0.85 coverage 84.60% selective risk 1.18%", "26.1× faster than standard Qwen JSON generation", "Jevify 90.0% / 167 ms CUDA graphs disabled", "Finding 1: Brier on stated confidence alone is a trap", "grpo_rlcr 0.78 / ECE 0.084", "reliability 0.007 but resolution 0.000", "27 900 schema-driven decisions", "13 600 / 13 600 questions", "candidate mass min 0.99999624", "22 configs · 166,054 rows · 4 calibration-gold", "sha a39eba3f", "Student B MAE 0.148 / Pearson 0.836 / 86.0%", "pngwn/open-jev-laya-bench README 404", "sha 9f69c742 likes 2", "HDFS 0.9933 (745/750) / retain 0.0084", "BGL ERROR/FATAL protection 1.0000", "2,479 / 2,500 HDFS uncertain", "cache hit 0.9648 (2412/2500)", "$0.153936 estimated", "E2 recomputes from saved probabilities", "Space sha eda59e0a", "MASSIVE English 0.783 / Khmer 0.033 / Hindi 0.133", "40–48 rows too small to ship T", "T never changes argmax", "siren2345/jev-single-decode-transformers ≠ siren2345/jev-single-decode", "Split Transformers experiment from llama.cpp runtime", "tanayvasishtha/jev-lab ≠ dairui1/jev-lab ≠ BrendanH18/jev-lab ≠ yibie/laya-jev-lab", "Four experiments stress-testing TypeSafe's Jev: calibration, bundle bias, label bias, and ensembling", "second pass must be $0.00 from cache", "The pages never call Jev", "Gemma 4 31B 77.0% / Jev 1.13.0 61.4% / Laya 322M 0.0%", "restriction state 95.0% against 84.4%", "None of the systems are particularly good at knowing when to stop and ask", "They skip the question and call a tool directly", "100% schema pass", "six-field joint 48.8% vs 72.8%", "ywchiu/jev_benchmark ≠ Running-Dolphins/jev-bench ≠ Praveenrajus/jev-bench", "ACT / REVIEW / FALLBACK", "A provider failure, timeout, malformed output, or missing answer is **not** a policy outcome", "confidence is descriptive provider output, not a substitute for probability", "Quality denominators include only valid scored answers", "an exact halfway tie chooses the lower level", "aiwithenoch/Jev-Skill ≠ simplosophy/jev-skill ≠ laguagu/jev-skills", "The local path does not claim to turn a smaller checkpoint into Jev", "Low support becomes decision: \"review\"", "MIT-0 SPDX NOASSERTION", "current-llm", "结构兼容,不是 Jev 模型能力", "altryne/jevify ≠ Mintzs/jevify ≠ gulagala001/jevify ≠ uspraveen/Jevify", "Find where Jev belongs. Design the questions. Measure the difference", "TypeAR-AI/TypeAR 301 → TypeLLM/TypeLLM", "TypeLLM/TypeLLM 16★", "SemIf 2241★ (+34 vs §111 2207)", "jevlike 1051★ (+8 vs 1043)", "AnotiaWang 98★ (+1 vs 97)", "yibie/awesome-jev 525★ (+19 vs 506)", "Laya likes 864 (was 822)", "tracker likes 67 (+3 vs 64)", "lastModified UNCHANGED `2026-09-20T04:29:16.000Z`", "do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#32", "hysteresis enter/exit / replay policy without inference", "calibration does not compose / hop-ECE permutation-invariant", "equal-width vs quantile ECE / ranking ≠ calibration", "Qwen2.5 ≠ Archer / Qwen 3.8 sparring ≠ Archer / Qwen/Qwen3.8-27B ≠ Archer", "Deferred Crispification / TCE / AMS", "g0runmezadam/what-is-jev IS tunahansahin897/what-is-jev", "pd.cut equal-width vs jeval quantile", "A hunch is a probability with a policy attached", "soundness theater / measurement theater / hourly 0843", , "Jev Capability Resolver / NiazMorshed2007/jcr", "one tool nested capability tree / returns context / does not execute", "skills vs capabilities / workflow+judgment vs operations", "format independent of Jev / proposed open standard", "JCR_BAND_RATIO 0.6 is application policy / soft scores ≠ hard gates", "routing ≠ permission / docs ≠ authority to run", "sol-vs-opus5-20 lookup+explain / n=1 / Not Harbor task-execution", "wall-time mixed / Sol slower with JCR in 19/20", "NiazMorshed2007/jcr ≠ skill-broker ≠ skillranker ≠ jev-sift ≠ jev-lens ≠ jevusher ≠ jev_select_capability", "do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34", "notes.md §116", "copy the SemIf/MLX installer?", "quote 5.21× as beating Jev?", "treat 0.845 as a TypeSafe replica?", "collapse SemIf into kw2828/zhihz/semif-rs/semif-serve", "softmax over options as a Noul", "llm prompt to jev primitives", "conversion assistant not equivalent behavior", "heuristic conversion ≠ calibrated Noul", "alexwestco/llm-to-jev ≠ altryne/jevify", "user-provided 0940 / notes.md §118", "judge ≠ actuator", "candidate_mass", "softmax over A–H ≠ Noul", "hourly 0947 / notes.md §119", "ggmlc GGUF is not llama.cpp", "serving substrate ≠ calibrated replica", "Qwen3.5-9B ≠ Archer", "planner writes JEV selects", "hourly 1049 / notes.md §120", "open recreation ≠ calibrated replica", "semantic lint is a sensor not a proof", "cutoff 0.8 still soft", "paired bootstrap CIs *theirs*", "Same accuracy, 35x faster *theirs*", "hourly 1143 / notes.md §121", "revisit HIGH / since-last-look", "catalogued repo changed", "star-noise vs material change", "densify prior notes without inventing equivalence", "decide is not generate", "tryDecide returns typed calibrated judgments not a token stream", "GLiNER/GLiClass ports are class members not Jev replicas", "93.5% *theirs* not Harbor", "74.9 *theirs* not Harbor", "8.7x *theirs* not Harbor", "Option-Marker joint attention", "openjev:0.2.1", "thinking=True/False per-field budget", "PLAN_Qwen35", "hyperspaceai/jevcache ≠ kushals256/jevcache", "wire-compat ≠ logit-equiv", "SHA move is not a replica", "hourly 1248 / notes.md §123", "typesafe-sdk 0.7 Pydantic response models", "msgspec dropped", "The server's output is unchanged and was never wrong", "SchemaError is 400 plain-string detail not 422 list", "Pydantic response models ≠ logit-equiv", "msgspec dropped is not a replica", "Error contract is not a Noul", "coverage-at-error-budget *theirs* not Harbor", "PLAN_Qwen35 still proposal for review", "GLiNER locate ports are class members not Jev replicas", "Locate ≠ decide", "~160 ms *theirs* not Harbor", "0.971 F1 *theirs* not Harbor", "hf:fr0stbit3/laya-gguf serving substrate ≠ calibrated replica", "jkcdarunday/SystemOne-Next ≠ TypeSafe System One", "hourly 1340 / notes.md §124", "vLLM NVIDIA + MLX Apple Silicon", "Codiv hosted free endpoint", "dual /v1/systemone + /v1/chat/completions", "chat 501 on MLX", "dual serving is not generate", "Hosted Codiv ≠ TypeSafe", "hr98w/jev-visual 167★ Apple Silicon visual candidate scoring", "37.30s → 2.40s at 64 decisions *theirs*", "Breakout 9 bricks 6 returns 2 lives *theirs*", "candidate probabilities are relative not correctness", "jkudish/jev-mcp 156★ ten MCP tools", "recommendation is advisory", "the server never blocks on its own", "TypeSafe CLERC 5% to 18% *theirs*", "jkudish/jev-mcp ≠ burnigtm/jev-mcp", "zhengxuyu/litjev off-the-shelf Qwen decision layer", "Probabilities are not calibrated by default", "Qwen/Qwen3.8-27B ≠ Archer", "zhengxuyu/litjev ≠ alexwestco/llm-to-jev", "Zefan-Cai/Open-Jev LoRA + scalar head", "2B 94.71% 9B 97.54% hard test *theirs*", "2B OOD 86.02% 9B OOD 91.97% *theirs*", "80,816 training rows", "27B still in progress", "LoRA ≠ RLCD replica", "Zefan-Cai/Open-Jev ≠ TheoLeeCJ/openjev ≠ razorback16/openjev", "cristianoliveira/jeq intelligence you can pipe", "pass-min 0.8 still soft", "JEQ does not own actions", "AndyInQtr/laya-coreai CoreML serving substrate ≠ calibrated replica", "AndyInQtr/laya-coreai ≠ mizorewww/laya-coreml", "hourly 1441 / notes.md §125", "TypeLLM/TypeLLM densify HEAD 6a48f9f1e623", "README densify 3k→12k B", "Batch 5.8x *theirs*", "Constrained AR ≠ calibrated Noul", "jaredpalmer/kev densify HEAD b339f446a0ef", "Kev-0.6B 4B 8B family", "4B new-source 0.790/0.806 *theirs*", "8B new-source 0.796/0.780 *theirs*", "Jev hosted 0.857 *theirs*", "Questions share the input text but cannot read each other", "No Jev outputs were used for training", "8.2% ≥0.9 on wrong *theirs*", "option order can change an answer", "Qwen3 ≠ Archer", "TheoOliveira/pi-jev 21★ fail-closed routing", "JEV_THRESHOLD 0.65 still soft", "harshwasan/jev-sentinel fail closed never auto-allows", "harshwasan/jev-sentinel ≠ leepokai/jev-guard", "jackbarunz/jev-tool-router ≠ esinocchi/jev-tool-router", "threshold 0.90 still soft", "76/81 vs 77/81 *theirs*", "0.419s vs 2.459s *theirs*", "$0.00486 vs $0.03673 *theirs*", "not a security boundary", "baronunread/leanest fail-open uncertainty means RUN", "classifier.dev default Jev/Laya pluggable", "openlayer-ai/jevals ≠ dayhaysoos/jevals", "estimates not Harbor", "classifier ≠ authorizer", "MrJev/awesome-jev 118 entries catalog ≠ endorsement", "MrJev/awesome-jev ≠ yibie/awesome-jev", "Koushik890/jev-firewall fail closed ask_below 0.7 still soft", "CompleteTech-LLC-AI-Research/jev-codex-approval experimental native not compiled", "confidence is not a measured probability", "rh-guard owns primary gates", "hf:rAVEUK/open-jev-deberta-v3-large encoder class member not Jev replica", "hf:p-yan/laya-quanto serving substrate ≠ calibrated replica", "hf:Gtrkrsk/laya serving substrate ≠ calibrated replica", "hourly 1542 / notes.md §126", "razorback16/openjev densify HEAD febf02e88989", "release 0.3.0", "re-pin vLLM PR #57250 restructured head", "MODEL_VERSION stays openjev-0.1", "uv.lock hygiene", "restructured vLLM head ≠ logit-equiv", "frostney/clean-code-review 7★ typed judgments not opinions", "documentation is read not judged", "morcoan/JMP Joint Model Participation", "Models participate. Real tools execute.", "Thresholds are policy not model", "Kelbie/hunch ≠ carldaws/hunch ≠ tpellet/hunch ≠ huncho", "Jev never generates prose JSX or code", "json-render is the only renderer", "game success ≠ calibrated Noul", "Shalimov04/open-jev ≠ razorback16/openjev", "MstyAI/laya-onnx empty repo", "hf:Praveenrajus/jev-bench HTTP 200 was 401", "hourly 1643 / notes.md §127", "TypeLLM/TypeLLM densify HEAD 702e6a287f3c", "truncated thinking then constrained decode", "0.8B thinking On 0/18 *theirs*", "forced closure 20/20 type-valid *theirs*", "jaredpalmer/kev densify live HEAD 8465c4c4c294", "Kev-0.8B completes family", "4B new-source 0.794/0.832 *theirs*", "9B new-source 0.812/0.837 *theirs*", "transfer-v9 Kev-9B 5% Jev 9% Kev-8B 26% *theirs*", "SemIf Kev-9B 0.917 Jev 0.965 *theirs*", "scienthoon 0.952/0.911 vs 0.897/0.914 *theirs*", "transformers >= 5.17", "Qwen3.5 ≠ Archer", "notque/vexjoy-agent 421★ /d routes /do fallback", "Facts go to code. Judgments go to Jev. Only facts can block.", "Jev never blocks", "jqueryscript/awesome-jev ≠ MrJev/awesome-jev ≠ yibie/awesome-jev", "five-lines threshold 0.80 still soft", "371ms $0.0000189 300-call *theirs*", "tpellet/jevify ≠ altryne/jevify", "seb4ez/jevguard-mcp ≠ seb4ez/jevguard", "resumocast/jev-mcp ≠ jkudish/jev-mcp", "Adrian-Ernesto/jevsort ≠ zzzzzec/jevsort", "MidasMulli/kev-ane 155/155 argmax *theirs*", "hourly 1746 / notes.md §128", "Fine-tuning on your own data", "--data JSONL", "--init_from warm-start LoRA/head PR #9", "from-scratch ≠ warm-start", "JSONL labels ≠ Harbor", "Kev-0.8B 4B 9B Qwen3.5 family", "0.33 vs 0.84 vs 0.83/0.88 *theirs*", "reconstruction ≠ replica", "assay-001 split verdict", "Promethe-us/awesome-jev ≠ MrJev/awesome-jev ≠ yibie/awesome-jev", "ThePFMind/jev-mcp ≠ jkudish/jev-mcp", "kyegomez/open-jev ≠ razorback16/openjev", "namenu/pi-jev-effort ≠ TheoOliveira/pi-jev", "samatv256/mini-Jev ≠ r-ms/mini-jev", "hourly 1843 / notes.md §129", "TypeSafe-compatible ≠ TypeSafe replica", "SystemOne.from_pretrained", "replica ≠ TypeSafe", "76.7% vs Jev 86.9% strict common subset *theirs*", "kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions", "DeBERTa-v3-large 0.855 / 42 ms *theirs*", "aisearchio 15-link census catalog ≠ endorsement", "user-provided 1936 / notes.md §130", or "cascade sign-flip / calibration theater": read `references/faq.md`, + other", "training confronts Choice other / none-of-the-above", "soft AGENTS.md rules vs the linter", "screenshot Choice / omni System One", "extractive quotes / pointer not generator", "compaction summarize vs pointer", "encoder vs Jev compaction backend", "shadow-mode compaction rollout", "CI flaky-vs-real merge gate", "fail-open VOI wake/resume", "claim vs session evidence", "S1 indexer escalate-S2", "Harbor on/off routing", "fail-open vs fail-closed wake vs CI gate", "encoder vs Jev computer-use backend", "hybrid local decide + remote fill", "DONE vs verified success", "stdout prune vs session compaction", "OpenCode jev-pruner vs Claude jev-pruner", "zen-chat vs jev-zen Noul", "hard envelope then Noul prune", "Cua-S1 vs TypeSafe Jev", "plan vs execute dry-run", "specialist computer-use vs general agent", "local drop-in vs stub scorer", "route vs memory", "when does it hold / extractable from state", "decision model vs constrained LLM", "dual-process S1/S2", "combinatorial grid vs extractive", "uncalibrated local likelihoods", "decision-native RAG", "classify-first / read selectively", "living applied-mappings atlas / class patterns", "silence as safer / draft-gate heartbeat", "robotics text-state vs pixels", "verbatim ledger vs summary", "judgment as language primitive", "Stagehand extract pick-and-copy", "harness observe-score-act vs demo loop", "public judgment wall / six parallel questions", "meaning-search without embeddings", "attention ≠ correctness", "skills→oxlint / AST prove ∩ remainder", "session-sticky first-prompt routing", "measured RAG rerank vs generative rerank", "capability kernel / secrets never in the agent", "Jev is SENSOR not policy", "type-safe ≠ correct", "typed control plane around DSPy", "native vs verbalized confidence", "engine owns truth / Jev owns judgment", "human-confirmed kill gate", "train specialist vs few-shot hosted", "decide→policy→LLM leftover", "Noul 0.5 cannot-tell never rounded", "calibration ≠ sortable / ORDER BY", "pairwise inversion / Score ordinality / two-decimal ties", "wire-compat GLiFormer /v1/systemone", "class-backend economics", "loopback gateway hosted + local", "do not distill Jev as teacher", "active-learning triage", "evidence-packet explorer", "meaning-grep AND/OR/NOT", "closed-vote-only / no planner LLM", "Jev vs PCD Harbor", "PCD O(1) ≠ Noul", "host-owned handlers × System One", "OMP/pi fail-open gate", "permission vs probability / operator owns thresholds", "judgment ≠ permission / Jev never grants access", "eval integrity / instrument not score", "constrained optimizer + S1 features / never sole hot-path gate", "privilege ≠ verdict / effect contracts not tokens", "attention filter / VOI for human review / never blocks / never green unless sure", "measurement owns endorsement / evidence-gated question packs", "Jev supplies evidence / code owns authority", "ranking ≠ calibration / never hard-threshold raw p as frequency", "hot-click CU / indexed element table", "Jev judges relevance / code decides structure", "local rules first then remainder / never auto-train on own hides", "combinators / System One as control plane", "receipts not leaderboard / type-safe ≠ correct jaggedness", "VOI over skill library / skillranker abstention", "OOD calibration / AUC ≠ ECE", "Jev vs thinking-budget small models", "turnstile / replayable evidence≠authority", "MLX one-pass schema→JSON / Apple Silicon replica economics", "memory leases ended by new evidence", "never confidently wrong / TLA+ compose / escalate instead of hard-gate", "no seal no advance / coverage ledger / mint ≠ product brain", "skill-broker sibling / judgment ≠ permission", "sureness bands / max_prob is generous", "JevBench / calibration not in Main Score", "CI typed gate before expensive review", "Codex MCP host adapter", "judgment as attention redirect / jev-preflight", "compress-before-first-send / dizk jev-lens", "tools≠use / SessionStart over hoping", "observational memory / pi-om keep-kind", "open-Jev class / openvons / JevPick", "physical-world System One / HA-Jev / not for locks", "judgment outside the store / jevql", "landed-script trust / headless≠auto-approve", "digital-design combinators / extended five", "VOI cache admission / same-intent skip LLM", "BM25 vs Jev skill routing Harbor harness", "zeroshot vs BERT / contamination DiD", "typed escalate continue abort baton / inverted loop", "worth-your-attention VOI / ThinkyMiner Winnow", "Jev WHETHER Python HOW LLM WHAT", "conflict vs ignorance / named Choice escape", "Playwright executes Jev chooses", "OpenJev /v1/decide not drop-in", "SemIf wire-compat runoff; SemIf rename densify / MLX backend / 5.21× systems≠semantic / Softmax ≠ Noul (`notes.md` §117)", "decision-as-memory flywheel", "record/replay CI / jevassert", "failure-finding arena / jevarena ≠ jev-arena", "BBQ not a bias cert", "decider≠executor", "sentence-as-rule lint / jevlint", "sentence-as-rule lint / jev-lint is jevlint rename", "VOI hunk prune", "whole-repo intent VERIFIED/VIOLATION/UNKNOWN", "GLiNER2 spec ≠ replica", "open replica substrates / grande / laya-jolt / JEV-CPU", "ONNX local-jev not equivalent", "persist constraints across compaction / pi-heed", "calibration+cost first-class gates", "Harbor-shaped Jev vs SGR LLM-as-judge / jev-judge-bench ≠ jevarena ≠ jevbench", "hand no-text steps / jev-use / Vercel drops confidence", "Pi System-One control plane / pi-jev-control", "never free-generates / jev-gpt tree of Choices", "OpenRouter recipe atlas / samples not benches", "personal history feed / jevfeed / no social graph", "competing NAR claims / dual-channel ECE / openJev-verdict ≠ OpenJev", "empty compaction-proxy skip / IPECTER", "throughput ≠ latency / like-for-like ECE", "1-token logprob endpoint ≠ Noul / coverage ≠ correctness", "open replica engine / jevinf", "unofficial Elixir SDK ≠ OTP peer", "jevex n=16 files-to-read VOI", "commit pre-review attention≠verdict / middle band", "Hermes plugin is Agnes not TypeSafe", "pi-jev-compact ≠ pi-jev-compaction", "empty Codex-proxy skip / IPECTER runway", "decision-native inbox / mailordinal", "unofficial jev-cli not ready / ≠ jevql", "laya-multilingual / English checkpoint confident-wrong OOD", "schema-scorer peaked ranking ≠ calibration", "HF 401 / GitHub 404 Hub-only", "productized System One HTTP / classifier.dev", "escalate-under-threshold / smart tier / multi-label ignores", "silent FALLBACK / granite 0.546 vs advertised 0.800", "vs_jev tracked JSON / read eval/README", "choxos/jev-reviewer ≠ egma-ai / systematic-review pointer", "two-pass Choice+Noul evidence extraction", "not-found is an answer", "human check as productized judgment", "githubnext/localjev ≠ kunchenguid/local-jev", "wire-compat ≠ logit-equiv / prompted JSON ≠ structured read", "self-reported probs / entropy confidence", "GitHub Next local /v1/systemone", "LM Studio runner gap / structured-read primitives", "NandhaKishorM/laya packaging ≠ Hub-only / Router script-before-p", "post-T ECE ≠ raw ECE / Banking77 token-budget", "0.85 still soft / not TypeSafe drop-in", "external census ≠ scored bake-off", "GLiNER2+routers class-boundary", "incomplete openjev census vs watch", "Harbor honesty watch / silent fallback", "JevBench v1.2 geometric mean / cal ON rank / weight sensitivity", "option-order 72→21 / instruction models in the class table", "self-host latency ×2 assumption / est. costs", "Laya absent is a gap not a named exclusion", "Qwen3.8 27B ≠ Archer", "hourly already-folded watch / apply-the-five / skip thin noise", "hard-gate Noul as PR/quality gate is soundness theater", "S1 never stalls waiting / S2 one-use advisory", "Local controller ≠ githubnext/localjev", "purple telemetry = consumed not arrived", "seed = geometry not async replay", "20% starting gate still soft", "no pixels to either provider", "OCR+AX observe-score-act / typesafe-computer-use", "never send screenshot to frontier for the decision", "overlapping CU options = false low confidence", "split kind/item/site", "155× one-screenshot ≠ Harbor taskset", "decision ≠ answer-reader capture", "ASR observe-score-act / jev-voice-browser", "partial-speech VOI / free-text waits", "spoken confirm ≠ hard auth", "numbered overlay without another model", "wrap-as-execution / AgentGhost ALLOW ASK DENY", "rules first then Jev remainder / ASK throws / fail-closed", "reddpy/AgentGhost ≠ jwen5419807/agentghost ≠ vventirozos", "JP genre atlas / studio_yebisu / stars ephemeral ≠ eval", "Jev Clearly Explained / akshay_pachaar / LLM hammer", "schema-safe ≠ correct / 200× 400× TypeSafe ceiling", "questions-as-code / shadow first / not a TypeSafe how-to", "proposition ≠ embedding / contrast-set", "boolean composition of soft Nouls / AND OR NOT", "uehaj/jev-semgrep ≠ semgrep.dev", "meaning-grep dedicated fold / not a gate", "decision-validated UI / Jev never authors text", "decision-as-assert / jevtest ambiguous band", "typed decisions drive UI / jev2ui", "hybrid S1 closed verb menu / anima3", "pointer-not-generator search / JevFind", "jev-frontier-bench ≠ frontier-100", "product bakeoff ≠ architecture duel / GLiClass", "four engines same questions / majority floor", "authorship named escape / not evidence", "ha-switchboard HA remains execution", "n8n classify/route/score / Low Confidence", "fast-jev-compaction-pi ≠ pi-jev-compact ≠ pi-jev-compaction", "jevloop full-distribution optimizer / no LLM in the loop", "laya-vision SmolVLM / score untrained", "Cerebellum-2B /v1/decide ≠ TypeSafe / wire-compat vs agent-routing", "laya-grounded not drop-in / Platt not temperature", "GestaltLabs/Jeff-1 ≠ logan-markewich/jeff / acc vs ECE n=9730", "stanley-code empty findings ≠ approval / human promote", "findme ≠ JevFind / NL memory beam-search FS", "jevsubrouter price workers not conversation / counts ≠ dollars", "feelings .feels() default 0.5 is Noul-0.5-never-rounded / ≠ hunch ≠ Probably", "apa-agent-harness ≠ AntonioCoppe/jev-harness / unpublished npm", "grok-bot-jev skill cannot force a bot that ignores it / A/B proxies not tokens", "Essentiel-Jev never authority / human every action", "enzo-mcp independently falsifiable claims / ≠ jev-sift", "pigeonhole OTHER skip / decision-as-filing", "jev-reliability Nothing about accuracy", "clduab11/jev-test ≠ realZachi/jevtest / Nothing runs yet", "jev-rag-benchmark Jev wins is not an assumption", "dairui1/jev-lab ≠ BrendanH18/jev-lab", "jevmail gmail.readonly / mailjay archive/trash", "ZHUBoer/ego-jev reserved __none__", "runWorkflow completed ≠ success", "jsort scores are relative", "Noul not Choice for scale", "groundedness-judge-bench native vs schema-guided", "implicit_true included in yes", "jev_playground 0 promotions", "routing-backtest 0.0447%", "yuyang2230/jev-agent-skill jev-1.13-free", "jev-techstack-classifier stack_config.json", "s1_ruby collapse late", "undecided? abstain", "2389-research/judgement license null", "confidence ≠ winner p", "typesafeai-sdk-community not a new species", "tpellet/hunch exit 3", "never-execute list", "jev-file-search scores not calibrated accuracy", "jev-linkmap Jev never sees S2 prose", "muhammedilyasy/jev-mail metadata only", "tidy none-of-folders stay", "tab-bouncer pinned/audio/current never closed", "lkclean Show fail-open", "jev-yt-time-saver Show anyway", "ORIGIN pause-if-no-Jev", "validResponse sums-to-1", "jev-crawlers risk bands never raw boolean", "jevbrain AUTO_ACT is not a Noul", "judgekit YAML classify/score/route/verify", "typed-judge-kit verdict-in-code", "alsoleg89/decide packing VOI", "0.8 ≠ 80% accuracy", "Jev-Calibration Platt ECE 0.117→0.052", "jev-calibration-arena never acts", "ctmx/openrouter-jev-mcp Decision-as-Plugin", "FrancoisChastel/jev-code ≠ npm jev-code", "claudecode-jev-marketplace fail-open not hot path", "pedroknigge/mcp_jev packs not ask_jev", "cyrusasco/typesafe-mcp noul deadband 0.35–0.65", "codaaiteam/jev-skill jevtypesafeai.com ≠ TypeSafe", "hermes-switchyard ≠ hermes-jev-router ≠ hermes-plugin-jev", "nanoprune 2.8MB ECE 2.58%", "smartdio/jev-browser-agent ≠ ZHUBoer/ego-jev", "Dakai/omp-jev-web DONE ≠ proof", "hari007sh/jev ≠ dannote/jev", "0thernet/system-one-skills deterministic verify", "typed-gate band [0.40,0.60] is refusal", "pi-jev-gate fail-closed; choice is the verdict", "Foq ~25ms/2.2GB local", "rev prefill-only + HF jev-0.5b", "robfrase/jev planning memo", "typesafe_agent_gates 27/27 / 31/31", "EpicEric/safe-sh static remainder", "pastepilot Confirm before act", "Jev-Reranker live Jev not yet measured", "sessionwise opt-in relevance", "jev-search pointer sieve", "400ms Salesforce WebMCP", "typesafe-scheduler-diagnostics advisory", "droidjev screenshot-free", "Tewoto1 jevcu planner still writes", "ha-conversation-jev Jev→Grok", "dsh-jev can only gate", "jev-classification-benchmark specified not run", "jev-luna-pagerduty p≥0.50", "meldltd/meldecision laya-go ONNX", "laya-doom never pixels", "logixism/laya-api empty README", "akpsahan/laya ≠ Archer", "choxos/jevchess engine owns truth", "jev-drive sim not AV", "story-arc Jev never authors", "jev-hs-assistant HS6", "golergka/jev-plays-starcraft-2 UI-verified ≠ API Victory", "awesome-jev-use-cases catalog", "Nibir1/typesafe-go ≠ official", "fingerprint after redact", "recall vs decide", "publish fingerprints+answers", "CI replay as Harbor cousin", "Cache hit ≠ correctness", "hyperspaceai/jevcache ≠ kushals256/jevcache", "human labels only", "score never auto-accepts", "production capture flywheel", "sutro-sh/jev-align ≠ caiovicentino/jev-align", "guidance ≠ hook", "catalysts ≠ summaries", "compile-time System One", "unofficial ≠ TypeSafe", "format_version modernbert-jev/1", "Argos1111/jev_local ≠ us/jev-local ≠ kunchenguid/local-jev", "LFM default ≠ ModernBERT backend", "Nemotron ≠ TypeSafe Jev", "not a calibrated replacement", "djev-dev complements djev-spark", "images as Choice options", "Laya essay numbers *theirs*", "Router/OOD confidence", "hosted bootstrap ≠ silent TypeSafe", "difficulty + policy thresholds + JSONL trace", "jev-codex-pilot model + reasoning depth", "keep/shadow/hybrid/reject", "quarry evidence projection", "Frank-ZY-Dou/awesome-jev robotics/3D/control", "one-dollar-tahoe TypeSafe Jev defense eval", "jevguard calibrator/cache/escape", "jev-ci-selector CI shadow mode", "llama-jev llama.cpp replica", "petercr/jev-orchestrator ≠ FleeexCorp/jev-orchestrator", "seb4ez/jevguard ≠ AseemPrasad/JevGuard ≠ pablozr/JevGuard", "webNeat/llama-jev ≠ WiktorB2004/llama-index-jev", "OpenCode jev-pruner context sieve", "observe→score-candidates→prune", "jev-zen / jev-1.13-free", "zen-chat ≠ Noul", "fail-open original", "keepScore >0.1 floor", "host port of tamaratran/jev-pruner", "indiejoseph/opencode-jev-pruner ≠ nrdz-labs/fast-jev-opencode", "jev-webagent-bench empty stub", "Kiln-AI/jev_jsonschema noul_threshold 0.5", "NSStudent/JevSwiftSDK unofficial", "GLiNER2 native Apple path", "unofficial Swift/Core ML GLiNER 2.5-small", "entity spans + confidence", "not Choice/Score/Noul", "not TypeSafe", "label descriptions as schema", "on-device ANE economics", "honesty locks", "shershah1024/gliner-native-runtime ≠ Fastino", "≠ gliner25-compaction ≠ gliner2-ultrafast ≠ Eran-BA/Jev_from_GLiNER2 ≠ NSStudent/JevSwiftSDK ≠ jevmlx", "default threshold 0.1 still soft", "soft Noul ≠ hard safety", "Decision Graph Protocol frame→assess→commit", "app retains permissions/effects", "Jev-first assessor-neutral", "guarded commit / receipt/next frame", "assessment batching", "hard-gating DGP as safety theater", "numerous-com/dgp ≠ TypeSafe official", "jegrep calibrated path+range Nouls", "no embeddings/index/daemon", "~$0.01–0.03 typical", "agent --json", "can1357/jegrep ≠ Bentlybro/jevgrep ≠ uehaj/jev-semgrep", "Archer-arch fidelity", "kev family OOD 0.76–0.77 vs Jev 0.86", "block-causal isolation", "pointer/readout CE-trained", "/v1/systemone drop-in", "replica honesty", "cost-sensitive decision theory × System One probabilities → control flow", "thresholds derived from costs not hard-coded", "YES / NO / UNSURE from cost_false_yes / cost_false_no / cost_human", "auto-batching same-object questions", "Kungie/gut ≠ tpellet/hunch ≠ carldaws/hunch", "judgment vs generation", "deterministic execution after probabilistic judgment", "exactly one app-owned callback", "explicit uncertain branch", "Illusion47586/judge ≠ lexingtonhibiki/judgekit ≠ Ascurse/typed-judge-kit", "variable-N option scoring as the trainable object", "dynamic candidate bags not fixed label sets", "zwliJay/jev-forge ≠ NanoJev", "open replica economics / latency vs closed Jev", "NAR local drop-in", "wfzyx/von late-catch HIGH", "competing NAR claims / replica honesty", "typed judgments vs chat judges on guardrailing", "ishaannk/llm-vs-jev cross-note only", "deeper integrity fold is rh-guard", "nothing wins outright", "can be argued out of guarding"", "Jev IS the if-statement", "judgments/probabilities drive branches", "text model only writes prose", "interpreter owns variables/loops/budgets/replay", "otherwise maybe / confidence gate", "chaos samples after the gate", "southpolesteve/probably ≠ carldaws/hunch ≠ feelings ≠ Kungie/gut ≠ Illusion47586/judge ≠ tidymodels/probably", "133★ / forks 10 live", "build calibrated classifiers from human feedback", "retrieve by relevance not resemblance", "one calibrated yes/no per memory in one request", "pointer mode 17/18 19/20 *theirs*", "embedding resemblance misses the allergy", "samdotmak/jev-recall ≠ jev-search ≠ jev-sift ≠ carryforward ≠ chopratejas/invalidate", "memory leases ended by new evidence", "six Nouls then fixed rules in code", "0 of 157 false invalidations", "questions/plans/directives are not evidence", "unsure → review queue", "host keeps the store", "name↔body / comment truth / test-claims", "mizchi/jev-lint is mizchi/jevlint rename", "no shipped rule has severity error", "~1 in 5 findings wrong *theirs*", "mizchi/jev-lint ≠ huntedman/JevLint ≠ MichitoSugawara/jev-lint", "JSON Schema → typed JSON via Jev", "noul_threshold 0.5 decoder not a proof", "IncompatibleSchemaError lists every bad property", "on-device Laya CoreML ANE", "~5 ms P50 short decisions", "189/189 FP16 checkpoint parity", "10× not achieved", "mizorewww/laya-coreml ≠ gliner-native-runtime ≠ jevmlx ≠ NandhaKishorM/laya", "softmax over allowed tokens ≠ Noul", "question-first cache", "Micha0827/snapjudge ≠ githubnext/localjev ≠ jevmlx ≠ cendress/SnapJudge", "Jev-first Pi agent loop", "slow-LLM fallback", "explicit action menu / CandidateSource unimplemented", "62 tests wiring not quality", "direwolfiy/JevPi ≠ standardagents/jevpilot ≠ pi-jev-control", "resume-screening bias audit methodology", "name×resume factorial independent Nouls", "callback determined by resume quality", "mean-probability name gaps operationally negligible", "natemoo-re/bias-bench ≠ BBQ", "Plan/PRD panel → code-owned pass|review|block", "cheerleading out of scope", "austindixson/planalyzer ≠ single-goodness Noul", "cost-aware multi-model routing/escalation", "decide vs do", "successful-task cost", "cannacre8ive/switchboard-ai ≠ ha-switchboard ≠ hermes-switchyard", "frozen-protocol zero-shot bench", "TypeSafe Jev vs PrismNLI vs Laya", "contamination caveat", "elcronos/jev-vs-open-decision-models ≠ JevBench ≠ DMB", "context-window admission control", "VOI gate which tokens are worth the expensive model", "fail polarity per lens", "on small inputs lenses lose money", "cvsgireesh/jevusher ≠ jev-sift ≠ winnow", "typed decision control plane", "receipt ≠ authorization", "historical-v0 zero retained cases", "MokiMeow/jev-fabric ≠ jev-forge ≠ dgp", "live 15-dim typed rubric re-score per pause", "scoring economics exemplar", "OpenJev/Codiv ≠ TypeSafe hosted", "jose-troche/live-rubric ~$0.000004 desc / ~$0.000006 README", "adversarial pre-registered Jev eval", "28 predictions before data", "123,805 requests", "confidence does not track ignorance", "polite injection 65% / crude 0%", "willkelly/jev-evaluation ≠ jevals ≠ jev-baselines-eval", "provider-neutral Elixir/BEAM Noul/Choice/Score SDK", "class infrastructure", "nshkrdotcom/system_one_sdk ≠ typesafe_sdk ≠ dannote/jev", "question-linting of Jev questions themselves", "nine jaggedness rules, no API key, no labelled data", "static lint ≠ measured separation", "yodablocks/jevq ≠ tenbin ≠ JevLint ≠ commitjev", "open-weights Laya as class exemplar (binding)", "Nx/Bumblebee runtime", "host chooses backend", "ChristianAlexander/laya_ex ≠ system_one_sdk ≠ dannote/jev ≠ NandhaKishorM/laya", "on-chain/edge Laya deploy", "parity_verified stays false", "model output never grants Tx", "humandebri/IC-Laya ≠ laya_ex", "auditable weekend replica", "Jev outputs never used for training", "soft human-vote distributions", "unpaired 0.577 vs 0.727", "agilabs-ai/jev48 ≠ JevBench ≠ Mapika/decider", "adversarial dual-judge / framing attack surface", "comparative framing is the usable judgment", "prior injection crowds out evidence", "copyleftdev/ember ≠ ember.js", "Laya specialist fine-tune pipeline", "training still GPU-pending", "PIXELZX0/XERON ≠ convaiinnovations/laya", "Hub Laya replica drop", "daliborsb/laya ≠ convaiinnovations/laya ≠ NandhaKishorM/laya", "System One student distillation corpus", "gold is programmatic", "teacher is closed-API clone", "do not distill Jev as teacher of record", "MagaBitmex/jev-4b-distill-data ≠ missing student checkpoint", "non-LLM VIN System One", "planning depth not chat", "lewislululu/jevon ≠ douglance/jevon", "source-bound evidence checks", "local quote mismatch needs no API", "exit 0 ≠ claim truth", "WaynezProg/jev-kit ≠ jonathanavis96/jev-kit (Airlock) ≠ jev-use ≠ jev-mcp", "independent System One evidence catalog", "scores not one leaderboard", "no external record currently reproduced", "TokenTrim no-Jev matched hybrid 62.4%", "reachjalil/system-one-bench ≠ mallahyari/system-one-benchmark", "21 tasks · 134 items · 208 questions", "scenes from public GitHub contracts, not production logs", "SivletLabs/jev-eval ≠ willkelly/jev-evaluation ≠ 4esv/jev-eval ≠ xxkuboxx/jev-eval ≠ onlyoneaman/jev-eval ≠ dayhaysoos/jevals", "option isolation (sibling-blind)", "permutation-equivariant", "Hub OWNER not published", "nafisazizir/hev ≠ jaredpalmer/kev", "frozen local LLM logits, no trained decision head", "residual-head 9,222-param decreased 73/96→67/96", "confidence = 1−normalized entropy, not P(correct)", "yuki-oshio/mini-jev ≠ r-ms/mini-jev", "Jev classifier as autoregressive next-token predictor", "ChatJev-style soundness theater", "erik-dunteman/ChatJev ≠ dannote/jev ≠ jev-gpt", "calibrated decision head × AlphaProof value head", "implementation-layer isomorphism, semantic difference", "timeout = censoring", "do not launder Noul as proof", "parallel rank-prediction vs serial selection", "independent questions can conflict", "zzzzzec/jevsort ≠ keltokhy/jsort", "curated open System One ecosystem catalog", "rupeshpoojary9/awesome-open-system-one ≠ AnotiaWang/awesome-jev", "arXiv paper radar with Jev relevance scoring", "ranking ≠ calibration / 0.5 still soft", "fail-open failed evals not marked seen", "train calibrated ~27M from scratch", "typed Q→prob dist / one forward pass / no LLM decode", "hyusi2003/MiniSystemOne ≠ Colvin0315/MiniSystemOne", "description-only stub / size 5", "ESCI hard probe fails four of six", "jev_bool ECE 0.242 inversion 0.255", "do not re-fold §60 six-gates as new", "jobbyjev one-request-per-company from batch-size result", "find/design/evaluate TypeSafe Jev decision loops", "karanb192/jev-architect ≠ samtay32/jev-system-architect", "Jairik/jev-distiller size 1", "distill-Jev UI stub / do not distill Jev as teacher of record", "post-launch scored use-case map / Jev self-scores then human curation", "licensedsaucer9-web/jev-opportunities", "Jev-inize a use case into classifier/router", "gavinHuang/jevinize → simple-jev not TypeSafe", "featherless-ai/simple-jev", "compare saved decisions / same label can still change the branch", "VihaanAgarwal/jev-diff ≠ Saik0s/diffusiongemma-jev-macos", "not tested with a live Jev API key", "constrained logprob + temp/Platt ≠ Noul", "OpenJevPro pastes openjev-sglang JevBench as own", "zhangcy122/OpenJevPro ≠ IamBusy/OpenJev ≠ ekzhang/openjev-sglang", "PolyForm Noncommercial", "SmolLM-135M / sub-70ms / 0 output tokens", "demo P(True) 0.5052 / Choice conf 0.2872 / Score conf 0.0055", "README claims MIT / GitHub license null / no LICENSE file", "patelvishwa112/jev-system-one-rlcd ≠ arnabgho/rlcd-lite ≠ blackwood-rlcd", "source-backed Awesome Jev radar / 306+ commit-pinned", "logicrw/awesome-jev-projects ≠ AnotiaWang/awesome-jev ≠ yibie/awesome-jev ≠ cobanov/awesome-jev ≠ rupeshpoojary9/awesome-open-system-one", "auto GitHub sync / Issue-only submissions", "hashed n-gram encoder / rival-aware attention", "olanotolu/jevbetter vs jevlike starter", "synthetic hard menus top-1 0.916 vs 0.873 / ECE 0.0182 vs 0.0367 / 40 vs 4608 menus/sec", "shuffled-context control 0.335", "Turn any open LLM into System-One Jev", "uspraveen/Jevify ≠ Mintzs/jevify ≠ gulagala001/jevify", "Jevify-any-LLM architecture probe", "description-only stub / size 0", "Train encoder-only calibrated decision models from a task sentence", "Exu is a toolkit, not a method", "strictly proper scoring rule", "Pre-alpha", "Ruivalim/exu-base", "scratch-trained calibrated decision model", "typed Q → probability dists", "Colvin0315/MiniSystemOne ≠ hyusi2003/MiniSystemOne", "no published weights download URL", "90.5 seconds / 29.2% pipeline evidence", "p_i/p_j independent of other candidates", "Recipe for calibrated decision models — small model out", "init → synth → train → eval → serve", "91.1 % / ECE 0.022 *theirs*", "Jev zero-shot 75.1", "scienthoon/luce", "Put Jev's three headline claims on trial", "0.5B local GPU", "46x speedup / accuracy identical", "ECE 0.624 sentiment catastrophe", "bigger model worse calibration", "RichardoMrMu/jev-mini ≠ yuki-oshio/mini-jev ≠ r-ms/mini-jev", "System-1 decision engine for local LLMs", "structured choices only", "JSON parse of generated text ≠ Noul", "TypefAI JEV / Journal Entry Voucher", "tapsin/jev-local ≠ us/jev-local ≠ Argos1111/jev_local", "Jev 1.13 reward-model eval across 8 benchmark tracks", "40,940 examples / 0 API errors", "RewardBench v1 92.58%", "Precise IF 50.63%", "goya4140/jev-reward-model-evaluation", "Scaffolding in progress", "Jev vs LLM support-ticket routing", "static + live decision bench", "TypeSafe's own published benchmark", "illustrative simulations, not live API calls", "JevBench v1 — smart/cheap/fast/reliable", "I/C/S/K 25% geometric mean", "classifier.dev fast tier 84.8 is Jev behind its own API", "do not re-fold §78 v1.2 board as new", "Laya (421M) 70.1 now on board", "Zero-shot/few-shot LLM routing", "hard budget filter before Jev", "Jev never asked to perform budget arithmetic", "Jev judges the next state, XState enforces transitions", "simulation uses synthetic keyword fixtures", "catalog gravity", "v-modal/awesome-jev-tools", "★339 live REST", "curation is not endorsement", "crawler-maintained directory", "Daily GitHub + npm sweep, human-merged", "RadRebelSam/awesome-jev ≠ AnotiaWang ≠ yibie ≠ cobanov ≠ logicrw ≠ v-modal", "HF peft SPLADE/BGE reranker", "rdxtremity/jev-reranking ≠ carlaiau/jev-reranking", "query-side encoders, not a Jev replica", "ONNX System One Qwen3.5-4B scorer", "source:pngwn/system-one-qwen3.5-4b-scorer", "CC-BY-NC-4.0", "temperature 1.75", "transformers.js AutoModel cannot load this graph", "Consistency benchmark Space", "This Space contains no benchmark result yet", "12-case plumbing fixture", "Benchmark-driven Jev router and judge", "cheap alone is not success", "Jev does not write, sum prices, or claim accuracy %", "Sol 94.2 / Luna 83.9 / Jev path 89.7", "19.2% Sol / 62.3% cost save / 4.5pp miss of 2pp non-inferiority", "p50 latency worse than Sol due to routing overhead", "erendikmenn/jev-llm-router-benchmark ≠ jev-rag-benchmark ≠ ryantsai/jev-llm-router", "Express + node:sqlite", "mock and Jev decision engines", "previous_ticket_count >= 3 is code", "MIN_CONFIDENCE 0.6 still soft", "substring false positives", "aesaganda/jev-ticket-router ≠ SarathChandraBellam/jev-vs-llm-ticket-router", "Universal Figure & Diagram Router", "confidence ≥ 0.85 hard-gate is theater", "generative AI banned from scientific plots", "six visual branches", "hoangngochuong24947-gif/jev-figure-router", "human-labeled (state, question, label)", "166,054 rows / 22 configs", "soft_label for human uncertainty", "Praveenrajus/jev-bench ≠ fstandhartinger/jevbench", "ternary bonsai System One GGUF", "openjev's mechanism, Bonsai's weights", "Hub does not ship weights", "100/100 easy T/F is not Harbor", "label_mass ≠ correctness", "stock llama.cpp Q2_0 silently gibberish", "NicolaiMTLassen/open-bonzi-jev ≠ NicolaiLassen", "transformers.js DeBERTa ONNX", "source:com-kotobalabs/open-jev-deberta-v3-large", "temperature 1.05", "AutoModel from_pretrained works", "onnx-community/open-jev-deberta-v3-large-ONNX ≠ system-one-qwen3.5-4b-scorer-ONNX", "107★ densify", "GH 151M vs README 149.6M", "PR #1 now closed unmerged", "do not re-fold §71 claim-audit as a beat", "typed decisions, RLCD, confidence-gated routing", "structured ≠ correct", "mock not live API", "26 tests", "wjdjdakf17/jev-study ≠ baekenough/jev-study", "bonzi-27b-v2 / ternary-8b / 27b-v1 GGUF family densify", "WANLI-256 74.6% / 65.2% / 71.1% *theirs*", "Bonsai 1 27B Q1_0 runs on stock llama.cpp", "ternary still needs PrismML fork", "hf:heman10x/openJev-verdict-2.0 twin tokenizer-only", "OpenJev Vision image classification + uncertainty", "CLEVR-4 held-out joint 0%", "hfdataset:IamBusy/OpenJev-Vision-Research-v0.1 12,832", "294,912 derived targets not independent samples", "Laya multilingual ONNX WebGPU typed-decisions port", "63/63 selected answers / 5.1e-4 CPU / 1.2e-2 WebGPU", "UpHash-Network/mini-jev is yuki-oshio transfer", "jev-injection-bench 11,900 labelled prompts", "Jev best ranking / Haiku better ECE 0.021 vs 0.058", "0.5–0.9 band is where Jev's numbers do not mean what they say", "Prompt wording moves panic 28%", "manojlds/jev-dspy-bench ≠ dspachos/jev-dspy ≠ jmanhype/jev-dspy-lab", "Jev agreement is similarity, never ground truth", "no aggregate quality grade or merge gate", "AbstentionBench-on-Jev rank 1 of 20 vs 2025 field", "question-asymmetry", "forward-looking 0.465 never extreme", "openkev calibration layer not a runtime", "ECE vs coverage independent", "select_threshold returns inf", "escalation catches uncertainty not ignorance", "misakaikato/openkev ≠ jaredpalmer/kev", "pdf-race Docling→Jev vs Gemini", "parser owns the wall clock", "12/12 tie is a tie", "titles selected not generated", "flopcheck 16 calibrated tweet judgments", "mechanical tells in code", "ZeroX-01/jev-atlas ≠ Zaious/jev-capability-atlas ≠ gorock007/jev-atlas", "Laya calibration lab Gradio MCP", "T never changes argmax", "confidence ≠ top-label p", "easy probe set refused", "40–48 rows too small to ship T", "Gemma-4 26B-A4B jevify classification+calibration", "LoRA adapter twin not independent eval", "Gemma-4 E4B jevify", "E4B LoRA stub card", "kushalpatil/jevify-gemma4 ≠ Mintzs/jevify ≠ gulagala001/jevify ≠ uspraveen/Jevify", "GH kushalpatil07/jevify 404", "PAWS 0.580/ece 0.288 is the weak cell", "smaller E4B slightly better OOD ECE than 26B-A4B", "Hub jevify merged LoRA ships weights", "bonzi Bonsai-8B v1 GGUF densify", "Bonsai-1.7B v1", "Bonsai-4B v1", "WANLI-256 64.5% / 60.2% / 52.0% *theirs*", "rank #4 / #5 / #6 of 6", "JulesHuisman/jev-eval scaffolding / README SHA c356a584 (was empty e69de29b)", "JulesHuisman/jev-eval ≠ SivletLabs/jev-eval ≠ willkelly/jev-evaluation ≠ 4esv ≠ xxkuboxx ≠ onlyoneaman ≠ dayhaysoos/jevals", "7 bands 6/10 vs 40 bands 0/10", "source receipts + confidence slider re-policy without re-inference", "32/32 synthetic is smoke not production", "classify HF datasets across typed semantic dimensions", "roadus2 watch misspelling; lock roadius2/ultra_laya", "ultra_laya REVIEW defects", "default branch claude/laya-jev-review-gg5ppo", "XNLI EN 88.3% ECE 0.032 → RU 77.3% ECE 0.096", "Δ −11.0 pp [−14.2,−7.8]; ECE +0.063", "MASSIVE no detectable difference at n=600", "confidence is function of p_max (r=1.000)", "pointer-not-generator 400 human-authored responses", "proposed ≠ authorized", "FewRel 160: Jev 85.0% vs lexical 13.125%", "gated 100% (95/95) coverage 59.375%", "J++ composable semantic computation language", "judge-jev 0.5 still soft", "947 repos scored; A 273 / B 302 / C 372", "LLM rubric ≠ benches", "No benchmark winner is claimed", "phishing: naive 62.6% vs regex 91.8%; 5-atomic + LR 95.0% *theirs*", "AITuber tension ±15", "README npm global; repo is Rust", "git-confess code owns counting/blame/ratio", "httpx exhibit 11% (13/119) *theirs*", "90d trend +12.40% vs random +12.75% vs BH +41.71%", "5m win rate 25%", "Awesomejev 656 entries / 38,160 stars", "tracker likes 64 (+4) lastModified UNCHANGED", "Laya present; Blackwood ABSENT; Archer still promised_not_landed", "Blackwood tracker ABSENT; likes 2 gated manual", "r = c - p_a", "ECE 0.021; acc 0.807 vs warmup 0.746", "Independent primitive", "11.57s vs 54.10s · 4.67× · 120/128 *theirs*", "default path is pretrained Gemma probs not trained RLCD head", "GH Meanblock 404; lock leesk212/JEV-CPU", "softmax over letter slots ≠ Noul", "WANLI 0.741 vs openjev v2 0.77 *theirs*", "3-way NLI ≠ Noul", "priority 0.464 = majority floor", "banking77 contaminated", "raw margins not probabilities", "do not distill Jev as teacher of record (they distilled Haiku)", "“0.9 is not one number”", "ranking ≠ calibration", "banking77 0.8–0.9 stated 0.86 actual 0.73 over-confident *theirs*", "≠ Praveenrajus/jev-bench ≠ fstandhartinger/jevbench", "$0.0000153–$0.0000226 vs circulating $0.0004 (~20×)", "Score is 0..n-1 expectation not 0–1", "Noul has no confidence field", "TCP floor 198.8 ms", "type reliability is not a reason to choose Jev (json_schema 5/5)", "gateway tax not one number", "Function-only 5/8 vs hybrid 8/8", "4/8 without Jev", "8 designed cases not conversion lift", "200-row pilot Jev 86.5% 173/200 vs Gemini Flash-Lite 86.0% 172/200 vs Pro 87.0% 174/200 *theirs*", "not a ranking", "情緒測謊器", "8-example Jev vs GPT-5.6 Sol ~64× cost 5.4× latency *theirs*", "synthetic; no inference", "≠ JevBench v1.2 §78", "Judged 3317 / listed 2560", "Jev judges, code applies policy", "APA “microsecond policy / zero hallucination” overclaim", "Client-side quiz; pointer from held docs; scanned-PDF warn", "Jev judges / agent reasons / user decides", "selecting an option is not permission to implement", "pattern exact, judgement must clear floor", "no matching pattern → no model call", "not a correctness oracle", "Spec vs artifact remainder", "treating 0.85 as 85% / minProbability hard-gate as Harbor", "VERIFY acquires discriminating evidence, never same-pool confidence-only rescoring", "fast/full/max are ceilings not sizes", "Solar writes, Jev chooses NEXT ACTION", "do not reopen or amend PR #23 or #24 or #25 or #26 or #27", , "Calibration is not alpha", "NO CURRENT ALPHA CANDIDATE", "ΔR² approximately +0.00084", "Brier 0.2131387", "ECE 0.0421875", "Adding Jev probability to deterministic volatility improved Brier by only 1.4058e-05", "default 0.5 keeps zero non pinned", "keepResult median 0.14 to 0.17", "keepCall median 0.28 to 0.35", "usable range is about 0.10 to 0.25", "7.8% to 57.9%", "judges results it never sees", "task-finish eval not built yet", "$0.002 per compaction", "slavadubrov/sgr-judge-bench ≠ slavadubrov/jev-judge-bench", "Jev 108/120 $0.083 0.34 s", "Luna SGR 114/120", "paired Jev accuracy-difference intervals include zero", "not evidence of equivalence", "GLM SGR 26/120 93 format failures", "Terra-planned Jev hybrid 55/120", "rule-based by default, optionally Jev-backed", "empty README", "missing key cannot break the experience", "prefill plus exactly one decode", "softmax over A/B/C ≠ Noul", "BBQ 9,053/10,000 (90.53%)", "ECE 0.0890", "Mean confidence 0.9943", "overconfident", "score and noul not implemented", "DGUI 12 rows (was 6)", "INSTRUCT 119 rows likes 2", "encode the state once, decide everything in parallel", "0.740 accuracy against a 0.508 majority", "ECE 0.047", "fine-tune's advantage ends where its 384-token training data does", "jasonkneen/open-jev ≠ pngwn/open-jev", "same sha d41dc3cd", "Space does not call Jev", "recomputes routing from saved probabilities", "200-case Jev 97.0% / 100.0% / 95.0% / MAE 9.22", "synthetic repository benchmark", "Jev evaluations are advisory", "YehuiTang0316/jev-nlgrep ≠ Bentlybro/jevgrep ≠ can1357/jegrep ≠ uehaj/jev-semgrep", "default threshold 0.8 still soft", "40-line windows cannot prove whole function", "token-native sequential start/end Choice", "Gemini/Haiku stubs not configured yet", "handful of hand-written examples, not a benchmark", "Jev judged exactly what it was given", "laguagu/jev-skills ≠ laguagu/jev-evidence-lab ≠ Pleo2/awesome-jev-agent-skills", "contract_passed is not a claim of guaranteed factual truth", "Wilson lower bound 0.85 floor", "fixture mode no savings claim", "SemIf 2207★ (+21 vs §110 2186)", "jevlike 1043★ (+5 vs 1038)", "TypeAR 15★ (+1 vs 14)", "AnotiaWang 97★ (+1 vs 96)", "yibie/awesome-jev 506★ (+16 vs 490)", "Laya likes 822 (was 802)", "tracker likes 64 flat, lastModified UNCHANGED", "do not reopen or amend PR #23/#24/#25/#26/#27/#28", "Heman10x-NGU/Verdict-open-jev ≠ Heman10x-NGU/openJev-verdict-2.0", "TF-IDF + LogReg ECE 0.0207 vs Jev 0.1440", "Verdict-open-jev 48.07% vs Jev 90.80%", "abstention combined recall 10.00%", "p50 35.58 ms", "K=25 (maximum capacity) 72.00%", "0.85 coverage 84.60% selective risk 1.18%", "26.1× faster than standard Qwen JSON generation", "Jevify 90.0% / 167 ms CUDA graphs disabled", "Finding 1: Brier on stated confidence alone is a trap", "grpo_rlcr 0.78 / ECE 0.084", "reliability 0.007 but resolution 0.000", "27 900 schema-driven decisions", "13 600 / 13 600 questions", "candidate mass min 0.99999624", "22 configs · 166,054 rows · 4 calibration-gold", "sha a39eba3f", "Student B MAE 0.148 / Pearson 0.836 / 86.0%", "pngwn/open-jev-laya-bench README 404", "sha 9f69c742 likes 2", "HDFS 0.9933 (745/750) / retain 0.0084", "BGL ERROR/FATAL protection 1.0000", "2,479 / 2,500 HDFS uncertain", "cache hit 0.9648 (2412/2500)", "$0.153936 estimated", "E2 recomputes from saved probabilities", "Space sha eda59e0a", "MASSIVE English 0.783 / Khmer 0.033 / Hindi 0.133", "40–48 rows too small to ship T", "T never changes argmax", "siren2345/jev-single-decode-transformers ≠ siren2345/jev-single-decode", "Split Transformers experiment from llama.cpp runtime", "tanayvasishtha/jev-lab ≠ dairui1/jev-lab ≠ BrendanH18/jev-lab ≠ yibie/laya-jev-lab", "Four experiments stress-testing TypeSafe's Jev: calibration, bundle bias, label bias, and ensembling", "second pass must be $0.00 from cache", "The pages never call Jev", "Gemma 4 31B 77.0% / Jev 1.13.0 61.4% / Laya 322M 0.0%", "restriction state 95.0% against 84.4%", "None of the systems are particularly good at knowing when to stop and ask", "They skip the question and call a tool directly", "100% schema pass", "six-field joint 48.8% vs 72.8%", "ywchiu/jev_benchmark ≠ Running-Dolphins/jev-bench ≠ Praveenrajus/jev-bench", "ACT / REVIEW / FALLBACK", "A provider failure, timeout, malformed output, or missing answer is **not** a policy outcome", "confidence is descriptive provider output, not a substitute for probability", "Quality denominators include only valid scored answers", "an exact halfway tie chooses the lower level", "aiwithenoch/Jev-Skill ≠ simplosophy/jev-skill ≠ laguagu/jev-skills", "The local path does not claim to turn a smaller checkpoint into Jev", "Low support becomes decision: \"review\"", "MIT-0 SPDX NOASSERTION", "current-llm", "结构兼容,不是 Jev 模型能力", "altryne/jevify ≠ Mintzs/jevify ≠ gulagala001/jevify ≠ uspraveen/Jevify", "Find where Jev belongs. Design the questions. Measure the difference", "TypeAR-AI/TypeAR 301 → TypeLLM/TypeLLM", "TypeLLM/TypeLLM 16★", "SemIf 2241★ (+34 vs §111 2207)", "jevlike 1051★ (+8 vs 1043)", "AnotiaWang 98★ (+1 vs 97)", "yibie/awesome-jev 525★ (+19 vs 506)", "Laya likes 864 (was 822)", "tracker likes 67 (+3 vs 64)", "lastModified UNCHANGED `2026-09-20T04:29:16.000Z`", "do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#32", "hysteresis enter/exit / replay policy without inference", "calibration does not compose / hop-ECE permutation-invariant", "equal-width vs quantile ECE / ranking ≠ calibration", "Qwen2.5 ≠ Archer / Qwen 3.8 sparring ≠ Archer / Qwen/Qwen3.8-27B ≠ Archer", "Deferred Crispification / TCE / AMS", "g0runmezadam/what-is-jev IS tunahansahin897/what-is-jev", "pd.cut equal-width vs jeval quantile", "A hunch is a probability with a policy attached", "soundness theater / measurement theater / hourly 0843", , "Jev Capability Resolver / NiazMorshed2007/jcr", "one tool nested capability tree / returns context / does not execute", "skills vs capabilities / workflow+judgment vs operations", "format independent of Jev / proposed open standard", "JCR_BAND_RATIO 0.6 is application policy / soft scores ≠ hard gates", "routing ≠ permission / docs ≠ authority to run", "sol-vs-opus5-20 lookup+explain / n=1 / Not Harbor task-execution", "wall-time mixed / Sol slower with JCR in 19/20", "NiazMorshed2007/jcr ≠ skill-broker ≠ skillranker ≠ jev-sift ≠ jev-lens ≠ jevusher ≠ jev_select_capability", "do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34", "notes.md §116", "copy the SemIf/MLX installer?", "quote 5.21× as beating Jev?", "treat 0.845 as a TypeSafe replica?", "collapse SemIf into kw2828/zhihz/semif-rs/semif-serve", "softmax over options as a Noul", "llm prompt to jev primitives", "conversion assistant not equivalent behavior", "heuristic conversion ≠ calibrated Noul", "alexwestco/llm-to-jev ≠ altryne/jevify", "user-provided 0940 / notes.md §118", "judge ≠ actuator", "candidate_mass", "softmax over A–H ≠ Noul", "hourly 0947 / notes.md §119", "ggmlc GGUF is not llama.cpp", "serving substrate ≠ calibrated replica", "Qwen3.5-9B ≠ Archer", "planner writes JEV selects", "hourly 1049 / notes.md §120", "open recreation ≠ calibrated replica", "semantic lint is a sensor not a proof", "cutoff 0.8 still soft", "paired bootstrap CIs *theirs*", "Same accuracy, 35x faster *theirs*", "hourly 1143 / notes.md §121", "revisit HIGH / since-last-look", "catalogued repo changed", "star-noise vs material change", "densify prior notes without inventing equivalence", "decide is not generate", "tryDecide returns typed calibrated judgments not a token stream", "GLiNER/GLiClass ports are class members not Jev replicas", "93.5% *theirs* not Harbor", "74.9 *theirs* not Harbor", "8.7x *theirs* not Harbor", "Option-Marker joint attention", "openjev:0.2.1", "thinking=True/False per-field budget", "PLAN_Qwen35", "hyperspaceai/jevcache ≠ kushals256/jevcache", "wire-compat ≠ logit-equiv", "SHA move is not a replica", "hourly 1248 / notes.md §123", "typesafe-sdk 0.7 Pydantic response models", "msgspec dropped", "The server's output is unchanged and was never wrong", "SchemaError is 400 plain-string detail not 422 list", "Pydantic response models ≠ logit-equiv", "msgspec dropped is not a replica", "Error contract is not a Noul", "coverage-at-error-budget *theirs* not Harbor", "PLAN_Qwen35 still proposal for review", "GLiNER locate ports are class members not Jev replicas", "Locate ≠ decide", "~160 ms *theirs* not Harbor", "0.971 F1 *theirs* not Harbor", "hf:fr0stbit3/laya-gguf serving substrate ≠ calibrated replica", "jkcdarunday/SystemOne-Next ≠ TypeSafe System One", "hourly 1340 / notes.md §124", "vLLM NVIDIA + MLX Apple Silicon", "Codiv hosted free endpoint", "dual /v1/systemone + /v1/chat/completions", "chat 501 on MLX", "dual serving is not generate", "Hosted Codiv ≠ TypeSafe", "hr98w/jev-visual 167★ Apple Silicon visual candidate scoring", "37.30s → 2.40s at 64 decisions *theirs*", "Breakout 9 bricks 6 returns 2 lives *theirs*", "candidate probabilities are relative not correctness", "jkudish/jev-mcp 156★ ten MCP tools", "recommendation is advisory", "the server never blocks on its own", "TypeSafe CLERC 5% to 18% *theirs*", "jkudish/jev-mcp ≠ burnigtm/jev-mcp", "zhengxuyu/litjev off-the-shelf Qwen decision layer", "Probabilities are not calibrated by default", "Qwen/Qwen3.8-27B ≠ Archer", "zhengxuyu/litjev ≠ alexwestco/llm-to-jev", "Zefan-Cai/Open-Jev LoRA + scalar head", "2B 94.71% 9B 97.54% hard test *theirs*", "2B OOD 86.02% 9B OOD 91.97% *theirs*", "80,816 training rows", "27B still in progress", "LoRA ≠ RLCD replica", "Zefan-Cai/Open-Jev ≠ TheoLeeCJ/openjev ≠ razorback16/openjev", "cristianoliveira/jeq intelligence you can pipe", "pass-min 0.8 still soft", "JEQ does not own actions", "AndyInQtr/laya-coreai CoreML serving substrate ≠ calibrated replica", "AndyInQtr/laya-coreai ≠ mizorewww/laya-coreml", "hourly 1441 / notes.md §125", "TypeLLM/TypeLLM densify HEAD 6a48f9f1e623", "README densify 3k→12k B", "Batch 5.8x *theirs*", "Constrained AR ≠ calibrated Noul", "jaredpalmer/kev densify HEAD b339f446a0ef", "Kev-0.6B 4B 8B family", "4B new-source 0.790/0.806 *theirs*", "8B new-source 0.796/0.780 *theirs*", "Jev hosted 0.857 *theirs*", "Questions share the input text but cannot read each other", "No Jev outputs were used for training", "8.2% ≥0.9 on wrong *theirs*", "option order can change an answer", "Qwen3 ≠ Archer", "TheoOliveira/pi-jev 21★ fail-closed routing", "JEV_THRESHOLD 0.65 still soft", "harshwasan/jev-sentinel fail closed never auto-allows", "harshwasan/jev-sentinel ≠ leepokai/jev-guard", "jackbarunz/jev-tool-router ≠ esinocchi/jev-tool-router", "threshold 0.90 still soft", "76/81 vs 77/81 *theirs*", "0.419s vs 2.459s *theirs*", "$0.00486 vs $0.03673 *theirs*", "not a security boundary", "baronunread/leanest fail-open uncertainty means RUN", "classifier.dev default Jev/Laya pluggable", "openlayer-ai/jevals ≠ dayhaysoos/jevals", "estimates not Harbor", "classifier ≠ authorizer", "MrJev/awesome-jev 118 entries catalog ≠ endorsement", "MrJev/awesome-jev ≠ yibie/awesome-jev", "Koushik890/jev-firewall fail closed ask_below 0.7 still soft", "CompleteTech-LLC-AI-Research/jev-codex-approval experimental native not compiled", "confidence is not a measured probability", "rh-guard owns primary gates", "hf:rAVEUK/open-jev-deberta-v3-large encoder class member not Jev replica", "hf:p-yan/laya-quanto serving substrate ≠ calibrated replica", "hf:Gtrkrsk/laya serving substrate ≠ calibrated replica", "hourly 1542 / notes.md §126", "razorback16/openjev densify HEAD febf02e88989", "release 0.3.0", "re-pin vLLM PR #57250 restructured head", "MODEL_VERSION stays openjev-0.1", "uv.lock hygiene", "restructured vLLM head ≠ logit-equiv", "frostney/clean-code-review 7★ typed judgments not opinions", "documentation is read not judged", "morcoan/JMP Joint Model Participation", "Models participate. Real tools execute.", "Thresholds are policy not model", "Kelbie/hunch ≠ carldaws/hunch ≠ tpellet/hunch ≠ huncho", "Jev never generates prose JSX or code", "json-render is the only renderer", "game success ≠ calibrated Noul", "Shalimov04/open-jev ≠ razorback16/openjev", "MstyAI/laya-onnx empty repo", "hf:Praveenrajus/jev-bench HTTP 200 was 401", "hourly 1643 / notes.md §127", "TypeLLM/TypeLLM densify HEAD 702e6a287f3c", "truncated thinking then constrained decode", "0.8B thinking On 0/18 *theirs*", "forced closure 20/20 type-valid *theirs*", "jaredpalmer/kev densify live HEAD 8465c4c4c294", "Kev-0.8B completes family", "4B new-source 0.794/0.832 *theirs*", "9B new-source 0.812/0.837 *theirs*", "transfer-v9 Kev-9B 5% Jev 9% Kev-8B 26% *theirs*", "SemIf Kev-9B 0.917 Jev 0.965 *theirs*", "scienthoon 0.952/0.911 vs 0.897/0.914 *theirs*", "transformers >= 5.17", "Qwen3.5 ≠ Archer", "notque/vexjoy-agent 421★ /d routes /do fallback", "Facts go to code. Judgments go to Jev. Only facts can block.", "Jev never blocks", "jqueryscript/awesome-jev ≠ MrJev/awesome-jev ≠ yibie/awesome-jev", "five-lines threshold 0.80 still soft", "371ms $0.0000189 300-call *theirs*", "tpellet/jevify ≠ altryne/jevify", "seb4ez/jevguard-mcp ≠ seb4ez/jevguard", "resumocast/jev-mcp ≠ jkudish/jev-mcp", "Adrian-Ernesto/jevsort ≠ zzzzzec/jevsort", "MidasMulli/kev-ane 155/155 argmax *theirs*", "hourly 1746 / notes.md §128", "Fine-tuning on your own data", "--data JSONL", "--init_from warm-start LoRA/head PR #9", "from-scratch ≠ warm-start", "JSONL labels ≠ Harbor", "Kev-0.8B 4B 9B Qwen3.5 family", "0.33 vs 0.84 vs 0.83/0.88 *theirs*", "reconstruction ≠ replica", "assay-001 split verdict", "Promethe-us/awesome-jev ≠ MrJev/awesome-jev ≠ yibie/awesome-jev", "ThePFMind/jev-mcp ≠ jkudish/jev-mcp", "kyegomez/open-jev ≠ razorback16/openjev", "namenu/pi-jev-effort ≠ TheoOliveira/pi-jev", "samatv256/mini-Jev ≠ r-ms/mini-jev", "hourly 1843 / notes.md §129", "TypeSafe-compatible ≠ TypeSafe replica", "SystemOne.from_pretrained", "replica ≠ TypeSafe", "76.7% vs Jev 86.9% strict common subset *theirs*", "kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions", "DeBERTa-v3-large 0.855 / 42 ms *theirs*", "aisearchio 15-link census catalog ≠ endorsement", "user-provided 1936 / notes.md §130", "systems latency ≠ semantic equivalence", "hard acc ≠ calibrated Noul", "Open-Jev TREC pending", "GPT Luna P50 918.13 ms Astra 1938.39 ms *theirs*", "TREC-DL Jev/Luna/Astra completed", "customer-service P50 local HTTP 85.03 ms vs Jev HTTPS 295.26 ms *theirs*", "1024 tokens/32 candidates Open-Jev slower 1015.90 vs 301.37 *theirs*", "prefix caching experimental/off by default", "not merged base models", "Open-Jev densify HEAD 4933ee84951f", "Astra TREC commit 1dd56990be7e", "densify §125 not a sibling first sighting", "Open-Jev densify / notes.md §125", "launch X thread https://x.com/Zefan_Cai/status/2101782158658695388", "2101786019607740436", "2101789698947793231", or "cascade sign-flip / calibration theater": read `references/faq.md`, then `references/mental-models.md`, then `references/mixed-architecture.md`, then `references/judgment-class.md` before any mapping. Proof, @@ -153,7 +153,7 @@ Merged #44 owns §121. This protocol is §122. | Familiar method | Judgment shape | Detail | |---|---|---| | Mental models across domains (not SWE-only) | EU, abstention, VOI, MCDA, SDT, search/control, Leveson, NATM/Norman/snap-fit; **extractable-from-state boundary map** (self-contained vs needs outside knowledge). **Apply queued 1047 follow-ons** (`notes.md` §88): **Replica honesty** / **Empty ≠ approve** / **Beam as control** / **Cache is the exact envelope**. Apply 1144 (`notes.md` §89): **Typed if** / **Shadow then honor** / **Human every action** / **Atom then sense** / **File by Choice** / **Question preflight** / **Inbox read-only vs write**. Apply 1241 (`notes.md` §90): **Observe→score→act (namesake lock)** / **Decision-as-ranking** / **Native vs schema-guided Harbor** / **0 promotions / authored vs real** / **Offload + classifier-not-generator** / **Collapse late** / **Unofficial toolbelt** / **Pointer shell** / **Preview-first VOI / rubric rewrite** / **Life fail-open covers** / **S1 decide / S2 plan** / **Seed/expand/judge/verify + local daemon ≠ Jev**. Apply 1347 (`notes.md` §91): **Judge harness as control API** / **Batch packing VOI** / **Calibration as product** / **Decision-as-Plugin** / **Policy-constrained skill select** / **Tiny local econ pruner** / **Observe→score→act cousins** / **Deterministic verify ≠ System One** / **Soft-score vs hard-argmax**. Apply 1441 (`notes.md` §92): **Self-hosted econ** / **Soft-judgment gate integrity** / **Retrieval as calibrated decision space** / **Enterprise reflexes** / **Screenshot-free CU** / **Hybrid S1/S2** / **Harbor-jevals / SRE** / **Laya densifies** / **Demos / unofficial toolbelt**. Apply SIGNAL jevcache/jev-align (`notes.md` §93): **Decision ledger / memoization** / **GEPA alignment loop** (fingerprint after redact; recall vs decide; publish fingerprints+answers; CI replay as Harbor cousin; Cache hit ≠ correctness; hyperspaceai/jevcache ≠ kushals256/jevcache; human labels only; score never auto-accepts; production capture flywheel; sutro-sh/jev-align ≠ caiovicentino/jev-align). Apply SIGNAL enzyme/JA ModernBERT/Gemma (`notes.md` §94): **Compile-time System One / questions-as-index** / **Unofficial JA ModernBERT cross-encoder** / **NAR class legitimacy / multimodal observe→decide / Router-OOD** (guidance ≠ hook; catalysts ≠ summaries; compile-time System One; unofficial ≠ TypeSafe; format_version modernbert-jev/1; Argos1111/jev_local ≠ us/jev-local ≠ kunchenguid/local-jev; LFM default ≠ ModernBERT backend; Nemotron ≠ TypeSafe Jev; not a calibrated replacement; djev-dev complements djev-spark; images as Choice options; Laya essay numbers *theirs*; Router/OOD confidence; hosted bootstrap ≠ silent TypeSafe). Apply 1541 (`notes.md` §95): **Decision-as-plugin** (difficulty + policy thresholds + JSONL trace; jev-codex-pilot model + reasoning depth; keep/shadow/hybrid/reject) / **Evidence projection** (quarry evidence projection) / **Soft judgment integrity** (jevguard calibrator/cache/escape; jev-ci-selector CI shadow mode) / **Physical/control** (Frank-ZY-Dou/awesome-jev robotics/3D/control) / **Harbor-jevals injection-firewall** (one-dollar-tahoe TypeSafe Jev defense eval) / **llama.cpp replica** (llama-jev llama.cpp replica; petercr/jev-orchestrator ≠ FleeexCorp/jev-orchestrator; seb4ez/jevguard ≠ AseemPrasad/JevGuard ≠ pablozr/JevGuard; webNeat/llama-jev ≠ WiktorB2004/llama-index-jev) Apply 1639 (`notes.md` §96): **OpenCode stdout-prune host port** (OpenCode jev-pruner context sieve; observe→score-candidates→prune; jev-zen / jev-1.13-free; zen-chat ≠ Noul; fail-open original; keepScore >0.1 floor; host port of tamaratran/jev-pruner; indiejoseph/opencode-jev-pruner ≠ nrdz-labs/fast-jev-opencode; jev-webagent-bench empty stub; Kiln-AI/jev_jsonschema noul_threshold 0.5; NSStudent/JevSwiftSDK unofficial). Apply SIGNAL gliner-native-runtime (`notes.md` §97): **GLiNER2 native Apple path** (unofficial Swift/Core ML GLiNER 2.5-small; entity spans + confidence; not Choice/Score/Noul; not TypeSafe; label descriptions as schema; on-device ANE economics; honesty locks; shershah1024/gliner-native-runtime ≠ Fastino; ≠ gliner25-compaction ≠ gliner2-ultrafast ≠ Eran-BA/Jev_from_GLiNER2 ≠ NSStudent/JevSwiftSDK ≠ jevmlx; default threshold 0.1 still soft). Apply 1740 (`notes.md` §98): **Decision Graph Protocol envelope** (Decision Graph Protocol frame→assess→commit; app retains permissions/effects; Jev-first assessor-neutral; guarded commit / receipt/next frame; assessment batching; hard-gating DGP as safety theater; numerous-com/dgp ≠ TypeSafe official) / **Calibrated meaning-grep live tree** (jegrep calibrated path+range Nouls; no embeddings/index/daemon; ~$0.01–0.03 typical; agent --json; can1357/jegrep ≠ Bentlybro/jevgrep ≠ uehaj/jev-semgrep) / **Archer-arch fidelity + measured calibration gap** (Archer-arch fidelity; kev family OOD 0.76–0.77 vs Jev 0.86; block-causal isolation; pointer/readout CE-trained; /v1/systemone drop-in; replica honesty). Apply 1843 (`notes.md` §99): **Cost-derived YES/NO/UNSURE control flow** (cost-sensitive decision theory × System One probabilities → control flow; thresholds derived from costs not hard-coded; YES / NO / UNSURE from cost_false_yes / cost_false_no / cost_human; auto-batching same-object questions; Kungie/gut ≠ tpellet/hunch ≠ carldaws/hunch) / **Typed-callback control flow** (judgment vs generation; deterministic execution after probabilistic judgment; exactly one app-owned callback; explicit uncertain branch; Illusion47586/judge ≠ lexingtonhibiki/judgekit ≠ Ascurse/typed-judge-kit) / **Variable-N option scoring as the trainable object** (dynamic candidate bags not fixed label sets; zwliJay/jev-forge ≠ NanoJev; not a new class-table species) / **Open NAR replica economics** (NAR local drop-in; open replica economics / latency vs closed Jev; wfzyx/von late-catch HIGH; competing NAR claims / replica honesty) / **Typed vs chat judges on guardrailing** (typed judgments vs chat judges on guardrailing; ishaannk/llm-vs-jev cross-note only; deeper integrity fold is rh-guard; nothing wins outright; can be argued out of guarding). Apply 1943 (`notes.md` §100): **Jev IS the if-statement** (judgments/probabilities drive branches; text model only writes prose; interpreter owns variables/loops/budgets/replay; otherwise maybe / confidence gate; chaos samples after the gate; southpolesteve/probably ≠ carldaws/hunch ≠ feelings ≠ Kungie/gut ≠ Illusion47586/judge ≠ tidymodels/probably) / **GEPA live delta** (133★ / forks 10 live; build calibrated classifiers from human feedback; HEAD/README SHA unchanged vs §93) / **Memory retrieve vs lease** (retrieve by relevance not resemblance; one calibrated yes/no per memory in one request; pointer mode 17/18 19/20 *theirs*; embedding resemblance misses the allergy; samdotmak/jev-recall ≠ jev-search ≠ jev-sift ≠ carryforward ≠ chopratejas/invalidate; memory leases ended by new evidence; six Nouls then fixed rules in code; 0 of 157 false invalidations; questions/plans/directives are not evidence; unsure → review queue; host keeps the store) / **Contract-of-artifact lint rename** (name↔body / comment truth / test-claims; mizchi/jev-lint is mizchi/jevlint rename; no shipped rule has severity error; ~1 in 5 findings wrong *theirs*; mizchi/jev-lint ≠ huntedman/JevLint ≠ MichitoSugawara/jev-lint) / **JSON Schema question compiler** (JSON Schema → typed JSON via Jev; noul_threshold 0.5 decoder not a proof; IncompatibleSchemaError lists every bad property) / **Local System One economics** (on-device Laya CoreML ANE; ~5 ms P50 short decisions; 189/189 FP16 checkpoint parity; 10× not achieved; mizorewww/laya-coreml ≠ gliner-native-runtime ≠ jevmlx ≠ NandhaKishorM/laya; softmax over allowed tokens ≠ Noul; question-first cache; Micha0827/snapjudge ≠ githubnext/localjev ≠ jevmlx ≠ cendress/SnapJudge; Jev-first Pi agent loop; slow-LLM fallback; explicit action menu / CandidateSource unimplemented; 62 tests wiring not quality; direwolfiy/JevPi ≠ standardagents/jevpilot ≠ pi-jev-control). Apply 2041 (`notes.md` §101): **Resume-screening bias audit** (resume-screening bias audit methodology; name×resume factorial independent Nouls; callback determined by resume quality; mean-probability name gaps operationally negligible; natemoo-re/bias-bench ≠ BBQ) / **MCDA panel code-owned verdict** (Plan/PRD panel → code-owned pass|review|block; cheerleading out of scope; austindixson/planalyzer ≠ single-goodness Noul) / **EU cost-aware routing** (cost-aware multi-model routing/escalation; decide vs do; successful-task cost; cannacre8ive/switchboard-ai ≠ ha-switchboard ≠ hermes-switchyard) / **Frozen-protocol bake-off** (frozen-protocol zero-shot bench; TypeSafe Jev vs PrismNLI vs Laya; contamination caveat; elcronos/jev-vs-open-decision-models ≠ JevBench ≠ DMB) / **VOI admission** (context-window admission control; VOI gate which tokens are worth the expensive model; fail polarity per lens; on small inputs lenses lose money; cvsgireesh/jevusher ≠ jev-sift ≠ winnow) / **Leveson control plane** (typed decision control plane; receipt ≠ authorization; historical-v0 zero retained cases; MokiMeow/jev-fabric ≠ jev-forge ≠ dgp) / **Scoring economics** (live 15-dim typed rubric re-score per pause; scoring economics exemplar; OpenJev/Codiv ≠ TypeSafe hosted; jose-troche/live-rubric ~$0.000004 desc / ~$0.000006 README) / **Pre-registered calibration science** (adversarial pre-registered Jev eval; 28 predictions before data; 123,805 requests; confidence does not track ignorance; polite injection 65% / crude 0%; willkelly/jev-evaluation ≠ jevals ≠ jev-baselines-eval) / **Class infrastructure SDK** (provider-neutral Elixir/BEAM Noul/Choice/Score SDK; class infrastructure; nshkrdotcom/system_one_sdk ≠ typesafe_sdk ≠ dannote/jev). Apply 2145 (`notes.md` §102): **Question-linting of Jev questions themselves** (question-linting of Jev questions themselves; nine jaggedness rules, no API key, no labelled data; static lint ≠ measured separation; yodablocks/jevq ≠ tenbin ≠ JevLint ≠ commitjev) / **Open-weights Laya as class exemplar (binding)** (Nx/Bumblebee runtime; host chooses backend; ChristianAlexander/laya_ex ≠ system_one_sdk ≠ dannote/jev ≠ NandhaKishorM/laya) / **On-chain/edge Laya deploy** (parity_verified stays false; model output never grants Tx; humandebri/IC-Laya ≠ laya_ex) / **Auditable weekend replica** (Jev outputs never used for training; unpaired 0.577 vs 0.727; agilabs-ai/jev48 ≠ JevBench ≠ Mapika/decider) / **Adversarial dual-judge / framing** (comparative framing is the usable judgment; prior injection crowds out evidence; copyleftdev/ember ≠ ember.js) / **Laya specialist + Hub replica** (training still GPU-pending; PIXELZX0/XERON ≠ convaiinnovations/laya; daliborsb/laya ≠ convaiinnovations/laya ≠ NandhaKishorM/laya) / **Distillation economics** (gold is programmatic; teacher is closed-API clone; do not distill Jev as teacher of record; MagaBitmex/jev-4b-distill-data ≠ missing student checkpoint) / **Non-LLM VIN System One** (planning depth not chat; lewislululu/jevon ≠ douglance/jevon) / **Source-bound evidence** (local quote mismatch needs no API; exit 0 ≠ claim truth; WaynezProg/jev-kit ≠ jonathanavis96/jev-kit (Airlock) ≠ jev-use ≠ jev-mcp) / Apply 2246 (`notes.md` §103): **Independent System One evidence catalog** (independent System One evidence catalog; 19 reviewed records; scores not one leaderboard; no external record currently reproduced; TokenTrim no-Jev matched hybrid 62.4%; reachjalil/system-one-bench ≠ mallahyari/system-one-benchmark) / **Typed eval freeze** (21 tasks · 134 items · 208 questions; scenes from public GitHub contracts, not production logs; SivletLabs/jev-eval ≠ willkelly/jev-evaluation ≠ 4esv/jev-eval ≠ xxkuboxx/jev-eval ≠ onlyoneaman/jev-eval ≠ dayhaysoos/jevals) / **Option-isolated tiny replica** (option isolation (sibling-blind); permutation-equivariant; Hub OWNER not published; nafisazizir/hev ≠ jaredpalmer/kev) / **Frozen-LLM typed decisions** (frozen local LLM logits, no trained decision head; residual-head 9,222-param decreased 73/96→67/96; confidence = 1−normalized entropy, not P(correct); yuki-oshio/mini-jev ≠ r-ms/mini-jev) / **AR next-token anti-pattern** (Jev classifier as autoregressive next-token predictor; ChatJev-style soundness theater; erik-dunteman/ChatJev ≠ dannote/jev ≠ jev-gpt) / **Formal compose with scoring** (calibrated decision head × AlphaProof value head; implementation-layer isomorphism, semantic difference; timeout = censoring; do not launder Noul as proof) / **Parallel rank vs serial selection** (parallel rank-prediction vs serial selection; independent questions can conflict; zzzzzec/jevsort ≠ keltokhy/jsort) / **Open-side ecosystem catalog** (curated open System One ecosystem catalog; rupeshpoojary9/awesome-open-system-one ≠ AnotiaWang/awesome-jev) / **Knowledge-work paper radar** (arXiv paper radar with Jev relevance scoring; ranking ≠ calibration / 0.5 still soft; fail-open failed evals not marked seen) / Apply 2340 (`notes.md` §104): **From-scratch calibrated decision model** (train calibrated ~27M from scratch; typed Q→prob dist / one forward pass / no LLM decode; hyusi2003/MiniSystemOne ≠ Colvin0315/MiniSystemOne; description-only stub / size 5) / **ORDER BY ranking upgrade** (ESCI hard probe fails four of six; jev_bool ECE 0.242 inversion 0.255; do not re-fold §60 six-gates as new; jobbyjev one-request-per-company from batch-size result) / **Find/design/evaluate decision loops** (find/design/evaluate TypeSafe Jev decision loops; karanb192/jev-architect ≠ samtay32/jev-system-architect) / **Distill-Jev UI stub** (Jairik/jev-distiller size 1; distill-Jev UI stub / do not distill Jev as teacher of record) / **Post-launch scored opportunity map** (post-launch scored use-case map / Jev self-scores then human curation; licensedsaucer9-web/jev-opportunities) / **Jev-inize a use case** (Jev-inize a use case into classifier/router; gavinHuang/jevinize → simple-jev not TypeSafe; featherless-ai/simple-jev) / **Saved-decision regression** (compare saved decisions / same label can still change the branch; VihaanAgarwal/jev-diff ≠ Saik0s/diffusiongemma-jev-macos; not tested with a live Jev API key) / **Constrained-logprob API** (constrained logprob + temp/Platt ≠ Noul; OpenJevPro pastes openjev-sglang JevBench as own; zhangcy122/OpenJevPro ≠ IamBusy/OpenJev ≠ ekzhang/openjev-sglang; PolyForm Noncommercial) / **SmolLM RLCD reproduction** (SmolLM-135M / sub-70ms / 0 output tokens; demo P(True) 0.5052 / Choice conf 0.2872 / Score conf 0.0055; README claims MIT / GitHub license null / no LICENSE file; patelvishwa112/jev-system-one-rlcd ≠ arnabgho/rlcd-lite ≠ blackwood-rlcd) / **Source-backed Awesome radar** (source-backed Awesome Jev radar / 306+ commit-pinned; logicrw/awesome-jev-projects ≠ AnotiaWang/awesome-jev ≠ yibie/awesome-jev ≠ cobanov/awesome-jev ≠ rupeshpoojary9/awesome-open-system-one; auto GitHub sync / Issue-only submissions) / **Rival-aware one-pass scorer** (hashed n-gram encoder / rival-aware attention; olanotolu/jevbetter vs jevlike starter; synthetic hard menus top-1 0.916 vs 0.873 / ECE 0.0182 vs 0.0367 / 40 vs 4608 menus/sec; shuffled-context control 0.335) | `references/mental-models.md` Apply 0042 (`notes.md` §105): structured probability readouts; distribution > argmax; Noul 0.5 midpoint; score is expectation not integer; bare HTTP not SDK; Arohtea/jev-readout) / **Ordinary-model Jev-shape** (Jev-style Choice/Score/Noul from ordinary models; optional DSH plugin; schema-valid ≠ calibrated; gulagala001/jevify ≠ Mintzs/jevify) / **Open-weight Laya measurement** (Laya RLCD benchmark; 40.3% below constant-answer; open-weight measurement; mourad-ghafiri/laya-rlcd-benchmark ≠ yibie/laya-jev-lab) / **Cheap fail-open semantic edge** (cheap fail-open semantic edge; second signal not sole; FastLoopError catch; SupremeDreamZ/jev-fastloop ≠ jev-ultrafast) / **Fan-out measurement** (asking more questions in one call; 0.980 at every N; nearly not fully deterministic; TheWebDevel/jev-fanout) / **VLM+Jev RL teacher** (Qwen3-VL perception + Jev decisions train RL; 0 model calls at deployment; VLM alone 1.7 vs +Jev 4.4; harneet2512/reflexrl ≠ khordoo/jev-reflex-autonomy-lab) / **Independent Jev API vs Laya** (independent Jev API vs Laya; cascade 0.60 matches 78% at 1.8×; noul facts not judgements; yibie/laya-jev-lab ≠ dairui1/jev-lab ≠ BrendanH18/jev-lab) / **Locate vs decide** (GLiNER vs GLiFormer vs Laya vs Jev; extractors ≠ decision engines; Laya dict-instructions collapse 58.3%; umstek/zero-shot-ie-bench) / **Throughput arena** (decisions-per-minute & cost; 204 moves vs 73; throughput not intelligence; angelgalvisc/snake-arena-jev-vs-llms ≠ vtrivedy/jev-plays-games) / **Behavioral contracts** (behavioral contracts; pin expectations eval upgrades; raw 0.94 is not a release; sathariels/jevcheck ≠ dayhaysoos/jevals ≠ SivletLabs/jev-eval) / **Evidence-linked upgrade review** (evidence-linked dependency upgrade; Jev never generates filenames; no_direct_evidence ≠ safe to merge; GaneshVG18/upgrade-radar ≠ LYchoon/paper-radar-jev) / **Knowledge-work discography** (discography theme/mood/complexity; five atomic questions one call; lirantal/discoprint) / Apply 0145 (`notes.md` §106): **Architecture probes PRIMARY** (Turn any open LLM into System-One Jev; uspraveen/Jevify ≠ Mintzs/jevify ≠ gulagala001/jevify; Jevify-any-LLM architecture probe; description-only stub / size 0; Train encoder-only calibrated decision models from a task sentence; Exu is a toolkit, not a method; strictly proper scoring rule; Pre-alpha; Ruivalim/exu-base; scratch-trained calibrated decision model; typed Q → probability dists; Colvin0315/MiniSystemOne ≠ hyusi2003/MiniSystemOne; no published weights download URL; 90.5 seconds / 29.2% pipeline evidence; p_i/p_j independent of other candidates; Recipe for calibrated decision models — small model out; init → synth → train → eval → serve; 91.1 % / ECE 0.022 *theirs*; Jev zero-shot 75.1; scienthoon/luce; Put Jev's three headline claims on trial; 0.5B local GPU; 46x speedup / accuracy identical; ECE 0.624 sentiment catastrophe; bigger model worse calibration; RichardoMrMu/jev-mini ≠ yuki-oshio/mini-jev ≠ r-ms/mini-jev; System-1 decision engine for local LLMs; structured choices only; JSON parse of generated text ≠ Noul; TypefAI JEV / Journal Entry Voucher; tapsin/jev-local ≠ us/jev-local ≠ Argos1111/jev_local) / **Measurement densifies** (Jev 1.13 reward-model eval across 8 benchmark tracks; 40,940 examples / 0 API errors; RewardBench v1 92.58%; Precise IF 50.63%; goya4140/jev-reward-model-evaluation; Scaffolding in progress; Jev vs LLM support-ticket routing; static + live decision bench; TypeSafe's own published benchmark; illustrative simulations, not live API calls; JevBench v1 — smart/cheap/fast/reliable; I/C/S/K 25% geometric mean; classifier.dev fast tier 84.8 is Jev behind its own API; do not re-fold §78 v1.2 board as new; Laya (421M) 70.1 now on board; Zero-shot/few-shot LLM routing; hard budget filter before Jev; Jev never asked to perform budget arithmetic; Jev judges the next state, XState enforces transitions; simulation uses synthetic keyword fixtures; Consistency benchmark Space; This Space contains no benchmark result yet; 12-case plumbing fixture) / **Catalog gravity** (catalog gravity; v-modal/awesome-jev-tools; ★339 live REST; curation is not endorsement; crawler-maintained directory; Daily GitHub + npm sweep, human-merged; RadRebelSam/awesome-jev ≠ AnotiaWang ≠ yibie ≠ cobanov ≠ logicrw ≠ v-modal) / **HF class ports** (HF peft SPLADE/BGE reranker; rdxtremity/jev-reranking ≠ carlaiau/jev-reranking; query-side encoders, not a Jev replica; ONNX System One Qwen3.5-4B scorer; source:pngwn/system-one-qwen3.5-4b-scorer; CC-BY-NC-4.0; temperature 1.75; transformers.js AutoModel cannot load this graph) / do not reopen or amend PR #23 / Apply 0243: Benchmark-driven Jev router and judge; cheap alone is not success; Jev does not write, sum prices, or claim accuracy %; Sol 94.2 / Luna 83.9 / Jev path 89.7; 19.2% Sol / 62.3% cost save / 4.5pp miss of 2pp non-inferiority; p50 latency worse than Sol due to routing overhead; erendikmenn/jev-llm-router-benchmark ≠ jev-rag-benchmark ≠ ryantsai/jev-llm-router; Express + node:sqlite; mock and Jev decision engines; previous_ticket_count >= 3 is code; MIN_CONFIDENCE 0.6 still soft; substring false positives; aesaganda/jev-ticket-router ≠ SarathChandraBellam/jev-vs-llm-ticket-router; Universal Figure & Diagram Router; confidence ≥ 0.85 hard-gate is theater; generative AI banned from scientific plots; six visual branches; human-labeled (state, question, label); 166,054 rows / 22 configs; soft_label for human uncertainty; Praveenrajus/jev-bench ≠ fstandhartinger/jevbench; ternary bonsai System One GGUF; Hub does not ship weights; 100/100 easy T/F is not Harbor; label_mass ≠ correctness; stock llama.cpp Q2_0 silently gibberish; NicolaiMTLassen/open-bonzi-jev ≠ NicolaiLassen; transformers.js DeBERTa ONNX; temperature 1.05; AutoModel from_pretrained works; onnx-community/open-jev-deberta-v3-large-ONNX ≠ system-one-qwen3.5-4b-scorer-ONNX; 107★ densify; GH 151M vs README 149.6M; PR #1 now closed unmerged; do not re-fold §71 claim-audit as a beat; typed decisions, RLCD, confidence-gated routing; structured ≠ correct; mock not live API; 26 tests; wjdjdakf17/jev-study ≠ baekenough/jev-study / Apply 0345 (`notes.md` §108): **Open reproduction densifies / measurement densifies PRIMARY** (bonzi-27b-v2 / ternary-8b / 27b-v1 GGUF family densify; WANLI-256 74.6% / 65.2% / 71.1% *theirs*; Bonsai 1 27B Q1_0 runs on stock llama.cpp; ternary still needs PrismML fork; hf:heman10x/openJev-verdict-2.0 twin tokenizer-only; OpenJev Vision image classification + uncertainty; CLEVR-4 held-out joint 0%; hfdataset:IamBusy/OpenJev-Vision-Research-v0.1 12,832; 294,912 derived targets not independent samples; Laya multilingual ONNX WebGPU typed-decisions port; 63/63 selected answers / 5.1e-4 CPU / 1.2e-2 WebGPU; UpHash-Network/mini-jev is yuki-oshio transfer; jev-injection-bench 11,900 labelled prompts; Jev best ranking / Haiku better ECE 0.021 vs 0.058; 0.5–0.9 band is where Jev's numbers do not mean what they say; Prompt wording moves panic 28%; manojlds/jev-dspy-bench ≠ dspachos/jev-dspy ≠ jmanhype/jev-dspy-lab; Jev agreement is similarity, never ground truth; no aggregate quality grade or merge gate; AbstentionBench-on-Jev rank 1 of 20 vs 2025 field; question-asymmetry; forward-looking 0.465 never extreme; openkev calibration layer not a runtime; ECE vs coverage independent; select_threshold returns inf; escalation catches uncertainty not ignorance; misakaikato/openkev ≠ jaredpalmer/kev; pdf-race Docling→Jev vs Gemini; parser owns the wall clock; 12/12 tie is a tie; titles selected not generated; flopcheck 16 calibrated tweet judgments; mechanical tells in code; ZeroX-01/jev-atlas ≠ Zaious/jev-capability-atlas ≠ gorock007/jev-atlas; Laya calibration lab Gradio MCP; T never changes argmax; confidence ≠ top-label p; easy probe set refused; 40–48 rows too small to ship T; do not reopen or amend PR #23 or #24 or #25. Apply 0439 (`notes.md` §109): Gemma-4 26B-A4B jevify classification+calibration; LoRA adapter twin not independent eval; Gemma-4 E4B jevify; E4B LoRA stub card; kushalpatil/jevify-gemma4 ≠ Mintzs/jevify ≠ gulagala001/jevify ≠ uspraveen/Jevify; GH kushalpatil07/jevify 404; PAWS 0.580/ece 0.288 is the weak cell; smaller E4B slightly better OOD ECE than 26B-A4B; Hub jevify merged LoRA ships weights; bonzi Bonsai-8B v1 GGUF densify; Bonsai-1.7B v1; Bonsai-4B v1; WANLI-256 64.5% / 60.2% / 52.0% *theirs*; rank #4 / #5 / #6 of 6; JulesHuisman/jev-eval scaffolding / README SHA c356a584 (was empty e69de29b); JulesHuisman/jev-eval ≠ SivletLabs/jev-eval ≠ willkelly/jev-evaluation ≠ 4esv ≠ xxkuboxx ≠ onlyoneaman ≠ dayhaysoos/jevals; 7 bands 6/10 vs 40 bands 0/10; source receipts + confidence slider re-policy without re-inference; 32/32 synthetic is smoke not production; classify HF datasets across typed semantic dimensions; roadus2 watch misspelling; lock roadius2/ultra_laya; ultra_laya REVIEW defects; default branch claude/laya-jev-review-gg5ppo; XNLI EN 88.3% ECE 0.032 → RU 77.3% ECE 0.096; Δ −11.0 pp [−14.2,−7.8]; ECE +0.063; MASSIVE no detectable difference at n=600; confidence is function of p_max (r=1.000); pointer-not-generator 400 human-authored responses; proposed ≠ authorized; FewRel 160: Jev 85.0% vs lexical 13.125%; gated 100% (95/95) coverage 59.375%; J++ composable semantic computation language; judge-jev 0.5 still soft; 947 repos scored; A 273 / B 302 / C 372; LLM rubric ≠ benches; No benchmark winner is claimed; phishing: naive 62.6% vs regex 91.8%; 5-atomic + LR 95.0% *theirs*; AITuber tension ±15; README npm global; repo is Rust; git-confess code owns counting/blame/ratio; httpx exhibit 11% (13/119) *theirs*; 90d trend +12.40% vs random +12.75% vs BH +41.71%; 5m win rate 25%; Awesomejev 656 entries / 38,160 stars; tracker likes 64 (+4) lastModified UNCHANGED; Laya present; Blackwood ABSENT; Archer still promised_not_landed; do not reopen or amend PR #23/#24/#25/#26. Apply 0541 (`notes.md` §110): Blackwood tracker ABSENT; likes 2 gated manual; r = c - p_a; ECE 0.021; acc 0.807 vs warmup 0.746; calibration beyond ~500 tokens unmeasured; Independent primitive; 11.57s vs 54.10s · 4.67× · 120/128 *theirs*; default path is pretrained Gemma probs not trained RLCD head; GH Meanblock 404; lock leesk212/JEV-CPU; softmax over letter slots ≠ Noul; WANLI 0.741 vs openjev v2 0.77 *theirs*; 3-way NLI ≠ Noul; priority 0.464 = majority floor; banking77 contaminated; raw margins not probabilities; GH jev-haiku-benchmarking 404; do not distill Jev as teacher of record (they distilled Haiku); “0.9 is not one number”; ranking ≠ calibration; banking77 0.8–0.9 stated 0.86 actual 0.73 over-confident *theirs*; ≠ Praveenrajus/jev-bench ≠ fstandhartinger/jevbench; $0.0000153–$0.0000226 vs circulating $0.0004 (~20×); Score is 0..n-1 expectation not 0–1; Noul has no confidence field; TCP floor 198.8 ms; type reliability is not a reason to choose Jev (json_schema 5/5); gateway tax not one number; Function-only 5/8 vs hybrid 8/8; 4/8 without Jev; 8 designed cases not conversion lift; ≠ RadRebelSam/awesome-jev; 200-row pilot Jev 86.5% 173/200 vs Gemini Flash-Lite 86.0% 172/200 vs Pro 87.0% 174/200 *theirs*; not a ranking; NLI Tetris argmax P(entail)−P(contradict); 情緒測謊器; 1q 396ms / 30q 567ms; ±0.03; 33q $0.000045 vs Gemini ~5× slower ~60× cost *theirs*; ≠ realZachi/jevtest; 8-example Jev vs GPT-5.6 Sol ~64× cost 5.4× latency *theirs*; synthetic; no inference; ≠ JevBench v1.2 §78; Judged 3317 / listed 2560; Jev judges, code applies policy; catalog ≠ endorsement; APA “microsecond policy / zero hallucination” overclaim; Client-side quiz; pointer from held docs; scanned-PDF warn; CSP only api.typesafe.ai; Jev judges / agent reasons / user decides; selecting an option is not permission to implement; degraded fallback; pattern exact, judgement must clear floor; no matching pattern → no model call; not a correctness oracle; $0.00022 vs chat $0.00306 *theirs*; Spec vs artifact remainder; treating 0.85 as 85% / minProbability hard-gate as Harbor; VERIFY acquires discriminating evidence, never same-pool confidence-only rescoring; fast/full/max are ceilings not sizes; Solar writes, Jev chooses NEXT ACTION; SemIf 2186★ (+20 vs §109 2166); jevlike 1038★ (+7 vs 1031); TypeAR 14★ flat; AnotiaWang 96★ (+1 vs 95); yibie/awesome-jev 490★; Laya likes 802 (was 783); tracker likes 64 flat, lastModified UNCHANGED; do not reopen or amend PR #23/#24/#25/#26/#27. Apply 0920 jcr (`notes.md` §116): **Retrieve-wide→decide→evidence-set** / **Skills vs capability catalogs** / **VOI of context admission** / **Measurement honesty (wall-time mixed)**. Apply 0743 (`notes.md` §113): Hourly 0743 uniqueness lock: Heman10x-NGU/Verdict-open-jev ≠ Heman10x-NGU/openJev-verdict-2.0; TF-IDF + LogReg ECE 0.0207 vs Jev 0.1440; Verdict-open-jev 48.07% vs Jev 90.80%; abstention combined recall 10.00%; p50 35.58 ms; K=25 (maximum capacity) 72.00%; 0.85 coverage 84.60% selective risk 1.18%; 26.1× faster than standard Qwen JSON generation; Jevify 90.0% / 167 ms CUDA graphs disabled; Finding 1: Brier on stated confidence alone is a trap; grpo_rlcr 0.78 / ECE 0.084; reliability 0.007 but resolution 0.000; 27 900 schema-driven decisions; 13 600 / 13 600 questions; candidate mass min 0.99999624; 22 configs · 166,054 rows · 4 calibration-gold; sha a39eba3f; Student B MAE 0.148 / Pearson 0.836 / 86.0%; pngwn/open-jev-laya-bench README 404; sha 9f69c742 likes 2; HDFS 0.9933 (745/750) / retain 0.0084; BGL ERROR/FATAL protection 1.0000; 2,479 / 2,500 HDFS uncertain; cache hit 0.9648 (2412/2500); $0.153936 estimated; E2 recomputes from saved probabilities; Space sha eda59e0a; MASSIVE English 0.783 / Khmer 0.033 / Hindi 0.133; 40–48 rows too small to ship T; T never changes argmax; siren2345/jev-single-decode-transformers ≠ siren2345/jev-single-decode; Split Transformers experiment from llama.cpp runtime; tanayvasishtha/jev-lab ≠ dairui1/jev-lab ≠ BrendanH18/jev-lab ≠ yibie/laya-jev-lab; Four experiments stress-testing TypeSafe's Jev: calibration, bundle bias, label bias, and ensembling; second pass must be $0.00 from cache; The pages never call Jev; Gemma 4 31B 77.0% / Jev 1.13.0 61.4% / Laya 322M 0.0%; restriction state 95.0% against 84.4%; None of the systems are particularly good at knowing when to stop and ask; They skip the question and call a tool directly; 100% schema pass; six-field joint 48.8% vs 72.8%; ywchiu/jev_benchmark ≠ Running-Dolphins/jev-bench ≠ Praveenrajus/jev-bench; ACT / REVIEW / FALLBACK; A provider failure, timeout, malformed output, or missing answer is **not** a policy outcome; confidence is descriptive provider output, not a substitute for probability; Quality denominators include only valid scored answers; an exact halfway tie chooses the lower level; aiwithenoch/Jev-Skill ≠ simplosophy/jev-skill ≠ laguagu/jev-skills; The local path does not claim to turn a smaller checkpoint into Jev; Low support becomes decision: "review"; MIT-0 SPDX NOASSERTION; current-llm; 结构兼容,不是 Jev 模型能力; altryne/jevify ≠ Mintzs/jevify ≠ gulagala001/jevify ≠ uspraveen/Jevify; Find where Jev belongs. Design the questions. Measure the difference; TypeAR-AI/TypeAR 301 → TypeLLM/TypeLLM; TypeLLM/TypeLLM 16★; SemIf 2241★ (+34 vs §111 2207); jevlike 1051★ (+8 vs 1043); AnotiaWang 98★ (+1 vs 97); yibie/awesome-jev 525★ (+19 vs 506); Laya likes 864 (was 822); tracker likes 67 (+3 vs 64); lastModified UNCHANGED `2026-09-20T04:29:16.000Z`; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#32/#33; notes.md §113 Apply 0646 (`notes.md` §111): Calibration is not alpha; NO CURRENT ALPHA CANDIDATE; ΔR² approximately +0.00084; Brier 0.2131387; ECE 0.0421875; Adding Jev probability to deterministic volatility improved Brier by only 1.4058e-05; default 0.5 keeps zero non pinned; keepResult median 0.14 to 0.17; keepCall median 0.28 to 0.35; usable range is about 0.10 to 0.25; 7.8% to 57.9%; judges results it never sees; task-finish eval not built yet; $0.002 per compaction; slavadubrov/sgr-judge-bench ≠ slavadubrov/jev-judge-bench; Jev 108/120 $0.083 0.34 s; Luna SGR 114/120; paired Jev accuracy-difference intervals include zero; not evidence of equivalence; GLM SGR 26/120 93 format failures; Terra-planned Jev hybrid 55/120; rule-based by default, optionally Jev-backed; empty README; missing key cannot break the experience; prefill plus exactly one decode; softmax over A/B/C ≠ Noul; BBQ 9,053/10,000 (90.53%); ECE 0.0890; Mean confidence 0.9943; overconfident; score and noul not implemented; DGUI 12 rows (was 6); INSTRUCT 119 rows likes 2; encode the state once, decide everything in parallel; 0.740 accuracy against a 0.508 majority; ECE 0.047; fine-tune's advantage ends where its 384-token training data does; jasonkneen/open-jev ≠ pngwn/open-jev; same sha d41dc3cd; Space does not call Jev; recomputes routing from saved probabilities; 200-case Jev 97.0% / 100.0% / 95.0% / MAE 9.22; synthetic repository benchmark; Jev evaluations are advisory; YehuiTang0316/jev-nlgrep ≠ Bentlybro/jevgrep ≠ can1357/jegrep ≠ uehaj/jev-semgrep; default threshold 0.8 still soft; 40-line windows cannot prove whole function; token-native sequential start/end Choice; Gemini/Haiku stubs not configured yet; handful of hand-written examples, not a benchmark; Jev judged exactly what it was given; laguagu/jev-skills ≠ laguagu/jev-evidence-lab ≠ Pleo2/awesome-jev-agent-skills; contract_passed is not a claim of guaranteed factual truth; Wilson lower bound 0.85 floor; fixture mode no savings claim; SemIf 2207★ (+21 vs §110 2186); jevlike 1043★ (+5 vs 1038); TypeAR 15★ (+1 vs 14); AnotiaWang 97★ (+1 vs 96); yibie/awesome-jev 506★ (+16 vs 490); Laya likes 822 (was 802); tracker likes 64 flat, lastModified UNCHANGED; do not reopen or amend PR #23/#24/#25/#26/#27/#28.) | -| Judgment-model class (Jev is exemplar, not monopoly) | Species: decide / locate (GLiNER) / categorize (GLiClass) / rank / perceive; open heads include encoder DeBERTa, LoRA distill, **domain specialist LoRA on independent gold**, openjev-lm, kev. Compaction job is backend-agnostic (Jev Score/Noul vs GLiNER2.5 encoder). Indexer cousin: GLiNER extract + escalate-S2 (10–50× unfilled). Computer-use observe→score-among-candidates→code-acts is backend-agnostic (Jev Ultrafast ↔ GLiNER2 Ultrafast ↔ Cua-S1 specialist ↔ Stagehand experimental Jev stack; **OCR+AX desktop:** typesafe-computer-use hosted Jev, never ships a screenshot for the *decision*; Cua-S1 is not TypeSafe Jev; Stagehand pick is a fast path, not a replacement; **≠** jev-macos-loop OmniParser **≠** camoufox; **ASR voice-browser:** jev-voice-browser hosted Jev, never ships a waveform). GLiFormer encoder serving `/v1/systemone` is a class-backend (jeff), not a Jev replica. Local MLX PCD is O(1) constrained-AR speed, **not** a calibrated Noul (system-one-benchmark Brier 0.3884 vs Jev 0.1096). **jevmlx** is the productized Apple Silicon one-pass schema→JSON+probs library (softmax ≠ Noul; no local leaderboard yet). **openvons** is an independent open-Jev class (LM/vision/voice finite-choice+prob; JevPick 3.2–4.8×; Flutter on-device; `/v1/systemone` wire-compat, not a TypeSafe replica). **OpenJev** (IamBusy) is a local 0.6B LoRA+scalar head on `/v1/decide` (45/60 *theirs*; **not** a TypeSafe drop-in; distinct from hraness/sysone OpenJev runners). **semif-serve** puts SemIf behind `/v1/systemone` (runoff, no option ceiling; 1164 vs 178 ms *theirs*; wire-compat ≠ replica). **grande** Rust/WebGPU kev-shaped branches (JGLUE *theirs* JNLI 0.614 ECE→0.088 / JCQA 0.853; 270M 0.710/0.710; isolation 0.098/0.996). **laya-jolt** Clojure/Jolt byte parity vs Python Laya. **JEV-CPU** SemIf on CPU (leesk212; Meanblock 404). **local-jev** ONNX ModernBERT measured not-equivalent (done 30% / shape 57% *theirs*). **GLiNER2→Choice/Score/Noul spec** (Eran-BA; design only, ≠ jeff GLiFormer). **openJev-verdict-2.0** competing NAR claims as **audit object not endorsement** (77.10%/0.0636/0.0144 *theirs*; PR #1; ≠ IamBusy/OpenJev `/v1/decide`). **chakuho** 1-token logprob local `/v1/systemone` (softmax ≠ Noul; coverage ≠ correctness; GUI 336 *theirs* 27B 95%/92% vs Jev 89%/82%). **jevinf** open replica engine (NanoJev/decider-2b/Laya; 2.57×/2.27× 100% argmax; MPS only). **laya-multilingual** mmBERT-base 322M (MASSIVE 0.366/0.387 vs English laya 0.227/0.733; Khmer 0.000@0.952 conf; ships uncalibrated). **schema-scorer** DeBERTa-v3-large scalar head (Hub; GitHub 404; v2 Choice 0.841 *theirs*; peaked ranking ≠ calibration). **githubnext/localjev** prompted-JSON `/v1/systemone` (MIT **261★**; TypeSafe SDK drop-in; wire-compat ≠ logit-equiv vs razorback16 structured-read; **≠** kunchenguid/local-jev; 1,200-req bake-off *theirs* Qwen3.6 76.7% / Gemma 26B 75.0% / DiffusionGemma 74.2% short; not calibrated). **NandhaKishorM/laya** PyPI+Router packaging of Hub Laya (Apache-2.0; **710★**; not a new species; T4 32.8 ms *theirs*; post-T ECE 0.081 vs Jev 0.246; Banking77 0.425 vs Jev 0.870; 0.766 is fine-tune not zero-shot; 0.85 still soft; **≠** TypeSafe `/v1/systemone`). **external openjev census** (@airesearch12 tweet ≠ jevbench v1.1; GLiNER2+routers class-boundary; incomplete vs watch). **JevBench v1.2 scored board** (geo-mean I/C/S/K 25% each; Jev 75.3 / SemIf 74.6 *theirs*; cal ON rank; instruction models in the table; ≠ v1.1 87.6; ≠ tweet census; Laya absent gap; Qwen3.8 27B ≠ Archer). **Hourly 0842:** already-folded class as a recipe (wire≠logit · product+FALLBACK · packaging honesty · pointer-not-generator · leaderboard VOI); skip thin noise. **Open LoRA replica, different jeff:** GestaltLabs/Jeff-1 LoRA Qwen3-4B ≠ logan-markewich/jeff GLiFormer; acc/ECE tradeoff n=9730 *theirs*; set reused. **Hourly 1241:** typesafeai-sdk-community not a new species; 2389-research/judgement license null; confidence ≠ winner p; jevbrain AUTO_ACT is not a Noul. **Hourly 1347:** FrancoisChastel/jev-code ≠ npm jev-code; ctmx/openrouter-jev-mcp Decision-as-Plugin; claudecode-jev-marketplace fail-open not hot path; pedroknigge/mcp_jev packs not ask_jev; 0thernet/system-one-skills deterministic verify; hari007sh/jev ≠ dannote/jev; nanoprune 2.8MB ECE 2.58%. **Hourly 1441:** Foq ~25ms/2.2GB local; rev prefill-only + HF jev-0.5b; robfrase/jev planning memo; meldltd/meldecision laya-go ONNX; laya-doom never pixels; logixism/laya-api empty README; akpsahan/laya ≠ Archer; Nibir1/typesafe-go ≠ official. **SIGNAL §94:** unofficial ≠ TypeSafe JA ModernBERT (argos1111/modernbert-ja-310m-jev; format_version modernbert-jev/1); Argos1111/jev_local ≠ us/jev-local ≠ kunchenguid/local-jev; LFM default ≠ ModernBERT backend; Nemotron ≠ TypeSafe Jev; djev-dev complements djev-spark (images as Choice options); Laya essay ≠ new species; hosted bootstrap ≠ silent TypeSafe. **Hourly 1541:** llama-jev llama.cpp replica; softmax ≠ Noul; **≠** TypeSafe **≠** pcdServer. **SIGNAL §97:** GLiNER2 native Apple path; unofficial Swift/Core ML GLiNER 2.5-small; entity spans + confidence; not Choice/Score/Noul; not TypeSafe; label descriptions as schema; on-device ANE economics; honesty locks; shershah1024/gliner-native-runtime ≠ Fastino; ≠ gliner25-compaction ≠ gliner2-ultrafast ≠ Eran-BA/Jev_from_GLiNER2 ≠ NSStudent/JevSwiftSDK ≠ jevmlx; default threshold 0.1 still soft. **Hourly 1740:** Decision Graph Protocol frame→assess→commit; app retains permissions/effects; Jev-first assessor-neutral; guarded commit / receipt/next frame; assessment batching; hard-gating DGP as safety theater; numerous-com/dgp ≠ TypeSafe official; jegrep calibrated path+range Nouls; no embeddings/index/daemon; ~$0.01–0.03 typical; agent --json; can1357/jegrep ≠ Bentlybro/jevgrep ≠ uehaj/jev-semgrep; Archer-arch fidelity; kev family OOD 0.76–0.77 vs Jev 0.86; block-causal isolation; pointer/readout CE-trained; /v1/systemone drop-in; replica honesty. **Hourly 1843:** variable-N option scoring as the trainable object; dynamic candidate bags not fixed label sets; zwliJay/jev-forge ≠ NanoJev; not a new class-table species; NAR local drop-in; wfzyx/von late-catch HIGH; open replica economics / latency vs closed Jev; competing NAR claims / replica honesty; gut/judge are control-flow overlays not species. **Hourly 1943:** softmax over allowed tokens ≠ Noul; on-device Laya CoreML ANE; JSON Schema → typed JSON via Jev; noul_threshold 0.5 decoder not a proof; mizorewww/laya-coreml ≠ gliner-native-runtime ≠ jevmlx ≠ NandhaKishorM/laya; Micha0827/snapjudge ≠ githubnext/localjev ≠ jevmlx ≠ cendress/SnapJudge; Jev IS the if-statement is a language primitive not a new species **Hourly 2041:** resume-screening bias audit methodology; Plan/PRD panel → code-owned pass|review|block; cost-aware multi-model routing/escalation; frozen-protocol zero-shot bench; context-window admission control; typed decision control plane; receipt ≠ authorization; live 15-dim typed rubric re-score per pause; adversarial pre-registered Jev eval; confidence does not track ignorance; provider-neutral Elixir/BEAM Noul/Choice/Score SDK **Hourly 2145:** question-linting of Jev questions themselves; nine jaggedness rules, no API key, no labelled data; static lint ≠ measured separation; yodablocks/jevq ≠ tenbin ≠ JevLint ≠ commitjev; open-weights Laya as class exemplar (binding); Nx/Bumblebee runtime; host chooses backend; ChristianAlexander/laya_ex ≠ system_one_sdk ≠ dannote/jev ≠ NandhaKishorM/laya; on-chain/edge Laya deploy; parity_verified stays false; model output never grants Tx; humandebri/IC-Laya ≠ laya_ex; auditable weekend replica; Jev outputs never used for training; unpaired 0.577 vs 0.727; agilabs-ai/jev48 ≠ JevBench ≠ Mapika/decider; adversarial dual-judge / framing attack surface; comparative framing is the usable judgment; prior injection crowds out evidence; copyleftdev/ember ≠ ember.js; Laya specialist fine-tune pipeline; training still GPU-pending; PIXELZX0/XERON ≠ convaiinnovations/laya; Hub Laya replica drop; daliborsb/laya ≠ convaiinnovations/laya ≠ NandhaKishorM/laya; System One student distillation corpus; gold is programmatic; teacher is closed-API clone; MagaBitmex/jev-4b-distill-data ≠ missing student checkpoint; non-LLM VIN System One; planning depth not chat; lewislululu/jevon ≠ douglance/jevon; source-bound evidence checks; local quote mismatch needs no API; exit 0 ≠ claim truth; WaynezProg/jev-kit ≠ jonathanavis96/jev-kit (Airlock) ≠ jev-use ≠ jev-mcp **Hourly 2246:** independent System One evidence catalog; 19 reviewed records; scores not one leaderboard; no external record currently reproduced; TokenTrim no-Jev matched hybrid 62.4%; reachjalil/system-one-bench ≠ mallahyari/system-one-benchmark; 21 tasks · 134 items · 208 questions; scenes from public GitHub contracts, not production logs; SivletLabs/jev-eval ≠ willkelly/jev-evaluation ≠ 4esv/jev-eval ≠ xxkuboxx/jev-eval ≠ onlyoneaman/jev-eval ≠ dayhaysoos/jevals; option isolation (sibling-blind); permutation-equivariant; Hub OWNER not published; nafisazizir/hev ≠ jaredpalmer/kev; frozen local LLM logits, no trained decision head; residual-head 9,222-param decreased 73/96→67/96; confidence = 1−normalized entropy, not P(correct); yuki-oshio/mini-jev ≠ r-ms/mini-jev; Jev classifier as autoregressive next-token predictor; ChatJev-style soundness theater; erik-dunteman/ChatJev ≠ dannote/jev ≠ jev-gpt; calibrated decision head × AlphaProof value head; implementation-layer isomorphism, semantic difference; timeout = censoring; do not launder Noul as proof; parallel rank-prediction vs serial selection; independent questions can conflict; zzzzzec/jevsort ≠ keltokhy/jsort; curated open System One ecosystem catalog; rupeshpoojary9/awesome-open-system-one ≠ AnotiaWang/awesome-jev; arXiv paper radar with Jev relevance scoring; ranking ≠ calibration / 0.5 still soft; fail-open failed evals not marked seen **Hourly 2340:** train calibrated ~27M from scratch; typed Q→prob dist / one forward pass / no LLM decode; hyusi2003/MiniSystemOne ≠ Colvin0315/MiniSystemOne; description-only stub / size 5; ESCI hard probe fails four of six; jev_bool ECE 0.242 inversion 0.255; do not re-fold §60 six-gates as new; jobbyjev one-request-per-company from batch-size result; find/design/evaluate TypeSafe Jev decision loops; karanb192/jev-architect ≠ samtay32/jev-system-architect; Jairik/jev-distiller size 1; distill-Jev UI stub / do not distill Jev as teacher of record; post-launch scored use-case map / Jev self-scores then human curation; licensedsaucer9-web/jev-opportunities; Jev-inize a use case into classifier/router; gavinHuang/jevinize → simple-jev not TypeSafe; featherless-ai/simple-jev; compare saved decisions / same label can still change the branch; VihaanAgarwal/jev-diff ≠ Saik0s/diffusiongemma-jev-macos; not tested with a live Jev API key; constrained logprob + temp/Platt ≠ Noul; OpenJevPro pastes openjev-sglang JevBench as own; zhangcy122/OpenJevPro ≠ IamBusy/OpenJev ≠ ekzhang/openjev-sglang; PolyForm Noncommercial; SmolLM-135M / sub-70ms / 0 output tokens; demo P(True) 0.5052 / Choice conf 0.2872 / Score conf 0.0055; README claims MIT / GitHub license null / no LICENSE file; patelvishwa112/jev-system-one-rlcd ≠ arnabgho/rlcd-lite ≠ blackwood-rlcd; source-backed Awesome Jev radar / 306+ commit-pinned; logicrw/awesome-jev-projects ≠ AnotiaWang/awesome-jev ≠ yibie/awesome-jev ≠ cobanov/awesome-jev ≠ rupeshpoojary9/awesome-open-system-one; auto GitHub sync / Issue-only submissions; hashed n-gram encoder / rival-aware attention; olanotolu/jevbetter vs jevlike starter; synthetic hard menus top-1 0.916 vs 0.873 / ECE 0.0182 vs 0.0367 / 40 vs 4608 menus/sec; shuffled-context control 0.335 | `references/judgment-class.md` Apply 0042 (`notes.md` §105): structured probability readouts; distribution > argmax; Noul 0.5 midpoint; score is expectation not integer; bare HTTP not SDK; Arohtea/jev-readout; Jev-style Choice/Score/Noul from ordinary models; optional DSH plugin; schema-valid ≠ calibrated; gulagala001/jevify ≠ Mintzs/jevify; Laya RLCD benchmark; 40.3% below constant-answer; open-weight measurement; mourad-ghafiri/laya-rlcd-benchmark ≠ yibie/laya-jev-lab; cheap fail-open semantic edge; second signal not sole; FastLoopError catch; SupremeDreamZ/jev-fastloop ≠ jev-ultrafast; asking more questions in one call; 0.980 at every N; nearly not fully deterministic; TheWebDevel/jev-fanout; Qwen3-VL perception + Jev decisions train RL; 0 model calls at deployment; VLM alone 1.7 vs +Jev 4.4; harneet2512/reflexrl ≠ khordoo/jev-reflex-autonomy-lab; independent Jev API vs Laya; cascade 0.60 matches 78% at 1.8×; noul facts not judgements; yibie/laya-jev-lab ≠ dairui1/jev-lab ≠ BrendanH18/jev-lab; GLiNER vs GLiFormer vs Laya vs Jev; extractors ≠ decision engines; Laya dict-instructions collapse 58.3%; umstek/zero-shot-ie-bench; decisions-per-minute & cost; 204 moves vs 73; throughput not intelligence; angelgalvisc/snake-arena-jev-vs-llms ≠ vtrivedy/jev-plays-games; behavioral contracts; pin expectations eval upgrades; raw 0.94 is not a release; sathariels/jevcheck ≠ dayhaysoos/jevals ≠ SivletLabs/jev-eval; evidence-linked dependency upgrade; Jev never generates filenames; no_direct_evidence ≠ safe to merge; GaneshVG18/upgrade-radar ≠ LYchoon/paper-radar-jev; discography theme/mood/complexity; five atomic questions one call; lirantal/discoprint / Apply 0145 (`notes.md` §106): Turn any open LLM into System-One Jev; description-only stub / size 0; Exu is a toolkit, not a method; typed Q → probability dists; JSON parse of generated text ≠ Noul; classifier.dev fast tier 84.8 is Jev behind its own API; hard budget filter before Jev; Jev judges the next state, XState enforces transitions; catalog gravity; ★339 live REST; query-side encoders, not a Jev replica; transformers.js AutoModel cannot load this graph; This Space contains no benchmark result yet; do not reopen or amend PR #23. / Apply 0243: Benchmark-driven Jev router and judge; cheap alone is not success; Jev does not write, sum prices, or claim accuracy %; Sol 94.2 / Luna 83.9 / Jev path 89.7; 19.2% Sol / 62.3% cost save / 4.5pp miss of 2pp non-inferiority; p50 latency worse than Sol due to routing overhead; erendikmenn/jev-llm-router-benchmark ≠ jev-rag-benchmark ≠ ryantsai/jev-llm-router; Express + node:sqlite; mock and Jev decision engines; previous_ticket_count >= 3 is code; MIN_CONFIDENCE 0.6 still soft; substring false positives; aesaganda/jev-ticket-router ≠ SarathChandraBellam/jev-vs-llm-ticket-router; Universal Figure & Diagram Router; confidence ≥ 0.85 hard-gate is theater; generative AI banned from scientific plots; six visual branches; human-labeled (state, question, label); 166,054 rows / 22 configs; soft_label for human uncertainty; Praveenrajus/jev-bench ≠ fstandhartinger/jevbench; ternary bonsai System One GGUF; Hub does not ship weights; 100/100 easy T/F is not Harbor; label_mass ≠ correctness; stock llama.cpp Q2_0 silently gibberish; NicolaiMTLassen/open-bonzi-jev ≠ NicolaiLassen; transformers.js DeBERTa ONNX; temperature 1.05; AutoModel from_pretrained works; onnx-community/open-jev-deberta-v3-large-ONNX ≠ system-one-qwen3.5-4b-scorer-ONNX; 107★ densify; GH 151M vs README 149.6M; PR #1 now closed unmerged; do not re-fold §71 claim-audit as a beat; typed decisions, RLCD, confidence-gated routing; structured ≠ correct; mock not live API; 26 tests; wjdjdakf17/jev-study ≠ baekenough/jev-study / Apply 0345 (`notes.md` §108): **Open reproduction densifies / measurement densifies PRIMARY** (bonzi-27b-v2 / ternary-8b / 27b-v1 GGUF family densify; WANLI-256 74.6% / 65.2% / 71.1% *theirs*; Bonsai 1 27B Q1_0 runs on stock llama.cpp; ternary still needs PrismML fork; hf:heman10x/openJev-verdict-2.0 twin tokenizer-only; OpenJev Vision image classification + uncertainty; CLEVR-4 held-out joint 0%; hfdataset:IamBusy/OpenJev-Vision-Research-v0.1 12,832; 294,912 derived targets not independent samples; Laya multilingual ONNX WebGPU typed-decisions port; 63/63 selected answers / 5.1e-4 CPU / 1.2e-2 WebGPU; UpHash-Network/mini-jev is yuki-oshio transfer; jev-injection-bench 11,900 labelled prompts; Jev best ranking / Haiku better ECE 0.021 vs 0.058; 0.5–0.9 band is where Jev's numbers do not mean what they say; Prompt wording moves panic 28%; manojlds/jev-dspy-bench ≠ dspachos/jev-dspy ≠ jmanhype/jev-dspy-lab; Jev agreement is similarity, never ground truth; no aggregate quality grade or merge gate; AbstentionBench-on-Jev rank 1 of 20 vs 2025 field; question-asymmetry; forward-looking 0.465 never extreme; openkev calibration layer not a runtime; ECE vs coverage independent; select_threshold returns inf; escalation catches uncertainty not ignorance; misakaikato/openkev ≠ jaredpalmer/kev; pdf-race Docling→Jev vs Gemini; parser owns the wall clock; 12/12 tie is a tie; titles selected not generated; flopcheck 16 calibrated tweet judgments; mechanical tells in code; ZeroX-01/jev-atlas ≠ Zaious/jev-capability-atlas ≠ gorock007/jev-atlas; Laya calibration lab Gradio MCP; T never changes argmax; confidence ≠ top-label p; easy probe set refused; 40–48 rows too small to ship T; do not reopen or amend PR #23 or #24 or #25. Apply 0439 (`notes.md` §109): Gemma-4 26B-A4B jevify classification+calibration; LoRA adapter twin not independent eval; Gemma-4 E4B jevify; E4B LoRA stub card; kushalpatil/jevify-gemma4 ≠ Mintzs/jevify ≠ gulagala001/jevify ≠ uspraveen/Jevify; GH kushalpatil07/jevify 404; PAWS 0.580/ece 0.288 is the weak cell; smaller E4B slightly better OOD ECE than 26B-A4B; Hub jevify merged LoRA ships weights; bonzi Bonsai-8B v1 GGUF densify; Bonsai-1.7B v1; Bonsai-4B v1; WANLI-256 64.5% / 60.2% / 52.0% *theirs*; rank #4 / #5 / #6 of 6; JulesHuisman/jev-eval scaffolding / README SHA c356a584 (was empty e69de29b); JulesHuisman/jev-eval ≠ SivletLabs/jev-eval ≠ willkelly/jev-evaluation ≠ 4esv ≠ xxkuboxx ≠ onlyoneaman ≠ dayhaysoos/jevals; 7 bands 6/10 vs 40 bands 0/10; source receipts + confidence slider re-policy without re-inference; 32/32 synthetic is smoke not production; classify HF datasets across typed semantic dimensions; roadus2 watch misspelling; lock roadius2/ultra_laya; ultra_laya REVIEW defects; default branch claude/laya-jev-review-gg5ppo; XNLI EN 88.3% ECE 0.032 → RU 77.3% ECE 0.096; Δ −11.0 pp [−14.2,−7.8]; ECE +0.063; MASSIVE no detectable difference at n=600; confidence is function of p_max (r=1.000); pointer-not-generator 400 human-authored responses; proposed ≠ authorized; FewRel 160: Jev 85.0% vs lexical 13.125%; gated 100% (95/95) coverage 59.375%; J++ composable semantic computation language; judge-jev 0.5 still soft; 947 repos scored; A 273 / B 302 / C 372; LLM rubric ≠ benches; No benchmark winner is claimed; phishing: naive 62.6% vs regex 91.8%; 5-atomic + LR 95.0% *theirs*; AITuber tension ±15; README npm global; repo is Rust; git-confess code owns counting/blame/ratio; httpx exhibit 11% (13/119) *theirs*; 90d trend +12.40% vs random +12.75% vs BH +41.71%; 5m win rate 25%; Awesomejev 656 entries / 38,160 stars; tracker likes 64 (+4) lastModified UNCHANGED; Laya present; Blackwood ABSENT; Archer still promised_not_landed; do not reopen or amend PR #23/#24/#25/#26. Apply 0541 (`notes.md` §110): Blackwood tracker ABSENT; likes 2 gated manual; r = c - p_a; ECE 0.021; acc 0.807 vs warmup 0.746; calibration beyond ~500 tokens unmeasured; Independent primitive; 11.57s vs 54.10s · 4.67× · 120/128 *theirs*; default path is pretrained Gemma probs not trained RLCD head; GH Meanblock 404; lock leesk212/JEV-CPU; softmax over letter slots ≠ Noul; WANLI 0.741 vs openjev v2 0.77 *theirs*; 3-way NLI ≠ Noul; priority 0.464 = majority floor; banking77 contaminated; raw margins not probabilities; GH jev-haiku-benchmarking 404; do not distill Jev as teacher of record (they distilled Haiku); “0.9 is not one number”; ranking ≠ calibration; banking77 0.8–0.9 stated 0.86 actual 0.73 over-confident *theirs*; ≠ Praveenrajus/jev-bench ≠ fstandhartinger/jevbench; $0.0000153–$0.0000226 vs circulating $0.0004 (~20×); Score is 0..n-1 expectation not 0–1; Noul has no confidence field; TCP floor 198.8 ms; type reliability is not a reason to choose Jev (json_schema 5/5); gateway tax not one number; Function-only 5/8 vs hybrid 8/8; 4/8 without Jev; 8 designed cases not conversion lift; ≠ RadRebelSam/awesome-jev; 200-row pilot Jev 86.5% 173/200 vs Gemini Flash-Lite 86.0% 172/200 vs Pro 87.0% 174/200 *theirs*; not a ranking; NLI Tetris argmax P(entail)−P(contradict); 情緒測謊器; 1q 396ms / 30q 567ms; ±0.03; 33q $0.000045 vs Gemini ~5× slower ~60× cost *theirs*; ≠ realZachi/jevtest; 8-example Jev vs GPT-5.6 Sol ~64× cost 5.4× latency *theirs*; synthetic; no inference; ≠ JevBench v1.2 §78; Judged 3317 / listed 2560; Jev judges, code applies policy; catalog ≠ endorsement; APA “microsecond policy / zero hallucination” overclaim; Client-side quiz; pointer from held docs; scanned-PDF warn; CSP only api.typesafe.ai; Jev judges / agent reasons / user decides; selecting an option is not permission to implement; degraded fallback; pattern exact, judgement must clear floor; no matching pattern → no model call; not a correctness oracle; $0.00022 vs chat $0.00306 *theirs*; Spec vs artifact remainder; treating 0.85 as 85% / minProbability hard-gate as Harbor; VERIFY acquires discriminating evidence, never same-pool confidence-only rescoring; fast/full/max are ceilings not sizes; Solar writes, Jev chooses NEXT ACTION; SemIf 2186★ (+20 vs §109 2166); jevlike 1038★ (+7 vs 1031); TypeAR 14★ flat; AnotiaWang 96★ (+1 vs 95); yibie/awesome-jev 490★; Laya likes 802 (was 783); tracker likes 64 flat, lastModified UNCHANGED; do not reopen or amend PR #23/#24/#25/#26/#27. Apply 0743 (`notes.md` §113): Hourly 0743 uniqueness lock: Heman10x-NGU/Verdict-open-jev ≠ Heman10x-NGU/openJev-verdict-2.0; TF-IDF + LogReg ECE 0.0207 vs Jev 0.1440; Verdict-open-jev 48.07% vs Jev 90.80%; abstention combined recall 10.00%; p50 35.58 ms; K=25 (maximum capacity) 72.00%; 0.85 coverage 84.60% selective risk 1.18%; 26.1× faster than standard Qwen JSON generation; Jevify 90.0% / 167 ms CUDA graphs disabled; Finding 1: Brier on stated confidence alone is a trap; grpo_rlcr 0.78 / ECE 0.084; reliability 0.007 but resolution 0.000; 27 900 schema-driven decisions; 13 600 / 13 600 questions; candidate mass min 0.99999624; 22 configs · 166,054 rows · 4 calibration-gold; sha a39eba3f; Student B MAE 0.148 / Pearson 0.836 / 86.0%; pngwn/open-jev-laya-bench README 404; sha 9f69c742 likes 2; HDFS 0.9933 (745/750) / retain 0.0084; BGL ERROR/FATAL protection 1.0000; 2,479 / 2,500 HDFS uncertain; cache hit 0.9648 (2412/2500); $0.153936 estimated; E2 recomputes from saved probabilities; Space sha eda59e0a; MASSIVE English 0.783 / Khmer 0.033 / Hindi 0.133; 40–48 rows too small to ship T; T never changes argmax; siren2345/jev-single-decode-transformers ≠ siren2345/jev-single-decode; Split Transformers experiment from llama.cpp runtime; tanayvasishtha/jev-lab ≠ dairui1/jev-lab ≠ BrendanH18/jev-lab ≠ yibie/laya-jev-lab; Four experiments stress-testing TypeSafe's Jev: calibration, bundle bias, label bias, and ensembling; second pass must be $0.00 from cache; The pages never call Jev; Gemma 4 31B 77.0% / Jev 1.13.0 61.4% / Laya 322M 0.0%; restriction state 95.0% against 84.4%; None of the systems are particularly good at knowing when to stop and ask; They skip the question and call a tool directly; 100% schema pass; six-field joint 48.8% vs 72.8%; ywchiu/jev_benchmark ≠ Running-Dolphins/jev-bench ≠ Praveenrajus/jev-bench; ACT / REVIEW / FALLBACK; A provider failure, timeout, malformed output, or missing answer is **not** a policy outcome; confidence is descriptive provider output, not a substitute for probability; Quality denominators include only valid scored answers; an exact halfway tie chooses the lower level; aiwithenoch/Jev-Skill ≠ simplosophy/jev-skill ≠ laguagu/jev-skills; The local path does not claim to turn a smaller checkpoint into Jev; Low support becomes decision: "review"; MIT-0 SPDX NOASSERTION; current-llm; 结构兼容,不是 Jev 模型能力; altryne/jevify ≠ Mintzs/jevify ≠ gulagala001/jevify ≠ uspraveen/Jevify; Find where Jev belongs. Design the questions. Measure the difference; TypeAR-AI/TypeAR 301 → TypeLLM/TypeLLM; TypeLLM/TypeLLM 16★; SemIf 2241★ (+34 vs §111 2207); jevlike 1051★ (+8 vs 1043); AnotiaWang 98★ (+1 vs 97); yibie/awesome-jev 525★ (+19 vs 506); Laya likes 864 (was 822); tracker likes 67 (+3 vs 64); lastModified UNCHANGED `2026-09-20T04:29:16.000Z`; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#32/#33; notes.md §113 Apply 0646 (`notes.md` §111): Calibration is not alpha; NO CURRENT ALPHA CANDIDATE; ΔR² approximately +0.00084; Brier 0.2131387; ECE 0.0421875; Adding Jev probability to deterministic volatility improved Brier by only 1.4058e-05; default 0.5 keeps zero non pinned; keepResult median 0.14 to 0.17; keepCall median 0.28 to 0.35; usable range is about 0.10 to 0.25; 7.8% to 57.9%; judges results it never sees; task-finish eval not built yet; $0.002 per compaction; slavadubrov/sgr-judge-bench ≠ slavadubrov/jev-judge-bench; Jev 108/120 $0.083 0.34 s; Luna SGR 114/120; paired Jev accuracy-difference intervals include zero; not evidence of equivalence; GLM SGR 26/120 93 format failures; Terra-planned Jev hybrid 55/120; rule-based by default, optionally Jev-backed; empty README; missing key cannot break the experience; prefill plus exactly one decode; softmax over A/B/C ≠ Noul; BBQ 9,053/10,000 (90.53%); ECE 0.0890; Mean confidence 0.9943; overconfident; score and noul not implemented; DGUI 12 rows (was 6); INSTRUCT 119 rows likes 2; encode the state once, decide everything in parallel; 0.740 accuracy against a 0.508 majority; ECE 0.047; fine-tune's advantage ends where its 384-token training data does; jasonkneen/open-jev ≠ pngwn/open-jev; same sha d41dc3cd; Space does not call Jev; recomputes routing from saved probabilities; 200-case Jev 97.0% / 100.0% / 95.0% / MAE 9.22; synthetic repository benchmark; Jev evaluations are advisory; YehuiTang0316/jev-nlgrep ≠ Bentlybro/jevgrep ≠ can1357/jegrep ≠ uehaj/jev-semgrep; default threshold 0.8 still soft; 40-line windows cannot prove whole function; token-native sequential start/end Choice; Gemini/Haiku stubs not configured yet; handful of hand-written examples, not a benchmark; Jev judged exactly what it was given; laguagu/jev-skills ≠ laguagu/jev-evidence-lab ≠ Pleo2/awesome-jev-agent-skills; contract_passed is not a claim of guaranteed factual truth; Wilson lower bound 0.85 floor; fixture mode no savings claim; SemIf 2207★ (+21 vs §110 2186); jevlike 1043★ (+5 vs 1038); TypeAR 15★ (+1 vs 14); AnotiaWang 97★ (+1 vs 96); yibie/awesome-jev 506★ (+16 vs 490); Laya likes 822 (was 802); tracker likes 64 flat, lastModified UNCHANGED; do not reopen or amend PR #23/#24/#25/#26/#27/#28. Hourly 1441 densify: dual /v1/systemone + /v1/chat/completions (dual serving is not generate; chat 501 on MLX); Hosted Codiv ≠ TypeSafe; hr98w/jev-visual candidate probabilities are relative not correctness; 37.30s → 2.40s at 64 decisions *theirs*; Breakout 9 bricks 6 returns 2 lives *theirs*; jkudish/jev-mcp recommendation is advisory / the server never blocks on its own; TypeSafe CLERC 5% to 18% *theirs*; Zefan-Cai/Open-Jev LoRA ≠ RLCD replica; 2B 94.71% 9B 97.54% hard test *theirs*; cristianoliveira/jeq pass-min 0.8 still soft; JEQ does not own actions. Apply 1746 (`notes.md` §128): truncated thinking then constrained decode; 0.8B thinking On 0/18 *theirs*; forced closure 20/20 type-valid *theirs*; Kev-0.8B completes family; 4B new-source 0.794/0.832 *theirs*; 9B new-source 0.812/0.837 *theirs*; Qwen3.5 ≠ Archer; Facts go to code. Judgments go to Jev. Only facts can block.; Jev never blocks.) | +| Judgment-model class (Jev is exemplar, not monopoly) | Species: decide / locate (GLiNER) / categorize (GLiClass) / rank / perceive; open heads include encoder DeBERTa, LoRA distill, **domain specialist LoRA on independent gold**, openjev-lm, kev. Compaction job is backend-agnostic (Jev Score/Noul vs GLiNER2.5 encoder). Indexer cousin: GLiNER extract + escalate-S2 (10–50× unfilled). Computer-use observe→score-among-candidates→code-acts is backend-agnostic (Jev Ultrafast ↔ GLiNER2 Ultrafast ↔ Cua-S1 specialist ↔ Stagehand experimental Jev stack; **OCR+AX desktop:** typesafe-computer-use hosted Jev, never ships a screenshot for the *decision*; Cua-S1 is not TypeSafe Jev; Stagehand pick is a fast path, not a replacement; **≠** jev-macos-loop OmniParser **≠** camoufox; **ASR voice-browser:** jev-voice-browser hosted Jev, never ships a waveform). GLiFormer encoder serving `/v1/systemone` is a class-backend (jeff), not a Jev replica. Local MLX PCD is O(1) constrained-AR speed, **not** a calibrated Noul (system-one-benchmark Brier 0.3884 vs Jev 0.1096). **jevmlx** is the productized Apple Silicon one-pass schema→JSON+probs library (softmax ≠ Noul; no local leaderboard yet). **openvons** is an independent open-Jev class (LM/vision/voice finite-choice+prob; JevPick 3.2–4.8×; Flutter on-device; `/v1/systemone` wire-compat, not a TypeSafe replica). **OpenJev** (IamBusy) is a local 0.6B LoRA+scalar head on `/v1/decide` (45/60 *theirs*; **not** a TypeSafe drop-in; distinct from hraness/sysone OpenJev runners). **semif-serve** puts SemIf behind `/v1/systemone` (runoff, no option ceiling; 1164 vs 178 ms *theirs*; wire-compat ≠ replica). **grande** Rust/WebGPU kev-shaped branches (JGLUE *theirs* JNLI 0.614 ECE→0.088 / JCQA 0.853; 270M 0.710/0.710; isolation 0.098/0.996). **laya-jolt** Clojure/Jolt byte parity vs Python Laya. **JEV-CPU** SemIf on CPU (leesk212; Meanblock 404). **local-jev** ONNX ModernBERT measured not-equivalent (done 30% / shape 57% *theirs*). **GLiNER2→Choice/Score/Noul spec** (Eran-BA; design only, ≠ jeff GLiFormer). **openJev-verdict-2.0** competing NAR claims as **audit object not endorsement** (77.10%/0.0636/0.0144 *theirs*; PR #1; ≠ IamBusy/OpenJev `/v1/decide`). **chakuho** 1-token logprob local `/v1/systemone` (softmax ≠ Noul; coverage ≠ correctness; GUI 336 *theirs* 27B 95%/92% vs Jev 89%/82%). **jevinf** open replica engine (NanoJev/decider-2b/Laya; 2.57×/2.27× 100% argmax; MPS only). **laya-multilingual** mmBERT-base 322M (MASSIVE 0.366/0.387 vs English laya 0.227/0.733; Khmer 0.000@0.952 conf; ships uncalibrated). **schema-scorer** DeBERTa-v3-large scalar head (Hub; GitHub 404; v2 Choice 0.841 *theirs*; peaked ranking ≠ calibration). **githubnext/localjev** prompted-JSON `/v1/systemone` (MIT **261★**; TypeSafe SDK drop-in; wire-compat ≠ logit-equiv vs razorback16 structured-read; **≠** kunchenguid/local-jev; 1,200-req bake-off *theirs* Qwen3.6 76.7% / Gemma 26B 75.0% / DiffusionGemma 74.2% short; not calibrated). **NandhaKishorM/laya** PyPI+Router packaging of Hub Laya (Apache-2.0; **710★**; not a new species; T4 32.8 ms *theirs*; post-T ECE 0.081 vs Jev 0.246; Banking77 0.425 vs Jev 0.870; 0.766 is fine-tune not zero-shot; 0.85 still soft; **≠** TypeSafe `/v1/systemone`). **external openjev census** (@airesearch12 tweet ≠ jevbench v1.1; GLiNER2+routers class-boundary; incomplete vs watch). **JevBench v1.2 scored board** (geo-mean I/C/S/K 25% each; Jev 75.3 / SemIf 74.6 *theirs*; cal ON rank; instruction models in the table; ≠ v1.1 87.6; ≠ tweet census; Laya absent gap; Qwen3.8 27B ≠ Archer). **Hourly 0842:** already-folded class as a recipe (wire≠logit · product+FALLBACK · packaging honesty · pointer-not-generator · leaderboard VOI); skip thin noise. **Open LoRA replica, different jeff:** GestaltLabs/Jeff-1 LoRA Qwen3-4B ≠ logan-markewich/jeff GLiFormer; acc/ECE tradeoff n=9730 *theirs*; set reused. **Hourly 1241:** typesafeai-sdk-community not a new species; 2389-research/judgement license null; confidence ≠ winner p; jevbrain AUTO_ACT is not a Noul. **Hourly 1347:** FrancoisChastel/jev-code ≠ npm jev-code; ctmx/openrouter-jev-mcp Decision-as-Plugin; claudecode-jev-marketplace fail-open not hot path; pedroknigge/mcp_jev packs not ask_jev; 0thernet/system-one-skills deterministic verify; hari007sh/jev ≠ dannote/jev; nanoprune 2.8MB ECE 2.58%. **Hourly 1441:** Foq ~25ms/2.2GB local; rev prefill-only + HF jev-0.5b; robfrase/jev planning memo; meldltd/meldecision laya-go ONNX; laya-doom never pixels; logixism/laya-api empty README; akpsahan/laya ≠ Archer; Nibir1/typesafe-go ≠ official. **SIGNAL §94:** unofficial ≠ TypeSafe JA ModernBERT (argos1111/modernbert-ja-310m-jev; format_version modernbert-jev/1); Argos1111/jev_local ≠ us/jev-local ≠ kunchenguid/local-jev; LFM default ≠ ModernBERT backend; Nemotron ≠ TypeSafe Jev; djev-dev complements djev-spark (images as Choice options); Laya essay ≠ new species; hosted bootstrap ≠ silent TypeSafe. **Hourly 1541:** llama-jev llama.cpp replica; softmax ≠ Noul; **≠** TypeSafe **≠** pcdServer. **SIGNAL §97:** GLiNER2 native Apple path; unofficial Swift/Core ML GLiNER 2.5-small; entity spans + confidence; not Choice/Score/Noul; not TypeSafe; label descriptions as schema; on-device ANE economics; honesty locks; shershah1024/gliner-native-runtime ≠ Fastino; ≠ gliner25-compaction ≠ gliner2-ultrafast ≠ Eran-BA/Jev_from_GLiNER2 ≠ NSStudent/JevSwiftSDK ≠ jevmlx; default threshold 0.1 still soft. **Hourly 1740:** Decision Graph Protocol frame→assess→commit; app retains permissions/effects; Jev-first assessor-neutral; guarded commit / receipt/next frame; assessment batching; hard-gating DGP as safety theater; numerous-com/dgp ≠ TypeSafe official; jegrep calibrated path+range Nouls; no embeddings/index/daemon; ~$0.01–0.03 typical; agent --json; can1357/jegrep ≠ Bentlybro/jevgrep ≠ uehaj/jev-semgrep; Archer-arch fidelity; kev family OOD 0.76–0.77 vs Jev 0.86; block-causal isolation; pointer/readout CE-trained; /v1/systemone drop-in; replica honesty. **Hourly 1843:** variable-N option scoring as the trainable object; dynamic candidate bags not fixed label sets; zwliJay/jev-forge ≠ NanoJev; not a new class-table species; NAR local drop-in; wfzyx/von late-catch HIGH; open replica economics / latency vs closed Jev; competing NAR claims / replica honesty; gut/judge are control-flow overlays not species. **Hourly 1943:** softmax over allowed tokens ≠ Noul; on-device Laya CoreML ANE; JSON Schema → typed JSON via Jev; noul_threshold 0.5 decoder not a proof; mizorewww/laya-coreml ≠ gliner-native-runtime ≠ jevmlx ≠ NandhaKishorM/laya; Micha0827/snapjudge ≠ githubnext/localjev ≠ jevmlx ≠ cendress/SnapJudge; Jev IS the if-statement is a language primitive not a new species **Hourly 2041:** resume-screening bias audit methodology; Plan/PRD panel → code-owned pass|review|block; cost-aware multi-model routing/escalation; frozen-protocol zero-shot bench; context-window admission control; typed decision control plane; receipt ≠ authorization; live 15-dim typed rubric re-score per pause; adversarial pre-registered Jev eval; confidence does not track ignorance; provider-neutral Elixir/BEAM Noul/Choice/Score SDK **Hourly 2145:** question-linting of Jev questions themselves; nine jaggedness rules, no API key, no labelled data; static lint ≠ measured separation; yodablocks/jevq ≠ tenbin ≠ JevLint ≠ commitjev; open-weights Laya as class exemplar (binding); Nx/Bumblebee runtime; host chooses backend; ChristianAlexander/laya_ex ≠ system_one_sdk ≠ dannote/jev ≠ NandhaKishorM/laya; on-chain/edge Laya deploy; parity_verified stays false; model output never grants Tx; humandebri/IC-Laya ≠ laya_ex; auditable weekend replica; Jev outputs never used for training; unpaired 0.577 vs 0.727; agilabs-ai/jev48 ≠ JevBench ≠ Mapika/decider; adversarial dual-judge / framing attack surface; comparative framing is the usable judgment; prior injection crowds out evidence; copyleftdev/ember ≠ ember.js; Laya specialist fine-tune pipeline; training still GPU-pending; PIXELZX0/XERON ≠ convaiinnovations/laya; Hub Laya replica drop; daliborsb/laya ≠ convaiinnovations/laya ≠ NandhaKishorM/laya; System One student distillation corpus; gold is programmatic; teacher is closed-API clone; MagaBitmex/jev-4b-distill-data ≠ missing student checkpoint; non-LLM VIN System One; planning depth not chat; lewislululu/jevon ≠ douglance/jevon; source-bound evidence checks; local quote mismatch needs no API; exit 0 ≠ claim truth; WaynezProg/jev-kit ≠ jonathanavis96/jev-kit (Airlock) ≠ jev-use ≠ jev-mcp **Hourly 2246:** independent System One evidence catalog; 19 reviewed records; scores not one leaderboard; no external record currently reproduced; TokenTrim no-Jev matched hybrid 62.4%; reachjalil/system-one-bench ≠ mallahyari/system-one-benchmark; 21 tasks · 134 items · 208 questions; scenes from public GitHub contracts, not production logs; SivletLabs/jev-eval ≠ willkelly/jev-evaluation ≠ 4esv/jev-eval ≠ xxkuboxx/jev-eval ≠ onlyoneaman/jev-eval ≠ dayhaysoos/jevals; option isolation (sibling-blind); permutation-equivariant; Hub OWNER not published; nafisazizir/hev ≠ jaredpalmer/kev; frozen local LLM logits, no trained decision head; residual-head 9,222-param decreased 73/96→67/96; confidence = 1−normalized entropy, not P(correct); yuki-oshio/mini-jev ≠ r-ms/mini-jev; Jev classifier as autoregressive next-token predictor; ChatJev-style soundness theater; erik-dunteman/ChatJev ≠ dannote/jev ≠ jev-gpt; calibrated decision head × AlphaProof value head; implementation-layer isomorphism, semantic difference; timeout = censoring; do not launder Noul as proof; parallel rank-prediction vs serial selection; independent questions can conflict; zzzzzec/jevsort ≠ keltokhy/jsort; curated open System One ecosystem catalog; rupeshpoojary9/awesome-open-system-one ≠ AnotiaWang/awesome-jev; arXiv paper radar with Jev relevance scoring; ranking ≠ calibration / 0.5 still soft; fail-open failed evals not marked seen **Hourly 2340:** train calibrated ~27M from scratch; typed Q→prob dist / one forward pass / no LLM decode; hyusi2003/MiniSystemOne ≠ Colvin0315/MiniSystemOne; description-only stub / size 5; ESCI hard probe fails four of six; jev_bool ECE 0.242 inversion 0.255; do not re-fold §60 six-gates as new; jobbyjev one-request-per-company from batch-size result; find/design/evaluate TypeSafe Jev decision loops; karanb192/jev-architect ≠ samtay32/jev-system-architect; Jairik/jev-distiller size 1; distill-Jev UI stub / do not distill Jev as teacher of record; post-launch scored use-case map / Jev self-scores then human curation; licensedsaucer9-web/jev-opportunities; Jev-inize a use case into classifier/router; gavinHuang/jevinize → simple-jev not TypeSafe; featherless-ai/simple-jev; compare saved decisions / same label can still change the branch; VihaanAgarwal/jev-diff ≠ Saik0s/diffusiongemma-jev-macos; not tested with a live Jev API key; constrained logprob + temp/Platt ≠ Noul; OpenJevPro pastes openjev-sglang JevBench as own; zhangcy122/OpenJevPro ≠ IamBusy/OpenJev ≠ ekzhang/openjev-sglang; PolyForm Noncommercial; SmolLM-135M / sub-70ms / 0 output tokens; demo P(True) 0.5052 / Choice conf 0.2872 / Score conf 0.0055; README claims MIT / GitHub license null / no LICENSE file; patelvishwa112/jev-system-one-rlcd ≠ arnabgho/rlcd-lite ≠ blackwood-rlcd; source-backed Awesome Jev radar / 306+ commit-pinned; logicrw/awesome-jev-projects ≠ AnotiaWang/awesome-jev ≠ yibie/awesome-jev ≠ cobanov/awesome-jev ≠ rupeshpoojary9/awesome-open-system-one; auto GitHub sync / Issue-only submissions; hashed n-gram encoder / rival-aware attention; olanotolu/jevbetter vs jevlike starter; synthetic hard menus top-1 0.916 vs 0.873 / ECE 0.0182 vs 0.0367 / 40 vs 4608 menus/sec; shuffled-context control 0.335 | `references/judgment-class.md` Apply 0042 (`notes.md` §105): structured probability readouts; distribution > argmax; Noul 0.5 midpoint; score is expectation not integer; bare HTTP not SDK; Arohtea/jev-readout; Jev-style Choice/Score/Noul from ordinary models; optional DSH plugin; schema-valid ≠ calibrated; gulagala001/jevify ≠ Mintzs/jevify; Laya RLCD benchmark; 40.3% below constant-answer; open-weight measurement; mourad-ghafiri/laya-rlcd-benchmark ≠ yibie/laya-jev-lab; cheap fail-open semantic edge; second signal not sole; FastLoopError catch; SupremeDreamZ/jev-fastloop ≠ jev-ultrafast; asking more questions in one call; 0.980 at every N; nearly not fully deterministic; TheWebDevel/jev-fanout; Qwen3-VL perception + Jev decisions train RL; 0 model calls at deployment; VLM alone 1.7 vs +Jev 4.4; harneet2512/reflexrl ≠ khordoo/jev-reflex-autonomy-lab; independent Jev API vs Laya; cascade 0.60 matches 78% at 1.8×; noul facts not judgements; yibie/laya-jev-lab ≠ dairui1/jev-lab ≠ BrendanH18/jev-lab; GLiNER vs GLiFormer vs Laya vs Jev; extractors ≠ decision engines; Laya dict-instructions collapse 58.3%; umstek/zero-shot-ie-bench; decisions-per-minute & cost; 204 moves vs 73; throughput not intelligence; angelgalvisc/snake-arena-jev-vs-llms ≠ vtrivedy/jev-plays-games; behavioral contracts; pin expectations eval upgrades; raw 0.94 is not a release; sathariels/jevcheck ≠ dayhaysoos/jevals ≠ SivletLabs/jev-eval; evidence-linked dependency upgrade; Jev never generates filenames; no_direct_evidence ≠ safe to merge; GaneshVG18/upgrade-radar ≠ LYchoon/paper-radar-jev; discography theme/mood/complexity; five atomic questions one call; lirantal/discoprint / Apply 0145 (`notes.md` §106): Turn any open LLM into System-One Jev; description-only stub / size 0; Exu is a toolkit, not a method; typed Q → probability dists; JSON parse of generated text ≠ Noul; classifier.dev fast tier 84.8 is Jev behind its own API; hard budget filter before Jev; Jev judges the next state, XState enforces transitions; catalog gravity; ★339 live REST; query-side encoders, not a Jev replica; transformers.js AutoModel cannot load this graph; This Space contains no benchmark result yet; do not reopen or amend PR #23. / Apply 0243: Benchmark-driven Jev router and judge; cheap alone is not success; Jev does not write, sum prices, or claim accuracy %; Sol 94.2 / Luna 83.9 / Jev path 89.7; 19.2% Sol / 62.3% cost save / 4.5pp miss of 2pp non-inferiority; p50 latency worse than Sol due to routing overhead; erendikmenn/jev-llm-router-benchmark ≠ jev-rag-benchmark ≠ ryantsai/jev-llm-router; Express + node:sqlite; mock and Jev decision engines; previous_ticket_count >= 3 is code; MIN_CONFIDENCE 0.6 still soft; substring false positives; aesaganda/jev-ticket-router ≠ SarathChandraBellam/jev-vs-llm-ticket-router; Universal Figure & Diagram Router; confidence ≥ 0.85 hard-gate is theater; generative AI banned from scientific plots; six visual branches; human-labeled (state, question, label); 166,054 rows / 22 configs; soft_label for human uncertainty; Praveenrajus/jev-bench ≠ fstandhartinger/jevbench; ternary bonsai System One GGUF; Hub does not ship weights; 100/100 easy T/F is not Harbor; label_mass ≠ correctness; stock llama.cpp Q2_0 silently gibberish; NicolaiMTLassen/open-bonzi-jev ≠ NicolaiLassen; transformers.js DeBERTa ONNX; temperature 1.05; AutoModel from_pretrained works; onnx-community/open-jev-deberta-v3-large-ONNX ≠ system-one-qwen3.5-4b-scorer-ONNX; 107★ densify; GH 151M vs README 149.6M; PR #1 now closed unmerged; do not re-fold §71 claim-audit as a beat; typed decisions, RLCD, confidence-gated routing; structured ≠ correct; mock not live API; 26 tests; wjdjdakf17/jev-study ≠ baekenough/jev-study / Apply 0345 (`notes.md` §108): **Open reproduction densifies / measurement densifies PRIMARY** (bonzi-27b-v2 / ternary-8b / 27b-v1 GGUF family densify; WANLI-256 74.6% / 65.2% / 71.1% *theirs*; Bonsai 1 27B Q1_0 runs on stock llama.cpp; ternary still needs PrismML fork; hf:heman10x/openJev-verdict-2.0 twin tokenizer-only; OpenJev Vision image classification + uncertainty; CLEVR-4 held-out joint 0%; hfdataset:IamBusy/OpenJev-Vision-Research-v0.1 12,832; 294,912 derived targets not independent samples; Laya multilingual ONNX WebGPU typed-decisions port; 63/63 selected answers / 5.1e-4 CPU / 1.2e-2 WebGPU; UpHash-Network/mini-jev is yuki-oshio transfer; jev-injection-bench 11,900 labelled prompts; Jev best ranking / Haiku better ECE 0.021 vs 0.058; 0.5–0.9 band is where Jev's numbers do not mean what they say; Prompt wording moves panic 28%; manojlds/jev-dspy-bench ≠ dspachos/jev-dspy ≠ jmanhype/jev-dspy-lab; Jev agreement is similarity, never ground truth; no aggregate quality grade or merge gate; AbstentionBench-on-Jev rank 1 of 20 vs 2025 field; question-asymmetry; forward-looking 0.465 never extreme; openkev calibration layer not a runtime; ECE vs coverage independent; select_threshold returns inf; escalation catches uncertainty not ignorance; misakaikato/openkev ≠ jaredpalmer/kev; pdf-race Docling→Jev vs Gemini; parser owns the wall clock; 12/12 tie is a tie; titles selected not generated; flopcheck 16 calibrated tweet judgments; mechanical tells in code; ZeroX-01/jev-atlas ≠ Zaious/jev-capability-atlas ≠ gorock007/jev-atlas; Laya calibration lab Gradio MCP; T never changes argmax; confidence ≠ top-label p; easy probe set refused; 40–48 rows too small to ship T; do not reopen or amend PR #23 or #24 or #25. Apply 0439 (`notes.md` §109): Gemma-4 26B-A4B jevify classification+calibration; LoRA adapter twin not independent eval; Gemma-4 E4B jevify; E4B LoRA stub card; kushalpatil/jevify-gemma4 ≠ Mintzs/jevify ≠ gulagala001/jevify ≠ uspraveen/Jevify; GH kushalpatil07/jevify 404; PAWS 0.580/ece 0.288 is the weak cell; smaller E4B slightly better OOD ECE than 26B-A4B; Hub jevify merged LoRA ships weights; bonzi Bonsai-8B v1 GGUF densify; Bonsai-1.7B v1; Bonsai-4B v1; WANLI-256 64.5% / 60.2% / 52.0% *theirs*; rank #4 / #5 / #6 of 6; JulesHuisman/jev-eval scaffolding / README SHA c356a584 (was empty e69de29b); JulesHuisman/jev-eval ≠ SivletLabs/jev-eval ≠ willkelly/jev-evaluation ≠ 4esv ≠ xxkuboxx ≠ onlyoneaman ≠ dayhaysoos/jevals; 7 bands 6/10 vs 40 bands 0/10; source receipts + confidence slider re-policy without re-inference; 32/32 synthetic is smoke not production; classify HF datasets across typed semantic dimensions; roadus2 watch misspelling; lock roadius2/ultra_laya; ultra_laya REVIEW defects; default branch claude/laya-jev-review-gg5ppo; XNLI EN 88.3% ECE 0.032 → RU 77.3% ECE 0.096; Δ −11.0 pp [−14.2,−7.8]; ECE +0.063; MASSIVE no detectable difference at n=600; confidence is function of p_max (r=1.000); pointer-not-generator 400 human-authored responses; proposed ≠ authorized; FewRel 160: Jev 85.0% vs lexical 13.125%; gated 100% (95/95) coverage 59.375%; J++ composable semantic computation language; judge-jev 0.5 still soft; 947 repos scored; A 273 / B 302 / C 372; LLM rubric ≠ benches; No benchmark winner is claimed; phishing: naive 62.6% vs regex 91.8%; 5-atomic + LR 95.0% *theirs*; AITuber tension ±15; README npm global; repo is Rust; git-confess code owns counting/blame/ratio; httpx exhibit 11% (13/119) *theirs*; 90d trend +12.40% vs random +12.75% vs BH +41.71%; 5m win rate 25%; Awesomejev 656 entries / 38,160 stars; tracker likes 64 (+4) lastModified UNCHANGED; Laya present; Blackwood ABSENT; Archer still promised_not_landed; do not reopen or amend PR #23/#24/#25/#26. Apply 0541 (`notes.md` §110): Blackwood tracker ABSENT; likes 2 gated manual; r = c - p_a; ECE 0.021; acc 0.807 vs warmup 0.746; calibration beyond ~500 tokens unmeasured; Independent primitive; 11.57s vs 54.10s · 4.67× · 120/128 *theirs*; default path is pretrained Gemma probs not trained RLCD head; GH Meanblock 404; lock leesk212/JEV-CPU; softmax over letter slots ≠ Noul; WANLI 0.741 vs openjev v2 0.77 *theirs*; 3-way NLI ≠ Noul; priority 0.464 = majority floor; banking77 contaminated; raw margins not probabilities; GH jev-haiku-benchmarking 404; do not distill Jev as teacher of record (they distilled Haiku); “0.9 is not one number”; ranking ≠ calibration; banking77 0.8–0.9 stated 0.86 actual 0.73 over-confident *theirs*; ≠ Praveenrajus/jev-bench ≠ fstandhartinger/jevbench; $0.0000153–$0.0000226 vs circulating $0.0004 (~20×); Score is 0..n-1 expectation not 0–1; Noul has no confidence field; TCP floor 198.8 ms; type reliability is not a reason to choose Jev (json_schema 5/5); gateway tax not one number; Function-only 5/8 vs hybrid 8/8; 4/8 without Jev; 8 designed cases not conversion lift; ≠ RadRebelSam/awesome-jev; 200-row pilot Jev 86.5% 173/200 vs Gemini Flash-Lite 86.0% 172/200 vs Pro 87.0% 174/200 *theirs*; not a ranking; NLI Tetris argmax P(entail)−P(contradict); 情緒測謊器; 1q 396ms / 30q 567ms; ±0.03; 33q $0.000045 vs Gemini ~5× slower ~60× cost *theirs*; ≠ realZachi/jevtest; 8-example Jev vs GPT-5.6 Sol ~64× cost 5.4× latency *theirs*; synthetic; no inference; ≠ JevBench v1.2 §78; Judged 3317 / listed 2560; Jev judges, code applies policy; catalog ≠ endorsement; APA “microsecond policy / zero hallucination” overclaim; Client-side quiz; pointer from held docs; scanned-PDF warn; CSP only api.typesafe.ai; Jev judges / agent reasons / user decides; selecting an option is not permission to implement; degraded fallback; pattern exact, judgement must clear floor; no matching pattern → no model call; not a correctness oracle; $0.00022 vs chat $0.00306 *theirs*; Spec vs artifact remainder; treating 0.85 as 85% / minProbability hard-gate as Harbor; VERIFY acquires discriminating evidence, never same-pool confidence-only rescoring; fast/full/max are ceilings not sizes; Solar writes, Jev chooses NEXT ACTION; SemIf 2186★ (+20 vs §109 2166); jevlike 1038★ (+7 vs 1031); TypeAR 14★ flat; AnotiaWang 96★ (+1 vs 95); yibie/awesome-jev 490★; Laya likes 802 (was 783); tracker likes 64 flat, lastModified UNCHANGED; do not reopen or amend PR #23/#24/#25/#26/#27. Apply 0743 (`notes.md` §113): Hourly 0743 uniqueness lock: Heman10x-NGU/Verdict-open-jev ≠ Heman10x-NGU/openJev-verdict-2.0; TF-IDF + LogReg ECE 0.0207 vs Jev 0.1440; Verdict-open-jev 48.07% vs Jev 90.80%; abstention combined recall 10.00%; p50 35.58 ms; K=25 (maximum capacity) 72.00%; 0.85 coverage 84.60% selective risk 1.18%; 26.1× faster than standard Qwen JSON generation; Jevify 90.0% / 167 ms CUDA graphs disabled; Finding 1: Brier on stated confidence alone is a trap; grpo_rlcr 0.78 / ECE 0.084; reliability 0.007 but resolution 0.000; 27 900 schema-driven decisions; 13 600 / 13 600 questions; candidate mass min 0.99999624; 22 configs · 166,054 rows · 4 calibration-gold; sha a39eba3f; Student B MAE 0.148 / Pearson 0.836 / 86.0%; pngwn/open-jev-laya-bench README 404; sha 9f69c742 likes 2; HDFS 0.9933 (745/750) / retain 0.0084; BGL ERROR/FATAL protection 1.0000; 2,479 / 2,500 HDFS uncertain; cache hit 0.9648 (2412/2500); $0.153936 estimated; E2 recomputes from saved probabilities; Space sha eda59e0a; MASSIVE English 0.783 / Khmer 0.033 / Hindi 0.133; 40–48 rows too small to ship T; T never changes argmax; siren2345/jev-single-decode-transformers ≠ siren2345/jev-single-decode; Split Transformers experiment from llama.cpp runtime; tanayvasishtha/jev-lab ≠ dairui1/jev-lab ≠ BrendanH18/jev-lab ≠ yibie/laya-jev-lab; Four experiments stress-testing TypeSafe's Jev: calibration, bundle bias, label bias, and ensembling; second pass must be $0.00 from cache; The pages never call Jev; Gemma 4 31B 77.0% / Jev 1.13.0 61.4% / Laya 322M 0.0%; restriction state 95.0% against 84.4%; None of the systems are particularly good at knowing when to stop and ask; They skip the question and call a tool directly; 100% schema pass; six-field joint 48.8% vs 72.8%; ywchiu/jev_benchmark ≠ Running-Dolphins/jev-bench ≠ Praveenrajus/jev-bench; ACT / REVIEW / FALLBACK; A provider failure, timeout, malformed output, or missing answer is **not** a policy outcome; confidence is descriptive provider output, not a substitute for probability; Quality denominators include only valid scored answers; an exact halfway tie chooses the lower level; aiwithenoch/Jev-Skill ≠ simplosophy/jev-skill ≠ laguagu/jev-skills; The local path does not claim to turn a smaller checkpoint into Jev; Low support becomes decision: "review"; MIT-0 SPDX NOASSERTION; current-llm; 结构兼容,不是 Jev 模型能力; altryne/jevify ≠ Mintzs/jevify ≠ gulagala001/jevify ≠ uspraveen/Jevify; Find where Jev belongs. Design the questions. Measure the difference; TypeAR-AI/TypeAR 301 → TypeLLM/TypeLLM; TypeLLM/TypeLLM 16★; SemIf 2241★ (+34 vs §111 2207); jevlike 1051★ (+8 vs 1043); AnotiaWang 98★ (+1 vs 97); yibie/awesome-jev 525★ (+19 vs 506); Laya likes 864 (was 822); tracker likes 67 (+3 vs 64); lastModified UNCHANGED `2026-09-20T04:29:16.000Z`; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#32/#33; notes.md §113 Apply 0646 (`notes.md` §111): Calibration is not alpha; NO CURRENT ALPHA CANDIDATE; ΔR² approximately +0.00084; Brier 0.2131387; ECE 0.0421875; Adding Jev probability to deterministic volatility improved Brier by only 1.4058e-05; default 0.5 keeps zero non pinned; keepResult median 0.14 to 0.17; keepCall median 0.28 to 0.35; usable range is about 0.10 to 0.25; 7.8% to 57.9%; judges results it never sees; task-finish eval not built yet; $0.002 per compaction; slavadubrov/sgr-judge-bench ≠ slavadubrov/jev-judge-bench; Jev 108/120 $0.083 0.34 s; Luna SGR 114/120; paired Jev accuracy-difference intervals include zero; not evidence of equivalence; GLM SGR 26/120 93 format failures; Terra-planned Jev hybrid 55/120; rule-based by default, optionally Jev-backed; empty README; missing key cannot break the experience; prefill plus exactly one decode; softmax over A/B/C ≠ Noul; BBQ 9,053/10,000 (90.53%); ECE 0.0890; Mean confidence 0.9943; overconfident; score and noul not implemented; DGUI 12 rows (was 6); INSTRUCT 119 rows likes 2; encode the state once, decide everything in parallel; 0.740 accuracy against a 0.508 majority; ECE 0.047; fine-tune's advantage ends where its 384-token training data does; jasonkneen/open-jev ≠ pngwn/open-jev; same sha d41dc3cd; Space does not call Jev; recomputes routing from saved probabilities; 200-case Jev 97.0% / 100.0% / 95.0% / MAE 9.22; synthetic repository benchmark; Jev evaluations are advisory; YehuiTang0316/jev-nlgrep ≠ Bentlybro/jevgrep ≠ can1357/jegrep ≠ uehaj/jev-semgrep; default threshold 0.8 still soft; 40-line windows cannot prove whole function; token-native sequential start/end Choice; Gemini/Haiku stubs not configured yet; handful of hand-written examples, not a benchmark; Jev judged exactly what it was given; laguagu/jev-skills ≠ laguagu/jev-evidence-lab ≠ Pleo2/awesome-jev-agent-skills; contract_passed is not a claim of guaranteed factual truth; Wilson lower bound 0.85 floor; fixture mode no savings claim; SemIf 2207★ (+21 vs §110 2186); jevlike 1043★ (+5 vs 1038); TypeAR 15★ (+1 vs 14); AnotiaWang 97★ (+1 vs 96); yibie/awesome-jev 506★ (+16 vs 490); Laya likes 822 (was 802); tracker likes 64 flat, lastModified UNCHANGED; do not reopen or amend PR #23/#24/#25/#26/#27/#28. Hourly 1441 densify: dual /v1/systemone + /v1/chat/completions (dual serving is not generate; chat 501 on MLX); Hosted Codiv ≠ TypeSafe; hr98w/jev-visual candidate probabilities are relative not correctness; 37.30s → 2.40s at 64 decisions *theirs*; Breakout 9 bricks 6 returns 2 lives *theirs*; jkudish/jev-mcp recommendation is advisory / the server never blocks on its own; TypeSafe CLERC 5% to 18% *theirs*; Zefan-Cai/Open-Jev LoRA ≠ RLCD replica; 2B 94.71% 9B 97.54% hard test *theirs*; cristianoliveira/jeq pass-min 0.8 still soft; JEQ does not own actions. Apply 1746 (`notes.md` §128): truncated thinking then constrained decode; 0.8B thinking On 0/18 *theirs*; forced closure 20/20 type-valid *theirs*; Kev-0.8B completes family; 4B new-source 0.794/0.832 *theirs*; 9B new-source 0.812/0.837 *theirs*; Qwen3.5 ≠ Archer; Facts go to code. Judgments go to Jev. Only facts can block.; Jev never blocks. Apply Open-Jev densify (`notes.md` §125): systems latency ≠ semantic equivalence; hard acc ≠ calibrated Noul; not merged base models; Open-Jev TREC pending; prefix caching experimental/off by default; LoRA ≠ RLCD replica; type-valid ≠ exact.) | | Open weights vs constrained decoding vs proprietary API | Three open paths: encoder open-jev / AR constrained decode (TypeAR + pcdServer; decision-token LoRA; packed one-forward logprob on open LLMs; **CUDA/PyTorch local replica** jevify — uncalibrated likelihoods ≠ Noul) / trained decision-only (Laya + ONNX port, Nimble, kev, **blackwood-rlcd** multimodal now, Archer Watch still Watch). **Domain LoRA specialist on independent gold** (not a Jev teacher-copy): train when downstream reads p; few-shot hosted when only argmax. Local `/v1/systemone` surfaces: jev-local (stub until `hf`), kev (trained pointer), von (§49 Needle SAN snapshot ≠ this-pass 395M / n=78; not a replica), **jeff** (GLiFormer-400M encoder, typesafe-sdk drop-in — not a Jev replica), **local-jev** (ModernBERT approximation — not equivalence). Laya ONNX: Mattepiu port vs **gqgs** complete browser int8 (distinct). Loopback **gateway** (sysone) routes hosted + local; does not run weights. Softmax over allowed tokens ≠ Noul. **jevmlx:** MLX one-pass schema→JSON+probs (Apple Silicon replica economics; not a Jev replica). **openvons:** independent open-Jev class (Apache-2.0 code; GitHub SPDX NOASSERTION); wire-compat `/v1/systemone`; NOTA + execute/confirm/reject. **OpenJev** `/v1/decide` ≠ TypeSafe. **semif-serve:** SemIf runoff wire (MIT pyproject / GitHub SPDX null). **grande** Rust/WebGPU `/v1/systemone` (license null; softmax ≠ Noul until T). **laya-jolt** Clojure Apache-2.0 byte-parity Laya. **JEV-CPU** CPU SemIf. **local-jev** ONNX NLI approximation (confidence omitted). **chakuho:** 1-token logprob constrained-AR endpoint (uncalibrated; coverage is format-mass). **jevinf:** Jev-kind engine + wire (not a replica). **laya-multilingual:** mmBERT-base for non-English; route by script. **githubnext/localjev:** Bun Chat Completions bridge; prompted JSON + entropy confidence; **≠** kunchenguid/local-jev; **≠** razorback16 logits. **NandhaKishorM/laya:** PyPI `laya` + Router over the three Hub ckpts; packaging ≠ new species; Jev still leads >20 options / soft-acc / raw ECE. **@airesearch12 census:** list ≠ rank; GLiNER2/routers counted as openjevs are a class-boundary. **JevBench v1.2:** same class table includes Luna/Gemini/DeepSeek/Qwen3.8; OpenJev on board = razorback16 DiffusionGemma ≠ IamBusy; SemIf formerly OpenJev. **≠** GestaltLabs/Jeff-1 Qwen3-4B LoRA replica. **SIGNAL §94:** unofficial JA ModernBERT cross-encoder (format_version modernbert-jev/1; unofficial ≠ TypeSafe; LFM default ≠ ModernBERT backend); Nemotron ≠ TypeSafe Jev / not a calibrated replacement; djev-dev complements djev-spark (native image / images as Choice options); Laya essay numbers *theirs* / Router/OOD confidence. **Hourly 1541:** llama-jev llama.cpp replica; softmax ≠ Noul; **≠** TypeSafe **≠** pcdServer. **Hourly 1740:** Archer-arch fidelity; kev family OOD 0.76–0.77 vs Jev 0.86; block-causal isolation; pointer/readout CE-trained; /v1/systemone drop-in; replica honesty. **Hourly 1843:** NAR local drop-in; wfzyx/von late-catch HIGH; open replica economics / latency vs closed Jev; competing NAR claims / replica honesty; variable-N option scoring as the trainable object; dynamic candidate bags not fixed label sets. candidate bags not fixed label sets; zwliJay/jev-forge ≠ NanoJev. **Hourly 1943:** on-device Laya CoreML ANE; ~5 ms P50 short decisions; 189/189 FP16 checkpoint parity; 10× not achieved; softmax over allowed tokens ≠ Noul; question-first cache **Hourly 2041:** resume-screening bias audit methodology; Plan/PRD panel → code-owned pass|review|block; cost-aware multi-model routing/escalation; frozen-protocol zero-shot bench; context-window admission control; typed decision control plane; receipt ≠ authorization; live 15-dim typed rubric re-score per pause; adversarial pre-registered Jev eval; confidence does not track ignorance; provider-neutral Elixir/BEAM Noul/Choice/Score SDK **Hourly 2145:** question-linting of Jev questions themselves; nine jaggedness rules, no API key, no labelled data; static lint ≠ measured separation; yodablocks/jevq ≠ tenbin ≠ JevLint ≠ commitjev; open-weights Laya as class exemplar (binding); Nx/Bumblebee runtime; host chooses backend; ChristianAlexander/laya_ex ≠ system_one_sdk ≠ dannote/jev ≠ NandhaKishorM/laya; on-chain/edge Laya deploy; parity_verified stays false; model output never grants Tx; humandebri/IC-Laya ≠ laya_ex; auditable weekend replica; Jev outputs never used for training; unpaired 0.577 vs 0.727; agilabs-ai/jev48 ≠ JevBench ≠ Mapika/decider; adversarial dual-judge / framing attack surface; comparative framing is the usable judgment; prior injection crowds out evidence; copyleftdev/ember ≠ ember.js; Laya specialist fine-tune pipeline; training still GPU-pending; PIXELZX0/XERON ≠ convaiinnovations/laya; Hub Laya replica drop; daliborsb/laya ≠ convaiinnovations/laya ≠ NandhaKishorM/laya; System One student distillation corpus; gold is programmatic; teacher is closed-API clone; MagaBitmex/jev-4b-distill-data ≠ missing student checkpoint; non-LLM VIN System One; planning depth not chat; lewislululu/jevon ≠ douglance/jevon; source-bound evidence checks; local quote mismatch needs no API; exit 0 ≠ claim truth; WaynezProg/jev-kit ≠ jonathanavis96/jev-kit (Airlock) ≠ jev-use ≠ jev-mcp **Hourly 2246:** independent System One evidence catalog; 19 reviewed records; scores not one leaderboard; no external record currently reproduced; TokenTrim no-Jev matched hybrid 62.4%; reachjalil/system-one-bench ≠ mallahyari/system-one-benchmark; 21 tasks · 134 items · 208 questions; scenes from public GitHub contracts, not production logs; SivletLabs/jev-eval ≠ willkelly/jev-evaluation ≠ 4esv/jev-eval ≠ xxkuboxx/jev-eval ≠ onlyoneaman/jev-eval ≠ dayhaysoos/jevals; option isolation (sibling-blind); permutation-equivariant; Hub OWNER not published; nafisazizir/hev ≠ jaredpalmer/kev; frozen local LLM logits, no trained decision head; residual-head 9,222-param decreased 73/96→67/96; confidence = 1−normalized entropy, not P(correct); yuki-oshio/mini-jev ≠ r-ms/mini-jev; Jev classifier as autoregressive next-token predictor; ChatJev-style soundness theater; erik-dunteman/ChatJev ≠ dannote/jev ≠ jev-gpt; calibrated decision head × AlphaProof value head; implementation-layer isomorphism, semantic difference; timeout = censoring; do not launder Noul as proof; parallel rank-prediction vs serial selection; independent questions can conflict; zzzzzec/jevsort ≠ keltokhy/jsort; curated open System One ecosystem catalog; rupeshpoojary9/awesome-open-system-one ≠ AnotiaWang/awesome-jev; arXiv paper radar with Jev relevance scoring; ranking ≠ calibration / 0.5 still soft; fail-open failed evals not marked seen **Hourly 2340:** train calibrated ~27M from scratch; typed Q→prob dist / one forward pass / no LLM decode; hyusi2003/MiniSystemOne ≠ Colvin0315/MiniSystemOne; description-only stub / size 5; ESCI hard probe fails four of six; jev_bool ECE 0.242 inversion 0.255; do not re-fold §60 six-gates as new; jobbyjev one-request-per-company from batch-size result; find/design/evaluate TypeSafe Jev decision loops; karanb192/jev-architect ≠ samtay32/jev-system-architect; Jairik/jev-distiller size 1; distill-Jev UI stub / do not distill Jev as teacher of record; post-launch scored use-case map / Jev self-scores then human curation; licensedsaucer9-web/jev-opportunities; Jev-inize a use case into classifier/router; gavinHuang/jevinize → simple-jev not TypeSafe; featherless-ai/simple-jev; compare saved decisions / same label can still change the branch; VihaanAgarwal/jev-diff ≠ Saik0s/diffusiongemma-jev-macos; not tested with a live Jev API key; constrained logprob + temp/Platt ≠ Noul; OpenJevPro pastes openjev-sglang JevBench as own; zhangcy122/OpenJevPro ≠ IamBusy/OpenJev ≠ ekzhang/openjev-sglang; PolyForm Noncommercial; SmolLM-135M / sub-70ms / 0 output tokens; demo P(True) 0.5052 / Choice conf 0.2872 / Score conf 0.0055; README claims MIT / GitHub license null / no LICENSE file; patelvishwa112/jev-system-one-rlcd ≠ arnabgho/rlcd-lite ≠ blackwood-rlcd; source-backed Awesome Jev radar / 306+ commit-pinned; logicrw/awesome-jev-projects ≠ AnotiaWang/awesome-jev ≠ yibie/awesome-jev ≠ cobanov/awesome-jev ≠ rupeshpoojary9/awesome-open-system-one; auto GitHub sync / Issue-only submissions; hashed n-gram encoder / rival-aware attention; olanotolu/jevbetter vs jevlike starter; synthetic hard menus top-1 0.916 vs 0.873 / ECE 0.0182 vs 0.0367 / 40 vs 4608 menus/sec; shuffled-context control 0.335 | `references/judgment-class.md` (when-to-use table) Apply 0042 (`notes.md` §105): structured probability readouts; distribution > argmax; Noul 0.5 midpoint; score is expectation not integer; bare HTTP not SDK; Arohtea/jev-readout; Jev-style Choice/Score/Noul from ordinary models; optional DSH plugin; schema-valid ≠ calibrated; gulagala001/jevify ≠ Mintzs/jevify; Laya RLCD benchmark; 40.3% below constant-answer; open-weight measurement; mourad-ghafiri/laya-rlcd-benchmark ≠ yibie/laya-jev-lab; cheap fail-open semantic edge; second signal not sole; FastLoopError catch; SupremeDreamZ/jev-fastloop ≠ jev-ultrafast; asking more questions in one call; 0.980 at every N; nearly not fully deterministic; TheWebDevel/jev-fanout; Qwen3-VL perception + Jev decisions train RL; 0 model calls at deployment; VLM alone 1.7 vs +Jev 4.4; harneet2512/reflexrl ≠ khordoo/jev-reflex-autonomy-lab; independent Jev API vs Laya; cascade 0.60 matches 78% at 1.8×; noul facts not judgements; yibie/laya-jev-lab ≠ dairui1/jev-lab ≠ BrendanH18/jev-lab; GLiNER vs GLiFormer vs Laya vs Jev; extractors ≠ decision engines; Laya dict-instructions collapse 58.3%; umstek/zero-shot-ie-bench; decisions-per-minute & cost; 204 moves vs 73; throughput not intelligence; angelgalvisc/snake-arena-jev-vs-llms ≠ vtrivedy/jev-plays-games; behavioral contracts; pin expectations eval upgrades; raw 0.94 is not a release; sathariels/jevcheck ≠ dayhaysoos/jevals ≠ SivletLabs/jev-eval; evidence-linked dependency upgrade; Jev never generates filenames; no_direct_evidence ≠ safe to merge; GaneshVG18/upgrade-radar ≠ LYchoon/paper-radar-jev; discography theme/mood/complexity; five atomic questions one call; lirantal/discoprint / Apply 0145 (`notes.md` §106): Turn any open LLM into System-One Jev; description-only stub / size 0; Exu is a toolkit, not a method; typed Q → probability dists; JSON parse of generated text ≠ Noul; classifier.dev fast tier 84.8 is Jev behind its own API; hard budget filter before Jev; Jev judges the next state, XState enforces transitions; catalog gravity; ★339 live REST; query-side encoders, not a Jev replica; transformers.js AutoModel cannot load this graph; This Space contains no benchmark result yet; do not reopen or amend PR #23. / Apply 0243: Benchmark-driven Jev router and judge; cheap alone is not success; Jev does not write, sum prices, or claim accuracy %; Sol 94.2 / Luna 83.9 / Jev path 89.7; 19.2% Sol / 62.3% cost save / 4.5pp miss of 2pp non-inferiority; p50 latency worse than Sol due to routing overhead; erendikmenn/jev-llm-router-benchmark ≠ jev-rag-benchmark ≠ ryantsai/jev-llm-router; Express + node:sqlite; mock and Jev decision engines; previous_ticket_count >= 3 is code; MIN_CONFIDENCE 0.6 still soft; substring false positives; aesaganda/jev-ticket-router ≠ SarathChandraBellam/jev-vs-llm-ticket-router; Universal Figure & Diagram Router; confidence ≥ 0.85 hard-gate is theater; generative AI banned from scientific plots; six visual branches; human-labeled (state, question, label); 166,054 rows / 22 configs; soft_label for human uncertainty; Praveenrajus/jev-bench ≠ fstandhartinger/jevbench; ternary bonsai System One GGUF; Hub does not ship weights; 100/100 easy T/F is not Harbor; label_mass ≠ correctness; stock llama.cpp Q2_0 silently gibberish; NicolaiMTLassen/open-bonzi-jev ≠ NicolaiLassen; transformers.js DeBERTa ONNX; temperature 1.05; AutoModel from_pretrained works; onnx-community/open-jev-deberta-v3-large-ONNX ≠ system-one-qwen3.5-4b-scorer-ONNX; 107★ densify; GH 151M vs README 149.6M; PR #1 now closed unmerged; do not re-fold §71 claim-audit as a beat; typed decisions, RLCD, confidence-gated routing; structured ≠ correct; mock not live API; 26 tests; wjdjdakf17/jev-study ≠ baekenough/jev-study / Apply 0345 (`notes.md` §108): **Open reproduction densifies / measurement densifies PRIMARY** (bonzi-27b-v2 / ternary-8b / 27b-v1 GGUF family densify; WANLI-256 74.6% / 65.2% / 71.1% *theirs*; Bonsai 1 27B Q1_0 runs on stock llama.cpp; ternary still needs PrismML fork; hf:heman10x/openJev-verdict-2.0 twin tokenizer-only; OpenJev Vision image classification + uncertainty; CLEVR-4 held-out joint 0%; hfdataset:IamBusy/OpenJev-Vision-Research-v0.1 12,832; 294,912 derived targets not independent samples; Laya multilingual ONNX WebGPU typed-decisions port; 63/63 selected answers / 5.1e-4 CPU / 1.2e-2 WebGPU; UpHash-Network/mini-jev is yuki-oshio transfer; jev-injection-bench 11,900 labelled prompts; Jev best ranking / Haiku better ECE 0.021 vs 0.058; 0.5–0.9 band is where Jev's numbers do not mean what they say; Prompt wording moves panic 28%; manojlds/jev-dspy-bench ≠ dspachos/jev-dspy ≠ jmanhype/jev-dspy-lab; Jev agreement is similarity, never ground truth; no aggregate quality grade or merge gate; AbstentionBench-on-Jev rank 1 of 20 vs 2025 field; question-asymmetry; forward-looking 0.465 never extreme; openkev calibration layer not a runtime; ECE vs coverage independent; select_threshold returns inf; escalation catches uncertainty not ignorance; misakaikato/openkev ≠ jaredpalmer/kev; pdf-race Docling→Jev vs Gemini; parser owns the wall clock; 12/12 tie is a tie; titles selected not generated; flopcheck 16 calibrated tweet judgments; mechanical tells in code; ZeroX-01/jev-atlas ≠ Zaious/jev-capability-atlas ≠ gorock007/jev-atlas; Laya calibration lab Gradio MCP; T never changes argmax; confidence ≠ top-label p; easy probe set refused; 40–48 rows too small to ship T; do not reopen or amend PR #23 or #24 or #25. Apply 0439 (`notes.md` §109): Gemma-4 26B-A4B jevify classification+calibration; LoRA adapter twin not independent eval; Gemma-4 E4B jevify; E4B LoRA stub card; kushalpatil/jevify-gemma4 ≠ Mintzs/jevify ≠ gulagala001/jevify ≠ uspraveen/Jevify; GH kushalpatil07/jevify 404; PAWS 0.580/ece 0.288 is the weak cell; smaller E4B slightly better OOD ECE than 26B-A4B; Hub jevify merged LoRA ships weights; bonzi Bonsai-8B v1 GGUF densify; Bonsai-1.7B v1; Bonsai-4B v1; WANLI-256 64.5% / 60.2% / 52.0% *theirs*; rank #4 / #5 / #6 of 6; JulesHuisman/jev-eval scaffolding / README SHA c356a584 (was empty e69de29b); JulesHuisman/jev-eval ≠ SivletLabs/jev-eval ≠ willkelly/jev-evaluation ≠ 4esv ≠ xxkuboxx ≠ onlyoneaman ≠ dayhaysoos/jevals; 7 bands 6/10 vs 40 bands 0/10; source receipts + confidence slider re-policy without re-inference; 32/32 synthetic is smoke not production; classify HF datasets across typed semantic dimensions; roadus2 watch misspelling; lock roadius2/ultra_laya; ultra_laya REVIEW defects; default branch claude/laya-jev-review-gg5ppo; XNLI EN 88.3% ECE 0.032 → RU 77.3% ECE 0.096; Δ −11.0 pp [−14.2,−7.8]; ECE +0.063; MASSIVE no detectable difference at n=600; confidence is function of p_max (r=1.000); pointer-not-generator 400 human-authored responses; proposed ≠ authorized; FewRel 160: Jev 85.0% vs lexical 13.125%; gated 100% (95/95) coverage 59.375%; J++ composable semantic computation language; judge-jev 0.5 still soft; 947 repos scored; A 273 / B 302 / C 372; LLM rubric ≠ benches; No benchmark winner is claimed; phishing: naive 62.6% vs regex 91.8%; 5-atomic + LR 95.0% *theirs*; AITuber tension ±15; README npm global; repo is Rust; git-confess code owns counting/blame/ratio; httpx exhibit 11% (13/119) *theirs*; 90d trend +12.40% vs random +12.75% vs BH +41.71%; 5m win rate 25%; Awesomejev 656 entries / 38,160 stars; tracker likes 64 (+4) lastModified UNCHANGED; Laya present; Blackwood ABSENT; Archer still promised_not_landed; do not reopen or amend PR #23/#24/#25/#26. Apply 0541 (`notes.md` §110): Blackwood tracker ABSENT; likes 2 gated manual; r = c - p_a; ECE 0.021; acc 0.807 vs warmup 0.746; calibration beyond ~500 tokens unmeasured; Independent primitive; 11.57s vs 54.10s · 4.67× · 120/128 *theirs*; default path is pretrained Gemma probs not trained RLCD head; GH Meanblock 404; lock leesk212/JEV-CPU; softmax over letter slots ≠ Noul; WANLI 0.741 vs openjev v2 0.77 *theirs*; 3-way NLI ≠ Noul; priority 0.464 = majority floor; banking77 contaminated; raw margins not probabilities; GH jev-haiku-benchmarking 404; do not distill Jev as teacher of record (they distilled Haiku); “0.9 is not one number”; ranking ≠ calibration; banking77 0.8–0.9 stated 0.86 actual 0.73 over-confident *theirs*; ≠ Praveenrajus/jev-bench ≠ fstandhartinger/jevbench; $0.0000153–$0.0000226 vs circulating $0.0004 (~20×); Score is 0..n-1 expectation not 0–1; Noul has no confidence field; TCP floor 198.8 ms; type reliability is not a reason to choose Jev (json_schema 5/5); gateway tax not one number; Function-only 5/8 vs hybrid 8/8; 4/8 without Jev; 8 designed cases not conversion lift; ≠ RadRebelSam/awesome-jev; 200-row pilot Jev 86.5% 173/200 vs Gemini Flash-Lite 86.0% 172/200 vs Pro 87.0% 174/200 *theirs*; not a ranking; NLI Tetris argmax P(entail)−P(contradict); 情緒測謊器; 1q 396ms / 30q 567ms; ±0.03; 33q $0.000045 vs Gemini ~5× slower ~60× cost *theirs*; ≠ realZachi/jevtest; 8-example Jev vs GPT-5.6 Sol ~64× cost 5.4× latency *theirs*; synthetic; no inference; ≠ JevBench v1.2 §78; Judged 3317 / listed 2560; Jev judges, code applies policy; catalog ≠ endorsement; APA “microsecond policy / zero hallucination” overclaim; Client-side quiz; pointer from held docs; scanned-PDF warn; CSP only api.typesafe.ai; Jev judges / agent reasons / user decides; selecting an option is not permission to implement; degraded fallback; pattern exact, judgement must clear floor; no matching pattern → no model call; not a correctness oracle; $0.00022 vs chat $0.00306 *theirs*; Spec vs artifact remainder; treating 0.85 as 85% / minProbability hard-gate as Harbor; VERIFY acquires discriminating evidence, never same-pool confidence-only rescoring; fast/full/max are ceilings not sizes; Solar writes, Jev chooses NEXT ACTION; SemIf 2186★ (+20 vs §109 2166); jevlike 1038★ (+7 vs 1031); TypeAR 14★ flat; AnotiaWang 96★ (+1 vs 95); yibie/awesome-jev 490★; Laya likes 802 (was 783); tracker likes 64 flat, lastModified UNCHANGED; do not reopen or amend PR #23/#24/#25/#26/#27. Apply 0743 (`notes.md` §113): Hourly 0743 uniqueness lock: Heman10x-NGU/Verdict-open-jev ≠ Heman10x-NGU/openJev-verdict-2.0; TF-IDF + LogReg ECE 0.0207 vs Jev 0.1440; Verdict-open-jev 48.07% vs Jev 90.80%; abstention combined recall 10.00%; p50 35.58 ms; K=25 (maximum capacity) 72.00%; 0.85 coverage 84.60% selective risk 1.18%; 26.1× faster than standard Qwen JSON generation; Jevify 90.0% / 167 ms CUDA graphs disabled; Finding 1: Brier on stated confidence alone is a trap; grpo_rlcr 0.78 / ECE 0.084; reliability 0.007 but resolution 0.000; 27 900 schema-driven decisions; 13 600 / 13 600 questions; candidate mass min 0.99999624; 22 configs · 166,054 rows · 4 calibration-gold; sha a39eba3f; Student B MAE 0.148 / Pearson 0.836 / 86.0%; pngwn/open-jev-laya-bench README 404; sha 9f69c742 likes 2; HDFS 0.9933 (745/750) / retain 0.0084; BGL ERROR/FATAL protection 1.0000; 2,479 / 2,500 HDFS uncertain; cache hit 0.9648 (2412/2500); $0.153936 estimated; E2 recomputes from saved probabilities; Space sha eda59e0a; MASSIVE English 0.783 / Khmer 0.033 / Hindi 0.133; 40–48 rows too small to ship T; T never changes argmax; siren2345/jev-single-decode-transformers ≠ siren2345/jev-single-decode; Split Transformers experiment from llama.cpp runtime; tanayvasishtha/jev-lab ≠ dairui1/jev-lab ≠ BrendanH18/jev-lab ≠ yibie/laya-jev-lab; Four experiments stress-testing TypeSafe's Jev: calibration, bundle bias, label bias, and ensembling; second pass must be $0.00 from cache; The pages never call Jev; Gemma 4 31B 77.0% / Jev 1.13.0 61.4% / Laya 322M 0.0%; restriction state 95.0% against 84.4%; None of the systems are particularly good at knowing when to stop and ask; They skip the question and call a tool directly; 100% schema pass; six-field joint 48.8% vs 72.8%; ywchiu/jev_benchmark ≠ Running-Dolphins/jev-bench ≠ Praveenrajus/jev-bench; ACT / REVIEW / FALLBACK; A provider failure, timeout, malformed output, or missing answer is **not** a policy outcome; confidence is descriptive provider output, not a substitute for probability; Quality denominators include only valid scored answers; an exact halfway tie chooses the lower level; aiwithenoch/Jev-Skill ≠ simplosophy/jev-skill ≠ laguagu/jev-skills; The local path does not claim to turn a smaller checkpoint into Jev; Low support becomes decision: "review"; MIT-0 SPDX NOASSERTION; current-llm; 结构兼容,不是 Jev 模型能力; altryne/jevify ≠ Mintzs/jevify ≠ gulagala001/jevify ≠ uspraveen/Jevify; Find where Jev belongs. Design the questions. Measure the difference; TypeAR-AI/TypeAR 301 → TypeLLM/TypeLLM; TypeLLM/TypeLLM 16★; SemIf 2241★ (+34 vs §111 2207); jevlike 1051★ (+8 vs 1043); AnotiaWang 98★ (+1 vs 97); yibie/awesome-jev 525★ (+19 vs 506); Laya likes 864 (was 822); tracker likes 67 (+3 vs 64); lastModified UNCHANGED `2026-09-20T04:29:16.000Z`; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#32/#33; notes.md §113 Apply 0646 (`notes.md` §111): Calibration is not alpha; NO CURRENT ALPHA CANDIDATE; ΔR² approximately +0.00084; Brier 0.2131387; ECE 0.0421875; Adding Jev probability to deterministic volatility improved Brier by only 1.4058e-05; default 0.5 keeps zero non pinned; keepResult median 0.14 to 0.17; keepCall median 0.28 to 0.35; usable range is about 0.10 to 0.25; 7.8% to 57.9%; judges results it never sees; task-finish eval not built yet; $0.002 per compaction; slavadubrov/sgr-judge-bench ≠ slavadubrov/jev-judge-bench; Jev 108/120 $0.083 0.34 s; Luna SGR 114/120; paired Jev accuracy-difference intervals include zero; not evidence of equivalence; GLM SGR 26/120 93 format failures; Terra-planned Jev hybrid 55/120; rule-based by default, optionally Jev-backed; empty README; missing key cannot break the experience; prefill plus exactly one decode; softmax over A/B/C ≠ Noul; BBQ 9,053/10,000 (90.53%); ECE 0.0890; Mean confidence 0.9943; overconfident; score and noul not implemented; DGUI 12 rows (was 6); INSTRUCT 119 rows likes 2; encode the state once, decide everything in parallel; 0.740 accuracy against a 0.508 majority; ECE 0.047; fine-tune's advantage ends where its 384-token training data does; jasonkneen/open-jev ≠ pngwn/open-jev; same sha d41dc3cd; Space does not call Jev; recomputes routing from saved probabilities; 200-case Jev 97.0% / 100.0% / 95.0% / MAE 9.22; synthetic repository benchmark; Jev evaluations are advisory; YehuiTang0316/jev-nlgrep ≠ Bentlybro/jevgrep ≠ can1357/jegrep ≠ uehaj/jev-semgrep; default threshold 0.8 still soft; 40-line windows cannot prove whole function; token-native sequential start/end Choice; Gemini/Haiku stubs not configured yet; handful of hand-written examples, not a benchmark; Jev judged exactly what it was given; laguagu/jev-skills ≠ laguagu/jev-evidence-lab ≠ Pleo2/awesome-jev-agent-skills; contract_passed is not a claim of guaranteed factual truth; Wilson lower bound 0.85 floor; fixture mode no savings claim; SemIf 2207★ (+21 vs §110 2186); jevlike 1043★ (+5 vs 1038); TypeAR 15★ (+1 vs 14); AnotiaWang 97★ (+1 vs 96); yibie/awesome-jev 506★ (+16 vs 490); Laya likes 822 (was 802); tracker likes 64 flat, lastModified UNCHANGED; do not reopen or amend PR #23/#24/#25/#26/#27/#28.) | Apply 0843 (`notes.md` §114): **Hysteresis is policy** / **Instruct-tuning honesty collapse** / **Equal-width ≠ quantile ECE** / **Calibration does not compose** / **Ranking ≠ calibration** / **Qwen2.5 ≠ Archer** / **Deferred Crispification** / **g0runmezadam IS tunahansahin897**. **Category error** (Jev vs GPT-5.6 bakeoffs)** / **Skill-issue thesis** / **opt for DeBERTa and ModernBERT ones** / **multimodal ZS perception front-end** / **softmax/ZS ≠ Noul**. people who compare Jev against GPT-5.6 has never fine-tuned BERTForXYZ for living and it shows; zero shot classifiers; scale them as much as decoder only models; many problems solved with LLMs could have been solved with them, it was a skill issue; opt for DeBERTa and ModernBERT ones; BERTForXYZ → DeBERTa → ModernBERT; Jev vs GPT-5.6 bakeoffs are a category error; encoder / ZS classifiers; institutional HF voice; quote *theirs*; do not invent accuracy numbers; softmax/ZS scores still ≠ calibrated Noul; soft scores ≠ hard gates; @mervenoyann; likes 421 / 189; impressions 35498 / 9613; multimodal image<>text ZS as perception front-end; hf:MoritzLaurer/deberta-v3-large-zeroshot-v2.0 likes 139; hf:MoritzLaurer/ModernBERT-large-zeroshot-v2.0 likes 72; Bart, bert, deberta, modernbert, these are all LLMs; Maziyar quoted; Jev is exemplar not the mandate; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29. Hourly 1441 densify: vLLM NVIDIA + MLX Apple Silicon on razorback16/openjev; STE README rewrite; serving-port densify; Codiv hosted free endpoint; zhengxuyu/litjev Probabilities are not calibrated by default; Qwen/Qwen3.8-27B ≠ Archer; AndyInQtr/laya-coreai CoreML serving substrate ≠ calibrated replica; AndyInQtr/laya-coreai ≠ mizorewww/laya-coreml; LoRA ≠ RLCD replica; 27B still in progress.)| Entropy as allocator (low / medium / high) | Typed low+medium decisions → System One marginals; high-entropy synthesis → frontier decoder. Product rhetoric, not a meter. **Hypothesis** | `references/judgment-class.md` | | Formal / semi-formal (proof vs judgment) | Sensor vs constraint vs searchlight; Alloy vs Apalache; DST trio; TOCTOU-of-Noul, AI×FM. Skills→oxlint: AST/precheck compose with remainder judgment without hard-gating a Noul as a proof. PR attention ≠ correctness (anti-soundness-theater). Engine owns truth / Jev owns judgment (Stockfish+Jev chess coach). Capability kernel: type-safe ≠ correct; irreversible behind threshold AND human. Eval integrity: check the instrument, not just the score (`dinostomp jev` tests a question like an if-statement). Effect contracts, not surface tokens (construct-auto-classifier; privilege ≠ verdict). **Jev supplies evidence, code owns authority** (actiongate-jev; a positive score never overrides a deterministic security failure). **Turnstile clone:** deterministic policy + Jev remainder + replay (evidence ≠ authority). **Type-safe ≠ correct as jaggedness receipts** (atlas; schema-valid ≠ picked-right). **TLA+ compose with a Jev-class oracle:** never confidently wrong; escalate is the safety valve (jev-labs; inverse of soundness theater is hard-gating without escalate). **Hourly 1347:** typed-gate band [0.40,0.60] is refusal; pi-jev-gate fail-closed; choice is the verdict; jev-calibration-arena never acts. **Hourly 1441:** typesafe_agent_gates 27/27 / 31/31; EpicEric/safe-sh static remainder; pastepilot Confirm before act; typesafe-scheduler-diagnostics advisory; choxos/jevchess engine owns truth. **Advance/coverage ledger:** Jev answers questions; SEAL answers whether the world may change (coverage.path auto|code|human|escalate; mint ≠ product brain). **Conflict ≠ ignorance:** Noul collapses both; named Choice escape separates (typed-evaluation-collapse; schema-as-interface). **Sentence-as-rule lint:** ast-grep matcher silent × Jev `ask:` loud (mizchi/jev-lint is mizchi/jevlint rename; ≠ huntedman/JevLint). **Whole-repo intent:** VERIFIED/VIOLATION/UNKNOWN; empty search ≠ proof (jev-intent-review). **Decision-as-assert:** meaning Noul vs exact `toContain`; ambiguous band fails both polarities (jevtest; 0.85 still soft; record/replay; hard-gating a matcher as a merge seal is soundness theater). **Authorship named escape:** `human`/`ai_generated`/`uncertain`; not courtroom evidence. **Empty findings as approval, or auto-promoting an agent-written workflow, is the same theater** (stanley-code `notChecked`; Soft Noul ≠ hard safety). **feelings `.feels()`** default 0.5 never rounded; **Essentiel-Jev never authority**; **enzo-mcp UNKNOWN**; **pigeonhole OTHER skip**. **Hourly 1241:** ZHUBoer/ego-jev reserved `__none__`; runWorkflow completed ≠ success; ORIGIN pause-if-no-Jev; jevbrain AUTO_ACT is not a Noul. **SIGNAL §93:** Cache hit ≠ correctness; score never auto-accepts. **SIGNAL §94:** guidance ≠ hook; unofficial ≠ TypeSafe; hosted bootstrap ≠ silent TypeSafe; Nemotron ≠ TypeSafe Jev; not a calibrated replacement; Router/OOD confidence. **Hourly 1541:** jevguard calibrator/cache/escape; jev-ci-selector CI shadow mode; one-dollar-tahoe TypeSafe Jev defense eval (rh-guard owns the gate cousin). **Hourly 1740:** Decision Graph Protocol frame→assess→commit; app retains permissions/effects; Jev-first assessor-neutral; guarded commit / receipt/next frame; assessment batching; hard-gating DGP as safety theater; numerous-com/dgp ≠ TypeSafe official. **Hourly 1843:** thresholds derived from costs not hard-coded; cost-sensitive decision theory × System One probabilities → control flow; hard-gating 0.038 as safety theater; "guaranteeing" calibration is theater; deeper integrity fold is rh-guard. **Hourly 1943:** noul_threshold 0.5 decoder not a proof; no shipped rule has severity error; 0 of 157 false invalidations; questions/plans/directives are not evidence; IncompatibleSchemaError lists every bad property **Hourly 2041:** resume-screening bias audit methodology; Plan/PRD panel → code-owned pass|review|block; cost-aware multi-model routing/escalation; frozen-protocol zero-shot bench; context-window admission control; typed decision control plane; receipt ≠ authorization; live 15-dim typed rubric re-score per pause; adversarial pre-registered Jev eval; confidence does not track ignorance; provider-neutral Elixir/BEAM Noul/Choice/Score SDK **Hourly 2145:** question-linting of Jev questions themselves; nine jaggedness rules, no API key, no labelled data; static lint ≠ measured separation; yodablocks/jevq ≠ tenbin ≠ JevLint ≠ commitjev; open-weights Laya as class exemplar (binding); Nx/Bumblebee runtime; host chooses backend; ChristianAlexander/laya_ex ≠ system_one_sdk ≠ dannote/jev ≠ NandhaKishorM/laya; on-chain/edge Laya deploy; parity_verified stays false; model output never grants Tx; humandebri/IC-Laya ≠ laya_ex; auditable weekend replica; Jev outputs never used for training; unpaired 0.577 vs 0.727; agilabs-ai/jev48 ≠ JevBench ≠ Mapika/decider; adversarial dual-judge / framing attack surface; comparative framing is the usable judgment; prior injection crowds out evidence; copyleftdev/ember ≠ ember.js; Laya specialist fine-tune pipeline; training still GPU-pending; PIXELZX0/XERON ≠ convaiinnovations/laya; Hub Laya replica drop; daliborsb/laya ≠ convaiinnovations/laya ≠ NandhaKishorM/laya; System One student distillation corpus; gold is programmatic; teacher is closed-API clone; MagaBitmex/jev-4b-distill-data ≠ missing student checkpoint; non-LLM VIN System One; planning depth not chat; lewislululu/jevon ≠ douglance/jevon; source-bound evidence checks; local quote mismatch needs no API; exit 0 ≠ claim truth; WaynezProg/jev-kit ≠ jonathanavis96/jev-kit (Airlock) ≠ jev-use ≠ jev-mcp **Hourly 2246:** independent System One evidence catalog; 19 reviewed records; scores not one leaderboard; no external record currently reproduced; TokenTrim no-Jev matched hybrid 62.4%; reachjalil/system-one-bench ≠ mallahyari/system-one-benchmark; 21 tasks · 134 items · 208 questions; scenes from public GitHub contracts, not production logs; SivletLabs/jev-eval ≠ willkelly/jev-evaluation ≠ 4esv/jev-eval ≠ xxkuboxx/jev-eval ≠ onlyoneaman/jev-eval ≠ dayhaysoos/jevals; option isolation (sibling-blind); permutation-equivariant; Hub OWNER not published; nafisazizir/hev ≠ jaredpalmer/kev; frozen local LLM logits, no trained decision head; residual-head 9,222-param decreased 73/96→67/96; confidence = 1−normalized entropy, not P(correct); yuki-oshio/mini-jev ≠ r-ms/mini-jev; Jev classifier as autoregressive next-token predictor; ChatJev-style soundness theater; erik-dunteman/ChatJev ≠ dannote/jev ≠ jev-gpt; calibrated decision head × AlphaProof value head; implementation-layer isomorphism, semantic difference; timeout = censoring; do not launder Noul as proof; parallel rank-prediction vs serial selection; independent questions can conflict; zzzzzec/jevsort ≠ keltokhy/jsort; curated open System One ecosystem catalog; rupeshpoojary9/awesome-open-system-one ≠ AnotiaWang/awesome-jev; arXiv paper radar with Jev relevance scoring; ranking ≠ calibration / 0.5 still soft; fail-open failed evals not marked seen **Hourly 2340:** train calibrated ~27M from scratch; typed Q→prob dist / one forward pass / no LLM decode; hyusi2003/MiniSystemOne ≠ Colvin0315/MiniSystemOne; description-only stub / size 5; ESCI hard probe fails four of six; jev_bool ECE 0.242 inversion 0.255; do not re-fold §60 six-gates as new; jobbyjev one-request-per-company from batch-size result; find/design/evaluate TypeSafe Jev decision loops; karanb192/jev-architect ≠ samtay32/jev-system-architect; Jairik/jev-distiller size 1; distill-Jev UI stub / do not distill Jev as teacher of record; post-launch scored use-case map / Jev self-scores then human curation; licensedsaucer9-web/jev-opportunities; Jev-inize a use case into classifier/router; gavinHuang/jevinize → simple-jev not TypeSafe; featherless-ai/simple-jev; compare saved decisions / same label can still change the branch; VihaanAgarwal/jev-diff ≠ Saik0s/diffusiongemma-jev-macos; not tested with a live Jev API key; constrained logprob + temp/Platt ≠ Noul; OpenJevPro pastes openjev-sglang JevBench as own; zhangcy122/OpenJevPro ≠ IamBusy/OpenJev ≠ ekzhang/openjev-sglang; PolyForm Noncommercial; SmolLM-135M / sub-70ms / 0 output tokens; demo P(True) 0.5052 / Choice conf 0.2872 / Score conf 0.0055; README claims MIT / GitHub license null / no LICENSE file; patelvishwa112/jev-system-one-rlcd ≠ arnabgho/rlcd-lite ≠ blackwood-rlcd; source-backed Awesome Jev radar / 306+ commit-pinned; logicrw/awesome-jev-projects ≠ AnotiaWang/awesome-jev ≠ yibie/awesome-jev ≠ cobanov/awesome-jev ≠ rupeshpoojary9/awesome-open-system-one; auto GitHub sync / Issue-only submissions; hashed n-gram encoder / rival-aware attention; olanotolu/jevbetter vs jevlike starter; synthetic hard menus top-1 0.916 vs 0.873 / ECE 0.0182 vs 0.0367 / 40 vs 4608 menus/sec; shuffled-context control 0.335 | `references/formal-methods.md` (one-screen: `references/formal-semi-formal.md`) Apply 0042 (`notes.md` §105): structured probability readouts; distribution > argmax; Noul 0.5 midpoint; score is expectation not integer; bare HTTP not SDK; Arohtea/jev-readout; Jev-style Choice/Score/Noul from ordinary models; optional DSH plugin; schema-valid ≠ calibrated; gulagala001/jevify ≠ Mintzs/jevify; Laya RLCD benchmark; 40.3% below constant-answer; open-weight measurement; mourad-ghafiri/laya-rlcd-benchmark ≠ yibie/laya-jev-lab; cheap fail-open semantic edge; second signal not sole; FastLoopError catch; SupremeDreamZ/jev-fastloop ≠ jev-ultrafast; asking more questions in one call; 0.980 at every N; nearly not fully deterministic; TheWebDevel/jev-fanout; Qwen3-VL perception + Jev decisions train RL; 0 model calls at deployment; VLM alone 1.7 vs +Jev 4.4; harneet2512/reflexrl ≠ khordoo/jev-reflex-autonomy-lab; independent Jev API vs Laya; cascade 0.60 matches 78% at 1.8×; noul facts not judgements; yibie/laya-jev-lab ≠ dairui1/jev-lab ≠ BrendanH18/jev-lab; GLiNER vs GLiFormer vs Laya vs Jev; extractors ≠ decision engines; Laya dict-instructions collapse 58.3%; umstek/zero-shot-ie-bench; decisions-per-minute & cost; 204 moves vs 73; throughput not intelligence; angelgalvisc/snake-arena-jev-vs-llms ≠ vtrivedy/jev-plays-games; behavioral contracts; pin expectations eval upgrades; raw 0.94 is not a release; sathariels/jevcheck ≠ dayhaysoos/jevals ≠ SivletLabs/jev-eval; evidence-linked dependency upgrade; Jev never generates filenames; no_direct_evidence ≠ safe to merge; GaneshVG18/upgrade-radar ≠ LYchoon/paper-radar-jev; discography theme/mood/complexity; five atomic questions one call; lirantal/discoprint / Apply 0145: hard budget filter before Jev; Jev never asked to perform budget arithmetic; Jev judges the next state, XState enforces transitions; do not reopen or amend PR #23. / Apply 0243: Benchmark-driven Jev router and judge; cheap alone is not success; Jev does not write, sum prices, or claim accuracy %; Sol 94.2 / Luna 83.9 / Jev path 89.7; 19.2% Sol / 62.3% cost save / 4.5pp miss of 2pp non-inferiority; p50 latency worse than Sol due to routing overhead; erendikmenn/jev-llm-router-benchmark ≠ jev-rag-benchmark ≠ ryantsai/jev-llm-router; Express + node:sqlite; mock and Jev decision engines; previous_ticket_count >= 3 is code; MIN_CONFIDENCE 0.6 still soft; substring false positives; aesaganda/jev-ticket-router ≠ SarathChandraBellam/jev-vs-llm-ticket-router; Universal Figure & Diagram Router; confidence ≥ 0.85 hard-gate is theater; generative AI banned from scientific plots; six visual branches; human-labeled (state, question, label); 166,054 rows / 22 configs; soft_label for human uncertainty; Praveenrajus/jev-bench ≠ fstandhartinger/jevbench; ternary bonsai System One GGUF; Hub does not ship weights; 100/100 easy T/F is not Harbor; label_mass ≠ correctness; stock llama.cpp Q2_0 silently gibberish; NicolaiMTLassen/open-bonzi-jev ≠ NicolaiLassen; transformers.js DeBERTa ONNX; temperature 1.05; AutoModel from_pretrained works; onnx-community/open-jev-deberta-v3-large-ONNX ≠ system-one-qwen3.5-4b-scorer-ONNX; 107★ densify; GH 151M vs README 149.6M; PR #1 now closed unmerged; do not re-fold §71 claim-audit as a beat; typed decisions, RLCD, confidence-gated routing; structured ≠ correct; mock not live API; 26 tests; wjdjdakf17/jev-study ≠ baekenough/jev-study / Apply 0345 (`notes.md` §108): **Open reproduction densifies / measurement densifies PRIMARY** (bonzi-27b-v2 / ternary-8b / 27b-v1 GGUF family densify; WANLI-256 74.6% / 65.2% / 71.1% *theirs*; Bonsai 1 27B Q1_0 runs on stock llama.cpp; ternary still needs PrismML fork; hf:heman10x/openJev-verdict-2.0 twin tokenizer-only; OpenJev Vision image classification + uncertainty; CLEVR-4 held-out joint 0%; hfdataset:IamBusy/OpenJev-Vision-Research-v0.1 12,832; 294,912 derived targets not independent samples; Laya multilingual ONNX WebGPU typed-decisions port; 63/63 selected answers / 5.1e-4 CPU / 1.2e-2 WebGPU; UpHash-Network/mini-jev is yuki-oshio transfer; jev-injection-bench 11,900 labelled prompts; Jev best ranking / Haiku better ECE 0.021 vs 0.058; 0.5–0.9 band is where Jev's numbers do not mean what they say; Prompt wording moves panic 28%; manojlds/jev-dspy-bench ≠ dspachos/jev-dspy ≠ jmanhype/jev-dspy-lab; Jev agreement is similarity, never ground truth; no aggregate quality grade or merge gate; AbstentionBench-on-Jev rank 1 of 20 vs 2025 field; question-asymmetry; forward-looking 0.465 never extreme; openkev calibration layer not a runtime; ECE vs coverage independent; select_threshold returns inf; escalation catches uncertainty not ignorance; misakaikato/openkev ≠ jaredpalmer/kev; pdf-race Docling→Jev vs Gemini; parser owns the wall clock; 12/12 tie is a tie; titles selected not generated; flopcheck 16 calibrated tweet judgments; mechanical tells in code; ZeroX-01/jev-atlas ≠ Zaious/jev-capability-atlas ≠ gorock007/jev-atlas; Laya calibration lab Gradio MCP; T never changes argmax; confidence ≠ top-label p; easy probe set refused; 40–48 rows too small to ship T; do not reopen or amend PR #23 or #24 or #25. Apply 0439 (`notes.md` §109): Gemma-4 26B-A4B jevify classification+calibration; LoRA adapter twin not independent eval; Gemma-4 E4B jevify; E4B LoRA stub card; kushalpatil/jevify-gemma4 ≠ Mintzs/jevify ≠ gulagala001/jevify ≠ uspraveen/Jevify; GH kushalpatil07/jevify 404; PAWS 0.580/ece 0.288 is the weak cell; smaller E4B slightly better OOD ECE than 26B-A4B; Hub jevify merged LoRA ships weights; bonzi Bonsai-8B v1 GGUF densify; Bonsai-1.7B v1; Bonsai-4B v1; WANLI-256 64.5% / 60.2% / 52.0% *theirs*; rank #4 / #5 / #6 of 6; JulesHuisman/jev-eval scaffolding / README SHA c356a584 (was empty e69de29b); JulesHuisman/jev-eval ≠ SivletLabs/jev-eval ≠ willkelly/jev-evaluation ≠ 4esv ≠ xxkuboxx ≠ onlyoneaman ≠ dayhaysoos/jevals; 7 bands 6/10 vs 40 bands 0/10; source receipts + confidence slider re-policy without re-inference; 32/32 synthetic is smoke not production; classify HF datasets across typed semantic dimensions; roadus2 watch misspelling; lock roadius2/ultra_laya; ultra_laya REVIEW defects; default branch claude/laya-jev-review-gg5ppo; XNLI EN 88.3% ECE 0.032 → RU 77.3% ECE 0.096; Δ −11.0 pp [−14.2,−7.8]; ECE +0.063; MASSIVE no detectable difference at n=600; confidence is function of p_max (r=1.000); pointer-not-generator 400 human-authored responses; proposed ≠ authorized; FewRel 160: Jev 85.0% vs lexical 13.125%; gated 100% (95/95) coverage 59.375%; J++ composable semantic computation language; judge-jev 0.5 still soft; 947 repos scored; A 273 / B 302 / C 372; LLM rubric ≠ benches; No benchmark winner is claimed; phishing: naive 62.6% vs regex 91.8%; 5-atomic + LR 95.0% *theirs*; AITuber tension ±15; README npm global; repo is Rust; git-confess code owns counting/blame/ratio; httpx exhibit 11% (13/119) *theirs*; 90d trend +12.40% vs random +12.75% vs BH +41.71%; 5m win rate 25%; Awesomejev 656 entries / 38,160 stars; tracker likes 64 (+4) lastModified UNCHANGED; Laya present; Blackwood ABSENT; Archer still promised_not_landed; do not reopen or amend PR #23/#24/#25/#26. Apply 0541 (`notes.md` §110): Blackwood tracker ABSENT; likes 2 gated manual; r = c - p_a; ECE 0.021; acc 0.807 vs warmup 0.746; calibration beyond ~500 tokens unmeasured; Independent primitive; 11.57s vs 54.10s · 4.67× · 120/128 *theirs*; default path is pretrained Gemma probs not trained RLCD head; GH Meanblock 404; lock leesk212/JEV-CPU; softmax over letter slots ≠ Noul; WANLI 0.741 vs openjev v2 0.77 *theirs*; 3-way NLI ≠ Noul; priority 0.464 = majority floor; banking77 contaminated; raw margins not probabilities; GH jev-haiku-benchmarking 404; do not distill Jev as teacher of record (they distilled Haiku); “0.9 is not one number”; ranking ≠ calibration; banking77 0.8–0.9 stated 0.86 actual 0.73 over-confident *theirs*; ≠ Praveenrajus/jev-bench ≠ fstandhartinger/jevbench; $0.0000153–$0.0000226 vs circulating $0.0004 (~20×); Score is 0..n-1 expectation not 0–1; Noul has no confidence field; TCP floor 198.8 ms; type reliability is not a reason to choose Jev (json_schema 5/5); gateway tax not one number; Function-only 5/8 vs hybrid 8/8; 4/8 without Jev; 8 designed cases not conversion lift; ≠ RadRebelSam/awesome-jev; 200-row pilot Jev 86.5% 173/200 vs Gemini Flash-Lite 86.0% 172/200 vs Pro 87.0% 174/200 *theirs*; not a ranking; NLI Tetris argmax P(entail)−P(contradict); 情緒測謊器; 1q 396ms / 30q 567ms; ±0.03; 33q $0.000045 vs Gemini ~5× slower ~60× cost *theirs*; ≠ realZachi/jevtest; 8-example Jev vs GPT-5.6 Sol ~64× cost 5.4× latency *theirs*; synthetic; no inference; ≠ JevBench v1.2 §78; Judged 3317 / listed 2560; Jev judges, code applies policy; catalog ≠ endorsement; APA “microsecond policy / zero hallucination” overclaim; Client-side quiz; pointer from held docs; scanned-PDF warn; CSP only api.typesafe.ai; Jev judges / agent reasons / user decides; selecting an option is not permission to implement; degraded fallback; pattern exact, judgement must clear floor; no matching pattern → no model call; not a correctness oracle; $0.00022 vs chat $0.00306 *theirs*; Spec vs artifact remainder; treating 0.85 as 85% / minProbability hard-gate as Harbor; VERIFY acquires discriminating evidence, never same-pool confidence-only rescoring; fast/full/max are ceilings not sizes; Solar writes, Jev chooses NEXT ACTION; SemIf 2186★ (+20 vs §109 2166); jevlike 1038★ (+7 vs 1031); TypeAR 14★ flat; AnotiaWang 96★ (+1 vs 95); yibie/awesome-jev 490★; Laya likes 802 (was 783); tracker likes 64 flat, lastModified UNCHANGED; do not reopen or amend PR #23/#24/#25/#26/#27. Apply 0743 (`notes.md` §113): Hourly 0743 uniqueness lock: Heman10x-NGU/Verdict-open-jev ≠ Heman10x-NGU/openJev-verdict-2.0; TF-IDF + LogReg ECE 0.0207 vs Jev 0.1440; Verdict-open-jev 48.07% vs Jev 90.80%; abstention combined recall 10.00%; p50 35.58 ms; K=25 (maximum capacity) 72.00%; 0.85 coverage 84.60% selective risk 1.18%; 26.1× faster than standard Qwen JSON generation; Jevify 90.0% / 167 ms CUDA graphs disabled; Finding 1: Brier on stated confidence alone is a trap; grpo_rlcr 0.78 / ECE 0.084; reliability 0.007 but resolution 0.000; 27 900 schema-driven decisions; 13 600 / 13 600 questions; candidate mass min 0.99999624; 22 configs · 166,054 rows · 4 calibration-gold; sha a39eba3f; Student B MAE 0.148 / Pearson 0.836 / 86.0%; pngwn/open-jev-laya-bench README 404; sha 9f69c742 likes 2; HDFS 0.9933 (745/750) / retain 0.0084; BGL ERROR/FATAL protection 1.0000; 2,479 / 2,500 HDFS uncertain; cache hit 0.9648 (2412/2500); $0.153936 estimated; E2 recomputes from saved probabilities; Space sha eda59e0a; MASSIVE English 0.783 / Khmer 0.033 / Hindi 0.133; 40–48 rows too small to ship T; T never changes argmax; siren2345/jev-single-decode-transformers ≠ siren2345/jev-single-decode; Split Transformers experiment from llama.cpp runtime; tanayvasishtha/jev-lab ≠ dairui1/jev-lab ≠ BrendanH18/jev-lab ≠ yibie/laya-jev-lab; Four experiments stress-testing TypeSafe's Jev: calibration, bundle bias, label bias, and ensembling; second pass must be $0.00 from cache; The pages never call Jev; Gemma 4 31B 77.0% / Jev 1.13.0 61.4% / Laya 322M 0.0%; restriction state 95.0% against 84.4%; None of the systems are particularly good at knowing when to stop and ask; They skip the question and call a tool directly; 100% schema pass; six-field joint 48.8% vs 72.8%; ywchiu/jev_benchmark ≠ Running-Dolphins/jev-bench ≠ Praveenrajus/jev-bench; ACT / REVIEW / FALLBACK; A provider failure, timeout, malformed output, or missing answer is **not** a policy outcome; confidence is descriptive provider output, not a substitute for probability; Quality denominators include only valid scored answers; an exact halfway tie chooses the lower level; aiwithenoch/Jev-Skill ≠ simplosophy/jev-skill ≠ laguagu/jev-skills; The local path does not claim to turn a smaller checkpoint into Jev; Low support becomes decision: "review"; MIT-0 SPDX NOASSERTION; current-llm; 结构兼容,不是 Jev 模型能力; altryne/jevify ≠ Mintzs/jevify ≠ gulagala001/jevify ≠ uspraveen/Jevify; Find where Jev belongs. Design the questions. Measure the difference; TypeAR-AI/TypeAR 301 → TypeLLM/TypeLLM; TypeLLM/TypeLLM 16★; SemIf 2241★ (+34 vs §111 2207); jevlike 1051★ (+8 vs 1043); AnotiaWang 98★ (+1 vs 97); yibie/awesome-jev 525★ (+19 vs 506); Laya likes 864 (was 822); tracker likes 67 (+3 vs 64); lastModified UNCHANGED `2026-09-20T04:29:16.000Z`; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#32/#33; notes.md §113 Apply 0646 (`notes.md` §111): Calibration is not alpha; NO CURRENT ALPHA CANDIDATE; ΔR² approximately +0.00084; Brier 0.2131387; ECE 0.0421875; Adding Jev probability to deterministic volatility improved Brier by only 1.4058e-05; default 0.5 keeps zero non pinned; keepResult median 0.14 to 0.17; keepCall median 0.28 to 0.35; usable range is about 0.10 to 0.25; 7.8% to 57.9%; judges results it never sees; task-finish eval not built yet; $0.002 per compaction; slavadubrov/sgr-judge-bench ≠ slavadubrov/jev-judge-bench; Jev 108/120 $0.083 0.34 s; Luna SGR 114/120; paired Jev accuracy-difference intervals include zero; not evidence of equivalence; GLM SGR 26/120 93 format failures; Terra-planned Jev hybrid 55/120; rule-based by default, optionally Jev-backed; empty README; missing key cannot break the experience; prefill plus exactly one decode; softmax over A/B/C ≠ Noul; BBQ 9,053/10,000 (90.53%); ECE 0.0890; Mean confidence 0.9943; overconfident; score and noul not implemented; DGUI 12 rows (was 6); INSTRUCT 119 rows likes 2; encode the state once, decide everything in parallel; 0.740 accuracy against a 0.508 majority; ECE 0.047; fine-tune's advantage ends where its 384-token training data does; jasonkneen/open-jev ≠ pngwn/open-jev; same sha d41dc3cd; Space does not call Jev; recomputes routing from saved probabilities; 200-case Jev 97.0% / 100.0% / 95.0% / MAE 9.22; synthetic repository benchmark; Jev evaluations are advisory; YehuiTang0316/jev-nlgrep ≠ Bentlybro/jevgrep ≠ can1357/jegrep ≠ uehaj/jev-semgrep; default threshold 0.8 still soft; 40-line windows cannot prove whole function; token-native sequential start/end Choice; Gemini/Haiku stubs not configured yet; handful of hand-written examples, not a benchmark; Jev judged exactly what it was given; laguagu/jev-skills ≠ laguagu/jev-evidence-lab ≠ Pleo2/awesome-jev-agent-skills; contract_passed is not a claim of guaranteed factual truth; Wilson lower bound 0.85 floor; fixture mode no savings claim; SemIf 2207★ (+21 vs §110 2186); jevlike 1043★ (+5 vs 1038); TypeAR 15★ (+1 vs 14); AnotiaWang 97★ (+1 vs 96); yibie/awesome-jev 506★ (+16 vs 490); Laya likes 822 (was 802); tracker likes 64 flat, lastModified UNCHANGED; do not reopen or amend PR #23/#24/#25/#26/#27/#28.) | | Mixed architecture (judgment model + LLM) | Provider judges, LLM writes, code owns control; not a stack replacement. Advisory sidecar never changes host routing. Dual-process: S1 decides, S2 generates (routing accuracy unmeasured); **Harbor-shaped cousin:** decide→policy→LLM leftover (shared Answer schema; jev vs gen-json vs gen-logprob; Noul 0.5 never rounded). **Closed-vote CU:** no planner LLM; code builds options, Jev only picks (JevOnly). **Host-owned product:** app retains handlers/permissions; Jev over live typed actions (waymode). S1 specialists + S2 coordinator is the same split (description-only greenfield). Internals ≠ FSM; placement is a **component node**. Browser-use strength = DOM-as-text + speculative fan-out, not vision. Fail polarity is per act: skip-wake fail-open vs merge-gate BLOCK fail-closed; OMP/pi acceptance+route **fail-open** (contrast pi-jev-approver fail-closed). OMP prompt suppression: operator owns the bar; not a sandbox (omp-greenlight). Skill-broker outline: Jev never grants access. Hybrid local decide + remote fill; `DONE` ≠ verified success. Specialist computer-use: plan ≠ execute, dry-run default (Cua-S1; not TypeSafe Jev). Judgment as a language primitive (Ruby `almost_certain?`/`pick`/`rate`). Decision-native RAG: retrieve wide → decide → evidence set → LLM. Classify-first MCP (topology A): content to the judge without entering main agent context first. Generative UI: model decides, compiler emits. Draft-gate silence ≠ safer (heartbeat). Living class-pattern atlas (not a 342-title dump). Stagehand experimental Jev: pick-and-copy extract + act tree + observe/cache-check; LLM fallback; draft stack. Public judgment wall (six parallel questions; policy-in-code; cost-to-1M). PR attention ≠ correctness. Session-sticky first-prompt route (fail-closed fallback). Capability kernel (LLM ring 3 / Interlock ring 0; secrets never in agent; Jev SENSOR; policy.py BLOCK/ASK/ALLOW; type-safe ≠ correct). Typed control plane around DSPy (drafts AFTER route+action). Engine owns truth / Jev owns judgment. Human-confirmed kill (mapped explanations; identity re-check). Wire-compat encoder backend (GLiFormer `/v1/systemone` drop-in; cheaper, less accurate on reasoning-heavy). Loopback gateway routes hosted + local (not a model). **Constrained optimizer + S1 features** (slo-router: Jev never the sole hot-path gate; fail-open local features). **Effect-based shell gate** (construct: privilege ≠ verdict; fail-closed). **Attention filter / VOI** (jev-lens: never blocks the agent; never green unless sure). **Jev supplies evidence, code owns authority** (actiongate-jev). **Measurement owns endorsement** (jev-packs evidence-gated). **Ranking ≠ calibration** (does-jev-confidence; never hard-threshold raw p). **Hot-click CU** (ego-jev: indexed table → operation+target; text model only for type). **Jev judges relevance, code decides structure** (jev-compactor; never rewrite; regex floor). **Local rules first / never auto-train on own hides** (x-reply-filter). **Control-plane combinators** (Then/Gate/Vote/Cascade/Weighted; not chat turns). **Skill VOI / abstention** (skillranker; hook fail-open). **Receipts not leaderboard** (atlas + frontier-100 + OOD). **Turnstile** evidence≠authority + replay. **MLX one-pass replica economics** (jevmlx; softmax ≠ Noul). **TLA+ consensus kernel** (jev-labs; never confidently wrong; escalate). **SEAL advance/coverage** (no seal, no advance; exception queue visible). **Sureness bands** (how-sure-is-jev; max_prob is generous). **JevBench v1.1** (calibration reported, not scored). **CI typed gate** (ci-gatekeeper before expensive review). **Codex MCP adapter** (jev-in-codex; ranking unbenchmarked; lexical fallback). **Stop-hook attention redirect** (jev-preflight; eight axes; assist=one reinspect; fail-open; not a merge blocker). **Pre-send view selection** (dizk/jev-lens; 79% fewer tokens *theirs*; compress-before-first-send; distinct from rashedInt32/jev-lens). **Observational memory** (pi-om; keep/kind verbatim; model-free compact). **tools≠use** (carryforward 0/4 recall; SessionStart > hoping). **Physical-world S1** (HA-Jev; sensors from typed answers; not for locks/heaters). **Judgment outside the store** (jevql CLI; DB never sees `jev()`). **Landed-script / headless≠auto-approve** (construct); **digital-design combinators** (jev-combinators rename + Router/Loop/Retry/Fallback/Memory; metaphor ≠ literal AND/OR); **VOI cache admission** (jevcache same-intent; 0 FP/100 *theirs*; fail-open); **worth-your-attention VOI** (ThinkyMiner/Winnow 80%/90%; ≠ kevinpita/winnow); **Jev WHETHER / Python HOW / LLM WHAT** (hermes-jev-router; license null); **typed escalate/continue/abort baton** (jev-handoff; inverted loop; gate never grants; fail-open); **Playwright executes, Jev chooses** (browser-jev; sample-from-distribution); **OpenJev `/v1/decide` ≠ drop-in** + **SemIf runoff wire**; **conflict ≠ ignorance** (named Choice escape); **decision-as-memory flywheel** (DGUI_HYPERMEM-JEV 6-row schema); **record/replay CI** (jevassert landed; accuracy+ECE+cost gates offline); **failure-finding arena** (chenmingtang830/jevarena ≠ meetr1912/jev-arena); **BBQ** 97.28%/0.04/0.34/$0.3429 *theirs*; **decider≠executor** (jeffrey: Jev next-tool, LLM fills args); **sentence-as-rule** (mizchi/jev-lint is jevlint rename); **VOI hunk prune** (prune-review ~20% target; 1.18% with outlier); **persist constraints across compaction** (pi-heed); **Harbor SGR-judge contract** (jev-judge-bench; canaries ≠ quality; ≠ jevarena/jevbench); **hand no-text steps** (jev-use; Vercel drops confidence; margin 0.4; fail-open gate; ≠ jev-ultrafast); **Pi System-One control plane** (pi-jev-control; GUI never force-click; compaction never writes session); **never free-generates** (jev-gpt tree of Choices; 400 calls / 75 s / 2¢ *theirs*); **OpenRouter recipe atlas** (jev-cookbook; 16–36 item samples not benches; 425 calls / $0.015); **personal-history feed** (jevfeed; no social graph; one request per batch of ten); **competing NAR claim-audit** (openJev-verdict-2.0; dual-channel ECE; PR #1; ≠ IamBusy/OpenJev); **empty compaction-proxy skip** (IPECTER context-pruner **and** jev-runway LICENSE-only); **1-token logprob endpoint ≠ Noul** (chakuho; coverage ≠ correctness); **open replica engine** (jevinf argmax-parity); **unofficial Elixir SDK ≠ OTP peer** (typesafe-elixir-sdk ≠ dannote/jev); **jevex n=16 files-to-read VOI** (rename of jev-semantic-explorer); **commit pre-review attention≠verdict** (commitjev; middle band never rounded); **Hermes plugin is Agnes not TypeSafe** (hermes-plugin-jev); **pi-jev-compact ≠ pi-jev-compaction** (verbatim summarizer replacement); **decision-native inbox** (mailordinal; humans own ambiguity); **unofficial jev-cli not ready** (≠ jevql); **laya-multilingual** English checkpoint confident-wrong OOD; **schema-scorer peaked ranking ≠ calibration**; **productized System One HTTP** (classifier.dev; label+p; batch ~1000; Jev primary / LLM fallback); **escalate-under-threshold** (smart single-label <0.7; multi-label ignores); **silent FALLBACK** (granite 0.546 vs advertised 0.800; rh-guard owns the gate); **systematic-review pointer (choxos/jev-reviewer ≠ egma-ai)** two-pass Choice+Noul; *Not found* is an answer; human check is the product; **githubnext/localjev** prompted JSON ≠ structured-read logits (wire-compat ≠ logit-equiv; **≠** kunchenguid/local-jev; GitHub Next **261★**; 1,200-req caveats *theirs*); **NandhaKishorM/laya packaging** Router script-before-p; post-T ECE ≠ raw ECE; 0.85 still soft; Banking77 token-budget; **≠** TypeSafe drop-in; **external census ≠ scored bake-off** (@airesearch12; GLiNER2+routers class-boundary; incomplete vs Laya/localjev/kev; Harbor honesty watch); **JevBench v1.2 geometric-mean product** (I/C/S/K 25% each; cal ON rank; Luna I=96.8 rank #7; ×2/est. Harbor honesty; option-order 72→21; Laya absent gap; Qwen3.8 27B ≠ Archer); **hourly 0842 apply-the-five** (already §73–§78; do not re-card); **skip thin noise**; **hard-gate a Noul as a PR/quality gate is soundness theater** (totally-tim/jev-gate / claude-jev-warden; ≠ jev-gateway / MongLong0214/jev-gate / jev-gate-student-b); **S1 keeps flying / S2 one-use** (khordoo delta: escalate without stall; purple = consumed; Local controller ≠ githubnext/localjev; seed = geometry; 20% still soft; no pixels; S2 never grants). **OCR+AX desktop CU:** typesafe-computer-use (hosted Jev; never screenshot-to-frontier for the decision; overlapping options = doubt; writer/decider; 155× *theirs* one screenshot; 0.4/0.5 still soft; **≠** jev-ultrafast **≠** cua-s1 **≠** camoufox). **ASR voice-browser CU:** jev-voice-browser (partial-speech VOI; pointer spans; spoken confirm ≠ auth; 27/27 *theirs* fixtures; **≠** jev-voice-control **≠** nikolas-j **≠** OCR desktop). **Wrap-as-execution ALLOW/ASK/DENY:** wrap *is* the tool function; rules first; ASK throws; fail-closed (AgentGhost; **≠** actiongate **≠** jev-use fail-open; rh-guard owns the gate cousin). **JP genre atlas:** apps by hole; stars research-time; not verified evals (@studio_yebisu; **≠** class census §77 **≠** v1.2). **External pedagogy:** Akshay “Jev Clearly Explained”; LLM hammer; schema-safe ≠ correct; 200×/400× TypeSafe ceiling; shadow + questions-as-code; **≠** official docs **≠** Flavio **≠** AgentGhost. **Meaning-grep dedicated:** proposition≠embedding; boolean composition of thresholded Nouls; Semgrep.dev collision; not a gate (jev-semgrep §86). **Hourly 1047 + deferred 0945:** decision-validated UI (gram-render never authors text; jev2ui Jev decides / Gemini writes); decision-as-assert (jevtest ambiguous band 0.15–0.85 fails both; 0.85 still soft); hybrid S1 (anima3 closed verb menu + hard safety first; jeff confidently flat on magnitude; Qwen logprob default — do not invent Laya as a backend); pointer search (JevFind path then window); Harbor bake-offs three shapes (frontier-bench ≠ frontier-100; GLiClass product bakeoff not architecture duel; four engines / majority floor / calibration ≠ discrimination); authorship named escape (not evidence); non-SWE (ha-switchboard HA remains execution ≠ HA-Jev; n8n Low Confidence abstention); compaction delta (fast-jev-compaction-pi ≠ pi-jev-compact ≠ pi-jev-compaction); full-distribution optimizer (jevloop UCB1+CEM; no LLM in the loop; mock default); deferred class (laya-vision SmolVLM `score` untrained ≠ blackwood ≠ Archer; Cerebellum-2B `/v1/decide` ≠ TypeSafe — wire-compat vs agent-routing as separate Harbor axes, competing NAR not endorsement; laya-grounded not drop-in / phishing 0.611→0.512 / Platt not temperature). Hard-gating a Noul as test/PR/HA write/authorship seal is soundness theater. **Queued SIGNALs:** Collapse GestaltLabs/Jeff-1 into logan-markewich/jeff; Quote Jeff-1 ECE as “better than Jev”; Treat empty stanley findings as approval; Treat findme beam score as file-identity; Price workers, not the conversation. Soft Noul ≠ hard safety. **Hourly 1144:** Treat `.feels()` 0.5 as a bool if; Collapse apa-agent-harness into AntonioCoppe/jev-harness; Quote apa "mathematically fulfilled" / npm @aipersona; Treat grok-bot-jev A/B as token savings; Let Essentiel Jev send / skip human approve; Collapse enzo-mcp into jev-sift / skip UNKNOWN; Treat pigeonhole OTHER as a move / 0.6 as Harbor τ; Treat the HF playground as live Jev / collapse into classifier.dev; Quote jev-reliability as accuracy; Collapse clduab11/jev-test into realZachi/jevtest / paste bars as results; Paste "Jev wins" from jev-rag-benchmark; Collapse dairui1/jev-lab into BrendanH18/jev-lab / re-card jev-desktop; Treat jevmail as mailordinal / mailjay as read-only. **Hourly 1241:** Collapse ZHUBoer/ego-jev into jiangkoumo / treat `completed` as success; Treat jsort logits as frequencies / Choice as the scale; Paste groundedness Macro-F1 as “Jev wins quality”; Quote jev_playground 83% / promote from authored bars; Copy `jev-latest` on Zen; Treat techstack ranks as a generated stack; Collapse s1_ruby into hunch/feelings; Treat judgement as jevql / confidence as winner p; Treat the Rust community SDK as official / a new species; Collapse tpellet/hunch into carldaws/hunch / skip exit 3; Quote file-search 15 matches as recall; Treat linkmap referee as gold / let Jev see S2 prose; Treat jev-mail as jevmail / tidy OTHER as a move; Close pinned/audio/current tabs / skip Show; Let ORIGIN LLM decide / continue without Jev; Gate crawlers on raw `bug_likely`; Sell jevbrain AUTO_ACT as a Noul. **Hourly 1347:** hermes-switchyard ≠ hermes-jev-router ≠ hermes-plugin-jev; cyrusasco/typesafe-mcp noul deadband 0.35–0.65; smartdio/jev-browser-agent ≠ ZHUBoer/ego-jev; Dakai/omp-jev-web DONE ≠ proof; typed-gate band [0.40,0.60] is refusal; pi-jev-gate fail-closed; choice is the verdict; Jev-Calibration Platt ECE 0.117→0.052. **Hourly 1441:** Jev-Reranker live Jev not yet measured; sessionwise opt-in relevance; jev-search pointer sieve; savka777/jev-search ≠ kazuhideoki/jev-search ≠ superagents-lab/jev-search; 400ms Salesforce WebMCP; droidjev screenshot-free; Tewoto1 jevcu planner still writes; ha-conversation-jev Jev→Grok; dsh-jev can only gate; jev-classification-benchmark specified not run; jev-luna-pagerduty p≥0.50; jev-drive sim not AV; story-arc Jev never authors; jev-hs-assistant HS6; golergka/jev-plays-starcraft-2 UI-verified ≠ API Victory; awesome-jev-use-cases catalog. **SIGNAL §93:** fingerprint after redact; recall vs decide; publish fingerprints+answers; CI replay as Harbor cousin; Cache hit ≠ correctness; hyperspaceai/jevcache ≠ kushals256/jevcache; human labels only; score never auto-accepts; production capture flywheel; sutro-sh/jev-align ≠ caiovicentino/jev-align. **SIGNAL §94:** guidance ≠ hook; catalysts ≠ summaries; compile-time System One; unofficial ≠ TypeSafe; format_version modernbert-jev/1; Argos1111/jev_local ≠ us/jev-local ≠ kunchenguid/local-jev; LFM default ≠ ModernBERT backend; Nemotron ≠ TypeSafe Jev; not a calibrated replacement; djev-dev complements djev-spark; images as Choice options; Laya essay numbers *theirs*; Router/OOD confidence; hosted bootstrap ≠ silent TypeSafe. **Hourly 1541:** difficulty + policy thresholds + JSONL trace; jev-codex-pilot model + reasoning depth; keep/shadow/hybrid/reject; quarry evidence projection; Frank-ZY-Dou/awesome-jev robotics/3D/control; one-dollar-tahoe TypeSafe Jev defense eval; jevguard calibrator/cache/escape; jev-ci-selector CI shadow mode; llama-jev llama.cpp replica; petercr/jev-orchestrator ≠ FleeexCorp/jev-orchestrator; seb4ez/jevguard ≠ AseemPrasad/JevGuard ≠ pablozr/JevGuard; webNeat/llama-jev ≠ WiktorB2004/llama-index-jev. **Hourly 1639:** OpenCode jev-pruner context sieve; observe→score-candidates→prune; jev-zen / jev-1.13-free; zen-chat ≠ Noul; fail-open original; keepScore >0.1 floor; host port of tamaratran/jev-pruner; indiejoseph/opencode-jev-pruner ≠ nrdz-labs/fast-jev-opencode; jev-webagent-bench empty stub; Kiln-AI/jev_jsonschema noul_threshold 0.5; NSStudent/JevSwiftSDK unofficial. **SIGNAL §97:** GLiNER2 native Apple path; unofficial Swift/Core ML GLiNER 2.5-small; entity spans + confidence; not Choice/Score/Noul; not TypeSafe; label descriptions as schema; on-device ANE economics; honesty locks; shershah1024/gliner-native-runtime ≠ Fastino; ≠ gliner25-compaction ≠ gliner2-ultrafast ≠ Eran-BA/Jev_from_GLiNER2 ≠ NSStudent/JevSwiftSDK ≠ jevmlx; default threshold 0.1 still soft. **Hourly 1740:** Decision Graph Protocol frame→assess→commit; app retains permissions/effects; Jev-first assessor-neutral; guarded commit / receipt/next frame; assessment batching; hard-gating DGP as safety theater; numerous-com/dgp ≠ TypeSafe official; jegrep calibrated path+range Nouls; no embeddings/index/daemon; ~$0.01–0.03 typical; agent --json; can1357/jegrep ≠ Bentlybro/jevgrep ≠ uehaj/jev-semgrep; Archer-arch fidelity; kev family OOD 0.76–0.77 vs Jev 0.86; block-causal isolation; pointer/readout CE-trained; /v1/systemone drop-in; replica honesty. **Hourly 1843:** judgment vs generation; deterministic execution after probabilistic judgment; exactly one app-owned callback; explicit uncertain branch; YES / NO / UNSURE from cost_false_yes / cost_false_no / cost_human; auto-batching same-object questions; Kungie/gut ≠ tpellet/hunch ≠ carldaws/hunch; Illusion47586/judge ≠ lexingtonhibiki/judgekit ≠ Ascurse/typed-judge-kit. **Hourly 1943:** Jev IS the if-statement; judgments/probabilities drive branches; text model only writes prose; interpreter owns variables/loops/budgets/replay; otherwise maybe / confidence gate; chaos samples after the gate; Jev-first Pi agent loop; slow-LLM fallback **Hourly 2041:** resume-screening bias audit methodology; Plan/PRD panel → code-owned pass|review|block; cost-aware multi-model routing/escalation; frozen-protocol zero-shot bench; context-window admission control; typed decision control plane; receipt ≠ authorization; live 15-dim typed rubric re-score per pause; adversarial pre-registered Jev eval; confidence does not track ignorance; provider-neutral Elixir/BEAM Noul/Choice/Score SDK **Hourly 2145:** question-linting of Jev questions themselves; nine jaggedness rules, no API key, no labelled data; static lint ≠ measured separation; yodablocks/jevq ≠ tenbin ≠ JevLint ≠ commitjev; open-weights Laya as class exemplar (binding); Nx/Bumblebee runtime; host chooses backend; ChristianAlexander/laya_ex ≠ system_one_sdk ≠ dannote/jev ≠ NandhaKishorM/laya; on-chain/edge Laya deploy; parity_verified stays false; model output never grants Tx; humandebri/IC-Laya ≠ laya_ex; auditable weekend replica; Jev outputs never used for training; unpaired 0.577 vs 0.727; agilabs-ai/jev48 ≠ JevBench ≠ Mapika/decider; adversarial dual-judge / framing attack surface; comparative framing is the usable judgment; prior injection crowds out evidence; copyleftdev/ember ≠ ember.js; Laya specialist fine-tune pipeline; training still GPU-pending; PIXELZX0/XERON ≠ convaiinnovations/laya; Hub Laya replica drop; daliborsb/laya ≠ convaiinnovations/laya ≠ NandhaKishorM/laya; System One student distillation corpus; gold is programmatic; teacher is closed-API clone; MagaBitmex/jev-4b-distill-data ≠ missing student checkpoint; non-LLM VIN System One; planning depth not chat; lewislululu/jevon ≠ douglance/jevon; source-bound evidence checks; local quote mismatch needs no API; exit 0 ≠ claim truth; WaynezProg/jev-kit ≠ jonathanavis96/jev-kit (Airlock) ≠ jev-use ≠ jev-mcp **Hourly 2246:** independent System One evidence catalog; 19 reviewed records; scores not one leaderboard; no external record currently reproduced; TokenTrim no-Jev matched hybrid 62.4%; reachjalil/system-one-bench ≠ mallahyari/system-one-benchmark; 21 tasks · 134 items · 208 questions; scenes from public GitHub contracts, not production logs; SivletLabs/jev-eval ≠ willkelly/jev-evaluation ≠ 4esv/jev-eval ≠ xxkuboxx/jev-eval ≠ onlyoneaman/jev-eval ≠ dayhaysoos/jevals; option isolation (sibling-blind); permutation-equivariant; Hub OWNER not published; nafisazizir/hev ≠ jaredpalmer/kev; frozen local LLM logits, no trained decision head; residual-head 9,222-param decreased 73/96→67/96; confidence = 1−normalized entropy, not P(correct); yuki-oshio/mini-jev ≠ r-ms/mini-jev; Jev classifier as autoregressive next-token predictor; ChatJev-style soundness theater; erik-dunteman/ChatJev ≠ dannote/jev ≠ jev-gpt; calibrated decision head × AlphaProof value head; implementation-layer isomorphism, semantic difference; timeout = censoring; do not launder Noul as proof; parallel rank-prediction vs serial selection; independent questions can conflict; zzzzzec/jevsort ≠ keltokhy/jsort; curated open System One ecosystem catalog; rupeshpoojary9/awesome-open-system-one ≠ AnotiaWang/awesome-jev; arXiv paper radar with Jev relevance scoring; ranking ≠ calibration / 0.5 still soft; fail-open failed evals not marked seen **Hourly 2340:** train calibrated ~27M from scratch; typed Q→prob dist / one forward pass / no LLM decode; hyusi2003/MiniSystemOne ≠ Colvin0315/MiniSystemOne; description-only stub / size 5; ESCI hard probe fails four of six; jev_bool ECE 0.242 inversion 0.255; do not re-fold §60 six-gates as new; jobbyjev one-request-per-company from batch-size result; find/design/evaluate TypeSafe Jev decision loops; karanb192/jev-architect ≠ samtay32/jev-system-architect; Jairik/jev-distiller size 1; distill-Jev UI stub / do not distill Jev as teacher of record; post-launch scored use-case map / Jev self-scores then human curation; licensedsaucer9-web/jev-opportunities; Jev-inize a use case into classifier/router; gavinHuang/jevinize → simple-jev not TypeSafe; featherless-ai/simple-jev; compare saved decisions / same label can still change the branch; VihaanAgarwal/jev-diff ≠ Saik0s/diffusiongemma-jev-macos; not tested with a live Jev API key; constrained logprob + temp/Platt ≠ Noul; OpenJevPro pastes openjev-sglang JevBench as own; zhangcy122/OpenJevPro ≠ IamBusy/OpenJev ≠ ekzhang/openjev-sglang; PolyForm Noncommercial; SmolLM-135M / sub-70ms / 0 output tokens; demo P(True) 0.5052 / Choice conf 0.2872 / Score conf 0.0055; README claims MIT / GitHub license null / no LICENSE file; patelvishwa112/jev-system-one-rlcd ≠ arnabgho/rlcd-lite ≠ blackwood-rlcd; source-backed Awesome Jev radar / 306+ commit-pinned; logicrw/awesome-jev-projects ≠ AnotiaWang/awesome-jev ≠ yibie/awesome-jev ≠ cobanov/awesome-jev ≠ rupeshpoojary9/awesome-open-system-one; auto GitHub sync / Issue-only submissions; hashed n-gram encoder / rival-aware attention; olanotolu/jevbetter vs jevlike starter; synthetic hard menus top-1 0.916 vs 0.873 / ECE 0.0182 vs 0.0367 / 40 vs 4608 menus/sec; shuffled-context control 0.335 | `references/mixed-architecture.md` Apply 0042 (`notes.md` §105): structured probability readouts; distribution > argmax; Noul 0.5 midpoint; score is expectation not integer; bare HTTP not SDK; Arohtea/jev-readout; Jev-style Choice/Score/Noul from ordinary models; optional DSH plugin; schema-valid ≠ calibrated; gulagala001/jevify ≠ Mintzs/jevify; Laya RLCD benchmark; 40.3% below constant-answer; open-weight measurement; mourad-ghafiri/laya-rlcd-benchmark ≠ yibie/laya-jev-lab; cheap fail-open semantic edge; second signal not sole; FastLoopError catch; SupremeDreamZ/jev-fastloop ≠ jev-ultrafast; asking more questions in one call; 0.980 at every N; nearly not fully deterministic; TheWebDevel/jev-fanout; Qwen3-VL perception + Jev decisions train RL; 0 model calls at deployment; VLM alone 1.7 vs +Jev 4.4; harneet2512/reflexrl ≠ khordoo/jev-reflex-autonomy-lab; independent Jev API vs Laya; cascade 0.60 matches 78% at 1.8×; noul facts not judgements; yibie/laya-jev-lab ≠ dairui1/jev-lab ≠ BrendanH18/jev-lab; GLiNER vs GLiFormer vs Laya vs Jev; extractors ≠ decision engines; Laya dict-instructions collapse 58.3%; umstek/zero-shot-ie-bench; decisions-per-minute & cost; 204 moves vs 73; throughput not intelligence; angelgalvisc/snake-arena-jev-vs-llms ≠ vtrivedy/jev-plays-games; behavioral contracts; pin expectations eval upgrades; raw 0.94 is not a release; sathariels/jevcheck ≠ dayhaysoos/jevals ≠ SivletLabs/jev-eval; evidence-linked dependency upgrade; Jev never generates filenames; no_direct_evidence ≠ safe to merge; GaneshVG18/upgrade-radar ≠ LYchoon/paper-radar-jev; discography theme/mood/complexity; five atomic questions one call; lirantal/discoprint / Apply 0145 (`notes.md` §106): Turn any open LLM into System-One Jev; description-only stub / size 0; Exu is a toolkit, not a method; typed Q → probability dists; JSON parse of generated text ≠ Noul; classifier.dev fast tier 84.8 is Jev behind its own API; hard budget filter before Jev; Jev judges the next state, XState enforces transitions; catalog gravity; ★339 live REST; query-side encoders, not a Jev replica; transformers.js AutoModel cannot load this graph; This Space contains no benchmark result yet; do not reopen or amend PR #23. / Apply 0243: Benchmark-driven Jev router and judge; cheap alone is not success; Jev does not write, sum prices, or claim accuracy %; Sol 94.2 / Luna 83.9 / Jev path 89.7; 19.2% Sol / 62.3% cost save / 4.5pp miss of 2pp non-inferiority; p50 latency worse than Sol due to routing overhead; erendikmenn/jev-llm-router-benchmark ≠ jev-rag-benchmark ≠ ryantsai/jev-llm-router; Express + node:sqlite; mock and Jev decision engines; previous_ticket_count >= 3 is code; MIN_CONFIDENCE 0.6 still soft; substring false positives; aesaganda/jev-ticket-router ≠ SarathChandraBellam/jev-vs-llm-ticket-router; Universal Figure & Diagram Router; confidence ≥ 0.85 hard-gate is theater; generative AI banned from scientific plots; six visual branches; human-labeled (state, question, label); 166,054 rows / 22 configs; soft_label for human uncertainty; Praveenrajus/jev-bench ≠ fstandhartinger/jevbench; ternary bonsai System One GGUF; Hub does not ship weights; 100/100 easy T/F is not Harbor; label_mass ≠ correctness; stock llama.cpp Q2_0 silently gibberish; NicolaiMTLassen/open-bonzi-jev ≠ NicolaiLassen; transformers.js DeBERTa ONNX; temperature 1.05; AutoModel from_pretrained works; onnx-community/open-jev-deberta-v3-large-ONNX ≠ system-one-qwen3.5-4b-scorer-ONNX; 107★ densify; GH 151M vs README 149.6M; PR #1 now closed unmerged; do not re-fold §71 claim-audit as a beat; typed decisions, RLCD, confidence-gated routing; structured ≠ correct; mock not live API; 26 tests; wjdjdakf17/jev-study ≠ baekenough/jev-study / Apply 0345 (`notes.md` §108): **Open reproduction densifies / measurement densifies PRIMARY** (bonzi-27b-v2 / ternary-8b / 27b-v1 GGUF family densify; WANLI-256 74.6% / 65.2% / 71.1% *theirs*; Bonsai 1 27B Q1_0 runs on stock llama.cpp; ternary still needs PrismML fork; hf:heman10x/openJev-verdict-2.0 twin tokenizer-only; OpenJev Vision image classification + uncertainty; CLEVR-4 held-out joint 0%; hfdataset:IamBusy/OpenJev-Vision-Research-v0.1 12,832; 294,912 derived targets not independent samples; Laya multilingual ONNX WebGPU typed-decisions port; 63/63 selected answers / 5.1e-4 CPU / 1.2e-2 WebGPU; UpHash-Network/mini-jev is yuki-oshio transfer; jev-injection-bench 11,900 labelled prompts; Jev best ranking / Haiku better ECE 0.021 vs 0.058; 0.5–0.9 band is where Jev's numbers do not mean what they say; Prompt wording moves panic 28%; manojlds/jev-dspy-bench ≠ dspachos/jev-dspy ≠ jmanhype/jev-dspy-lab; Jev agreement is similarity, never ground truth; no aggregate quality grade or merge gate; AbstentionBench-on-Jev rank 1 of 20 vs 2025 field; question-asymmetry; forward-looking 0.465 never extreme; openkev calibration layer not a runtime; ECE vs coverage independent; select_threshold returns inf; escalation catches uncertainty not ignorance; misakaikato/openkev ≠ jaredpalmer/kev; pdf-race Docling→Jev vs Gemini; parser owns the wall clock; 12/12 tie is a tie; titles selected not generated; flopcheck 16 calibrated tweet judgments; mechanical tells in code; ZeroX-01/jev-atlas ≠ Zaious/jev-capability-atlas ≠ gorock007/jev-atlas; Laya calibration lab Gradio MCP; T never changes argmax; confidence ≠ top-label p; easy probe set refused; 40–48 rows too small to ship T; do not reopen or amend PR #23 or #24 or #25. Apply 0439 (`notes.md` §109): Gemma-4 26B-A4B jevify classification+calibration; LoRA adapter twin not independent eval; Gemma-4 E4B jevify; E4B LoRA stub card; kushalpatil/jevify-gemma4 ≠ Mintzs/jevify ≠ gulagala001/jevify ≠ uspraveen/Jevify; GH kushalpatil07/jevify 404; PAWS 0.580/ece 0.288 is the weak cell; smaller E4B slightly better OOD ECE than 26B-A4B; Hub jevify merged LoRA ships weights; bonzi Bonsai-8B v1 GGUF densify; Bonsai-1.7B v1; Bonsai-4B v1; WANLI-256 64.5% / 60.2% / 52.0% *theirs*; rank #4 / #5 / #6 of 6; JulesHuisman/jev-eval scaffolding / README SHA c356a584 (was empty e69de29b); JulesHuisman/jev-eval ≠ SivletLabs/jev-eval ≠ willkelly/jev-evaluation ≠ 4esv ≠ xxkuboxx ≠ onlyoneaman ≠ dayhaysoos/jevals; 7 bands 6/10 vs 40 bands 0/10; source receipts + confidence slider re-policy without re-inference; 32/32 synthetic is smoke not production; classify HF datasets across typed semantic dimensions; roadus2 watch misspelling; lock roadius2/ultra_laya; ultra_laya REVIEW defects; default branch claude/laya-jev-review-gg5ppo; XNLI EN 88.3% ECE 0.032 → RU 77.3% ECE 0.096; Δ −11.0 pp [−14.2,−7.8]; ECE +0.063; MASSIVE no detectable difference at n=600; confidence is function of p_max (r=1.000); pointer-not-generator 400 human-authored responses; proposed ≠ authorized; FewRel 160: Jev 85.0% vs lexical 13.125%; gated 100% (95/95) coverage 59.375%; J++ composable semantic computation language; judge-jev 0.5 still soft; 947 repos scored; A 273 / B 302 / C 372; LLM rubric ≠ benches; No benchmark winner is claimed; phishing: naive 62.6% vs regex 91.8%; 5-atomic + LR 95.0% *theirs*; AITuber tension ±15; README npm global; repo is Rust; git-confess code owns counting/blame/ratio; httpx exhibit 11% (13/119) *theirs*; 90d trend +12.40% vs random +12.75% vs BH +41.71%; 5m win rate 25%; Awesomejev 656 entries / 38,160 stars; tracker likes 64 (+4) lastModified UNCHANGED; Laya present; Blackwood ABSENT; Archer still promised_not_landed; do not reopen or amend PR #23/#24/#25/#26. Apply 0541 (`notes.md` §110): Blackwood tracker ABSENT; likes 2 gated manual; r = c - p_a; ECE 0.021; acc 0.807 vs warmup 0.746; calibration beyond ~500 tokens unmeasured; Independent primitive; 11.57s vs 54.10s · 4.67× · 120/128 *theirs*; default path is pretrained Gemma probs not trained RLCD head; GH Meanblock 404; lock leesk212/JEV-CPU; softmax over letter slots ≠ Noul; WANLI 0.741 vs openjev v2 0.77 *theirs*; 3-way NLI ≠ Noul; priority 0.464 = majority floor; banking77 contaminated; raw margins not probabilities; GH jev-haiku-benchmarking 404; do not distill Jev as teacher of record (they distilled Haiku); “0.9 is not one number”; ranking ≠ calibration; banking77 0.8–0.9 stated 0.86 actual 0.73 over-confident *theirs*; ≠ Praveenrajus/jev-bench ≠ fstandhartinger/jevbench; $0.0000153–$0.0000226 vs circulating $0.0004 (~20×); Score is 0..n-1 expectation not 0–1; Noul has no confidence field; TCP floor 198.8 ms; type reliability is not a reason to choose Jev (json_schema 5/5); gateway tax not one number; Function-only 5/8 vs hybrid 8/8; 4/8 without Jev; 8 designed cases not conversion lift; ≠ RadRebelSam/awesome-jev; 200-row pilot Jev 86.5% 173/200 vs Gemini Flash-Lite 86.0% 172/200 vs Pro 87.0% 174/200 *theirs*; not a ranking; NLI Tetris argmax P(entail)−P(contradict); 情緒測謊器; 1q 396ms / 30q 567ms; ±0.03; 33q $0.000045 vs Gemini ~5× slower ~60× cost *theirs*; ≠ realZachi/jevtest; 8-example Jev vs GPT-5.6 Sol ~64× cost 5.4× latency *theirs*; synthetic; no inference; ≠ JevBench v1.2 §78; Judged 3317 / listed 2560; Jev judges, code applies policy; catalog ≠ endorsement; APA “microsecond policy / zero hallucination” overclaim; Client-side quiz; pointer from held docs; scanned-PDF warn; CSP only api.typesafe.ai; Jev judges / agent reasons / user decides; selecting an option is not permission to implement; degraded fallback; pattern exact, judgement must clear floor; no matching pattern → no model call; not a correctness oracle; $0.00022 vs chat $0.00306 *theirs*; Spec vs artifact remainder; treating 0.85 as 85% / minProbability hard-gate as Harbor; VERIFY acquires discriminating evidence, never same-pool confidence-only rescoring; fast/full/max are ceilings not sizes; Solar writes, Jev chooses NEXT ACTION; SemIf 2186★ (+20 vs §109 2166); jevlike 1038★ (+7 vs 1031); TypeAR 14★ flat; AnotiaWang 96★ (+1 vs 95); yibie/awesome-jev 490★; Laya likes 802 (was 783); tracker likes 64 flat, lastModified UNCHANGED; do not reopen or amend PR #23/#24/#25/#26/#27. Apply 0743 (`notes.md` §113): Hourly 0743 uniqueness lock: Heman10x-NGU/Verdict-open-jev ≠ Heman10x-NGU/openJev-verdict-2.0; TF-IDF + LogReg ECE 0.0207 vs Jev 0.1440; Verdict-open-jev 48.07% vs Jev 90.80%; abstention combined recall 10.00%; p50 35.58 ms; K=25 (maximum capacity) 72.00%; 0.85 coverage 84.60% selective risk 1.18%; 26.1× faster than standard Qwen JSON generation; Jevify 90.0% / 167 ms CUDA graphs disabled; Finding 1: Brier on stated confidence alone is a trap; grpo_rlcr 0.78 / ECE 0.084; reliability 0.007 but resolution 0.000; 27 900 schema-driven decisions; 13 600 / 13 600 questions; candidate mass min 0.99999624; 22 configs · 166,054 rows · 4 calibration-gold; sha a39eba3f; Student B MAE 0.148 / Pearson 0.836 / 86.0%; pngwn/open-jev-laya-bench README 404; sha 9f69c742 likes 2; HDFS 0.9933 (745/750) / retain 0.0084; BGL ERROR/FATAL protection 1.0000; 2,479 / 2,500 HDFS uncertain; cache hit 0.9648 (2412/2500); $0.153936 estimated; E2 recomputes from saved probabilities; Space sha eda59e0a; MASSIVE English 0.783 / Khmer 0.033 / Hindi 0.133; 40–48 rows too small to ship T; T never changes argmax; siren2345/jev-single-decode-transformers ≠ siren2345/jev-single-decode; Split Transformers experiment from llama.cpp runtime; tanayvasishtha/jev-lab ≠ dairui1/jev-lab ≠ BrendanH18/jev-lab ≠ yibie/laya-jev-lab; Four experiments stress-testing TypeSafe's Jev: calibration, bundle bias, label bias, and ensembling; second pass must be $0.00 from cache; The pages never call Jev; Gemma 4 31B 77.0% / Jev 1.13.0 61.4% / Laya 322M 0.0%; restriction state 95.0% against 84.4%; None of the systems are particularly good at knowing when to stop and ask; They skip the question and call a tool directly; 100% schema pass; six-field joint 48.8% vs 72.8%; ywchiu/jev_benchmark ≠ Running-Dolphins/jev-bench ≠ Praveenrajus/jev-bench; ACT / REVIEW / FALLBACK; A provider failure, timeout, malformed output, or missing answer is **not** a policy outcome; confidence is descriptive provider output, not a substitute for probability; Quality denominators include only valid scored answers; an exact halfway tie chooses the lower level; aiwithenoch/Jev-Skill ≠ simplosophy/jev-skill ≠ laguagu/jev-skills; The local path does not claim to turn a smaller checkpoint into Jev; Low support becomes decision: "review"; MIT-0 SPDX NOASSERTION; current-llm; 结构兼容,不是 Jev 模型能力; altryne/jevify ≠ Mintzs/jevify ≠ gulagala001/jevify ≠ uspraveen/Jevify; Find where Jev belongs. Design the questions. Measure the difference; TypeAR-AI/TypeAR 301 → TypeLLM/TypeLLM; TypeLLM/TypeLLM 16★; SemIf 2241★ (+34 vs §111 2207); jevlike 1051★ (+8 vs 1043); AnotiaWang 98★ (+1 vs 97); yibie/awesome-jev 525★ (+19 vs 506); Laya likes 864 (was 822); tracker likes 67 (+3 vs 64); lastModified UNCHANGED `2026-09-20T04:29:16.000Z`; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#32/#33; notes.md §113 Apply 0646 (`notes.md` §111): Calibration is not alpha; NO CURRENT ALPHA CANDIDATE; ΔR² approximately +0.00084; Brier 0.2131387; ECE 0.0421875; Adding Jev probability to deterministic volatility improved Brier by only 1.4058e-05; default 0.5 keeps zero non pinned; keepResult median 0.14 to 0.17; keepCall median 0.28 to 0.35; usable range is about 0.10 to 0.25; 7.8% to 57.9%; judges results it never sees; task-finish eval not built yet; $0.002 per compaction; slavadubrov/sgr-judge-bench ≠ slavadubrov/jev-judge-bench; Jev 108/120 $0.083 0.34 s; Luna SGR 114/120; paired Jev accuracy-difference intervals include zero; not evidence of equivalence; GLM SGR 26/120 93 format failures; Terra-planned Jev hybrid 55/120; rule-based by default, optionally Jev-backed; empty README; missing key cannot break the experience; prefill plus exactly one decode; softmax over A/B/C ≠ Noul; BBQ 9,053/10,000 (90.53%); ECE 0.0890; Mean confidence 0.9943; overconfident; score and noul not implemented; DGUI 12 rows (was 6); INSTRUCT 119 rows likes 2; encode the state once, decide everything in parallel; 0.740 accuracy against a 0.508 majority; ECE 0.047; fine-tune's advantage ends where its 384-token training data does; jasonkneen/open-jev ≠ pngwn/open-jev; same sha d41dc3cd; Space does not call Jev; recomputes routing from saved probabilities; 200-case Jev 97.0% / 100.0% / 95.0% / MAE 9.22; synthetic repository benchmark; Jev evaluations are advisory; YehuiTang0316/jev-nlgrep ≠ Bentlybro/jevgrep ≠ can1357/jegrep ≠ uehaj/jev-semgrep; default threshold 0.8 still soft; 40-line windows cannot prove whole function; token-native sequential start/end Choice; Gemini/Haiku stubs not configured yet; handful of hand-written examples, not a benchmark; Jev judged exactly what it was given; laguagu/jev-skills ≠ laguagu/jev-evidence-lab ≠ Pleo2/awesome-jev-agent-skills; contract_passed is not a claim of guaranteed factual truth; Wilson lower bound 0.85 floor; fixture mode no savings claim; SemIf 2207★ (+21 vs §110 2186); jevlike 1043★ (+5 vs 1038); TypeAR 15★ (+1 vs 14); AnotiaWang 97★ (+1 vs 96); yibie/awesome-jev 506★ (+16 vs 490); Laya likes 822 (was 802); tracker likes 64 flat, lastModified UNCHANGED; do not reopen or amend PR #23/#24/#25/#26/#27/#28.) | @@ -531,6 +531,28 @@ Do not copy keys. Fresh PR off `fea7c0b` (merged #47). Do not reopen or amend PR #23–#47. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. +## Open-Jev densify (`notes.md` §125) + +DENSIFY the original 1441 card. Do not mint a sibling first sighting. +HEAD 4933ee84951f README SHA ce1a587219e4. pushed 2026-09-21T01:34Z. +Astra TREC commit 1dd56990be7e. live 3★ (star-noise is not the fold). +LoRA adapters plus a trained scalar decision head and calibration +temperature. not merged base models. Independent of TypeSafe. +no RLCD/parity claims. LoRA ≠ RLCD replica. +customer-service P50 local HTTP 85.03 ms vs Jev HTTPS 295.26 ms *theirs*. +1024 tokens/32 candidates Open-Jev slower 1015.90 vs 301.37 *theirs*. +prefix caching experimental/off by default. +systems latency ≠ semantic equivalence. +GPT Luna P50 918.13 ms Astra 1938.39 ms *theirs*. +TREC-DL Jev/Luna/Astra completed. Open-Jev TREC pending. +80,816 training rows. 2B 94.71% / OOD 86.02%. 9B 97.54% / 91.97% *theirs* not Harbor. +hard acc ≠ calibrated Noul. type-valid ≠ exact. +Qwen/Qwen3.8-27B ≠ Archer. +website https://zefan-cai.github.io/open-jev/. +SHA move is not a replica. Do not copy train flags. +Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. +`invented_signal: false`. + **Hourly 1441 HIGH (`notes.md` §125).** vLLM NVIDIA + MLX Apple Silicon. Codiv hosted free endpoint. dual /v1/systemone + /v1/chat/completions. chat 501 on MLX. dual serving is not generate. Hosted Codiv ≠ TypeSafe. candidate probabilities are relative not correctness. recommendation is advisory. the server never blocks on its own. LoRA ≠ RLCD replica. pass-min 0.8 still soft. 37.30s → 2.40s at 64 decisions *theirs*. 2B 94.71% 9B 97.54% hard test *theirs*. wire-compat ≠ logit-equiv. SHA move is not a replica. catalog ≠ endorsement. Do not reopen or amend PR #23–#47. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. Hourly 1441 uniqueness lock: vLLM NVIDIA + MLX Apple Silicon; Codiv hosted free endpoint; dual /v1/systemone + /v1/chat/completions; razorback16/openjev densify HEAD cddbd962c88a README SHA a5943415cb92; STE README rewrite; serving-port densify; chat 501 on MLX; dual serving is not generate; Hosted Codiv ≠ TypeSafe; wire-compat ≠ logit-equiv; SHA move is not a replica; Error contract is not a Noul; hr98w/jev-visual 167★ Apple Silicon visual candidate scoring; 37.30s → 2.40s at 64 decisions *theirs*; Breakout 9 bricks 6 returns 2 lives *theirs*; candidate probabilities are relative not correctness; jkudish/jev-mcp 156★ ten MCP tools; recommendation is advisory; the server never blocks on its own; TypeSafe CLERC 5% to 18% *theirs*; jkudish/jev-mcp ≠ burnigtm/jev-mcp; zhengxuyu/litjev off-the-shelf Qwen decision layer; Probabilities are not calibrated by default; Qwen/Qwen3.8-27B ≠ Archer; zhengxuyu/litjev ≠ alexwestco/llm-to-jev; Zefan-Cai/Open-Jev LoRA + scalar head; 2B 94.71% 9B 97.54% hard test *theirs*; 2B OOD 86.02% 9B OOD 91.97% *theirs*; 80,816 training rows; 27B still in progress; LoRA ≠ RLCD replica; Zefan-Cai/Open-Jev ≠ TheoLeeCJ/openjev ≠ razorback16/openjev; cristianoliveira/jeq intelligence you can pipe; pass-min 0.8 still soft; JEQ does not own actions; AndyInQtr/laya-coreai CoreML serving substrate ≠ calibrated replica; AndyInQtr/laya-coreai ≠ mizorewww/laya-coreml; catalog ≠ endorsement; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47; notes.md §125 @@ -568,3 +590,5 @@ Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. **User-provided 1936 HIGH (`notes.md` §130).** sgoedecke/system-one first-sighting. SystemOne.from_pretrained. TypeSafe-compatible ≠ TypeSafe replica. mithalouni/system-one-open first-sighting. 76.7% vs Jev 86.9% *theirs*. replica ≠ TypeSafe. kotoba-lang/typed-decisions first-sighting. DeBERTa-v3-large 0.855 / 42 ms *theirs*. kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions. aisearchio 15-link census catalog ≠ endorsement. soft scores ≠ hard gates. SHA move is not a replica. Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. User-provided 1936 uniqueness lock: sgoedecke/system-one 20★ HEAD ebde2a2db706 README SHA d331b567e2c3; SystemOne.from_pretrained; Batched single-token choice inference; TypeSafe-compatible; cache_prefix=True; LICENSE absent; sgoedecke/system-one ≠ mithalouni/system-one-open ≠ KathanModh259/system-one ≠ babybear-labs/system-one; TypeSafe-compatible ≠ TypeSafe replica; mithalouni/system-one-open 18★ MIT HEAD 77f1f7cccf8a README SHA 535f33028a68 LICENSE SHA 2f6f2cf1064e; Gemma 4 E2B / Gemma 3 270M Modal; 76.7% vs Jev 86.9% strict common subset *theirs*; 97 ms H100 *theirs*; 74.8% held-out *theirs*; replica ≠ TypeSafe; HF upload pending; kotoba-lang/typed-decisions 1★ Apache-2.0 HEAD 10d7834d3b99 README SHA 4d6bbf4c4e44 LICENSE SHA 513bb5e3cb4c; ModernBERT / DeBERTa / LLaDA-MoE; DeBERTa-v3-large 0.855 / 42 ms *theirs*; ModernBERT-base 0.717 / 68 ms *theirs*; LLaDA-MoE 0.835 / 676 ms *theirs*; kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions; encoder class member not Jev replica; aisearchio 15-link census catalog ≠ endorsement; 12 already carded 3 gaps this fold; soft scores ≠ hard gates; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §130 +**Open-Jev densify (`notes.md` §125).** DENSIFY the original 1441 card, not a sibling first sighting. HEAD 4933ee84951f README SHA ce1a587219e4. LoRA + scalar head + calibration temperature. not merged base models. customer-service P50 85.03 vs Jev 295.26 *theirs*. 1024/32 slower 1015.90 vs 301.37 *theirs*. systems latency ≠ semantic equivalence. Open-Jev TREC pending. hard acc ≠ calibrated Noul. type-valid ≠ exact. LoRA ≠ RLCD replica. Qwen/Qwen3.8-27B ≠ Archer. SHA move is not a replica. Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. +User-provided Open-Jev densify uniqueness lock: Zefan-Cai/Open-Jev densify HEAD 4933ee84951f README SHA ce1a587219e4; pushed 2026-09-21T01:34Z; Astra TREC commit 1dd56990be7e pushed 2026-09-21T01:17Z; live 3★ (was 0★; star-noise is not the fold); LoRA adapters plus trained scalar decision head and calibration temperature; not merged base models; dataset ZefanCai/Open-Jev rev c67699e13d0a; Open-Jev-2B rev 0c7aa498b162; Open-Jev-9B rev 47e966881e48; 27B still in progress; Independent of TypeSafe; no RLCD/parity claims; LoRA ≠ RLCD replica; customer-service P50 local HTTP 85.03 ms vs Jev HTTPS 295.26 ms *theirs*; 1024 tokens/32 candidates Open-Jev slower 1015.90 vs 301.37 *theirs*; prefix caching experimental/off by default; CUDA prefix caching exceeded tolerance on 9/11 workloads; systems latency ≠ semantic equivalence; GPT Luna P50 918.13 ms Astra 1938.39 ms *theirs*; TREC-DL Jev/Luna/Astra completed; Open-Jev TREC pending; 80,816 training rows; 2B 94.71% / OOD 86.02%; 9B 97.54% / 91.97% *theirs* not Harbor; hard acc ≠ calibrated Noul; type-valid ≠ exact; Qwen/Qwen3.8-27B ≠ Archer; website https://zefan-cai.github.io/open-jev/; launch X thread https://x.com/Zefan_Cai/status/2101782158658695388 https://x.com/Zefan_Cai/status/2101786019607740436 https://x.com/Zefan_Cai/status/2101789698947793231; densify §125 not a sibling first sighting; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §125 diff --git a/.agents/skills/augustus/references/agent-self-assessment.md b/.agents/skills/augustus/references/agent-self-assessment.md index c7bdfc6..054b559 100644 --- a/.agents/skills/augustus/references/agent-self-assessment.md +++ b/.agents/skills/augustus/references/agent-self-assessment.md @@ -972,3 +972,6 @@ Hourly 1843 uniqueness lock: jaredpalmer/kev densify HEAD bd058057ad0a README SH **User-provided 1936 HIGH (`notes.md` §130).** sgoedecke/system-one first-sighting. SystemOne.from_pretrained. TypeSafe-compatible ≠ TypeSafe replica. mithalouni/system-one-open first-sighting. 76.7% vs Jev 86.9% *theirs*. replica ≠ TypeSafe. kotoba-lang/typed-decisions first-sighting. DeBERTa-v3-large 0.855 / 42 ms *theirs*. kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions. aisearchio 15-link census catalog ≠ endorsement. soft scores ≠ hard gates. SHA move is not a replica. Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. User-provided 1936 uniqueness lock: sgoedecke/system-one 20★ HEAD ebde2a2db706 README SHA d331b567e2c3; SystemOne.from_pretrained; Batched single-token choice inference; TypeSafe-compatible; cache_prefix=True; LICENSE absent; sgoedecke/system-one ≠ mithalouni/system-one-open ≠ KathanModh259/system-one ≠ babybear-labs/system-one; TypeSafe-compatible ≠ TypeSafe replica; mithalouni/system-one-open 18★ MIT HEAD 77f1f7cccf8a README SHA 535f33028a68 LICENSE SHA 2f6f2cf1064e; Gemma 4 E2B / Gemma 3 270M Modal; 76.7% vs Jev 86.9% strict common subset *theirs*; 97 ms H100 *theirs*; 74.8% held-out *theirs*; replica ≠ TypeSafe; HF upload pending; kotoba-lang/typed-decisions 1★ Apache-2.0 HEAD 10d7834d3b99 README SHA 4d6bbf4c4e44 LICENSE SHA 513bb5e3cb4c; ModernBERT / DeBERTa / LLaDA-MoE; DeBERTa-v3-large 0.855 / 42 ms *theirs*; ModernBERT-base 0.717 / 68 ms *theirs*; LLaDA-MoE 0.835 / 676 ms *theirs*; kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions; encoder class member not Jev replica; aisearchio 15-link census catalog ≠ endorsement; 12 already carded 3 gaps this fold; soft scores ≠ hard gates; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §130 + +**Open-Jev densify (`notes.md` §125).** DENSIFY the original 1441 card, not a sibling first sighting. HEAD 4933ee84951f README SHA ce1a587219e4. LoRA + scalar head + calibration temperature. not merged base models. customer-service P50 85.03 vs Jev 295.26 *theirs*. 1024/32 slower 1015.90 vs 301.37 *theirs*. systems latency ≠ semantic equivalence. Open-Jev TREC pending. hard acc ≠ calibrated Noul. type-valid ≠ exact. LoRA ≠ RLCD replica. Qwen/Qwen3.8-27B ≠ Archer. SHA move is not a replica. Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. +User-provided Open-Jev densify uniqueness lock: Zefan-Cai/Open-Jev densify HEAD 4933ee84951f README SHA ce1a587219e4; pushed 2026-09-21T01:34Z; Astra TREC commit 1dd56990be7e pushed 2026-09-21T01:17Z; live 3★ (was 0★; star-noise is not the fold); LoRA adapters plus trained scalar decision head and calibration temperature; not merged base models; dataset ZefanCai/Open-Jev rev c67699e13d0a; Open-Jev-2B rev 0c7aa498b162; Open-Jev-9B rev 47e966881e48; 27B still in progress; Independent of TypeSafe; no RLCD/parity claims; LoRA ≠ RLCD replica; customer-service P50 local HTTP 85.03 ms vs Jev HTTPS 295.26 ms *theirs*; 1024 tokens/32 candidates Open-Jev slower 1015.90 vs 301.37 *theirs*; prefix caching experimental/off by default; CUDA prefix caching exceeded tolerance on 9/11 workloads; systems latency ≠ semantic equivalence; GPT Luna P50 918.13 ms Astra 1938.39 ms *theirs*; TREC-DL Jev/Luna/Astra completed; Open-Jev TREC pending; 80,816 training rows; 2B 94.71% / OOD 86.02%; 9B 97.54% / 91.97% *theirs* not Harbor; hard acc ≠ calibrated Noul; type-valid ≠ exact; Qwen/Qwen3.8-27B ≠ Archer; website https://zefan-cai.github.io/open-jev/; launch X thread https://x.com/Zefan_Cai/status/2101782158658695388 https://x.com/Zefan_Cai/status/2101786019607740436 https://x.com/Zefan_Cai/status/2101789698947793231; densify §125 not a sibling first sighting; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §125 diff --git a/.agents/skills/augustus/references/applied-mappings.md b/.agents/skills/augustus/references/applied-mappings.md index f4c2023..abf578a 100644 --- a/.agents/skills/augustus/references/applied-mappings.md +++ b/.agents/skills/augustus/references/applied-mappings.md @@ -2551,3 +2551,6 @@ Hourly 1843 uniqueness lock: jaredpalmer/kev densify HEAD bd058057ad0a README SH **User-provided 1936 HIGH (`notes.md` §130).** sgoedecke/system-one first-sighting. SystemOne.from_pretrained. TypeSafe-compatible ≠ TypeSafe replica. mithalouni/system-one-open first-sighting. 76.7% vs Jev 86.9% *theirs*. replica ≠ TypeSafe. kotoba-lang/typed-decisions first-sighting. DeBERTa-v3-large 0.855 / 42 ms *theirs*. kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions. aisearchio 15-link census catalog ≠ endorsement. soft scores ≠ hard gates. SHA move is not a replica. Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. User-provided 1936 uniqueness lock: sgoedecke/system-one 20★ HEAD ebde2a2db706 README SHA d331b567e2c3; SystemOne.from_pretrained; Batched single-token choice inference; TypeSafe-compatible; cache_prefix=True; LICENSE absent; sgoedecke/system-one ≠ mithalouni/system-one-open ≠ KathanModh259/system-one ≠ babybear-labs/system-one; TypeSafe-compatible ≠ TypeSafe replica; mithalouni/system-one-open 18★ MIT HEAD 77f1f7cccf8a README SHA 535f33028a68 LICENSE SHA 2f6f2cf1064e; Gemma 4 E2B / Gemma 3 270M Modal; 76.7% vs Jev 86.9% strict common subset *theirs*; 97 ms H100 *theirs*; 74.8% held-out *theirs*; replica ≠ TypeSafe; HF upload pending; kotoba-lang/typed-decisions 1★ Apache-2.0 HEAD 10d7834d3b99 README SHA 4d6bbf4c4e44 LICENSE SHA 513bb5e3cb4c; ModernBERT / DeBERTa / LLaDA-MoE; DeBERTa-v3-large 0.855 / 42 ms *theirs*; ModernBERT-base 0.717 / 68 ms *theirs*; LLaDA-MoE 0.835 / 676 ms *theirs*; kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions; encoder class member not Jev replica; aisearchio 15-link census catalog ≠ endorsement; 12 already carded 3 gaps this fold; soft scores ≠ hard gates; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §130 + +**Open-Jev densify (`notes.md` §125).** DENSIFY the original 1441 card, not a sibling first sighting. HEAD 4933ee84951f README SHA ce1a587219e4. LoRA + scalar head + calibration temperature. not merged base models. customer-service P50 85.03 vs Jev 295.26 *theirs*. 1024/32 slower 1015.90 vs 301.37 *theirs*. systems latency ≠ semantic equivalence. Open-Jev TREC pending. hard acc ≠ calibrated Noul. type-valid ≠ exact. LoRA ≠ RLCD replica. Qwen/Qwen3.8-27B ≠ Archer. SHA move is not a replica. Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. +User-provided Open-Jev densify uniqueness lock: Zefan-Cai/Open-Jev densify HEAD 4933ee84951f README SHA ce1a587219e4; pushed 2026-09-21T01:34Z; Astra TREC commit 1dd56990be7e pushed 2026-09-21T01:17Z; live 3★ (was 0★; star-noise is not the fold); LoRA adapters plus trained scalar decision head and calibration temperature; not merged base models; dataset ZefanCai/Open-Jev rev c67699e13d0a; Open-Jev-2B rev 0c7aa498b162; Open-Jev-9B rev 47e966881e48; 27B still in progress; Independent of TypeSafe; no RLCD/parity claims; LoRA ≠ RLCD replica; customer-service P50 local HTTP 85.03 ms vs Jev HTTPS 295.26 ms *theirs*; 1024 tokens/32 candidates Open-Jev slower 1015.90 vs 301.37 *theirs*; prefix caching experimental/off by default; CUDA prefix caching exceeded tolerance on 9/11 workloads; systems latency ≠ semantic equivalence; GPT Luna P50 918.13 ms Astra 1938.39 ms *theirs*; TREC-DL Jev/Luna/Astra completed; Open-Jev TREC pending; 80,816 training rows; 2B 94.71% / OOD 86.02%; 9B 97.54% / 91.97% *theirs* not Harbor; hard acc ≠ calibrated Noul; type-valid ≠ exact; Qwen/Qwen3.8-27B ≠ Archer; website https://zefan-cai.github.io/open-jev/; launch X thread https://x.com/Zefan_Cai/status/2101782158658695388 https://x.com/Zefan_Cai/status/2101786019607740436 https://x.com/Zefan_Cai/status/2101789698947793231; densify §125 not a sibling first sighting; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §125 diff --git a/.agents/skills/augustus/references/composition-algebra.md b/.agents/skills/augustus/references/composition-algebra.md index f730760..3b1ccb6 100644 --- a/.agents/skills/augustus/references/composition-algebra.md +++ b/.agents/skills/augustus/references/composition-algebra.md @@ -2553,10 +2553,22 @@ Soft Noul ≠ hard safety. 2B 94.71% 9B 97.54% hard test *theirs*. LoRA ≠ RLCD replica. Zefan-Cai/Open-Jev ≠ TheoLeeCJ/openjev ≠ razorback16/openjev. + Since last look 2026-09-21 densify: HEAD 4933ee84951f README SHA + ce1a587219e4. not merged base models. Independent of TypeSafe. + customer-service P50 local HTTP 85.03 ms vs Jev HTTPS 295.26 ms + *theirs*. 1024 tokens/32 candidates Open-Jev slower 1015.90 vs + 301.37 *theirs*. systems latency ≠ semantic equivalence. + prefix caching experimental/off by default. Full cards: `judgment-class.md`, `validation.md`. 426. **Open-Jev OOD + 27B in progress** (ZefanCai/Open-Jev-2B / 9B): 2B OOD 86.02% 9B OOD 91.97% *theirs*. 80,816 training rows. 27B still in progress. Hub 2B 0c7aa498b162 / 9B 47e966881e48. + Since last look: dataset ZefanCai/Open-Jev rev c67699e13d0a. + TREC-DL Jev/Luna/Astra completed. Open-Jev TREC pending. + hard acc ≠ calibrated Noul. type-valid ≠ exact. + GPT Luna P50 918.13 ms Astra 1938.39 ms *theirs*. + website https://zefan-cai.github.io/open-jev/. + densify §125 not a sibling first sighting. Full cards: `validation.md`. 427. **jeq pipe judgments** (cristianoliveira/jeq): 3★ HEAD 44ea80c90903. intelligence you can pipe. @@ -3098,3 +3110,5 @@ Soft Noul ≠ hard safety. **User-provided 1936 HIGH (`notes.md` §130).** sgoedecke/system-one first-sighting. SystemOne.from_pretrained. TypeSafe-compatible ≠ TypeSafe replica. mithalouni/system-one-open first-sighting. 76.7% vs Jev 86.9% *theirs*. replica ≠ TypeSafe. kotoba-lang/typed-decisions first-sighting. DeBERTa-v3-large 0.855 / 42 ms *theirs*. kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions. aisearchio 15-link census catalog ≠ endorsement. soft scores ≠ hard gates. SHA move is not a replica. Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. User-provided 1936 uniqueness lock: sgoedecke/system-one 20★ HEAD ebde2a2db706 README SHA d331b567e2c3; SystemOne.from_pretrained; Batched single-token choice inference; TypeSafe-compatible; cache_prefix=True; LICENSE absent; sgoedecke/system-one ≠ mithalouni/system-one-open ≠ KathanModh259/system-one ≠ babybear-labs/system-one; TypeSafe-compatible ≠ TypeSafe replica; mithalouni/system-one-open 18★ MIT HEAD 77f1f7cccf8a README SHA 535f33028a68 LICENSE SHA 2f6f2cf1064e; Gemma 4 E2B / Gemma 3 270M Modal; 76.7% vs Jev 86.9% strict common subset *theirs*; 97 ms H100 *theirs*; 74.8% held-out *theirs*; replica ≠ TypeSafe; HF upload pending; kotoba-lang/typed-decisions 1★ Apache-2.0 HEAD 10d7834d3b99 README SHA 4d6bbf4c4e44 LICENSE SHA 513bb5e3cb4c; ModernBERT / DeBERTa / LLaDA-MoE; DeBERTa-v3-large 0.855 / 42 ms *theirs*; ModernBERT-base 0.717 / 68 ms *theirs*; LLaDA-MoE 0.835 / 676 ms *theirs*; kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions; encoder class member not Jev replica; aisearchio 15-link census catalog ≠ endorsement; 12 already carded 3 gaps this fold; soft scores ≠ hard gates; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §130 +**Open-Jev densify (`notes.md` §125).** DENSIFY the original 1441 card, not a sibling first sighting. HEAD 4933ee84951f README SHA ce1a587219e4. LoRA + scalar head + calibration temperature. not merged base models. customer-service P50 85.03 vs Jev 295.26 *theirs*. 1024/32 slower 1015.90 vs 301.37 *theirs*. systems latency ≠ semantic equivalence. Open-Jev TREC pending. hard acc ≠ calibrated Noul. type-valid ≠ exact. LoRA ≠ RLCD replica. Qwen/Qwen3.8-27B ≠ Archer. SHA move is not a replica. Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. +User-provided Open-Jev densify uniqueness lock: Zefan-Cai/Open-Jev densify HEAD 4933ee84951f README SHA ce1a587219e4; pushed 2026-09-21T01:34Z; Astra TREC commit 1dd56990be7e pushed 2026-09-21T01:17Z; live 3★ (was 0★; star-noise is not the fold); LoRA adapters plus trained scalar decision head and calibration temperature; not merged base models; dataset ZefanCai/Open-Jev rev c67699e13d0a; Open-Jev-2B rev 0c7aa498b162; Open-Jev-9B rev 47e966881e48; 27B still in progress; Independent of TypeSafe; no RLCD/parity claims; LoRA ≠ RLCD replica; customer-service P50 local HTTP 85.03 ms vs Jev HTTPS 295.26 ms *theirs*; 1024 tokens/32 candidates Open-Jev slower 1015.90 vs 301.37 *theirs*; prefix caching experimental/off by default; CUDA prefix caching exceeded tolerance on 9/11 workloads; systems latency ≠ semantic equivalence; GPT Luna P50 918.13 ms Astra 1938.39 ms *theirs*; TREC-DL Jev/Luna/Astra completed; Open-Jev TREC pending; 80,816 training rows; 2B 94.71% / OOD 86.02%; 9B 97.54% / 91.97% *theirs* not Harbor; hard acc ≠ calibrated Noul; type-valid ≠ exact; Qwen/Qwen3.8-27B ≠ Archer; website https://zefan-cai.github.io/open-jev/; launch X thread https://x.com/Zefan_Cai/status/2101782158658695388 https://x.com/Zefan_Cai/status/2101786019607740436 https://x.com/Zefan_Cai/status/2101789698947793231; densify §125 not a sibling first sighting; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §125 diff --git a/.agents/skills/augustus/references/faq.md b/.agents/skills/augustus/references/faq.md index 8acb596..ce389c7 100644 --- a/.agents/skills/augustus/references/faq.md +++ b/.agents/skills/augustus/references/faq.md @@ -3761,6 +3761,18 @@ catalog ≠ endorsement. game success ≠ calibrated Noul. Do not reopen or amend PR #23–#51. `invented_signal: false`. `notes.md` §129. +## Treat Open-Jev 85 ms as Jev parity? Treat 94.71% as a Noul? Treat LoRA as RLCD? + +No. systems latency ≠ semantic equivalence. hard acc ≠ calibrated Noul. +not merged base models. LoRA ≠ RLCD replica. type-valid ≠ exact. +customer-service P50 local HTTP 85.03 ms vs Jev HTTPS 295.26 ms *theirs*. +1024 tokens/32 candidates Open-Jev slower 1015.90 vs 301.37 *theirs*. +prefix caching experimental/off by default. TREC-DL Jev/Luna/Astra completed. +Open-Jev TREC pending. Qwen/Qwen3.8-27B ≠ Archer. +Zefan-Cai/Open-Jev ≠ TheoLeeCJ/openjev ≠ razorback16/openjev. +Do not reopen or amend PR #23–#52. +`invented_signal: false`. `notes.md` §125. + **Hourly 1843 HIGH (`notes.md` §129).** kev own-data JSONL densify. --init_from warm-start LoRA/head PR #9. from-scratch ≠ warm-start. JSONL labels ≠ Harbor. 4B new-source 0.794/0.832 *theirs*. 9B new-source 0.812/0.837 *theirs*. 0.33 vs 0.84 vs 0.83/0.88 *theirs*. reconstruction ≠ replica. assay-001 split verdict. catalog ≠ endorsement. SHA move is not a replica. Do not reopen or amend PR #23–#51. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. Hourly 1843 uniqueness lock: jaredpalmer/kev densify HEAD bd058057ad0a README SHA 84b872488915; Fine-tuning on your own data; --data JSONL; --init_from warm-start LoRA/head PR #9; Kev-0.8B 4B 9B Qwen3.5 family; 4B new-source 0.794/0.832 *theirs*; 9B new-source 0.812/0.837 *theirs*; 0.33 vs 0.84 vs 0.83/0.88 *theirs*; from-scratch ≠ warm-start; JSONL labels ≠ Harbor; Kev-0.5B card Qwen3.5 family pointer; No Jev outputs were used for training; option order can change an answer; 8.2% ≥0.9 on wrong *theirs*; Kev-9B 7.5% ≥0.9 on wrong *theirs*; dabit3/jev-experiments densify 340★; simota/tenbin densify neighbor skill; Promethe-us/awesome-jev ≠ MrJev/awesome-jev ≠ yibie/awesome-jev; kyegomez/open-jev reconstruction ≠ replica; unofficial research implementation with random weights; kyegomez/open-jev ≠ razorback16/openjev ≠ TheoLeeCJ/openjev ≠ Zefan-Cai/Open-Jev ≠ Shalimov04/open-jev; jourdanlabs/assay-001 split verdict; CLINC150 ECE 0.0204 *theirs*; Banking77 ECE 0.0936 *theirs*; 8,576 responses zero type errors *theirs*; brnyxx/jev-ra 3-5x / ~300 ms *theirs*; 8.50× Wikipedia *theirs*; ThePFMind/jev-mcp ≠ jkudish/jev-mcp ≠ burnigtm/jev-mcp; namenu/pi-jev-effort ≠ TheoOliveira/pi-jev; samatv256/mini-Jev ≠ r-ms/mini-jev; comoc/jev-minesweeper ≠ EnesYilmazcode/JevMinesweeper; game success ≠ calibrated Noul; Nutlope/jev-fraud Kimi K3; jeffloo886/jev-notion; hf:akhilaaa3/openjev-v1-allmix-r512-merged ≠ hf:akhilaaa3/openjev-r512-handoff-demo; serving substrate ≠ calibrated replica; catalog ≠ endorsement; SHA move is not a replica; wire-compat ≠ logit-equiv; Qwen3.5 ≠ Archer; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51; notes.md §129 @@ -3776,3 +3788,5 @@ Do not reopen or amend PR #23–#52. **User-provided 1936 HIGH (`notes.md` §130).** sgoedecke/system-one first-sighting. SystemOne.from_pretrained. TypeSafe-compatible ≠ TypeSafe replica. mithalouni/system-one-open first-sighting. 76.7% vs Jev 86.9% *theirs*. replica ≠ TypeSafe. kotoba-lang/typed-decisions first-sighting. DeBERTa-v3-large 0.855 / 42 ms *theirs*. kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions. aisearchio 15-link census catalog ≠ endorsement. soft scores ≠ hard gates. SHA move is not a replica. Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. User-provided 1936 uniqueness lock: sgoedecke/system-one 20★ HEAD ebde2a2db706 README SHA d331b567e2c3; SystemOne.from_pretrained; Batched single-token choice inference; TypeSafe-compatible; cache_prefix=True; LICENSE absent; sgoedecke/system-one ≠ mithalouni/system-one-open ≠ KathanModh259/system-one ≠ babybear-labs/system-one; TypeSafe-compatible ≠ TypeSafe replica; mithalouni/system-one-open 18★ MIT HEAD 77f1f7cccf8a README SHA 535f33028a68 LICENSE SHA 2f6f2cf1064e; Gemma 4 E2B / Gemma 3 270M Modal; 76.7% vs Jev 86.9% strict common subset *theirs*; 97 ms H100 *theirs*; 74.8% held-out *theirs*; replica ≠ TypeSafe; HF upload pending; kotoba-lang/typed-decisions 1★ Apache-2.0 HEAD 10d7834d3b99 README SHA 4d6bbf4c4e44 LICENSE SHA 513bb5e3cb4c; ModernBERT / DeBERTa / LLaDA-MoE; DeBERTa-v3-large 0.855 / 42 ms *theirs*; ModernBERT-base 0.717 / 68 ms *theirs*; LLaDA-MoE 0.835 / 676 ms *theirs*; kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions; encoder class member not Jev replica; aisearchio 15-link census catalog ≠ endorsement; 12 already carded 3 gaps this fold; soft scores ≠ hard gates; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §130 +**Open-Jev densify (`notes.md` §125).** DENSIFY the original 1441 card, not a sibling first sighting. HEAD 4933ee84951f README SHA ce1a587219e4. LoRA + scalar head + calibration temperature. not merged base models. customer-service P50 85.03 vs Jev 295.26 *theirs*. 1024/32 slower 1015.90 vs 301.37 *theirs*. systems latency ≠ semantic equivalence. Open-Jev TREC pending. hard acc ≠ calibrated Noul. type-valid ≠ exact. LoRA ≠ RLCD replica. Qwen/Qwen3.8-27B ≠ Archer. SHA move is not a replica. Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. +User-provided Open-Jev densify uniqueness lock: Zefan-Cai/Open-Jev densify HEAD 4933ee84951f README SHA ce1a587219e4; pushed 2026-09-21T01:34Z; Astra TREC commit 1dd56990be7e pushed 2026-09-21T01:17Z; live 3★ (was 0★; star-noise is not the fold); LoRA adapters plus trained scalar decision head and calibration temperature; not merged base models; dataset ZefanCai/Open-Jev rev c67699e13d0a; Open-Jev-2B rev 0c7aa498b162; Open-Jev-9B rev 47e966881e48; 27B still in progress; Independent of TypeSafe; no RLCD/parity claims; LoRA ≠ RLCD replica; customer-service P50 local HTTP 85.03 ms vs Jev HTTPS 295.26 ms *theirs*; 1024 tokens/32 candidates Open-Jev slower 1015.90 vs 301.37 *theirs*; prefix caching experimental/off by default; CUDA prefix caching exceeded tolerance on 9/11 workloads; systems latency ≠ semantic equivalence; GPT Luna P50 918.13 ms Astra 1938.39 ms *theirs*; TREC-DL Jev/Luna/Astra completed; Open-Jev TREC pending; 80,816 training rows; 2B 94.71% / OOD 86.02%; 9B 97.54% / 91.97% *theirs* not Harbor; hard acc ≠ calibrated Noul; type-valid ≠ exact; Qwen/Qwen3.8-27B ≠ Archer; website https://zefan-cai.github.io/open-jev/; launch X thread https://x.com/Zefan_Cai/status/2101782158658695388 https://x.com/Zefan_Cai/status/2101786019607740436 https://x.com/Zefan_Cai/status/2101789698947793231; densify §125 not a sibling first sighting; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §125 diff --git a/.agents/skills/augustus/references/formal-methods.md b/.agents/skills/augustus/references/formal-methods.md index ba9cdcc..2f748bd 100644 --- a/.agents/skills/augustus/references/formal-methods.md +++ b/.agents/skills/augustus/references/formal-methods.md @@ -1401,3 +1401,6 @@ Hourly 1843 uniqueness lock: jaredpalmer/kev densify HEAD bd058057ad0a README SH **User-provided 1936 HIGH (`notes.md` §130).** sgoedecke/system-one first-sighting. SystemOne.from_pretrained. TypeSafe-compatible ≠ TypeSafe replica. mithalouni/system-one-open first-sighting. 76.7% vs Jev 86.9% *theirs*. replica ≠ TypeSafe. kotoba-lang/typed-decisions first-sighting. DeBERTa-v3-large 0.855 / 42 ms *theirs*. kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions. aisearchio 15-link census catalog ≠ endorsement. soft scores ≠ hard gates. SHA move is not a replica. Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. User-provided 1936 uniqueness lock: sgoedecke/system-one 20★ HEAD ebde2a2db706 README SHA d331b567e2c3; SystemOne.from_pretrained; Batched single-token choice inference; TypeSafe-compatible; cache_prefix=True; LICENSE absent; sgoedecke/system-one ≠ mithalouni/system-one-open ≠ KathanModh259/system-one ≠ babybear-labs/system-one; TypeSafe-compatible ≠ TypeSafe replica; mithalouni/system-one-open 18★ MIT HEAD 77f1f7cccf8a README SHA 535f33028a68 LICENSE SHA 2f6f2cf1064e; Gemma 4 E2B / Gemma 3 270M Modal; 76.7% vs Jev 86.9% strict common subset *theirs*; 97 ms H100 *theirs*; 74.8% held-out *theirs*; replica ≠ TypeSafe; HF upload pending; kotoba-lang/typed-decisions 1★ Apache-2.0 HEAD 10d7834d3b99 README SHA 4d6bbf4c4e44 LICENSE SHA 513bb5e3cb4c; ModernBERT / DeBERTa / LLaDA-MoE; DeBERTa-v3-large 0.855 / 42 ms *theirs*; ModernBERT-base 0.717 / 68 ms *theirs*; LLaDA-MoE 0.835 / 676 ms *theirs*; kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions; encoder class member not Jev replica; aisearchio 15-link census catalog ≠ endorsement; 12 already carded 3 gaps this fold; soft scores ≠ hard gates; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §130 + +**Open-Jev densify (`notes.md` §125).** DENSIFY the original 1441 card, not a sibling first sighting. HEAD 4933ee84951f README SHA ce1a587219e4. LoRA + scalar head + calibration temperature. not merged base models. customer-service P50 85.03 vs Jev 295.26 *theirs*. 1024/32 slower 1015.90 vs 301.37 *theirs*. systems latency ≠ semantic equivalence. Open-Jev TREC pending. hard acc ≠ calibrated Noul. type-valid ≠ exact. LoRA ≠ RLCD replica. Qwen/Qwen3.8-27B ≠ Archer. SHA move is not a replica. Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. +User-provided Open-Jev densify uniqueness lock: Zefan-Cai/Open-Jev densify HEAD 4933ee84951f README SHA ce1a587219e4; pushed 2026-09-21T01:34Z; Astra TREC commit 1dd56990be7e pushed 2026-09-21T01:17Z; live 3★ (was 0★; star-noise is not the fold); LoRA adapters plus trained scalar decision head and calibration temperature; not merged base models; dataset ZefanCai/Open-Jev rev c67699e13d0a; Open-Jev-2B rev 0c7aa498b162; Open-Jev-9B rev 47e966881e48; 27B still in progress; Independent of TypeSafe; no RLCD/parity claims; LoRA ≠ RLCD replica; customer-service P50 local HTTP 85.03 ms vs Jev HTTPS 295.26 ms *theirs*; 1024 tokens/32 candidates Open-Jev slower 1015.90 vs 301.37 *theirs*; prefix caching experimental/off by default; CUDA prefix caching exceeded tolerance on 9/11 workloads; systems latency ≠ semantic equivalence; GPT Luna P50 918.13 ms Astra 1938.39 ms *theirs*; TREC-DL Jev/Luna/Astra completed; Open-Jev TREC pending; 80,816 training rows; 2B 94.71% / OOD 86.02%; 9B 97.54% / 91.97% *theirs* not Harbor; hard acc ≠ calibrated Noul; type-valid ≠ exact; Qwen/Qwen3.8-27B ≠ Archer; website https://zefan-cai.github.io/open-jev/; launch X thread https://x.com/Zefan_Cai/status/2101782158658695388 https://x.com/Zefan_Cai/status/2101786019607740436 https://x.com/Zefan_Cai/status/2101789698947793231; densify §125 not a sibling first sighting; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §125 diff --git a/.agents/skills/augustus/references/formal-semi-formal.md b/.agents/skills/augustus/references/formal-semi-formal.md index ac39893..706a934 100644 --- a/.agents/skills/augustus/references/formal-semi-formal.md +++ b/.agents/skills/augustus/references/formal-semi-formal.md @@ -114,3 +114,6 @@ Hourly 1843 uniqueness lock: jaredpalmer/kev densify HEAD bd058057ad0a README SH **User-provided 1936 HIGH (`notes.md` §130).** sgoedecke/system-one first-sighting. SystemOne.from_pretrained. TypeSafe-compatible ≠ TypeSafe replica. mithalouni/system-one-open first-sighting. 76.7% vs Jev 86.9% *theirs*. replica ≠ TypeSafe. kotoba-lang/typed-decisions first-sighting. DeBERTa-v3-large 0.855 / 42 ms *theirs*. kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions. aisearchio 15-link census catalog ≠ endorsement. soft scores ≠ hard gates. SHA move is not a replica. Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. User-provided 1936 uniqueness lock: sgoedecke/system-one 20★ HEAD ebde2a2db706 README SHA d331b567e2c3; SystemOne.from_pretrained; Batched single-token choice inference; TypeSafe-compatible; cache_prefix=True; LICENSE absent; sgoedecke/system-one ≠ mithalouni/system-one-open ≠ KathanModh259/system-one ≠ babybear-labs/system-one; TypeSafe-compatible ≠ TypeSafe replica; mithalouni/system-one-open 18★ MIT HEAD 77f1f7cccf8a README SHA 535f33028a68 LICENSE SHA 2f6f2cf1064e; Gemma 4 E2B / Gemma 3 270M Modal; 76.7% vs Jev 86.9% strict common subset *theirs*; 97 ms H100 *theirs*; 74.8% held-out *theirs*; replica ≠ TypeSafe; HF upload pending; kotoba-lang/typed-decisions 1★ Apache-2.0 HEAD 10d7834d3b99 README SHA 4d6bbf4c4e44 LICENSE SHA 513bb5e3cb4c; ModernBERT / DeBERTa / LLaDA-MoE; DeBERTa-v3-large 0.855 / 42 ms *theirs*; ModernBERT-base 0.717 / 68 ms *theirs*; LLaDA-MoE 0.835 / 676 ms *theirs*; kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions; encoder class member not Jev replica; aisearchio 15-link census catalog ≠ endorsement; 12 already carded 3 gaps this fold; soft scores ≠ hard gates; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §130 + +**Open-Jev densify (`notes.md` §125).** DENSIFY the original 1441 card, not a sibling first sighting. HEAD 4933ee84951f README SHA ce1a587219e4. LoRA + scalar head + calibration temperature. not merged base models. customer-service P50 85.03 vs Jev 295.26 *theirs*. 1024/32 slower 1015.90 vs 301.37 *theirs*. systems latency ≠ semantic equivalence. Open-Jev TREC pending. hard acc ≠ calibrated Noul. type-valid ≠ exact. LoRA ≠ RLCD replica. Qwen/Qwen3.8-27B ≠ Archer. SHA move is not a replica. Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. +User-provided Open-Jev densify uniqueness lock: Zefan-Cai/Open-Jev densify HEAD 4933ee84951f README SHA ce1a587219e4; pushed 2026-09-21T01:34Z; Astra TREC commit 1dd56990be7e pushed 2026-09-21T01:17Z; live 3★ (was 0★; star-noise is not the fold); LoRA adapters plus trained scalar decision head and calibration temperature; not merged base models; dataset ZefanCai/Open-Jev rev c67699e13d0a; Open-Jev-2B rev 0c7aa498b162; Open-Jev-9B rev 47e966881e48; 27B still in progress; Independent of TypeSafe; no RLCD/parity claims; LoRA ≠ RLCD replica; customer-service P50 local HTTP 85.03 ms vs Jev HTTPS 295.26 ms *theirs*; 1024 tokens/32 candidates Open-Jev slower 1015.90 vs 301.37 *theirs*; prefix caching experimental/off by default; CUDA prefix caching exceeded tolerance on 9/11 workloads; systems latency ≠ semantic equivalence; GPT Luna P50 918.13 ms Astra 1938.39 ms *theirs*; TREC-DL Jev/Luna/Astra completed; Open-Jev TREC pending; 80,816 training rows; 2B 94.71% / OOD 86.02%; 9B 97.54% / 91.97% *theirs* not Harbor; hard acc ≠ calibrated Noul; type-valid ≠ exact; Qwen/Qwen3.8-27B ≠ Archer; website https://zefan-cai.github.io/open-jev/; launch X thread https://x.com/Zefan_Cai/status/2101782158658695388 https://x.com/Zefan_Cai/status/2101786019607740436 https://x.com/Zefan_Cai/status/2101789698947793231; densify §125 not a sibling first sighting; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §125 diff --git a/.agents/skills/augustus/references/judgment-class.md b/.agents/skills/augustus/references/judgment-class.md index 6b0851d..eba2613 100644 --- a/.agents/skills/augustus/references/judgment-class.md +++ b/.agents/skills/augustus/references/judgment-class.md @@ -1517,3 +1517,5 @@ Hourly 1843 uniqueness lock: jaredpalmer/kev densify HEAD bd058057ad0a README SH Open LM logit-trick (sgoedecke/system-one) is TypeSafe-compatible ≠ TypeSafe replica. Gemma LoRA replica (mithalouni/system-one-open) is replica ≠ TypeSafe. kotoba-lang/typed-decisions is encoder class member not Jev replica. `notes.md` §130. **User-provided 1936 HIGH (`notes.md` §130).** sgoedecke/system-one first-sighting. SystemOne.from_pretrained. TypeSafe-compatible ≠ TypeSafe replica. mithalouni/system-one-open first-sighting. 76.7% vs Jev 86.9% *theirs*. replica ≠ TypeSafe. kotoba-lang/typed-decisions first-sighting. DeBERTa-v3-large 0.855 / 42 ms *theirs*. kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions. aisearchio 15-link census catalog ≠ endorsement. soft scores ≠ hard gates. SHA move is not a replica. Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. User-provided 1936 uniqueness lock: sgoedecke/system-one 20★ HEAD ebde2a2db706 README SHA d331b567e2c3; SystemOne.from_pretrained; Batched single-token choice inference; TypeSafe-compatible; cache_prefix=True; LICENSE absent; sgoedecke/system-one ≠ mithalouni/system-one-open ≠ KathanModh259/system-one ≠ babybear-labs/system-one; TypeSafe-compatible ≠ TypeSafe replica; mithalouni/system-one-open 18★ MIT HEAD 77f1f7cccf8a README SHA 535f33028a68 LICENSE SHA 2f6f2cf1064e; Gemma 4 E2B / Gemma 3 270M Modal; 76.7% vs Jev 86.9% strict common subset *theirs*; 97 ms H100 *theirs*; 74.8% held-out *theirs*; replica ≠ TypeSafe; HF upload pending; kotoba-lang/typed-decisions 1★ Apache-2.0 HEAD 10d7834d3b99 README SHA 4d6bbf4c4e44 LICENSE SHA 513bb5e3cb4c; ModernBERT / DeBERTa / LLaDA-MoE; DeBERTa-v3-large 0.855 / 42 ms *theirs*; ModernBERT-base 0.717 / 68 ms *theirs*; LLaDA-MoE 0.835 / 676 ms *theirs*; kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions; encoder class member not Jev replica; aisearchio 15-link census catalog ≠ endorsement; 12 already carded 3 gaps this fold; soft scores ≠ hard gates; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §130 +**Open-Jev densify (`notes.md` §125).** DENSIFY the original 1441 card, not a sibling first sighting. HEAD 4933ee84951f README SHA ce1a587219e4. LoRA + scalar head + calibration temperature. not merged base models. customer-service P50 85.03 vs Jev 295.26 *theirs*. 1024/32 slower 1015.90 vs 301.37 *theirs*. systems latency ≠ semantic equivalence. Open-Jev TREC pending. hard acc ≠ calibrated Noul. type-valid ≠ exact. LoRA ≠ RLCD replica. Qwen/Qwen3.8-27B ≠ Archer. SHA move is not a replica. Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. +User-provided Open-Jev densify uniqueness lock: Zefan-Cai/Open-Jev densify HEAD 4933ee84951f README SHA ce1a587219e4; pushed 2026-09-21T01:34Z; Astra TREC commit 1dd56990be7e pushed 2026-09-21T01:17Z; live 3★ (was 0★; star-noise is not the fold); LoRA adapters plus trained scalar decision head and calibration temperature; not merged base models; dataset ZefanCai/Open-Jev rev c67699e13d0a; Open-Jev-2B rev 0c7aa498b162; Open-Jev-9B rev 47e966881e48; 27B still in progress; Independent of TypeSafe; no RLCD/parity claims; LoRA ≠ RLCD replica; customer-service P50 local HTTP 85.03 ms vs Jev HTTPS 295.26 ms *theirs*; 1024 tokens/32 candidates Open-Jev slower 1015.90 vs 301.37 *theirs*; prefix caching experimental/off by default; CUDA prefix caching exceeded tolerance on 9/11 workloads; systems latency ≠ semantic equivalence; GPT Luna P50 918.13 ms Astra 1938.39 ms *theirs*; TREC-DL Jev/Luna/Astra completed; Open-Jev TREC pending; 80,816 training rows; 2B 94.71% / OOD 86.02%; 9B 97.54% / 91.97% *theirs* not Harbor; hard acc ≠ calibrated Noul; type-valid ≠ exact; Qwen/Qwen3.8-27B ≠ Archer; website https://zefan-cai.github.io/open-jev/; launch X thread https://x.com/Zefan_Cai/status/2101782158658695388 https://x.com/Zefan_Cai/status/2101786019607740436 https://x.com/Zefan_Cai/status/2101789698947793231; densify §125 not a sibling first sighting; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §125 diff --git a/.agents/skills/augustus/references/mappings.md b/.agents/skills/augustus/references/mappings.md index 3d42331..651ad8a 100644 --- a/.agents/skills/augustus/references/mappings.md +++ b/.agents/skills/augustus/references/mappings.md @@ -2506,3 +2506,6 @@ Hourly 1843 uniqueness lock: jaredpalmer/kev densify HEAD bd058057ad0a README SH **User-provided 1936 HIGH (`notes.md` §130).** sgoedecke/system-one first-sighting. SystemOne.from_pretrained. TypeSafe-compatible ≠ TypeSafe replica. mithalouni/system-one-open first-sighting. 76.7% vs Jev 86.9% *theirs*. replica ≠ TypeSafe. kotoba-lang/typed-decisions first-sighting. DeBERTa-v3-large 0.855 / 42 ms *theirs*. kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions. aisearchio 15-link census catalog ≠ endorsement. soft scores ≠ hard gates. SHA move is not a replica. Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. User-provided 1936 uniqueness lock: sgoedecke/system-one 20★ HEAD ebde2a2db706 README SHA d331b567e2c3; SystemOne.from_pretrained; Batched single-token choice inference; TypeSafe-compatible; cache_prefix=True; LICENSE absent; sgoedecke/system-one ≠ mithalouni/system-one-open ≠ KathanModh259/system-one ≠ babybear-labs/system-one; TypeSafe-compatible ≠ TypeSafe replica; mithalouni/system-one-open 18★ MIT HEAD 77f1f7cccf8a README SHA 535f33028a68 LICENSE SHA 2f6f2cf1064e; Gemma 4 E2B / Gemma 3 270M Modal; 76.7% vs Jev 86.9% strict common subset *theirs*; 97 ms H100 *theirs*; 74.8% held-out *theirs*; replica ≠ TypeSafe; HF upload pending; kotoba-lang/typed-decisions 1★ Apache-2.0 HEAD 10d7834d3b99 README SHA 4d6bbf4c4e44 LICENSE SHA 513bb5e3cb4c; ModernBERT / DeBERTa / LLaDA-MoE; DeBERTa-v3-large 0.855 / 42 ms *theirs*; ModernBERT-base 0.717 / 68 ms *theirs*; LLaDA-MoE 0.835 / 676 ms *theirs*; kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions; encoder class member not Jev replica; aisearchio 15-link census catalog ≠ endorsement; 12 already carded 3 gaps this fold; soft scores ≠ hard gates; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §130 + +**Open-Jev densify (`notes.md` §125).** DENSIFY the original 1441 card, not a sibling first sighting. HEAD 4933ee84951f README SHA ce1a587219e4. LoRA + scalar head + calibration temperature. not merged base models. customer-service P50 85.03 vs Jev 295.26 *theirs*. 1024/32 slower 1015.90 vs 301.37 *theirs*. systems latency ≠ semantic equivalence. Open-Jev TREC pending. hard acc ≠ calibrated Noul. type-valid ≠ exact. LoRA ≠ RLCD replica. Qwen/Qwen3.8-27B ≠ Archer. SHA move is not a replica. Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. +User-provided Open-Jev densify uniqueness lock: Zefan-Cai/Open-Jev densify HEAD 4933ee84951f README SHA ce1a587219e4; pushed 2026-09-21T01:34Z; Astra TREC commit 1dd56990be7e pushed 2026-09-21T01:17Z; live 3★ (was 0★; star-noise is not the fold); LoRA adapters plus trained scalar decision head and calibration temperature; not merged base models; dataset ZefanCai/Open-Jev rev c67699e13d0a; Open-Jev-2B rev 0c7aa498b162; Open-Jev-9B rev 47e966881e48; 27B still in progress; Independent of TypeSafe; no RLCD/parity claims; LoRA ≠ RLCD replica; customer-service P50 local HTTP 85.03 ms vs Jev HTTPS 295.26 ms *theirs*; 1024 tokens/32 candidates Open-Jev slower 1015.90 vs 301.37 *theirs*; prefix caching experimental/off by default; CUDA prefix caching exceeded tolerance on 9/11 workloads; systems latency ≠ semantic equivalence; GPT Luna P50 918.13 ms Astra 1938.39 ms *theirs*; TREC-DL Jev/Luna/Astra completed; Open-Jev TREC pending; 80,816 training rows; 2B 94.71% / OOD 86.02%; 9B 97.54% / 91.97% *theirs* not Harbor; hard acc ≠ calibrated Noul; type-valid ≠ exact; Qwen/Qwen3.8-27B ≠ Archer; website https://zefan-cai.github.io/open-jev/; launch X thread https://x.com/Zefan_Cai/status/2101782158658695388 https://x.com/Zefan_Cai/status/2101786019607740436 https://x.com/Zefan_Cai/status/2101789698947793231; densify §125 not a sibling first sighting; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §125 diff --git a/.agents/skills/augustus/references/mental-models.md b/.agents/skills/augustus/references/mental-models.md index b4541b0..c734edb 100644 --- a/.agents/skills/augustus/references/mental-models.md +++ b/.agents/skills/augustus/references/mental-models.md @@ -3082,6 +3082,12 @@ LoRA ≠ RLCD replica. 2B 94.71% 9B 97.54% hard test *theirs*. 27B still in progress. pass-min 0.8 still soft. JEQ does not own actions. AndyInQtr/laya-coreai CoreML serving substrate ≠ calibrated replica. SHA move is not a replica. Do not copy keys. +Since last look 2026-09-21 Open-Jev densify: not merged base models. +customer-service P50 85.03 vs Jev 295.26 *theirs*. +1024/32 slower 1015.90 vs 301.37 *theirs*. +systems latency ≠ semantic equivalence. Open-Jev TREC pending. +hard acc ≠ calibrated Noul. type-valid ≠ exact. +prefix caching experimental/off by default. ## Apply 1340 (`notes.md` §124) @@ -3144,3 +3150,5 @@ SHA move is not a replica. Do not reopen or amend PR #23–#52. **User-provided 1936 HIGH (`notes.md` §130).** sgoedecke/system-one first-sighting. SystemOne.from_pretrained. TypeSafe-compatible ≠ TypeSafe replica. mithalouni/system-one-open first-sighting. 76.7% vs Jev 86.9% *theirs*. replica ≠ TypeSafe. kotoba-lang/typed-decisions first-sighting. DeBERTa-v3-large 0.855 / 42 ms *theirs*. kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions. aisearchio 15-link census catalog ≠ endorsement. soft scores ≠ hard gates. SHA move is not a replica. Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. User-provided 1936 uniqueness lock: sgoedecke/system-one 20★ HEAD ebde2a2db706 README SHA d331b567e2c3; SystemOne.from_pretrained; Batched single-token choice inference; TypeSafe-compatible; cache_prefix=True; LICENSE absent; sgoedecke/system-one ≠ mithalouni/system-one-open ≠ KathanModh259/system-one ≠ babybear-labs/system-one; TypeSafe-compatible ≠ TypeSafe replica; mithalouni/system-one-open 18★ MIT HEAD 77f1f7cccf8a README SHA 535f33028a68 LICENSE SHA 2f6f2cf1064e; Gemma 4 E2B / Gemma 3 270M Modal; 76.7% vs Jev 86.9% strict common subset *theirs*; 97 ms H100 *theirs*; 74.8% held-out *theirs*; replica ≠ TypeSafe; HF upload pending; kotoba-lang/typed-decisions 1★ Apache-2.0 HEAD 10d7834d3b99 README SHA 4d6bbf4c4e44 LICENSE SHA 513bb5e3cb4c; ModernBERT / DeBERTa / LLaDA-MoE; DeBERTa-v3-large 0.855 / 42 ms *theirs*; ModernBERT-base 0.717 / 68 ms *theirs*; LLaDA-MoE 0.835 / 676 ms *theirs*; kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions; encoder class member not Jev replica; aisearchio 15-link census catalog ≠ endorsement; 12 already carded 3 gaps this fold; soft scores ≠ hard gates; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §130 +**Open-Jev densify (`notes.md` §125).** DENSIFY the original 1441 card, not a sibling first sighting. HEAD 4933ee84951f README SHA ce1a587219e4. LoRA + scalar head + calibration temperature. not merged base models. customer-service P50 85.03 vs Jev 295.26 *theirs*. 1024/32 slower 1015.90 vs 301.37 *theirs*. systems latency ≠ semantic equivalence. Open-Jev TREC pending. hard acc ≠ calibrated Noul. type-valid ≠ exact. LoRA ≠ RLCD replica. Qwen/Qwen3.8-27B ≠ Archer. SHA move is not a replica. Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. +User-provided Open-Jev densify uniqueness lock: Zefan-Cai/Open-Jev densify HEAD 4933ee84951f README SHA ce1a587219e4; pushed 2026-09-21T01:34Z; Astra TREC commit 1dd56990be7e pushed 2026-09-21T01:17Z; live 3★ (was 0★; star-noise is not the fold); LoRA adapters plus trained scalar decision head and calibration temperature; not merged base models; dataset ZefanCai/Open-Jev rev c67699e13d0a; Open-Jev-2B rev 0c7aa498b162; Open-Jev-9B rev 47e966881e48; 27B still in progress; Independent of TypeSafe; no RLCD/parity claims; LoRA ≠ RLCD replica; customer-service P50 local HTTP 85.03 ms vs Jev HTTPS 295.26 ms *theirs*; 1024 tokens/32 candidates Open-Jev slower 1015.90 vs 301.37 *theirs*; prefix caching experimental/off by default; CUDA prefix caching exceeded tolerance on 9/11 workloads; systems latency ≠ semantic equivalence; GPT Luna P50 918.13 ms Astra 1938.39 ms *theirs*; TREC-DL Jev/Luna/Astra completed; Open-Jev TREC pending; 80,816 training rows; 2B 94.71% / OOD 86.02%; 9B 97.54% / 91.97% *theirs* not Harbor; hard acc ≠ calibrated Noul; type-valid ≠ exact; Qwen/Qwen3.8-27B ≠ Archer; website https://zefan-cai.github.io/open-jev/; launch X thread https://x.com/Zefan_Cai/status/2101782158658695388 https://x.com/Zefan_Cai/status/2101786019607740436 https://x.com/Zefan_Cai/status/2101789698947793231; densify §125 not a sibling first sighting; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §125 diff --git a/.agents/skills/augustus/references/methods-catalog.md b/.agents/skills/augustus/references/methods-catalog.md index d09de91..a857d28 100644 --- a/.agents/skills/augustus/references/methods-catalog.md +++ b/.agents/skills/augustus/references/methods-catalog.md @@ -308,3 +308,6 @@ Hourly 1843 uniqueness lock: jaredpalmer/kev densify HEAD bd058057ad0a README SH **User-provided 1936 HIGH (`notes.md` §130).** sgoedecke/system-one first-sighting. SystemOne.from_pretrained. TypeSafe-compatible ≠ TypeSafe replica. mithalouni/system-one-open first-sighting. 76.7% vs Jev 86.9% *theirs*. replica ≠ TypeSafe. kotoba-lang/typed-decisions first-sighting. DeBERTa-v3-large 0.855 / 42 ms *theirs*. kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions. aisearchio 15-link census catalog ≠ endorsement. soft scores ≠ hard gates. SHA move is not a replica. Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. User-provided 1936 uniqueness lock: sgoedecke/system-one 20★ HEAD ebde2a2db706 README SHA d331b567e2c3; SystemOne.from_pretrained; Batched single-token choice inference; TypeSafe-compatible; cache_prefix=True; LICENSE absent; sgoedecke/system-one ≠ mithalouni/system-one-open ≠ KathanModh259/system-one ≠ babybear-labs/system-one; TypeSafe-compatible ≠ TypeSafe replica; mithalouni/system-one-open 18★ MIT HEAD 77f1f7cccf8a README SHA 535f33028a68 LICENSE SHA 2f6f2cf1064e; Gemma 4 E2B / Gemma 3 270M Modal; 76.7% vs Jev 86.9% strict common subset *theirs*; 97 ms H100 *theirs*; 74.8% held-out *theirs*; replica ≠ TypeSafe; HF upload pending; kotoba-lang/typed-decisions 1★ Apache-2.0 HEAD 10d7834d3b99 README SHA 4d6bbf4c4e44 LICENSE SHA 513bb5e3cb4c; ModernBERT / DeBERTa / LLaDA-MoE; DeBERTa-v3-large 0.855 / 42 ms *theirs*; ModernBERT-base 0.717 / 68 ms *theirs*; LLaDA-MoE 0.835 / 676 ms *theirs*; kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions; encoder class member not Jev replica; aisearchio 15-link census catalog ≠ endorsement; 12 already carded 3 gaps this fold; soft scores ≠ hard gates; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §130 + +**Open-Jev densify (`notes.md` §125).** DENSIFY the original 1441 card, not a sibling first sighting. HEAD 4933ee84951f README SHA ce1a587219e4. LoRA + scalar head + calibration temperature. not merged base models. customer-service P50 85.03 vs Jev 295.26 *theirs*. 1024/32 slower 1015.90 vs 301.37 *theirs*. systems latency ≠ semantic equivalence. Open-Jev TREC pending. hard acc ≠ calibrated Noul. type-valid ≠ exact. LoRA ≠ RLCD replica. Qwen/Qwen3.8-27B ≠ Archer. SHA move is not a replica. Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. +User-provided Open-Jev densify uniqueness lock: Zefan-Cai/Open-Jev densify HEAD 4933ee84951f README SHA ce1a587219e4; pushed 2026-09-21T01:34Z; Astra TREC commit 1dd56990be7e pushed 2026-09-21T01:17Z; live 3★ (was 0★; star-noise is not the fold); LoRA adapters plus trained scalar decision head and calibration temperature; not merged base models; dataset ZefanCai/Open-Jev rev c67699e13d0a; Open-Jev-2B rev 0c7aa498b162; Open-Jev-9B rev 47e966881e48; 27B still in progress; Independent of TypeSafe; no RLCD/parity claims; LoRA ≠ RLCD replica; customer-service P50 local HTTP 85.03 ms vs Jev HTTPS 295.26 ms *theirs*; 1024 tokens/32 candidates Open-Jev slower 1015.90 vs 301.37 *theirs*; prefix caching experimental/off by default; CUDA prefix caching exceeded tolerance on 9/11 workloads; systems latency ≠ semantic equivalence; GPT Luna P50 918.13 ms Astra 1938.39 ms *theirs*; TREC-DL Jev/Luna/Astra completed; Open-Jev TREC pending; 80,816 training rows; 2B 94.71% / OOD 86.02%; 9B 97.54% / 91.97% *theirs* not Harbor; hard acc ≠ calibrated Noul; type-valid ≠ exact; Qwen/Qwen3.8-27B ≠ Archer; website https://zefan-cai.github.io/open-jev/; launch X thread https://x.com/Zefan_Cai/status/2101782158658695388 https://x.com/Zefan_Cai/status/2101786019607740436 https://x.com/Zefan_Cai/status/2101789698947793231; densify §125 not a sibling first sighting; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §125 diff --git a/.agents/skills/augustus/references/mixed-architecture.md b/.agents/skills/augustus/references/mixed-architecture.md index c438245..0021707 100644 --- a/.agents/skills/augustus/references/mixed-architecture.md +++ b/.agents/skills/augustus/references/mixed-architecture.md @@ -1479,6 +1479,8 @@ Hourly 1643 uniqueness lock: razorback16/openjev densify HEAD febf02e88989 READM Apply 1746 (`notes.md` §128): Facts go to code. Judgments go to Jev. Only facts can block. Jev never blocks. /d routes /do fallback is routing ≠ permission. Observe then honor (371ms *theirs*); rh-guard owns primary gates. +Apply Open-Jev densify (`notes.md` §125): systems latency ≠ semantic equivalence. hard acc ≠ calibrated Noul. not merged base models. Open-Jev TREC pending. prefix caching experimental/off by default. LoRA ≠ RLCD replica. type-valid ≠ exact. + **Hourly 1746 HIGH (`notes.md` §128).** TypeLLM truncated thinking densify. 0.8B thinking On 0/18 *theirs*. forced closure 20/20 type-valid *theirs*. Constrained AR ≠ calibrated Noul. Kev-0.8B completes family. 4B new-source 0.794/0.832 *theirs*. 9B new-source 0.812/0.837 *theirs*. transfer-v9 5%/9%/26% *theirs*. Facts go to code. Judgments go to Jev. Only facts can block. Jev never blocks. five-lines threshold 0.80 still soft. 155/155 argmax *theirs*. Qwen3.5 ≠ Archer. catalog ≠ endorsement. SHA move is not a replica. Do not reopen or amend PR #23–#50. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. Hourly 1746 uniqueness lock: TypeLLM/TypeLLM densify HEAD 702e6a287f3c README SHA 08180db0450b; truncated thinking then constrained decode; typellm_runtime.py typellm_sglang.py; evals/qwen35_small; 0.8B thinking On 0/18 *theirs*; forced closure 20/20 type-valid *theirs*; Constrained AR ≠ calibrated Noul; type safety does not guarantee factual accuracy; Qwen/Qwen3.8-27B ≠ Archer; jaredpalmer/kev densify live HEAD 8465c4c4c294 watch 38087aa0301d README SHA 19664b9ae546; Kev-0.8B completes family; Kev-0.8B 4B 9B Qwen3.5; 4B new-source 0.794/0.832 *theirs*; 9B new-source 0.812/0.837 *theirs*; transfer-v9 Kev-9B 5% Jev 9% Kev-8B 26% *theirs*; SemIf Kev-9B 0.917 Jev 0.965 *theirs*; scienthoon 0.952/0.911 vs 0.897/0.914 *theirs*; transformers >= 5.17; Qwen3.5 ≠ Archer; wire-compat ≠ logit-equiv; SHA move is not a replica; notque/vexjoy-agent 421★ /d routes /do fallback; tamaratran/jev-pruner densify HEAD 47d017c34eab; qkal/Canny Facts go to code. Judgments go to Jev. Only facts can block.; Jev never blocks; jqueryscript/awesome-jev 231 entries catalog ≠ endorsement; jqueryscript/awesome-jev ≠ MrJev/awesome-jev ≠ yibie/awesome-jev; jamescazzetta/five-lines threshold 0.80 still soft; eugeniughelbur/jev-engineering 371ms $0.0000189 300-call *theirs*; tpellet/jevify ≠ altryne/jevify; seb4ez/jevguard-mcp ≠ seb4ez/jevguard; resumocast/jev-mcp ≠ jkudish/jev-mcp; dtduc-git/jev-table first sighting; Adrian-Ernesto/jevsort ≠ zzzzzec/jevsort; MidasMulli/kev-ane 155/155 argmax *theirs*; MidasMulli/kev-ane ≠ jaredpalmer/kev; serving substrate ≠ calibrated replica; empty repo skip-thin; loktar00/llm-lan-party empty repo; rh-guard owns primary gates; catalog ≠ endorsement; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50; notes.md §128 **Hourly 1843 HIGH (`notes.md` §129).** kev own-data JSONL densify. --init_from warm-start LoRA/head PR #9. from-scratch ≠ warm-start. JSONL labels ≠ Harbor. 4B new-source 0.794/0.832 *theirs*. 9B new-source 0.812/0.837 *theirs*. 0.33 vs 0.84 vs 0.83/0.88 *theirs*. reconstruction ≠ replica. assay-001 split verdict. catalog ≠ endorsement. SHA move is not a replica. Do not reopen or amend PR #23–#51. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. @@ -1488,3 +1490,5 @@ Hourly 1843 uniqueness lock: jaredpalmer/kev densify HEAD bd058057ad0a README SH Batched single-token Choice after one prefill is still not hosted Jev. TypeSafe-compatible ≠ TypeSafe replica. replica ≠ TypeSafe. `notes.md` §130. **User-provided 1936 HIGH (`notes.md` §130).** sgoedecke/system-one first-sighting. SystemOne.from_pretrained. TypeSafe-compatible ≠ TypeSafe replica. mithalouni/system-one-open first-sighting. 76.7% vs Jev 86.9% *theirs*. replica ≠ TypeSafe. kotoba-lang/typed-decisions first-sighting. DeBERTa-v3-large 0.855 / 42 ms *theirs*. kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions. aisearchio 15-link census catalog ≠ endorsement. soft scores ≠ hard gates. SHA move is not a replica. Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. User-provided 1936 uniqueness lock: sgoedecke/system-one 20★ HEAD ebde2a2db706 README SHA d331b567e2c3; SystemOne.from_pretrained; Batched single-token choice inference; TypeSafe-compatible; cache_prefix=True; LICENSE absent; sgoedecke/system-one ≠ mithalouni/system-one-open ≠ KathanModh259/system-one ≠ babybear-labs/system-one; TypeSafe-compatible ≠ TypeSafe replica; mithalouni/system-one-open 18★ MIT HEAD 77f1f7cccf8a README SHA 535f33028a68 LICENSE SHA 2f6f2cf1064e; Gemma 4 E2B / Gemma 3 270M Modal; 76.7% vs Jev 86.9% strict common subset *theirs*; 97 ms H100 *theirs*; 74.8% held-out *theirs*; replica ≠ TypeSafe; HF upload pending; kotoba-lang/typed-decisions 1★ Apache-2.0 HEAD 10d7834d3b99 README SHA 4d6bbf4c4e44 LICENSE SHA 513bb5e3cb4c; ModernBERT / DeBERTa / LLaDA-MoE; DeBERTa-v3-large 0.855 / 42 ms *theirs*; ModernBERT-base 0.717 / 68 ms *theirs*; LLaDA-MoE 0.835 / 676 ms *theirs*; kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions; encoder class member not Jev replica; aisearchio 15-link census catalog ≠ endorsement; 12 already carded 3 gaps this fold; soft scores ≠ hard gates; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §130 +**Open-Jev densify (`notes.md` §125).** DENSIFY the original 1441 card, not a sibling first sighting. HEAD 4933ee84951f README SHA ce1a587219e4. LoRA + scalar head + calibration temperature. not merged base models. customer-service P50 85.03 vs Jev 295.26 *theirs*. 1024/32 slower 1015.90 vs 301.37 *theirs*. systems latency ≠ semantic equivalence. Open-Jev TREC pending. hard acc ≠ calibrated Noul. type-valid ≠ exact. LoRA ≠ RLCD replica. Qwen/Qwen3.8-27B ≠ Archer. SHA move is not a replica. Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. +User-provided Open-Jev densify uniqueness lock: Zefan-Cai/Open-Jev densify HEAD 4933ee84951f README SHA ce1a587219e4; pushed 2026-09-21T01:34Z; Astra TREC commit 1dd56990be7e pushed 2026-09-21T01:17Z; live 3★ (was 0★; star-noise is not the fold); LoRA adapters plus trained scalar decision head and calibration temperature; not merged base models; dataset ZefanCai/Open-Jev rev c67699e13d0a; Open-Jev-2B rev 0c7aa498b162; Open-Jev-9B rev 47e966881e48; 27B still in progress; Independent of TypeSafe; no RLCD/parity claims; LoRA ≠ RLCD replica; customer-service P50 local HTTP 85.03 ms vs Jev HTTPS 295.26 ms *theirs*; 1024 tokens/32 candidates Open-Jev slower 1015.90 vs 301.37 *theirs*; prefix caching experimental/off by default; CUDA prefix caching exceeded tolerance on 9/11 workloads; systems latency ≠ semantic equivalence; GPT Luna P50 918.13 ms Astra 1938.39 ms *theirs*; TREC-DL Jev/Luna/Astra completed; Open-Jev TREC pending; 80,816 training rows; 2B 94.71% / OOD 86.02%; 9B 97.54% / 91.97% *theirs* not Harbor; hard acc ≠ calibrated Noul; type-valid ≠ exact; Qwen/Qwen3.8-27B ≠ Archer; website https://zefan-cai.github.io/open-jev/; launch X thread https://x.com/Zefan_Cai/status/2101782158658695388 https://x.com/Zefan_Cai/status/2101786019607740436 https://x.com/Zefan_Cai/status/2101789698947793231; densify §125 not a sibling first sighting; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §125 diff --git a/.agents/skills/augustus/references/question-design.md b/.agents/skills/augustus/references/question-design.md index c31c457..85e10c7 100644 --- a/.agents/skills/augustus/references/question-design.md +++ b/.agents/skills/augustus/references/question-design.md @@ -455,3 +455,6 @@ Hourly 1843 uniqueness lock: jaredpalmer/kev densify HEAD bd058057ad0a README SH **User-provided 1936 HIGH (`notes.md` §130).** sgoedecke/system-one first-sighting. SystemOne.from_pretrained. TypeSafe-compatible ≠ TypeSafe replica. mithalouni/system-one-open first-sighting. 76.7% vs Jev 86.9% *theirs*. replica ≠ TypeSafe. kotoba-lang/typed-decisions first-sighting. DeBERTa-v3-large 0.855 / 42 ms *theirs*. kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions. aisearchio 15-link census catalog ≠ endorsement. soft scores ≠ hard gates. SHA move is not a replica. Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. User-provided 1936 uniqueness lock: sgoedecke/system-one 20★ HEAD ebde2a2db706 README SHA d331b567e2c3; SystemOne.from_pretrained; Batched single-token choice inference; TypeSafe-compatible; cache_prefix=True; LICENSE absent; sgoedecke/system-one ≠ mithalouni/system-one-open ≠ KathanModh259/system-one ≠ babybear-labs/system-one; TypeSafe-compatible ≠ TypeSafe replica; mithalouni/system-one-open 18★ MIT HEAD 77f1f7cccf8a README SHA 535f33028a68 LICENSE SHA 2f6f2cf1064e; Gemma 4 E2B / Gemma 3 270M Modal; 76.7% vs Jev 86.9% strict common subset *theirs*; 97 ms H100 *theirs*; 74.8% held-out *theirs*; replica ≠ TypeSafe; HF upload pending; kotoba-lang/typed-decisions 1★ Apache-2.0 HEAD 10d7834d3b99 README SHA 4d6bbf4c4e44 LICENSE SHA 513bb5e3cb4c; ModernBERT / DeBERTa / LLaDA-MoE; DeBERTa-v3-large 0.855 / 42 ms *theirs*; ModernBERT-base 0.717 / 68 ms *theirs*; LLaDA-MoE 0.835 / 676 ms *theirs*; kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions; encoder class member not Jev replica; aisearchio 15-link census catalog ≠ endorsement; 12 already carded 3 gaps this fold; soft scores ≠ hard gates; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §130 + +**Open-Jev densify (`notes.md` §125).** DENSIFY the original 1441 card, not a sibling first sighting. HEAD 4933ee84951f README SHA ce1a587219e4. LoRA + scalar head + calibration temperature. not merged base models. customer-service P50 85.03 vs Jev 295.26 *theirs*. 1024/32 slower 1015.90 vs 301.37 *theirs*. systems latency ≠ semantic equivalence. Open-Jev TREC pending. hard acc ≠ calibrated Noul. type-valid ≠ exact. LoRA ≠ RLCD replica. Qwen/Qwen3.8-27B ≠ Archer. SHA move is not a replica. Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. +User-provided Open-Jev densify uniqueness lock: Zefan-Cai/Open-Jev densify HEAD 4933ee84951f README SHA ce1a587219e4; pushed 2026-09-21T01:34Z; Astra TREC commit 1dd56990be7e pushed 2026-09-21T01:17Z; live 3★ (was 0★; star-noise is not the fold); LoRA adapters plus trained scalar decision head and calibration temperature; not merged base models; dataset ZefanCai/Open-Jev rev c67699e13d0a; Open-Jev-2B rev 0c7aa498b162; Open-Jev-9B rev 47e966881e48; 27B still in progress; Independent of TypeSafe; no RLCD/parity claims; LoRA ≠ RLCD replica; customer-service P50 local HTTP 85.03 ms vs Jev HTTPS 295.26 ms *theirs*; 1024 tokens/32 candidates Open-Jev slower 1015.90 vs 301.37 *theirs*; prefix caching experimental/off by default; CUDA prefix caching exceeded tolerance on 9/11 workloads; systems latency ≠ semantic equivalence; GPT Luna P50 918.13 ms Astra 1938.39 ms *theirs*; TREC-DL Jev/Luna/Astra completed; Open-Jev TREC pending; 80,816 training rows; 2B 94.71% / OOD 86.02%; 9B 97.54% / 91.97% *theirs* not Harbor; hard acc ≠ calibrated Noul; type-valid ≠ exact; Qwen/Qwen3.8-27B ≠ Archer; website https://zefan-cai.github.io/open-jev/; launch X thread https://x.com/Zefan_Cai/status/2101782158658695388 https://x.com/Zefan_Cai/status/2101786019607740436 https://x.com/Zefan_Cai/status/2101789698947793231; densify §125 not a sibling first sighting; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §125 diff --git a/.agents/skills/augustus/references/toolbox-mapping.md b/.agents/skills/augustus/references/toolbox-mapping.md index 17e9b26..aa3a08c 100644 --- a/.agents/skills/augustus/references/toolbox-mapping.md +++ b/.agents/skills/augustus/references/toolbox-mapping.md @@ -384,3 +384,6 @@ Hourly 1843 uniqueness lock: jaredpalmer/kev densify HEAD bd058057ad0a README SH **User-provided 1936 HIGH (`notes.md` §130).** sgoedecke/system-one first-sighting. SystemOne.from_pretrained. TypeSafe-compatible ≠ TypeSafe replica. mithalouni/system-one-open first-sighting. 76.7% vs Jev 86.9% *theirs*. replica ≠ TypeSafe. kotoba-lang/typed-decisions first-sighting. DeBERTa-v3-large 0.855 / 42 ms *theirs*. kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions. aisearchio 15-link census catalog ≠ endorsement. soft scores ≠ hard gates. SHA move is not a replica. Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. User-provided 1936 uniqueness lock: sgoedecke/system-one 20★ HEAD ebde2a2db706 README SHA d331b567e2c3; SystemOne.from_pretrained; Batched single-token choice inference; TypeSafe-compatible; cache_prefix=True; LICENSE absent; sgoedecke/system-one ≠ mithalouni/system-one-open ≠ KathanModh259/system-one ≠ babybear-labs/system-one; TypeSafe-compatible ≠ TypeSafe replica; mithalouni/system-one-open 18★ MIT HEAD 77f1f7cccf8a README SHA 535f33028a68 LICENSE SHA 2f6f2cf1064e; Gemma 4 E2B / Gemma 3 270M Modal; 76.7% vs Jev 86.9% strict common subset *theirs*; 97 ms H100 *theirs*; 74.8% held-out *theirs*; replica ≠ TypeSafe; HF upload pending; kotoba-lang/typed-decisions 1★ Apache-2.0 HEAD 10d7834d3b99 README SHA 4d6bbf4c4e44 LICENSE SHA 513bb5e3cb4c; ModernBERT / DeBERTa / LLaDA-MoE; DeBERTa-v3-large 0.855 / 42 ms *theirs*; ModernBERT-base 0.717 / 68 ms *theirs*; LLaDA-MoE 0.835 / 676 ms *theirs*; kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions; encoder class member not Jev replica; aisearchio 15-link census catalog ≠ endorsement; 12 already carded 3 gaps this fold; soft scores ≠ hard gates; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §130 + +**Open-Jev densify (`notes.md` §125).** DENSIFY the original 1441 card, not a sibling first sighting. HEAD 4933ee84951f README SHA ce1a587219e4. LoRA + scalar head + calibration temperature. not merged base models. customer-service P50 85.03 vs Jev 295.26 *theirs*. 1024/32 slower 1015.90 vs 301.37 *theirs*. systems latency ≠ semantic equivalence. Open-Jev TREC pending. hard acc ≠ calibrated Noul. type-valid ≠ exact. LoRA ≠ RLCD replica. Qwen/Qwen3.8-27B ≠ Archer. SHA move is not a replica. Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. +User-provided Open-Jev densify uniqueness lock: Zefan-Cai/Open-Jev densify HEAD 4933ee84951f README SHA ce1a587219e4; pushed 2026-09-21T01:34Z; Astra TREC commit 1dd56990be7e pushed 2026-09-21T01:17Z; live 3★ (was 0★; star-noise is not the fold); LoRA adapters plus trained scalar decision head and calibration temperature; not merged base models; dataset ZefanCai/Open-Jev rev c67699e13d0a; Open-Jev-2B rev 0c7aa498b162; Open-Jev-9B rev 47e966881e48; 27B still in progress; Independent of TypeSafe; no RLCD/parity claims; LoRA ≠ RLCD replica; customer-service P50 local HTTP 85.03 ms vs Jev HTTPS 295.26 ms *theirs*; 1024 tokens/32 candidates Open-Jev slower 1015.90 vs 301.37 *theirs*; prefix caching experimental/off by default; CUDA prefix caching exceeded tolerance on 9/11 workloads; systems latency ≠ semantic equivalence; GPT Luna P50 918.13 ms Astra 1938.39 ms *theirs*; TREC-DL Jev/Luna/Astra completed; Open-Jev TREC pending; 80,816 training rows; 2B 94.71% / OOD 86.02%; 9B 97.54% / 91.97% *theirs* not Harbor; hard acc ≠ calibrated Noul; type-valid ≠ exact; Qwen/Qwen3.8-27B ≠ Archer; website https://zefan-cai.github.io/open-jev/; launch X thread https://x.com/Zefan_Cai/status/2101782158658695388 https://x.com/Zefan_Cai/status/2101786019607740436 https://x.com/Zefan_Cai/status/2101789698947793231; densify §125 not a sibling first sighting; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §125 diff --git a/.agents/skills/augustus/references/validation.md b/.agents/skills/augustus/references/validation.md index 69283a7..3924b92 100644 --- a/.agents/skills/augustus/references/validation.md +++ b/.agents/skills/augustus/references/validation.md @@ -1192,3 +1192,5 @@ Hourly 1843 uniqueness lock: jaredpalmer/kev densify HEAD bd058057ad0a README SH 76.7% vs Jev 86.9% *theirs*. DeBERTa-v3-large 0.855 / 42 ms *theirs*. Soft scores ≠ hard gates. *theirs* not Harbor. `notes.md` §130. **User-provided 1936 HIGH (`notes.md` §130).** sgoedecke/system-one first-sighting. SystemOne.from_pretrained. TypeSafe-compatible ≠ TypeSafe replica. mithalouni/system-one-open first-sighting. 76.7% vs Jev 86.9% *theirs*. replica ≠ TypeSafe. kotoba-lang/typed-decisions first-sighting. DeBERTa-v3-large 0.855 / 42 ms *theirs*. kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions. aisearchio 15-link census catalog ≠ endorsement. soft scores ≠ hard gates. SHA move is not a replica. Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. User-provided 1936 uniqueness lock: sgoedecke/system-one 20★ HEAD ebde2a2db706 README SHA d331b567e2c3; SystemOne.from_pretrained; Batched single-token choice inference; TypeSafe-compatible; cache_prefix=True; LICENSE absent; sgoedecke/system-one ≠ mithalouni/system-one-open ≠ KathanModh259/system-one ≠ babybear-labs/system-one; TypeSafe-compatible ≠ TypeSafe replica; mithalouni/system-one-open 18★ MIT HEAD 77f1f7cccf8a README SHA 535f33028a68 LICENSE SHA 2f6f2cf1064e; Gemma 4 E2B / Gemma 3 270M Modal; 76.7% vs Jev 86.9% strict common subset *theirs*; 97 ms H100 *theirs*; 74.8% held-out *theirs*; replica ≠ TypeSafe; HF upload pending; kotoba-lang/typed-decisions 1★ Apache-2.0 HEAD 10d7834d3b99 README SHA 4d6bbf4c4e44 LICENSE SHA 513bb5e3cb4c; ModernBERT / DeBERTa / LLaDA-MoE; DeBERTa-v3-large 0.855 / 42 ms *theirs*; ModernBERT-base 0.717 / 68 ms *theirs*; LLaDA-MoE 0.835 / 676 ms *theirs*; kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions; encoder class member not Jev replica; aisearchio 15-link census catalog ≠ endorsement; 12 already carded 3 gaps this fold; soft scores ≠ hard gates; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §130 +**Open-Jev densify (`notes.md` §125).** DENSIFY the original 1441 card, not a sibling first sighting. HEAD 4933ee84951f README SHA ce1a587219e4. LoRA + scalar head + calibration temperature. not merged base models. customer-service P50 85.03 vs Jev 295.26 *theirs*. 1024/32 slower 1015.90 vs 301.37 *theirs*. systems latency ≠ semantic equivalence. Open-Jev TREC pending. hard acc ≠ calibrated Noul. type-valid ≠ exact. LoRA ≠ RLCD replica. Qwen/Qwen3.8-27B ≠ Archer. SHA move is not a replica. Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. +User-provided Open-Jev densify uniqueness lock: Zefan-Cai/Open-Jev densify HEAD 4933ee84951f README SHA ce1a587219e4; pushed 2026-09-21T01:34Z; Astra TREC commit 1dd56990be7e pushed 2026-09-21T01:17Z; live 3★ (was 0★; star-noise is not the fold); LoRA adapters plus trained scalar decision head and calibration temperature; not merged base models; dataset ZefanCai/Open-Jev rev c67699e13d0a; Open-Jev-2B rev 0c7aa498b162; Open-Jev-9B rev 47e966881e48; 27B still in progress; Independent of TypeSafe; no RLCD/parity claims; LoRA ≠ RLCD replica; customer-service P50 local HTTP 85.03 ms vs Jev HTTPS 295.26 ms *theirs*; 1024 tokens/32 candidates Open-Jev slower 1015.90 vs 301.37 *theirs*; prefix caching experimental/off by default; CUDA prefix caching exceeded tolerance on 9/11 workloads; systems latency ≠ semantic equivalence; GPT Luna P50 918.13 ms Astra 1938.39 ms *theirs*; TREC-DL Jev/Luna/Astra completed; Open-Jev TREC pending; 80,816 training rows; 2B 94.71% / OOD 86.02%; 9B 97.54% / 91.97% *theirs* not Harbor; hard acc ≠ calibrated Noul; type-valid ≠ exact; Qwen/Qwen3.8-27B ≠ Archer; website https://zefan-cai.github.io/open-jev/; launch X thread https://x.com/Zefan_Cai/status/2101782158658695388 https://x.com/Zefan_Cai/status/2101786019607740436 https://x.com/Zefan_Cai/status/2101789698947793231; densify §125 not a sibling first sighting; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §125 diff --git a/.agents/skills/augustus/scripts/evaluate_decisions.py b/.agents/skills/augustus/scripts/evaluate_decisions.py index 7d4900a..5e38689 100755 --- a/.agents/skills/augustus/scripts/evaluate_decisions.py +++ b/.agents/skills/augustus/scripts/evaluate_decisions.py @@ -29,6 +29,13 @@ * typesafe-sdk 0.7 Pydantic response models are not logit-equiv * MLX SchemaError 400 plain-string is the same contract as vLLM * coverage-at-error-budget stays *theirs* (not Harbor) + * systems latency is not semantic equivalence (Open-Jev vs Jev P50) + * hard accuracy is not a calibrated Noul + * LoRA + scalar-head packs are not merged base models + * prefix caching experimental/off by default + * TREC-DL Jev/Luna/Astra completed; Open-Jev TREC pending + * Qwen/Qwen3.8-27B is not Archer + * latency does not establish equal task quality Select thresholds on one split, evaluate on another: run twice with different files. Missing labels or costs produce a stated limitation, not defaults. @@ -521,6 +528,51 @@ def soft_scores_are_not_hard_gates(score, hard_gate=False): if not (0.0 <= score <= 1.0 or score > 1.0): raise ValueError("unexpected score") return hard_gate is False +def systems_latency_is_not_semantic_equivalence(lat_a_ms, lat_b_ms, claimed_equiv=False): + """Observed deployment latency is not semantic equivalence.""" + if lat_a_ms <= 0 or lat_b_ms <= 0: + raise ValueError("latencies must be positive") + return claimed_equiv is False + + +def hard_acc_is_not_calibrated_noul(hard_acc, noul_claimed=False): + """Hard accuracy on synthetic rows ≠ calibrated Noul.""" + if not (0.0 <= hard_acc <= 1.0): + raise ValueError("acc must be a rate") + return noul_claimed is False + + +def lora_pack_is_not_merged_base(kind, merged=False): + """Open-Jev 2B/9B packs are LoRA + scalar head, not merged base models.""" + if kind != "lora_plus_scalar_head": + return False + return merged is False + + +def prefix_cache_off_by_default(enabled=False, experimental=True): + """Prefix caching is experimental and off by default.""" + return enabled is False and experimental is True + + +def open_jev_trec_is_pending(completed_providers, open_jev_done=False): + """TREC-DL Jev/Luna/Astra completed. Open-Jev TREC pending.""" + expected = {"jev", "luna", "astra"} + if set(completed_providers) != expected: + raise ValueError("unexpected providers") + return open_jev_done is False + + +def qwen38_27b_is_not_archer(checkpoint, claimed_archer=False): + """Qwen/Qwen3.8-27B ≠ Archer (Open-Jev 27B still in progress; litjev default).""" + if checkpoint != "Qwen/Qwen3.8-27B": + raise ValueError("unexpected checkpoint") + return claimed_archer is False + + +def latency_is_not_task_quality(quality_claimed=False): + """Latency does not establish equal task quality *theirs*.""" + return quality_claimed is False + def hop_ece_permutation_invariant(rows, bins=10, key="p"): """Shuffle order; equal-width ECE must not move. @@ -744,6 +796,29 @@ def self_test(): assert not soft_scores_are_not_hard_gates(0.767, True) assert theirs_bench_is_not_harbor(343, "mithalouni-76.7-vs-jev-86.9") assert theirs_bench_is_not_harbor(1500, "kotoba-deberta-0.855-42ms") + # Open-Jev densify §125: systems latency ≠ semantic equivalence / + # hard acc ≠ calibrated Noul / LoRA pack ≠ merged base / + # prefix cache off / TREC pending / *theirs* not Harbor. + assert systems_latency_is_not_semantic_equivalence(85.03, 295.26, False) + assert not systems_latency_is_not_semantic_equivalence(85.03, 295.26, True) + assert systems_latency_is_not_semantic_equivalence(1015.90, 301.37, False) + assert hard_acc_is_not_calibrated_noul(0.9471, False) + assert not hard_acc_is_not_calibrated_noul(0.9471, True) + assert lora_pack_is_not_merged_base("lora_plus_scalar_head", False) + assert not lora_pack_is_not_merged_base("lora_plus_scalar_head", True) + assert prefix_cache_off_by_default(False, True) + assert not prefix_cache_off_by_default(True, True) + assert open_jev_trec_is_pending(("jev", "luna", "astra"), False) + assert not open_jev_trec_is_pending(("jev", "luna", "astra"), True) + assert type_valid_is_not_exact(True, False) + assert lora_is_not_rlcd_replica("lora_plus_scalar_head", False) + assert theirs_bench_is_not_harbor(85, "open-jev-cs-p50-85ms") + assert theirs_bench_is_not_harbor(1015, "open-jev-1024-32-1015ms") + assert theirs_bench_is_not_harbor(97, "trec-dl-jev-luna-astra") + assert qwen38_27b_is_not_archer("Qwen/Qwen3.8-27B", False) + assert not qwen38_27b_is_not_archer("Qwen/Qwen3.8-27B", True) + assert latency_is_not_task_quality(False) + assert not latency_is_not_task_quality(True) print("self-test ok") diff --git a/.agents/skills/augustus/scripts/uniqueness_gate.py b/.agents/skills/augustus/scripts/uniqueness_gate.py index ec9a07f..4331984 100644 --- a/.agents/skills/augustus/scripts/uniqueness_gate.py +++ b/.agents/skills/augustus/scripts/uniqueness_gate.py @@ -2,7 +2,7 @@ """Uniqueness gate for merged 0843 (§114), merged 0915 NanoJev (§115), merged 0920 jcr (§116), merged 0922 SemIf (§117), merged 0940 llm-to-jev (§118), hourly 0947 HIGH (§119), hourly 1049 HIGH (§120), -hourly 1143 HIGH (§121), hourly 1248 HIGH (§123), hourly 1340 HIGH (§124), hourly 1441 HIGH (§125), hourly 1542 HIGH (§126), hourly 1643 HIGH (§127), hourly 1746 HIGH (§128), hourly 1843 HIGH (§129), and user-provided 1936 HIGH (§130). +hourly 1143 HIGH (§121), hourly 1248 HIGH (§123), hourly 1340 HIGH (§124), hourly 1441 HIGH (§125), hourly 1542 HIGH (§126), hourly 1643 HIGH (§127), hourly 1746 HIGH (§128), hourly 1843 HIGH (§129), user-provided 1936 HIGH (§130), and Open-Jev densify (§125). Each lock must appear as one consecutive substring in every listed overlay. Fragments scattered across files do not count. @@ -169,6 +169,10 @@ "Hourly 1843 uniqueness lock: jaredpalmer/kev densify HEAD bd058057ad0a README SHA 84b872488915; Fine-tuning on your own data; --data JSONL; --init_from warm-start LoRA/head PR #9; Kev-0.8B 4B 9B Qwen3.5 family; 4B new-source 0.794/0.832 *theirs*; 9B new-source 0.812/0.837 *theirs*; 0.33 vs 0.84 vs 0.83/0.88 *theirs*; from-scratch ≠ warm-start; JSONL labels ≠ Harbor; Kev-0.5B card Qwen3.5 family pointer; No Jev outputs were used for training; option order can change an answer; 8.2% ≥0.9 on wrong *theirs*; Kev-9B 7.5% ≥0.9 on wrong *theirs*; dabit3/jev-experiments densify 340★; simota/tenbin densify neighbor skill; Promethe-us/awesome-jev ≠ MrJev/awesome-jev ≠ yibie/awesome-jev; kyegomez/open-jev reconstruction ≠ replica; unofficial research implementation with random weights; kyegomez/open-jev ≠ razorback16/openjev ≠ TheoLeeCJ/openjev ≠ Zefan-Cai/Open-Jev ≠ Shalimov04/open-jev; jourdanlabs/assay-001 split verdict; CLINC150 ECE 0.0204 *theirs*; Banking77 ECE 0.0936 *theirs*; 8,576 responses zero type errors *theirs*; brnyxx/jev-ra 3-5x / ~300 ms *theirs*; 8.50× Wikipedia *theirs*; ThePFMind/jev-mcp ≠ jkudish/jev-mcp ≠ burnigtm/jev-mcp; namenu/pi-jev-effort ≠ TheoOliveira/pi-jev; samatv256/mini-Jev ≠ r-ms/mini-jev; comoc/jev-minesweeper ≠ EnesYilmazcode/JevMinesweeper; game success ≠ calibrated Noul; Nutlope/jev-fraud Kimi K3; jeffloo886/jev-notion; hf:akhilaaa3/openjev-v1-allmix-r512-merged ≠ hf:akhilaaa3/openjev-r512-handoff-demo; serving substrate ≠ calibrated replica; catalog ≠ endorsement; SHA move is not a replica; wire-compat ≠ logit-equiv; Qwen3.5 ≠ Archer; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51; notes.md §129" ) +UNIQ_OPENJEV = ( + 'User-provided Open-Jev densify uniqueness lock: Zefan-Cai/Open-Jev densify HEAD 4933ee84951f README SHA ce1a587219e4; pushed 2026-09-21T01:34Z; Astra TREC commit 1dd56990be7e pushed 2026-09-21T01:17Z; live 3★ (was 0★; star-noise is not the fold); LoRA adapters plus trained scalar decision head and calibration temperature; not merged base models; dataset ZefanCai/Open-Jev rev c67699e13d0a; Open-Jev-2B rev 0c7aa498b162; Open-Jev-9B rev 47e966881e48; 27B still in progress; Independent of TypeSafe; no RLCD/parity claims; LoRA ≠ RLCD replica; customer-service P50 local HTTP 85.03 ms vs Jev HTTPS 295.26 ms *theirs*; 1024 tokens/32 candidates Open-Jev slower 1015.90 vs 301.37 *theirs*; prefix caching experimental/off by default; CUDA prefix caching exceeded tolerance on 9/11 workloads; systems latency ≠ semantic equivalence; GPT Luna P50 918.13 ms Astra 1938.39 ms *theirs*; TREC-DL Jev/Luna/Astra completed; Open-Jev TREC pending; 80,816 training rows; 2B 94.71% / OOD 86.02%; 9B 97.54% / 91.97% *theirs* not Harbor; hard acc ≠ calibrated Noul; type-valid ≠ exact; Qwen/Qwen3.8-27B ≠ Archer; website https://zefan-cai.github.io/open-jev/; launch X thread https://x.com/Zefan_Cai/status/2101782158658695388 https://x.com/Zefan_Cai/status/2101786019607740436 https://x.com/Zefan_Cai/status/2101789698947793231; densify §125 not a sibling first sighting; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §125' +) + REVISIT_LOCK = ( "Revisit / since-last-look lock: catalogued repos are not done; " "store fingerprints default_sha, pushed_at, description_hash, release_tag; " @@ -264,6 +268,8 @@ def main() -> int: failed.append(f"1843 lock missing as one substring: {rel}") if UNIQ_1936 not in body: failed.append(f"1936 lock missing as one substring: {rel}") + if UNIQ_OPENJEV not in body: + failed.append(f"openjev densify lock missing as one substring: {rel}") for rel in REVISIT_OVERLAYS: path = ROOT / rel if not path.is_file(): @@ -603,6 +609,22 @@ def main() -> int: '12 already carded 3 gaps this fold', 'user-provided 1936 / notes.md §130', "hourly 1843 / notes.md §129", + "Open-Jev densify HEAD 4933ee84951f", + "Astra TREC commit 1dd56990be7e", + "not merged base models", + "customer-service P50 local HTTP 85.03 ms vs Jev HTTPS 295.26 ms *theirs*", + "1024 tokens/32 candidates Open-Jev slower 1015.90 vs 301.37 *theirs*", + "prefix caching experimental/off by default", + "systems latency ≠ semantic equivalence", + "GPT Luna P50 918.13 ms Astra 1938.39 ms *theirs*", + "TREC-DL Jev/Luna/Astra completed", + "Open-Jev TREC pending", + "hard acc ≠ calibrated Noul", + "densify §125 not a sibling first sighting", + "Open-Jev densify / notes.md §125", + "launch X thread https://x.com/Zefan_Cai/status/2101782158658695388", + "2101786019607740436", + "2101789698947793231", ): if frag not in haystack: failed.append(f"SKILL.md missing fragment {frag!r}") @@ -807,6 +829,22 @@ def main() -> int: 'aisearchio 15-link census catalog ≠ endorsement', 'user-provided 1936 / notes.md §130', "hourly 1843 / notes.md §129", + "Open-Jev densify HEAD 4933ee84951f", + "Astra TREC commit 1dd56990be7e", + "not merged base models", + "customer-service P50 local HTTP 85.03 ms vs Jev HTTPS 295.26 ms *theirs*", + "1024 tokens/32 candidates Open-Jev slower 1015.90 vs 301.37 *theirs*", + "prefix caching experimental/off by default", + "systems latency ≠ semantic equivalence", + "GPT Luna P50 918.13 ms Astra 1938.39 ms *theirs*", + "TREC-DL Jev/Luna/Astra completed", + "Open-Jev TREC pending", + "hard acc ≠ calibrated Noul", + "densify §125 not a sibling first sighting", + "Open-Jev densify / notes.md §125", + "launch X thread https://x.com/Zefan_Cai/status/2101782158658695388", + "2101786019607740436", + "2101789698947793231", ): if frag not in proto_line: failed.append(f"SKILL.md protocol missing {frag!r}") @@ -828,6 +866,7 @@ def main() -> int: ("1746", UNIQ_1746), ("1843", UNIQ_1843), ("1936", UNIQ_1936), + ("openjev_densify", UNIQ_OPENJEV), ): if lock in changelog: failed.append( @@ -905,6 +944,7 @@ def main() -> int: f"1746 chars={len(UNIQ_1746)} " f"1843 chars={len(UNIQ_1843)} " f"1936 chars={len(UNIQ_1936)} " + f"openjev_densify chars={len(UNIQ_OPENJEV)} " f"revisit chars={len(REVISIT_LOCK)} " f"overlays={len(OVERLAYS)} " f"revisit_overlays={len(REVISIT_OVERLAYS)}" diff --git a/CHANGELOG.md b/CHANGELOG.md index 21f8572..882f248 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -16,6 +16,36 @@ folds: `research/notes.md`. ## [Unreleased] +Open-Jev densify (`research/notes.md` §125). Does **not** bump the +0.5.0 pin. Uniqueness dumps live in +[`research/changelog-hourly.md`](research/changelog-hourly.md). +Do not reopen or amend PR #23–#52. +Do not amend released 0.5.0 (#42). Merged #52 owns §129. + +### Added + +- **Open-Jev densify (`notes.md` §125).** HEAD 4933ee84951f / + README SHA ce1a587219e4 / Astra TREC commit 1dd56990be7e. + LoRA + scalar head + calibration temperature. not merged base + models. customer-service P50 85.03 vs Jev 295.26 *theirs*. + 1024/32 slower 1015.90 vs 301.37 *theirs*. systems latency ≠ + semantic equivalence. Open-Jev TREC pending. hard acc ≠ + calibrated Noul. type-valid ≠ exact. LoRA ≠ RLCD replica. + Evaluator: systems latency ≠ semantic equivalence / hard acc ≠ + Noul / LoRA pack ≠ merged base / prefix cache off / TREC pending. + uniqueness_gate.py now checks 0843 + 0915 + jcr + 0922 + 0940 + + 0947 + 1049 + 1143 + 1248 + 1340 + 1441 + 1542 + 1643 + 1746 + + 1843 + Open-Jev densify. Densify original section. Do not mint + a sibling first sighting. + **HARD RULE:** do not reopen or amend PR #23–#52. Does **not** bump + 0.5.0. + +- **Recipe (class, not Jev-only).** Without Augustus: treat 85 ms as + parity, 94.71% as a Noul, or a LoRA pack as a merged RLCD replica. + With Augustus: systems latency ≠ semantic equivalence; hard acc ≠ + calibrated Noul; LoRA ≠ RLCD replica; not merged base models; + Open-Jev TREC pending. Same split for any Choice/Score/Noul-style + head, not only hosted Jev. User-provided 1936 HIGH (`research/notes.md` §130 / composition items 497–504 / findings batch #112). Does **not** bump the 0.5.0 pin. Uniqueness dumps live in @@ -49,7 +79,6 @@ Do not amend released 0.5.0 (#42). Merged #52 owns §129. Merged #51 owns §128. catalog ≠ endorsement; *theirs* not Harbor. Same split for any Choice/Score/Noul-style head, not only hosted Jev. - Hourly 1843 HIGH (`research/notes.md` §129 / composition items 481–496 / findings batch #111). Does **not** bump the 0.5.0 pin. Uniqueness dumps live in diff --git a/README.md b/README.md index f829ae6..ff4342e 100644 --- a/README.md +++ b/README.md @@ -182,3 +182,6 @@ Hourly 1843 uniqueness lock: jaredpalmer/kev densify HEAD bd058057ad0a README SH **User-provided 1936 HIGH (`notes.md` §130).** sgoedecke/system-one first-sighting. SystemOne.from_pretrained. TypeSafe-compatible ≠ TypeSafe replica. mithalouni/system-one-open first-sighting. 76.7% vs Jev 86.9% *theirs*. replica ≠ TypeSafe. kotoba-lang/typed-decisions first-sighting. DeBERTa-v3-large 0.855 / 42 ms *theirs*. kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions. aisearchio 15-link census catalog ≠ endorsement. soft scores ≠ hard gates. SHA move is not a replica. Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. User-provided 1936 uniqueness lock: sgoedecke/system-one 20★ HEAD ebde2a2db706 README SHA d331b567e2c3; SystemOne.from_pretrained; Batched single-token choice inference; TypeSafe-compatible; cache_prefix=True; LICENSE absent; sgoedecke/system-one ≠ mithalouni/system-one-open ≠ KathanModh259/system-one ≠ babybear-labs/system-one; TypeSafe-compatible ≠ TypeSafe replica; mithalouni/system-one-open 18★ MIT HEAD 77f1f7cccf8a README SHA 535f33028a68 LICENSE SHA 2f6f2cf1064e; Gemma 4 E2B / Gemma 3 270M Modal; 76.7% vs Jev 86.9% strict common subset *theirs*; 97 ms H100 *theirs*; 74.8% held-out *theirs*; replica ≠ TypeSafe; HF upload pending; kotoba-lang/typed-decisions 1★ Apache-2.0 HEAD 10d7834d3b99 README SHA 4d6bbf4c4e44 LICENSE SHA 513bb5e3cb4c; ModernBERT / DeBERTa / LLaDA-MoE; DeBERTa-v3-large 0.855 / 42 ms *theirs*; ModernBERT-base 0.717 / 68 ms *theirs*; LLaDA-MoE 0.835 / 676 ms *theirs*; kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions; encoder class member not Jev replica; aisearchio 15-link census catalog ≠ endorsement; 12 already carded 3 gaps this fold; soft scores ≠ hard gates; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §130 + +**Open-Jev densify (`notes.md` §125).** DENSIFY the original 1441 card, not a sibling first sighting. HEAD 4933ee84951f README SHA ce1a587219e4. LoRA + scalar head + calibration temperature. not merged base models. customer-service P50 85.03 vs Jev 295.26 *theirs*. 1024/32 slower 1015.90 vs 301.37 *theirs*. systems latency ≠ semantic equivalence. Open-Jev TREC pending. hard acc ≠ calibrated Noul. type-valid ≠ exact. LoRA ≠ RLCD replica. Qwen/Qwen3.8-27B ≠ Archer. SHA move is not a replica. Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. +User-provided Open-Jev densify uniqueness lock: Zefan-Cai/Open-Jev densify HEAD 4933ee84951f README SHA ce1a587219e4; pushed 2026-09-21T01:34Z; Astra TREC commit 1dd56990be7e pushed 2026-09-21T01:17Z; live 3★ (was 0★; star-noise is not the fold); LoRA adapters plus trained scalar decision head and calibration temperature; not merged base models; dataset ZefanCai/Open-Jev rev c67699e13d0a; Open-Jev-2B rev 0c7aa498b162; Open-Jev-9B rev 47e966881e48; 27B still in progress; Independent of TypeSafe; no RLCD/parity claims; LoRA ≠ RLCD replica; customer-service P50 local HTTP 85.03 ms vs Jev HTTPS 295.26 ms *theirs*; 1024 tokens/32 candidates Open-Jev slower 1015.90 vs 301.37 *theirs*; prefix caching experimental/off by default; CUDA prefix caching exceeded tolerance on 9/11 workloads; systems latency ≠ semantic equivalence; GPT Luna P50 918.13 ms Astra 1938.39 ms *theirs*; TREC-DL Jev/Luna/Astra completed; Open-Jev TREC pending; 80,816 training rows; 2B 94.71% / OOD 86.02%; 9B 97.54% / 91.97% *theirs* not Harbor; hard acc ≠ calibrated Noul; type-valid ≠ exact; Qwen/Qwen3.8-27B ≠ Archer; website https://zefan-cai.github.io/open-jev/; launch X thread https://x.com/Zefan_Cai/status/2101782158658695388 https://x.com/Zefan_Cai/status/2101786019607740436 https://x.com/Zefan_Cai/status/2101789698947793231; densify §125 not a sibling first sighting; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §125 diff --git a/docs/_includes/recipes.html b/docs/_includes/recipes.html index e1da63e..1ebe8f9 100644 --- a/docs/_includes/recipes.html +++ b/docs/_includes/recipes.html @@ -321,6 +321,20 @@

Random weights vs replica

Namesake lock first. Third-party benches stay *theirs*.
+
+

Open LoRA specialist

+

Latency vs meaning

+
+
Problem
+
A local P50 treated as proof the head matches hosted Jev.
+
Without
+
Quote 85 ms as parity. Skip the 1024/32 slowdown. Treat 94.71% as a Noul.
+
With
+
systems latency ≠ semantic equivalence. hard acc ≠ calibrated Noul. LoRA ≠ RLCD replica. not merged base models.
+
Measure
+
customer-service P50 85.03 vs 295.26 *theirs*. 1024/32 1015.90 vs 301.37 *theirs*. Open-Jev TREC pending.
+
+

Measurement recipe (hysteresis, equal-width vs quantile ECE, hop-ECE, diff --git a/docs/ecosystem.md b/docs/ecosystem.md index 600ea2b..64bebe1 100644 --- a/docs/ecosystem.md +++ b/docs/ecosystem.md @@ -73,6 +73,7 @@ Per-keystroke launchers (104ms median, sequence-tagged staleness), firehose mode - **DECRUX9812/openjev-lm** — Qwen2.5-0.5B+LoRA distilled from hosted Jev answers; 65/70 = 92.9% on 70 hand-labelled rows (one annotator, one domain, one seed) overnight on 6 vCPU, $0/call. Its 98.1% on fresh rows is teacher *agreement*, not gold. Receipts pattern: `notes.md` §25, §44. - **convaiinnovations/laya** — open Choice/Score/Noul head, text-only, 512 tok. Companion packaging: [`laya-typed-decisions`](https://huggingface.co/convaiinnovations/laya-typed-decisions) (421.3M, acc 0.766 / Brier 0.066 unverified). Shared bake-off: [`pngwn/open-jev-laya-bench`](https://huggingface.co/datasets/pngwn/open-jev-laya-bench) (26+9 tasks, 11959 items; ECE/NLL/Brier; not TypeSafe Jev vs Laya). ONNX replica: [`Mattepiu/laya-onnx`](https://huggingface.co/Mattepiu/laya-onnx) (~15 ms CPU; do not copy vs-Jev table). **GitHub/PyPI face this hour:** [`NandhaKishorM/laya`](https://github.com/NandhaKishorM/laya) (Apache-2.0; **710★**; Router; vs-Jev unpublished-here; `notes.md` §76). `notes.md` §18, §42, §46, §48, §72, §76. - **jaredpalmer/kev** — Qwen2.5-0.5B LoRA + pointer readout; Apache-2.0; Hub [`jaredpalmer/kev-0.5b`](https://huggingface.co/jaredpalmer/kev-0.5b) plus GitHub release tarball. Runnable Archer reconstruction (`POST /v1/systemone`). Public gold, not a Jev teacher. Isolation exact; ID ECE 0.065 (0.031 after T); acc 0.799 / 1,350. NOTA training must confront `"other"` as a wrong alternative (`notes.md` §45 delta). **100★** this pass (light activity delta; `notes.md` §49). Not a knowledge/frontier substitute. +- **Zefan-Cai/Open-Jev** — independent LoRA + scalar decision head + calibration temperature on Qwen3.5-2B/9B (27B in progress). Public HF packs `ZefanCai/Open-Jev-{2,9}B` are adapters, not merged base models. customer-service P50 85.03 vs Jev 295.26 *theirs*; 1024/32 slower 1015.90 vs 301.37 *theirs*. systems latency ≠ semantic equivalence. Open-Jev TREC pending. hard acc ≠ calibrated Noul. LoRA ≠ RLCD replica. website https://zefan-cai.github.io/open-jev/. Launch thread https://x.com/Zefan_Cai/status/2101782158658695388. Densify `notes.md` §125, not a sibling first sighting. - **ikermoel/open-alternative-jev** — packed one-forward logprob System One on open LLMs (HF + vLLM). Apache-2.0. **Not a Jev reproduction.** RACE-H 92.9% @ 4.55 q/s on Qwen3.6-27B 8-bit; interference 6–9%. Space demo. `notes.md` §49. - **wfzyx/von** — **§49 snapshot:** 14 MB Needle SAN; local `POST /v1/systemone`; sub-15 ms CPU claim / ~38 ms embed. authored144 needle 52.6% — **not a calibrated Jev replica**. Distinguish from jev-local stub and kev pointer. Do not copy vs-Jev table. Late-catch rewrite is `notes.md` §99 (395M / n=78 93.0% *theirs*); do **not** merge the two tables. `notes.md` §49. - **Mintzs/jevify** — CUDA/PyTorch packed-logprob cousin on Qwen2.5-1.5B (`ora_decision_engine`). Uncalibrated likelihoods ≠ Noul. Independent of Distillation. No LICENSE this pass. `notes.md` §55. @@ -1172,3 +1173,6 @@ Hourly 1843 uniqueness lock: jaredpalmer/kev densify HEAD bd058057ad0a README SH **User-provided 1936 HIGH (`notes.md` §130).** sgoedecke/system-one first-sighting. SystemOne.from_pretrained. TypeSafe-compatible ≠ TypeSafe replica. mithalouni/system-one-open first-sighting. 76.7% vs Jev 86.9% *theirs*. replica ≠ TypeSafe. kotoba-lang/typed-decisions first-sighting. DeBERTa-v3-large 0.855 / 42 ms *theirs*. kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions. aisearchio 15-link census catalog ≠ endorsement. soft scores ≠ hard gates. SHA move is not a replica. Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. User-provided 1936 uniqueness lock: sgoedecke/system-one 20★ HEAD ebde2a2db706 README SHA d331b567e2c3; SystemOne.from_pretrained; Batched single-token choice inference; TypeSafe-compatible; cache_prefix=True; LICENSE absent; sgoedecke/system-one ≠ mithalouni/system-one-open ≠ KathanModh259/system-one ≠ babybear-labs/system-one; TypeSafe-compatible ≠ TypeSafe replica; mithalouni/system-one-open 18★ MIT HEAD 77f1f7cccf8a README SHA 535f33028a68 LICENSE SHA 2f6f2cf1064e; Gemma 4 E2B / Gemma 3 270M Modal; 76.7% vs Jev 86.9% strict common subset *theirs*; 97 ms H100 *theirs*; 74.8% held-out *theirs*; replica ≠ TypeSafe; HF upload pending; kotoba-lang/typed-decisions 1★ Apache-2.0 HEAD 10d7834d3b99 README SHA 4d6bbf4c4e44 LICENSE SHA 513bb5e3cb4c; ModernBERT / DeBERTa / LLaDA-MoE; DeBERTa-v3-large 0.855 / 42 ms *theirs*; ModernBERT-base 0.717 / 68 ms *theirs*; LLaDA-MoE 0.835 / 676 ms *theirs*; kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions; encoder class member not Jev replica; aisearchio 15-link census catalog ≠ endorsement; 12 already carded 3 gaps this fold; soft scores ≠ hard gates; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §130 + +**Open-Jev densify (`notes.md` §125).** DENSIFY the original 1441 card, not a sibling first sighting. HEAD 4933ee84951f README SHA ce1a587219e4. LoRA + scalar head + calibration temperature. not merged base models. customer-service P50 85.03 vs Jev 295.26 *theirs*. 1024/32 slower 1015.90 vs 301.37 *theirs*. systems latency ≠ semantic equivalence. Open-Jev TREC pending. hard acc ≠ calibrated Noul. type-valid ≠ exact. LoRA ≠ RLCD replica. Qwen/Qwen3.8-27B ≠ Archer. SHA move is not a replica. Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. +User-provided Open-Jev densify uniqueness lock: Zefan-Cai/Open-Jev densify HEAD 4933ee84951f README SHA ce1a587219e4; pushed 2026-09-21T01:34Z; Astra TREC commit 1dd56990be7e pushed 2026-09-21T01:17Z; live 3★ (was 0★; star-noise is not the fold); LoRA adapters plus trained scalar decision head and calibration temperature; not merged base models; dataset ZefanCai/Open-Jev rev c67699e13d0a; Open-Jev-2B rev 0c7aa498b162; Open-Jev-9B rev 47e966881e48; 27B still in progress; Independent of TypeSafe; no RLCD/parity claims; LoRA ≠ RLCD replica; customer-service P50 local HTTP 85.03 ms vs Jev HTTPS 295.26 ms *theirs*; 1024 tokens/32 candidates Open-Jev slower 1015.90 vs 301.37 *theirs*; prefix caching experimental/off by default; CUDA prefix caching exceeded tolerance on 9/11 workloads; systems latency ≠ semantic equivalence; GPT Luna P50 918.13 ms Astra 1938.39 ms *theirs*; TREC-DL Jev/Luna/Astra completed; Open-Jev TREC pending; 80,816 training rows; 2B 94.71% / OOD 86.02%; 9B 97.54% / 91.97% *theirs* not Harbor; hard acc ≠ calibrated Noul; type-valid ≠ exact; Qwen/Qwen3.8-27B ≠ Archer; website https://zefan-cai.github.io/open-jev/; launch X thread https://x.com/Zefan_Cai/status/2101782158658695388 https://x.com/Zefan_Cai/status/2101786019607740436 https://x.com/Zefan_Cai/status/2101789698947793231; densify §125 not a sibling first sighting; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §125 diff --git a/research/archive/findings.md b/research/archive/findings.md index 0023926..80f126f 100644 --- a/research/archive/findings.md +++ b/research/archive/findings.md @@ -215,6 +215,15 @@ Pulse: Archer still NOT landed. Hub archerhume/4rcherhume HTTP **401**. openjev **195★**; jev-visual **167★**; jev-mcp **156★**; litjev **28★**. `invented_signal: false`. +### Open-Jev densify (2026-09-21T01:34Z; still batch #107 / §125) + +DENSIFY the original Open-Jev card. Do not mint a sibling first sighting. +HEAD 4933ee84951f README SHA ce1a587219e4. live 3★ (star-noise). +not merged base models. systems latency ≠ semantic equivalence. +customer-service P50 85.03 vs Jev 295.26 *theirs*. +1024/32 1015.90 vs 301.37 *theirs*. Open-Jev TREC pending. +hard acc ≠ calibrated Noul. LoRA ≠ RLCD replica. *theirs* not Harbor. + ## Batch #106 (2026-09-20 ~13:40 Boise / ~19:40 UTC) - hourly 1340 HIGH @@ -4881,3 +4890,6 @@ Hourly 1746 uniqueness lock: TypeLLM/TypeLLM densify HEAD 702e6a287f3c README SH **User-provided 1936 HIGH (`notes.md` §130).** sgoedecke/system-one first-sighting. SystemOne.from_pretrained. TypeSafe-compatible ≠ TypeSafe replica. mithalouni/system-one-open first-sighting. 76.7% vs Jev 86.9% *theirs*. replica ≠ TypeSafe. kotoba-lang/typed-decisions first-sighting. DeBERTa-v3-large 0.855 / 42 ms *theirs*. kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions. aisearchio 15-link census catalog ≠ endorsement. soft scores ≠ hard gates. SHA move is not a replica. Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. User-provided 1936 uniqueness lock: sgoedecke/system-one 20★ HEAD ebde2a2db706 README SHA d331b567e2c3; SystemOne.from_pretrained; Batched single-token choice inference; TypeSafe-compatible; cache_prefix=True; LICENSE absent; sgoedecke/system-one ≠ mithalouni/system-one-open ≠ KathanModh259/system-one ≠ babybear-labs/system-one; TypeSafe-compatible ≠ TypeSafe replica; mithalouni/system-one-open 18★ MIT HEAD 77f1f7cccf8a README SHA 535f33028a68 LICENSE SHA 2f6f2cf1064e; Gemma 4 E2B / Gemma 3 270M Modal; 76.7% vs Jev 86.9% strict common subset *theirs*; 97 ms H100 *theirs*; 74.8% held-out *theirs*; replica ≠ TypeSafe; HF upload pending; kotoba-lang/typed-decisions 1★ Apache-2.0 HEAD 10d7834d3b99 README SHA 4d6bbf4c4e44 LICENSE SHA 513bb5e3cb4c; ModernBERT / DeBERTa / LLaDA-MoE; DeBERTa-v3-large 0.855 / 42 ms *theirs*; ModernBERT-base 0.717 / 68 ms *theirs*; LLaDA-MoE 0.835 / 676 ms *theirs*; kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions; encoder class member not Jev replica; aisearchio 15-link census catalog ≠ endorsement; 12 already carded 3 gaps this fold; soft scores ≠ hard gates; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §130 + +**Open-Jev densify (`notes.md` §125).** DENSIFY the original 1441 card, not a sibling first sighting. HEAD 4933ee84951f README SHA ce1a587219e4. LoRA + scalar head + calibration temperature. not merged base models. customer-service P50 85.03 vs Jev 295.26 *theirs*. 1024/32 slower 1015.90 vs 301.37 *theirs*. systems latency ≠ semantic equivalence. Open-Jev TREC pending. hard acc ≠ calibrated Noul. type-valid ≠ exact. LoRA ≠ RLCD replica. Qwen/Qwen3.8-27B ≠ Archer. SHA move is not a replica. Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. +User-provided Open-Jev densify uniqueness lock: Zefan-Cai/Open-Jev densify HEAD 4933ee84951f README SHA ce1a587219e4; pushed 2026-09-21T01:34Z; Astra TREC commit 1dd56990be7e pushed 2026-09-21T01:17Z; live 3★ (was 0★; star-noise is not the fold); LoRA adapters plus trained scalar decision head and calibration temperature; not merged base models; dataset ZefanCai/Open-Jev rev c67699e13d0a; Open-Jev-2B rev 0c7aa498b162; Open-Jev-9B rev 47e966881e48; 27B still in progress; Independent of TypeSafe; no RLCD/parity claims; LoRA ≠ RLCD replica; customer-service P50 local HTTP 85.03 ms vs Jev HTTPS 295.26 ms *theirs*; 1024 tokens/32 candidates Open-Jev slower 1015.90 vs 301.37 *theirs*; prefix caching experimental/off by default; CUDA prefix caching exceeded tolerance on 9/11 workloads; systems latency ≠ semantic equivalence; GPT Luna P50 918.13 ms Astra 1938.39 ms *theirs*; TREC-DL Jev/Luna/Astra completed; Open-Jev TREC pending; 80,816 training rows; 2B 94.71% / OOD 86.02%; 9B 97.54% / 91.97% *theirs* not Harbor; hard acc ≠ calibrated Noul; type-valid ≠ exact; Qwen/Qwen3.8-27B ≠ Archer; website https://zefan-cai.github.io/open-jev/; launch X thread https://x.com/Zefan_Cai/status/2101782158658695388 https://x.com/Zefan_Cai/status/2101786019607740436 https://x.com/Zefan_Cai/status/2101789698947793231; densify §125 not a sibling first sighting; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §125 diff --git a/research/changelog-hourly.md b/research/changelog-hourly.md index 96c0208..decb222 100644 --- a/research/changelog-hourly.md +++ b/research/changelog-hourly.md @@ -30,6 +30,20 @@ This is the uniqueness-lock archive after hourly folds (#2–#40 / notes aisearchio 15-link census catalog ≠ endorsement. SHA move is not a replica. - User-provided 1936 uniqueness lock: sgoedecke/system-one 20★ HEAD ebde2a2db706 README SHA d331b567e2c3; SystemOne.from_pretrained; Batched single-token choice inference; TypeSafe-compatible; cache_prefix=True; LICENSE absent; sgoedecke/system-one ≠ mithalouni/system-one-open ≠ KathanModh259/system-one ≠ babybear-labs/system-one; TypeSafe-compatible ≠ TypeSafe replica; mithalouni/system-one-open 18★ MIT HEAD 77f1f7cccf8a README SHA 535f33028a68 LICENSE SHA 2f6f2cf1064e; Gemma 4 E2B / Gemma 3 270M Modal; 76.7% vs Jev 86.9% strict common subset *theirs*; 97 ms H100 *theirs*; 74.8% held-out *theirs*; replica ≠ TypeSafe; HF upload pending; kotoba-lang/typed-decisions 1★ Apache-2.0 HEAD 10d7834d3b99 README SHA 4d6bbf4c4e44 LICENSE SHA 513bb5e3cb4c; ModernBERT / DeBERTa / LLaDA-MoE; DeBERTa-v3-large 0.855 / 42 ms *theirs*; ModernBERT-base 0.717 / 68 ms *theirs*; LLaDA-MoE 0.835 / 676 ms *theirs*; kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions; encoder class member not Jev replica; aisearchio 15-link census catalog ≠ endorsement; 12 already carded 3 gaps this fold; soft scores ≠ hard gates; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §130 +## Open-Jev densify (notes.md §125; 2026-09-21T01:34Z) + +- Fresh PR off latest `main` after merged #52 (hourly 1843 / §129). + **HARD RULE:** do not reopen or amend PR #23–#52. Does not bump 0.5.0. + Densify original §125. Do not mint a sibling first sighting. + Quote *theirs*. No wrappers. `invented_signal: false`. +- HEAD 4933ee84951f README SHA ce1a587219e4. Astra TREC commit 1dd56990be7e. + not merged base models. systems latency ≠ semantic equivalence. + Open-Jev TREC pending. hard acc ≠ calibrated Noul. LoRA ≠ RLCD replica. + SHA move is not a replica. Launch thread + https://x.com/Zefan_Cai/status/2101782158658695388 + https://x.com/Zefan_Cai/status/2101786019607740436 + https://x.com/Zefan_Cai/status/2101789698947793231. +- User-provided Open-Jev densify uniqueness lock: Zefan-Cai/Open-Jev densify HEAD 4933ee84951f README SHA ce1a587219e4; pushed 2026-09-21T01:34Z; Astra TREC commit 1dd56990be7e pushed 2026-09-21T01:17Z; live 3★ (was 0★; star-noise is not the fold); LoRA adapters plus trained scalar decision head and calibration temperature; not merged base models; dataset ZefanCai/Open-Jev rev c67699e13d0a; Open-Jev-2B rev 0c7aa498b162; Open-Jev-9B rev 47e966881e48; 27B still in progress; Independent of TypeSafe; no RLCD/parity claims; LoRA ≠ RLCD replica; customer-service P50 local HTTP 85.03 ms vs Jev HTTPS 295.26 ms *theirs*; 1024 tokens/32 candidates Open-Jev slower 1015.90 vs 301.37 *theirs*; prefix caching experimental/off by default; CUDA prefix caching exceeded tolerance on 9/11 workloads; systems latency ≠ semantic equivalence; GPT Luna P50 918.13 ms Astra 1938.39 ms *theirs*; TREC-DL Jev/Luna/Astra completed; Open-Jev TREC pending; 80,816 training rows; 2B 94.71% / OOD 86.02%; 9B 97.54% / 91.97% *theirs* not Harbor; hard acc ≠ calibrated Noul; type-valid ≠ exact; Qwen/Qwen3.8-27B ≠ Archer; website https://zefan-cai.github.io/open-jev/; launch X thread https://x.com/Zefan_Cai/status/2101782158658695388 https://x.com/Zefan_Cai/status/2101786019607740436 https://x.com/Zefan_Cai/status/2101789698947793231; densify §125 not a sibling first sighting; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §125 ## Hourly 1843 HIGH (notes.md §129 / items 481–496 / batch #111) @@ -3921,3 +3935,6 @@ Hourly 1746 uniqueness lock: TypeLLM/TypeLLM densify HEAD 702e6a287f3c README SH **User-provided 1936 HIGH (`notes.md` §130).** sgoedecke/system-one first-sighting. SystemOne.from_pretrained. TypeSafe-compatible ≠ TypeSafe replica. mithalouni/system-one-open first-sighting. 76.7% vs Jev 86.9% *theirs*. replica ≠ TypeSafe. kotoba-lang/typed-decisions first-sighting. DeBERTa-v3-large 0.855 / 42 ms *theirs*. kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions. aisearchio 15-link census catalog ≠ endorsement. soft scores ≠ hard gates. SHA move is not a replica. Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. User-provided 1936 uniqueness lock: sgoedecke/system-one 20★ HEAD ebde2a2db706 README SHA d331b567e2c3; SystemOne.from_pretrained; Batched single-token choice inference; TypeSafe-compatible; cache_prefix=True; LICENSE absent; sgoedecke/system-one ≠ mithalouni/system-one-open ≠ KathanModh259/system-one ≠ babybear-labs/system-one; TypeSafe-compatible ≠ TypeSafe replica; mithalouni/system-one-open 18★ MIT HEAD 77f1f7cccf8a README SHA 535f33028a68 LICENSE SHA 2f6f2cf1064e; Gemma 4 E2B / Gemma 3 270M Modal; 76.7% vs Jev 86.9% strict common subset *theirs*; 97 ms H100 *theirs*; 74.8% held-out *theirs*; replica ≠ TypeSafe; HF upload pending; kotoba-lang/typed-decisions 1★ Apache-2.0 HEAD 10d7834d3b99 README SHA 4d6bbf4c4e44 LICENSE SHA 513bb5e3cb4c; ModernBERT / DeBERTa / LLaDA-MoE; DeBERTa-v3-large 0.855 / 42 ms *theirs*; ModernBERT-base 0.717 / 68 ms *theirs*; LLaDA-MoE 0.835 / 676 ms *theirs*; kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions; encoder class member not Jev replica; aisearchio 15-link census catalog ≠ endorsement; 12 already carded 3 gaps this fold; soft scores ≠ hard gates; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §130 + +**Open-Jev densify (`notes.md` §125).** DENSIFY the original 1441 card, not a sibling first sighting. HEAD 4933ee84951f README SHA ce1a587219e4. LoRA + scalar head + calibration temperature. not merged base models. customer-service P50 85.03 vs Jev 295.26 *theirs*. 1024/32 slower 1015.90 vs 301.37 *theirs*. systems latency ≠ semantic equivalence. Open-Jev TREC pending. hard acc ≠ calibrated Noul. type-valid ≠ exact. LoRA ≠ RLCD replica. Qwen/Qwen3.8-27B ≠ Archer. SHA move is not a replica. Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. +User-provided Open-Jev densify uniqueness lock: Zefan-Cai/Open-Jev densify HEAD 4933ee84951f README SHA ce1a587219e4; pushed 2026-09-21T01:34Z; Astra TREC commit 1dd56990be7e pushed 2026-09-21T01:17Z; live 3★ (was 0★; star-noise is not the fold); LoRA adapters plus trained scalar decision head and calibration temperature; not merged base models; dataset ZefanCai/Open-Jev rev c67699e13d0a; Open-Jev-2B rev 0c7aa498b162; Open-Jev-9B rev 47e966881e48; 27B still in progress; Independent of TypeSafe; no RLCD/parity claims; LoRA ≠ RLCD replica; customer-service P50 local HTTP 85.03 ms vs Jev HTTPS 295.26 ms *theirs*; 1024 tokens/32 candidates Open-Jev slower 1015.90 vs 301.37 *theirs*; prefix caching experimental/off by default; CUDA prefix caching exceeded tolerance on 9/11 workloads; systems latency ≠ semantic equivalence; GPT Luna P50 918.13 ms Astra 1938.39 ms *theirs*; TREC-DL Jev/Luna/Astra completed; Open-Jev TREC pending; 80,816 training rows; 2B 94.71% / OOD 86.02%; 9B 97.54% / 91.97% *theirs* not Harbor; hard acc ≠ calibrated Noul; type-valid ≠ exact; Qwen/Qwen3.8-27B ≠ Archer; website https://zefan-cai.github.io/open-jev/; launch X thread https://x.com/Zefan_Cai/status/2101782158658695388 https://x.com/Zefan_Cai/status/2101786019607740436 https://x.com/Zefan_Cai/status/2101789698947793231; densify §125 not a sibling first sighting; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §125 diff --git a/research/notes.md b/research/notes.md index 8744c65..55d8553 100644 --- a/research/notes.md +++ b/research/notes.md @@ -31216,7 +31216,7 @@ replica; a 0.8 pass-min is application policy, not a proof. burnigtm/jev-mcp. Ten MCP tools. Gate cousins stay measurement notes; rh-guard owns primary gates. 5. **LoRA ≠ RLCD replica** - (Zefan-Cai/Open-Jev; zhengxuyu/litjev). Quote *theirs* Open-Jev: + (Zefan-Cai/Open-Jev; zhengxuyu/litjev). Quote *theirs* Open-Jev: LoRA adapters plus a trained scalar decision head and calibration temperature. Does not reproduce proprietary RLCD. 2B 94.71% / 9B 97.54% hard test *theirs*. 2B OOD 86.02% / 9B OOD 91.97% @@ -31224,6 +31224,13 @@ replica; a 0.8 pass-min is application policy, not a proof. *theirs* litjev: Probabilities are not calibrated by default. Qwen/Qwen3.8-27B ≠ Archer. zhengxuyu/litjev ≠ alexwestco/llm-to-jev. Zefan-Cai/Open-Jev ≠ TheoLeeCJ/openjev ≠ razorback16/openjev. + Since last look (2026-09-21): not merged base models. + customer-service P50 local HTTP 85.03 ms vs Jev HTTPS 295.26 ms + *theirs*. 1024 tokens/32 candidates Open-Jev slower 1015.90 vs + 301.37 *theirs*. systems latency ≠ semantic equivalence. + prefix caching experimental/off by default. TREC-DL Jev/Luna/Astra + completed. Open-Jev TREC pending. hard acc ≠ calibrated Noul. + type-valid ≠ exact. ### HIGH (revisit densify; keep original section ids) @@ -31284,6 +31291,11 @@ replica; a 0.8 pass-min is application policy, not a proof. Hub revisions 2B `0c7aa498b162` / 9B `47e966881e48`. Zefan-Cai/Open-Jev ≠ TheoLeeCJ/openjev ≠ razorback16/openjev. Do **not** copy train flags. + Since last look 2026-09-21: densify this card (HEAD + `4933ee84951f` / README SHA `ce1a587219e4` / live **3★**). + not merged base models. systems latency ≠ semantic equivalence. + Open-Jev TREC pending. hard acc ≠ calibrated Noul. + See **Since last look** below. Do not mint a sibling section. 6. **[cristianoliveira/jeq](https://github.com/cristianoliveira/jeq)** - NEW HIGH measurement (MIT; **3★**; HEAD `44ea80c90903`; @@ -31390,6 +31402,69 @@ Hooks for the reviewer: Hourly 1441 uniqueness lock: vLLM NVIDIA + MLX Apple Silicon; Codiv hosted free endpoint; dual /v1/systemone + /v1/chat/completions; razorback16/openjev densify HEAD cddbd962c88a README SHA a5943415cb92; STE README rewrite; serving-port densify; chat 501 on MLX; dual serving is not generate; Hosted Codiv ≠ TypeSafe; wire-compat ≠ logit-equiv; SHA move is not a replica; Error contract is not a Noul; hr98w/jev-visual 167★ Apple Silicon visual candidate scoring; 37.30s → 2.40s at 64 decisions *theirs*; Breakout 9 bricks 6 returns 2 lives *theirs*; candidate probabilities are relative not correctness; jkudish/jev-mcp 156★ ten MCP tools; recommendation is advisory; the server never blocks on its own; TypeSafe CLERC 5% to 18% *theirs*; jkudish/jev-mcp ≠ burnigtm/jev-mcp; zhengxuyu/litjev off-the-shelf Qwen decision layer; Probabilities are not calibrated by default; Qwen/Qwen3.8-27B ≠ Archer; zhengxuyu/litjev ≠ alexwestco/llm-to-jev; Zefan-Cai/Open-Jev LoRA + scalar head; 2B 94.71% 9B 97.54% hard test *theirs*; 2B OOD 86.02% 9B OOD 91.97% *theirs*; 80,816 training rows; 27B still in progress; LoRA ≠ RLCD replica; Zefan-Cai/Open-Jev ≠ TheoLeeCJ/openjev ≠ razorback16/openjev; cristianoliveira/jeq intelligence you can pipe; pass-min 0.8 still soft; JEQ does not own actions; AndyInQtr/laya-coreai CoreML serving substrate ≠ calibrated replica; AndyInQtr/laya-coreai ≠ mizorewww/laya-coreml; catalog ≠ endorsement; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47; notes.md §125 +### Since last look (2026-09-21T01:34Z Open-Jev densify) + +DENSIFY §125 item 5. Do **not** mint a sibling first-sighting +section. Material change is README / latency report / provider +comparison / TREC-DL / public HF packs, not the 0★→3★ star +move. SHA move is not a replica. + +**[Zefan-Cai/Open-Jev](https://github.com/Zefan-Cai/Open-Jev)** +- DENSIFY §125 (MIT source; Apache-2.0 adapters; live **3★**; + HEAD `4933ee84951f`; README SHA `ce1a587219e4`; pushed + `2026-09-21T01:34:47Z`; was HEAD `6d8de5ed72a0` / README SHA + `771bf3135f50` / 0★ at first sighting). Astra TREC commit + `1dd56990be7e` (pushed 2026-09-21T01:17Z). Quote *theirs*: + the 2B/9B artifacts are LoRA adapters plus a trained scalar + decision head and calibration temperature. They are not + merged base models or ordinary text-generation checkpoints. + Independent of TypeSafe. Does not reproduce proprietary RLCD + and does not claim TypeSafe speedups or parity. LoRA ≠ RLCD + replica. not merged base models. +- Public HF packs (revisions unchanged this look): dataset + [`ZefanCai/Open-Jev`](https://huggingface.co/datasets/ZefanCai/Open-Jev) + rev `c67699e13d0a`; [`Open-Jev-2B`](https://huggingface.co/ZefanCai/Open-Jev-2B) + rev `0c7aa498b162`; [`Open-Jev-9B`](https://huggingface.co/ZefanCai/Open-Jev-9B) + rev `47e966881e48`. 27B still in progress. +- Latency vs Jev-1.13.0 *theirs*: customer-service P50 local + HTTP **85.03 ms** versus Jev HTTPS **295.26 ms**. At 1024 + state tokens and 32 candidates Open-Jev is slower: + **1015.90 ms** versus **301.37 ms**. Hardware and network + paths differ; this is observed deployment latency, not + matched-hardware speedup. systems latency ≠ semantic + equivalence. Prefix caching is experimental and off by + default. CUDA prefix caching exceeded the probability + tolerance on 9/11 workloads; all selected decisions matched. +- Provider comparison *theirs*: same customer-service request + OpenAI Luna P50 **918.13 ms** and Astra **1938.39 ms** with + recorded reasoning settings. Latency does not establish + equal task quality. +- TREC-DL *theirs*: Jev / Luna / Astra completed (97 queries). + Open-Jev TREC pending. Do not quote Jev/Luna/Astra nDCG as + an Open-Jev result. +- Training restated *theirs* not Harbor: 80,816 training rows; + 2B hard test 9,515 / 10,046 (94.71%) / OOD 13,287 / 15,446 + (86.02%); 9B 9,799 / 10,046 (97.54%) / OOD 14,205 / 15,446 + (91.97%). hard acc ≠ calibrated Noul. type-valid ≠ exact. + Qwen/Qwen3.8-27B ≠ Archer. +- Website https://zefan-cai.github.io/open-jev/ plus launch X + thread *theirs*. Quote *theirs*: Inspired by Jev, we built + Open-Jev: open-source decision models. 2B/9B: LoRA adapters + + decision heads. Demo + [2101782158658695388](https://x.com/Zefan_Cai/status/2101782158658695388). + Intro reply + [2101786019607740436](https://x.com/Zefan_Cai/status/2101786019607740436) + (52-second intro; HF collection + ZefanCai/open-jev-6ab049b9d43a267bae4dedc8). Website reply + [2101789698947793231](https://x.com/Zefan_Cai/status/2101789698947793231). + Do **not** copy `pip` / train flags / `hf download`. + Zefan-Cai/Open-Jev ≠ TheoLeeCJ/openjev ≠ razorback16/openjev + ≠ Shalimov04/open-jev ≠ kyegomez/open-jev. + Do not reopen or amend PR #23–#52. Does not bump 0.5.0. + `invented_signal: false`. + +User-provided Open-Jev densify uniqueness lock: Zefan-Cai/Open-Jev densify HEAD 4933ee84951f README SHA ce1a587219e4; pushed 2026-09-21T01:34Z; Astra TREC commit 1dd56990be7e pushed 2026-09-21T01:17Z; live 3★ (was 0★; star-noise is not the fold); LoRA adapters plus trained scalar decision head and calibration temperature; not merged base models; dataset ZefanCai/Open-Jev rev c67699e13d0a; Open-Jev-2B rev 0c7aa498b162; Open-Jev-9B rev 47e966881e48; 27B still in progress; Independent of TypeSafe; no RLCD/parity claims; LoRA ≠ RLCD replica; customer-service P50 local HTTP 85.03 ms vs Jev HTTPS 295.26 ms *theirs*; 1024 tokens/32 candidates Open-Jev slower 1015.90 vs 301.37 *theirs*; prefix caching experimental/off by default; CUDA prefix caching exceeded tolerance on 9/11 workloads; systems latency ≠ semantic equivalence; GPT Luna P50 918.13 ms Astra 1938.39 ms *theirs*; TREC-DL Jev/Luna/Astra completed; Open-Jev TREC pending; 80,816 training rows; 2B 94.71% / OOD 86.02%; 9B 97.54% / 91.97% *theirs* not Harbor; hard acc ≠ calibrated Noul; type-valid ≠ exact; Qwen/Qwen3.8-27B ≠ Archer; website https://zefan-cai.github.io/open-jev/; launch X thread https://x.com/Zefan_Cai/status/2101782158658695388 https://x.com/Zefan_Cai/status/2101786019607740436 https://x.com/Zefan_Cai/status/2101789698947793231; densify §125 not a sibling first sighting; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §125 + ## 126. Hourly 1542 HIGH (2026-09-20 ~15:42 Boise / 2026-09-20T21:42Z) @@ -32795,3 +32870,6 @@ Hooks for the reviewer: User-provided 1936 uniqueness lock: sgoedecke/system-one 20★ HEAD ebde2a2db706 README SHA d331b567e2c3; SystemOne.from_pretrained; Batched single-token choice inference; TypeSafe-compatible; cache_prefix=True; LICENSE absent; sgoedecke/system-one ≠ mithalouni/system-one-open ≠ KathanModh259/system-one ≠ babybear-labs/system-one; TypeSafe-compatible ≠ TypeSafe replica; mithalouni/system-one-open 18★ MIT HEAD 77f1f7cccf8a README SHA 535f33028a68 LICENSE SHA 2f6f2cf1064e; Gemma 4 E2B / Gemma 3 270M Modal; 76.7% vs Jev 86.9% strict common subset *theirs*; 97 ms H100 *theirs*; 74.8% held-out *theirs*; replica ≠ TypeSafe; HF upload pending; kotoba-lang/typed-decisions 1★ Apache-2.0 HEAD 10d7834d3b99 README SHA 4d6bbf4c4e44 LICENSE SHA 513bb5e3cb4c; ModernBERT / DeBERTa / LLaDA-MoE; DeBERTa-v3-large 0.855 / 42 ms *theirs*; ModernBERT-base 0.717 / 68 ms *theirs*; LLaDA-MoE 0.835 / 676 ms *theirs*; kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions; encoder class member not Jev replica; aisearchio 15-link census catalog ≠ endorsement; 12 already carded 3 gaps this fold; soft scores ≠ hard gates; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §130 **User-provided 1936 HIGH (`notes.md` §130).** sgoedecke/system-one first-sighting. SystemOne.from_pretrained. TypeSafe-compatible ≠ TypeSafe replica. mithalouni/system-one-open first-sighting. 76.7% vs Jev 86.9% *theirs*. replica ≠ TypeSafe. kotoba-lang/typed-decisions first-sighting. DeBERTa-v3-large 0.855 / 42 ms *theirs*. kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions. aisearchio 15-link census catalog ≠ endorsement. soft scores ≠ hard gates. SHA move is not a replica. Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. + +**Open-Jev densify (`notes.md` §125).** DENSIFY the original 1441 card, not a sibling first sighting. HEAD 4933ee84951f README SHA ce1a587219e4. LoRA + scalar head + calibration temperature. not merged base models. customer-service P50 85.03 vs Jev 295.26 *theirs*. 1024/32 slower 1015.90 vs 301.37 *theirs*. systems latency ≠ semantic equivalence. Open-Jev TREC pending. hard acc ≠ calibrated Noul. type-valid ≠ exact. LoRA ≠ RLCD replica. Qwen/Qwen3.8-27B ≠ Archer. SHA move is not a replica. Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. +User-provided Open-Jev densify uniqueness lock: Zefan-Cai/Open-Jev densify HEAD 4933ee84951f README SHA ce1a587219e4; pushed 2026-09-21T01:34Z; Astra TREC commit 1dd56990be7e pushed 2026-09-21T01:17Z; live 3★ (was 0★; star-noise is not the fold); LoRA adapters plus trained scalar decision head and calibration temperature; not merged base models; dataset ZefanCai/Open-Jev rev c67699e13d0a; Open-Jev-2B rev 0c7aa498b162; Open-Jev-9B rev 47e966881e48; 27B still in progress; Independent of TypeSafe; no RLCD/parity claims; LoRA ≠ RLCD replica; customer-service P50 local HTTP 85.03 ms vs Jev HTTPS 295.26 ms *theirs*; 1024 tokens/32 candidates Open-Jev slower 1015.90 vs 301.37 *theirs*; prefix caching experimental/off by default; CUDA prefix caching exceeded tolerance on 9/11 workloads; systems latency ≠ semantic equivalence; GPT Luna P50 918.13 ms Astra 1938.39 ms *theirs*; TREC-DL Jev/Luna/Astra completed; Open-Jev TREC pending; 80,816 training rows; 2B 94.71% / OOD 86.02%; 9B 97.54% / 91.97% *theirs* not Harbor; hard acc ≠ calibrated Noul; type-valid ≠ exact; Qwen/Qwen3.8-27B ≠ Archer; website https://zefan-cai.github.io/open-jev/; launch X thread https://x.com/Zefan_Cai/status/2101782158658695388 https://x.com/Zefan_Cai/status/2101786019607740436 https://x.com/Zefan_Cai/status/2101789698947793231; densify §125 not a sibling first sighting; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §125 diff --git a/research/refresh-log.md b/research/refresh-log.md index f5c3c99..1a9af3d 100644 --- a/research/refresh-log.md +++ b/research/refresh-log.md @@ -14,6 +14,21 @@ kotoba ≠ laya / census ≠ endorsement. Quote *theirs*. No wrappers. `invented_signal: false`. - User-provided 1936 uniqueness lock: sgoedecke/system-one 20★ HEAD ebde2a2db706 README SHA d331b567e2c3; SystemOne.from_pretrained; Batched single-token choice inference; TypeSafe-compatible; cache_prefix=True; LICENSE absent; sgoedecke/system-one ≠ mithalouni/system-one-open ≠ KathanModh259/system-one ≠ babybear-labs/system-one; TypeSafe-compatible ≠ TypeSafe replica; mithalouni/system-one-open 18★ MIT HEAD 77f1f7cccf8a README SHA 535f33028a68 LICENSE SHA 2f6f2cf1064e; Gemma 4 E2B / Gemma 3 270M Modal; 76.7% vs Jev 86.9% strict common subset *theirs*; 97 ms H100 *theirs*; 74.8% held-out *theirs*; replica ≠ TypeSafe; HF upload pending; kotoba-lang/typed-decisions 1★ Apache-2.0 HEAD 10d7834d3b99 README SHA 4d6bbf4c4e44 LICENSE SHA 513bb5e3cb4c; ModernBERT / DeBERTa / LLaDA-MoE; DeBERTa-v3-large 0.855 / 42 ms *theirs*; ModernBERT-base 0.717 / 68 ms *theirs*; LLaDA-MoE 0.835 / 676 ms *theirs*; kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions; encoder class member not Jev replica; aisearchio 15-link census catalog ≠ endorsement; 12 already carded 3 gaps this fold; soft scores ≠ hard gates; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §130 +## 2026-09-21 ~01:34 UTC / ~19:34 Boise - Open-Jev densify +- Fresh PR off latest `main` after merged #52 (hourly 1843 / `notes.md` + §129 / items 481–496 / batch #111). Densify original `notes.md` §125. + Do **not** mint a sibling first sighting. + **HARD RULE:** do not reopen or amend PR #23–#52. +- PRIMARY: Zefan-Cai/Open-Jev latency / providers / TREC / HF packs. + HEAD 4933ee84951f README SHA ce1a587219e4. Astra TREC 1dd56990be7e. + systems latency ≠ semantic equivalence. Open-Jev TREC pending. + hard acc ≠ calibrated Noul. LoRA ≠ RLCD replica. not merged base models. + SHA move is not a replica. +- uniqueness_gate 0843+0915+jcr+0922+0940+0947+1049+1143+1248+1340+1441+1542+1643+1746+1843+Open-Jev densify. + Evaluator: systems latency ≠ semantic equivalence / hard acc ≠ Noul / + LoRA pack ≠ merged base / prefix cache off / TREC pending. + Quote *theirs*. No wrappers. `invented_signal: false`. +- User-provided Open-Jev densify uniqueness lock: Zefan-Cai/Open-Jev densify HEAD 4933ee84951f README SHA ce1a587219e4; pushed 2026-09-21T01:34Z; Astra TREC commit 1dd56990be7e pushed 2026-09-21T01:17Z; live 3★ (was 0★; star-noise is not the fold); LoRA adapters plus trained scalar decision head and calibration temperature; not merged base models; dataset ZefanCai/Open-Jev rev c67699e13d0a; Open-Jev-2B rev 0c7aa498b162; Open-Jev-9B rev 47e966881e48; 27B still in progress; Independent of TypeSafe; no RLCD/parity claims; LoRA ≠ RLCD replica; customer-service P50 local HTTP 85.03 ms vs Jev HTTPS 295.26 ms *theirs*; 1024 tokens/32 candidates Open-Jev slower 1015.90 vs 301.37 *theirs*; prefix caching experimental/off by default; CUDA prefix caching exceeded tolerance on 9/11 workloads; systems latency ≠ semantic equivalence; GPT Luna P50 918.13 ms Astra 1938.39 ms *theirs*; TREC-DL Jev/Luna/Astra completed; Open-Jev TREC pending; 80,816 training rows; 2B 94.71% / OOD 86.02%; 9B 97.54% / 91.97% *theirs* not Harbor; hard acc ≠ calibrated Noul; type-valid ≠ exact; Qwen/Qwen3.8-27B ≠ Archer; website https://zefan-cai.github.io/open-jev/; launch X thread https://x.com/Zefan_Cai/status/2101782158658695388 https://x.com/Zefan_Cai/status/2101786019607740436 https://x.com/Zefan_Cai/status/2101789698947793231; densify §125 not a sibling first sighting; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §125 ## 2026-09-21 ~00:43 UTC / ~18:43 Boise - Hourly 1843 HIGH @@ -3377,3 +3392,6 @@ Hourly 1746 uniqueness lock: TypeLLM/TypeLLM densify HEAD 702e6a287f3c README SH **User-provided 1936 HIGH (`notes.md` §130).** sgoedecke/system-one first-sighting. SystemOne.from_pretrained. TypeSafe-compatible ≠ TypeSafe replica. mithalouni/system-one-open first-sighting. 76.7% vs Jev 86.9% *theirs*. replica ≠ TypeSafe. kotoba-lang/typed-decisions first-sighting. DeBERTa-v3-large 0.855 / 42 ms *theirs*. kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions. aisearchio 15-link census catalog ≠ endorsement. soft scores ≠ hard gates. SHA move is not a replica. Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. User-provided 1936 uniqueness lock: sgoedecke/system-one 20★ HEAD ebde2a2db706 README SHA d331b567e2c3; SystemOne.from_pretrained; Batched single-token choice inference; TypeSafe-compatible; cache_prefix=True; LICENSE absent; sgoedecke/system-one ≠ mithalouni/system-one-open ≠ KathanModh259/system-one ≠ babybear-labs/system-one; TypeSafe-compatible ≠ TypeSafe replica; mithalouni/system-one-open 18★ MIT HEAD 77f1f7cccf8a README SHA 535f33028a68 LICENSE SHA 2f6f2cf1064e; Gemma 4 E2B / Gemma 3 270M Modal; 76.7% vs Jev 86.9% strict common subset *theirs*; 97 ms H100 *theirs*; 74.8% held-out *theirs*; replica ≠ TypeSafe; HF upload pending; kotoba-lang/typed-decisions 1★ Apache-2.0 HEAD 10d7834d3b99 README SHA 4d6bbf4c4e44 LICENSE SHA 513bb5e3cb4c; ModernBERT / DeBERTa / LLaDA-MoE; DeBERTa-v3-large 0.855 / 42 ms *theirs*; ModernBERT-base 0.717 / 68 ms *theirs*; LLaDA-MoE 0.835 / 676 ms *theirs*; kotoba-lang/typed-decisions ≠ convaiinnovations/laya-typed-decisions; encoder class member not Jev replica; aisearchio 15-link census catalog ≠ endorsement; 12 already carded 3 gaps this fold; soft scores ≠ hard gates; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §130 + +**Open-Jev densify (`notes.md` §125).** DENSIFY the original 1441 card, not a sibling first sighting. HEAD 4933ee84951f README SHA ce1a587219e4. LoRA + scalar head + calibration temperature. not merged base models. customer-service P50 85.03 vs Jev 295.26 *theirs*. 1024/32 slower 1015.90 vs 301.37 *theirs*. systems latency ≠ semantic equivalence. Open-Jev TREC pending. hard acc ≠ calibrated Noul. type-valid ≠ exact. LoRA ≠ RLCD replica. Qwen/Qwen3.8-27B ≠ Archer. SHA move is not a replica. Do not reopen or amend PR #23–#52. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. +User-provided Open-Jev densify uniqueness lock: Zefan-Cai/Open-Jev densify HEAD 4933ee84951f README SHA ce1a587219e4; pushed 2026-09-21T01:34Z; Astra TREC commit 1dd56990be7e pushed 2026-09-21T01:17Z; live 3★ (was 0★; star-noise is not the fold); LoRA adapters plus trained scalar decision head and calibration temperature; not merged base models; dataset ZefanCai/Open-Jev rev c67699e13d0a; Open-Jev-2B rev 0c7aa498b162; Open-Jev-9B rev 47e966881e48; 27B still in progress; Independent of TypeSafe; no RLCD/parity claims; LoRA ≠ RLCD replica; customer-service P50 local HTTP 85.03 ms vs Jev HTTPS 295.26 ms *theirs*; 1024 tokens/32 candidates Open-Jev slower 1015.90 vs 301.37 *theirs*; prefix caching experimental/off by default; CUDA prefix caching exceeded tolerance on 9/11 workloads; systems latency ≠ semantic equivalence; GPT Luna P50 918.13 ms Astra 1938.39 ms *theirs*; TREC-DL Jev/Luna/Astra completed; Open-Jev TREC pending; 80,816 training rows; 2B 94.71% / OOD 86.02%; 9B 97.54% / 91.97% *theirs* not Harbor; hard acc ≠ calibrated Noul; type-valid ≠ exact; Qwen/Qwen3.8-27B ≠ Archer; website https://zefan-cai.github.io/open-jev/; launch X thread https://x.com/Zefan_Cai/status/2101782158658695388 https://x.com/Zefan_Cai/status/2101786019607740436 https://x.com/Zefan_Cai/status/2101789698947793231; densify §125 not a sibling first sighting; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48/#49/#50/#51/#52; notes.md §125 diff --git a/research/revisit_fingerprints.json b/research/revisit_fingerprints.json index de6b2b6..15c9c8a 100644 --- a/research/revisit_fingerprints.json +++ b/research/revisit_fingerprints.json @@ -20,7 +20,7 @@ "forks_count", "likes" ], - "note": "Last-look snapshots for already-catalogued sources. Hourly diffs these four fingerprints. Star-noise is not a fold. SHA move is not a replica. Seeded from merged notes (SemIf \u00a7117, NanoJev \u00a7115) plus hourly 1248 densify (openjev \u00a775, von \u00a749, verdict \u00a771, kev \u00a745, jeff \u00a760, TypeLLM \u00a7113) plus hourly 1340 densify (openjev sdk 0.7 + MLX 400, kev PLAN_Qwen35) plus hourly 1441 densify (openjev STE backends+Codiv dual serving) and 1441 first sightings plus hourly 1542 densify (TypeLLM README 3k\u219212k B, kev family new-source) and 1542 first sightings (pi-jev densify, jev-sentinel, tool-routers, leanest, jevals, MrJev, jev-firewall, jev-codex-approval, HF encoder/quanto). default_sha is the full HEAD. pushed_at is the GitHub push clock. description_hash is sha256[:12] of the GitHub/Space description, or null when notes do not quote it. README SHA is an optional extra, not a substitute for description_hash. plus hourly 1746 densify (TypeLLM truncated thinking + qwen35_small, kev 0.8B Qwen3.5 family, jev-pruner \u00a753, jevassert \u00a770, jev-packs \u00a764) and 1746 first sightings (vexjoy, Canny, jev-engineering, five-lines, jev-table, kev-ane) plus hourly 1843 densify (kev own-data JSONL / --init_from warm-start, SHA 8465c4c4c294 \u2192 bd058057ad0a) and 1843 first sightings (assay-001, kyegomez reconstruction, jev-ra, catalogs). plus user-provided 1936 first sightings (sgoedecke/system-one, mithalouni/system-one-open, kotoba-lang/typed-decisions).", + "note": "Last-look snapshots for already-catalogued sources. Hourly diffs these four fingerprints. Star-noise is not a fold. SHA move is not a replica. Seeded from merged notes (SemIf \u00a7117, NanoJev \u00a7115) plus hourly 1248 densify (openjev \u00a775, von \u00a749, verdict \u00a771, kev \u00a745, jeff \u00a760, TypeLLM \u00a7113) plus hourly 1340 densify (openjev sdk 0.7 + MLX 400, kev PLAN_Qwen35) plus hourly 1441 densify (openjev STE backends+Codiv dual serving) and 1441 first sightings plus hourly 1542 densify (TypeLLM README 3k\u219212k B, kev family new-source) and 1542 first sightings (pi-jev densify, jev-sentinel, tool-routers, leanest, jevals, MrJev, jev-firewall, jev-codex-approval, HF encoder/quanto). default_sha is the full HEAD. pushed_at is the GitHub push clock. description_hash is sha256[:12] of the GitHub/Space description, or null when notes do not quote it. README SHA is an optional extra, not a substitute for description_hash. plus hourly 1746 densify (TypeLLM truncated thinking + qwen35_small, kev 0.8B Qwen3.5 family, jev-pruner \u00a753, jevassert \u00a770, jev-packs \u00a764) and 1746 first sightings (vexjoy, Canny, jev-engineering, five-lines, jev-table, kev-ane) plus hourly 1843 densify (kev own-data JSONL / --init_from warm-start, SHA 8465c4c4c294 \u2192 bd058057ad0a) and 1843 first sightings (assay-001, kyegomez reconstruction, jev-ra, catalogs). plus user-provided 1936 first sightings (sgoedecke/system-one, mithalouni/system-one-open, kotoba-lang/typed-decisions). plus Open-Jev densify 2026-09-21 (github:Zefan-Cai/Open-Jev SHA 6d8de5ed72a0 → 4933ee84951f; README 771bf3135f50 → ce1a587219e4; HF 2B/9B packs + dataset last_look; densify §125 not a sibling).", "looks": [ { "id": "github:TheoLeeCJ/SemIf", @@ -169,14 +169,38 @@ { "id": "github:Zefan-Cai/Open-Jev", "notes_section": "125", - "last_look": "2026-09-20T20:41Z", + "last_look": "2026-09-21T01:34Z", + "fingerprints": { + "default_sha": "4933ee84951f1a3b94b8be1f7490f02a4fa4ba24", + "pushed_at": "2026-09-21T01:34:47Z", + "description_hash": null, + "release_tag": null + }, + "readme_sha": "ce1a587219e4b9bcb5db3787a654f248d9a84b92" + }, + { + "id": "hf:ZefanCai/Open-Jev-2B", + "notes_section": "125", + "last_look": "2026-09-21T01:34Z", + "fingerprints": { + "default_sha": "0c7aa498b1627be8da4acf34c863ff0ee0a92785", + "pushed_at": "2026-09-20T20:22:14.000Z", + "description_hash": null, + "release_tag": null + }, + "readme_sha": null + }, + { + "id": "hf:ZefanCai/Open-Jev-9B", + "notes_section": "125", + "last_look": "2026-09-21T01:34Z", "fingerprints": { - "default_sha": "6d8de5ed72a09e56cc6ee331f9181ea6ab2a5858", - "pushed_at": "2026-09-20T20:42:40Z", + "default_sha": "47e966881e489511c0c7f5633a9e1960a676a551", + "pushed_at": "2026-09-20T20:22:28.000Z", "description_hash": null, "release_tag": null }, - "readme_sha": "771bf3135f50fd08c8607b25d52e0549a130e025" + "readme_sha": null }, { "id": "github:cristianoliveira/jeq", @@ -982,7 +1006,7 @@ { "id": "hf:ds:ZefanCai/Open-Jev", "notes_section": "125", - "last_look": "2026-09-20T22:43Z", + "last_look": "2026-09-21T01:34Z", "fingerprints": { "default_sha": "c67699e13d0ae25e35b77165a4b6b079bedc8aba", "pushed_at": "2026-09-20T22:49:59.000Z", diff --git a/research/revisit_fingerprints.py b/research/revisit_fingerprints.py index 0fed3ad..2db7508 100644 --- a/research/revisit_fingerprints.py +++ b/research/revisit_fingerprints.py @@ -276,6 +276,15 @@ def self_test() -> None: assert nano["release_tag"] == "unified-games-v1" nano_readme = by_id["github:TianyuCodings/NanoJev"].get("readme_sha") assert isinstance(nano_readme, str) and nano_readme.startswith("4190093c64ee") + openjev = by_id["github:Zefan-Cai/Open-Jev"]["fingerprints"] + assert openjev["default_sha"] == "4933ee84951f1a3b94b8be1f7490f02a4fa4ba24" + assert openjev["pushed_at"] == "2026-09-21T01:34:47Z" + openjev_readme = by_id["github:Zefan-Cai/Open-Jev"].get("readme_sha") + assert isinstance(openjev_readme, str) and openjev_readme.startswith("ce1a587219e4") + assert by_id["hf:ZefanCai/Open-Jev-2B"]["fingerprints"]["default_sha"].startswith("0c7aa498b162") + assert by_id["hf:ZefanCai/Open-Jev-9B"]["fingerprints"]["default_sha"].startswith("47e966881e48") + assert by_id["hf:ds:ZefanCai/Open-Jev"]["fingerprints"]["default_sha"].startswith("c67699e13d0a") + assert by_id["hf:ds:ZefanCai/Open-Jev"]["notes_section"] == "125" semif_readme = by_id["github:TheoLeeCJ/SemIf"].get("readme_sha") assert isinstance(semif_readme, str) and semif_readme.startswith("74ab7f7f") densify_original_ids = { @@ -289,6 +298,10 @@ def self_test() -> None: "github:tamaratran/jev-pruner": "53", "github:dtduc-git/jevassert": "70", "github:dtduc-git/jev-packs": "64", + "github:Zefan-Cai/Open-Jev": "125", + "hf:ZefanCai/Open-Jev-2B": "125", + "hf:ZefanCai/Open-Jev-9B": "125", + "hf:ds:ZefanCai/Open-Jev": "125", } for look_id, section in densify_original_ids.items(): assert look_id in by_id, look_id diff --git a/research/sources.json b/research/sources.json index ee18fe6..e21e843 100644 --- a/research/sources.json +++ b/research/sources.json @@ -5323,6 +5323,54 @@ "title": "aisearchio open-source Jev alternatives list", "url": "https://x.com/aisearchio/status/2101720039779086414", "note": "15-link census catalog ≠ endorsement. 12 already carded 3 gaps this fold. notes.md §130." + }, + { + "kind": "github", + "title": "Zefan-Cai/Open-Jev", + "url": "https://github.com/Zefan-Cai/Open-Jev", + "note": "Densify notes.md §125. HEAD 4933ee84951f README SHA ce1a587219e4. LoRA + scalar head, not merged base. LoRA ≠ RLCD replica. systems latency ≠ semantic equivalence. Open-Jev TREC pending." + }, + { + "kind": "hf-model", + "title": "ZefanCai/Open-Jev-2B", + "url": "https://huggingface.co/ZefanCai/Open-Jev-2B", + "note": "rev 0c7aa498b162. LoRA adapters plus scalar head, not merged base. notes.md §125." + }, + { + "kind": "hf-model", + "title": "ZefanCai/Open-Jev-9B", + "url": "https://huggingface.co/ZefanCai/Open-Jev-9B", + "note": "rev 47e966881e48. LoRA adapters plus scalar head, not merged base. notes.md §125." + }, + { + "kind": "hf-dataset", + "title": "ZefanCai/Open-Jev", + "url": "https://huggingface.co/datasets/ZefanCai/Open-Jev", + "note": "rev c67699e13d0a. 80,816 training rows *theirs* not Harbor. notes.md §125." + }, + { + "kind": "site", + "title": "Open-Jev website", + "url": "https://zefan-cai.github.io/open-jev/", + "note": "Project site + demos. notes.md §125 densify." + }, + { + "kind": "tweet", + "title": "Zefan_Cai Open-Jev launch demo", + "url": "https://x.com/Zefan_Cai/status/2101782158658695388", + "note": "Launch quote / demo. conversation 2101782158658695388. notes.md §125." + }, + { + "kind": "tweet", + "title": "Zefan_Cai Open-Jev introduction reply", + "url": "https://x.com/Zefan_Cai/status/2101786019607740436", + "note": "Introduction reply in launch thread. notes.md §125." + }, + { + "kind": "tweet", + "title": "Zefan_Cai Open-Jev website reply", + "url": "https://x.com/Zefan_Cai/status/2101789698947793231", + "note": "Website reply in launch thread. notes.md §125." } ] }