diff --git a/.agents/skills/augustus/SKILL.md b/.agents/skills/augustus/SKILL.md index e4343fc..c6fdb86 100644 --- a/.agents/skills/augustus/SKILL.md +++ b/.agents/skills/augustus/SKILL.md @@ -54,7 +54,7 @@ classical method you already trust, substitute it, classify the win "paraphrase brittleness", "allowlist then judge", "TOCTOU-of-Noul", "Jev inside the database / sqlite-jev", "Jev picks bitrate / join order / the model", "wait for Archer", "lint the request / missing - other", "training confronts Choice other / none-of-the-above", "soft AGENTS.md rules vs the linter", "screenshot Choice / omni System One", "extractive quotes / pointer not generator", "compaction summarize vs pointer", "encoder vs Jev compaction backend", "shadow-mode compaction rollout", "CI flaky-vs-real merge gate", "fail-open VOI wake/resume", "claim vs session evidence", "S1 indexer escalate-S2", "Harbor on/off routing", "fail-open vs fail-closed wake vs CI gate", "encoder vs Jev computer-use backend", "hybrid local decide + remote fill", "DONE vs verified success", "stdout prune vs session compaction", "OpenCode jev-pruner vs Claude jev-pruner", "zen-chat vs jev-zen Noul", "hard envelope then Noul prune", "Cua-S1 vs TypeSafe Jev", "plan vs execute dry-run", "specialist computer-use vs general agent", "local drop-in vs stub scorer", "route vs memory", "when does it hold / extractable from state", "decision model vs constrained LLM", "dual-process S1/S2", "combinatorial grid vs extractive", "uncalibrated local likelihoods", "decision-native RAG", "classify-first / read selectively", "living applied-mappings atlas / class patterns", "silence as safer / draft-gate heartbeat", "robotics text-state vs pixels", "verbatim ledger vs summary", "judgment as language primitive", "Stagehand extract pick-and-copy", "harness observe-score-act vs demo loop", "public judgment wall / six parallel questions", "meaning-search without embeddings", "attention ≠ correctness", "skills→oxlint / AST prove ∩ remainder", "session-sticky first-prompt routing", "measured RAG rerank vs generative rerank", "capability kernel / secrets never in the agent", "Jev is SENSOR not policy", "type-safe ≠ correct", "typed control plane around DSPy", "native vs verbalized confidence", "engine owns truth / Jev owns judgment", "human-confirmed kill gate", "train specialist vs few-shot hosted", "decide→policy→LLM leftover", "Noul 0.5 cannot-tell never rounded", "calibration ≠ sortable / ORDER BY", "pairwise inversion / Score ordinality / two-decimal ties", "wire-compat GLiFormer /v1/systemone", "class-backend economics", "loopback gateway hosted + local", "do not distill Jev as teacher", "active-learning triage", "evidence-packet explorer", "meaning-grep AND/OR/NOT", "closed-vote-only / no planner LLM", "Jev vs PCD Harbor", "PCD O(1) ≠ Noul", "host-owned handlers × System One", "OMP/pi fail-open gate", "permission vs probability / operator owns thresholds", "judgment ≠ permission / Jev never grants access", "eval integrity / instrument not score", "constrained optimizer + S1 features / never sole hot-path gate", "privilege ≠ verdict / effect contracts not tokens", "attention filter / VOI for human review / never blocks / never green unless sure", "measurement owns endorsement / evidence-gated question packs", "Jev supplies evidence / code owns authority", "ranking ≠ calibration / never hard-threshold raw p as frequency", "hot-click CU / indexed element table", "Jev judges relevance / code decides structure", "local rules first then remainder / never auto-train on own hides", "combinators / System One as control plane", "receipts not leaderboard / type-safe ≠ correct jaggedness", "VOI over skill library / skillranker abstention", "OOD calibration / AUC ≠ ECE", "Jev vs thinking-budget small models", "turnstile / replayable evidence≠authority", "MLX one-pass schema→JSON / Apple Silicon replica economics", "memory leases ended by new evidence", "never confidently wrong / TLA+ compose / escalate instead of hard-gate", "no seal no advance / coverage ledger / mint ≠ product brain", "skill-broker sibling / judgment ≠ permission", "sureness bands / max_prob is generous", "JevBench / calibration not in Main Score", "CI typed gate before expensive review", "Codex MCP host adapter", "judgment as attention redirect / jev-preflight", "compress-before-first-send / dizk jev-lens", "tools≠use / SessionStart over hoping", "observational memory / pi-om keep-kind", "open-Jev class / openvons / JevPick", "physical-world System One / HA-Jev / not for locks", "judgment outside the store / jevql", "landed-script trust / headless≠auto-approve", "digital-design combinators / extended five", "VOI cache admission / same-intent skip LLM", "BM25 vs Jev skill routing Harbor harness", "zeroshot vs BERT / contamination DiD", "typed escalate continue abort baton / inverted loop", "worth-your-attention VOI / ThinkyMiner Winnow", "Jev WHETHER Python HOW LLM WHAT", "conflict vs ignorance / named Choice escape", "Playwright executes Jev chooses", "OpenJev /v1/decide not drop-in", "SemIf wire-compat runoff; SemIf rename densify / MLX backend / 5.21× systems≠semantic / Softmax ≠ Noul (`notes.md` §117)", "decision-as-memory flywheel", "record/replay CI / jevassert", "failure-finding arena / jevarena ≠ jev-arena", "BBQ not a bias cert", "decider≠executor", "sentence-as-rule lint / jevlint", "sentence-as-rule lint / jev-lint is jevlint rename", "VOI hunk prune", "whole-repo intent VERIFIED/VIOLATION/UNKNOWN", "GLiNER2 spec ≠ replica", "open replica substrates / grande / laya-jolt / JEV-CPU", "ONNX local-jev not equivalent", "persist constraints across compaction / pi-heed", "calibration+cost first-class gates", "Harbor-shaped Jev vs SGR LLM-as-judge / jev-judge-bench ≠ jevarena ≠ jevbench", "hand no-text steps / jev-use / Vercel drops confidence", "Pi System-One control plane / pi-jev-control", "never free-generates / jev-gpt tree of Choices", "OpenRouter recipe atlas / samples not benches", "personal history feed / jevfeed / no social graph", "competing NAR claims / dual-channel ECE / openJev-verdict ≠ OpenJev", "empty compaction-proxy skip / IPECTER", "throughput ≠ latency / like-for-like ECE", "1-token logprob endpoint ≠ Noul / coverage ≠ correctness", "open replica engine / jevinf", "unofficial Elixir SDK ≠ OTP peer", "jevex n=16 files-to-read VOI", "commit pre-review attention≠verdict / middle band", "Hermes plugin is Agnes not TypeSafe", "pi-jev-compact ≠ pi-jev-compaction", "empty Codex-proxy skip / IPECTER runway", "decision-native inbox / mailordinal", "unofficial jev-cli not ready / ≠ jevql", "laya-multilingual / English checkpoint confident-wrong OOD", "schema-scorer peaked ranking ≠ calibration", "HF 401 / GitHub 404 Hub-only", "productized System One HTTP / classifier.dev", "escalate-under-threshold / smart tier / multi-label ignores", "silent FALLBACK / granite 0.546 vs advertised 0.800", "vs_jev tracked JSON / read eval/README", "choxos/jev-reviewer ≠ egma-ai / systematic-review pointer", "two-pass Choice+Noul evidence extraction", "not-found is an answer", "human check as productized judgment", "githubnext/localjev ≠ kunchenguid/local-jev", "wire-compat ≠ logit-equiv / prompted JSON ≠ structured read", "self-reported probs / entropy confidence", "GitHub Next local /v1/systemone", "LM Studio runner gap / structured-read primitives", "NandhaKishorM/laya packaging ≠ Hub-only / Router script-before-p", "post-T ECE ≠ raw ECE / Banking77 token-budget", "0.85 still soft / not TypeSafe drop-in", "external census ≠ scored bake-off", "GLiNER2+routers class-boundary", "incomplete openjev census vs watch", "Harbor honesty watch / silent fallback", "JevBench v1.2 geometric mean / cal ON rank / weight sensitivity", "option-order 72→21 / instruction models in the class table", "self-host latency ×2 assumption / est. costs", "Laya absent is a gap not a named exclusion", "Qwen3.8 27B ≠ Archer", "hourly already-folded watch / apply-the-five / skip thin noise", "hard-gate Noul as PR/quality gate is soundness theater", "S1 never stalls waiting / S2 one-use advisory", "Local controller ≠ githubnext/localjev", "purple telemetry = consumed not arrived", "seed = geometry not async replay", "20% starting gate still soft", "no pixels to either provider", "OCR+AX observe-score-act / typesafe-computer-use", "never send screenshot to frontier for the decision", "overlapping CU options = false low confidence", "split kind/item/site", "155× one-screenshot ≠ Harbor taskset", "decision ≠ answer-reader capture", "ASR observe-score-act / jev-voice-browser", "partial-speech VOI / free-text waits", "spoken confirm ≠ hard auth", "numbered overlay without another model", "wrap-as-execution / AgentGhost ALLOW ASK DENY", "rules first then Jev remainder / ASK throws / fail-closed", "reddpy/AgentGhost ≠ jwen5419807/agentghost ≠ vventirozos", "JP genre atlas / studio_yebisu / stars ephemeral ≠ eval", "Jev Clearly Explained / akshay_pachaar / LLM hammer", "schema-safe ≠ correct / 200× 400× TypeSafe ceiling", "questions-as-code / shadow first / not a TypeSafe how-to", "proposition ≠ embedding / contrast-set", "boolean composition of soft Nouls / AND OR NOT", "uehaj/jev-semgrep ≠ semgrep.dev", "meaning-grep dedicated fold / not a gate", "decision-validated UI / Jev never authors text", "decision-as-assert / jevtest ambiguous band", "typed decisions drive UI / jev2ui", "hybrid S1 closed verb menu / anima3", "pointer-not-generator search / JevFind", "jev-frontier-bench ≠ frontier-100", "product bakeoff ≠ architecture duel / GLiClass", "four engines same questions / majority floor", "authorship named escape / not evidence", "ha-switchboard HA remains execution", "n8n classify/route/score / Low Confidence", "fast-jev-compaction-pi ≠ pi-jev-compact ≠ pi-jev-compaction", "jevloop full-distribution optimizer / no LLM in the loop", "laya-vision SmolVLM / score untrained", "Cerebellum-2B /v1/decide ≠ TypeSafe / wire-compat vs agent-routing", "laya-grounded not drop-in / Platt not temperature", "GestaltLabs/Jeff-1 ≠ logan-markewich/jeff / acc vs ECE n=9730", "stanley-code empty findings ≠ approval / human promote", "findme ≠ JevFind / NL memory beam-search FS", "jevsubrouter price workers not conversation / counts ≠ dollars", "feelings .feels() default 0.5 is Noul-0.5-never-rounded / ≠ hunch ≠ Probably", "apa-agent-harness ≠ AntonioCoppe/jev-harness / unpublished npm", "grok-bot-jev skill cannot force a bot that ignores it / A/B proxies not tokens", "Essentiel-Jev never authority / human every action", "enzo-mcp independently falsifiable claims / ≠ jev-sift", "pigeonhole OTHER skip / decision-as-filing", "jev-reliability Nothing about accuracy", "clduab11/jev-test ≠ realZachi/jevtest / Nothing runs yet", "jev-rag-benchmark Jev wins is not an assumption", "dairui1/jev-lab ≠ BrendanH18/jev-lab", "jevmail gmail.readonly / mailjay archive/trash", "ZHUBoer/ego-jev reserved __none__", "runWorkflow completed ≠ success", "jsort scores are relative", "Noul not Choice for scale", "groundedness-judge-bench native vs schema-guided", "implicit_true included in yes", "jev_playground 0 promotions", "routing-backtest 0.0447%", "yuyang2230/jev-agent-skill jev-1.13-free", "jev-techstack-classifier stack_config.json", "s1_ruby collapse late", "undecided? abstain", "2389-research/judgement license null", "confidence ≠ winner p", "typesafeai-sdk-community not a new species", "tpellet/hunch exit 3", "never-execute list", "jev-file-search scores not calibrated accuracy", "jev-linkmap Jev never sees S2 prose", "muhammedilyasy/jev-mail metadata only", "tidy none-of-folders stay", "tab-bouncer pinned/audio/current never closed", "lkclean Show fail-open", "jev-yt-time-saver Show anyway", "ORIGIN pause-if-no-Jev", "validResponse sums-to-1", "jev-crawlers risk bands never raw boolean", "jevbrain AUTO_ACT is not a Noul", "judgekit YAML classify/score/route/verify", "typed-judge-kit verdict-in-code", "alsoleg89/decide packing VOI", "0.8 ≠ 80% accuracy", "Jev-Calibration Platt ECE 0.117→0.052", "jev-calibration-arena never acts", "ctmx/openrouter-jev-mcp Decision-as-Plugin", "FrancoisChastel/jev-code ≠ npm jev-code", "claudecode-jev-marketplace fail-open not hot path", "pedroknigge/mcp_jev packs not ask_jev", "cyrusasco/typesafe-mcp noul deadband 0.35–0.65", "codaaiteam/jev-skill jevtypesafeai.com ≠ TypeSafe", "hermes-switchyard ≠ hermes-jev-router ≠ hermes-plugin-jev", "nanoprune 2.8MB ECE 2.58%", "smartdio/jev-browser-agent ≠ ZHUBoer/ego-jev", "Dakai/omp-jev-web DONE ≠ proof", "hari007sh/jev ≠ dannote/jev", "0thernet/system-one-skills deterministic verify", "typed-gate band [0.40,0.60] is refusal", "pi-jev-gate fail-closed; choice is the verdict", "Foq ~25ms/2.2GB local", "rev prefill-only + HF jev-0.5b", "robfrase/jev planning memo", "typesafe_agent_gates 27/27 / 31/31", "EpicEric/safe-sh static remainder", "pastepilot Confirm before act", "Jev-Reranker live Jev not yet measured", "sessionwise opt-in relevance", "jev-search pointer sieve", "400ms Salesforce WebMCP", "typesafe-scheduler-diagnostics advisory", "droidjev screenshot-free", "Tewoto1 jevcu planner still writes", "ha-conversation-jev Jev→Grok", "dsh-jev can only gate", "jev-classification-benchmark specified not run", "jev-luna-pagerduty p≥0.50", "meldltd/meldecision laya-go ONNX", "laya-doom never pixels", "logixism/laya-api empty README", "akpsahan/laya ≠ Archer", "choxos/jevchess engine owns truth", "jev-drive sim not AV", "story-arc Jev never authors", "jev-hs-assistant HS6", "golergka/jev-plays-starcraft-2 UI-verified ≠ API Victory", "awesome-jev-use-cases catalog", "Nibir1/typesafe-go ≠ official", "fingerprint after redact", "recall vs decide", "publish fingerprints+answers", "CI replay as Harbor cousin", "Cache hit ≠ correctness", "hyperspaceai/jevcache ≠ kushals256/jevcache", "human labels only", "score never auto-accepts", "production capture flywheel", "sutro-sh/jev-align ≠ caiovicentino/jev-align", "guidance ≠ hook", "catalysts ≠ summaries", "compile-time System One", "unofficial ≠ TypeSafe", "format_version modernbert-jev/1", "Argos1111/jev_local ≠ us/jev-local ≠ kunchenguid/local-jev", "LFM default ≠ ModernBERT backend", "Nemotron ≠ TypeSafe Jev", "not a calibrated replacement", "djev-dev complements djev-spark", "images as Choice options", "Laya essay numbers *theirs*", "Router/OOD confidence", "hosted bootstrap ≠ silent TypeSafe", "difficulty + policy thresholds + JSONL trace", "jev-codex-pilot model + reasoning depth", "keep/shadow/hybrid/reject", "quarry evidence projection", "Frank-ZY-Dou/awesome-jev robotics/3D/control", "one-dollar-tahoe TypeSafe Jev defense eval", "jevguard calibrator/cache/escape", "jev-ci-selector CI shadow mode", "llama-jev llama.cpp replica", "petercr/jev-orchestrator ≠ FleeexCorp/jev-orchestrator", "seb4ez/jevguard ≠ AseemPrasad/JevGuard ≠ pablozr/JevGuard", "webNeat/llama-jev ≠ WiktorB2004/llama-index-jev", "OpenCode jev-pruner context sieve", "observe→score-candidates→prune", "jev-zen / jev-1.13-free", "zen-chat ≠ Noul", "fail-open original", "keepScore >0.1 floor", "host port of tamaratran/jev-pruner", "indiejoseph/opencode-jev-pruner ≠ nrdz-labs/fast-jev-opencode", "jev-webagent-bench empty stub", "Kiln-AI/jev_jsonschema noul_threshold 0.5", "NSStudent/JevSwiftSDK unofficial", "GLiNER2 native Apple path", "unofficial Swift/Core ML GLiNER 2.5-small", "entity spans + confidence", "not Choice/Score/Noul", "not TypeSafe", "label descriptions as schema", "on-device ANE economics", "honesty locks", "shershah1024/gliner-native-runtime ≠ Fastino", "≠ gliner25-compaction ≠ gliner2-ultrafast ≠ Eran-BA/Jev_from_GLiNER2 ≠ NSStudent/JevSwiftSDK ≠ jevmlx", "default threshold 0.1 still soft", "soft Noul ≠ hard safety", "Decision Graph Protocol frame→assess→commit", "app retains permissions/effects", "Jev-first assessor-neutral", "guarded commit / receipt/next frame", "assessment batching", "hard-gating DGP as safety theater", "numerous-com/dgp ≠ TypeSafe official", "jegrep calibrated path+range Nouls", "no embeddings/index/daemon", "~$0.01–0.03 typical", "agent --json", "can1357/jegrep ≠ Bentlybro/jevgrep ≠ uehaj/jev-semgrep", "Archer-arch fidelity", "kev family OOD 0.76–0.77 vs Jev 0.86", "block-causal isolation", "pointer/readout CE-trained", "/v1/systemone drop-in", "replica honesty", "cost-sensitive decision theory × System One probabilities → control flow", "thresholds derived from costs not hard-coded", "YES / NO / UNSURE from cost_false_yes / cost_false_no / cost_human", "auto-batching same-object questions", "Kungie/gut ≠ tpellet/hunch ≠ carldaws/hunch", "judgment vs generation", "deterministic execution after probabilistic judgment", "exactly one app-owned callback", "explicit uncertain branch", "Illusion47586/judge ≠ lexingtonhibiki/judgekit ≠ Ascurse/typed-judge-kit", "variable-N option scoring as the trainable object", "dynamic candidate bags not fixed label sets", "zwliJay/jev-forge ≠ NanoJev", "open replica economics / latency vs closed Jev", "NAR local drop-in", "wfzyx/von late-catch HIGH", "competing NAR claims / replica honesty", "typed judgments vs chat judges on guardrailing", "ishaannk/llm-vs-jev cross-note only", "deeper integrity fold is rh-guard", "nothing wins outright", "can be argued out of guarding"", "Jev IS the if-statement", "judgments/probabilities drive branches", "text model only writes prose", "interpreter owns variables/loops/budgets/replay", "otherwise maybe / confidence gate", "chaos samples after the gate", "southpolesteve/probably ≠ carldaws/hunch ≠ feelings ≠ Kungie/gut ≠ Illusion47586/judge ≠ tidymodels/probably", "133★ / forks 10 live", "build calibrated classifiers from human feedback", "retrieve by relevance not resemblance", "one calibrated yes/no per memory in one request", "pointer mode 17/18 19/20 *theirs*", "embedding resemblance misses the allergy", "samdotmak/jev-recall ≠ jev-search ≠ jev-sift ≠ carryforward ≠ chopratejas/invalidate", "memory leases ended by new evidence", "six Nouls then fixed rules in code", "0 of 157 false invalidations", "questions/plans/directives are not evidence", "unsure → review queue", "host keeps the store", "name↔body / comment truth / test-claims", "mizchi/jev-lint is mizchi/jevlint rename", "no shipped rule has severity error", "~1 in 5 findings wrong *theirs*", "mizchi/jev-lint ≠ huntedman/JevLint ≠ MichitoSugawara/jev-lint", "JSON Schema → typed JSON via Jev", "noul_threshold 0.5 decoder not a proof", "IncompatibleSchemaError lists every bad property", "on-device Laya CoreML ANE", "~5 ms P50 short decisions", "189/189 FP16 checkpoint parity", "10× not achieved", "mizorewww/laya-coreml ≠ gliner-native-runtime ≠ jevmlx ≠ NandhaKishorM/laya", "softmax over allowed tokens ≠ Noul", "question-first cache", "Micha0827/snapjudge ≠ githubnext/localjev ≠ jevmlx ≠ cendress/SnapJudge", "Jev-first Pi agent loop", "slow-LLM fallback", "explicit action menu / CandidateSource unimplemented", "62 tests wiring not quality", "direwolfiy/JevPi ≠ standardagents/jevpilot ≠ pi-jev-control", "resume-screening bias audit methodology", "name×resume factorial independent Nouls", "callback determined by resume quality", "mean-probability name gaps operationally negligible", "natemoo-re/bias-bench ≠ BBQ", "Plan/PRD panel → code-owned pass|review|block", "cheerleading out of scope", "austindixson/planalyzer ≠ single-goodness Noul", "cost-aware multi-model routing/escalation", "decide vs do", "successful-task cost", "cannacre8ive/switchboard-ai ≠ ha-switchboard ≠ hermes-switchyard", "frozen-protocol zero-shot bench", "TypeSafe Jev vs PrismNLI vs Laya", "contamination caveat", "elcronos/jev-vs-open-decision-models ≠ JevBench ≠ DMB", "context-window admission control", "VOI gate which tokens are worth the expensive model", "fail polarity per lens", "on small inputs lenses lose money", "cvsgireesh/jevusher ≠ jev-sift ≠ winnow", "typed decision control plane", "receipt ≠ authorization", "historical-v0 zero retained cases", "MokiMeow/jev-fabric ≠ jev-forge ≠ dgp", "live 15-dim typed rubric re-score per pause", "scoring economics exemplar", "OpenJev/Codiv ≠ TypeSafe hosted", "jose-troche/live-rubric ~$0.000004 desc / ~$0.000006 README", "adversarial pre-registered Jev eval", "28 predictions before data", "123,805 requests", "confidence does not track ignorance", "polite injection 65% / crude 0%", "willkelly/jev-evaluation ≠ jevals ≠ jev-baselines-eval", "provider-neutral Elixir/BEAM Noul/Choice/Score SDK", "class infrastructure", "nshkrdotcom/system_one_sdk ≠ typesafe_sdk ≠ dannote/jev", "question-linting of Jev questions themselves", "nine jaggedness rules, no API key, no labelled data", "static lint ≠ measured separation", "yodablocks/jevq ≠ tenbin ≠ JevLint ≠ commitjev", "open-weights Laya as class exemplar (binding)", "Nx/Bumblebee runtime", "host chooses backend", "ChristianAlexander/laya_ex ≠ system_one_sdk ≠ dannote/jev ≠ NandhaKishorM/laya", "on-chain/edge Laya deploy", "parity_verified stays false", "model output never grants Tx", "humandebri/IC-Laya ≠ laya_ex", "auditable weekend replica", "Jev outputs never used for training", "soft human-vote distributions", "unpaired 0.577 vs 0.727", "agilabs-ai/jev48 ≠ JevBench ≠ Mapika/decider", "adversarial dual-judge / framing attack surface", "comparative framing is the usable judgment", "prior injection crowds out evidence", "copyleftdev/ember ≠ ember.js", "Laya specialist fine-tune pipeline", "training still GPU-pending", "PIXELZX0/XERON ≠ convaiinnovations/laya", "Hub Laya replica drop", "daliborsb/laya ≠ convaiinnovations/laya ≠ NandhaKishorM/laya", "System One student distillation corpus", "gold is programmatic", "teacher is closed-API clone", "do not distill Jev as teacher of record", "MagaBitmex/jev-4b-distill-data ≠ missing student checkpoint", "non-LLM VIN System One", "planning depth not chat", "lewislululu/jevon ≠ douglance/jevon", "source-bound evidence checks", "local quote mismatch needs no API", "exit 0 ≠ claim truth", "WaynezProg/jev-kit ≠ jonathanavis96/jev-kit (Airlock) ≠ jev-use ≠ jev-mcp", "independent System One evidence catalog", "scores not one leaderboard", "no external record currently reproduced", "TokenTrim no-Jev matched hybrid 62.4%", "reachjalil/system-one-bench ≠ mallahyari/system-one-benchmark", "21 tasks · 134 items · 208 questions", "scenes from public GitHub contracts, not production logs", "SivletLabs/jev-eval ≠ willkelly/jev-evaluation ≠ 4esv/jev-eval ≠ xxkuboxx/jev-eval ≠ onlyoneaman/jev-eval ≠ dayhaysoos/jevals", "option isolation (sibling-blind)", "permutation-equivariant", "Hub OWNER not published", "nafisazizir/hev ≠ jaredpalmer/kev", "frozen local LLM logits, no trained decision head", "residual-head 9,222-param decreased 73/96→67/96", "confidence = 1−normalized entropy, not P(correct)", "yuki-oshio/mini-jev ≠ r-ms/mini-jev", "Jev classifier as autoregressive next-token predictor", "ChatJev-style soundness theater", "erik-dunteman/ChatJev ≠ dannote/jev ≠ jev-gpt", "calibrated decision head × AlphaProof value head", "implementation-layer isomorphism, semantic difference", "timeout = censoring", "do not launder Noul as proof", "parallel rank-prediction vs serial selection", "independent questions can conflict", "zzzzzec/jevsort ≠ keltokhy/jsort", "curated open System One ecosystem catalog", "rupeshpoojary9/awesome-open-system-one ≠ AnotiaWang/awesome-jev", "arXiv paper radar with Jev relevance scoring", "ranking ≠ calibration / 0.5 still soft", "fail-open failed evals not marked seen", "train calibrated ~27M from scratch", "typed Q→prob dist / one forward pass / no LLM decode", "hyusi2003/MiniSystemOne ≠ Colvin0315/MiniSystemOne", "description-only stub / size 5", "ESCI hard probe fails four of six", "jev_bool ECE 0.242 inversion 0.255", "do not re-fold §60 six-gates as new", "jobbyjev one-request-per-company from batch-size result", "find/design/evaluate TypeSafe Jev decision loops", "karanb192/jev-architect ≠ samtay32/jev-system-architect", "Jairik/jev-distiller size 1", "distill-Jev UI stub / do not distill Jev as teacher of record", "post-launch scored use-case map / Jev self-scores then human curation", "licensedsaucer9-web/jev-opportunities", "Jev-inize a use case into classifier/router", "gavinHuang/jevinize → simple-jev not TypeSafe", "featherless-ai/simple-jev", "compare saved decisions / same label can still change the branch", "VihaanAgarwal/jev-diff ≠ Saik0s/diffusiongemma-jev-macos", "not tested with a live Jev API key", "constrained logprob + temp/Platt ≠ Noul", "OpenJevPro pastes openjev-sglang JevBench as own", "zhangcy122/OpenJevPro ≠ IamBusy/OpenJev ≠ ekzhang/openjev-sglang", "PolyForm Noncommercial", "SmolLM-135M / sub-70ms / 0 output tokens", "demo P(True) 0.5052 / Choice conf 0.2872 / Score conf 0.0055", "README claims MIT / GitHub license null / no LICENSE file", "patelvishwa112/jev-system-one-rlcd ≠ arnabgho/rlcd-lite ≠ blackwood-rlcd", "source-backed Awesome Jev radar / 306+ commit-pinned", "logicrw/awesome-jev-projects ≠ AnotiaWang/awesome-jev ≠ yibie/awesome-jev ≠ cobanov/awesome-jev ≠ rupeshpoojary9/awesome-open-system-one", "auto GitHub sync / Issue-only submissions", "hashed n-gram encoder / rival-aware attention", "olanotolu/jevbetter vs jevlike starter", "synthetic hard menus top-1 0.916 vs 0.873 / ECE 0.0182 vs 0.0367 / 40 vs 4608 menus/sec", "shuffled-context control 0.335", "Turn any open LLM into System-One Jev", "uspraveen/Jevify ≠ Mintzs/jevify ≠ gulagala001/jevify", "Jevify-any-LLM architecture probe", "description-only stub / size 0", "Train encoder-only calibrated decision models from a task sentence", "Exu is a toolkit, not a method", "strictly proper scoring rule", "Pre-alpha", "Ruivalim/exu-base", "scratch-trained calibrated decision model", "typed Q → probability dists", "Colvin0315/MiniSystemOne ≠ hyusi2003/MiniSystemOne", "no published weights download URL", "90.5 seconds / 29.2% pipeline evidence", "p_i/p_j independent of other candidates", "Recipe for calibrated decision models — small model out", "init → synth → train → eval → serve", "91.1 % / ECE 0.022 *theirs*", "Jev zero-shot 75.1", "scienthoon/luce", "Put Jev's three headline claims on trial", "0.5B local GPU", "46x speedup / accuracy identical", "ECE 0.624 sentiment catastrophe", "bigger model worse calibration", "RichardoMrMu/jev-mini ≠ yuki-oshio/mini-jev ≠ r-ms/mini-jev", "System-1 decision engine for local LLMs", "structured choices only", "JSON parse of generated text ≠ Noul", "TypefAI JEV / Journal Entry Voucher", "tapsin/jev-local ≠ us/jev-local ≠ Argos1111/jev_local", "Jev 1.13 reward-model eval across 8 benchmark tracks", "40,940 examples / 0 API errors", "RewardBench v1 92.58%", "Precise IF 50.63%", "goya4140/jev-reward-model-evaluation", "Scaffolding in progress", "Jev vs LLM support-ticket routing", "static + live decision bench", "TypeSafe's own published benchmark", "illustrative simulations, not live API calls", "JevBench v1 — smart/cheap/fast/reliable", "I/C/S/K 25% geometric mean", "classifier.dev fast tier 84.8 is Jev behind its own API", "do not re-fold §78 v1.2 board as new", "Laya (421M) 70.1 now on board", "Zero-shot/few-shot LLM routing", "hard budget filter before Jev", "Jev never asked to perform budget arithmetic", "Jev judges the next state, XState enforces transitions", "simulation uses synthetic keyword fixtures", "catalog gravity", "v-modal/awesome-jev-tools", "★339 live REST", "curation is not endorsement", "crawler-maintained directory", "Daily GitHub + npm sweep, human-merged", "RadRebelSam/awesome-jev ≠ AnotiaWang ≠ yibie ≠ cobanov ≠ logicrw ≠ v-modal", "HF peft SPLADE/BGE reranker", "rdxtremity/jev-reranking ≠ carlaiau/jev-reranking", "query-side encoders, not a Jev replica", "ONNX System One Qwen3.5-4B scorer", "source:pngwn/system-one-qwen3.5-4b-scorer", "CC-BY-NC-4.0", "temperature 1.75", "transformers.js AutoModel cannot load this graph", "Consistency benchmark Space", "This Space contains no benchmark result yet", "12-case plumbing fixture", "Benchmark-driven Jev router and judge", "cheap alone is not success", "Jev does not write, sum prices, or claim accuracy %", "Sol 94.2 / Luna 83.9 / Jev path 89.7", "19.2% Sol / 62.3% cost save / 4.5pp miss of 2pp non-inferiority", "p50 latency worse than Sol due to routing overhead", "erendikmenn/jev-llm-router-benchmark ≠ jev-rag-benchmark ≠ ryantsai/jev-llm-router", "Express + node:sqlite", "mock and Jev decision engines", "previous_ticket_count >= 3 is code", "MIN_CONFIDENCE 0.6 still soft", "substring false positives", "aesaganda/jev-ticket-router ≠ SarathChandraBellam/jev-vs-llm-ticket-router", "Universal Figure & Diagram Router", "confidence ≥ 0.85 hard-gate is theater", "generative AI banned from scientific plots", "six visual branches", "hoangngochuong24947-gif/jev-figure-router", "human-labeled (state, question, label)", "166,054 rows / 22 configs", "soft_label for human uncertainty", "Praveenrajus/jev-bench ≠ fstandhartinger/jevbench", "ternary bonsai System One GGUF", "openjev's mechanism, Bonsai's weights", "Hub does not ship weights", "100/100 easy T/F is not Harbor", "label_mass ≠ correctness", "stock llama.cpp Q2_0 silently gibberish", "NicolaiMTLassen/open-bonzi-jev ≠ NicolaiLassen", "transformers.js DeBERTa ONNX", "source:com-kotobalabs/open-jev-deberta-v3-large", "temperature 1.05", "AutoModel from_pretrained works", "onnx-community/open-jev-deberta-v3-large-ONNX ≠ system-one-qwen3.5-4b-scorer-ONNX", "107★ densify", "GH 151M vs README 149.6M", "PR #1 now closed unmerged", "do not re-fold §71 claim-audit as a beat", "typed decisions, RLCD, confidence-gated routing", "structured ≠ correct", "mock not live API", "26 tests", "wjdjdakf17/jev-study ≠ baekenough/jev-study", "bonzi-27b-v2 / ternary-8b / 27b-v1 GGUF family densify", "WANLI-256 74.6% / 65.2% / 71.1% *theirs*", "Bonsai 1 27B Q1_0 runs on stock llama.cpp", "ternary still needs PrismML fork", "hf:heman10x/openJev-verdict-2.0 twin tokenizer-only", "OpenJev Vision image classification + uncertainty", "CLEVR-4 held-out joint 0%", "hfdataset:IamBusy/OpenJev-Vision-Research-v0.1 12,832", "294,912 derived targets not independent samples", "Laya multilingual ONNX WebGPU typed-decisions port", "63/63 selected answers / 5.1e-4 CPU / 1.2e-2 WebGPU", "UpHash-Network/mini-jev is yuki-oshio transfer", "jev-injection-bench 11,900 labelled prompts", "Jev best ranking / Haiku better ECE 0.021 vs 0.058", "0.5–0.9 band is where Jev's numbers do not mean what they say", "Prompt wording moves panic 28%", "manojlds/jev-dspy-bench ≠ dspachos/jev-dspy ≠ jmanhype/jev-dspy-lab", "Jev agreement is similarity, never ground truth", "no aggregate quality grade or merge gate", "AbstentionBench-on-Jev rank 1 of 20 vs 2025 field", "question-asymmetry", "forward-looking 0.465 never extreme", "openkev calibration layer not a runtime", "ECE vs coverage independent", "select_threshold returns inf", "escalation catches uncertainty not ignorance", "misakaikato/openkev ≠ jaredpalmer/kev", "pdf-race Docling→Jev vs Gemini", "parser owns the wall clock", "12/12 tie is a tie", "titles selected not generated", "flopcheck 16 calibrated tweet judgments", "mechanical tells in code", "ZeroX-01/jev-atlas ≠ Zaious/jev-capability-atlas ≠ gorock007/jev-atlas", "Laya calibration lab Gradio MCP", "T never changes argmax", "confidence ≠ top-label p", "easy probe set refused", "40–48 rows too small to ship T", "Gemma-4 26B-A4B jevify classification+calibration", "LoRA adapter twin not independent eval", "Gemma-4 E4B jevify", "E4B LoRA stub card", "kushalpatil/jevify-gemma4 ≠ Mintzs/jevify ≠ gulagala001/jevify ≠ uspraveen/Jevify", "GH kushalpatil07/jevify 404", "PAWS 0.580/ece 0.288 is the weak cell", "smaller E4B slightly better OOD ECE than 26B-A4B", "Hub jevify merged LoRA ships weights", "bonzi Bonsai-8B v1 GGUF densify", "Bonsai-1.7B v1", "Bonsai-4B v1", "WANLI-256 64.5% / 60.2% / 52.0% *theirs*", "rank #4 / #5 / #6 of 6", "JulesHuisman/jev-eval scaffolding / README SHA c356a584 (was empty e69de29b)", "JulesHuisman/jev-eval ≠ SivletLabs/jev-eval ≠ willkelly/jev-evaluation ≠ 4esv ≠ xxkuboxx ≠ onlyoneaman ≠ dayhaysoos/jevals", "7 bands 6/10 vs 40 bands 0/10", "source receipts + confidence slider re-policy without re-inference", "32/32 synthetic is smoke not production", "classify HF datasets across typed semantic dimensions", "roadus2 watch misspelling; lock roadius2/ultra_laya", "ultra_laya REVIEW defects", "default branch claude/laya-jev-review-gg5ppo", "XNLI EN 88.3% ECE 0.032 → RU 77.3% ECE 0.096", "Δ −11.0 pp [−14.2,−7.8]; ECE +0.063", "MASSIVE no detectable difference at n=600", "confidence is function of p_max (r=1.000)", "pointer-not-generator 400 human-authored responses", "proposed ≠ authorized", "FewRel 160: Jev 85.0% vs lexical 13.125%", "gated 100% (95/95) coverage 59.375%", "J++ composable semantic computation language", "judge-jev 0.5 still soft", "947 repos scored; A 273 / B 302 / C 372", "LLM rubric ≠ benches", "No benchmark winner is claimed", "phishing: naive 62.6% vs regex 91.8%; 5-atomic + LR 95.0% *theirs*", "AITuber tension ±15", "README npm global; repo is Rust", "git-confess code owns counting/blame/ratio", "httpx exhibit 11% (13/119) *theirs*", "90d trend +12.40% vs random +12.75% vs BH +41.71%", "5m win rate 25%", "Awesomejev 656 entries / 38,160 stars", "tracker likes 64 (+4) lastModified UNCHANGED", "Laya present; Blackwood ABSENT; Archer still promised_not_landed", "Blackwood tracker ABSENT; likes 2 gated manual", "r = c - p_a", "ECE 0.021; acc 0.807 vs warmup 0.746", "Independent primitive", "11.57s vs 54.10s · 4.67× · 120/128 *theirs*", "default path is pretrained Gemma probs not trained RLCD head", "GH Meanblock 404; lock leesk212/JEV-CPU", "softmax over letter slots ≠ Noul", "WANLI 0.741 vs openjev v2 0.77 *theirs*", "3-way NLI ≠ Noul", "priority 0.464 = majority floor", "banking77 contaminated", "raw margins not probabilities", "do not distill Jev as teacher of record (they distilled Haiku)", "“0.9 is not one number”", "ranking ≠ calibration", "banking77 0.8–0.9 stated 0.86 actual 0.73 over-confident *theirs*", "≠ Praveenrajus/jev-bench ≠ fstandhartinger/jevbench", "$0.0000153–$0.0000226 vs circulating $0.0004 (~20×)", "Score is 0..n-1 expectation not 0–1", "Noul has no confidence field", "TCP floor 198.8 ms", "type reliability is not a reason to choose Jev (json_schema 5/5)", "gateway tax not one number", "Function-only 5/8 vs hybrid 8/8", "4/8 without Jev", "8 designed cases not conversion lift", "200-row pilot Jev 86.5% 173/200 vs Gemini Flash-Lite 86.0% 172/200 vs Pro 87.0% 174/200 *theirs*", "not a ranking", "情緒測謊器", "8-example Jev vs GPT-5.6 Sol ~64× cost 5.4× latency *theirs*", "synthetic; no inference", "≠ JevBench v1.2 §78", "Judged 3317 / listed 2560", "Jev judges, code applies policy", "APA “microsecond policy / zero hallucination” overclaim", "Client-side quiz; pointer from held docs; scanned-PDF warn", "Jev judges / agent reasons / user decides", "selecting an option is not permission to implement", "pattern exact, judgement must clear floor", "no matching pattern → no model call", "not a correctness oracle", "Spec vs artifact remainder", "treating 0.85 as 85% / minProbability hard-gate as Harbor", "VERIFY acquires discriminating evidence, never same-pool confidence-only rescoring", "fast/full/max are ceilings not sizes", "Solar writes, Jev chooses NEXT ACTION", "do not reopen or amend PR #23 or #24 or #25 or #26 or #27", , "Calibration is not alpha", "NO CURRENT ALPHA CANDIDATE", "ΔR² approximately +0.00084", "Brier 0.2131387", "ECE 0.0421875", "Adding Jev probability to deterministic volatility improved Brier by only 1.4058e-05", "default 0.5 keeps zero non pinned", "keepResult median 0.14 to 0.17", "keepCall median 0.28 to 0.35", "usable range is about 0.10 to 0.25", "7.8% to 57.9%", "judges results it never sees", "task-finish eval not built yet", "$0.002 per compaction", "slavadubrov/sgr-judge-bench ≠ slavadubrov/jev-judge-bench", "Jev 108/120 $0.083 0.34 s", "Luna SGR 114/120", "paired Jev accuracy-difference intervals include zero", "not evidence of equivalence", "GLM SGR 26/120 93 format failures", "Terra-planned Jev hybrid 55/120", "rule-based by default, optionally Jev-backed", "empty README", "missing key cannot break the experience", "prefill plus exactly one decode", "softmax over A/B/C ≠ Noul", "BBQ 9,053/10,000 (90.53%)", "ECE 0.0890", "Mean confidence 0.9943", "overconfident", "score and noul not implemented", "DGUI 12 rows (was 6)", "INSTRUCT 119 rows likes 2", "encode the state once, decide everything in parallel", "0.740 accuracy against a 0.508 majority", "ECE 0.047", "fine-tune's advantage ends where its 384-token training data does", "jasonkneen/open-jev ≠ pngwn/open-jev", "same sha d41dc3cd", "Space does not call Jev", "recomputes routing from saved probabilities", "200-case Jev 97.0% / 100.0% / 95.0% / MAE 9.22", "synthetic repository benchmark", "Jev evaluations are advisory", "YehuiTang0316/jev-nlgrep ≠ Bentlybro/jevgrep ≠ can1357/jegrep ≠ uehaj/jev-semgrep", "default threshold 0.8 still soft", "40-line windows cannot prove whole function", "token-native sequential start/end Choice", "Gemini/Haiku stubs not configured yet", "handful of hand-written examples, not a benchmark", "Jev judged exactly what it was given", "laguagu/jev-skills ≠ laguagu/jev-evidence-lab ≠ Pleo2/awesome-jev-agent-skills", "contract_passed is not a claim of guaranteed factual truth", "Wilson lower bound 0.85 floor", "fixture mode no savings claim", "SemIf 2207★ (+21 vs §110 2186)", "jevlike 1043★ (+5 vs 1038)", "TypeAR 15★ (+1 vs 14)", "AnotiaWang 97★ (+1 vs 96)", "yibie/awesome-jev 506★ (+16 vs 490)", "Laya likes 822 (was 802)", "tracker likes 64 flat, lastModified UNCHANGED", "do not reopen or amend PR #23/#24/#25/#26/#27/#28", "Heman10x-NGU/Verdict-open-jev ≠ Heman10x-NGU/openJev-verdict-2.0", "TF-IDF + LogReg ECE 0.0207 vs Jev 0.1440", "Verdict-open-jev 48.07% vs Jev 90.80%", "abstention combined recall 10.00%", "p50 35.58 ms", "K=25 (maximum capacity) 72.00%", "0.85 coverage 84.60% selective risk 1.18%", "26.1× faster than standard Qwen JSON generation", "Jevify 90.0% / 167 ms CUDA graphs disabled", "Finding 1: Brier on stated confidence alone is a trap", "grpo_rlcr 0.78 / ECE 0.084", "reliability 0.007 but resolution 0.000", "27 900 schema-driven decisions", "13 600 / 13 600 questions", "candidate mass min 0.99999624", "22 configs · 166,054 rows · 4 calibration-gold", "sha a39eba3f", "Student B MAE 0.148 / Pearson 0.836 / 86.0%", "pngwn/open-jev-laya-bench README 404", "sha 9f69c742 likes 2", "HDFS 0.9933 (745/750) / retain 0.0084", "BGL ERROR/FATAL protection 1.0000", "2,479 / 2,500 HDFS uncertain", "cache hit 0.9648 (2412/2500)", "$0.153936 estimated", "E2 recomputes from saved probabilities", "Space sha eda59e0a", "MASSIVE English 0.783 / Khmer 0.033 / Hindi 0.133", "40–48 rows too small to ship T", "T never changes argmax", "siren2345/jev-single-decode-transformers ≠ siren2345/jev-single-decode", "Split Transformers experiment from llama.cpp runtime", "tanayvasishtha/jev-lab ≠ dairui1/jev-lab ≠ BrendanH18/jev-lab ≠ yibie/laya-jev-lab", "Four experiments stress-testing TypeSafe's Jev: calibration, bundle bias, label bias, and ensembling", "second pass must be $0.00 from cache", "The pages never call Jev", "Gemma 4 31B 77.0% / Jev 1.13.0 61.4% / Laya 322M 0.0%", "restriction state 95.0% against 84.4%", "None of the systems are particularly good at knowing when to stop and ask", "They skip the question and call a tool directly", "100% schema pass", "six-field joint 48.8% vs 72.8%", "ywchiu/jev_benchmark ≠ Running-Dolphins/jev-bench ≠ Praveenrajus/jev-bench", "ACT / REVIEW / FALLBACK", "A provider failure, timeout, malformed output, or missing answer is **not** a policy outcome", "confidence is descriptive provider output, not a substitute for probability", "Quality denominators include only valid scored answers", "an exact halfway tie chooses the lower level", "aiwithenoch/Jev-Skill ≠ simplosophy/jev-skill ≠ laguagu/jev-skills", "The local path does not claim to turn a smaller checkpoint into Jev", "Low support becomes decision: \"review\"", "MIT-0 SPDX NOASSERTION", "current-llm", "结构兼容,不是 Jev 模型能力", "altryne/jevify ≠ Mintzs/jevify ≠ gulagala001/jevify ≠ uspraveen/Jevify", "Find where Jev belongs. Design the questions. Measure the difference", "TypeAR-AI/TypeAR 301 → TypeLLM/TypeLLM", "TypeLLM/TypeLLM 16★", "SemIf 2241★ (+34 vs §111 2207)", "jevlike 1051★ (+8 vs 1043)", "AnotiaWang 98★ (+1 vs 97)", "yibie/awesome-jev 525★ (+19 vs 506)", "Laya likes 864 (was 822)", "tracker likes 67 (+3 vs 64)", "lastModified UNCHANGED `2026-09-20T04:29:16.000Z`", "do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#32", "hysteresis enter/exit / replay policy without inference", "calibration does not compose / hop-ECE permutation-invariant", "equal-width vs quantile ECE / ranking ≠ calibration", "Qwen2.5 ≠ Archer / Qwen 3.8 sparring ≠ Archer / Qwen/Qwen3.8-27B ≠ Archer", "Deferred Crispification / TCE / AMS", "g0runmezadam/what-is-jev IS tunahansahin897/what-is-jev", "pd.cut equal-width vs jeval quantile", "A hunch is a probability with a policy attached", "soundness theater / measurement theater / hourly 0843", , "Jev Capability Resolver / NiazMorshed2007/jcr", "one tool nested capability tree / returns context / does not execute", "skills vs capabilities / workflow+judgment vs operations", "format independent of Jev / proposed open standard", "JCR_BAND_RATIO 0.6 is application policy / soft scores ≠ hard gates", "routing ≠ permission / docs ≠ authority to run", "sol-vs-opus5-20 lookup+explain / n=1 / Not Harbor task-execution", "wall-time mixed / Sol slower with JCR in 19/20", "NiazMorshed2007/jcr ≠ skill-broker ≠ skillranker ≠ jev-sift ≠ jev-lens ≠ jevusher ≠ jev_select_capability", "do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34", "notes.md §116", "copy the SemIf/MLX installer?", "quote 5.21× as beating Jev?", "treat 0.845 as a TypeSafe replica?", "collapse SemIf into kw2828/zhihz/semif-rs/semif-serve", "softmax over options as a Noul", "llm prompt to jev primitives", "conversion assistant not equivalent behavior", "heuristic conversion ≠ calibrated Noul", "alexwestco/llm-to-jev ≠ altryne/jevify", "user-provided 0940 / notes.md §118", "judge ≠ actuator", "candidate_mass", "softmax over A–H ≠ Noul", "hourly 0947 / notes.md §119", "ggmlc GGUF is not llama.cpp", "serving substrate ≠ calibrated replica", "Qwen3.5-9B ≠ Archer", "planner writes JEV selects", "hourly 1049 / notes.md §120", "open recreation ≠ calibrated replica", "semantic lint is a sensor not a proof", "cutoff 0.8 still soft", "paired bootstrap CIs *theirs*", "Same accuracy, 35x faster *theirs*", "hourly 1143 / notes.md §121", "revisit HIGH / since-last-look", "catalogued repo changed", "star-noise vs material change", "densify prior notes without inventing equivalence", "decide is not generate", "tryDecide returns typed calibrated judgments not a token stream", "GLiNER/GLiClass ports are class members not Jev replicas", "93.5% *theirs* not Harbor", "74.9 *theirs* not Harbor", "8.7x *theirs* not Harbor", "Option-Marker joint attention", "openjev:0.2.1", "thinking=True/False per-field budget", "PLAN_Qwen35", "hyperspaceai/jevcache ≠ kushals256/jevcache", "wire-compat ≠ logit-equiv", "SHA move is not a replica", "hourly 1248 / notes.md §123", "typesafe-sdk 0.7 Pydantic response models", "msgspec dropped", "The server's output is unchanged and was never wrong", "SchemaError is 400 plain-string detail not 422 list", "Pydantic response models ≠ logit-equiv", "msgspec dropped is not a replica", "Error contract is not a Noul", "coverage-at-error-budget *theirs* not Harbor", "PLAN_Qwen35 still proposal for review", "GLiNER locate ports are class members not Jev replicas", "Locate ≠ decide", "~160 ms *theirs* not Harbor", "0.971 F1 *theirs* not Harbor", "hf:fr0stbit3/laya-gguf serving substrate ≠ calibrated replica", "jkcdarunday/SystemOne-Next ≠ TypeSafe System One", "hourly 1340 / notes.md §124", "vLLM NVIDIA + MLX Apple Silicon", "Codiv hosted free endpoint", "dual /v1/systemone + /v1/chat/completions", "chat 501 on MLX", "dual serving is not generate", "Hosted Codiv ≠ TypeSafe", "hr98w/jev-visual 167★ Apple Silicon visual candidate scoring", "37.30s → 2.40s at 64 decisions *theirs*", "Breakout 9 bricks 6 returns 2 lives *theirs*", "candidate probabilities are relative not correctness", "jkudish/jev-mcp 156★ ten MCP tools", "recommendation is advisory", "the server never blocks on its own", "TypeSafe CLERC 5% to 18% *theirs*", "jkudish/jev-mcp ≠ burnigtm/jev-mcp", "zhengxuyu/litjev off-the-shelf Qwen decision layer", "Probabilities are not calibrated by default", "Qwen/Qwen3.8-27B ≠ Archer", "zhengxuyu/litjev ≠ alexwestco/llm-to-jev", "Zefan-Cai/Open-Jev LoRA + scalar head", "2B 94.71% 9B 97.54% hard test *theirs*", "2B OOD 86.02% 9B OOD 91.97% *theirs*", "80,816 training rows", "27B still in progress", "LoRA ≠ RLCD replica", "Zefan-Cai/Open-Jev ≠ TheoLeeCJ/openjev ≠ razorback16/openjev", "cristianoliveira/jeq intelligence you can pipe", "pass-min 0.8 still soft", "JEQ does not own actions", "AndyInQtr/laya-coreai CoreML serving substrate ≠ calibrated replica", "AndyInQtr/laya-coreai ≠ mizorewww/laya-coreml", "hourly 1441 / notes.md §125", or "cascade sign-flip / calibration theater": read `references/faq.md`, + other", "training confronts Choice other / none-of-the-above", "soft AGENTS.md rules vs the linter", "screenshot Choice / omni System One", "extractive quotes / pointer not generator", "compaction summarize vs pointer", "encoder vs Jev compaction backend", "shadow-mode compaction rollout", "CI flaky-vs-real merge gate", "fail-open VOI wake/resume", "claim vs session evidence", "S1 indexer escalate-S2", "Harbor on/off routing", "fail-open vs fail-closed wake vs CI gate", "encoder vs Jev computer-use backend", "hybrid local decide + remote fill", "DONE vs verified success", "stdout prune vs session compaction", "OpenCode jev-pruner vs Claude jev-pruner", "zen-chat vs jev-zen Noul", "hard envelope then Noul prune", "Cua-S1 vs TypeSafe Jev", "plan vs execute dry-run", "specialist computer-use vs general agent", "local drop-in vs stub scorer", "route vs memory", "when does it hold / extractable from state", "decision model vs constrained LLM", "dual-process S1/S2", "combinatorial grid vs extractive", "uncalibrated local likelihoods", "decision-native RAG", "classify-first / read selectively", "living applied-mappings atlas / class patterns", "silence as safer / draft-gate heartbeat", "robotics text-state vs pixels", "verbatim ledger vs summary", "judgment as language primitive", "Stagehand extract pick-and-copy", "harness observe-score-act vs demo loop", "public judgment wall / six parallel questions", "meaning-search without embeddings", "attention ≠ correctness", "skills→oxlint / AST prove ∩ remainder", "session-sticky first-prompt routing", "measured RAG rerank vs generative rerank", "capability kernel / secrets never in the agent", "Jev is SENSOR not policy", "type-safe ≠ correct", "typed control plane around DSPy", "native vs verbalized confidence", "engine owns truth / Jev owns judgment", "human-confirmed kill gate", "train specialist vs few-shot hosted", "decide→policy→LLM leftover", "Noul 0.5 cannot-tell never rounded", "calibration ≠ sortable / ORDER BY", "pairwise inversion / Score ordinality / two-decimal ties", "wire-compat GLiFormer /v1/systemone", "class-backend economics", "loopback gateway hosted + local", "do not distill Jev as teacher", "active-learning triage", "evidence-packet explorer", "meaning-grep AND/OR/NOT", "closed-vote-only / no planner LLM", "Jev vs PCD Harbor", "PCD O(1) ≠ Noul", "host-owned handlers × System One", "OMP/pi fail-open gate", "permission vs probability / operator owns thresholds", "judgment ≠ permission / Jev never grants access", "eval integrity / instrument not score", "constrained optimizer + S1 features / never sole hot-path gate", "privilege ≠ verdict / effect contracts not tokens", "attention filter / VOI for human review / never blocks / never green unless sure", "measurement owns endorsement / evidence-gated question packs", "Jev supplies evidence / code owns authority", "ranking ≠ calibration / never hard-threshold raw p as frequency", "hot-click CU / indexed element table", "Jev judges relevance / code decides structure", "local rules first then remainder / never auto-train on own hides", "combinators / System One as control plane", "receipts not leaderboard / type-safe ≠ correct jaggedness", "VOI over skill library / skillranker abstention", "OOD calibration / AUC ≠ ECE", "Jev vs thinking-budget small models", "turnstile / replayable evidence≠authority", "MLX one-pass schema→JSON / Apple Silicon replica economics", "memory leases ended by new evidence", "never confidently wrong / TLA+ compose / escalate instead of hard-gate", "no seal no advance / coverage ledger / mint ≠ product brain", "skill-broker sibling / judgment ≠ permission", "sureness bands / max_prob is generous", "JevBench / calibration not in Main Score", "CI typed gate before expensive review", "Codex MCP host adapter", "judgment as attention redirect / jev-preflight", "compress-before-first-send / dizk jev-lens", "tools≠use / SessionStart over hoping", "observational memory / pi-om keep-kind", "open-Jev class / openvons / JevPick", "physical-world System One / HA-Jev / not for locks", "judgment outside the store / jevql", "landed-script trust / headless≠auto-approve", "digital-design combinators / extended five", "VOI cache admission / same-intent skip LLM", "BM25 vs Jev skill routing Harbor harness", "zeroshot vs BERT / contamination DiD", "typed escalate continue abort baton / inverted loop", "worth-your-attention VOI / ThinkyMiner Winnow", "Jev WHETHER Python HOW LLM WHAT", "conflict vs ignorance / named Choice escape", "Playwright executes Jev chooses", "OpenJev /v1/decide not drop-in", "SemIf wire-compat runoff; SemIf rename densify / MLX backend / 5.21× systems≠semantic / Softmax ≠ Noul (`notes.md` §117)", "decision-as-memory flywheel", "record/replay CI / jevassert", "failure-finding arena / jevarena ≠ jev-arena", "BBQ not a bias cert", "decider≠executor", "sentence-as-rule lint / jevlint", "sentence-as-rule lint / jev-lint is jevlint rename", "VOI hunk prune", "whole-repo intent VERIFIED/VIOLATION/UNKNOWN", "GLiNER2 spec ≠ replica", "open replica substrates / grande / laya-jolt / JEV-CPU", "ONNX local-jev not equivalent", "persist constraints across compaction / pi-heed", "calibration+cost first-class gates", "Harbor-shaped Jev vs SGR LLM-as-judge / jev-judge-bench ≠ jevarena ≠ jevbench", "hand no-text steps / jev-use / Vercel drops confidence", "Pi System-One control plane / pi-jev-control", "never free-generates / jev-gpt tree of Choices", "OpenRouter recipe atlas / samples not benches", "personal history feed / jevfeed / no social graph", "competing NAR claims / dual-channel ECE / openJev-verdict ≠ OpenJev", "empty compaction-proxy skip / IPECTER", "throughput ≠ latency / like-for-like ECE", "1-token logprob endpoint ≠ Noul / coverage ≠ correctness", "open replica engine / jevinf", "unofficial Elixir SDK ≠ OTP peer", "jevex n=16 files-to-read VOI", "commit pre-review attention≠verdict / middle band", "Hermes plugin is Agnes not TypeSafe", "pi-jev-compact ≠ pi-jev-compaction", "empty Codex-proxy skip / IPECTER runway", "decision-native inbox / mailordinal", "unofficial jev-cli not ready / ≠ jevql", "laya-multilingual / English checkpoint confident-wrong OOD", "schema-scorer peaked ranking ≠ calibration", "HF 401 / GitHub 404 Hub-only", "productized System One HTTP / classifier.dev", "escalate-under-threshold / smart tier / multi-label ignores", "silent FALLBACK / granite 0.546 vs advertised 0.800", "vs_jev tracked JSON / read eval/README", "choxos/jev-reviewer ≠ egma-ai / systematic-review pointer", "two-pass Choice+Noul evidence extraction", "not-found is an answer", "human check as productized judgment", "githubnext/localjev ≠ kunchenguid/local-jev", "wire-compat ≠ logit-equiv / prompted JSON ≠ structured read", "self-reported probs / entropy confidence", "GitHub Next local /v1/systemone", "LM Studio runner gap / structured-read primitives", "NandhaKishorM/laya packaging ≠ Hub-only / Router script-before-p", "post-T ECE ≠ raw ECE / Banking77 token-budget", "0.85 still soft / not TypeSafe drop-in", "external census ≠ scored bake-off", "GLiNER2+routers class-boundary", "incomplete openjev census vs watch", "Harbor honesty watch / silent fallback", "JevBench v1.2 geometric mean / cal ON rank / weight sensitivity", "option-order 72→21 / instruction models in the class table", "self-host latency ×2 assumption / est. costs", "Laya absent is a gap not a named exclusion", "Qwen3.8 27B ≠ Archer", "hourly already-folded watch / apply-the-five / skip thin noise", "hard-gate Noul as PR/quality gate is soundness theater", "S1 never stalls waiting / S2 one-use advisory", "Local controller ≠ githubnext/localjev", "purple telemetry = consumed not arrived", "seed = geometry not async replay", "20% starting gate still soft", "no pixels to either provider", "OCR+AX observe-score-act / typesafe-computer-use", "never send screenshot to frontier for the decision", "overlapping CU options = false low confidence", "split kind/item/site", "155× one-screenshot ≠ Harbor taskset", "decision ≠ answer-reader capture", "ASR observe-score-act / jev-voice-browser", "partial-speech VOI / free-text waits", "spoken confirm ≠ hard auth", "numbered overlay without another model", "wrap-as-execution / AgentGhost ALLOW ASK DENY", "rules first then Jev remainder / ASK throws / fail-closed", "reddpy/AgentGhost ≠ jwen5419807/agentghost ≠ vventirozos", "JP genre atlas / studio_yebisu / stars ephemeral ≠ eval", "Jev Clearly Explained / akshay_pachaar / LLM hammer", "schema-safe ≠ correct / 200× 400× TypeSafe ceiling", "questions-as-code / shadow first / not a TypeSafe how-to", "proposition ≠ embedding / contrast-set", "boolean composition of soft Nouls / AND OR NOT", "uehaj/jev-semgrep ≠ semgrep.dev", "meaning-grep dedicated fold / not a gate", "decision-validated UI / Jev never authors text", "decision-as-assert / jevtest ambiguous band", "typed decisions drive UI / jev2ui", "hybrid S1 closed verb menu / anima3", "pointer-not-generator search / JevFind", "jev-frontier-bench ≠ frontier-100", "product bakeoff ≠ architecture duel / GLiClass", "four engines same questions / majority floor", "authorship named escape / not evidence", "ha-switchboard HA remains execution", "n8n classify/route/score / Low Confidence", "fast-jev-compaction-pi ≠ pi-jev-compact ≠ pi-jev-compaction", "jevloop full-distribution optimizer / no LLM in the loop", "laya-vision SmolVLM / score untrained", "Cerebellum-2B /v1/decide ≠ TypeSafe / wire-compat vs agent-routing", "laya-grounded not drop-in / Platt not temperature", "GestaltLabs/Jeff-1 ≠ logan-markewich/jeff / acc vs ECE n=9730", "stanley-code empty findings ≠ approval / human promote", "findme ≠ JevFind / NL memory beam-search FS", "jevsubrouter price workers not conversation / counts ≠ dollars", "feelings .feels() default 0.5 is Noul-0.5-never-rounded / ≠ hunch ≠ Probably", "apa-agent-harness ≠ AntonioCoppe/jev-harness / unpublished npm", "grok-bot-jev skill cannot force a bot that ignores it / A/B proxies not tokens", "Essentiel-Jev never authority / human every action", "enzo-mcp independently falsifiable claims / ≠ jev-sift", "pigeonhole OTHER skip / decision-as-filing", "jev-reliability Nothing about accuracy", "clduab11/jev-test ≠ realZachi/jevtest / Nothing runs yet", "jev-rag-benchmark Jev wins is not an assumption", "dairui1/jev-lab ≠ BrendanH18/jev-lab", "jevmail gmail.readonly / mailjay archive/trash", "ZHUBoer/ego-jev reserved __none__", "runWorkflow completed ≠ success", "jsort scores are relative", "Noul not Choice for scale", "groundedness-judge-bench native vs schema-guided", "implicit_true included in yes", "jev_playground 0 promotions", "routing-backtest 0.0447%", "yuyang2230/jev-agent-skill jev-1.13-free", "jev-techstack-classifier stack_config.json", "s1_ruby collapse late", "undecided? abstain", "2389-research/judgement license null", "confidence ≠ winner p", "typesafeai-sdk-community not a new species", "tpellet/hunch exit 3", "never-execute list", "jev-file-search scores not calibrated accuracy", "jev-linkmap Jev never sees S2 prose", "muhammedilyasy/jev-mail metadata only", "tidy none-of-folders stay", "tab-bouncer pinned/audio/current never closed", "lkclean Show fail-open", "jev-yt-time-saver Show anyway", "ORIGIN pause-if-no-Jev", "validResponse sums-to-1", "jev-crawlers risk bands never raw boolean", "jevbrain AUTO_ACT is not a Noul", "judgekit YAML classify/score/route/verify", "typed-judge-kit verdict-in-code", "alsoleg89/decide packing VOI", "0.8 ≠ 80% accuracy", "Jev-Calibration Platt ECE 0.117→0.052", "jev-calibration-arena never acts", "ctmx/openrouter-jev-mcp Decision-as-Plugin", "FrancoisChastel/jev-code ≠ npm jev-code", "claudecode-jev-marketplace fail-open not hot path", "pedroknigge/mcp_jev packs not ask_jev", "cyrusasco/typesafe-mcp noul deadband 0.35–0.65", "codaaiteam/jev-skill jevtypesafeai.com ≠ TypeSafe", "hermes-switchyard ≠ hermes-jev-router ≠ hermes-plugin-jev", "nanoprune 2.8MB ECE 2.58%", "smartdio/jev-browser-agent ≠ ZHUBoer/ego-jev", "Dakai/omp-jev-web DONE ≠ proof", "hari007sh/jev ≠ dannote/jev", "0thernet/system-one-skills deterministic verify", "typed-gate band [0.40,0.60] is refusal", "pi-jev-gate fail-closed; choice is the verdict", "Foq ~25ms/2.2GB local", "rev prefill-only + HF jev-0.5b", "robfrase/jev planning memo", "typesafe_agent_gates 27/27 / 31/31", "EpicEric/safe-sh static remainder", "pastepilot Confirm before act", "Jev-Reranker live Jev not yet measured", "sessionwise opt-in relevance", "jev-search pointer sieve", "400ms Salesforce WebMCP", "typesafe-scheduler-diagnostics advisory", "droidjev screenshot-free", "Tewoto1 jevcu planner still writes", "ha-conversation-jev Jev→Grok", "dsh-jev can only gate", "jev-classification-benchmark specified not run", "jev-luna-pagerduty p≥0.50", "meldltd/meldecision laya-go ONNX", "laya-doom never pixels", "logixism/laya-api empty README", "akpsahan/laya ≠ Archer", "choxos/jevchess engine owns truth", "jev-drive sim not AV", "story-arc Jev never authors", "jev-hs-assistant HS6", "golergka/jev-plays-starcraft-2 UI-verified ≠ API Victory", "awesome-jev-use-cases catalog", "Nibir1/typesafe-go ≠ official", "fingerprint after redact", "recall vs decide", "publish fingerprints+answers", "CI replay as Harbor cousin", "Cache hit ≠ correctness", "hyperspaceai/jevcache ≠ kushals256/jevcache", "human labels only", "score never auto-accepts", "production capture flywheel", "sutro-sh/jev-align ≠ caiovicentino/jev-align", "guidance ≠ hook", "catalysts ≠ summaries", "compile-time System One", "unofficial ≠ TypeSafe", "format_version modernbert-jev/1", "Argos1111/jev_local ≠ us/jev-local ≠ kunchenguid/local-jev", "LFM default ≠ ModernBERT backend", "Nemotron ≠ TypeSafe Jev", "not a calibrated replacement", "djev-dev complements djev-spark", "images as Choice options", "Laya essay numbers *theirs*", "Router/OOD confidence", "hosted bootstrap ≠ silent TypeSafe", "difficulty + policy thresholds + JSONL trace", "jev-codex-pilot model + reasoning depth", "keep/shadow/hybrid/reject", "quarry evidence projection", "Frank-ZY-Dou/awesome-jev robotics/3D/control", "one-dollar-tahoe TypeSafe Jev defense eval", "jevguard calibrator/cache/escape", "jev-ci-selector CI shadow mode", "llama-jev llama.cpp replica", "petercr/jev-orchestrator ≠ FleeexCorp/jev-orchestrator", "seb4ez/jevguard ≠ AseemPrasad/JevGuard ≠ pablozr/JevGuard", "webNeat/llama-jev ≠ WiktorB2004/llama-index-jev", "OpenCode jev-pruner context sieve", "observe→score-candidates→prune", "jev-zen / jev-1.13-free", "zen-chat ≠ Noul", "fail-open original", "keepScore >0.1 floor", "host port of tamaratran/jev-pruner", "indiejoseph/opencode-jev-pruner ≠ nrdz-labs/fast-jev-opencode", "jev-webagent-bench empty stub", "Kiln-AI/jev_jsonschema noul_threshold 0.5", "NSStudent/JevSwiftSDK unofficial", "GLiNER2 native Apple path", "unofficial Swift/Core ML GLiNER 2.5-small", "entity spans + confidence", "not Choice/Score/Noul", "not TypeSafe", "label descriptions as schema", "on-device ANE economics", "honesty locks", "shershah1024/gliner-native-runtime ≠ Fastino", "≠ gliner25-compaction ≠ gliner2-ultrafast ≠ Eran-BA/Jev_from_GLiNER2 ≠ NSStudent/JevSwiftSDK ≠ jevmlx", "default threshold 0.1 still soft", "soft Noul ≠ hard safety", "Decision Graph Protocol frame→assess→commit", "app retains permissions/effects", "Jev-first assessor-neutral", "guarded commit / receipt/next frame", "assessment batching", "hard-gating DGP as safety theater", "numerous-com/dgp ≠ TypeSafe official", "jegrep calibrated path+range Nouls", "no embeddings/index/daemon", "~$0.01–0.03 typical", "agent --json", "can1357/jegrep ≠ Bentlybro/jevgrep ≠ uehaj/jev-semgrep", "Archer-arch fidelity", "kev family OOD 0.76–0.77 vs Jev 0.86", "block-causal isolation", "pointer/readout CE-trained", "/v1/systemone drop-in", "replica honesty", "cost-sensitive decision theory × System One probabilities → control flow", "thresholds derived from costs not hard-coded", "YES / NO / UNSURE from cost_false_yes / cost_false_no / cost_human", "auto-batching same-object questions", "Kungie/gut ≠ tpellet/hunch ≠ carldaws/hunch", "judgment vs generation", "deterministic execution after probabilistic judgment", "exactly one app-owned callback", "explicit uncertain branch", "Illusion47586/judge ≠ lexingtonhibiki/judgekit ≠ Ascurse/typed-judge-kit", "variable-N option scoring as the trainable object", "dynamic candidate bags not fixed label sets", "zwliJay/jev-forge ≠ NanoJev", "open replica economics / latency vs closed Jev", "NAR local drop-in", "wfzyx/von late-catch HIGH", "competing NAR claims / replica honesty", "typed judgments vs chat judges on guardrailing", "ishaannk/llm-vs-jev cross-note only", "deeper integrity fold is rh-guard", "nothing wins outright", "can be argued out of guarding"", "Jev IS the if-statement", "judgments/probabilities drive branches", "text model only writes prose", "interpreter owns variables/loops/budgets/replay", "otherwise maybe / confidence gate", "chaos samples after the gate", "southpolesteve/probably ≠ carldaws/hunch ≠ feelings ≠ Kungie/gut ≠ Illusion47586/judge ≠ tidymodels/probably", "133★ / forks 10 live", "build calibrated classifiers from human feedback", "retrieve by relevance not resemblance", "one calibrated yes/no per memory in one request", "pointer mode 17/18 19/20 *theirs*", "embedding resemblance misses the allergy", "samdotmak/jev-recall ≠ jev-search ≠ jev-sift ≠ carryforward ≠ chopratejas/invalidate", "memory leases ended by new evidence", "six Nouls then fixed rules in code", "0 of 157 false invalidations", "questions/plans/directives are not evidence", "unsure → review queue", "host keeps the store", "name↔body / comment truth / test-claims", "mizchi/jev-lint is mizchi/jevlint rename", "no shipped rule has severity error", "~1 in 5 findings wrong *theirs*", "mizchi/jev-lint ≠ huntedman/JevLint ≠ MichitoSugawara/jev-lint", "JSON Schema → typed JSON via Jev", "noul_threshold 0.5 decoder not a proof", "IncompatibleSchemaError lists every bad property", "on-device Laya CoreML ANE", "~5 ms P50 short decisions", "189/189 FP16 checkpoint parity", "10× not achieved", "mizorewww/laya-coreml ≠ gliner-native-runtime ≠ jevmlx ≠ NandhaKishorM/laya", "softmax over allowed tokens ≠ Noul", "question-first cache", "Micha0827/snapjudge ≠ githubnext/localjev ≠ jevmlx ≠ cendress/SnapJudge", "Jev-first Pi agent loop", "slow-LLM fallback", "explicit action menu / CandidateSource unimplemented", "62 tests wiring not quality", "direwolfiy/JevPi ≠ standardagents/jevpilot ≠ pi-jev-control", "resume-screening bias audit methodology", "name×resume factorial independent Nouls", "callback determined by resume quality", "mean-probability name gaps operationally negligible", "natemoo-re/bias-bench ≠ BBQ", "Plan/PRD panel → code-owned pass|review|block", "cheerleading out of scope", "austindixson/planalyzer ≠ single-goodness Noul", "cost-aware multi-model routing/escalation", "decide vs do", "successful-task cost", "cannacre8ive/switchboard-ai ≠ ha-switchboard ≠ hermes-switchyard", "frozen-protocol zero-shot bench", "TypeSafe Jev vs PrismNLI vs Laya", "contamination caveat", "elcronos/jev-vs-open-decision-models ≠ JevBench ≠ DMB", "context-window admission control", "VOI gate which tokens are worth the expensive model", "fail polarity per lens", "on small inputs lenses lose money", "cvsgireesh/jevusher ≠ jev-sift ≠ winnow", "typed decision control plane", "receipt ≠ authorization", "historical-v0 zero retained cases", "MokiMeow/jev-fabric ≠ jev-forge ≠ dgp", "live 15-dim typed rubric re-score per pause", "scoring economics exemplar", "OpenJev/Codiv ≠ TypeSafe hosted", "jose-troche/live-rubric ~$0.000004 desc / ~$0.000006 README", "adversarial pre-registered Jev eval", "28 predictions before data", "123,805 requests", "confidence does not track ignorance", "polite injection 65% / crude 0%", "willkelly/jev-evaluation ≠ jevals ≠ jev-baselines-eval", "provider-neutral Elixir/BEAM Noul/Choice/Score SDK", "class infrastructure", "nshkrdotcom/system_one_sdk ≠ typesafe_sdk ≠ dannote/jev", "question-linting of Jev questions themselves", "nine jaggedness rules, no API key, no labelled data", "static lint ≠ measured separation", "yodablocks/jevq ≠ tenbin ≠ JevLint ≠ commitjev", "open-weights Laya as class exemplar (binding)", "Nx/Bumblebee runtime", "host chooses backend", "ChristianAlexander/laya_ex ≠ system_one_sdk ≠ dannote/jev ≠ NandhaKishorM/laya", "on-chain/edge Laya deploy", "parity_verified stays false", "model output never grants Tx", "humandebri/IC-Laya ≠ laya_ex", "auditable weekend replica", "Jev outputs never used for training", "soft human-vote distributions", "unpaired 0.577 vs 0.727", "agilabs-ai/jev48 ≠ JevBench ≠ Mapika/decider", "adversarial dual-judge / framing attack surface", "comparative framing is the usable judgment", "prior injection crowds out evidence", "copyleftdev/ember ≠ ember.js", "Laya specialist fine-tune pipeline", "training still GPU-pending", "PIXELZX0/XERON ≠ convaiinnovations/laya", "Hub Laya replica drop", "daliborsb/laya ≠ convaiinnovations/laya ≠ NandhaKishorM/laya", "System One student distillation corpus", "gold is programmatic", "teacher is closed-API clone", "do not distill Jev as teacher of record", "MagaBitmex/jev-4b-distill-data ≠ missing student checkpoint", "non-LLM VIN System One", "planning depth not chat", "lewislululu/jevon ≠ douglance/jevon", "source-bound evidence checks", "local quote mismatch needs no API", "exit 0 ≠ claim truth", "WaynezProg/jev-kit ≠ jonathanavis96/jev-kit (Airlock) ≠ jev-use ≠ jev-mcp", "independent System One evidence catalog", "scores not one leaderboard", "no external record currently reproduced", "TokenTrim no-Jev matched hybrid 62.4%", "reachjalil/system-one-bench ≠ mallahyari/system-one-benchmark", "21 tasks · 134 items · 208 questions", "scenes from public GitHub contracts, not production logs", "SivletLabs/jev-eval ≠ willkelly/jev-evaluation ≠ 4esv/jev-eval ≠ xxkuboxx/jev-eval ≠ onlyoneaman/jev-eval ≠ dayhaysoos/jevals", "option isolation (sibling-blind)", "permutation-equivariant", "Hub OWNER not published", "nafisazizir/hev ≠ jaredpalmer/kev", "frozen local LLM logits, no trained decision head", "residual-head 9,222-param decreased 73/96→67/96", "confidence = 1−normalized entropy, not P(correct)", "yuki-oshio/mini-jev ≠ r-ms/mini-jev", "Jev classifier as autoregressive next-token predictor", "ChatJev-style soundness theater", "erik-dunteman/ChatJev ≠ dannote/jev ≠ jev-gpt", "calibrated decision head × AlphaProof value head", "implementation-layer isomorphism, semantic difference", "timeout = censoring", "do not launder Noul as proof", "parallel rank-prediction vs serial selection", "independent questions can conflict", "zzzzzec/jevsort ≠ keltokhy/jsort", "curated open System One ecosystem catalog", "rupeshpoojary9/awesome-open-system-one ≠ AnotiaWang/awesome-jev", "arXiv paper radar with Jev relevance scoring", "ranking ≠ calibration / 0.5 still soft", "fail-open failed evals not marked seen", "train calibrated ~27M from scratch", "typed Q→prob dist / one forward pass / no LLM decode", "hyusi2003/MiniSystemOne ≠ Colvin0315/MiniSystemOne", "description-only stub / size 5", "ESCI hard probe fails four of six", "jev_bool ECE 0.242 inversion 0.255", "do not re-fold §60 six-gates as new", "jobbyjev one-request-per-company from batch-size result", "find/design/evaluate TypeSafe Jev decision loops", "karanb192/jev-architect ≠ samtay32/jev-system-architect", "Jairik/jev-distiller size 1", "distill-Jev UI stub / do not distill Jev as teacher of record", "post-launch scored use-case map / Jev self-scores then human curation", "licensedsaucer9-web/jev-opportunities", "Jev-inize a use case into classifier/router", "gavinHuang/jevinize → simple-jev not TypeSafe", "featherless-ai/simple-jev", "compare saved decisions / same label can still change the branch", "VihaanAgarwal/jev-diff ≠ Saik0s/diffusiongemma-jev-macos", "not tested with a live Jev API key", "constrained logprob + temp/Platt ≠ Noul", "OpenJevPro pastes openjev-sglang JevBench as own", "zhangcy122/OpenJevPro ≠ IamBusy/OpenJev ≠ ekzhang/openjev-sglang", "PolyForm Noncommercial", "SmolLM-135M / sub-70ms / 0 output tokens", "demo P(True) 0.5052 / Choice conf 0.2872 / Score conf 0.0055", "README claims MIT / GitHub license null / no LICENSE file", "patelvishwa112/jev-system-one-rlcd ≠ arnabgho/rlcd-lite ≠ blackwood-rlcd", "source-backed Awesome Jev radar / 306+ commit-pinned", "logicrw/awesome-jev-projects ≠ AnotiaWang/awesome-jev ≠ yibie/awesome-jev ≠ cobanov/awesome-jev ≠ rupeshpoojary9/awesome-open-system-one", "auto GitHub sync / Issue-only submissions", "hashed n-gram encoder / rival-aware attention", "olanotolu/jevbetter vs jevlike starter", "synthetic hard menus top-1 0.916 vs 0.873 / ECE 0.0182 vs 0.0367 / 40 vs 4608 menus/sec", "shuffled-context control 0.335", "Turn any open LLM into System-One Jev", "uspraveen/Jevify ≠ Mintzs/jevify ≠ gulagala001/jevify", "Jevify-any-LLM architecture probe", "description-only stub / size 0", "Train encoder-only calibrated decision models from a task sentence", "Exu is a toolkit, not a method", "strictly proper scoring rule", "Pre-alpha", "Ruivalim/exu-base", "scratch-trained calibrated decision model", "typed Q → probability dists", "Colvin0315/MiniSystemOne ≠ hyusi2003/MiniSystemOne", "no published weights download URL", "90.5 seconds / 29.2% pipeline evidence", "p_i/p_j independent of other candidates", "Recipe for calibrated decision models — small model out", "init → synth → train → eval → serve", "91.1 % / ECE 0.022 *theirs*", "Jev zero-shot 75.1", "scienthoon/luce", "Put Jev's three headline claims on trial", "0.5B local GPU", "46x speedup / accuracy identical", "ECE 0.624 sentiment catastrophe", "bigger model worse calibration", "RichardoMrMu/jev-mini ≠ yuki-oshio/mini-jev ≠ r-ms/mini-jev", "System-1 decision engine for local LLMs", "structured choices only", "JSON parse of generated text ≠ Noul", "TypefAI JEV / Journal Entry Voucher", "tapsin/jev-local ≠ us/jev-local ≠ Argos1111/jev_local", "Jev 1.13 reward-model eval across 8 benchmark tracks", "40,940 examples / 0 API errors", "RewardBench v1 92.58%", "Precise IF 50.63%", "goya4140/jev-reward-model-evaluation", "Scaffolding in progress", "Jev vs LLM support-ticket routing", "static + live decision bench", "TypeSafe's own published benchmark", "illustrative simulations, not live API calls", "JevBench v1 — smart/cheap/fast/reliable", "I/C/S/K 25% geometric mean", "classifier.dev fast tier 84.8 is Jev behind its own API", "do not re-fold §78 v1.2 board as new", "Laya (421M) 70.1 now on board", "Zero-shot/few-shot LLM routing", "hard budget filter before Jev", "Jev never asked to perform budget arithmetic", "Jev judges the next state, XState enforces transitions", "simulation uses synthetic keyword fixtures", "catalog gravity", "v-modal/awesome-jev-tools", "★339 live REST", "curation is not endorsement", "crawler-maintained directory", "Daily GitHub + npm sweep, human-merged", "RadRebelSam/awesome-jev ≠ AnotiaWang ≠ yibie ≠ cobanov ≠ logicrw ≠ v-modal", "HF peft SPLADE/BGE reranker", "rdxtremity/jev-reranking ≠ carlaiau/jev-reranking", "query-side encoders, not a Jev replica", "ONNX System One Qwen3.5-4B scorer", "source:pngwn/system-one-qwen3.5-4b-scorer", "CC-BY-NC-4.0", "temperature 1.75", "transformers.js AutoModel cannot load this graph", "Consistency benchmark Space", "This Space contains no benchmark result yet", "12-case plumbing fixture", "Benchmark-driven Jev router and judge", "cheap alone is not success", "Jev does not write, sum prices, or claim accuracy %", "Sol 94.2 / Luna 83.9 / Jev path 89.7", "19.2% Sol / 62.3% cost save / 4.5pp miss of 2pp non-inferiority", "p50 latency worse than Sol due to routing overhead", "erendikmenn/jev-llm-router-benchmark ≠ jev-rag-benchmark ≠ ryantsai/jev-llm-router", "Express + node:sqlite", "mock and Jev decision engines", "previous_ticket_count >= 3 is code", "MIN_CONFIDENCE 0.6 still soft", "substring false positives", "aesaganda/jev-ticket-router ≠ SarathChandraBellam/jev-vs-llm-ticket-router", "Universal Figure & Diagram Router", "confidence ≥ 0.85 hard-gate is theater", "generative AI banned from scientific plots", "six visual branches", "hoangngochuong24947-gif/jev-figure-router", "human-labeled (state, question, label)", "166,054 rows / 22 configs", "soft_label for human uncertainty", "Praveenrajus/jev-bench ≠ fstandhartinger/jevbench", "ternary bonsai System One GGUF", "openjev's mechanism, Bonsai's weights", "Hub does not ship weights", "100/100 easy T/F is not Harbor", "label_mass ≠ correctness", "stock llama.cpp Q2_0 silently gibberish", "NicolaiMTLassen/open-bonzi-jev ≠ NicolaiLassen", "transformers.js DeBERTa ONNX", "source:com-kotobalabs/open-jev-deberta-v3-large", "temperature 1.05", "AutoModel from_pretrained works", "onnx-community/open-jev-deberta-v3-large-ONNX ≠ system-one-qwen3.5-4b-scorer-ONNX", "107★ densify", "GH 151M vs README 149.6M", "PR #1 now closed unmerged", "do not re-fold §71 claim-audit as a beat", "typed decisions, RLCD, confidence-gated routing", "structured ≠ correct", "mock not live API", "26 tests", "wjdjdakf17/jev-study ≠ baekenough/jev-study", "bonzi-27b-v2 / ternary-8b / 27b-v1 GGUF family densify", "WANLI-256 74.6% / 65.2% / 71.1% *theirs*", "Bonsai 1 27B Q1_0 runs on stock llama.cpp", "ternary still needs PrismML fork", "hf:heman10x/openJev-verdict-2.0 twin tokenizer-only", "OpenJev Vision image classification + uncertainty", "CLEVR-4 held-out joint 0%", "hfdataset:IamBusy/OpenJev-Vision-Research-v0.1 12,832", "294,912 derived targets not independent samples", "Laya multilingual ONNX WebGPU typed-decisions port", "63/63 selected answers / 5.1e-4 CPU / 1.2e-2 WebGPU", "UpHash-Network/mini-jev is yuki-oshio transfer", "jev-injection-bench 11,900 labelled prompts", "Jev best ranking / Haiku better ECE 0.021 vs 0.058", "0.5–0.9 band is where Jev's numbers do not mean what they say", "Prompt wording moves panic 28%", "manojlds/jev-dspy-bench ≠ dspachos/jev-dspy ≠ jmanhype/jev-dspy-lab", "Jev agreement is similarity, never ground truth", "no aggregate quality grade or merge gate", "AbstentionBench-on-Jev rank 1 of 20 vs 2025 field", "question-asymmetry", "forward-looking 0.465 never extreme", "openkev calibration layer not a runtime", "ECE vs coverage independent", "select_threshold returns inf", "escalation catches uncertainty not ignorance", "misakaikato/openkev ≠ jaredpalmer/kev", "pdf-race Docling→Jev vs Gemini", "parser owns the wall clock", "12/12 tie is a tie", "titles selected not generated", "flopcheck 16 calibrated tweet judgments", "mechanical tells in code", "ZeroX-01/jev-atlas ≠ Zaious/jev-capability-atlas ≠ gorock007/jev-atlas", "Laya calibration lab Gradio MCP", "T never changes argmax", "confidence ≠ top-label p", "easy probe set refused", "40–48 rows too small to ship T", "Gemma-4 26B-A4B jevify classification+calibration", "LoRA adapter twin not independent eval", "Gemma-4 E4B jevify", "E4B LoRA stub card", "kushalpatil/jevify-gemma4 ≠ Mintzs/jevify ≠ gulagala001/jevify ≠ uspraveen/Jevify", "GH kushalpatil07/jevify 404", "PAWS 0.580/ece 0.288 is the weak cell", "smaller E4B slightly better OOD ECE than 26B-A4B", "Hub jevify merged LoRA ships weights", "bonzi Bonsai-8B v1 GGUF densify", "Bonsai-1.7B v1", "Bonsai-4B v1", "WANLI-256 64.5% / 60.2% / 52.0% *theirs*", "rank #4 / #5 / #6 of 6", "JulesHuisman/jev-eval scaffolding / README SHA c356a584 (was empty e69de29b)", "JulesHuisman/jev-eval ≠ SivletLabs/jev-eval ≠ willkelly/jev-evaluation ≠ 4esv ≠ xxkuboxx ≠ onlyoneaman ≠ dayhaysoos/jevals", "7 bands 6/10 vs 40 bands 0/10", "source receipts + confidence slider re-policy without re-inference", "32/32 synthetic is smoke not production", "classify HF datasets across typed semantic dimensions", "roadus2 watch misspelling; lock roadius2/ultra_laya", "ultra_laya REVIEW defects", "default branch claude/laya-jev-review-gg5ppo", "XNLI EN 88.3% ECE 0.032 → RU 77.3% ECE 0.096", "Δ −11.0 pp [−14.2,−7.8]; ECE +0.063", "MASSIVE no detectable difference at n=600", "confidence is function of p_max (r=1.000)", "pointer-not-generator 400 human-authored responses", "proposed ≠ authorized", "FewRel 160: Jev 85.0% vs lexical 13.125%", "gated 100% (95/95) coverage 59.375%", "J++ composable semantic computation language", "judge-jev 0.5 still soft", "947 repos scored; A 273 / B 302 / C 372", "LLM rubric ≠ benches", "No benchmark winner is claimed", "phishing: naive 62.6% vs regex 91.8%; 5-atomic + LR 95.0% *theirs*", "AITuber tension ±15", "README npm global; repo is Rust", "git-confess code owns counting/blame/ratio", "httpx exhibit 11% (13/119) *theirs*", "90d trend +12.40% vs random +12.75% vs BH +41.71%", "5m win rate 25%", "Awesomejev 656 entries / 38,160 stars", "tracker likes 64 (+4) lastModified UNCHANGED", "Laya present; Blackwood ABSENT; Archer still promised_not_landed", "Blackwood tracker ABSENT; likes 2 gated manual", "r = c - p_a", "ECE 0.021; acc 0.807 vs warmup 0.746", "Independent primitive", "11.57s vs 54.10s · 4.67× · 120/128 *theirs*", "default path is pretrained Gemma probs not trained RLCD head", "GH Meanblock 404; lock leesk212/JEV-CPU", "softmax over letter slots ≠ Noul", "WANLI 0.741 vs openjev v2 0.77 *theirs*", "3-way NLI ≠ Noul", "priority 0.464 = majority floor", "banking77 contaminated", "raw margins not probabilities", "do not distill Jev as teacher of record (they distilled Haiku)", "“0.9 is not one number”", "ranking ≠ calibration", "banking77 0.8–0.9 stated 0.86 actual 0.73 over-confident *theirs*", "≠ Praveenrajus/jev-bench ≠ fstandhartinger/jevbench", "$0.0000153–$0.0000226 vs circulating $0.0004 (~20×)", "Score is 0..n-1 expectation not 0–1", "Noul has no confidence field", "TCP floor 198.8 ms", "type reliability is not a reason to choose Jev (json_schema 5/5)", "gateway tax not one number", "Function-only 5/8 vs hybrid 8/8", "4/8 without Jev", "8 designed cases not conversion lift", "200-row pilot Jev 86.5% 173/200 vs Gemini Flash-Lite 86.0% 172/200 vs Pro 87.0% 174/200 *theirs*", "not a ranking", "情緒測謊器", "8-example Jev vs GPT-5.6 Sol ~64× cost 5.4× latency *theirs*", "synthetic; no inference", "≠ JevBench v1.2 §78", "Judged 3317 / listed 2560", "Jev judges, code applies policy", "APA “microsecond policy / zero hallucination” overclaim", "Client-side quiz; pointer from held docs; scanned-PDF warn", "Jev judges / agent reasons / user decides", "selecting an option is not permission to implement", "pattern exact, judgement must clear floor", "no matching pattern → no model call", "not a correctness oracle", "Spec vs artifact remainder", "treating 0.85 as 85% / minProbability hard-gate as Harbor", "VERIFY acquires discriminating evidence, never same-pool confidence-only rescoring", "fast/full/max are ceilings not sizes", "Solar writes, Jev chooses NEXT ACTION", "do not reopen or amend PR #23 or #24 or #25 or #26 or #27", , "Calibration is not alpha", "NO CURRENT ALPHA CANDIDATE", "ΔR² approximately +0.00084", "Brier 0.2131387", "ECE 0.0421875", "Adding Jev probability to deterministic volatility improved Brier by only 1.4058e-05", "default 0.5 keeps zero non pinned", "keepResult median 0.14 to 0.17", "keepCall median 0.28 to 0.35", "usable range is about 0.10 to 0.25", "7.8% to 57.9%", "judges results it never sees", "task-finish eval not built yet", "$0.002 per compaction", "slavadubrov/sgr-judge-bench ≠ slavadubrov/jev-judge-bench", "Jev 108/120 $0.083 0.34 s", "Luna SGR 114/120", "paired Jev accuracy-difference intervals include zero", "not evidence of equivalence", "GLM SGR 26/120 93 format failures", "Terra-planned Jev hybrid 55/120", "rule-based by default, optionally Jev-backed", "empty README", "missing key cannot break the experience", "prefill plus exactly one decode", "softmax over A/B/C ≠ Noul", "BBQ 9,053/10,000 (90.53%)", "ECE 0.0890", "Mean confidence 0.9943", "overconfident", "score and noul not implemented", "DGUI 12 rows (was 6)", "INSTRUCT 119 rows likes 2", "encode the state once, decide everything in parallel", "0.740 accuracy against a 0.508 majority", "ECE 0.047", "fine-tune's advantage ends where its 384-token training data does", "jasonkneen/open-jev ≠ pngwn/open-jev", "same sha d41dc3cd", "Space does not call Jev", "recomputes routing from saved probabilities", "200-case Jev 97.0% / 100.0% / 95.0% / MAE 9.22", "synthetic repository benchmark", "Jev evaluations are advisory", "YehuiTang0316/jev-nlgrep ≠ Bentlybro/jevgrep ≠ can1357/jegrep ≠ uehaj/jev-semgrep", "default threshold 0.8 still soft", "40-line windows cannot prove whole function", "token-native sequential start/end Choice", "Gemini/Haiku stubs not configured yet", "handful of hand-written examples, not a benchmark", "Jev judged exactly what it was given", "laguagu/jev-skills ≠ laguagu/jev-evidence-lab ≠ Pleo2/awesome-jev-agent-skills", "contract_passed is not a claim of guaranteed factual truth", "Wilson lower bound 0.85 floor", "fixture mode no savings claim", "SemIf 2207★ (+21 vs §110 2186)", "jevlike 1043★ (+5 vs 1038)", "TypeAR 15★ (+1 vs 14)", "AnotiaWang 97★ (+1 vs 96)", "yibie/awesome-jev 506★ (+16 vs 490)", "Laya likes 822 (was 802)", "tracker likes 64 flat, lastModified UNCHANGED", "do not reopen or amend PR #23/#24/#25/#26/#27/#28", "Heman10x-NGU/Verdict-open-jev ≠ Heman10x-NGU/openJev-verdict-2.0", "TF-IDF + LogReg ECE 0.0207 vs Jev 0.1440", "Verdict-open-jev 48.07% vs Jev 90.80%", "abstention combined recall 10.00%", "p50 35.58 ms", "K=25 (maximum capacity) 72.00%", "0.85 coverage 84.60% selective risk 1.18%", "26.1× faster than standard Qwen JSON generation", "Jevify 90.0% / 167 ms CUDA graphs disabled", "Finding 1: Brier on stated confidence alone is a trap", "grpo_rlcr 0.78 / ECE 0.084", "reliability 0.007 but resolution 0.000", "27 900 schema-driven decisions", "13 600 / 13 600 questions", "candidate mass min 0.99999624", "22 configs · 166,054 rows · 4 calibration-gold", "sha a39eba3f", "Student B MAE 0.148 / Pearson 0.836 / 86.0%", "pngwn/open-jev-laya-bench README 404", "sha 9f69c742 likes 2", "HDFS 0.9933 (745/750) / retain 0.0084", "BGL ERROR/FATAL protection 1.0000", "2,479 / 2,500 HDFS uncertain", "cache hit 0.9648 (2412/2500)", "$0.153936 estimated", "E2 recomputes from saved probabilities", "Space sha eda59e0a", "MASSIVE English 0.783 / Khmer 0.033 / Hindi 0.133", "40–48 rows too small to ship T", "T never changes argmax", "siren2345/jev-single-decode-transformers ≠ siren2345/jev-single-decode", "Split Transformers experiment from llama.cpp runtime", "tanayvasishtha/jev-lab ≠ dairui1/jev-lab ≠ BrendanH18/jev-lab ≠ yibie/laya-jev-lab", "Four experiments stress-testing TypeSafe's Jev: calibration, bundle bias, label bias, and ensembling", "second pass must be $0.00 from cache", "The pages never call Jev", "Gemma 4 31B 77.0% / Jev 1.13.0 61.4% / Laya 322M 0.0%", "restriction state 95.0% against 84.4%", "None of the systems are particularly good at knowing when to stop and ask", "They skip the question and call a tool directly", "100% schema pass", "six-field joint 48.8% vs 72.8%", "ywchiu/jev_benchmark ≠ Running-Dolphins/jev-bench ≠ Praveenrajus/jev-bench", "ACT / REVIEW / FALLBACK", "A provider failure, timeout, malformed output, or missing answer is **not** a policy outcome", "confidence is descriptive provider output, not a substitute for probability", "Quality denominators include only valid scored answers", "an exact halfway tie chooses the lower level", "aiwithenoch/Jev-Skill ≠ simplosophy/jev-skill ≠ laguagu/jev-skills", "The local path does not claim to turn a smaller checkpoint into Jev", "Low support becomes decision: \"review\"", "MIT-0 SPDX NOASSERTION", "current-llm", "结构兼容,不是 Jev 模型能力", "altryne/jevify ≠ Mintzs/jevify ≠ gulagala001/jevify ≠ uspraveen/Jevify", "Find where Jev belongs. Design the questions. Measure the difference", "TypeAR-AI/TypeAR 301 → TypeLLM/TypeLLM", "TypeLLM/TypeLLM 16★", "SemIf 2241★ (+34 vs §111 2207)", "jevlike 1051★ (+8 vs 1043)", "AnotiaWang 98★ (+1 vs 97)", "yibie/awesome-jev 525★ (+19 vs 506)", "Laya likes 864 (was 822)", "tracker likes 67 (+3 vs 64)", "lastModified UNCHANGED `2026-09-20T04:29:16.000Z`", "do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#32", "hysteresis enter/exit / replay policy without inference", "calibration does not compose / hop-ECE permutation-invariant", "equal-width vs quantile ECE / ranking ≠ calibration", "Qwen2.5 ≠ Archer / Qwen 3.8 sparring ≠ Archer / Qwen/Qwen3.8-27B ≠ Archer", "Deferred Crispification / TCE / AMS", "g0runmezadam/what-is-jev IS tunahansahin897/what-is-jev", "pd.cut equal-width vs jeval quantile", "A hunch is a probability with a policy attached", "soundness theater / measurement theater / hourly 0843", , "Jev Capability Resolver / NiazMorshed2007/jcr", "one tool nested capability tree / returns context / does not execute", "skills vs capabilities / workflow+judgment vs operations", "format independent of Jev / proposed open standard", "JCR_BAND_RATIO 0.6 is application policy / soft scores ≠ hard gates", "routing ≠ permission / docs ≠ authority to run", "sol-vs-opus5-20 lookup+explain / n=1 / Not Harbor task-execution", "wall-time mixed / Sol slower with JCR in 19/20", "NiazMorshed2007/jcr ≠ skill-broker ≠ skillranker ≠ jev-sift ≠ jev-lens ≠ jevusher ≠ jev_select_capability", "do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34", "notes.md §116", "copy the SemIf/MLX installer?", "quote 5.21× as beating Jev?", "treat 0.845 as a TypeSafe replica?", "collapse SemIf into kw2828/zhihz/semif-rs/semif-serve", "softmax over options as a Noul", "llm prompt to jev primitives", "conversion assistant not equivalent behavior", "heuristic conversion ≠ calibrated Noul", "alexwestco/llm-to-jev ≠ altryne/jevify", "user-provided 0940 / notes.md §118", "judge ≠ actuator", "candidate_mass", "softmax over A–H ≠ Noul", "hourly 0947 / notes.md §119", "ggmlc GGUF is not llama.cpp", "serving substrate ≠ calibrated replica", "Qwen3.5-9B ≠ Archer", "planner writes JEV selects", "hourly 1049 / notes.md §120", "open recreation ≠ calibrated replica", "semantic lint is a sensor not a proof", "cutoff 0.8 still soft", "paired bootstrap CIs *theirs*", "Same accuracy, 35x faster *theirs*", "hourly 1143 / notes.md §121", "revisit HIGH / since-last-look", "catalogued repo changed", "star-noise vs material change", "densify prior notes without inventing equivalence", "decide is not generate", "tryDecide returns typed calibrated judgments not a token stream", "GLiNER/GLiClass ports are class members not Jev replicas", "93.5% *theirs* not Harbor", "74.9 *theirs* not Harbor", "8.7x *theirs* not Harbor", "Option-Marker joint attention", "openjev:0.2.1", "thinking=True/False per-field budget", "PLAN_Qwen35", "hyperspaceai/jevcache ≠ kushals256/jevcache", "wire-compat ≠ logit-equiv", "SHA move is not a replica", "hourly 1248 / notes.md §123", "typesafe-sdk 0.7 Pydantic response models", "msgspec dropped", "The server's output is unchanged and was never wrong", "SchemaError is 400 plain-string detail not 422 list", "Pydantic response models ≠ logit-equiv", "msgspec dropped is not a replica", "Error contract is not a Noul", "coverage-at-error-budget *theirs* not Harbor", "PLAN_Qwen35 still proposal for review", "GLiNER locate ports are class members not Jev replicas", "Locate ≠ decide", "~160 ms *theirs* not Harbor", "0.971 F1 *theirs* not Harbor", "hf:fr0stbit3/laya-gguf serving substrate ≠ calibrated replica", "jkcdarunday/SystemOne-Next ≠ TypeSafe System One", "hourly 1340 / notes.md §124", "vLLM NVIDIA + MLX Apple Silicon", "Codiv hosted free endpoint", "dual /v1/systemone + /v1/chat/completions", "chat 501 on MLX", "dual serving is not generate", "Hosted Codiv ≠ TypeSafe", "hr98w/jev-visual 167★ Apple Silicon visual candidate scoring", "37.30s → 2.40s at 64 decisions *theirs*", "Breakout 9 bricks 6 returns 2 lives *theirs*", "candidate probabilities are relative not correctness", "jkudish/jev-mcp 156★ ten MCP tools", "recommendation is advisory", "the server never blocks on its own", "TypeSafe CLERC 5% to 18% *theirs*", "jkudish/jev-mcp ≠ burnigtm/jev-mcp", "zhengxuyu/litjev off-the-shelf Qwen decision layer", "Probabilities are not calibrated by default", "Qwen/Qwen3.8-27B ≠ Archer", "zhengxuyu/litjev ≠ alexwestco/llm-to-jev", "Zefan-Cai/Open-Jev LoRA + scalar head", "2B 94.71% 9B 97.54% hard test *theirs*", "2B OOD 86.02% 9B OOD 91.97% *theirs*", "80,816 training rows", "27B still in progress", "LoRA ≠ RLCD replica", "Zefan-Cai/Open-Jev ≠ TheoLeeCJ/openjev ≠ razorback16/openjev", "cristianoliveira/jeq intelligence you can pipe", "pass-min 0.8 still soft", "JEQ does not own actions", "AndyInQtr/laya-coreai CoreML serving substrate ≠ calibrated replica", "AndyInQtr/laya-coreai ≠ mizorewww/laya-coreml", "hourly 1441 / notes.md §125", "TypeLLM/TypeLLM densify HEAD 6a48f9f1e623", "README densify 3k→12k B", "Batch 5.8x *theirs*", "Constrained AR ≠ calibrated Noul", "jaredpalmer/kev densify HEAD b339f446a0ef", "Kev-0.6B 4B 8B family", "4B new-source 0.790/0.806 *theirs*", "8B new-source 0.796/0.780 *theirs*", "Jev hosted 0.857 *theirs*", "Questions share the input text but cannot read each other", "No Jev outputs were used for training", "8.2% ≥0.9 on wrong *theirs*", "option order can change an answer", "Qwen3 ≠ Archer", "TheoOliveira/pi-jev 21★ fail-closed routing", "JEV_THRESHOLD 0.65 still soft", "harshwasan/jev-sentinel fail closed never auto-allows", "harshwasan/jev-sentinel ≠ leepokai/jev-guard", "jackbarunz/jev-tool-router ≠ esinocchi/jev-tool-router", "threshold 0.90 still soft", "76/81 vs 77/81 *theirs*", "0.419s vs 2.459s *theirs*", "$0.00486 vs $0.03673 *theirs*", "not a security boundary", "baronunread/leanest fail-open uncertainty means RUN", "classifier.dev default Jev/Laya pluggable", "openlayer-ai/jevals ≠ dayhaysoos/jevals", "estimates not Harbor", "classifier ≠ authorizer", "MrJev/awesome-jev 118 entries catalog ≠ endorsement", "MrJev/awesome-jev ≠ yibie/awesome-jev", "Koushik890/jev-firewall fail closed ask_below 0.7 still soft", "CompleteTech-LLC-AI-Research/jev-codex-approval experimental native not compiled", "confidence is not a measured probability", "rh-guard owns primary gates", "hf:rAVEUK/open-jev-deberta-v3-large encoder class member not Jev replica", "hf:p-yan/laya-quanto serving substrate ≠ calibrated replica", "hf:Gtrkrsk/laya serving substrate ≠ calibrated replica", "hourly 1542 / notes.md §126", or "cascade sign-flip / calibration theater": read `references/faq.md`, then `references/mental-models.md`, then `references/mixed-architecture.md`, then `references/judgment-class.md` before any mapping. Proof, @@ -381,6 +381,48 @@ Do not reopen or amend PR #23–#46. Does not bump 0.5.0. Skip Archer. **Hourly 1340 HIGH (`notes.md` §124).** typesafe-sdk 0.7 Pydantic response models. msgspec dropped. MLX backend 400 plain-text error contract. Pydantic response models ≠ logit-equiv. msgspec dropped is not a replica. Error contract is not a Noul. PLAN_Qwen35 densify. coverage-at-error-budget *theirs* not Harbor. GLiNER locate ports are class members not Jev replicas. Locate ≠ decide. ~160 ms *theirs* not Harbor. SHA move is not a replica. catalog ≠ endorsement. Do not reopen or amend PR #23–#46. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. Hourly 1340 uniqueness lock: typesafe-sdk 0.7 Pydantic response models; msgspec dropped; The server's output is unchanged and was never wrong; MLX backend 400 plain-text error contract; SchemaError is 400 plain-string detail not 422 list; razorback16/openjev densify HEAD 6e91dfc031bc README SHA cbdcc8de0304; Pydantic response models ≠ logit-equiv; msgspec dropped is not a replica; Error contract is not a Noul; wire-compat ≠ logit-equiv; PLAN_Qwen35 densify; corrected Qwen3.5 LoRA target names verified; in_proj_qkv in_proj_z in_proj_a in_proj_b out_proj; peft 0.21 existence proof; OOD-calibration study; coverage-at-error-budget metric in Phase 0; PLAN_Qwen35 still proposal for review; deadline 0.53→0.82 at 9B *theirs*; isolation would fail by construction on DeltaNet; Qwen3.5-9B ≠ Archer; jaredpalmer/kev densify HEAD 75cc15ddb8e2 PLAN SHA eca543246f50; GLiNER locate ports are class members not Jev replicas; urchade/GLiNER ≠ fbilhaut/gline-rs ≠ lmoe/gliner-onnx.js ≠ shershah1024/gliner-native-runtime; Locate ≠ decide; Jev-Vision skip 0.936 effect 0.967 done 0.896 157 ms *theirs*; ~160 ms *theirs* not Harbor; 0.971 F1 *theirs* not Harbor; coverage-at-error-budget *theirs* not Harbor; hf:fr0stbit3/laya-gguf serving substrate ≠ calibrated replica; jkcdarunday/SystemOne-Next ≠ TypeSafe System One; catalog ≠ endorsement; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46; notes.md §124 +## Hourly 1542 HIGH (`notes.md` §126) + +TypeLLM/TypeLLM densify HEAD 6a48f9f1e623 README SHA dbdc1f193537. +README densify 3k→12k B. thinking=True/False per-field budget. +type safety does not guarantee factual accuracy. +Batch 5.8x *theirs*. Constrained AR ≠ calibrated Noul. +Qwen/Qwen3.8-27B ≠ Archer. +jaredpalmer/kev densify HEAD b339f446a0ef README SHA 86b0a19909f3. +Kev-0.6B 4B 8B family. 4B new-source 0.790/0.806 *theirs*. +8B new-source 0.796/0.780 *theirs*. Jev hosted 0.857 *theirs*. +Questions share the input text but cannot read each other. +No Jev outputs were used for training. +8.2% ≥0.9 on wrong *theirs*. option order can change an answer. +Qwen3 ≠ Archer. +TheoOliveira/pi-jev 21★ fail-closed routing. +JEV_THRESHOLD 0.65 still soft. routing ≠ permission. +harshwasan/jev-sentinel fail closed never auto-allows. +harshwasan/jev-sentinel ≠ leepokai/jev-guard. +jackbarunz/jev-tool-router ≠ esinocchi/jev-tool-router. +threshold 0.90 still soft. +76/81 vs 77/81 *theirs*. 0.419s vs 2.459s *theirs*. +$0.00486 vs $0.03673 *theirs*. +does not execute. not a security boundary. +baronunread/leanest fail-open uncertainty means RUN. +classifier.dev default Jev/Laya pluggable. +openlayer-ai/jevals ≠ dayhaysoos/jevals. +estimates not Harbor. classifier ≠ authorizer. +MrJev/awesome-jev 118 entries catalog ≠ endorsement. +MrJev/awesome-jev ≠ yibie/awesome-jev. +Koushik890/jev-firewall fail closed ask_below 0.7 still soft. +CompleteTech-LLC-AI-Research/jev-codex-approval experimental native not compiled. +confidence is not a measured probability. rh-guard owns primary gates. +hf:rAVEUK/open-jev-deberta-v3-large encoder class member not Jev replica. +hf:p-yan/laya-quanto serving substrate ≠ calibrated replica. +hf:Gtrkrsk/laya serving substrate ≠ calibrated replica. +wire-compat ≠ logit-equiv. SHA move is not a replica. catalog ≠ endorsement. +Archer still promised_not_landed. Hub archerhume/4rcherhume HTTP 401. +Do not copy keys. Fresh PR off `f40139c` (merged #48). +Do not reopen or amend PR #23–#48. Does not bump 0.5.0. Skip Archer. +`invented_signal: false`. + + ## Hourly 1441 HIGH (`notes.md` §125) vLLM NVIDIA + MLX Apple Silicon. Codiv hosted free endpoint. @@ -417,3 +459,7 @@ Do not reopen or amend PR #23–#47. Does not bump 0.5.0. Skip Archer. **Hourly 1441 HIGH (`notes.md` §125).** vLLM NVIDIA + MLX Apple Silicon. Codiv hosted free endpoint. dual /v1/systemone + /v1/chat/completions. chat 501 on MLX. dual serving is not generate. Hosted Codiv ≠ TypeSafe. candidate probabilities are relative not correctness. recommendation is advisory. the server never blocks on its own. LoRA ≠ RLCD replica. pass-min 0.8 still soft. 37.30s → 2.40s at 64 decisions *theirs*. 2B 94.71% 9B 97.54% hard test *theirs*. wire-compat ≠ logit-equiv. SHA move is not a replica. catalog ≠ endorsement. Do not reopen or amend PR #23–#47. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. Hourly 1441 uniqueness lock: vLLM NVIDIA + MLX Apple Silicon; Codiv hosted free endpoint; dual /v1/systemone + /v1/chat/completions; razorback16/openjev densify HEAD cddbd962c88a README SHA a5943415cb92; STE README rewrite; serving-port densify; chat 501 on MLX; dual serving is not generate; Hosted Codiv ≠ TypeSafe; wire-compat ≠ logit-equiv; SHA move is not a replica; Error contract is not a Noul; hr98w/jev-visual 167★ Apple Silicon visual candidate scoring; 37.30s → 2.40s at 64 decisions *theirs*; Breakout 9 bricks 6 returns 2 lives *theirs*; candidate probabilities are relative not correctness; jkudish/jev-mcp 156★ ten MCP tools; recommendation is advisory; the server never blocks on its own; TypeSafe CLERC 5% to 18% *theirs*; jkudish/jev-mcp ≠ burnigtm/jev-mcp; zhengxuyu/litjev off-the-shelf Qwen decision layer; Probabilities are not calibrated by default; Qwen/Qwen3.8-27B ≠ Archer; zhengxuyu/litjev ≠ alexwestco/llm-to-jev; Zefan-Cai/Open-Jev LoRA + scalar head; 2B 94.71% 9B 97.54% hard test *theirs*; 2B OOD 86.02% 9B OOD 91.97% *theirs*; 80,816 training rows; 27B still in progress; LoRA ≠ RLCD replica; Zefan-Cai/Open-Jev ≠ TheoLeeCJ/openjev ≠ razorback16/openjev; cristianoliveira/jeq intelligence you can pipe; pass-min 0.8 still soft; JEQ does not own actions; AndyInQtr/laya-coreai CoreML serving substrate ≠ calibrated replica; AndyInQtr/laya-coreai ≠ mizorewww/laya-coreml; catalog ≠ endorsement; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47; notes.md §125 + + +**Hourly 1542 HIGH (`notes.md` §126).** TypeLLM README densify 3k→12k B. Batch 5.8x *theirs*. Constrained AR ≠ calibrated Noul. kev family densify. 4B new-source 0.790/0.806 *theirs*. 8.2% ≥0.9 on wrong *theirs*. option order can change an answer. fail-closed routing vs fail-open test selection. classifier ≠ authorizer. estimates not Harbor. catalog ≠ endorsement. SHA move is not a replica. Do not reopen or amend PR #23–#48. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. +Hourly 1542 uniqueness lock: TypeLLM/TypeLLM densify HEAD 6a48f9f1e623 README SHA dbdc1f193537; README densify 3k→12k B; thinking=True/False per-field budget; type safety does not guarantee factual accuracy; Batch 5.8x *theirs*; Constrained AR ≠ calibrated Noul; Qwen/Qwen3.8-27B ≠ Archer; jaredpalmer/kev densify HEAD b339f446a0ef README SHA 86b0a19909f3; Kev-0.6B 4B 8B family; 4B new-source 0.790/0.806 *theirs*; 8B new-source 0.796/0.780 *theirs*; Jev hosted 0.857 *theirs*; Questions share the input text but cannot read each other; No Jev outputs were used for training; 8.2% ≥0.9 on wrong *theirs*; option order can change an answer; Qwen3 ≠ Archer; wire-compat ≠ logit-equiv; SHA move is not a replica; TheoOliveira/pi-jev 21★ fail-closed routing; JEV_THRESHOLD 0.65 still soft; routing ≠ permission; harshwasan/jev-sentinel fail closed never auto-allows; harshwasan/jev-sentinel ≠ leepokai/jev-guard; jackbarunz/jev-tool-router ≠ esinocchi/jev-tool-router; threshold 0.90 still soft; 76/81 vs 77/81 *theirs*; 0.419s vs 2.459s *theirs*; $0.00486 vs $0.03673 *theirs*; does not execute; not a security boundary; baronunread/leanest fail-open uncertainty means RUN; classifier.dev default Jev/Laya pluggable; openlayer-ai/jevals ≠ dayhaysoos/jevals; estimates not Harbor; classifier ≠ authorizer; MrJev/awesome-jev 118 entries catalog ≠ endorsement; MrJev/awesome-jev ≠ yibie/awesome-jev; Koushik890/jev-firewall fail closed ask_below 0.7 still soft; CompleteTech-LLC-AI-Research/jev-codex-approval experimental native not compiled; confidence is not a measured probability; rh-guard owns primary gates; hf:rAVEUK/open-jev-deberta-v3-large encoder class member not Jev replica; hf:p-yan/laya-quanto serving substrate ≠ calibrated replica; hf:Gtrkrsk/laya serving substrate ≠ calibrated replica; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48; notes.md §126 diff --git a/.agents/skills/augustus/references/agent-self-assessment.md b/.agents/skills/augustus/references/agent-self-assessment.md index 83ac75d..fc7dbdd 100644 --- a/.agents/skills/augustus/references/agent-self-assessment.md +++ b/.agents/skills/augustus/references/agent-self-assessment.md @@ -958,3 +958,7 @@ Hourly 1340 uniqueness lock: typesafe-sdk 0.7 Pydantic response models; msgspec **Hourly 1441 HIGH (`notes.md` §125).** vLLM NVIDIA + MLX Apple Silicon. Codiv hosted free endpoint. dual /v1/systemone + /v1/chat/completions. chat 501 on MLX. dual serving is not generate. Hosted Codiv ≠ TypeSafe. candidate probabilities are relative not correctness. recommendation is advisory. the server never blocks on its own. LoRA ≠ RLCD replica. pass-min 0.8 still soft. 37.30s → 2.40s at 64 decisions *theirs*. 2B 94.71% 9B 97.54% hard test *theirs*. wire-compat ≠ logit-equiv. SHA move is not a replica. catalog ≠ endorsement. Do not reopen or amend PR #23–#47. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. Hourly 1441 uniqueness lock: vLLM NVIDIA + MLX Apple Silicon; Codiv hosted free endpoint; dual /v1/systemone + /v1/chat/completions; razorback16/openjev densify HEAD cddbd962c88a README SHA a5943415cb92; STE README rewrite; serving-port densify; chat 501 on MLX; dual serving is not generate; Hosted Codiv ≠ TypeSafe; wire-compat ≠ logit-equiv; SHA move is not a replica; Error contract is not a Noul; hr98w/jev-visual 167★ Apple Silicon visual candidate scoring; 37.30s → 2.40s at 64 decisions *theirs*; Breakout 9 bricks 6 returns 2 lives *theirs*; candidate probabilities are relative not correctness; jkudish/jev-mcp 156★ ten MCP tools; recommendation is advisory; the server never blocks on its own; TypeSafe CLERC 5% to 18% *theirs*; jkudish/jev-mcp ≠ burnigtm/jev-mcp; zhengxuyu/litjev off-the-shelf Qwen decision layer; Probabilities are not calibrated by default; Qwen/Qwen3.8-27B ≠ Archer; zhengxuyu/litjev ≠ alexwestco/llm-to-jev; Zefan-Cai/Open-Jev LoRA + scalar head; 2B 94.71% 9B 97.54% hard test *theirs*; 2B OOD 86.02% 9B OOD 91.97% *theirs*; 80,816 training rows; 27B still in progress; LoRA ≠ RLCD replica; Zefan-Cai/Open-Jev ≠ TheoLeeCJ/openjev ≠ razorback16/openjev; cristianoliveira/jeq intelligence you can pipe; pass-min 0.8 still soft; JEQ does not own actions; AndyInQtr/laya-coreai CoreML serving substrate ≠ calibrated replica; AndyInQtr/laya-coreai ≠ mizorewww/laya-coreml; catalog ≠ endorsement; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47; notes.md §125 + + +**Hourly 1542 HIGH (`notes.md` §126).** TypeLLM README densify 3k→12k B. Batch 5.8x *theirs*. Constrained AR ≠ calibrated Noul. kev family densify. 4B new-source 0.790/0.806 *theirs*. 8.2% ≥0.9 on wrong *theirs*. option order can change an answer. fail-closed routing vs fail-open test selection. classifier ≠ authorizer. estimates not Harbor. catalog ≠ endorsement. SHA move is not a replica. Do not reopen or amend PR #23–#48. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. +Hourly 1542 uniqueness lock: TypeLLM/TypeLLM densify HEAD 6a48f9f1e623 README SHA dbdc1f193537; README densify 3k→12k B; thinking=True/False per-field budget; type safety does not guarantee factual accuracy; Batch 5.8x *theirs*; Constrained AR ≠ calibrated Noul; Qwen/Qwen3.8-27B ≠ Archer; jaredpalmer/kev densify HEAD b339f446a0ef README SHA 86b0a19909f3; Kev-0.6B 4B 8B family; 4B new-source 0.790/0.806 *theirs*; 8B new-source 0.796/0.780 *theirs*; Jev hosted 0.857 *theirs*; Questions share the input text but cannot read each other; No Jev outputs were used for training; 8.2% ≥0.9 on wrong *theirs*; option order can change an answer; Qwen3 ≠ Archer; wire-compat ≠ logit-equiv; SHA move is not a replica; TheoOliveira/pi-jev 21★ fail-closed routing; JEV_THRESHOLD 0.65 still soft; routing ≠ permission; harshwasan/jev-sentinel fail closed never auto-allows; harshwasan/jev-sentinel ≠ leepokai/jev-guard; jackbarunz/jev-tool-router ≠ esinocchi/jev-tool-router; threshold 0.90 still soft; 76/81 vs 77/81 *theirs*; 0.419s vs 2.459s *theirs*; $0.00486 vs $0.03673 *theirs*; does not execute; not a security boundary; baronunread/leanest fail-open uncertainty means RUN; classifier.dev default Jev/Laya pluggable; openlayer-ai/jevals ≠ dayhaysoos/jevals; estimates not Harbor; classifier ≠ authorizer; MrJev/awesome-jev 118 entries catalog ≠ endorsement; MrJev/awesome-jev ≠ yibie/awesome-jev; Koushik890/jev-firewall fail closed ask_below 0.7 still soft; CompleteTech-LLC-AI-Research/jev-codex-approval experimental native not compiled; confidence is not a measured probability; rh-guard owns primary gates; hf:rAVEUK/open-jev-deberta-v3-large encoder class member not Jev replica; hf:p-yan/laya-quanto serving substrate ≠ calibrated replica; hf:Gtrkrsk/laya serving substrate ≠ calibrated replica; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48; notes.md §126 diff --git a/.agents/skills/augustus/references/applied-mappings.md b/.agents/skills/augustus/references/applied-mappings.md index 92e462c..94f1dba 100644 --- a/.agents/skills/augustus/references/applied-mappings.md +++ b/.agents/skills/augustus/references/applied-mappings.md @@ -2537,3 +2537,7 @@ Hourly 1340 uniqueness lock: typesafe-sdk 0.7 Pydantic response models; msgspec **Hourly 1441 HIGH (`notes.md` §125).** vLLM NVIDIA + MLX Apple Silicon. Codiv hosted free endpoint. dual /v1/systemone + /v1/chat/completions. chat 501 on MLX. dual serving is not generate. Hosted Codiv ≠ TypeSafe. candidate probabilities are relative not correctness. recommendation is advisory. the server never blocks on its own. LoRA ≠ RLCD replica. pass-min 0.8 still soft. 37.30s → 2.40s at 64 decisions *theirs*. 2B 94.71% 9B 97.54% hard test *theirs*. wire-compat ≠ logit-equiv. SHA move is not a replica. catalog ≠ endorsement. Do not reopen or amend PR #23–#47. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. Hourly 1441 uniqueness lock: vLLM NVIDIA + MLX Apple Silicon; Codiv hosted free endpoint; dual /v1/systemone + /v1/chat/completions; razorback16/openjev densify HEAD cddbd962c88a README SHA a5943415cb92; STE README rewrite; serving-port densify; chat 501 on MLX; dual serving is not generate; Hosted Codiv ≠ TypeSafe; wire-compat ≠ logit-equiv; SHA move is not a replica; Error contract is not a Noul; hr98w/jev-visual 167★ Apple Silicon visual candidate scoring; 37.30s → 2.40s at 64 decisions *theirs*; Breakout 9 bricks 6 returns 2 lives *theirs*; candidate probabilities are relative not correctness; jkudish/jev-mcp 156★ ten MCP tools; recommendation is advisory; the server never blocks on its own; TypeSafe CLERC 5% to 18% *theirs*; jkudish/jev-mcp ≠ burnigtm/jev-mcp; zhengxuyu/litjev off-the-shelf Qwen decision layer; Probabilities are not calibrated by default; Qwen/Qwen3.8-27B ≠ Archer; zhengxuyu/litjev ≠ alexwestco/llm-to-jev; Zefan-Cai/Open-Jev LoRA + scalar head; 2B 94.71% 9B 97.54% hard test *theirs*; 2B OOD 86.02% 9B OOD 91.97% *theirs*; 80,816 training rows; 27B still in progress; LoRA ≠ RLCD replica; Zefan-Cai/Open-Jev ≠ TheoLeeCJ/openjev ≠ razorback16/openjev; cristianoliveira/jeq intelligence you can pipe; pass-min 0.8 still soft; JEQ does not own actions; AndyInQtr/laya-coreai CoreML serving substrate ≠ calibrated replica; AndyInQtr/laya-coreai ≠ mizorewww/laya-coreml; catalog ≠ endorsement; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47; notes.md §125 + + +**Hourly 1542 HIGH (`notes.md` §126).** TypeLLM README densify 3k→12k B. Batch 5.8x *theirs*. Constrained AR ≠ calibrated Noul. kev family densify. 4B new-source 0.790/0.806 *theirs*. 8.2% ≥0.9 on wrong *theirs*. option order can change an answer. fail-closed routing vs fail-open test selection. classifier ≠ authorizer. estimates not Harbor. catalog ≠ endorsement. SHA move is not a replica. Do not reopen or amend PR #23–#48. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. +Hourly 1542 uniqueness lock: TypeLLM/TypeLLM densify HEAD 6a48f9f1e623 README SHA dbdc1f193537; README densify 3k→12k B; thinking=True/False per-field budget; type safety does not guarantee factual accuracy; Batch 5.8x *theirs*; Constrained AR ≠ calibrated Noul; Qwen/Qwen3.8-27B ≠ Archer; jaredpalmer/kev densify HEAD b339f446a0ef README SHA 86b0a19909f3; Kev-0.6B 4B 8B family; 4B new-source 0.790/0.806 *theirs*; 8B new-source 0.796/0.780 *theirs*; Jev hosted 0.857 *theirs*; Questions share the input text but cannot read each other; No Jev outputs were used for training; 8.2% ≥0.9 on wrong *theirs*; option order can change an answer; Qwen3 ≠ Archer; wire-compat ≠ logit-equiv; SHA move is not a replica; TheoOliveira/pi-jev 21★ fail-closed routing; JEV_THRESHOLD 0.65 still soft; routing ≠ permission; harshwasan/jev-sentinel fail closed never auto-allows; harshwasan/jev-sentinel ≠ leepokai/jev-guard; jackbarunz/jev-tool-router ≠ esinocchi/jev-tool-router; threshold 0.90 still soft; 76/81 vs 77/81 *theirs*; 0.419s vs 2.459s *theirs*; $0.00486 vs $0.03673 *theirs*; does not execute; not a security boundary; baronunread/leanest fail-open uncertainty means RUN; classifier.dev default Jev/Laya pluggable; openlayer-ai/jevals ≠ dayhaysoos/jevals; estimates not Harbor; classifier ≠ authorizer; MrJev/awesome-jev 118 entries catalog ≠ endorsement; MrJev/awesome-jev ≠ yibie/awesome-jev; Koushik890/jev-firewall fail closed ask_below 0.7 still soft; CompleteTech-LLC-AI-Research/jev-codex-approval experimental native not compiled; confidence is not a measured probability; rh-guard owns primary gates; hf:rAVEUK/open-jev-deberta-v3-large encoder class member not Jev replica; hf:p-yan/laya-quanto serving substrate ≠ calibrated replica; hf:Gtrkrsk/laya serving substrate ≠ calibrated replica; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48; notes.md §126 diff --git a/.agents/skills/augustus/references/composition-algebra.md b/.agents/skills/augustus/references/composition-algebra.md index 05f7b20..79f8aa3 100644 --- a/.agents/skills/augustus/references/composition-algebra.md +++ b/.agents/skills/augustus/references/composition-algebra.md @@ -2578,6 +2578,91 @@ Soft Noul ≠ hard safety. Qwen/Qwen3.8-27B ≠ Archer. Hub archerhume/4rcherhume HTTP 401. Archer still promised_not_landed. Full cards: `faq.md`. + +433. **TypeLLM densify PRIMARY** (TypeLLM/TypeLLM): + densify §113. HEAD 6a48f9f1e623 README SHA dbdc1f193537. 16★. + README densify 3k→12k B. thinking=True/False per-field budget. + type safety does not guarantee factual accuracy. + Constrained AR ≠ calibrated Noul. Qwen/Qwen3.8-27B ≠ Archer. + Full cards: `judgment-class.md`, `validation.md`. +434. **TypeLLM Batch 5.8x *theirs*** (TypeLLM/TypeLLM): + Sequential 9.35 s vs batch 1.61 s K=16 5.8x *theirs*. + Batch 5.8x *theirs*. not Harbor. + Full cards: `validation.md`. +435. **kev family densify PRIMARY** (jaredpalmer/kev): + densify §45. HEAD b339f446a0ef README SHA 86b0a19909f3. + Kev-0.6B 4B 8B family. 4B new-source 0.790/0.806 *theirs*. + 8B new-source 0.796/0.780 *theirs*. Jev hosted 0.857 *theirs*. + Questions share the input text but cannot read each other. + No Jev outputs were used for training. Qwen3 ≠ Archer. + Full cards: `judgment-class.md`, `validation.md`. +436. **kev 8.2% / option-order *theirs*** (jaredpalmer/kev): + 8.2% ≥0.9 on wrong *theirs*. option order can change an answer. + isolation ≠ option-order immunity. Full cards: `validation.md`. +437. **pi-jev fail-closed routing densify** (TheoOliveira/pi-jev): + densify §42. 21★ fail-closed routing. JEV_THRESHOLD 0.65 still soft. + routing ≠ permission. Full cards: `mixed-architecture.md`, `faq.md`. +438. **jev-sentinel fail closed** (harshwasan/jev-sentinel): + 8★ HEAD 4ae67df78c95. fail closed never auto-allows. + harshwasan/jev-sentinel ≠ leepokai/jev-guard. + rh-guard owns primary gates. Full cards: `faq.md`. +439. **MCP tool routers namesake** (jackbarunz/jev-tool-router ≠ esinocchi/jev-tool-router): + jackbarunz/jev-tool-router ≠ esinocchi/jev-tool-router. + threshold 0.90 still soft. none_of_the_above. + Full cards: `faq.md`, `mixed-architecture.md`. +440. **76/81 routing bench *theirs*** (esinocchi/jev-tool-router): + 76/81 vs 77/81 *theirs*. 0.419s vs 2.459s *theirs*. + $0.00486 vs $0.03673 *theirs*. does not execute. + not a security boundary. Full cards: `validation.md`. +441. **leanest fail-open** (baronunread/leanest): + fail-open uncertainty means RUN. + classifier.dev default Jev/Laya pluggable. + Full cards: `faq.md`, `mixed-architecture.md`. +442. **jevals estimates not Harbor** (openlayer-ai/jevals): + estimates not Harbor. classifier ≠ authorizer. + openlayer-ai/jevals ≠ dayhaysoos/jevals. + Full cards: `validation.md`, `faq.md`. +443. **MrJev catalog** (MrJev/awesome-jev): + 118 entries catalog ≠ endorsement. + MrJev/awesome-jev ≠ yibie/awesome-jev. + Full cards: `faq.md`. +444. **jev-firewall fail closed** (Koushik890/jev-firewall): + fail closed ask_below 0.7 still soft. Rules can only tighten. + rh-guard owns primary gates. Full cards: `faq.md`. +445. **jev-codex-approval experimental** (CompleteTech-LLC-AI-Research/jev-codex-approval): + experimental native not compiled. + confidence is not a measured probability. + Full cards: `faq.md`. +446. **HF encoder / quanto serving** (rAVEUK / p-yan / Gtrkrsk): + hf:rAVEUK/open-jev-deberta-v3-large encoder class member not Jev replica. + hf:p-yan/laya-quanto serving substrate ≠ calibrated replica. + hf:Gtrkrsk/laya serving substrate ≠ calibrated replica. + p-yan/laya-q8 and q4 Hub HTTP 401. + Full cards: `judgment-class.md`, `faq.md`. +447. **already-catalogued remainder / skip-thin**: + densify is not a second census. SHA move is not a replica. + catalog ≠ endorsement. Full cards: `faq.md`. +448. **skip Archer** (promised_not_landed): + Qwen3 ≠ Archer. Qwen/Qwen3.8-27B ≠ Archer. + Hub archerhume/4rcherhume HTTP 401. + Archer still promised_not_landed. Full cards: `faq.md`. + +Hourly 1542 items 433–448 (`notes.md` §126). Do **not** +re-fold §125 items 417–432 / §124 items 401–416 / §123 items 385–400 +/ §122 protocol / §121 items 369–384 / §120 items 353–368 +/ §119 items 337–352 / §118 items 322–329 / §117 items 330–336 +/ §116 items 309–316 / §115 items 303–308 / §114 items 289–302. +Skip Archer rewrite. +Constrained AR ≠ calibrated Noul; Batch 5.8x *theirs*; +4B new-source 0.790/0.806 *theirs*; 8.2% ≥0.9 on wrong *theirs*; +JEV_THRESHOLD 0.65 still soft; fail closed never auto-allows; +fail-open uncertainty means RUN; classifier ≠ authorizer; +estimates not Harbor; wire-compat ≠ logit-equiv; +SHA move is not a replica; catalog ≠ endorsement. +do not reopen or amend PR #23–#48. +Soft Noul ≠ hard safety. + + Hourly 1441 items 417–432 (`notes.md` §125). Do **not** re-fold §124 items 401–416 / §123 items 385–400 / §122 protocol / §121 items 369–384 / §120 items 353–368 / §119 items 337–352 / §118 items 322–329 @@ -2744,3 +2829,7 @@ Hourly 1340 uniqueness lock: typesafe-sdk 0.7 Pydantic response models; msgspec **Hourly 1441 HIGH (`notes.md` §125).** vLLM NVIDIA + MLX Apple Silicon. Codiv hosted free endpoint. dual /v1/systemone + /v1/chat/completions. chat 501 on MLX. dual serving is not generate. Hosted Codiv ≠ TypeSafe. candidate probabilities are relative not correctness. recommendation is advisory. the server never blocks on its own. LoRA ≠ RLCD replica. pass-min 0.8 still soft. 37.30s → 2.40s at 64 decisions *theirs*. 2B 94.71% 9B 97.54% hard test *theirs*. wire-compat ≠ logit-equiv. SHA move is not a replica. catalog ≠ endorsement. Do not reopen or amend PR #23–#47. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. Hourly 1441 uniqueness lock: vLLM NVIDIA + MLX Apple Silicon; Codiv hosted free endpoint; dual /v1/systemone + /v1/chat/completions; razorback16/openjev densify HEAD cddbd962c88a README SHA a5943415cb92; STE README rewrite; serving-port densify; chat 501 on MLX; dual serving is not generate; Hosted Codiv ≠ TypeSafe; wire-compat ≠ logit-equiv; SHA move is not a replica; Error contract is not a Noul; hr98w/jev-visual 167★ Apple Silicon visual candidate scoring; 37.30s → 2.40s at 64 decisions *theirs*; Breakout 9 bricks 6 returns 2 lives *theirs*; candidate probabilities are relative not correctness; jkudish/jev-mcp 156★ ten MCP tools; recommendation is advisory; the server never blocks on its own; TypeSafe CLERC 5% to 18% *theirs*; jkudish/jev-mcp ≠ burnigtm/jev-mcp; zhengxuyu/litjev off-the-shelf Qwen decision layer; Probabilities are not calibrated by default; Qwen/Qwen3.8-27B ≠ Archer; zhengxuyu/litjev ≠ alexwestco/llm-to-jev; Zefan-Cai/Open-Jev LoRA + scalar head; 2B 94.71% 9B 97.54% hard test *theirs*; 2B OOD 86.02% 9B OOD 91.97% *theirs*; 80,816 training rows; 27B still in progress; LoRA ≠ RLCD replica; Zefan-Cai/Open-Jev ≠ TheoLeeCJ/openjev ≠ razorback16/openjev; cristianoliveira/jeq intelligence you can pipe; pass-min 0.8 still soft; JEQ does not own actions; AndyInQtr/laya-coreai CoreML serving substrate ≠ calibrated replica; AndyInQtr/laya-coreai ≠ mizorewww/laya-coreml; catalog ≠ endorsement; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47; notes.md §125 + + +**Hourly 1542 HIGH (`notes.md` §126).** TypeLLM README densify 3k→12k B. Batch 5.8x *theirs*. Constrained AR ≠ calibrated Noul. kev family densify. 4B new-source 0.790/0.806 *theirs*. 8.2% ≥0.9 on wrong *theirs*. option order can change an answer. fail-closed routing vs fail-open test selection. classifier ≠ authorizer. estimates not Harbor. catalog ≠ endorsement. SHA move is not a replica. Do not reopen or amend PR #23–#48. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. +Hourly 1542 uniqueness lock: TypeLLM/TypeLLM densify HEAD 6a48f9f1e623 README SHA dbdc1f193537; README densify 3k→12k B; thinking=True/False per-field budget; type safety does not guarantee factual accuracy; Batch 5.8x *theirs*; Constrained AR ≠ calibrated Noul; Qwen/Qwen3.8-27B ≠ Archer; jaredpalmer/kev densify HEAD b339f446a0ef README SHA 86b0a19909f3; Kev-0.6B 4B 8B family; 4B new-source 0.790/0.806 *theirs*; 8B new-source 0.796/0.780 *theirs*; Jev hosted 0.857 *theirs*; Questions share the input text but cannot read each other; No Jev outputs were used for training; 8.2% ≥0.9 on wrong *theirs*; option order can change an answer; Qwen3 ≠ Archer; wire-compat ≠ logit-equiv; SHA move is not a replica; TheoOliveira/pi-jev 21★ fail-closed routing; JEV_THRESHOLD 0.65 still soft; routing ≠ permission; harshwasan/jev-sentinel fail closed never auto-allows; harshwasan/jev-sentinel ≠ leepokai/jev-guard; jackbarunz/jev-tool-router ≠ esinocchi/jev-tool-router; threshold 0.90 still soft; 76/81 vs 77/81 *theirs*; 0.419s vs 2.459s *theirs*; $0.00486 vs $0.03673 *theirs*; does not execute; not a security boundary; baronunread/leanest fail-open uncertainty means RUN; classifier.dev default Jev/Laya pluggable; openlayer-ai/jevals ≠ dayhaysoos/jevals; estimates not Harbor; classifier ≠ authorizer; MrJev/awesome-jev 118 entries catalog ≠ endorsement; MrJev/awesome-jev ≠ yibie/awesome-jev; Koushik890/jev-firewall fail closed ask_below 0.7 still soft; CompleteTech-LLC-AI-Research/jev-codex-approval experimental native not compiled; confidence is not a measured probability; rh-guard owns primary gates; hf:rAVEUK/open-jev-deberta-v3-large encoder class member not Jev replica; hf:p-yan/laya-quanto serving substrate ≠ calibrated replica; hf:Gtrkrsk/laya serving substrate ≠ calibrated replica; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48; notes.md §126 diff --git a/.agents/skills/augustus/references/faq.md b/.agents/skills/augustus/references/faq.md index 0f25527..6e4003f 100644 --- a/.agents/skills/augustus/references/faq.md +++ b/.agents/skills/augustus/references/faq.md @@ -3715,3 +3715,11 @@ Hourly 1340 uniqueness lock: typesafe-sdk 0.7 Pydantic response models; msgspec **Hourly 1441 HIGH (`notes.md` §125).** vLLM NVIDIA + MLX Apple Silicon. Codiv hosted free endpoint. dual /v1/systemone + /v1/chat/completions. chat 501 on MLX. dual serving is not generate. Hosted Codiv ≠ TypeSafe. candidate probabilities are relative not correctness. recommendation is advisory. the server never blocks on its own. LoRA ≠ RLCD replica. pass-min 0.8 still soft. 37.30s → 2.40s at 64 decisions *theirs*. 2B 94.71% 9B 97.54% hard test *theirs*. wire-compat ≠ logit-equiv. SHA move is not a replica. catalog ≠ endorsement. Do not reopen or amend PR #23–#47. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. Hourly 1441 uniqueness lock: vLLM NVIDIA + MLX Apple Silicon; Codiv hosted free endpoint; dual /v1/systemone + /v1/chat/completions; razorback16/openjev densify HEAD cddbd962c88a README SHA a5943415cb92; STE README rewrite; serving-port densify; chat 501 on MLX; dual serving is not generate; Hosted Codiv ≠ TypeSafe; wire-compat ≠ logit-equiv; SHA move is not a replica; Error contract is not a Noul; hr98w/jev-visual 167★ Apple Silicon visual candidate scoring; 37.30s → 2.40s at 64 decisions *theirs*; Breakout 9 bricks 6 returns 2 lives *theirs*; candidate probabilities are relative not correctness; jkudish/jev-mcp 156★ ten MCP tools; recommendation is advisory; the server never blocks on its own; TypeSafe CLERC 5% to 18% *theirs*; jkudish/jev-mcp ≠ burnigtm/jev-mcp; zhengxuyu/litjev off-the-shelf Qwen decision layer; Probabilities are not calibrated by default; Qwen/Qwen3.8-27B ≠ Archer; zhengxuyu/litjev ≠ alexwestco/llm-to-jev; Zefan-Cai/Open-Jev LoRA + scalar head; 2B 94.71% 9B 97.54% hard test *theirs*; 2B OOD 86.02% 9B OOD 91.97% *theirs*; 80,816 training rows; 27B still in progress; LoRA ≠ RLCD replica; Zefan-Cai/Open-Jev ≠ TheoLeeCJ/openjev ≠ razorback16/openjev; cristianoliveira/jeq intelligence you can pipe; pass-min 0.8 still soft; JEQ does not own actions; AndyInQtr/laya-coreai CoreML serving substrate ≠ calibrated replica; AndyInQtr/laya-coreai ≠ mizorewww/laya-coreml; catalog ≠ endorsement; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47; notes.md §125 + + +## Treat TypeLLM thinking / 5.8x as a Noul? Treat kev 0.790 as Harbor? Treat 0.65 as a hard gate? + +No. Constrained AR ≠ calibrated Noul. Batch 5.8x *theirs*. type safety does not guarantee factual accuracy. 4B new-source 0.790/0.806 *theirs*. 8.2% ≥0.9 on wrong *theirs*. option order can change an answer. JEV_THRESHOLD 0.65 still soft. routing ≠ permission. fail closed never auto-allows. fail-open uncertainty means RUN. classifier ≠ authorizer. estimates not Harbor. catalog ≠ endorsement. Do not reopen or amend PR #23–#48. `invented_signal: false`. `notes.md` §126. + +**Hourly 1542 HIGH (`notes.md` §126).** TypeLLM README densify 3k→12k B. Batch 5.8x *theirs*. Constrained AR ≠ calibrated Noul. kev family densify. 4B new-source 0.790/0.806 *theirs*. 8.2% ≥0.9 on wrong *theirs*. option order can change an answer. fail-closed routing vs fail-open test selection. classifier ≠ authorizer. estimates not Harbor. catalog ≠ endorsement. SHA move is not a replica. Do not reopen or amend PR #23–#48. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. +Hourly 1542 uniqueness lock: TypeLLM/TypeLLM densify HEAD 6a48f9f1e623 README SHA dbdc1f193537; README densify 3k→12k B; thinking=True/False per-field budget; type safety does not guarantee factual accuracy; Batch 5.8x *theirs*; Constrained AR ≠ calibrated Noul; Qwen/Qwen3.8-27B ≠ Archer; jaredpalmer/kev densify HEAD b339f446a0ef README SHA 86b0a19909f3; Kev-0.6B 4B 8B family; 4B new-source 0.790/0.806 *theirs*; 8B new-source 0.796/0.780 *theirs*; Jev hosted 0.857 *theirs*; Questions share the input text but cannot read each other; No Jev outputs were used for training; 8.2% ≥0.9 on wrong *theirs*; option order can change an answer; Qwen3 ≠ Archer; wire-compat ≠ logit-equiv; SHA move is not a replica; TheoOliveira/pi-jev 21★ fail-closed routing; JEV_THRESHOLD 0.65 still soft; routing ≠ permission; harshwasan/jev-sentinel fail closed never auto-allows; harshwasan/jev-sentinel ≠ leepokai/jev-guard; jackbarunz/jev-tool-router ≠ esinocchi/jev-tool-router; threshold 0.90 still soft; 76/81 vs 77/81 *theirs*; 0.419s vs 2.459s *theirs*; $0.00486 vs $0.03673 *theirs*; does not execute; not a security boundary; baronunread/leanest fail-open uncertainty means RUN; classifier.dev default Jev/Laya pluggable; openlayer-ai/jevals ≠ dayhaysoos/jevals; estimates not Harbor; classifier ≠ authorizer; MrJev/awesome-jev 118 entries catalog ≠ endorsement; MrJev/awesome-jev ≠ yibie/awesome-jev; Koushik890/jev-firewall fail closed ask_below 0.7 still soft; CompleteTech-LLC-AI-Research/jev-codex-approval experimental native not compiled; confidence is not a measured probability; rh-guard owns primary gates; hf:rAVEUK/open-jev-deberta-v3-large encoder class member not Jev replica; hf:p-yan/laya-quanto serving substrate ≠ calibrated replica; hf:Gtrkrsk/laya serving substrate ≠ calibrated replica; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48; notes.md §126 diff --git a/.agents/skills/augustus/references/formal-methods.md b/.agents/skills/augustus/references/formal-methods.md index 8d1c68c..9e189f6 100644 --- a/.agents/skills/augustus/references/formal-methods.md +++ b/.agents/skills/augustus/references/formal-methods.md @@ -1387,3 +1387,7 @@ Hourly 1340 uniqueness lock: typesafe-sdk 0.7 Pydantic response models; msgspec **Hourly 1441 HIGH (`notes.md` §125).** vLLM NVIDIA + MLX Apple Silicon. Codiv hosted free endpoint. dual /v1/systemone + /v1/chat/completions. chat 501 on MLX. dual serving is not generate. Hosted Codiv ≠ TypeSafe. candidate probabilities are relative not correctness. recommendation is advisory. the server never blocks on its own. LoRA ≠ RLCD replica. pass-min 0.8 still soft. 37.30s → 2.40s at 64 decisions *theirs*. 2B 94.71% 9B 97.54% hard test *theirs*. wire-compat ≠ logit-equiv. SHA move is not a replica. catalog ≠ endorsement. Do not reopen or amend PR #23–#47. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. Hourly 1441 uniqueness lock: vLLM NVIDIA + MLX Apple Silicon; Codiv hosted free endpoint; dual /v1/systemone + /v1/chat/completions; razorback16/openjev densify HEAD cddbd962c88a README SHA a5943415cb92; STE README rewrite; serving-port densify; chat 501 on MLX; dual serving is not generate; Hosted Codiv ≠ TypeSafe; wire-compat ≠ logit-equiv; SHA move is not a replica; Error contract is not a Noul; hr98w/jev-visual 167★ Apple Silicon visual candidate scoring; 37.30s → 2.40s at 64 decisions *theirs*; Breakout 9 bricks 6 returns 2 lives *theirs*; candidate probabilities are relative not correctness; jkudish/jev-mcp 156★ ten MCP tools; recommendation is advisory; the server never blocks on its own; TypeSafe CLERC 5% to 18% *theirs*; jkudish/jev-mcp ≠ burnigtm/jev-mcp; zhengxuyu/litjev off-the-shelf Qwen decision layer; Probabilities are not calibrated by default; Qwen/Qwen3.8-27B ≠ Archer; zhengxuyu/litjev ≠ alexwestco/llm-to-jev; Zefan-Cai/Open-Jev LoRA + scalar head; 2B 94.71% 9B 97.54% hard test *theirs*; 2B OOD 86.02% 9B OOD 91.97% *theirs*; 80,816 training rows; 27B still in progress; LoRA ≠ RLCD replica; Zefan-Cai/Open-Jev ≠ TheoLeeCJ/openjev ≠ razorback16/openjev; cristianoliveira/jeq intelligence you can pipe; pass-min 0.8 still soft; JEQ does not own actions; AndyInQtr/laya-coreai CoreML serving substrate ≠ calibrated replica; AndyInQtr/laya-coreai ≠ mizorewww/laya-coreml; catalog ≠ endorsement; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47; notes.md §125 + + +**Hourly 1542 HIGH (`notes.md` §126).** TypeLLM README densify 3k→12k B. Batch 5.8x *theirs*. Constrained AR ≠ calibrated Noul. kev family densify. 4B new-source 0.790/0.806 *theirs*. 8.2% ≥0.9 on wrong *theirs*. option order can change an answer. fail-closed routing vs fail-open test selection. classifier ≠ authorizer. estimates not Harbor. catalog ≠ endorsement. SHA move is not a replica. Do not reopen or amend PR #23–#48. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. +Hourly 1542 uniqueness lock: TypeLLM/TypeLLM densify HEAD 6a48f9f1e623 README SHA dbdc1f193537; README densify 3k→12k B; thinking=True/False per-field budget; type safety does not guarantee factual accuracy; Batch 5.8x *theirs*; Constrained AR ≠ calibrated Noul; Qwen/Qwen3.8-27B ≠ Archer; jaredpalmer/kev densify HEAD b339f446a0ef README SHA 86b0a19909f3; Kev-0.6B 4B 8B family; 4B new-source 0.790/0.806 *theirs*; 8B new-source 0.796/0.780 *theirs*; Jev hosted 0.857 *theirs*; Questions share the input text but cannot read each other; No Jev outputs were used for training; 8.2% ≥0.9 on wrong *theirs*; option order can change an answer; Qwen3 ≠ Archer; wire-compat ≠ logit-equiv; SHA move is not a replica; TheoOliveira/pi-jev 21★ fail-closed routing; JEV_THRESHOLD 0.65 still soft; routing ≠ permission; harshwasan/jev-sentinel fail closed never auto-allows; harshwasan/jev-sentinel ≠ leepokai/jev-guard; jackbarunz/jev-tool-router ≠ esinocchi/jev-tool-router; threshold 0.90 still soft; 76/81 vs 77/81 *theirs*; 0.419s vs 2.459s *theirs*; $0.00486 vs $0.03673 *theirs*; does not execute; not a security boundary; baronunread/leanest fail-open uncertainty means RUN; classifier.dev default Jev/Laya pluggable; openlayer-ai/jevals ≠ dayhaysoos/jevals; estimates not Harbor; classifier ≠ authorizer; MrJev/awesome-jev 118 entries catalog ≠ endorsement; MrJev/awesome-jev ≠ yibie/awesome-jev; Koushik890/jev-firewall fail closed ask_below 0.7 still soft; CompleteTech-LLC-AI-Research/jev-codex-approval experimental native not compiled; confidence is not a measured probability; rh-guard owns primary gates; hf:rAVEUK/open-jev-deberta-v3-large encoder class member not Jev replica; hf:p-yan/laya-quanto serving substrate ≠ calibrated replica; hf:Gtrkrsk/laya serving substrate ≠ calibrated replica; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48; notes.md §126 diff --git a/.agents/skills/augustus/references/formal-semi-formal.md b/.agents/skills/augustus/references/formal-semi-formal.md index c1616e3..600f05c 100644 --- a/.agents/skills/augustus/references/formal-semi-formal.md +++ b/.agents/skills/augustus/references/formal-semi-formal.md @@ -100,3 +100,7 @@ Hourly 1340 uniqueness lock: typesafe-sdk 0.7 Pydantic response models; msgspec **Hourly 1441 HIGH (`notes.md` §125).** vLLM NVIDIA + MLX Apple Silicon. Codiv hosted free endpoint. dual /v1/systemone + /v1/chat/completions. chat 501 on MLX. dual serving is not generate. Hosted Codiv ≠ TypeSafe. candidate probabilities are relative not correctness. recommendation is advisory. the server never blocks on its own. LoRA ≠ RLCD replica. pass-min 0.8 still soft. 37.30s → 2.40s at 64 decisions *theirs*. 2B 94.71% 9B 97.54% hard test *theirs*. wire-compat ≠ logit-equiv. SHA move is not a replica. catalog ≠ endorsement. Do not reopen or amend PR #23–#47. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. Hourly 1441 uniqueness lock: vLLM NVIDIA + MLX Apple Silicon; Codiv hosted free endpoint; dual /v1/systemone + /v1/chat/completions; razorback16/openjev densify HEAD cddbd962c88a README SHA a5943415cb92; STE README rewrite; serving-port densify; chat 501 on MLX; dual serving is not generate; Hosted Codiv ≠ TypeSafe; wire-compat ≠ logit-equiv; SHA move is not a replica; Error contract is not a Noul; hr98w/jev-visual 167★ Apple Silicon visual candidate scoring; 37.30s → 2.40s at 64 decisions *theirs*; Breakout 9 bricks 6 returns 2 lives *theirs*; candidate probabilities are relative not correctness; jkudish/jev-mcp 156★ ten MCP tools; recommendation is advisory; the server never blocks on its own; TypeSafe CLERC 5% to 18% *theirs*; jkudish/jev-mcp ≠ burnigtm/jev-mcp; zhengxuyu/litjev off-the-shelf Qwen decision layer; Probabilities are not calibrated by default; Qwen/Qwen3.8-27B ≠ Archer; zhengxuyu/litjev ≠ alexwestco/llm-to-jev; Zefan-Cai/Open-Jev LoRA + scalar head; 2B 94.71% 9B 97.54% hard test *theirs*; 2B OOD 86.02% 9B OOD 91.97% *theirs*; 80,816 training rows; 27B still in progress; LoRA ≠ RLCD replica; Zefan-Cai/Open-Jev ≠ TheoLeeCJ/openjev ≠ razorback16/openjev; cristianoliveira/jeq intelligence you can pipe; pass-min 0.8 still soft; JEQ does not own actions; AndyInQtr/laya-coreai CoreML serving substrate ≠ calibrated replica; AndyInQtr/laya-coreai ≠ mizorewww/laya-coreml; catalog ≠ endorsement; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47; notes.md §125 + + +**Hourly 1542 HIGH (`notes.md` §126).** TypeLLM README densify 3k→12k B. Batch 5.8x *theirs*. Constrained AR ≠ calibrated Noul. kev family densify. 4B new-source 0.790/0.806 *theirs*. 8.2% ≥0.9 on wrong *theirs*. option order can change an answer. fail-closed routing vs fail-open test selection. classifier ≠ authorizer. estimates not Harbor. catalog ≠ endorsement. SHA move is not a replica. Do not reopen or amend PR #23–#48. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. +Hourly 1542 uniqueness lock: TypeLLM/TypeLLM densify HEAD 6a48f9f1e623 README SHA dbdc1f193537; README densify 3k→12k B; thinking=True/False per-field budget; type safety does not guarantee factual accuracy; Batch 5.8x *theirs*; Constrained AR ≠ calibrated Noul; Qwen/Qwen3.8-27B ≠ Archer; jaredpalmer/kev densify HEAD b339f446a0ef README SHA 86b0a19909f3; Kev-0.6B 4B 8B family; 4B new-source 0.790/0.806 *theirs*; 8B new-source 0.796/0.780 *theirs*; Jev hosted 0.857 *theirs*; Questions share the input text but cannot read each other; No Jev outputs were used for training; 8.2% ≥0.9 on wrong *theirs*; option order can change an answer; Qwen3 ≠ Archer; wire-compat ≠ logit-equiv; SHA move is not a replica; TheoOliveira/pi-jev 21★ fail-closed routing; JEV_THRESHOLD 0.65 still soft; routing ≠ permission; harshwasan/jev-sentinel fail closed never auto-allows; harshwasan/jev-sentinel ≠ leepokai/jev-guard; jackbarunz/jev-tool-router ≠ esinocchi/jev-tool-router; threshold 0.90 still soft; 76/81 vs 77/81 *theirs*; 0.419s vs 2.459s *theirs*; $0.00486 vs $0.03673 *theirs*; does not execute; not a security boundary; baronunread/leanest fail-open uncertainty means RUN; classifier.dev default Jev/Laya pluggable; openlayer-ai/jevals ≠ dayhaysoos/jevals; estimates not Harbor; classifier ≠ authorizer; MrJev/awesome-jev 118 entries catalog ≠ endorsement; MrJev/awesome-jev ≠ yibie/awesome-jev; Koushik890/jev-firewall fail closed ask_below 0.7 still soft; CompleteTech-LLC-AI-Research/jev-codex-approval experimental native not compiled; confidence is not a measured probability; rh-guard owns primary gates; hf:rAVEUK/open-jev-deberta-v3-large encoder class member not Jev replica; hf:p-yan/laya-quanto serving substrate ≠ calibrated replica; hf:Gtrkrsk/laya serving substrate ≠ calibrated replica; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48; notes.md §126 diff --git a/.agents/skills/augustus/references/judgment-class.md b/.agents/skills/augustus/references/judgment-class.md index b16c356..45641c5 100644 --- a/.agents/skills/augustus/references/judgment-class.md +++ b/.agents/skills/augustus/references/judgment-class.md @@ -675,9 +675,11 @@ is the generator, not a sixth surface. | **Tiny LoRA distill** (jev-gate-student-b) | Teacher-copy. P(relevant) from yes/no logits. Held-out n=60 vs vanilla 0.5B; 148,160-row corpus. HF card **unchanged** ~17:48 vs §33 (MAE 0.187 / Pearson 0.791 / 90%; ~59 ms RTX 3060; fail-open) | Memory-gating / context sieve; **fail-open** on errors | Qwen2.5-0.5B LoRA; ~59 ms RTX 3060 | Local, apache-2.0 | Text | Binary relevance | | **Domain LoRA specialist** (Domain-jev-maker) | Independent CLINC gold, soft targets, pointer readout. **Not** a Jev teacher-copy. Calibration gap is the product: KL 0.168 vs hosted 0.580 banking; few-shot hosted matches argmax (McNemar n.s.) | Threshold / deferral / EU that *reads* p; skip when only argmax | ~0.5 s / request on 8 GB GPU *theirs*; 1.5B LoRA | Self-host; MIT | Text | Domain K + abstain; one forward pass | | **Nimble** (open LoRA recipe, not a distill) | Hard synthetic labels. They say temperature was not tuned to correctness rates. 324-row agreement is their receipt, not an ECE (`notes.md` §35) | Not a gather primitive | Their latency table, not re-run | Self-host the adapter. Model card Apache-2.0; repo license absent | Text only | Enum ≤26; 2,048 tokens | -| **kev** (Qwen2.5-0.5B LoRA + pointer; Apache-2.0) | Public gold, CE. Held-out ECE 0.065 (0.031 after T=1.47); acc 0.799 on 1,350 ID questions. Isolation exact. **Not** a Jev teacher-copy (`notes.md` §45). **Family delta (`notes.md` §98):** Archer-arch fidelity; kev family OOD 0.76–0.77 vs Jev 0.86; block-causal isolation; pointer/readout CE-trained; `/v1/systemone` drop-in; replica honesty | Laptop-local System One drop-in for development/eval; independent questions, one prefill. Family bake-off candidate, not a Jev substitute | ~160 ms / 6 questions; ~1h45m train on M5; 38 MB adapter. Family: kev-4b ~1 s / kev-8b ~2 s bf16 *theirs* | Self-host; official `typesafe-sdk` with `base_url` | Text. Not multimodal. 0.5B knowledge; 4B/8B OOD still a gap | noul / choice 2–255 / score | +| **kev** (Qwen3 0.6B/4B/8B family + pointer; Apache-2.0) | Public gold, CE. Held-out ECE 0.065 (0.031 after T=1.47); acc 0.799 on 1,350 ID questions. Isolation exact. **Not** a Jev teacher-copy (`notes.md` §45). **Family delta (`notes.md` §98):** Archer-arch fidelity; kev family OOD 0.76–0.77 vs Jev 0.86; block-causal isolation; pointer/readout CE-trained; `/v1/systemone` drop-in; replica honesty | Laptop-local System One drop-in for development/eval; independent questions, one prefill. Family bake-off candidate, not a Jev substitute | ~160 ms / 6 questions; ~1h45m train on M5; 38 MB adapter. Family: kev-4b ~1 s / kev-8b ~2 s bf16 *theirs* | Self-host; official `typesafe-sdk` with `base_url` | Text. Not multimodal. 0.5B knowledge; 4B/8B OOD still a gap | noul / choice 2–255 / score | | **Diffusion structured reads** (djev-spark) | Interface claim only. **Hypothesis** it beats a decision head on your labels (`notes.md` §36) | Optional sequential chunks, text-only | Their GX10 tables, not a class benchmark | DGX Spark container. Do not copy the route | Images are an extension; think and sequential reject images | README criteria, not copied here | + **1542 densify:** Constrained AR ≠ calibrated Noul (TypeLLM Batch 5.8x *theirs*). kev 4B new-source 0.790/0.806 *theirs*; 8.2% ≥0.9 on wrong *theirs*; option order can change an answer (`notes.md` §126). + **Three open paths** (not three species, not extra when-to-use rows): encoder open-jev (DeBERTa, public gold); AR constrained decode (TypeAR Python/SGLang, pcdServer native GGUF; decision-token LoRA on that graph); @@ -1334,6 +1336,8 @@ Soft Noul ≠ hard safety: 0.76 / 0.77 / 0.86 / ECE ~0.1 enough to ship as Jev” is theater. Archer still **NOT landed**. +**Family delta (`notes.md` §126 hourly 1542).** HEAD `b339f446a0ef` README SHA `86b0a19909f3`. Kev-0.6B 4B 8B family. 4B new-source 0.790/0.806 *theirs*. 8B new-source 0.796/0.780 *theirs*. Jev hosted 0.857 *theirs*. Questions share the input text but cannot read each other. No Jev outputs were used for training. 8.2% ≥0.9 on wrong *theirs*. option order can change an answer. isolation ≠ option-order immunity. Qwen3 ≠ Archer. Do not copy `uv run`. Soft Noul ≠ hard safety. + ### blackwood-rlcd — open multimodal RLCD (not Archer, not CLIP) [`BlackwoodAI/blackwood-rlcd`](https://huggingface.co/BlackwoodAI/blackwood-rlcd) @@ -1424,6 +1428,10 @@ rotated-option tests — not a TypeAR how-to and not a Noul. Option order and an irrelevant extra option are properties to test, not a proof (`formal-methods.md`). +### Since last look (2026-09-20T21 hourly 1542) — TypeLLM/TypeLLM + +Live name [TypeLLM/TypeLLM](https://github.com/TypeLLM/TypeLLM). HEAD `6a48f9f1e623` README SHA `dbdc1f193537`. README densify 3k→12k B. thinking=True/False per-field budget. type safety does not guarantee factual accuracy. Sequential 9.35 s vs batch 1.61 s K=16 Boolean fields, 5.8x *theirs*. Constrained AR ≠ calibrated Noul. Qwen/Qwen3.8-27B ≠ Archer. Do not copy SGLang flags. `notes.md` §126. + ## Decision-design extras for class choice When the request is "Jev vs GLiNER vs GLiClass vs CLIP vs a cross-encoder": @@ -1486,3 +1494,7 @@ Hourly 1340 uniqueness lock: typesafe-sdk 0.7 Pydantic response models; msgspec **Hourly 1441 HIGH (`notes.md` §125).** vLLM NVIDIA + MLX Apple Silicon. Codiv hosted free endpoint. dual /v1/systemone + /v1/chat/completions. chat 501 on MLX. dual serving is not generate. Hosted Codiv ≠ TypeSafe. candidate probabilities are relative not correctness. recommendation is advisory. the server never blocks on its own. LoRA ≠ RLCD replica. pass-min 0.8 still soft. 37.30s → 2.40s at 64 decisions *theirs*. 2B 94.71% 9B 97.54% hard test *theirs*. wire-compat ≠ logit-equiv. SHA move is not a replica. catalog ≠ endorsement. Do not reopen or amend PR #23–#47. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. Hourly 1441 uniqueness lock: vLLM NVIDIA + MLX Apple Silicon; Codiv hosted free endpoint; dual /v1/systemone + /v1/chat/completions; razorback16/openjev densify HEAD cddbd962c88a README SHA a5943415cb92; STE README rewrite; serving-port densify; chat 501 on MLX; dual serving is not generate; Hosted Codiv ≠ TypeSafe; wire-compat ≠ logit-equiv; SHA move is not a replica; Error contract is not a Noul; hr98w/jev-visual 167★ Apple Silicon visual candidate scoring; 37.30s → 2.40s at 64 decisions *theirs*; Breakout 9 bricks 6 returns 2 lives *theirs*; candidate probabilities are relative not correctness; jkudish/jev-mcp 156★ ten MCP tools; recommendation is advisory; the server never blocks on its own; TypeSafe CLERC 5% to 18% *theirs*; jkudish/jev-mcp ≠ burnigtm/jev-mcp; zhengxuyu/litjev off-the-shelf Qwen decision layer; Probabilities are not calibrated by default; Qwen/Qwen3.8-27B ≠ Archer; zhengxuyu/litjev ≠ alexwestco/llm-to-jev; Zefan-Cai/Open-Jev LoRA + scalar head; 2B 94.71% 9B 97.54% hard test *theirs*; 2B OOD 86.02% 9B OOD 91.97% *theirs*; 80,816 training rows; 27B still in progress; LoRA ≠ RLCD replica; Zefan-Cai/Open-Jev ≠ TheoLeeCJ/openjev ≠ razorback16/openjev; cristianoliveira/jeq intelligence you can pipe; pass-min 0.8 still soft; JEQ does not own actions; AndyInQtr/laya-coreai CoreML serving substrate ≠ calibrated replica; AndyInQtr/laya-coreai ≠ mizorewww/laya-coreml; catalog ≠ endorsement; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47; notes.md §125 + + +**Hourly 1542 HIGH (`notes.md` §126).** TypeLLM README densify 3k→12k B. Batch 5.8x *theirs*. Constrained AR ≠ calibrated Noul. kev family densify. 4B new-source 0.790/0.806 *theirs*. 8.2% ≥0.9 on wrong *theirs*. option order can change an answer. fail-closed routing vs fail-open test selection. classifier ≠ authorizer. estimates not Harbor. catalog ≠ endorsement. SHA move is not a replica. Do not reopen or amend PR #23–#48. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. +Hourly 1542 uniqueness lock: TypeLLM/TypeLLM densify HEAD 6a48f9f1e623 README SHA dbdc1f193537; README densify 3k→12k B; thinking=True/False per-field budget; type safety does not guarantee factual accuracy; Batch 5.8x *theirs*; Constrained AR ≠ calibrated Noul; Qwen/Qwen3.8-27B ≠ Archer; jaredpalmer/kev densify HEAD b339f446a0ef README SHA 86b0a19909f3; Kev-0.6B 4B 8B family; 4B new-source 0.790/0.806 *theirs*; 8B new-source 0.796/0.780 *theirs*; Jev hosted 0.857 *theirs*; Questions share the input text but cannot read each other; No Jev outputs were used for training; 8.2% ≥0.9 on wrong *theirs*; option order can change an answer; Qwen3 ≠ Archer; wire-compat ≠ logit-equiv; SHA move is not a replica; TheoOliveira/pi-jev 21★ fail-closed routing; JEV_THRESHOLD 0.65 still soft; routing ≠ permission; harshwasan/jev-sentinel fail closed never auto-allows; harshwasan/jev-sentinel ≠ leepokai/jev-guard; jackbarunz/jev-tool-router ≠ esinocchi/jev-tool-router; threshold 0.90 still soft; 76/81 vs 77/81 *theirs*; 0.419s vs 2.459s *theirs*; $0.00486 vs $0.03673 *theirs*; does not execute; not a security boundary; baronunread/leanest fail-open uncertainty means RUN; classifier.dev default Jev/Laya pluggable; openlayer-ai/jevals ≠ dayhaysoos/jevals; estimates not Harbor; classifier ≠ authorizer; MrJev/awesome-jev 118 entries catalog ≠ endorsement; MrJev/awesome-jev ≠ yibie/awesome-jev; Koushik890/jev-firewall fail closed ask_below 0.7 still soft; CompleteTech-LLC-AI-Research/jev-codex-approval experimental native not compiled; confidence is not a measured probability; rh-guard owns primary gates; hf:rAVEUK/open-jev-deberta-v3-large encoder class member not Jev replica; hf:p-yan/laya-quanto serving substrate ≠ calibrated replica; hf:Gtrkrsk/laya serving substrate ≠ calibrated replica; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48; notes.md §126 diff --git a/.agents/skills/augustus/references/mappings.md b/.agents/skills/augustus/references/mappings.md index 080952e..59e7b16 100644 --- a/.agents/skills/augustus/references/mappings.md +++ b/.agents/skills/augustus/references/mappings.md @@ -2492,3 +2492,7 @@ Hourly 1340 uniqueness lock: typesafe-sdk 0.7 Pydantic response models; msgspec **Hourly 1441 HIGH (`notes.md` §125).** vLLM NVIDIA + MLX Apple Silicon. Codiv hosted free endpoint. dual /v1/systemone + /v1/chat/completions. chat 501 on MLX. dual serving is not generate. Hosted Codiv ≠ TypeSafe. candidate probabilities are relative not correctness. recommendation is advisory. the server never blocks on its own. LoRA ≠ RLCD replica. pass-min 0.8 still soft. 37.30s → 2.40s at 64 decisions *theirs*. 2B 94.71% 9B 97.54% hard test *theirs*. wire-compat ≠ logit-equiv. SHA move is not a replica. catalog ≠ endorsement. Do not reopen or amend PR #23–#47. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. Hourly 1441 uniqueness lock: vLLM NVIDIA + MLX Apple Silicon; Codiv hosted free endpoint; dual /v1/systemone + /v1/chat/completions; razorback16/openjev densify HEAD cddbd962c88a README SHA a5943415cb92; STE README rewrite; serving-port densify; chat 501 on MLX; dual serving is not generate; Hosted Codiv ≠ TypeSafe; wire-compat ≠ logit-equiv; SHA move is not a replica; Error contract is not a Noul; hr98w/jev-visual 167★ Apple Silicon visual candidate scoring; 37.30s → 2.40s at 64 decisions *theirs*; Breakout 9 bricks 6 returns 2 lives *theirs*; candidate probabilities are relative not correctness; jkudish/jev-mcp 156★ ten MCP tools; recommendation is advisory; the server never blocks on its own; TypeSafe CLERC 5% to 18% *theirs*; jkudish/jev-mcp ≠ burnigtm/jev-mcp; zhengxuyu/litjev off-the-shelf Qwen decision layer; Probabilities are not calibrated by default; Qwen/Qwen3.8-27B ≠ Archer; zhengxuyu/litjev ≠ alexwestco/llm-to-jev; Zefan-Cai/Open-Jev LoRA + scalar head; 2B 94.71% 9B 97.54% hard test *theirs*; 2B OOD 86.02% 9B OOD 91.97% *theirs*; 80,816 training rows; 27B still in progress; LoRA ≠ RLCD replica; Zefan-Cai/Open-Jev ≠ TheoLeeCJ/openjev ≠ razorback16/openjev; cristianoliveira/jeq intelligence you can pipe; pass-min 0.8 still soft; JEQ does not own actions; AndyInQtr/laya-coreai CoreML serving substrate ≠ calibrated replica; AndyInQtr/laya-coreai ≠ mizorewww/laya-coreml; catalog ≠ endorsement; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47; notes.md §125 + + +**Hourly 1542 HIGH (`notes.md` §126).** TypeLLM README densify 3k→12k B. Batch 5.8x *theirs*. Constrained AR ≠ calibrated Noul. kev family densify. 4B new-source 0.790/0.806 *theirs*. 8.2% ≥0.9 on wrong *theirs*. option order can change an answer. fail-closed routing vs fail-open test selection. classifier ≠ authorizer. estimates not Harbor. catalog ≠ endorsement. SHA move is not a replica. Do not reopen or amend PR #23–#48. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. +Hourly 1542 uniqueness lock: TypeLLM/TypeLLM densify HEAD 6a48f9f1e623 README SHA dbdc1f193537; README densify 3k→12k B; thinking=True/False per-field budget; type safety does not guarantee factual accuracy; Batch 5.8x *theirs*; Constrained AR ≠ calibrated Noul; Qwen/Qwen3.8-27B ≠ Archer; jaredpalmer/kev densify HEAD b339f446a0ef README SHA 86b0a19909f3; Kev-0.6B 4B 8B family; 4B new-source 0.790/0.806 *theirs*; 8B new-source 0.796/0.780 *theirs*; Jev hosted 0.857 *theirs*; Questions share the input text but cannot read each other; No Jev outputs were used for training; 8.2% ≥0.9 on wrong *theirs*; option order can change an answer; Qwen3 ≠ Archer; wire-compat ≠ logit-equiv; SHA move is not a replica; TheoOliveira/pi-jev 21★ fail-closed routing; JEV_THRESHOLD 0.65 still soft; routing ≠ permission; harshwasan/jev-sentinel fail closed never auto-allows; harshwasan/jev-sentinel ≠ leepokai/jev-guard; jackbarunz/jev-tool-router ≠ esinocchi/jev-tool-router; threshold 0.90 still soft; 76/81 vs 77/81 *theirs*; 0.419s vs 2.459s *theirs*; $0.00486 vs $0.03673 *theirs*; does not execute; not a security boundary; baronunread/leanest fail-open uncertainty means RUN; classifier.dev default Jev/Laya pluggable; openlayer-ai/jevals ≠ dayhaysoos/jevals; estimates not Harbor; classifier ≠ authorizer; MrJev/awesome-jev 118 entries catalog ≠ endorsement; MrJev/awesome-jev ≠ yibie/awesome-jev; Koushik890/jev-firewall fail closed ask_below 0.7 still soft; CompleteTech-LLC-AI-Research/jev-codex-approval experimental native not compiled; confidence is not a measured probability; rh-guard owns primary gates; hf:rAVEUK/open-jev-deberta-v3-large encoder class member not Jev replica; hf:p-yan/laya-quanto serving substrate ≠ calibrated replica; hf:Gtrkrsk/laya serving substrate ≠ calibrated replica; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48; notes.md §126 diff --git a/.agents/skills/augustus/references/mental-models.md b/.agents/skills/augustus/references/mental-models.md index 7efb63a..69ce039 100644 --- a/.agents/skills/augustus/references/mental-models.md +++ b/.agents/skills/augustus/references/mental-models.md @@ -3021,6 +3021,18 @@ Do not copy keys. **Hourly 1248 HIGH (`notes.md` §123).** decide is not generate. tryDecide returns typed calibrated judgments not a token stream. GLiNER/GLiClass ports are class members not Jev replicas. 93.5% *theirs* not Harbor. 74.9 *theirs* not Harbor. 8.7x *theirs* not Harbor. wire-compat ≠ logit-equiv. SHA move is not a replica. catalog ≠ endorsement. Do not reopen or amend PR #23–#45. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. Hourly 1248 uniqueness lock: decide is not generate; tryDecide returns typed calibrated judgments not a token stream; juspay/neurolink 133★ MIT HEAD 268b0fe83130 README SHA e709cadfa6b6 tag v12.19.0; GLiNER/GLiClass ports are class members not Jev replicas; MacPaw/Gliner2Swift ≠ Knowledgator/GLiClass.c ≠ fbilhaut/gliclass-rs ≠ Knowledgator/GLiClass.js ≠ gravitee-io/GLiNER4j ≠ apiplant/gliner-rs ≠ codesoda/gliner2-rs; 8.7x faster 4.4x fewer prompts *theirs*; 153 was a reporting error; corrected 156-case 9.8x faster 4.2x fewer prompts *theirs*; independent v0.2.1 1.24x vs Mini *theirs*; Approvals only; anpicasso/hermes-jev-approvals ≠ hermes-switchyard; scx-router GLiClass ranks candidate LLMs in one non-generative pass; typesafeai-dotnet-sdk Not affiliated with TypeSafe AI; hyperspaceai/jevcache ≠ kushals256/jevcache; ST-jeved measures each reply; 400 plain-text for unaskable question; razorback16/openjev:0.2.1 Docker densify HEAD 794a81b87131; wire-compat ≠ logit-equiv; Option-Marker joint attention 93.5% macro *theirs*; 93.6% micro *theirs*; n=78; T = 1.0367 vs T = 1.1692 two temperatures; guaranteeing is soundness theater; wfzyx/von densify HEAD bed7e7337791; Benchmark Heaven leaderboard #2 74.9 *theirs*; NLL calibration assets; 77.10% still §71 claim-audit; do not re-fold as a beat; Heman10x-NGU/openJev-verdict-2.0 densify HEAD bff28567cff4; kev-family weight tarballs; PLAN_Qwen35 proposal for review; deadline 0.53→0.82 at 9B *theirs*; Qwen3.5-9B ≠ Archer; isolation would fail by construction on DeltaNet; jaredpalmer/kev densify; JevBench v1.2.2 jeff 66.9 (#9) jev 75.3 (#2) *theirs*; logan-markewich/jeff densify HEAD 34b32f99a727; thinking=True/False per-field budget; type safety does not guarantee factual accuracy; TypeLLM/TypeLLM densify HEAD c4b03ba9e792; us/jev-local stub until hf; Eran-BA/Jev_from_GLiNER2 spec ≠ replica; lsu-ub-uu/systemone ≠ TypeSafe System One; Layan/Laya HF spaces name-match; catalog ≠ endorsement; decide ≠ generate ≠ stream; 93.5% *theirs* not Harbor; 74.9 *theirs* not Harbor; 8.7x *theirs* not Harbor; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45; notes.md §123 +## Apply 1542 (`notes.md` §126) + +Constrained AR ≠ calibrated Noul. Batch 5.8x *theirs*. +thinking=True/False per-field budget. type safety does not guarantee factual accuracy. +Kev-0.6B 4B 8B family. 4B new-source 0.790/0.806 *theirs*. +8.2% ≥0.9 on wrong *theirs*. option order can change an answer. +Questions share the input text but cannot read each other. +JEV_THRESHOLD 0.65 still soft. routing ≠ permission. +fail closed never auto-allows. fail-open uncertainty means RUN. +classifier ≠ authorizer. estimates not Harbor. +catalog ≠ endorsement. SHA move is not a replica. Do not copy keys. + ## Apply 1441 (`notes.md` §125) dual serving is not generate. Hosted Codiv ≠ TypeSafe. @@ -3056,3 +3068,7 @@ Hourly 1340 uniqueness lock: typesafe-sdk 0.7 Pydantic response models; msgspec **Hourly 1441 HIGH (`notes.md` §125).** vLLM NVIDIA + MLX Apple Silicon. Codiv hosted free endpoint. dual /v1/systemone + /v1/chat/completions. chat 501 on MLX. dual serving is not generate. Hosted Codiv ≠ TypeSafe. candidate probabilities are relative not correctness. recommendation is advisory. the server never blocks on its own. LoRA ≠ RLCD replica. pass-min 0.8 still soft. 37.30s → 2.40s at 64 decisions *theirs*. 2B 94.71% 9B 97.54% hard test *theirs*. wire-compat ≠ logit-equiv. SHA move is not a replica. catalog ≠ endorsement. Do not reopen or amend PR #23–#47. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. Hourly 1441 uniqueness lock: vLLM NVIDIA + MLX Apple Silicon; Codiv hosted free endpoint; dual /v1/systemone + /v1/chat/completions; razorback16/openjev densify HEAD cddbd962c88a README SHA a5943415cb92; STE README rewrite; serving-port densify; chat 501 on MLX; dual serving is not generate; Hosted Codiv ≠ TypeSafe; wire-compat ≠ logit-equiv; SHA move is not a replica; Error contract is not a Noul; hr98w/jev-visual 167★ Apple Silicon visual candidate scoring; 37.30s → 2.40s at 64 decisions *theirs*; Breakout 9 bricks 6 returns 2 lives *theirs*; candidate probabilities are relative not correctness; jkudish/jev-mcp 156★ ten MCP tools; recommendation is advisory; the server never blocks on its own; TypeSafe CLERC 5% to 18% *theirs*; jkudish/jev-mcp ≠ burnigtm/jev-mcp; zhengxuyu/litjev off-the-shelf Qwen decision layer; Probabilities are not calibrated by default; Qwen/Qwen3.8-27B ≠ Archer; zhengxuyu/litjev ≠ alexwestco/llm-to-jev; Zefan-Cai/Open-Jev LoRA + scalar head; 2B 94.71% 9B 97.54% hard test *theirs*; 2B OOD 86.02% 9B OOD 91.97% *theirs*; 80,816 training rows; 27B still in progress; LoRA ≠ RLCD replica; Zefan-Cai/Open-Jev ≠ TheoLeeCJ/openjev ≠ razorback16/openjev; cristianoliveira/jeq intelligence you can pipe; pass-min 0.8 still soft; JEQ does not own actions; AndyInQtr/laya-coreai CoreML serving substrate ≠ calibrated replica; AndyInQtr/laya-coreai ≠ mizorewww/laya-coreml; catalog ≠ endorsement; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47; notes.md §125 + + +**Hourly 1542 HIGH (`notes.md` §126).** TypeLLM README densify 3k→12k B. Batch 5.8x *theirs*. Constrained AR ≠ calibrated Noul. kev family densify. 4B new-source 0.790/0.806 *theirs*. 8.2% ≥0.9 on wrong *theirs*. option order can change an answer. fail-closed routing vs fail-open test selection. classifier ≠ authorizer. estimates not Harbor. catalog ≠ endorsement. SHA move is not a replica. Do not reopen or amend PR #23–#48. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. +Hourly 1542 uniqueness lock: TypeLLM/TypeLLM densify HEAD 6a48f9f1e623 README SHA dbdc1f193537; README densify 3k→12k B; thinking=True/False per-field budget; type safety does not guarantee factual accuracy; Batch 5.8x *theirs*; Constrained AR ≠ calibrated Noul; Qwen/Qwen3.8-27B ≠ Archer; jaredpalmer/kev densify HEAD b339f446a0ef README SHA 86b0a19909f3; Kev-0.6B 4B 8B family; 4B new-source 0.790/0.806 *theirs*; 8B new-source 0.796/0.780 *theirs*; Jev hosted 0.857 *theirs*; Questions share the input text but cannot read each other; No Jev outputs were used for training; 8.2% ≥0.9 on wrong *theirs*; option order can change an answer; Qwen3 ≠ Archer; wire-compat ≠ logit-equiv; SHA move is not a replica; TheoOliveira/pi-jev 21★ fail-closed routing; JEV_THRESHOLD 0.65 still soft; routing ≠ permission; harshwasan/jev-sentinel fail closed never auto-allows; harshwasan/jev-sentinel ≠ leepokai/jev-guard; jackbarunz/jev-tool-router ≠ esinocchi/jev-tool-router; threshold 0.90 still soft; 76/81 vs 77/81 *theirs*; 0.419s vs 2.459s *theirs*; $0.00486 vs $0.03673 *theirs*; does not execute; not a security boundary; baronunread/leanest fail-open uncertainty means RUN; classifier.dev default Jev/Laya pluggable; openlayer-ai/jevals ≠ dayhaysoos/jevals; estimates not Harbor; classifier ≠ authorizer; MrJev/awesome-jev 118 entries catalog ≠ endorsement; MrJev/awesome-jev ≠ yibie/awesome-jev; Koushik890/jev-firewall fail closed ask_below 0.7 still soft; CompleteTech-LLC-AI-Research/jev-codex-approval experimental native not compiled; confidence is not a measured probability; rh-guard owns primary gates; hf:rAVEUK/open-jev-deberta-v3-large encoder class member not Jev replica; hf:p-yan/laya-quanto serving substrate ≠ calibrated replica; hf:Gtrkrsk/laya serving substrate ≠ calibrated replica; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48; notes.md §126 diff --git a/.agents/skills/augustus/references/methods-catalog.md b/.agents/skills/augustus/references/methods-catalog.md index 42259f6..ad64e67 100644 --- a/.agents/skills/augustus/references/methods-catalog.md +++ b/.agents/skills/augustus/references/methods-catalog.md @@ -294,3 +294,7 @@ Hourly 1340 uniqueness lock: typesafe-sdk 0.7 Pydantic response models; msgspec **Hourly 1441 HIGH (`notes.md` §125).** vLLM NVIDIA + MLX Apple Silicon. Codiv hosted free endpoint. dual /v1/systemone + /v1/chat/completions. chat 501 on MLX. dual serving is not generate. Hosted Codiv ≠ TypeSafe. candidate probabilities are relative not correctness. recommendation is advisory. the server never blocks on its own. LoRA ≠ RLCD replica. pass-min 0.8 still soft. 37.30s → 2.40s at 64 decisions *theirs*. 2B 94.71% 9B 97.54% hard test *theirs*. wire-compat ≠ logit-equiv. SHA move is not a replica. catalog ≠ endorsement. Do not reopen or amend PR #23–#47. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. Hourly 1441 uniqueness lock: vLLM NVIDIA + MLX Apple Silicon; Codiv hosted free endpoint; dual /v1/systemone + /v1/chat/completions; razorback16/openjev densify HEAD cddbd962c88a README SHA a5943415cb92; STE README rewrite; serving-port densify; chat 501 on MLX; dual serving is not generate; Hosted Codiv ≠ TypeSafe; wire-compat ≠ logit-equiv; SHA move is not a replica; Error contract is not a Noul; hr98w/jev-visual 167★ Apple Silicon visual candidate scoring; 37.30s → 2.40s at 64 decisions *theirs*; Breakout 9 bricks 6 returns 2 lives *theirs*; candidate probabilities are relative not correctness; jkudish/jev-mcp 156★ ten MCP tools; recommendation is advisory; the server never blocks on its own; TypeSafe CLERC 5% to 18% *theirs*; jkudish/jev-mcp ≠ burnigtm/jev-mcp; zhengxuyu/litjev off-the-shelf Qwen decision layer; Probabilities are not calibrated by default; Qwen/Qwen3.8-27B ≠ Archer; zhengxuyu/litjev ≠ alexwestco/llm-to-jev; Zefan-Cai/Open-Jev LoRA + scalar head; 2B 94.71% 9B 97.54% hard test *theirs*; 2B OOD 86.02% 9B OOD 91.97% *theirs*; 80,816 training rows; 27B still in progress; LoRA ≠ RLCD replica; Zefan-Cai/Open-Jev ≠ TheoLeeCJ/openjev ≠ razorback16/openjev; cristianoliveira/jeq intelligence you can pipe; pass-min 0.8 still soft; JEQ does not own actions; AndyInQtr/laya-coreai CoreML serving substrate ≠ calibrated replica; AndyInQtr/laya-coreai ≠ mizorewww/laya-coreml; catalog ≠ endorsement; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47; notes.md §125 + + +**Hourly 1542 HIGH (`notes.md` §126).** TypeLLM README densify 3k→12k B. Batch 5.8x *theirs*. Constrained AR ≠ calibrated Noul. kev family densify. 4B new-source 0.790/0.806 *theirs*. 8.2% ≥0.9 on wrong *theirs*. option order can change an answer. fail-closed routing vs fail-open test selection. classifier ≠ authorizer. estimates not Harbor. catalog ≠ endorsement. SHA move is not a replica. Do not reopen or amend PR #23–#48. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. +Hourly 1542 uniqueness lock: TypeLLM/TypeLLM densify HEAD 6a48f9f1e623 README SHA dbdc1f193537; README densify 3k→12k B; thinking=True/False per-field budget; type safety does not guarantee factual accuracy; Batch 5.8x *theirs*; Constrained AR ≠ calibrated Noul; Qwen/Qwen3.8-27B ≠ Archer; jaredpalmer/kev densify HEAD b339f446a0ef README SHA 86b0a19909f3; Kev-0.6B 4B 8B family; 4B new-source 0.790/0.806 *theirs*; 8B new-source 0.796/0.780 *theirs*; Jev hosted 0.857 *theirs*; Questions share the input text but cannot read each other; No Jev outputs were used for training; 8.2% ≥0.9 on wrong *theirs*; option order can change an answer; Qwen3 ≠ Archer; wire-compat ≠ logit-equiv; SHA move is not a replica; TheoOliveira/pi-jev 21★ fail-closed routing; JEV_THRESHOLD 0.65 still soft; routing ≠ permission; harshwasan/jev-sentinel fail closed never auto-allows; harshwasan/jev-sentinel ≠ leepokai/jev-guard; jackbarunz/jev-tool-router ≠ esinocchi/jev-tool-router; threshold 0.90 still soft; 76/81 vs 77/81 *theirs*; 0.419s vs 2.459s *theirs*; $0.00486 vs $0.03673 *theirs*; does not execute; not a security boundary; baronunread/leanest fail-open uncertainty means RUN; classifier.dev default Jev/Laya pluggable; openlayer-ai/jevals ≠ dayhaysoos/jevals; estimates not Harbor; classifier ≠ authorizer; MrJev/awesome-jev 118 entries catalog ≠ endorsement; MrJev/awesome-jev ≠ yibie/awesome-jev; Koushik890/jev-firewall fail closed ask_below 0.7 still soft; CompleteTech-LLC-AI-Research/jev-codex-approval experimental native not compiled; confidence is not a measured probability; rh-guard owns primary gates; hf:rAVEUK/open-jev-deberta-v3-large encoder class member not Jev replica; hf:p-yan/laya-quanto serving substrate ≠ calibrated replica; hf:Gtrkrsk/laya serving substrate ≠ calibrated replica; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48; notes.md §126 diff --git a/.agents/skills/augustus/references/mixed-architecture.md b/.agents/skills/augustus/references/mixed-architecture.md index d1531d8..dad3d9f 100644 --- a/.agents/skills/augustus/references/mixed-architecture.md +++ b/.agents/skills/augustus/references/mixed-architecture.md @@ -1461,3 +1461,7 @@ Hourly 1340 uniqueness lock: typesafe-sdk 0.7 Pydantic response models; msgspec **Hourly 1441 HIGH (`notes.md` §125).** vLLM NVIDIA + MLX Apple Silicon. Codiv hosted free endpoint. dual /v1/systemone + /v1/chat/completions. chat 501 on MLX. dual serving is not generate. Hosted Codiv ≠ TypeSafe. candidate probabilities are relative not correctness. recommendation is advisory. the server never blocks on its own. LoRA ≠ RLCD replica. pass-min 0.8 still soft. 37.30s → 2.40s at 64 decisions *theirs*. 2B 94.71% 9B 97.54% hard test *theirs*. wire-compat ≠ logit-equiv. SHA move is not a replica. catalog ≠ endorsement. Do not reopen or amend PR #23–#47. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. Hourly 1441 uniqueness lock: vLLM NVIDIA + MLX Apple Silicon; Codiv hosted free endpoint; dual /v1/systemone + /v1/chat/completions; razorback16/openjev densify HEAD cddbd962c88a README SHA a5943415cb92; STE README rewrite; serving-port densify; chat 501 on MLX; dual serving is not generate; Hosted Codiv ≠ TypeSafe; wire-compat ≠ logit-equiv; SHA move is not a replica; Error contract is not a Noul; hr98w/jev-visual 167★ Apple Silicon visual candidate scoring; 37.30s → 2.40s at 64 decisions *theirs*; Breakout 9 bricks 6 returns 2 lives *theirs*; candidate probabilities are relative not correctness; jkudish/jev-mcp 156★ ten MCP tools; recommendation is advisory; the server never blocks on its own; TypeSafe CLERC 5% to 18% *theirs*; jkudish/jev-mcp ≠ burnigtm/jev-mcp; zhengxuyu/litjev off-the-shelf Qwen decision layer; Probabilities are not calibrated by default; Qwen/Qwen3.8-27B ≠ Archer; zhengxuyu/litjev ≠ alexwestco/llm-to-jev; Zefan-Cai/Open-Jev LoRA + scalar head; 2B 94.71% 9B 97.54% hard test *theirs*; 2B OOD 86.02% 9B OOD 91.97% *theirs*; 80,816 training rows; 27B still in progress; LoRA ≠ RLCD replica; Zefan-Cai/Open-Jev ≠ TheoLeeCJ/openjev ≠ razorback16/openjev; cristianoliveira/jeq intelligence you can pipe; pass-min 0.8 still soft; JEQ does not own actions; AndyInQtr/laya-coreai CoreML serving substrate ≠ calibrated replica; AndyInQtr/laya-coreai ≠ mizorewww/laya-coreml; catalog ≠ endorsement; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47; notes.md §125 + + +**Hourly 1542 HIGH (`notes.md` §126).** TypeLLM README densify 3k→12k B. Batch 5.8x *theirs*. Constrained AR ≠ calibrated Noul. kev family densify. 4B new-source 0.790/0.806 *theirs*. 8.2% ≥0.9 on wrong *theirs*. option order can change an answer. fail-closed routing vs fail-open test selection. classifier ≠ authorizer. estimates not Harbor. catalog ≠ endorsement. SHA move is not a replica. Do not reopen or amend PR #23–#48. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. +Hourly 1542 uniqueness lock: TypeLLM/TypeLLM densify HEAD 6a48f9f1e623 README SHA dbdc1f193537; README densify 3k→12k B; thinking=True/False per-field budget; type safety does not guarantee factual accuracy; Batch 5.8x *theirs*; Constrained AR ≠ calibrated Noul; Qwen/Qwen3.8-27B ≠ Archer; jaredpalmer/kev densify HEAD b339f446a0ef README SHA 86b0a19909f3; Kev-0.6B 4B 8B family; 4B new-source 0.790/0.806 *theirs*; 8B new-source 0.796/0.780 *theirs*; Jev hosted 0.857 *theirs*; Questions share the input text but cannot read each other; No Jev outputs were used for training; 8.2% ≥0.9 on wrong *theirs*; option order can change an answer; Qwen3 ≠ Archer; wire-compat ≠ logit-equiv; SHA move is not a replica; TheoOliveira/pi-jev 21★ fail-closed routing; JEV_THRESHOLD 0.65 still soft; routing ≠ permission; harshwasan/jev-sentinel fail closed never auto-allows; harshwasan/jev-sentinel ≠ leepokai/jev-guard; jackbarunz/jev-tool-router ≠ esinocchi/jev-tool-router; threshold 0.90 still soft; 76/81 vs 77/81 *theirs*; 0.419s vs 2.459s *theirs*; $0.00486 vs $0.03673 *theirs*; does not execute; not a security boundary; baronunread/leanest fail-open uncertainty means RUN; classifier.dev default Jev/Laya pluggable; openlayer-ai/jevals ≠ dayhaysoos/jevals; estimates not Harbor; classifier ≠ authorizer; MrJev/awesome-jev 118 entries catalog ≠ endorsement; MrJev/awesome-jev ≠ yibie/awesome-jev; Koushik890/jev-firewall fail closed ask_below 0.7 still soft; CompleteTech-LLC-AI-Research/jev-codex-approval experimental native not compiled; confidence is not a measured probability; rh-guard owns primary gates; hf:rAVEUK/open-jev-deberta-v3-large encoder class member not Jev replica; hf:p-yan/laya-quanto serving substrate ≠ calibrated replica; hf:Gtrkrsk/laya serving substrate ≠ calibrated replica; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48; notes.md §126 diff --git a/.agents/skills/augustus/references/question-design.md b/.agents/skills/augustus/references/question-design.md index df28f5b..277a91d 100644 --- a/.agents/skills/augustus/references/question-design.md +++ b/.agents/skills/augustus/references/question-design.md @@ -441,3 +441,7 @@ Hourly 1340 uniqueness lock: typesafe-sdk 0.7 Pydantic response models; msgspec **Hourly 1441 HIGH (`notes.md` §125).** vLLM NVIDIA + MLX Apple Silicon. Codiv hosted free endpoint. dual /v1/systemone + /v1/chat/completions. chat 501 on MLX. dual serving is not generate. Hosted Codiv ≠ TypeSafe. candidate probabilities are relative not correctness. recommendation is advisory. the server never blocks on its own. LoRA ≠ RLCD replica. pass-min 0.8 still soft. 37.30s → 2.40s at 64 decisions *theirs*. 2B 94.71% 9B 97.54% hard test *theirs*. wire-compat ≠ logit-equiv. SHA move is not a replica. catalog ≠ endorsement. Do not reopen or amend PR #23–#47. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. Hourly 1441 uniqueness lock: vLLM NVIDIA + MLX Apple Silicon; Codiv hosted free endpoint; dual /v1/systemone + /v1/chat/completions; razorback16/openjev densify HEAD cddbd962c88a README SHA a5943415cb92; STE README rewrite; serving-port densify; chat 501 on MLX; dual serving is not generate; Hosted Codiv ≠ TypeSafe; wire-compat ≠ logit-equiv; SHA move is not a replica; Error contract is not a Noul; hr98w/jev-visual 167★ Apple Silicon visual candidate scoring; 37.30s → 2.40s at 64 decisions *theirs*; Breakout 9 bricks 6 returns 2 lives *theirs*; candidate probabilities are relative not correctness; jkudish/jev-mcp 156★ ten MCP tools; recommendation is advisory; the server never blocks on its own; TypeSafe CLERC 5% to 18% *theirs*; jkudish/jev-mcp ≠ burnigtm/jev-mcp; zhengxuyu/litjev off-the-shelf Qwen decision layer; Probabilities are not calibrated by default; Qwen/Qwen3.8-27B ≠ Archer; zhengxuyu/litjev ≠ alexwestco/llm-to-jev; Zefan-Cai/Open-Jev LoRA + scalar head; 2B 94.71% 9B 97.54% hard test *theirs*; 2B OOD 86.02% 9B OOD 91.97% *theirs*; 80,816 training rows; 27B still in progress; LoRA ≠ RLCD replica; Zefan-Cai/Open-Jev ≠ TheoLeeCJ/openjev ≠ razorback16/openjev; cristianoliveira/jeq intelligence you can pipe; pass-min 0.8 still soft; JEQ does not own actions; AndyInQtr/laya-coreai CoreML serving substrate ≠ calibrated replica; AndyInQtr/laya-coreai ≠ mizorewww/laya-coreml; catalog ≠ endorsement; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47; notes.md §125 + + +**Hourly 1542 HIGH (`notes.md` §126).** TypeLLM README densify 3k→12k B. Batch 5.8x *theirs*. Constrained AR ≠ calibrated Noul. kev family densify. 4B new-source 0.790/0.806 *theirs*. 8.2% ≥0.9 on wrong *theirs*. option order can change an answer. fail-closed routing vs fail-open test selection. classifier ≠ authorizer. estimates not Harbor. catalog ≠ endorsement. SHA move is not a replica. Do not reopen or amend PR #23–#48. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. +Hourly 1542 uniqueness lock: TypeLLM/TypeLLM densify HEAD 6a48f9f1e623 README SHA dbdc1f193537; README densify 3k→12k B; thinking=True/False per-field budget; type safety does not guarantee factual accuracy; Batch 5.8x *theirs*; Constrained AR ≠ calibrated Noul; Qwen/Qwen3.8-27B ≠ Archer; jaredpalmer/kev densify HEAD b339f446a0ef README SHA 86b0a19909f3; Kev-0.6B 4B 8B family; 4B new-source 0.790/0.806 *theirs*; 8B new-source 0.796/0.780 *theirs*; Jev hosted 0.857 *theirs*; Questions share the input text but cannot read each other; No Jev outputs were used for training; 8.2% ≥0.9 on wrong *theirs*; option order can change an answer; Qwen3 ≠ Archer; wire-compat ≠ logit-equiv; SHA move is not a replica; TheoOliveira/pi-jev 21★ fail-closed routing; JEV_THRESHOLD 0.65 still soft; routing ≠ permission; harshwasan/jev-sentinel fail closed never auto-allows; harshwasan/jev-sentinel ≠ leepokai/jev-guard; jackbarunz/jev-tool-router ≠ esinocchi/jev-tool-router; threshold 0.90 still soft; 76/81 vs 77/81 *theirs*; 0.419s vs 2.459s *theirs*; $0.00486 vs $0.03673 *theirs*; does not execute; not a security boundary; baronunread/leanest fail-open uncertainty means RUN; classifier.dev default Jev/Laya pluggable; openlayer-ai/jevals ≠ dayhaysoos/jevals; estimates not Harbor; classifier ≠ authorizer; MrJev/awesome-jev 118 entries catalog ≠ endorsement; MrJev/awesome-jev ≠ yibie/awesome-jev; Koushik890/jev-firewall fail closed ask_below 0.7 still soft; CompleteTech-LLC-AI-Research/jev-codex-approval experimental native not compiled; confidence is not a measured probability; rh-guard owns primary gates; hf:rAVEUK/open-jev-deberta-v3-large encoder class member not Jev replica; hf:p-yan/laya-quanto serving substrate ≠ calibrated replica; hf:Gtrkrsk/laya serving substrate ≠ calibrated replica; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48; notes.md §126 diff --git a/.agents/skills/augustus/references/toolbox-mapping.md b/.agents/skills/augustus/references/toolbox-mapping.md index a8fa313..275902f 100644 --- a/.agents/skills/augustus/references/toolbox-mapping.md +++ b/.agents/skills/augustus/references/toolbox-mapping.md @@ -370,3 +370,7 @@ Hourly 1340 uniqueness lock: typesafe-sdk 0.7 Pydantic response models; msgspec **Hourly 1441 HIGH (`notes.md` §125).** vLLM NVIDIA + MLX Apple Silicon. Codiv hosted free endpoint. dual /v1/systemone + /v1/chat/completions. chat 501 on MLX. dual serving is not generate. Hosted Codiv ≠ TypeSafe. candidate probabilities are relative not correctness. recommendation is advisory. the server never blocks on its own. LoRA ≠ RLCD replica. pass-min 0.8 still soft. 37.30s → 2.40s at 64 decisions *theirs*. 2B 94.71% 9B 97.54% hard test *theirs*. wire-compat ≠ logit-equiv. SHA move is not a replica. catalog ≠ endorsement. Do not reopen or amend PR #23–#47. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. Hourly 1441 uniqueness lock: vLLM NVIDIA + MLX Apple Silicon; Codiv hosted free endpoint; dual /v1/systemone + /v1/chat/completions; razorback16/openjev densify HEAD cddbd962c88a README SHA a5943415cb92; STE README rewrite; serving-port densify; chat 501 on MLX; dual serving is not generate; Hosted Codiv ≠ TypeSafe; wire-compat ≠ logit-equiv; SHA move is not a replica; Error contract is not a Noul; hr98w/jev-visual 167★ Apple Silicon visual candidate scoring; 37.30s → 2.40s at 64 decisions *theirs*; Breakout 9 bricks 6 returns 2 lives *theirs*; candidate probabilities are relative not correctness; jkudish/jev-mcp 156★ ten MCP tools; recommendation is advisory; the server never blocks on its own; TypeSafe CLERC 5% to 18% *theirs*; jkudish/jev-mcp ≠ burnigtm/jev-mcp; zhengxuyu/litjev off-the-shelf Qwen decision layer; Probabilities are not calibrated by default; Qwen/Qwen3.8-27B ≠ Archer; zhengxuyu/litjev ≠ alexwestco/llm-to-jev; Zefan-Cai/Open-Jev LoRA + scalar head; 2B 94.71% 9B 97.54% hard test *theirs*; 2B OOD 86.02% 9B OOD 91.97% *theirs*; 80,816 training rows; 27B still in progress; LoRA ≠ RLCD replica; Zefan-Cai/Open-Jev ≠ TheoLeeCJ/openjev ≠ razorback16/openjev; cristianoliveira/jeq intelligence you can pipe; pass-min 0.8 still soft; JEQ does not own actions; AndyInQtr/laya-coreai CoreML serving substrate ≠ calibrated replica; AndyInQtr/laya-coreai ≠ mizorewww/laya-coreml; catalog ≠ endorsement; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47; notes.md §125 + + +**Hourly 1542 HIGH (`notes.md` §126).** TypeLLM README densify 3k→12k B. Batch 5.8x *theirs*. Constrained AR ≠ calibrated Noul. kev family densify. 4B new-source 0.790/0.806 *theirs*. 8.2% ≥0.9 on wrong *theirs*. option order can change an answer. fail-closed routing vs fail-open test selection. classifier ≠ authorizer. estimates not Harbor. catalog ≠ endorsement. SHA move is not a replica. Do not reopen or amend PR #23–#48. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. +Hourly 1542 uniqueness lock: TypeLLM/TypeLLM densify HEAD 6a48f9f1e623 README SHA dbdc1f193537; README densify 3k→12k B; thinking=True/False per-field budget; type safety does not guarantee factual accuracy; Batch 5.8x *theirs*; Constrained AR ≠ calibrated Noul; Qwen/Qwen3.8-27B ≠ Archer; jaredpalmer/kev densify HEAD b339f446a0ef README SHA 86b0a19909f3; Kev-0.6B 4B 8B family; 4B new-source 0.790/0.806 *theirs*; 8B new-source 0.796/0.780 *theirs*; Jev hosted 0.857 *theirs*; Questions share the input text but cannot read each other; No Jev outputs were used for training; 8.2% ≥0.9 on wrong *theirs*; option order can change an answer; Qwen3 ≠ Archer; wire-compat ≠ logit-equiv; SHA move is not a replica; TheoOliveira/pi-jev 21★ fail-closed routing; JEV_THRESHOLD 0.65 still soft; routing ≠ permission; harshwasan/jev-sentinel fail closed never auto-allows; harshwasan/jev-sentinel ≠ leepokai/jev-guard; jackbarunz/jev-tool-router ≠ esinocchi/jev-tool-router; threshold 0.90 still soft; 76/81 vs 77/81 *theirs*; 0.419s vs 2.459s *theirs*; $0.00486 vs $0.03673 *theirs*; does not execute; not a security boundary; baronunread/leanest fail-open uncertainty means RUN; classifier.dev default Jev/Laya pluggable; openlayer-ai/jevals ≠ dayhaysoos/jevals; estimates not Harbor; classifier ≠ authorizer; MrJev/awesome-jev 118 entries catalog ≠ endorsement; MrJev/awesome-jev ≠ yibie/awesome-jev; Koushik890/jev-firewall fail closed ask_below 0.7 still soft; CompleteTech-LLC-AI-Research/jev-codex-approval experimental native not compiled; confidence is not a measured probability; rh-guard owns primary gates; hf:rAVEUK/open-jev-deberta-v3-large encoder class member not Jev replica; hf:p-yan/laya-quanto serving substrate ≠ calibrated replica; hf:Gtrkrsk/laya serving substrate ≠ calibrated replica; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48; notes.md §126 diff --git a/.agents/skills/augustus/references/validation.md b/.agents/skills/augustus/references/validation.md index d03e955..bab0def 100644 --- a/.agents/skills/augustus/references/validation.md +++ b/.agents/skills/augustus/references/validation.md @@ -1174,3 +1174,7 @@ Hourly 1340 uniqueness lock: typesafe-sdk 0.7 Pydantic response models; msgspec **Hourly 1441 HIGH (`notes.md` §125).** vLLM NVIDIA + MLX Apple Silicon. Codiv hosted free endpoint. dual /v1/systemone + /v1/chat/completions. chat 501 on MLX. dual serving is not generate. Hosted Codiv ≠ TypeSafe. candidate probabilities are relative not correctness. recommendation is advisory. the server never blocks on its own. LoRA ≠ RLCD replica. pass-min 0.8 still soft. 37.30s → 2.40s at 64 decisions *theirs*. 2B 94.71% 9B 97.54% hard test *theirs*. wire-compat ≠ logit-equiv. SHA move is not a replica. catalog ≠ endorsement. Do not reopen or amend PR #23–#47. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. Hourly 1441 uniqueness lock: vLLM NVIDIA + MLX Apple Silicon; Codiv hosted free endpoint; dual /v1/systemone + /v1/chat/completions; razorback16/openjev densify HEAD cddbd962c88a README SHA a5943415cb92; STE README rewrite; serving-port densify; chat 501 on MLX; dual serving is not generate; Hosted Codiv ≠ TypeSafe; wire-compat ≠ logit-equiv; SHA move is not a replica; Error contract is not a Noul; hr98w/jev-visual 167★ Apple Silicon visual candidate scoring; 37.30s → 2.40s at 64 decisions *theirs*; Breakout 9 bricks 6 returns 2 lives *theirs*; candidate probabilities are relative not correctness; jkudish/jev-mcp 156★ ten MCP tools; recommendation is advisory; the server never blocks on its own; TypeSafe CLERC 5% to 18% *theirs*; jkudish/jev-mcp ≠ burnigtm/jev-mcp; zhengxuyu/litjev off-the-shelf Qwen decision layer; Probabilities are not calibrated by default; Qwen/Qwen3.8-27B ≠ Archer; zhengxuyu/litjev ≠ alexwestco/llm-to-jev; Zefan-Cai/Open-Jev LoRA + scalar head; 2B 94.71% 9B 97.54% hard test *theirs*; 2B OOD 86.02% 9B OOD 91.97% *theirs*; 80,816 training rows; 27B still in progress; LoRA ≠ RLCD replica; Zefan-Cai/Open-Jev ≠ TheoLeeCJ/openjev ≠ razorback16/openjev; cristianoliveira/jeq intelligence you can pipe; pass-min 0.8 still soft; JEQ does not own actions; AndyInQtr/laya-coreai CoreML serving substrate ≠ calibrated replica; AndyInQtr/laya-coreai ≠ mizorewww/laya-coreml; catalog ≠ endorsement; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47; notes.md §125 + + +**Hourly 1542 HIGH (`notes.md` §126).** TypeLLM README densify 3k→12k B. Batch 5.8x *theirs*. Constrained AR ≠ calibrated Noul. kev family densify. 4B new-source 0.790/0.806 *theirs*. 8.2% ≥0.9 on wrong *theirs*. option order can change an answer. fail-closed routing vs fail-open test selection. classifier ≠ authorizer. estimates not Harbor. catalog ≠ endorsement. SHA move is not a replica. Do not reopen or amend PR #23–#48. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. +Hourly 1542 uniqueness lock: TypeLLM/TypeLLM densify HEAD 6a48f9f1e623 README SHA dbdc1f193537; README densify 3k→12k B; thinking=True/False per-field budget; type safety does not guarantee factual accuracy; Batch 5.8x *theirs*; Constrained AR ≠ calibrated Noul; Qwen/Qwen3.8-27B ≠ Archer; jaredpalmer/kev densify HEAD b339f446a0ef README SHA 86b0a19909f3; Kev-0.6B 4B 8B family; 4B new-source 0.790/0.806 *theirs*; 8B new-source 0.796/0.780 *theirs*; Jev hosted 0.857 *theirs*; Questions share the input text but cannot read each other; No Jev outputs were used for training; 8.2% ≥0.9 on wrong *theirs*; option order can change an answer; Qwen3 ≠ Archer; wire-compat ≠ logit-equiv; SHA move is not a replica; TheoOliveira/pi-jev 21★ fail-closed routing; JEV_THRESHOLD 0.65 still soft; routing ≠ permission; harshwasan/jev-sentinel fail closed never auto-allows; harshwasan/jev-sentinel ≠ leepokai/jev-guard; jackbarunz/jev-tool-router ≠ esinocchi/jev-tool-router; threshold 0.90 still soft; 76/81 vs 77/81 *theirs*; 0.419s vs 2.459s *theirs*; $0.00486 vs $0.03673 *theirs*; does not execute; not a security boundary; baronunread/leanest fail-open uncertainty means RUN; classifier.dev default Jev/Laya pluggable; openlayer-ai/jevals ≠ dayhaysoos/jevals; estimates not Harbor; classifier ≠ authorizer; MrJev/awesome-jev 118 entries catalog ≠ endorsement; MrJev/awesome-jev ≠ yibie/awesome-jev; Koushik890/jev-firewall fail closed ask_below 0.7 still soft; CompleteTech-LLC-AI-Research/jev-codex-approval experimental native not compiled; confidence is not a measured probability; rh-guard owns primary gates; hf:rAVEUK/open-jev-deberta-v3-large encoder class member not Jev replica; hf:p-yan/laya-quanto serving substrate ≠ calibrated replica; hf:Gtrkrsk/laya serving substrate ≠ calibrated replica; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48; notes.md §126 diff --git a/.agents/skills/augustus/scripts/evaluate_decisions.py b/.agents/skills/augustus/scripts/evaluate_decisions.py index f9617aa..5e3129a 100755 --- a/.agents/skills/augustus/scripts/evaluate_decisions.py +++ b/.agents/skills/augustus/scripts/evaluate_decisions.py @@ -355,6 +355,50 @@ def pass_min_still_soft(threshold, hard_gate=False): return hard_gate is False + + +def constrained_ar_is_not_noul(kind, noul_claimed=False): + """Constrained AR ≠ calibrated Noul.""" + if kind != "constrained_ar": + raise ValueError("unexpected kind") + return noul_claimed is False + + +def fail_closed_never_auto_allows(polarity, auto_allows=False): + """harshwasan/jev-sentinel fail closed never auto-allows.""" + if polarity != "fail_closed": + raise ValueError("unexpected polarity") + return auto_allows is False + + +def fail_open_uncertainty_means_run(polarity, uncertainty_action): + """baronunread/leanest fail-open uncertainty means RUN.""" + if polarity != "fail_open": + raise ValueError("unexpected polarity") + return uncertainty_action == "RUN" + + +def routing_threshold_still_soft(threshold, hard_gate=False): + """JEV_THRESHOLD 0.65 / 0.90 / ask_below 0.7 still soft.""" + if threshold not in (0.65, 0.90, 0.7): + raise ValueError("unexpected threshold") + return hard_gate is False + + +def classifier_is_not_authorizer(role, grants_permission=False): + """classifier ≠ authorizer.""" + if role != "classifier": + raise ValueError("unexpected role") + return grants_permission is False + + +def estimates_are_not_harbor(kind, harbor=False): + """estimates not Harbor.""" + if kind != "estimate": + raise ValueError("unexpected kind") + return harbor is False + + def hop_ece_permutation_invariant(rows, bins=10, key="p"): """Shuffle order; equal-width ECE must not move. @@ -496,6 +540,26 @@ def self_test(): assert theirs_bench_is_not_harbor(64, "jev-visual-37.30s-2.40s") assert theirs_bench_is_not_harbor(10046, "open-jev-2b-94.71") + + # 1542: Constrained AR ≠ Noul / fail-closed never auto-allows / + # fail-open RUN / routing threshold still soft / classifier ≠ + # authorizer / estimates not Harbor. + assert constrained_ar_is_not_noul("constrained_ar", False) + assert not constrained_ar_is_not_noul("constrained_ar", True) + assert fail_closed_never_auto_allows("fail_closed", False) + assert not fail_closed_never_auto_allows("fail_closed", True) + assert fail_open_uncertainty_means_run("fail_open", "RUN") + assert not fail_open_uncertainty_means_run("fail_open", "SKIP") + assert routing_threshold_still_soft(0.65, hard_gate=False) + assert not routing_threshold_still_soft(0.90, hard_gate=True) + assert routing_threshold_still_soft(0.7, hard_gate=False) + assert classifier_is_not_authorizer("classifier", False) + assert not classifier_is_not_authorizer("classifier", True) + assert estimates_are_not_harbor("estimate", harbor=False) + assert not estimates_are_not_harbor("estimate", harbor=True) + assert theirs_bench_is_not_harbor(16, "typellm-batch-5.8x") + assert theirs_bench_is_not_harbor(81, "esinocchi-76-81") + print("self-test ok") diff --git a/.agents/skills/augustus/scripts/uniqueness_gate.py b/.agents/skills/augustus/scripts/uniqueness_gate.py index 8360999..78d8d67 100644 --- a/.agents/skills/augustus/scripts/uniqueness_gate.py +++ b/.agents/skills/augustus/scripts/uniqueness_gate.py @@ -2,7 +2,7 @@ """Uniqueness gate for merged 0843 (§114), merged 0915 NanoJev (§115), merged 0920 jcr (§116), merged 0922 SemIf (§117), merged 0940 llm-to-jev (§118), hourly 0947 HIGH (§119), hourly 1049 HIGH (§120), -hourly 1143 HIGH (§121), hourly 1248 HIGH (§123), hourly 1340 HIGH (§124), and hourly 1441 HIGH (§125). +hourly 1143 HIGH (§121), hourly 1248 HIGH (§123), hourly 1340 HIGH (§124), hourly 1441 HIGH (§125), and hourly 1542 HIGH (§126). Each lock must appear as one consecutive substring in every listed overlay. Fragments scattered across files do not count. @@ -11,9 +11,9 @@ substring in the skill + research files (not a 21-overlay dump wall). Hourly must treat revisit HIGH like novel HIGH. Star-noise is not a fold. -Also: YAML-parse SKILL.md frontmatter; notes.md owns §114–§125; -composition items 289–316, 322–329, 330–336, 337–352, 353–368, 369–384, 385–400, 401–416, and 417–432 exist; -findings batches #97–#107 exist. Items 317–321 stay unused. +Also: YAML-parse SKILL.md frontmatter; notes.md owns §114–§126; +composition items 289–316, 322–329, 330–336, 337–352, 353–368, 369–384, 385–400, 401–416, 417–432, and 433–448 exist; +findings batches #97–#108 exist. Items 317–321 stay unused. CHANGELOG.md must not hold uniqueness dump walls (dumps live in changelog-hourly.md). README.md must not hold the 0743 dump wall. Pages greps stay in docs/index.md and docs/_layouts/default.html. @@ -146,6 +146,10 @@ 'Hourly 1441 uniqueness lock: vLLM NVIDIA + MLX Apple Silicon; Codiv hosted free endpoint; dual /v1/systemone + /v1/chat/completions; razorback16/openjev densify HEAD cddbd962c88a README SHA a5943415cb92; STE README rewrite; serving-port densify; chat 501 on MLX; dual serving is not generate; Hosted Codiv ≠ TypeSafe; wire-compat ≠ logit-equiv; SHA move is not a replica; Error contract is not a Noul; hr98w/jev-visual 167★ Apple Silicon visual candidate scoring; 37.30s → 2.40s at 64 decisions *theirs*; Breakout 9 bricks 6 returns 2 lives *theirs*; candidate probabilities are relative not correctness; jkudish/jev-mcp 156★ ten MCP tools; recommendation is advisory; the server never blocks on its own; TypeSafe CLERC 5% to 18% *theirs*; jkudish/jev-mcp ≠ burnigtm/jev-mcp; zhengxuyu/litjev off-the-shelf Qwen decision layer; Probabilities are not calibrated by default; Qwen/Qwen3.8-27B ≠ Archer; zhengxuyu/litjev ≠ alexwestco/llm-to-jev; Zefan-Cai/Open-Jev LoRA + scalar head; 2B 94.71% 9B 97.54% hard test *theirs*; 2B OOD 86.02% 9B OOD 91.97% *theirs*; 80,816 training rows; 27B still in progress; LoRA ≠ RLCD replica; Zefan-Cai/Open-Jev ≠ TheoLeeCJ/openjev ≠ razorback16/openjev; cristianoliveira/jeq intelligence you can pipe; pass-min 0.8 still soft; JEQ does not own actions; AndyInQtr/laya-coreai CoreML serving substrate ≠ calibrated replica; AndyInQtr/laya-coreai ≠ mizorewww/laya-coreml; catalog ≠ endorsement; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47; notes.md §125' ) +UNIQ_1542 = ( + 'Hourly 1542 uniqueness lock: TypeLLM/TypeLLM densify HEAD 6a48f9f1e623 README SHA dbdc1f193537; README densify 3k→12k B; thinking=True/False per-field budget; type safety does not guarantee factual accuracy; Batch 5.8x *theirs*; Constrained AR ≠ calibrated Noul; Qwen/Qwen3.8-27B ≠ Archer; jaredpalmer/kev densify HEAD b339f446a0ef README SHA 86b0a19909f3; Kev-0.6B 4B 8B family; 4B new-source 0.790/0.806 *theirs*; 8B new-source 0.796/0.780 *theirs*; Jev hosted 0.857 *theirs*; Questions share the input text but cannot read each other; No Jev outputs were used for training; 8.2% ≥0.9 on wrong *theirs*; option order can change an answer; Qwen3 ≠ Archer; wire-compat ≠ logit-equiv; SHA move is not a replica; TheoOliveira/pi-jev 21★ fail-closed routing; JEV_THRESHOLD 0.65 still soft; routing ≠ permission; harshwasan/jev-sentinel fail closed never auto-allows; harshwasan/jev-sentinel ≠ leepokai/jev-guard; jackbarunz/jev-tool-router ≠ esinocchi/jev-tool-router; threshold 0.90 still soft; 76/81 vs 77/81 *theirs*; 0.419s vs 2.459s *theirs*; $0.00486 vs $0.03673 *theirs*; does not execute; not a security boundary; baronunread/leanest fail-open uncertainty means RUN; classifier.dev default Jev/Laya pluggable; openlayer-ai/jevals ≠ dayhaysoos/jevals; estimates not Harbor; classifier ≠ authorizer; MrJev/awesome-jev 118 entries catalog ≠ endorsement; MrJev/awesome-jev ≠ yibie/awesome-jev; Koushik890/jev-firewall fail closed ask_below 0.7 still soft; CompleteTech-LLC-AI-Research/jev-codex-approval experimental native not compiled; confidence is not a measured probability; rh-guard owns primary gates; hf:rAVEUK/open-jev-deberta-v3-large encoder class member not Jev replica; hf:p-yan/laya-quanto serving substrate ≠ calibrated replica; hf:Gtrkrsk/laya serving substrate ≠ calibrated replica; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48; notes.md §126' +) + REVISIT_LOCK = ( "Revisit / since-last-look lock: catalogued repos are not done; " "store fingerprints default_sha, pushed_at, description_hash, release_tag; " @@ -231,6 +235,8 @@ def main() -> int: failed.append(f"1340 lock missing as one substring: {rel}") if UNIQ_1441 not in body: failed.append(f"1441 lock missing as one substring: {rel}") + if UNIQ_1542 not in body: + failed.append(f"1542 lock missing as one substring: {rel}") for rel in REVISIT_OVERLAYS: path = ROOT / rel if not path.is_file(): @@ -264,10 +270,12 @@ def main() -> int: failed.append("notes.md missing §124 heading") if "## 125. Hourly 1441 HIGH" not in notes: failed.append("notes.md missing §125 heading") + if "## 126. Hourly 1542 HIGH" not in notes: + failed.append("notes.md missing §126 heading") algebra = (ROOT / ".agents/skills/augustus/references/composition-algebra.md").read_text( encoding="utf-8" ) - for n in list(range(289, 317)) + list(range(322, 330)) + list(range(330, 337)) + list(range(337, 353)) + list(range(353, 369)) + list(range(369, 385)) + list(range(385, 401)) + list(range(401, 417)) + list(range(417, 433)): + for n in list(range(289, 317)) + list(range(322, 330)) + list(range(330, 337)) + list(range(337, 353)) + list(range(353, 369)) + list(range(369, 385)) + list(range(385, 401)) + list(range(401, 417)) + list(range(417, 433)) + list(range(433, 449)): needle = f"{n}. **" if needle not in algebra: failed.append(f"composition-algebra missing item {n}") @@ -288,6 +296,7 @@ def main() -> int: "## Batch #105", "## Batch #106", "## Batch #107", + "## Batch #108", ): if batch not in findings: failed.append(f"findings.md missing {batch}") @@ -403,6 +412,45 @@ def main() -> int: 'JEQ does not own actions', 'AndyInQtr/laya-coreai CoreML serving substrate ≠ calibrated replica', 'AndyInQtr/laya-coreai ≠ mizorewww/laya-coreml', + 'TypeLLM/TypeLLM densify HEAD 6a48f9f1e623', + 'README densify 3k→12k B', + 'Batch 5.8x *theirs*', + 'Constrained AR ≠ calibrated Noul', + 'jaredpalmer/kev densify HEAD b339f446a0ef', + 'Kev-0.6B 4B 8B family', + '4B new-source 0.790/0.806 *theirs*', + '8B new-source 0.796/0.780 *theirs*', + 'Jev hosted 0.857 *theirs*', + 'Questions share the input text but cannot read each other', + 'No Jev outputs were used for training', + '8.2% ≥0.9 on wrong *theirs*', + 'option order can change an answer', + 'Qwen3 ≠ Archer', + 'TheoOliveira/pi-jev 21★ fail-closed routing', + 'JEV_THRESHOLD 0.65 still soft', + 'harshwasan/jev-sentinel fail closed never auto-allows', + 'harshwasan/jev-sentinel ≠ leepokai/jev-guard', + 'jackbarunz/jev-tool-router ≠ esinocchi/jev-tool-router', + 'threshold 0.90 still soft', + '76/81 vs 77/81 *theirs*', + '0.419s vs 2.459s *theirs*', + '$0.00486 vs $0.03673 *theirs*', + 'not a security boundary', + 'baronunread/leanest fail-open uncertainty means RUN', + 'classifier.dev default Jev/Laya pluggable', + 'openlayer-ai/jevals ≠ dayhaysoos/jevals', + 'estimates not Harbor', + 'classifier ≠ authorizer', + 'MrJev/awesome-jev 118 entries catalog ≠ endorsement', + 'MrJev/awesome-jev ≠ yibie/awesome-jev', + 'Koushik890/jev-firewall fail closed ask_below 0.7 still soft', + 'CompleteTech-LLC-AI-Research/jev-codex-approval experimental native not compiled', + 'confidence is not a measured probability', + 'rh-guard owns primary gates', + 'hf:rAVEUK/open-jev-deberta-v3-large encoder class member not Jev replica', + 'hf:p-yan/laya-quanto serving substrate ≠ calibrated replica', + 'hf:Gtrkrsk/laya serving substrate ≠ calibrated replica', + 'hourly 1542 / notes.md §126', ): if frag not in haystack: failed.append(f"SKILL.md missing fragment {frag!r}") @@ -501,6 +549,45 @@ def main() -> int: 'AndyInQtr/laya-coreai CoreML serving substrate ≠ calibrated replica', 'AndyInQtr/laya-coreai ≠ mizorewww/laya-coreml', 'hourly 1441 / notes.md §125', + 'TypeLLM/TypeLLM densify HEAD 6a48f9f1e623', + 'README densify 3k→12k B', + 'Batch 5.8x *theirs*', + 'Constrained AR ≠ calibrated Noul', + 'jaredpalmer/kev densify HEAD b339f446a0ef', + 'Kev-0.6B 4B 8B family', + '4B new-source 0.790/0.806 *theirs*', + '8B new-source 0.796/0.780 *theirs*', + 'Jev hosted 0.857 *theirs*', + 'Questions share the input text but cannot read each other', + 'No Jev outputs were used for training', + '8.2% ≥0.9 on wrong *theirs*', + 'option order can change an answer', + 'Qwen3 ≠ Archer', + 'TheoOliveira/pi-jev 21★ fail-closed routing', + 'JEV_THRESHOLD 0.65 still soft', + 'harshwasan/jev-sentinel fail closed never auto-allows', + 'harshwasan/jev-sentinel ≠ leepokai/jev-guard', + 'jackbarunz/jev-tool-router ≠ esinocchi/jev-tool-router', + 'threshold 0.90 still soft', + '76/81 vs 77/81 *theirs*', + '0.419s vs 2.459s *theirs*', + '$0.00486 vs $0.03673 *theirs*', + 'not a security boundary', + 'baronunread/leanest fail-open uncertainty means RUN', + 'classifier.dev default Jev/Laya pluggable', + 'openlayer-ai/jevals ≠ dayhaysoos/jevals', + 'estimates not Harbor', + 'classifier ≠ authorizer', + 'MrJev/awesome-jev 118 entries catalog ≠ endorsement', + 'MrJev/awesome-jev ≠ yibie/awesome-jev', + 'Koushik890/jev-firewall fail closed ask_below 0.7 still soft', + 'CompleteTech-LLC-AI-Research/jev-codex-approval experimental native not compiled', + 'confidence is not a measured probability', + 'rh-guard owns primary gates', + 'hf:rAVEUK/open-jev-deberta-v3-large encoder class member not Jev replica', + 'hf:p-yan/laya-quanto serving substrate ≠ calibrated replica', + 'hf:Gtrkrsk/laya serving substrate ≠ calibrated replica', + 'hourly 1542 / notes.md §126', ): if frag not in proto_line: failed.append(f"SKILL.md protocol missing {frag!r}") @@ -517,6 +604,7 @@ def main() -> int: ("1248", UNIQ_1248), ("1340", UNIQ_1340), ("1441", UNIQ_1441), + ("1542", UNIQ_1542), ): if lock in changelog: failed.append( @@ -589,6 +677,7 @@ def main() -> int: f"1248 chars={len(UNIQ_1248)} " f"1340 chars={len(UNIQ_1340)} " f"1441 chars={len(UNIQ_1441)} " + f"1542 chars={len(UNIQ_1542)} " f"revisit chars={len(REVISIT_LOCK)} " f"overlays={len(OVERLAYS)} " f"revisit_overlays={len(REVISIT_OVERLAYS)}" diff --git a/CHANGELOG.md b/CHANGELOG.md index 637563f..f049228 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -16,6 +16,39 @@ folds: `research/notes.md`. ## [Unreleased] +Hourly 1542 HIGH (`research/notes.md` §126 / composition items +433–448 / findings batch #108). Does **not** bump the 0.5.0 pin. +Uniqueness dumps live in +[`research/changelog-hourly.md`](research/changelog-hourly.md). +Do not reopen or amend PR #23–#48. Do not amend released 0.5.0 +(#42). Merged #48 owns §125. Merged #47 owns §124. + +### Added + +- **Hourly 1542 HIGH (`notes.md` §126).** TypeLLM README densify 3k→12k B / + Batch 5.8x *theirs* / Constrained AR ≠ calibrated Noul / kev family + new-source *theirs* / fail-closed routing / fail-open test selection / + classifier ≠ authorizer / estimates not Harbor. JEV_THRESHOLD 0.65 + still soft. threshold 0.90 still soft. ask_below 0.7 still soft. + 8.2% ≥0.9 on wrong *theirs*. 76/81 vs 77/81 *theirs*. wire-compat is not + logit-equiv. SHA move is not a replica. + Evaluator: Constrained AR ≠ Noul / fail-closed never auto-allows / + fail-open uncertainty means RUN / routing threshold still soft / + classifier ≠ authorizer / estimates not Harbor. uniqueness_gate.py now + checks 0843 + 0915 + jcr + 0922 + 0940 + 0947 + 1049 + 1143 + + 1248 + 1340 + 1441 + 1542. Composition items 433–448 / batch #108. + **HARD RULE:** do not reopen or amend PR #23–#48. Does **not** bump + 0.5.0. + +- **Recipe (class, not Jev-only).** Without Augustus: treat constrained + AR as a Noul, 5.8x as Harbor, 0.65 as a hard gate, fail-open RUN as + fail-closed, estimates as Harbor, or a catalog as endorsement. With + Augustus: Constrained AR ≠ calibrated Noul; Batch 5.8x *theirs*; + JEV_THRESHOLD 0.65 still soft; fail-open uncertainty means RUN; + classifier ≠ authorizer; estimates not Harbor. Same split for any + Choice/Score/Noul-style head, not only hosted Jev. + + Hourly 1441 HIGH (`research/notes.md` §125 / composition items 417–432 / findings batch #107). Does **not** bump the 0.5.0 pin. Uniqueness dumps live in diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 28b15ba..cbf08f4 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -45,7 +45,7 @@ re-opened as "new." Before folding: `.agents/skills/augustus/SKILL.md` - Do not re-fold an already-landed section as a new beat - Do not reopen or amend a merged fold PR (#23–#45) -- uniqueness_gate.py checks 0843 + 0915 + jcr + 0922 + 0940 + 0947 + 1049 + 1143 + 1248 + 1340 + 1441 consecutive locks, plus the revisit / since-last-look protocol substring in the skill and research files. +- uniqueness_gate.py checks 0843 + 0915 + jcr + 0922 + 0940 + 0947 + 1049 + 1143 + 1248 + 1340 + 1441 + 1542 consecutive locks, plus the revisit / since-last-look protocol substring in the skill and research files. - Hourly uniqueness dump: `research/changelog-hourly.md` (archive, not release notes) - Treat **revisit HIGH like novel HIGH**. Catalogued repos are not diff --git a/README.md b/README.md index ebe7ca9..9ab2177 100644 --- a/README.md +++ b/README.md @@ -144,6 +144,7 @@ User-provided 0920 jcr uniqueness lock: NiazMorshed2007/jcr MIT; site https://jc User-provided 0922 uniqueness lock: SemIf was formerly OpenJev; independent; not affiliated with Jev or TypeSafe; homepage openjev.com; default master; MIT; HEAD ca3ba65f1429; Tolerate float roundoff in MLX evidence verification; pushed 2026-09-19; live REST 2282★ / 140 forks; size 9177; README SHA 74ab7f7f; LICENSE SHA ca562883; interface pattern reproduction with open models; does not reproduce Jev undisclosed model/training; Direct option logits; 0 output tokens; shared-state parallel; MLX backend for Apple Silicon (`--backend mlx`); Qwen3.5-4B 3090 direct 1.023s vs AR JSON 5.332s (**5.21×**); argmax agree 18/21; systems comparison ≠ semantic equivalence; Parallel suffixes 20.03 dec/s on 777 decisions; Browser ladder Qwen3.5-4B authored BA 0.813, pert 0.766, TypeSafe subset agreement 0.845 vs Published Jev 0.883 (102 across 20 cases); Softmax over options ≠ calibrated Noul; typed output does not guarantee semantic correctness; wire/agreement ≠ replica of TypeSafe; SemIf ≠ kw2828/OpenJev playground ≠ zhihz/openjev ≠ apiplant/semif-rs port ≠ dddanielliu/semif-serve; rename is densify not a second census; JevBench 74.6 is §78 not this ladder; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#38; do not push onto open #39/#40; notes.md §117 User-provided 0940 uniqueness lock: Turn decision-shaped LLM prompts into proposed Jev primitives; This is a conversion assistant, not an automatic guarantee of equivalent behavior; The compiler uses deterministic heuristics, not an LLM or evaluation model; It understands a deliberately small set of common prompt patterns; Generated instructions and criteria must be reviewed before production use; Score ranges such as 0 to 1 are translated into ordered Jev criteria; Prompts requiring open-ended prose are not a fit; suitability strong/partial/not_a_fit; compatibility full/partial/none; Writing new text stays with an LLM; Review the generated Score rubric; Jev scores ordered criteria, not an arbitrary 0-to-1 range; Everything runs locally in the browser; There is no framework, database, account, API, or server-side prompt processing; The key is read from the process environment and is never stored or printed; connect-src 'none'; alexwestco/llm-to-jev ≠ altryne/jevify ≠ ryana/jevify ≠ fidecastro/jevify ≠ Mintzs/jevify ≠ gulagala001/jevify ≠ uspraveen/Jevify; HEAD 234058ab372d; README SHA 43cd94fb; LICENSE SHA 5f334006; compiler SHA fdf235d0; 2★; MIT; JavaScript; size 29; Pages https://alexwestco.github.io/llm-to-jev/; invented_signal false; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35; notes.md §118 Hourly 0947 uniqueness lock: Fast and cheap agent evals. jev as judge.; 18,041 skills from the 200 most-starred repos; Not a security scanner; 最简 Jev 调用演示器; confidence 不是正确率; q93304989-bit/jev-lab ≠ tanayvasishtha/jev-lab ≠ dairui1/jev-lab ≠ BrendanH18/jev-lab ≠ yibie/laya-jev-lab; 75% cheaper and 18% faster withdrawn; jev @0.15 100% recall 87% savings; 33Audits/jev-auto ≠ gargpratyush/jev-router; no Typesafe key, no PI_API_BASE, zero deps; tool-emitted Score/Noul ≠ calibrated Noul; semantic_compatibility: false; candidate_mass; Qwen3.5-2B ≠ Archer; Jev evaluates decisions; it cannot run a coding-agent session; Status: no model yet; S1LV3RJ1NX/openjev ≠ TheoLeeCJ/openjev; 28 accepted decisions; 3 targets; score 800; health 100; arcade game not a flight trainer; A successful live TypeSafe call has not been verified for v0.1.0; abhibansal60/tidy ≠ MANISH007700/tidy; No model, Jev included, predicted which channels its owner keeps; seed 1 selected on a held-out 400-item validation split; Brier 0.342 → 0.378; more accurate and more overconfident; Qwen3.5-4B ≠ Archer; static quants of kushalpatil/jevify-gemma4-26b-a4b; The labels were corrected, and one earlier result was retracted; zero of 23,869 eligible rows; Do not compare cost without checking task success; Exit 1 is not a proof; kisshan13/typesafe-ai-go ≠ Nibir1/typesafe-go ≠ official; 38 tests that cannot fail in a 356-model warehouse; if a parser can answer it, Jev is never asked; 359 of them; Games & Simulation 82; Education & Learning 1; Ratings are heuristics; syedabbasshaheer-art/jev-atlas ≠ ZeroX-01/jev-atlas ≠ Zaious/jev-capability-atlas ≠ gorock007/jev-atlas; anandi1989/awesome-jev-usecases ≠ whyashthakker/awesome-jev-use-cases ≠ walidboulanouar/awesome-jev-use-cases ≠ vamsikrishna2421/jev-usecases; Every headline result above is self-reported; Archer Hume 84.6% MMLU-Pro is a third-party probe not landed Archer; catalog ≠ endorsement; judge ≠ actuator; softmax over A–H ≠ Noul; SemIf 2270★; jevlike 1054★; TypeLLM/TypeLLM 16★; AnotiaWang 98★; yibie/awesome-jev 538★; Laya likes 889; tracker likes 68 lastModified UNCHANGED; Blackwood likes 2 gated manual; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#35/#36/#37/#38/#40; do not push onto open #39; notes.md §119 +- Hourly 1542 HIGH (`research/notes.md` §126 / items 433–448 / batch #108). TypeLLM + kev densify / Constrained AR ≠ Noul / fail-closed routing / fail-open RUN / classifier ≠ authorizer. uniqueness_gate 0843+0915+jcr+0922+0940+0947+1049+1143+1248+1340+1441+1542. Does not bump 0.5.0. Merged #48 owns §125. Merged #47 owns §124. - Hourly 1441 HIGH (`research/notes.md` §125 / items 417–432 / batch #107). openjev STE backends+Codiv / dual serving / jev-visual *theirs* / jev-mcp advisory. uniqueness_gate 0843+0915+jcr+0922+0940+0947+1049+1143+1248+1340+1441. Does not bump 0.5.0. Merged #47 owns §124. Merged #46 owns §123. - Hourly 1340 HIGH (`research/notes.md` §124 / items 401–416 / batch #106). typesafe-sdk 0.7 Pydantic / MLX 400 / PLAN_Qwen35 densify / GLiNER locate. uniqueness_gate 0843+0915+jcr+0922+0940+0947+1049+1143+1248+1340. Does not bump 0.5.0. Merged #46 owns §123. Merged #45 owns §122. - Hourly 1248 HIGH (`research/notes.md` §123 / items 385–400 / batch #105). decide is not generate / GLiNER-GLiClass ports / third-party benches *theirs* / openjev 0.2.1 densify. uniqueness_gate 0843+0915+jcr+0922+0940+0947+1049+1143+1248. Does not bump 0.5.0. Merged #45 owns §122. Merged #44 owns §121. @@ -164,3 +165,7 @@ Hourly 1340 uniqueness lock: typesafe-sdk 0.7 Pydantic response models; msgspec **Hourly 1441 HIGH (`notes.md` §125).** vLLM NVIDIA + MLX Apple Silicon. Codiv hosted free endpoint. dual /v1/systemone + /v1/chat/completions. chat 501 on MLX. dual serving is not generate. Hosted Codiv ≠ TypeSafe. candidate probabilities are relative not correctness. recommendation is advisory. the server never blocks on its own. LoRA ≠ RLCD replica. pass-min 0.8 still soft. 37.30s → 2.40s at 64 decisions *theirs*. 2B 94.71% 9B 97.54% hard test *theirs*. wire-compat ≠ logit-equiv. SHA move is not a replica. catalog ≠ endorsement. Do not reopen or amend PR #23–#47. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. Hourly 1441 uniqueness lock: vLLM NVIDIA + MLX Apple Silicon; Codiv hosted free endpoint; dual /v1/systemone + /v1/chat/completions; razorback16/openjev densify HEAD cddbd962c88a README SHA a5943415cb92; STE README rewrite; serving-port densify; chat 501 on MLX; dual serving is not generate; Hosted Codiv ≠ TypeSafe; wire-compat ≠ logit-equiv; SHA move is not a replica; Error contract is not a Noul; hr98w/jev-visual 167★ Apple Silicon visual candidate scoring; 37.30s → 2.40s at 64 decisions *theirs*; Breakout 9 bricks 6 returns 2 lives *theirs*; candidate probabilities are relative not correctness; jkudish/jev-mcp 156★ ten MCP tools; recommendation is advisory; the server never blocks on its own; TypeSafe CLERC 5% to 18% *theirs*; jkudish/jev-mcp ≠ burnigtm/jev-mcp; zhengxuyu/litjev off-the-shelf Qwen decision layer; Probabilities are not calibrated by default; Qwen/Qwen3.8-27B ≠ Archer; zhengxuyu/litjev ≠ alexwestco/llm-to-jev; Zefan-Cai/Open-Jev LoRA + scalar head; 2B 94.71% 9B 97.54% hard test *theirs*; 2B OOD 86.02% 9B OOD 91.97% *theirs*; 80,816 training rows; 27B still in progress; LoRA ≠ RLCD replica; Zefan-Cai/Open-Jev ≠ TheoLeeCJ/openjev ≠ razorback16/openjev; cristianoliveira/jeq intelligence you can pipe; pass-min 0.8 still soft; JEQ does not own actions; AndyInQtr/laya-coreai CoreML serving substrate ≠ calibrated replica; AndyInQtr/laya-coreai ≠ mizorewww/laya-coreml; catalog ≠ endorsement; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47; notes.md §125 + + +**Hourly 1542 HIGH (`notes.md` §126).** TypeLLM README densify 3k→12k B. Batch 5.8x *theirs*. Constrained AR ≠ calibrated Noul. kev family densify. 4B new-source 0.790/0.806 *theirs*. 8.2% ≥0.9 on wrong *theirs*. option order can change an answer. fail-closed routing vs fail-open test selection. classifier ≠ authorizer. estimates not Harbor. catalog ≠ endorsement. SHA move is not a replica. Do not reopen or amend PR #23–#48. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. +Hourly 1542 uniqueness lock: TypeLLM/TypeLLM densify HEAD 6a48f9f1e623 README SHA dbdc1f193537; README densify 3k→12k B; thinking=True/False per-field budget; type safety does not guarantee factual accuracy; Batch 5.8x *theirs*; Constrained AR ≠ calibrated Noul; Qwen/Qwen3.8-27B ≠ Archer; jaredpalmer/kev densify HEAD b339f446a0ef README SHA 86b0a19909f3; Kev-0.6B 4B 8B family; 4B new-source 0.790/0.806 *theirs*; 8B new-source 0.796/0.780 *theirs*; Jev hosted 0.857 *theirs*; Questions share the input text but cannot read each other; No Jev outputs were used for training; 8.2% ≥0.9 on wrong *theirs*; option order can change an answer; Qwen3 ≠ Archer; wire-compat ≠ logit-equiv; SHA move is not a replica; TheoOliveira/pi-jev 21★ fail-closed routing; JEV_THRESHOLD 0.65 still soft; routing ≠ permission; harshwasan/jev-sentinel fail closed never auto-allows; harshwasan/jev-sentinel ≠ leepokai/jev-guard; jackbarunz/jev-tool-router ≠ esinocchi/jev-tool-router; threshold 0.90 still soft; 76/81 vs 77/81 *theirs*; 0.419s vs 2.459s *theirs*; $0.00486 vs $0.03673 *theirs*; does not execute; not a security boundary; baronunread/leanest fail-open uncertainty means RUN; classifier.dev default Jev/Laya pluggable; openlayer-ai/jevals ≠ dayhaysoos/jevals; estimates not Harbor; classifier ≠ authorizer; MrJev/awesome-jev 118 entries catalog ≠ endorsement; MrJev/awesome-jev ≠ yibie/awesome-jev; Koushik890/jev-firewall fail closed ask_below 0.7 still soft; CompleteTech-LLC-AI-Research/jev-codex-approval experimental native not compiled; confidence is not a measured probability; rh-guard owns primary gates; hf:rAVEUK/open-jev-deberta-v3-large encoder class member not Jev replica; hf:p-yan/laya-quanto serving substrate ≠ calibrated replica; hf:Gtrkrsk/laya serving substrate ≠ calibrated replica; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48; notes.md §126 diff --git a/docs/_includes/comparison.html b/docs/_includes/comparison.html index cec817d..cb04d2b 100644 --- a/docs/_includes/comparison.html +++ b/docs/_includes/comparison.html @@ -16,6 +16,7 @@

With vs without Augustus

  • SemIf
  • NanoJev
  • Jeff-1
  • +
  • TypeLLM
  • diff --git a/docs/_includes/recipes.html b/docs/_includes/recipes.html index 8005f90..9d52f12 100644 --- a/docs/_includes/recipes.html +++ b/docs/_includes/recipes.html @@ -150,6 +150,62 @@

    Relative p, advisory MCP, LoRA

    Relative ranking separately from ECE. The server never blocks on its own.
    +
    +

    Constrained AR

    +

    TypeLLM vs Noul

    +
    +
    Problem
    +
    Typed constrained decode treated as a calibrated Noul.
    +
    Without
    +
    Quote 5.8x. Ship thinking=True as a probability.
    +
    With
    +
    Constrained AR ≠ calibrated Noul. type safety does not guarantee factual accuracy.
    +
    Measure
    +
    Batch 5.8x *theirs*. Thinking budget separately from ECE.
    +
    +
    +
    +

    Open heads

    +

    kev family isolation

    +
    +
    Problem
    +
    Sibling-blind questions treated as option-order immunity.
    +
    Without
    +
    Quote 0.790 as Harbor. Hard-gate 0.9.
    +
    With
    +
    Questions share the input text but cannot read each other. option order can change an answer.
    +
    Measure
    +
    4B new-source 0.790/0.806 *theirs*. 8.2% ≥0.9 on wrong *theirs*.
    +
    +
    +
    +

    Fail polarity

    +

    Closed routing vs open RUN

    +
    +
    Problem
    +
    One cutoff treated as both a router and a permission.
    +
    Without
    +
    0.65 auto-allows. Uncertainty skips the suite.
    +
    With
    +
    JEV_THRESHOLD 0.65 still soft. routing ≠ permission. fail-open uncertainty means RUN. fail closed never auto-allows.
    +
    Measure
    +
    Name the polarity per act. Thresholds stay application policy.
    +
    +
    +
    +

    Eval / catalog

    +

    classifier ≠ authorizer

    +
    +
    Problem
    +
    An estimate row or a curated list treated as a grant.
    +
    Without
    +
    Ship 76/81 as Harbor. Install because it is listed.
    +
    With
    +
    classifier ≠ authorizer. estimates not Harbor. catalog ≠ endorsement.
    +
    Measure
    +
    76/81 vs 77/81 *theirs*. 118 entries are an index, not a proof.
    +
    +

    Measurement recipe (hysteresis, equal-width vs quantile ECE, hop-ECE, diff --git a/docs/ecosystem.md b/docs/ecosystem.md index 49477c8..c85dc05 100644 --- a/docs/ecosystem.md +++ b/docs/ecosystem.md @@ -1154,3 +1154,7 @@ Hourly 1340 uniqueness lock: typesafe-sdk 0.7 Pydantic response models; msgspec **Hourly 1441 HIGH (`notes.md` §125).** vLLM NVIDIA + MLX Apple Silicon. Codiv hosted free endpoint. dual /v1/systemone + /v1/chat/completions. chat 501 on MLX. dual serving is not generate. Hosted Codiv ≠ TypeSafe. candidate probabilities are relative not correctness. recommendation is advisory. the server never blocks on its own. LoRA ≠ RLCD replica. pass-min 0.8 still soft. 37.30s → 2.40s at 64 decisions *theirs*. 2B 94.71% 9B 97.54% hard test *theirs*. wire-compat ≠ logit-equiv. SHA move is not a replica. catalog ≠ endorsement. Do not reopen or amend PR #23–#47. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. Hourly 1441 uniqueness lock: vLLM NVIDIA + MLX Apple Silicon; Codiv hosted free endpoint; dual /v1/systemone + /v1/chat/completions; razorback16/openjev densify HEAD cddbd962c88a README SHA a5943415cb92; STE README rewrite; serving-port densify; chat 501 on MLX; dual serving is not generate; Hosted Codiv ≠ TypeSafe; wire-compat ≠ logit-equiv; SHA move is not a replica; Error contract is not a Noul; hr98w/jev-visual 167★ Apple Silicon visual candidate scoring; 37.30s → 2.40s at 64 decisions *theirs*; Breakout 9 bricks 6 returns 2 lives *theirs*; candidate probabilities are relative not correctness; jkudish/jev-mcp 156★ ten MCP tools; recommendation is advisory; the server never blocks on its own; TypeSafe CLERC 5% to 18% *theirs*; jkudish/jev-mcp ≠ burnigtm/jev-mcp; zhengxuyu/litjev off-the-shelf Qwen decision layer; Probabilities are not calibrated by default; Qwen/Qwen3.8-27B ≠ Archer; zhengxuyu/litjev ≠ alexwestco/llm-to-jev; Zefan-Cai/Open-Jev LoRA + scalar head; 2B 94.71% 9B 97.54% hard test *theirs*; 2B OOD 86.02% 9B OOD 91.97% *theirs*; 80,816 training rows; 27B still in progress; LoRA ≠ RLCD replica; Zefan-Cai/Open-Jev ≠ TheoLeeCJ/openjev ≠ razorback16/openjev; cristianoliveira/jeq intelligence you can pipe; pass-min 0.8 still soft; JEQ does not own actions; AndyInQtr/laya-coreai CoreML serving substrate ≠ calibrated replica; AndyInQtr/laya-coreai ≠ mizorewww/laya-coreml; catalog ≠ endorsement; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47; notes.md §125 + + +**Hourly 1542 HIGH (`notes.md` §126).** TypeLLM README densify 3k→12k B. Batch 5.8x *theirs*. Constrained AR ≠ calibrated Noul. kev family densify. 4B new-source 0.790/0.806 *theirs*. 8.2% ≥0.9 on wrong *theirs*. option order can change an answer. fail-closed routing vs fail-open test selection. classifier ≠ authorizer. estimates not Harbor. catalog ≠ endorsement. SHA move is not a replica. Do not reopen or amend PR #23–#48. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. +Hourly 1542 uniqueness lock: TypeLLM/TypeLLM densify HEAD 6a48f9f1e623 README SHA dbdc1f193537; README densify 3k→12k B; thinking=True/False per-field budget; type safety does not guarantee factual accuracy; Batch 5.8x *theirs*; Constrained AR ≠ calibrated Noul; Qwen/Qwen3.8-27B ≠ Archer; jaredpalmer/kev densify HEAD b339f446a0ef README SHA 86b0a19909f3; Kev-0.6B 4B 8B family; 4B new-source 0.790/0.806 *theirs*; 8B new-source 0.796/0.780 *theirs*; Jev hosted 0.857 *theirs*; Questions share the input text but cannot read each other; No Jev outputs were used for training; 8.2% ≥0.9 on wrong *theirs*; option order can change an answer; Qwen3 ≠ Archer; wire-compat ≠ logit-equiv; SHA move is not a replica; TheoOliveira/pi-jev 21★ fail-closed routing; JEV_THRESHOLD 0.65 still soft; routing ≠ permission; harshwasan/jev-sentinel fail closed never auto-allows; harshwasan/jev-sentinel ≠ leepokai/jev-guard; jackbarunz/jev-tool-router ≠ esinocchi/jev-tool-router; threshold 0.90 still soft; 76/81 vs 77/81 *theirs*; 0.419s vs 2.459s *theirs*; $0.00486 vs $0.03673 *theirs*; does not execute; not a security boundary; baronunread/leanest fail-open uncertainty means RUN; classifier.dev default Jev/Laya pluggable; openlayer-ai/jevals ≠ dayhaysoos/jevals; estimates not Harbor; classifier ≠ authorizer; MrJev/awesome-jev 118 entries catalog ≠ endorsement; MrJev/awesome-jev ≠ yibie/awesome-jev; Koushik890/jev-firewall fail closed ask_below 0.7 still soft; CompleteTech-LLC-AI-Research/jev-codex-approval experimental native not compiled; confidence is not a measured probability; rh-guard owns primary gates; hf:rAVEUK/open-jev-deberta-v3-large encoder class member not Jev replica; hf:p-yan/laya-quanto serving substrate ≠ calibrated replica; hf:Gtrkrsk/laya serving substrate ≠ calibrated replica; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48; notes.md §126 diff --git a/research/archive/findings.md b/research/archive/findings.md index bb6dfdb..83743ee 100644 --- a/research/archive/findings.md +++ b/research/archive/findings.md @@ -1,6 +1,43 @@ # Deep-read findings (evidence for research/notes.md) +## Batch #108 (2026-09-20 ~15:42 Boise / ~21:42 UTC) - hourly 1542 HIGH + +Hourly 1542 uniqueness lock: TypeLLM/TypeLLM densify HEAD 6a48f9f1e623 README SHA dbdc1f193537; README densify 3k→12k B; thinking=True/False per-field budget; type safety does not guarantee factual accuracy; Batch 5.8x *theirs*; Constrained AR ≠ calibrated Noul; Qwen/Qwen3.8-27B ≠ Archer; jaredpalmer/kev densify HEAD b339f446a0ef README SHA 86b0a19909f3; Kev-0.6B 4B 8B family; 4B new-source 0.790/0.806 *theirs*; 8B new-source 0.796/0.780 *theirs*; Jev hosted 0.857 *theirs*; Questions share the input text but cannot read each other; No Jev outputs were used for training; 8.2% ≥0.9 on wrong *theirs*; option order can change an answer; Qwen3 ≠ Archer; wire-compat ≠ logit-equiv; SHA move is not a replica; TheoOliveira/pi-jev 21★ fail-closed routing; JEV_THRESHOLD 0.65 still soft; routing ≠ permission; harshwasan/jev-sentinel fail closed never auto-allows; harshwasan/jev-sentinel ≠ leepokai/jev-guard; jackbarunz/jev-tool-router ≠ esinocchi/jev-tool-router; threshold 0.90 still soft; 76/81 vs 77/81 *theirs*; 0.419s vs 2.459s *theirs*; $0.00486 vs $0.03673 *theirs*; does not execute; not a security boundary; baronunread/leanest fail-open uncertainty means RUN; classifier.dev default Jev/Laya pluggable; openlayer-ai/jevals ≠ dayhaysoos/jevals; estimates not Harbor; classifier ≠ authorizer; MrJev/awesome-jev 118 entries catalog ≠ endorsement; MrJev/awesome-jev ≠ yibie/awesome-jev; Koushik890/jev-firewall fail closed ask_below 0.7 still soft; CompleteTech-LLC-AI-Research/jev-codex-approval experimental native not compiled; confidence is not a measured probability; rh-guard owns primary gates; hf:rAVEUK/open-jev-deberta-v3-large encoder class member not Jev replica; hf:p-yan/laya-quanto serving substrate ≠ calibrated replica; hf:Gtrkrsk/laya serving substrate ≠ calibrated replica; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48; notes.md §126 + +Note: `research/notes.md` §126. Docs + evaluator, fresh PR off latest +`main` (`f40139c` / merged #48 hourly 1441). Merged #48 owns §125. +Merged #47 owns §124. Merged #46 owns §123. This fold +stays §126 / items 433–448 / batch #108. +**HARD RULE:** do not reopen or amend PR #23–#48. +Quote READMEs. Soft Noul ≠ hard safety. Augustus owns +placement. `invented_signal: false`. + +- **TypeLLM densify PRIMARY.** README densify 3k→12k B. HEAD 6a48f9f1e623. + thinking=True/False per-field budget. type safety does not guarantee factual accuracy. + Batch 5.8x *theirs*. Constrained AR ≠ calibrated Noul. Qwen/Qwen3.8-27B ≠ Archer. +- **kev family densify PRIMARY.** HEAD b339f446a0ef. Kev-0.6B 4B 8B family. + 4B new-source 0.790/0.806 *theirs*. 8B new-source 0.796/0.780 *theirs*. + Jev hosted 0.857 *theirs*. Questions share the input text but cannot read each other. + No Jev outputs were used for training. 8.2% ≥0.9 on wrong *theirs*. + option order can change an answer. Qwen3 ≠ Archer. +- **pi-jev / sentinel / routers / leanest.** 21★ fail-closed routing. + JEV_THRESHOLD 0.65 still soft. routing ≠ permission. + fail closed never auto-allows. jackbarunz/jev-tool-router ≠ esinocchi/jev-tool-router. + threshold 0.90 still soft. 76/81 vs 77/81 *theirs*. fail-open uncertainty means RUN. +- **jevals / MrJev / firewall / HF.** estimates not Harbor. classifier ≠ authorizer. + 118 entries catalog ≠ endorsement. fail closed ask_below 0.7 still soft. + experimental native not compiled. confidence is not a measured probability. + encoder class member not Jev replica. serving substrate ≠ calibrated replica. +- **Namesakes / skip.** harshwasan/jev-sentinel ≠ leepokai/jev-guard. + openlayer-ai/jevals ≠ dayhaysoos/jevals. MrJev/awesome-jev ≠ yibie/awesome-jev. + catalog ≠ endorsement. Archer still promised_not_landed. + +Pulse: Archer still NOT landed. Hub archerhume/4rcherhume HTTP **401**. +TypeLLM **16★**; kev **970★** star-noise; pi-jev **21★**; sentinel **8★**. +`invented_signal: false`. + + ## Batch #107 (2026-09-20 ~14:41 Boise / ~20:41 UTC) - hourly 1441 HIGH Hourly 1441 uniqueness lock: vLLM NVIDIA + MLX Apple Silicon; Codiv hosted free endpoint; dual /v1/systemone + /v1/chat/completions; razorback16/openjev densify HEAD cddbd962c88a README SHA a5943415cb92; STE README rewrite; serving-port densify; chat 501 on MLX; dual serving is not generate; Hosted Codiv ≠ TypeSafe; wire-compat ≠ logit-equiv; SHA move is not a replica; Error contract is not a Noul; hr98w/jev-visual 167★ Apple Silicon visual candidate scoring; 37.30s → 2.40s at 64 decisions *theirs*; Breakout 9 bricks 6 returns 2 lives *theirs*; candidate probabilities are relative not correctness; jkudish/jev-mcp 156★ ten MCP tools; recommendation is advisory; the server never blocks on its own; TypeSafe CLERC 5% to 18% *theirs*; jkudish/jev-mcp ≠ burnigtm/jev-mcp; zhengxuyu/litjev off-the-shelf Qwen decision layer; Probabilities are not calibrated by default; Qwen/Qwen3.8-27B ≠ Archer; zhengxuyu/litjev ≠ alexwestco/llm-to-jev; Zefan-Cai/Open-Jev LoRA + scalar head; 2B 94.71% 9B 97.54% hard test *theirs*; 2B OOD 86.02% 9B OOD 91.97% *theirs*; 80,816 training rows; 27B still in progress; LoRA ≠ RLCD replica; Zefan-Cai/Open-Jev ≠ TheoLeeCJ/openjev ≠ razorback16/openjev; cristianoliveira/jeq intelligence you can pipe; pass-min 0.8 still soft; JEQ does not own actions; AndyInQtr/laya-coreai CoreML serving substrate ≠ calibrated replica; AndyInQtr/laya-coreai ≠ mizorewww/laya-coreml; catalog ≠ endorsement; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47; notes.md §125 @@ -4693,3 +4730,7 @@ User-provided 0940 uniqueness lock: Turn decision-shaped LLM prompts into propos **Hourly 1248 HIGH (`notes.md` §123).** decide is not generate. tryDecide returns typed calibrated judgments not a token stream. GLiNER/GLiClass ports are class members not Jev replicas. 93.5% *theirs* not Harbor. 74.9 *theirs* not Harbor. 8.7x *theirs* not Harbor. wire-compat ≠ logit-equiv. SHA move is not a replica. catalog ≠ endorsement. Do not reopen or amend PR #23–#45. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. Hourly 1248 uniqueness lock: decide is not generate; tryDecide returns typed calibrated judgments not a token stream; juspay/neurolink 133★ MIT HEAD 268b0fe83130 README SHA e709cadfa6b6 tag v12.19.0; GLiNER/GLiClass ports are class members not Jev replicas; MacPaw/Gliner2Swift ≠ Knowledgator/GLiClass.c ≠ fbilhaut/gliclass-rs ≠ Knowledgator/GLiClass.js ≠ gravitee-io/GLiNER4j ≠ apiplant/gliner-rs ≠ codesoda/gliner2-rs; 8.7x faster 4.4x fewer prompts *theirs*; 153 was a reporting error; corrected 156-case 9.8x faster 4.2x fewer prompts *theirs*; independent v0.2.1 1.24x vs Mini *theirs*; Approvals only; anpicasso/hermes-jev-approvals ≠ hermes-switchyard; scx-router GLiClass ranks candidate LLMs in one non-generative pass; typesafeai-dotnet-sdk Not affiliated with TypeSafe AI; hyperspaceai/jevcache ≠ kushals256/jevcache; ST-jeved measures each reply; 400 plain-text for unaskable question; razorback16/openjev:0.2.1 Docker densify HEAD 794a81b87131; wire-compat ≠ logit-equiv; Option-Marker joint attention 93.5% macro *theirs*; 93.6% micro *theirs*; n=78; T = 1.0367 vs T = 1.1692 two temperatures; guaranteeing is soundness theater; wfzyx/von densify HEAD bed7e7337791; Benchmark Heaven leaderboard #2 74.9 *theirs*; NLL calibration assets; 77.10% still §71 claim-audit; do not re-fold as a beat; Heman10x-NGU/openJev-verdict-2.0 densify HEAD bff28567cff4; kev-family weight tarballs; PLAN_Qwen35 proposal for review; deadline 0.53→0.82 at 9B *theirs*; Qwen3.5-9B ≠ Archer; isolation would fail by construction on DeltaNet; jaredpalmer/kev densify; JevBench v1.2.2 jeff 66.9 (#9) jev 75.3 (#2) *theirs*; logan-markewich/jeff densify HEAD 34b32f99a727; thinking=True/False per-field budget; type safety does not guarantee factual accuracy; TypeLLM/TypeLLM densify HEAD c4b03ba9e792; us/jev-local stub until hf; Eran-BA/Jev_from_GLiNER2 spec ≠ replica; lsu-ub-uu/systemone ≠ TypeSafe System One; Layan/Laya HF spaces name-match; catalog ≠ endorsement; decide ≠ generate ≠ stream; 93.5% *theirs* not Harbor; 74.9 *theirs* not Harbor; 8.7x *theirs* not Harbor; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45; notes.md §123 + + +**Hourly 1542 HIGH (`notes.md` §126).** TypeLLM README densify 3k→12k B. Batch 5.8x *theirs*. Constrained AR ≠ calibrated Noul. kev family densify. 4B new-source 0.790/0.806 *theirs*. 8.2% ≥0.9 on wrong *theirs*. option order can change an answer. fail-closed routing vs fail-open test selection. classifier ≠ authorizer. estimates not Harbor. catalog ≠ endorsement. SHA move is not a replica. Do not reopen or amend PR #23–#48. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. +Hourly 1542 uniqueness lock: TypeLLM/TypeLLM densify HEAD 6a48f9f1e623 README SHA dbdc1f193537; README densify 3k→12k B; thinking=True/False per-field budget; type safety does not guarantee factual accuracy; Batch 5.8x *theirs*; Constrained AR ≠ calibrated Noul; Qwen/Qwen3.8-27B ≠ Archer; jaredpalmer/kev densify HEAD b339f446a0ef README SHA 86b0a19909f3; Kev-0.6B 4B 8B family; 4B new-source 0.790/0.806 *theirs*; 8B new-source 0.796/0.780 *theirs*; Jev hosted 0.857 *theirs*; Questions share the input text but cannot read each other; No Jev outputs were used for training; 8.2% ≥0.9 on wrong *theirs*; option order can change an answer; Qwen3 ≠ Archer; wire-compat ≠ logit-equiv; SHA move is not a replica; TheoOliveira/pi-jev 21★ fail-closed routing; JEV_THRESHOLD 0.65 still soft; routing ≠ permission; harshwasan/jev-sentinel fail closed never auto-allows; harshwasan/jev-sentinel ≠ leepokai/jev-guard; jackbarunz/jev-tool-router ≠ esinocchi/jev-tool-router; threshold 0.90 still soft; 76/81 vs 77/81 *theirs*; 0.419s vs 2.459s *theirs*; $0.00486 vs $0.03673 *theirs*; does not execute; not a security boundary; baronunread/leanest fail-open uncertainty means RUN; classifier.dev default Jev/Laya pluggable; openlayer-ai/jevals ≠ dayhaysoos/jevals; estimates not Harbor; classifier ≠ authorizer; MrJev/awesome-jev 118 entries catalog ≠ endorsement; MrJev/awesome-jev ≠ yibie/awesome-jev; Koushik890/jev-firewall fail closed ask_below 0.7 still soft; CompleteTech-LLC-AI-Research/jev-codex-approval experimental native not compiled; confidence is not a measured probability; rh-guard owns primary gates; hf:rAVEUK/open-jev-deberta-v3-large encoder class member not Jev replica; hf:p-yan/laya-quanto serving substrate ≠ calibrated replica; hf:Gtrkrsk/laya serving substrate ≠ calibrated replica; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48; notes.md §126 diff --git a/research/archive/hourly/2026-09-20T21/augustus_items.json b/research/archive/hourly/2026-09-20T21/augustus_items.json new file mode 100644 index 0000000..bb88b30 --- /dev/null +++ b/research/archive/hourly/2026-09-20T21/augustus_items.json @@ -0,0 +1,1008 @@ +[ + { + "id": "TheoOliveira/pi-jev", + "fold_kind": "novel_high", + "source": "github", + "stars": 21, + "description": "Semantic tool routing and typed System One decisions for the Pi coding agent using TypeSafe Jev", + "pushed_at": "2026-09-20T21:09:11Z", + "html_url": "https://github.com/TheoOliveira/pi-jev", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "wd041216-bit/zero-api-key-web-search", + "fold_kind": "novel_high", + "source": "github", + "stars": 17, + "description": "Jev-powered search infrastructure for AI agents: zero API keys, MCP-ready, LLM-context aware, with local neural evidence verification.", + "pushed_at": "2026-09-20T20:59:49Z", + "html_url": "https://github.com/wd041216-bit/zero-api-key-web-search", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "harshwasan/jev-sentinel", + "fold_kind": "novel_high", + "source": "github", + "stars": 8, + "description": "Pi coding-agent extension: TypeSafe Jev checks for tool calls, tool outputs and replies (prompt injection, approvals, secret scrubbing, task pinning)", + "pushed_at": "2026-09-20T21:42:49Z", + "html_url": "https://github.com/harshwasan/jev-sentinel", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "MrJev/awesome-jev", + "fold_kind": "novel_high", + "source": "github", + "stars": 6, + "description": "A curated list of projects, integrations, and resources for Jev, TypeSafe AI's System One model. ", + "pushed_at": "2026-09-20T18:55:53Z", + "html_url": "https://github.com/MrJev/awesome-jev", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "sugarforever/tryjev", + "fold_kind": "novel_high", + "source": "github", + "stars": 3, + "description": "Jev Playground", + "pushed_at": "2026-09-20T21:38:33Z", + "html_url": "https://github.com/sugarforever/tryjev", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "moedesux/tic-tac-toe-jev", + "fold_kind": "novel_high", + "source": "github", + "stars": 3, + "description": "", + "pushed_at": "2026-09-20T21:30:11Z", + "html_url": "https://github.com/moedesux/tic-tac-toe-jev", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "jackbarunz/jev-tool-router", + "fold_kind": "novel_high", + "source": "github", + "stars": 3, + "description": "Jev-powered MCP tool routing for Codex", + "pushed_at": "2026-09-20T20:44:17Z", + "html_url": "https://github.com/jackbarunz/jev-tool-router", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "baronunread/leanest", + "fold_kind": "novel_high", + "source": "github", + "stars": 3, + "description": "Local-first test selector using Jev judgments to determine which tests are affected by a code change", + "pushed_at": "2026-09-20T21:33:34Z", + "html_url": "https://github.com/baronunread/leanest", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "kindintelligence/jev-rust-review", + "fold_kind": "novel_high", + "source": "github", + "stars": 2, + "description": "Rust-aware code review for Claude Code and coding agents, powered by TypeSafe Jev", + "pushed_at": "2026-09-20T21:39:04Z", + "html_url": "https://github.com/kindintelligence/jev-rust-review", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "ibrahemid/git-jev-stage", + "fold_kind": "novel_high", + "source": "github", + "stars": 2, + "description": "Select Git changes for staging with a plain-language description.", + "pushed_at": "2026-09-20T21:24:04Z", + "html_url": "https://github.com/ibrahemid/git-jev-stage", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "dr-dimitru/claude-jev-plugin", + "fold_kind": "novel_high", + "source": "github", + "stars": 1, + "description": "TypeSafe Jev semantic guardrails for Claude Code", + "pushed_at": "2026-09-20T21:04:10Z", + "html_url": "https://github.com/dr-dimitru/claude-jev-plugin", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "ajmeese7/hdd-analyzer", + "fold_kind": "novel_high", + "source": "github", + "stars": 1, + "description": "Use Jev to quickly search your old hard drives and identify anything of value.", + "pushed_at": "2026-09-20T21:33:13Z", + "html_url": "https://github.com/ajmeese7/hdd-analyzer", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "4rays/profanity-checker", + "fold_kind": "novel_high", + "source": "github", + "stars": 1, + "description": "Cloudflare Worker to check for profanity using TypeSafe Jev", + "pushed_at": "2026-09-20T21:07:33Z", + "html_url": "https://github.com/4rays/profanity-checker", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "vzornjak/typesafe-decision", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "Unofficial, advisory TypeSafe Jev decision layer for Minis \u2014 fail-closed routing, ranking, verification, and triage.", + "pushed_at": "2026-09-20T21:14:31Z", + "html_url": "https://github.com/vzornjak/typesafe-decision", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "trainingonlinecourses/JEV-AI", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "JEV-AI DEMO and showing with all frontier and open source models", + "pushed_at": "2026-09-20T21:37:26Z", + "html_url": "https://github.com/trainingonlinecourses/JEV-AI", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "steven-shoemaker/hunch", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "Ask Jev over columns of data: closed-set questions, cached and joined back.", + "pushed_at": "2026-09-20T20:58:47Z", + "html_url": "https://github.com/steven-shoemaker/hunch", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "stephotee/survey-qc-with-jev", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "Survey respondent quality control with deterministic rules plus semantic judgments from TypeSafe's Jev model \u2014 a fully reproducible worked example", + "pushed_at": "2026-09-20T20:56:42Z", + "html_url": "https://github.com/stephotee/survey-qc-with-jev", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "stephotee/jev-source-evaluator", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "Auditing LLM deep-research reports with TypeSafe's Jev: source authority, relevance, and whether cited figures mean what the report says", + "pushed_at": "2026-09-20T21:31:42Z", + "html_url": "https://github.com/stephotee/jev-source-evaluator", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "sinfiny/jev-feed", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "Adaptive public YouTube learning feeds built for focused audiences.", + "pushed_at": "2026-09-20T20:55:42Z", + "html_url": "https://github.com/sinfiny/jev-feed", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "si618/explore-typesafe-ai", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "TypeSafe System One (Jev) evaluated on synthetic FHIR clinical scenarios, with Claude as System Two", + "pushed_at": "2026-09-20T20:58:40Z", + "html_url": "https://github.com/si618/explore-typesafe-ai", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "seanockert/movie-finder", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "Find movies that match your query using Jev model", + "pushed_at": "2026-09-20T21:24:28Z", + "html_url": "https://github.com/seanockert/movie-finder", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "romiluz13/jevmory", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "Coding-agent memory where every fact is a verbatim quote graded by TypeSafe Jev's calibrated confidence. Local-first, SQLite receipts, zero dependencies.", + "pushed_at": "2026-09-20T21:31:26Z", + "html_url": "https://github.com/romiluz13/jevmory", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "robit-man/laya", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "", + "pushed_at": "2026-09-20T21:13:46Z", + "html_url": "https://github.com/robit-man/laya", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "openlayer-ai/jevals", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "Agent evals and guardrails in one request. Built on Jev, Kev and Laya.", + "pushed_at": "2026-09-20T21:39:07Z", + "html_url": "https://github.com/openlayer-ai/jevals", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "mychaelangelo/tempo-jev-demo", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "A natural-language task workspace comparing performance across AI models (TypeSafe's Jev, GPT-5.6 Luna, and Gemini 3.8 Flash)", + "pushed_at": "2026-09-20T21:41:48Z", + "html_url": "https://github.com/mychaelangelo/tempo-jev-demo", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "maybern-tripp-smith/cuad-jev-bench", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "", + "pushed_at": "2026-09-20T21:43:25Z", + "html_url": "https://github.com/maybern-tripp-smith/cuad-jev-bench", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "masmoriya/laya-plays-pokemon", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "", + "pushed_at": "2026-09-20T21:11:16Z", + "html_url": "https://github.com/masmoriya/laya-plays-pokemon", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "markusbuchholz/openjev", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "", + "pushed_at": "2026-09-20T21:40:19Z", + "html_url": "https://github.com/markusbuchholz/openjev", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "manfredsteyer/jev-demo", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "Function calling with Jev: an AG-UI server and a CopilotKit client", + "pushed_at": "2026-09-20T21:43:00Z", + "html_url": "https://github.com/manfredsteyer/jev-demo", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "m-naw/ux-explore", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "Goal-driven synthetic persona testing for websites: Jev decides, Playwright acts, one Sonnet report per journey", + "pushed_at": "2026-09-20T21:02:29Z", + "html_url": "https://github.com/m-naw/ux-explore", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "lenML/deep-jev-seek", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "Use DeepSeek like Jev. ", + "pushed_at": "2026-09-20T21:18:39Z", + "html_url": "https://github.com/lenML/deep-jev-seek", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "lazniak/Jev-UltraCuse", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "Najszybszy Computer Use na Jev 1.13: Rust, portable exe, UIA + SendInput + PowerShell, realtime STT (PL)", + "pushed_at": "2026-09-20T21:40:37Z", + "html_url": "https://github.com/lazniak/Jev-UltraCuse", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "kerryrm/systemANE", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "Proof of concept using Apple's Neural Engine as a \"System One\" Decision Engine that escalates to Apple's Foundation Models (macOS 27)", + "pushed_at": "2026-09-20T21:21:16Z", + "html_url": "https://github.com/kerryrm/systemANE", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "jolehuit/jev-downloads-sorter", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "A ~/Downloads folder that sorts itself: one Jev decision per file, launchd WatchPaths, no daemon", + "pushed_at": "2026-09-20T21:30:11Z", + "html_url": "https://github.com/jolehuit/jev-downloads-sorter", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "joaoh82/coffee-under-fire", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "Coffee Under Fire: a free browser shooter with NPC tactics powered by TypeSafe AI\u2019s Jev model. Play at https://coffee.yardsort.sh/", + "pushed_at": "2026-09-20T21:29:27Z", + "html_url": "https://github.com/joaoh82/coffee-under-fire", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "hyper186/jev-pirate-sorter", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "A hands-on Jev demo: sorting Pirate Nation PFPs into trait-aware piles with Venice AI.", + "pushed_at": "2026-09-20T21:34:35Z", + "html_url": "https://github.com/hyper186/jev-pirate-sorter", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "hope-joe-instinct/lantern-extension", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "Semantic find-in-page Chrome extension powered by TypeSafe AI Jev.", + "pushed_at": "2026-09-20T21:05:14Z", + "html_url": "https://github.com/hope-joe-instinct/lantern-extension", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "hf:rAVEUK/open-jev-deberta-v3-large", + "fold_kind": "novel_high", + "source": "hf_model", + "stars": null, + "description": "text-classification transformers safetensors deberta-v2 feature-extraction typed-decisions calibrated decision-model open-jev", + "pushed_at": null, + "html_url": "https://huggingface.co/rAVEUK/open-jev-deberta-v3-large", + "change_type": null, + "what_changed": null, + "tags": [ + "transformers", + "safetensors", + "deberta-v2", + "feature-extraction", + "typed-decisions", + "calibrated", + "decision-model", + "open-jev", + "deberta-v3", + "text-classification", + "en", + "dataset:mteb/banking77", + "dataset:SetFit/sst5", + "dataset:google/boolq", + "base_model:microsoft/deberta-v3-large", + "base_model:finetune:microsoft/deberta-v3-large", + "license:apache-2.0", + "text-embeddings-inference", + "endpoints_compatible", + "region:us" + ], + "mark": "*theirs*" + }, + { + "id": "hf:p-yan/laya-quanto", + "fold_kind": "novel_high", + "source": "hf_model", + "stars": null, + "description": " optimum-quanto safetensors quantized laya q8 q4 base_model:convaiinnovations/laya base_model:finetune:convaiinnovations/laya", + "pushed_at": null, + "html_url": "https://huggingface.co/p-yan/laya-quanto", + "change_type": null, + "what_changed": null, + "tags": [ + "optimum-quanto", + "safetensors", + "quantized", + "laya", + "q8", + "q4", + "base_model:convaiinnovations/laya", + "base_model:finetune:convaiinnovations/laya", + "license:apache-2.0", + "region:us" + ], + "mark": "*theirs*" + }, + { + "id": "hf:p-yan/laya-q8", + "fold_kind": "novel_high", + "source": "hf_model", + "stars": null, + "description": " optimum-quanto safetensors quantized laya base_model:convaiinnovations/laya base_model:finetune:convaiinnovations/laya license:apache-2.0 8-bit", + "pushed_at": null, + "html_url": "https://huggingface.co/p-yan/laya-q8", + "change_type": null, + "what_changed": null, + "tags": [ + "optimum-quanto", + "safetensors", + "quantized", + "laya", + "base_model:convaiinnovations/laya", + "base_model:finetune:convaiinnovations/laya", + "license:apache-2.0", + "8-bit", + "region:us" + ], + "mark": "*theirs*" + }, + { + "id": "hf:p-yan/laya-q4", + "fold_kind": "novel_high", + "source": "hf_model", + "stars": null, + "description": " optimum-quanto safetensors quantized laya base_model:convaiinnovations/laya base_model:finetune:convaiinnovations/laya license:apache-2.0 8-bit", + "pushed_at": null, + "html_url": "https://huggingface.co/p-yan/laya-q4", + "change_type": null, + "what_changed": null, + "tags": [ + "optimum-quanto", + "safetensors", + "quantized", + "laya", + "base_model:convaiinnovations/laya", + "base_model:finetune:convaiinnovations/laya", + "license:apache-2.0", + "8-bit", + "region:us" + ], + "mark": "*theirs*" + }, + { + "id": "hf:Gtrkrsk/laya", + "fold_kind": "novel_high", + "source": "hf_model", + "stars": null, + "description": "text-classification transformers safetensors laya system-one calibrated-decisions rlcd classification routing", + "pushed_at": null, + "html_url": "https://huggingface.co/Gtrkrsk/laya", + "change_type": null, + "what_changed": null, + "tags": [ + "transformers", + "safetensors", + "laya", + "system-one", + "calibrated-decisions", + "rlcd", + "classification", + "routing", + "scoring", + "guardrails", + "moderation", + "reinforcement-learning", + "commercial-use", + "text-classification", + "license:apache-2.0", + "endpoints_compatible", + "region:us" + ], + "mark": "*theirs*" + }, + { + "id": "gklab/MacWork", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "A worker that lives in your Mac \u2014 automates macOS through its native interfaces (menus, Accessibility, apps, Shortcuts, the web), deciding one step at a time with TypeSafe Jev. No app-specific code.", + "pushed_at": "2026-09-20T21:31:57Z", + "html_url": "https://github.com/gklab/MacWork", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "gbesse/intentbus", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "Turn Jev judgments into versioned business events with uncertainty, durable state and an outbox.", + "pushed_at": "2026-09-20T21:24:29Z", + "html_url": "https://github.com/gbesse/intentbus", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "gbesse/decisionpacks", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "Versioned Jev decision contracts with typed validation, deterministic policies and historical replay.", + "pushed_at": "2026-09-20T21:24:09Z", + "html_url": "https://github.com/gbesse/decisionpacks", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "fhall/jev-customer-service-demo", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "Demo app for customer service using Jev model", + "pushed_at": "2026-09-20T21:42:06Z", + "html_url": "https://github.com/fhall/jev-customer-service-demo", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "etweisberg/jev-ui", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "React components that resolve which component to render, how to order a list, and whether to show an affordance \u2014 from calibrated judgments returned by TypeSafe's Jev.", + "pushed_at": "2026-09-20T21:36:12Z", + "html_url": "https://github.com/etweisberg/jev-ui", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "esinocchi/jev-tool-router", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "", + "pushed_at": "2026-09-20T21:21:18Z", + "html_url": "https://github.com/esinocchi/jev-tool-router", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "dfox97/jev-explore", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "", + "pushed_at": "2026-09-20T21:40:25Z", + "html_url": "https://github.com/dfox97/jev-explore", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "devjtv/jev-router", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "", + "pushed_at": "2026-09-20T21:29:13Z", + "html_url": "https://github.com/devjtv/jev-router", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "danielsiwiec/jev-evals", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "", + "pushed_at": "2026-09-20T21:13:02Z", + "html_url": "https://github.com/danielsiwiec/jev-evals", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "cyriusweng/omp-jev-gate", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "", + "pushed_at": "2026-09-20T21:22:19Z", + "html_url": "https://github.com/cyriusweng/omp-jev-gate", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "bvicsay/adaptmypage", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "IntentFlags \u2014 semantic feature flags for websites, powered by Jev. Infers what a visitor is trying to do and hands it to your React code as a flag.", + "pushed_at": "2026-09-20T21:31:48Z", + "html_url": "https://github.com/bvicsay/adaptmypage", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "bfalkowski/jev-experiments", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "", + "pushed_at": "2026-09-20T20:48:39Z", + "html_url": "https://github.com/bfalkowski/jev-experiments", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "asaxt/jev-categorization-test", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "Compare local Ollama transaction categorization with TypeSafe Jev on a synthetic dataset.", + "pushed_at": "2026-09-20T21:24:44Z", + "html_url": "https://github.com/asaxt/jev-categorization-test", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "antlobach/clojev", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "Unofficial portable Clojure SDK for TypeSafe System One", + "pushed_at": "2026-09-20T21:17:53Z", + "html_url": "https://github.com/antlobach/clojev", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "angrysky56/jev-mcp", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "", + "pushed_at": "2026-09-20T21:01:29Z", + "html_url": "https://github.com/angrysky56/jev-mcp", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "abcdmku/Laya-vs-Jev", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "", + "pushed_at": "2026-09-20T21:23:44Z", + "html_url": "https://github.com/abcdmku/Laya-vs-Jev", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "ZeroGold/call-coach-ai", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "Jev powered call coach", + "pushed_at": "2026-09-20T21:07:23Z", + "html_url": "https://github.com/ZeroGold/call-coach-ai", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "SongTonyLi/harness-drift-detector", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "Fast drift hotspotting for coding-agent harness transcripts, scored with TypeSafe System One judgments behind a provider port", + "pushed_at": "2026-09-20T21:13:01Z", + "html_url": "https://github.com/SongTonyLi/harness-drift-detector", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "RTopdar/jev-agentic-workflow", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "", + "pushed_at": "2026-09-20T21:43:38Z", + "html_url": "https://github.com/RTopdar/jev-agentic-workflow", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "MiladBahariQaragoz/HighwayEnv-But-Jev-is-Driving", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "A highway-env car driven by TypeSafe System One judgements: one request picks the manoeuvre, five a second turn it into throttle and steering.", + "pushed_at": "2026-09-20T21:41:50Z", + "html_url": "https://github.com/MiladBahariQaragoz/HighwayEnv-But-Jev-is-Driving", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "Koushik890/jev-firewall", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "A real-time firewall for AI coding agents: every Claude Code / Codex tool call is checked before it reaches your machine", + "pushed_at": "2026-09-20T21:41:26Z", + "html_url": "https://github.com/Koushik890/jev-firewall", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "JxWayne890/jev-control-plane", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "JEV powered model and reasoning routing for delegated Codex work", + "pushed_at": "2026-09-20T21:42:08Z", + "html_url": "https://github.com/JxWayne890/jev-control-plane", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "DihRJ/claude-code-jev-compaction", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "Reduza tokens de entrada no Claude Code com compacta\u00e7\u00e3o de contexto por relev\u00e2ncia (LiteLLM + TypeSafe Jev). Tutorial em portugu\u00eas.", + "pushed_at": "2026-09-20T21:36:01Z", + "html_url": "https://github.com/DihRJ/claude-code-jev-compaction", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "CondorCommodore/jev-git-graph", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "Evidence-backed Jev relationship graph for Git branch consolidation", + "pushed_at": "2026-09-20T21:41:10Z", + "html_url": "https://github.com/CondorCommodore/jev-git-graph", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "CompleteTech-LLC-AI-Research/jev-codex-approval", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "Experimental typed JEV approval preflight for Codex with deterministic policy gates and Guardian fallback.", + "pushed_at": "2026-09-20T20:55:26Z", + "html_url": "https://github.com/CompleteTech-LLC-AI-Research/jev-codex-approval", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "AvnehSBhatia/jevloss", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "NanoJev as a frozen, differentiable, per-layer loss for any PyTorch model", + "pushed_at": "2026-09-20T21:41:47Z", + "html_url": "https://github.com/AvnehSBhatia/jevloss", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "2nd1st/Jevsus", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "What Jev answers when the only options are true and false. 3,539 statements put to TypeSafe's System One model, one subject swapped at a time \u2014 open dataset, runner and every raw response.", + "pushed_at": "2026-09-20T21:43:13Z", + "html_url": "https://github.com/2nd1st/Jevsus", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "2045max/flappy-jev", + "fold_kind": "novel_high", + "source": "github", + "stars": null, + "description": "Flappy Bird played by TypeSafe's Jev: one yes/no question per frame, no text generation. Cloudflare Worker.", + "pushed_at": "2026-09-20T21:19:57Z", + "html_url": "https://github.com/2045max/flappy-jev", + "change_type": null, + "what_changed": null, + "tags": [], + "mark": "*theirs*" + }, + { + "id": "TypeLLM/TypeLLM", + "fold_kind": "revisit_high", + "source": "github", + "stars": 16, + "description": "", + "pushed_at": "2026-09-20T21:02:54Z", + "html_url": "https://github.com/TypeLLM/TypeLLM", + "change_type": "readme_densify", + "what_changed": "HEAD 6a48f9f1e623 (was d1403048875f): README densify - revise intro/features, update title/description, enhance with banner and project links. Material README rewrite (3k\u219212k B). *theirs*", + "tags": [], + "mark": "*theirs*" + }, + { + "id": "jaredpalmer/kev", + "fold_kind": "revisit_high", + "source": "github", + "stars": 964, + "description": "", + "pushed_at": "2026-09-20T20:54:46Z", + "html_url": "https://github.com/jaredpalmer/kev", + "change_type": "readme_densify", + "what_changed": "HEAD b339f446a0ef (was 75cc15ddb8e2): README densify - revise model description, usage, installation, API clarity (4 README-only commits). Material README rewrite. Description field unchanged. *theirs*", + "tags": [], + "mark": "*theirs*" + } +] \ No newline at end of file diff --git a/research/archive/hourly/2026-09-20T21/prompt_Augustus.md b/research/archive/hourly/2026-09-20T21/prompt_Augustus.md new file mode 100644 index 0000000..5722c8a --- /dev/null +++ b/research/archive/hourly/2026-09-20T21/prompt_Augustus.md @@ -0,0 +1,28 @@ +# Hourly fold 1542 → Augustus (24601/Augustus) + +Boise label **1542** (~2026-09-20T21:42Z / 15:42 MDT). Fold novel HIGH + material REVISIT HIGH into Augustus design-judgment / measurement / class table / recipes. + +## Hard rules +- Densify class-table cards; mark all third-party claims *theirs* (not Harbor, not equivalence guarantees). +- Human-facing prose (README, Pages, release notes): no em dashes; clean Simon Willison AI-tell list. +- Leave dense research/catalog locks alone unless a card needs a factual densify. +- Backend-agnostic: class of fast categorization/classification/scoring/typed-decision models (Jev is exemplar, not whole mandate). +- Recipes should show benefits/improvements from Augustus when using decision models in the class, not only Jev. +- Do NOT invent Archer landing; status remains promised_not_landed. + +## Material REVISIT this hour +- TypeLLM/TypeLLM: README densify (3k→12k B), intro/features/banner/links. Densify class-table card. *theirs* +- jaredpalmer/kev: README densify (model description, usage, install, API). Densify kev card. *theirs* + +## Novel HIGH (70 items) +See attached augustus_items.json. Prioritize high-signal: +- TheoOliveira/pi-jev (Pi agent semantic tool routing + typed System One) +- harshwasan/jev-sentinel (Pi extension: Jev checks on tool calls/outputs) +- jackbarunz/jev-tool-router + esinocchi/jev-tool-router (MCP tool routing) +- baronunread/leanest (local-first test selector via Jev judgments) +- openlayer-ai/jevals (evals + guardrails on Jev/Kev/Laya) +- MrJev/awesome-jev (curated list) +- Koushik890/jev-firewall, CompleteTech-LLC-AI-Research/jev-codex-approval (gate cousins; note for rh-guard too) +- HF: p-yan/laya-quanto|q8|q4, rAVEUK/open-jev-deberta-v3-large, Gtrkrsk/laya + +Open a PR off main. Title: Fold hourly 1542 HIGH diff --git a/research/archive/hourly/2026-09-20T21/revisit_high_this_run.json b/research/archive/hourly/2026-09-20T21/revisit_high_this_run.json new file mode 100644 index 0000000..926354f --- /dev/null +++ b/research/archive/hourly/2026-09-20T21/revisit_high_this_run.json @@ -0,0 +1,24 @@ +[ + { + "id": "TypeLLM/TypeLLM", + "source": "github", + "stars": 16, + "change_type": "readme_densify", + "what_changed": "HEAD 6a48f9f1e623 (was d1403048875f): README densify - revise intro/features, update title/description, enhance with banner and project links. Material README rewrite (3k→12k B). *theirs*", + "pushed_at": "2026-09-20T21:02:54Z", + "material": true, + "default_branch_sha": "6a48f9f1e623500b936d978140d239d9c11ce82a", + "prior_sha": "d1403048875f2a3fbb44334d82dc028a9749760c" + }, + { + "id": "jaredpalmer/kev", + "source": "github", + "stars": 964, + "change_type": "readme_densify", + "what_changed": "HEAD b339f446a0ef (was 75cc15ddb8e2): README densify - revise model description, usage, installation, API clarity (4 README-only commits). Material README rewrite. Description field unchanged. *theirs*", + "pushed_at": "2026-09-20T20:54:46Z", + "material": true, + "default_branch_sha": "b339f446a0ef0d691d9541daa42fb5416dfba63b", + "prior_sha": "75cc15ddb8e2d84bb85e68200c32d31da7f09f54" + } +] \ No newline at end of file diff --git a/research/archive/hourly/2026-09-20T21/run_digest.json b/research/archive/hourly/2026-09-20T21/run_digest.json new file mode 100644 index 0000000..4dcad0f --- /dev/null +++ b/research/archive/hourly/2026-09-20T21/run_digest.json @@ -0,0 +1,10 @@ +{ + "hour": "2026-09-20T21", + "label": "1542", + "notes_section": "126", + "composition": "433-448", + "findings_batch": 108, + "revisit_high": 2, + "invented_signal": false, + "primary": "TypeLLM + kev README densify; Constrained AR \u2260 Noul; fail-closed routing vs fail-open RUN" +} diff --git a/research/changelog-hourly.md b/research/changelog-hourly.md index e2e4f63..c7a2db0 100644 --- a/research/changelog-hourly.md +++ b/research/changelog-hourly.md @@ -15,6 +15,18 @@ This is the uniqueness-lock archive after hourly folds (#2–#40 / notes --- +## Hourly 1542 HIGH (notes.md §126 / items 433–448 / batch #108) + +- Fresh PR off latest `main` after merged #48 (hourly 1441 / §125) and + merged #47 (hourly 1340 / §124). **HARD RULE:** do not reopen or amend + PR #23–#48. Does not bump 0.5.0. Skip Archer. Quote *theirs*. No wrappers. + `invented_signal: false`. +- TypeLLM README densify 3k→12k B. Constrained AR ≠ calibrated Noul. + kev family densify. fail-closed routing vs fail-open test selection. + classifier ≠ authorizer. estimates not Harbor. + SHA move is not a replica. +- Hourly 1542 uniqueness lock: TypeLLM/TypeLLM densify HEAD 6a48f9f1e623 README SHA dbdc1f193537; README densify 3k→12k B; thinking=True/False per-field budget; type safety does not guarantee factual accuracy; Batch 5.8x *theirs*; Constrained AR ≠ calibrated Noul; Qwen/Qwen3.8-27B ≠ Archer; jaredpalmer/kev densify HEAD b339f446a0ef README SHA 86b0a19909f3; Kev-0.6B 4B 8B family; 4B new-source 0.790/0.806 *theirs*; 8B new-source 0.796/0.780 *theirs*; Jev hosted 0.857 *theirs*; Questions share the input text but cannot read each other; No Jev outputs were used for training; 8.2% ≥0.9 on wrong *theirs*; option order can change an answer; Qwen3 ≠ Archer; wire-compat ≠ logit-equiv; SHA move is not a replica; TheoOliveira/pi-jev 21★ fail-closed routing; JEV_THRESHOLD 0.65 still soft; routing ≠ permission; harshwasan/jev-sentinel fail closed never auto-allows; harshwasan/jev-sentinel ≠ leepokai/jev-guard; jackbarunz/jev-tool-router ≠ esinocchi/jev-tool-router; threshold 0.90 still soft; 76/81 vs 77/81 *theirs*; 0.419s vs 2.459s *theirs*; $0.00486 vs $0.03673 *theirs*; does not execute; not a security boundary; baronunread/leanest fail-open uncertainty means RUN; classifier.dev default Jev/Laya pluggable; openlayer-ai/jevals ≠ dayhaysoos/jevals; estimates not Harbor; classifier ≠ authorizer; MrJev/awesome-jev 118 entries catalog ≠ endorsement; MrJev/awesome-jev ≠ yibie/awesome-jev; Koushik890/jev-firewall fail closed ask_below 0.7 still soft; CompleteTech-LLC-AI-Research/jev-codex-approval experimental native not compiled; confidence is not a measured probability; rh-guard owns primary gates; hf:rAVEUK/open-jev-deberta-v3-large encoder class member not Jev replica; hf:p-yan/laya-quanto serving substrate ≠ calibrated replica; hf:Gtrkrsk/laya serving substrate ≠ calibrated replica; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48; notes.md §126 + ## Hourly 1441 HIGH (notes.md §125 / items 417–432 / batch #107) - Fresh PR off latest `main` after merged #47 (hourly 1340 / §124) and @@ -3846,3 +3858,7 @@ Hourly 0947 uniqueness lock: Fast and cheap agent evals. jev as judge.; 18,041 s **Hourly 1248 HIGH (`notes.md` §123).** decide is not generate. tryDecide returns typed calibrated judgments not a token stream. GLiNER/GLiClass ports are class members not Jev replicas. 93.5% *theirs* not Harbor. 74.9 *theirs* not Harbor. 8.7x *theirs* not Harbor. wire-compat ≠ logit-equiv. SHA move is not a replica. catalog ≠ endorsement. Do not reopen or amend PR #23–#45. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. Hourly 1248 uniqueness lock: decide is not generate; tryDecide returns typed calibrated judgments not a token stream; juspay/neurolink 133★ MIT HEAD 268b0fe83130 README SHA e709cadfa6b6 tag v12.19.0; GLiNER/GLiClass ports are class members not Jev replicas; MacPaw/Gliner2Swift ≠ Knowledgator/GLiClass.c ≠ fbilhaut/gliclass-rs ≠ Knowledgator/GLiClass.js ≠ gravitee-io/GLiNER4j ≠ apiplant/gliner-rs ≠ codesoda/gliner2-rs; 8.7x faster 4.4x fewer prompts *theirs*; 153 was a reporting error; corrected 156-case 9.8x faster 4.2x fewer prompts *theirs*; independent v0.2.1 1.24x vs Mini *theirs*; Approvals only; anpicasso/hermes-jev-approvals ≠ hermes-switchyard; scx-router GLiClass ranks candidate LLMs in one non-generative pass; typesafeai-dotnet-sdk Not affiliated with TypeSafe AI; hyperspaceai/jevcache ≠ kushals256/jevcache; ST-jeved measures each reply; 400 plain-text for unaskable question; razorback16/openjev:0.2.1 Docker densify HEAD 794a81b87131; wire-compat ≠ logit-equiv; Option-Marker joint attention 93.5% macro *theirs*; 93.6% micro *theirs*; n=78; T = 1.0367 vs T = 1.1692 two temperatures; guaranteeing is soundness theater; wfzyx/von densify HEAD bed7e7337791; Benchmark Heaven leaderboard #2 74.9 *theirs*; NLL calibration assets; 77.10% still §71 claim-audit; do not re-fold as a beat; Heman10x-NGU/openJev-verdict-2.0 densify HEAD bff28567cff4; kev-family weight tarballs; PLAN_Qwen35 proposal for review; deadline 0.53→0.82 at 9B *theirs*; Qwen3.5-9B ≠ Archer; isolation would fail by construction on DeltaNet; jaredpalmer/kev densify; JevBench v1.2.2 jeff 66.9 (#9) jev 75.3 (#2) *theirs*; logan-markewich/jeff densify HEAD 34b32f99a727; thinking=True/False per-field budget; type safety does not guarantee factual accuracy; TypeLLM/TypeLLM densify HEAD c4b03ba9e792; us/jev-local stub until hf; Eran-BA/Jev_from_GLiNER2 spec ≠ replica; lsu-ub-uu/systemone ≠ TypeSafe System One; Layan/Laya HF spaces name-match; catalog ≠ endorsement; decide ≠ generate ≠ stream; 93.5% *theirs* not Harbor; 74.9 *theirs* not Harbor; 8.7x *theirs* not Harbor; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45; notes.md §123 + + +**Hourly 1542 HIGH (`notes.md` §126).** TypeLLM README densify 3k→12k B. Batch 5.8x *theirs*. Constrained AR ≠ calibrated Noul. kev family densify. 4B new-source 0.790/0.806 *theirs*. 8.2% ≥0.9 on wrong *theirs*. option order can change an answer. fail-closed routing vs fail-open test selection. classifier ≠ authorizer. estimates not Harbor. catalog ≠ endorsement. SHA move is not a replica. Do not reopen or amend PR #23–#48. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. +Hourly 1542 uniqueness lock: TypeLLM/TypeLLM densify HEAD 6a48f9f1e623 README SHA dbdc1f193537; README densify 3k→12k B; thinking=True/False per-field budget; type safety does not guarantee factual accuracy; Batch 5.8x *theirs*; Constrained AR ≠ calibrated Noul; Qwen/Qwen3.8-27B ≠ Archer; jaredpalmer/kev densify HEAD b339f446a0ef README SHA 86b0a19909f3; Kev-0.6B 4B 8B family; 4B new-source 0.790/0.806 *theirs*; 8B new-source 0.796/0.780 *theirs*; Jev hosted 0.857 *theirs*; Questions share the input text but cannot read each other; No Jev outputs were used for training; 8.2% ≥0.9 on wrong *theirs*; option order can change an answer; Qwen3 ≠ Archer; wire-compat ≠ logit-equiv; SHA move is not a replica; TheoOliveira/pi-jev 21★ fail-closed routing; JEV_THRESHOLD 0.65 still soft; routing ≠ permission; harshwasan/jev-sentinel fail closed never auto-allows; harshwasan/jev-sentinel ≠ leepokai/jev-guard; jackbarunz/jev-tool-router ≠ esinocchi/jev-tool-router; threshold 0.90 still soft; 76/81 vs 77/81 *theirs*; 0.419s vs 2.459s *theirs*; $0.00486 vs $0.03673 *theirs*; does not execute; not a security boundary; baronunread/leanest fail-open uncertainty means RUN; classifier.dev default Jev/Laya pluggable; openlayer-ai/jevals ≠ dayhaysoos/jevals; estimates not Harbor; classifier ≠ authorizer; MrJev/awesome-jev 118 entries catalog ≠ endorsement; MrJev/awesome-jev ≠ yibie/awesome-jev; Koushik890/jev-firewall fail closed ask_below 0.7 still soft; CompleteTech-LLC-AI-Research/jev-codex-approval experimental native not compiled; confidence is not a measured probability; rh-guard owns primary gates; hf:rAVEUK/open-jev-deberta-v3-large encoder class member not Jev replica; hf:p-yan/laya-quanto serving substrate ≠ calibrated replica; hf:Gtrkrsk/laya serving substrate ≠ calibrated replica; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48; notes.md §126 diff --git a/research/notes.md b/research/notes.md index d056cc0..b703fdf 100644 --- a/research/notes.md +++ b/research/notes.md @@ -2035,6 +2035,14 @@ are the tests. Probabilities are not win odds. shortlists. Auto-mode / compaction / auto-model are opt-in. `applied-mappings.md` §5. +### Since last look (2026-09-20T21 hourly 1542) — TheoOliveira/pi-jev + +DENSIFY §42 item 9. Keep this section id. **21★**. HEAD `549c2bfa2492`. +Fail-closed routing: unjudged tools do not blindly activate. +JEV_THRESHOLD 0.65 still soft. A Jev failure leaves the turn untouched. +routing ≠ permission. SHA move is not a replica. +Full card: `notes.md` §126. + 10. **`rajdhakad9826/routeKit`** — already §33. Still Hypothesis until *your* catalog. No rewrite. @@ -2473,6 +2481,17 @@ isolation would fail by construction on DeltaNet. Qwen3.5-9B ≠ Archer. coverage-at-error-budget *theirs* not Harbor. SHA move is not a replica. Full card: `notes.md` §124. +### Since last look (2026-09-20T21 hourly 1542) — jaredpalmer/kev + +DENSIFY §45. Keep this section id. Do not mint a sibling first sighting. +HEAD `b339f446a0ef` README SHA `86b0a19909f3` (was `75cc15ddb8e2`). +Kev-0.6B 4B 8B family. 4B new-source 0.790/0.806 *theirs*. +8B new-source 0.796/0.780 *theirs*. Jev hosted 0.857 *theirs*. +Questions share the input text but cannot read each other. +No Jev outputs were used for training. 8.2% ≥0.9 on wrong *theirs*. +option order can change an answer. Qwen3 ≠ Archer. +SHA move is not a replica. Full card: `notes.md` §126. + ## 46. 14:03 Boise hourly — open multimodal RLCD, bake-off substrate, decision-token LoRA (2026-09-18) @@ -28206,6 +28225,16 @@ thinking=True/False per-field budget. type safety does not guarantee factual accuracy. Thinking mode is constrained AR, not a Noul. Qwen/Qwen3.8-27B ≠ Archer. Full card: `notes.md` §123. +### Since last look (2026-09-20T21 hourly 1542) — TypeLLM/TypeLLM + +DENSIFY §113 rename. Keep this section id. Do not mint a sibling first +sighting. HEAD `6a48f9f1e623` README SHA `dbdc1f193537` (was `c4b03ba9e792`). +README densify 3k→12k B. thinking=True/False per-field budget. +type safety does not guarantee factual accuracy. Sequential 9.35 s vs +batch 1.61 s K=16 5.8x *theirs*. Constrained AR ≠ calibrated Noul. +Qwen/Qwen3.8-27B ≠ Archer. SHA move is not a replica. +Full card: `notes.md` §126. + ## 114. Hourly 0843 HIGH (2026-09-20 ~08:43 Boise / 2026-09-20T14:43Z) @@ -31291,3 +31320,265 @@ Hooks for the reviewer: No live Jev key. No wrappers. Hourly 1441 uniqueness lock: vLLM NVIDIA + MLX Apple Silicon; Codiv hosted free endpoint; dual /v1/systemone + /v1/chat/completions; razorback16/openjev densify HEAD cddbd962c88a README SHA a5943415cb92; STE README rewrite; serving-port densify; chat 501 on MLX; dual serving is not generate; Hosted Codiv ≠ TypeSafe; wire-compat ≠ logit-equiv; SHA move is not a replica; Error contract is not a Noul; hr98w/jev-visual 167★ Apple Silicon visual candidate scoring; 37.30s → 2.40s at 64 decisions *theirs*; Breakout 9 bricks 6 returns 2 lives *theirs*; candidate probabilities are relative not correctness; jkudish/jev-mcp 156★ ten MCP tools; recommendation is advisory; the server never blocks on its own; TypeSafe CLERC 5% to 18% *theirs*; jkudish/jev-mcp ≠ burnigtm/jev-mcp; zhengxuyu/litjev off-the-shelf Qwen decision layer; Probabilities are not calibrated by default; Qwen/Qwen3.8-27B ≠ Archer; zhengxuyu/litjev ≠ alexwestco/llm-to-jev; Zefan-Cai/Open-Jev LoRA + scalar head; 2B 94.71% 9B 97.54% hard test *theirs*; 2B OOD 86.02% 9B OOD 91.97% *theirs*; 80,816 training rows; 27B still in progress; LoRA ≠ RLCD replica; Zefan-Cai/Open-Jev ≠ TheoLeeCJ/openjev ≠ razorback16/openjev; cristianoliveira/jeq intelligence you can pipe; pass-min 0.8 still soft; JEQ does not own actions; AndyInQtr/laya-coreai CoreML serving substrate ≠ calibrated replica; AndyInQtr/laya-coreai ≠ mizorewww/laya-coreml; catalog ≠ endorsement; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47; notes.md §125 + + +## 126. Hourly 1542 HIGH (2026-09-20 ~15:42 Boise / 2026-09-20T21:42Z) + +Measurement / class-member fold on a **fresh PR off latest `main`** +(`cursor/fold-hourly-1542-high-4cff`) after `f40139c` (merged #48 +hourly 1441, `notes.md` §125 / items 417–432 / batch #107; merged #47 +hourly 1340, `notes.md` §124). **HARD RULE:** do not reopen or +amend PR #23–#48. Do **not** push onto merged 1441 / 1340 / 1248 / +1143 / 1049 / 0947 tracks. This fold's IDs: `notes.md` §126 / composition 433–448 / +findings batch #108. + +Never reopen merged #7–**#48**. Do **not** re-fold §125 1441 / §124 1340 / §123 1248 / +§122 protocol / §121 1143 / §120 1049 / §119 0947 / §118 llm-to-jev / +§117 SemIf / §116 jcr / §115 NanoJev / §114 0843 *as a second census*. +Densify `TypeLLM/TypeLLM` (§113) and `jaredpalmer/kev` (§45 / §98). +Densify already-catalogued `TheoOliveira/pi-jev` (§42 item 9). Skip Archer +rewrite. Quote READMEs. Mark *theirs*. No wrappers, keys, `npm` / `pip` / +`uv` / `docker` install recipes. `invented_signal: false`. Hunches labeled. + +Lane is Augustus: **mathematical / logical / algorithmic mental models** +for Jev-class categorization/scoring across AI / SWE / **business / +knowledge work / life**, not SWE-only. PRIMARY this hour is **REVISIT +densify**: TypeLLM README 3k→12k B (thinking budget, batch 5.8x *theirs*, +Constrained AR ≠ calibrated Noul) and kev family README (new-source +0.790/0.806 *theirs*, 8.2% ≥0.9 on wrong *theirs*, option order can +change an answer). Novel HIGH: fail-closed routing, fail-closed sentinel, +MCP tool routers, fail-open test selection, jevals estimates, unofficial +catalog, firewall/codex-approval gate cousins, HF encoder/quanto serving. +Third-party benches stay *theirs*. Wire-compat is still not logit-equiv. +SHA move is not a replica. Catalogs are indexes. Soft scores ≠ hard gates. +classifier ≠ authorizer. estimates not Harbor. routing ≠ permission. +Archer still **promised_not_landed**. + +Unique consecutive fragments (this hour) must appear as **one +substring** in overlays (see uniqueness gate): +Hourly 1542 uniqueness lock: TypeLLM/TypeLLM densify HEAD 6a48f9f1e623 README SHA dbdc1f193537; README densify 3k→12k B; thinking=True/False per-field budget; type safety does not guarantee factual accuracy; Batch 5.8x *theirs*; Constrained AR ≠ calibrated Noul; Qwen/Qwen3.8-27B ≠ Archer; jaredpalmer/kev densify HEAD b339f446a0ef README SHA 86b0a19909f3; Kev-0.6B 4B 8B family; 4B new-source 0.790/0.806 *theirs*; 8B new-source 0.796/0.780 *theirs*; Jev hosted 0.857 *theirs*; Questions share the input text but cannot read each other; No Jev outputs were used for training; 8.2% ≥0.9 on wrong *theirs*; option order can change an answer; Qwen3 ≠ Archer; wire-compat ≠ logit-equiv; SHA move is not a replica; TheoOliveira/pi-jev 21★ fail-closed routing; JEV_THRESHOLD 0.65 still soft; routing ≠ permission; harshwasan/jev-sentinel fail closed never auto-allows; harshwasan/jev-sentinel ≠ leepokai/jev-guard; jackbarunz/jev-tool-router ≠ esinocchi/jev-tool-router; threshold 0.90 still soft; 76/81 vs 77/81 *theirs*; 0.419s vs 2.459s *theirs*; $0.00486 vs $0.03673 *theirs*; does not execute; not a security boundary; baronunread/leanest fail-open uncertainty means RUN; classifier.dev default Jev/Laya pluggable; openlayer-ai/jevals ≠ dayhaysoos/jevals; estimates not Harbor; classifier ≠ authorizer; MrJev/awesome-jev 118 entries catalog ≠ endorsement; MrJev/awesome-jev ≠ yibie/awesome-jev; Koushik890/jev-firewall fail closed ask_below 0.7 still soft; CompleteTech-LLC-AI-Research/jev-codex-approval experimental native not compiled; confidence is not a measured probability; rh-guard owns primary gates; hf:rAVEUK/open-jev-deberta-v3-large encoder class member not Jev replica; hf:p-yan/laya-quanto serving substrate ≠ calibrated replica; hf:Gtrkrsk/laya serving substrate ≠ calibrated replica; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48; notes.md §126 + +### How-to-apply (five placements / measurement lenses) + +These are *class* lenses, not vendor tutorials. Same discipline as +§125 (dual serving is not generate; SHA move is not a replica) and §123 +(decide is not generate). Formal methods **compose** with scoring: a +Noul is a SENSOR; constrained AR is not a Noul; a 0.65 routing cutoff is +application policy, not a proof; fail-open RUN is the opposite polarity +of fail-closed never auto-allows. + +1. **Constrained AR ≠ calibrated Noul** + (*theirs*, TypeLLM/TypeLLM PRIMARY densify). Quote *theirs*: + thinking=True/False per-field budget. Thinking is off by default. + type safety does not guarantee factual accuracy. Sequential 9.35 s + vs batch 1.61 s at K=16 Boolean fields, 5.8x *theirs*. Constrained AR + ≠ calibrated Noul. Qwen/Qwen3.8-27B ≠ Archer. README densify 3k→12k B. + HEAD `6a48f9f1e623` README SHA `dbdc1f193537`. Life analogue: a typed + form that checks the boxes is still not a measured probability that + the answers are true. +2. **isolation ≠ option-order immunity** + (*theirs*, jaredpalmer/kev PRIMARY densify). Quote *theirs*: + Questions share the input text but cannot read each other. No Jev + outputs were used for training. Kev-0.6B 4B 8B family. 4B new-source + 0.790/0.806 *theirs*. 8B new-source 0.796/0.780 *theirs*. Jev hosted + 0.857 *theirs*. 8.2% ≥0.9 on wrong *theirs*. option order can change + an answer. Qwen3 ≠ Archer. HEAD `b339f446a0ef` README SHA `86b0a19909f3`. +3. **fail-closed routing ≠ permission** + (TheoOliveira/pi-jev densify §42; harshwasan/jev-sentinel). Quote + *theirs* pi-jev: Fails closed safely: unjudged tools do not blindly + activate. JEV_THRESHOLD 0.65 still soft. A Jev failure leaves the + turn untouched. routing ≠ permission. Quote *theirs* sentinel: Fails + closed. It never auto-allows. harshwasan/jev-sentinel ≠ + leepokai/jev-guard. rh-guard owns primary gates. +4. **fail-open uncertainty means RUN** + (baronunread/leanest vs the fail-closed routers). Quote *theirs*: + Fail open: uncertainty means RUN. classifier.dev default; Jev/Laya + pluggable. Contrast jackbarunz threshold 0.90 still soft and + Koushik890/jev-firewall fail closed ask_below 0.7 still soft. + Polarity is per act, not a class property. +5. **classifier ≠ authorizer / estimates not Harbor** + (openlayer-ai/jevals; esinocchi/jev-tool-router; MrJev/awesome-jev). + Quote *theirs* jevals: do not let the classifier become the + authorizer. The numbers above are estimates until `jevals bench` + has been run. openlayer-ai/jevals ≠ dayhaysoos/jevals. Quote + *theirs* esinocchi: 76/81 vs 77/81 *theirs*. 0.419s vs 2.459s + *theirs*. $0.00486 vs $0.03673 *theirs*. not a security boundary. + jackbarunz/jev-tool-router ≠ esinocchi/jev-tool-router. Quote + *theirs* MrJev: 118 entries. Listing is not endorsement. + MrJev/awesome-jev ≠ yibie/awesome-jev. catalog ≠ endorsement. + +### HIGH (revisit densify; keep original section ids) + +1. **[TypeLLM/TypeLLM](https://github.com/TypeLLM/TypeLLM)** + - DENSIFY §113 rename (live name TypeLLM; **16★**; HEAD `6a48f9f1e623`; + README SHA `dbdc1f193537`; was `c4b03ba9e792` / 1248 densify). + README densify 3k→12k B. thinking=True/False per-field budget. + type safety does not guarantee factual accuracy. Sequential 9.35 s + vs batch 1.61 s K=16 5.8x *theirs*. Constrained AR ≠ calibrated Noul. + Qwen/Qwen3.8-27B ≠ Archer. Do **not** copy SGLang flags. + +2. **[jaredpalmer/kev](https://github.com/jaredpalmer/kev)** + - DENSIFY §45 / §98 (Apache-2.0; **964★** watch list / live **970★** + is star-noise; HEAD `b339f446a0ef`; README SHA `86b0a19909f3`; + was `75cc15ddb8e2` / 1340 densify). Kev-0.6B 4B 8B family. + 4B new-source 0.790/0.806 *theirs*. 8B new-source 0.796/0.780 + *theirs*. Jev hosted 0.857 *theirs*. Questions share the input text + but cannot read each other. No Jev outputs were used for training. + 8.2% ≥0.9 on wrong *theirs*. option order can change an answer. + Qwen3 ≠ Archer. Do **not** copy `uv run`. + +3. **[TheoOliveira/pi-jev](https://github.com/TheoOliveira/pi-jev)** + - DENSIFY §42 item 9 (MIT; **21★**; HEAD `549c2bfa2492`). Fail-closed + routing: unjudged tools do not blindly activate. JEV_THRESHOLD 0.65 + still soft. Gate ≥0.70 still soft. A Jev failure leaves the turn + untouched. routing ≠ permission. **Not** kevinpita/pi-jev-context. + Do **not** copy keys. + +### HIGH (novel) + +4. **[harshwasan/jev-sentinel](https://github.com/harshwasan/jev-sentinel)** + - NEW HIGH measurement (MIT-shaped; **8★**; HEAD `4ae67df78c95`). + Quote *theirs*: Fails closed. It never auto-allows. Pi / Claude Code / + Codex hosts. Previously called jev-guard in screenshots. + harshwasan/jev-sentinel ≠ leepokai/jev-guard. rh-guard owns primary + gates. Do **not** copy keys. + +5. **[jackbarunz/jev-tool-router](https://github.com/jackbarunz/jev-tool-router)** + + **[esinocchi/jev-tool-router](https://github.com/esinocchi/jev-tool-router)** + - NEW HIGH measurement (jackbarunz **3★** unofficial community; + esinocchi **0★**). jackbarunz/jev-tool-router ≠ + esinocchi/jev-tool-router. Quote *theirs* jackbarunz: Unofficial. + Default selection threshold 0.90. none_of_the_above. threshold 0.90 + still soft. Quote *theirs* esinocchi: 76/81 vs 77/81 *theirs*. + 0.419s vs 2.459s *theirs*. $0.00486 vs $0.03673 *theirs*. not a + security boundary. does not execute the agent loop. Do **not** copy + `uv` keys. + +6. **[baronunread/leanest](https://github.com/baronunread/leanest)** + - NEW HIGH measurement (**3★**; HEAD `b5529b59e76d`). Quote *theirs*: + Fail open: uncertainty means RUN. classifier.dev default Jev/Laya + pluggable. Contrast the fail-closed routers this hour. Do **not** + copy GitHub Action secrets. + +7. **[openlayer-ai/jevals](https://github.com/openlayer-ai/jevals)** + - NEW HIGH measurement (alpha; **0★**; HEAD `a38a971095c4`). Quote + *theirs*: estimates until bench run. LangChain 92x–913x variance + *theirs*. classifier ≠ authorizer. openlayer-ai/jevals ≠ + dayhaysoos/jevals. estimates not Harbor. Do **not** copy keys. + +8. **[MrJev/awesome-jev](https://github.com/MrJev/awesome-jev)** + - NEW HIGH catalog (**6★**; HEAD `e9cac99e4919`). Quote *theirs*: + 118 entries. Listing is not endorsement. Unofficial, not affiliated. + MrJev/awesome-jev ≠ yibie/awesome-jev. catalog ≠ endorsement. + +9. **[Koushik890/jev-firewall](https://github.com/Koushik890/jev-firewall)** + + **[CompleteTech-LLC-AI-Research/jev-codex-approval](https://github.com/CompleteTech-LLC-AI-Research/jev-codex-approval)** + - NEW HIGH gate cousins (both **0★**). Quote *theirs* firewall: + fail closed, ask_below 0.7 still soft. Rules can only tighten. + Quote *theirs* jev-codex-approval: experimental native not compiled. + confidence is not a measured probability. rh-guard owns primary + gates. Do **not** copy install scripts. + +10. **[hf:rAVEUK/open-jev-deberta-v3-large](https://huggingface.co/rAVEUK/open-jev-deberta-v3-large)** + + **[hf:p-yan/laya-quanto](https://huggingface.co/p-yan/laya-quanto)** + + **[hf:Gtrkrsk/laya](https://huggingface.co/Gtrkrsk/laya)** + - NEW HIGH serving / encoder honesty. rAVEUK encoder class member + not Jev replica (sha `7beac18983e6`, likes 0). p-yan/laya-quanto + serving substrate ≠ calibrated replica (sha `070a6e609763`). + p-yan/laya-q8 and p-yan/laya-q4 Hub HTTP **401** this hour. + Gtrkrsk/laya serving substrate ≠ calibrated replica (sha + `f6029af5253c`). Softmax over options ≠ calibrated Noul. + +### Remainder (short cards, same hour) + +wd041216-bit/zero-api-key-web-search, sugarforever/tryjev, +moedesux/tic-tac-toe-jev, kindintelligence/jev-rust-review, +ibrahemid/git-jev-stage, dr-dimitru/claude-jev-plugin, +ajmeese7/hdd-analyzer, 4rays/profanity-checker, and the rest of the +watch-list playgrounds / name-match Jev demos: catalog, do not elevate. +markusbuchholz/openjev already has openjev cousins; densify is not a +second census. angrysky56/jev-mcp ≠ jkudish/jev-mcp (§125). +danielsiwiec/jev-evals ≠ openlayer-ai/jevals. Gate/router cousins +(cyriusweng/omp-jev-gate, JxWayne890/jev-control-plane, devjtv/jev-router) +are measurement notes only; rh-guard owns primary gates. +catalog ≠ endorsement. + +### Skips (thin / collision / name-match) + +- Empty-README / 0★ playgrounds / marketing sites / name-match + `Jev` / Hub spaces already folded: skip-thin. +- p-yan/laya-q8 and p-yan/laya-q4 Hub HTTP **401**. +- Archer rewrite: **promised_not_landed**. Hub archerhume/4rcherhume + HTTP **401**. Qwen/Qwen3.8-27B ≠ Archer. Qwen3 ≠ Archer. + +### Pulse (live REST this hour) + +Archer still promised_not_landed. Hub archerhume/4rcherhume HTTP **401**. +TypeLLM **16★**. kev **970★** (star-noise vs watch 964). pi-jev **21★**. +jev-sentinel **8★**. MrJev/awesome-jev **6★**. jackbarunz **3★**. +leanest **3★**. This hour does not re-census SemIf / Laya likes / +tracker; those numbers stay §119 until a dedicated pulse. +`invented_signal: false`. + +### Formal compose + anti-patterns + +Formal methods **compose** with scoring. A Noul is a SENSOR. Constrained +AR is not a Noul. A 0.65 / 0.90 / 0.7 cutoff is policy. Fail-open RUN is +not fail-closed never auto-allows. Treating 5.8x as Harbor, new-source +0.790 as Harbor, 76/81 as Harbor, 118 entries as endorsement, a +classifier as an authorizer, estimates as Harbor, quanto as a +calibrated replica, DeBERTa as a Jev replica, or Qwen3 as Archer is +soundness theater. Constrained AR ≠ calibrated Noul. JEV_THRESHOLD 0.65 +still soft. threshold 0.90 still soft. ask_below 0.7 still soft. +classifier ≠ authorizer. estimates not Harbor. routing ≠ permission. +wire-compat ≠ logit-equiv. SHA move is not a replica. catalog ≠ +endorsement. + +### Adversarial review + testing hooks (Basit standing +order) + +Parent merge only after **CLEAN** adversarial review **AND** testing. +Hooks for the reviewer: + +- Uniqueness-gate: the consecutive `Hourly 1542 uniqueness lock:` + string must appear in every overlay listed below. Prior walls 0843 / + 0915 / jcr / 0922 / 0940 / 0947 / 1049 / 1143 / 1248 / 1340 / 1441 stay one + substring each (do not mutate them; do not reopen #23–#48). +- Namesake locks: TypeLLM/TypeLLM densify §113; jaredpalmer/kev densify + §45; TheoOliveira/pi-jev densify §42; harshwasan/jev-sentinel ≠ + leepokai/jev-guard; jackbarunz/jev-tool-router ≠ + esinocchi/jev-tool-router; openlayer-ai/jevals ≠ dayhaysoos/jevals; + MrJev/awesome-jev ≠ yibie/awesome-jev; Qwen3 ≠ Archer; + Constrained AR ≠ calibrated Noul. +- Densify vs new: TypeLLM densifies §113. kev densifies §45. pi-jev + densifies §42. Do not mint sibling first-sighting sections. + sentinel / routers / leanest / jevals / MrJev / firewall / + jev-codex-approval / HF encoder-quanto are first sightings this hour. +- Harbor-jevals: Batch 5.8x / 4B new-source 0.790/0.806 / 8B 0.796/0.780 / + Jev hosted 0.857 / 8.2% ≥0.9 on wrong / 76/81 vs 77/81 / 0.419s vs 2.459s / + $0.00486 vs $0.03673 are *theirs*, not Harbor. 37.30s → 2.40s / CLERC + 5% to 18% / 94.71% stay §125. 93.5% / 74.9 / 8.7x stay §123. 74.6 + JevBench stays §78. +- Anti-patterns to refuse: TypeSafe drop-in; Qwen3 as Archer; + Constrained AR as Noul; 0.65 / 0.90 / 0.7 as a hard gate; fail-open as + fail-closed; classifier as authorizer; estimates as Harbor; catalog as + endorsement; copying keys / `npm` / `pip` / `uv` / `docker`. +- Overlay set: SKILL.md body (protocol fragments + class-table densify + + Hourly 1542), mental-models Apply 1542, composition-algebra items + 433–448, faq, mixed-architecture, validation, toolbox-mapping, + methods-catalog, formal-methods, formal-semi-formal, + applied-mappings, judgment-class, question-design, + agent-self-assessment, mappings, CHANGELOG, README, docs/ecosystem, + findings batch #108, refresh-log, sources.json, changelog-hourly.md, + revisit_fingerprints.json (TypeLLM + kev SHA move; novel HIGH seeds). +- Offline check: `evaluate_decisions.py --self-test` (now includes + Constrained AR ≠ Noul / fail-closed never auto-allows / fail-open + uncertainty means RUN / routing threshold still soft / classifier ≠ + authorizer / estimates not Harbor) and `uniqueness_gate.py` (0843 + + 0915 + jcr + 0922 + 0940 + 0947 + 1049 + 1143 + 1248 + 1340 + 1441 + 1542). + No live Jev key. No wrappers. + +Hourly 1542 uniqueness lock: TypeLLM/TypeLLM densify HEAD 6a48f9f1e623 README SHA dbdc1f193537; README densify 3k→12k B; thinking=True/False per-field budget; type safety does not guarantee factual accuracy; Batch 5.8x *theirs*; Constrained AR ≠ calibrated Noul; Qwen/Qwen3.8-27B ≠ Archer; jaredpalmer/kev densify HEAD b339f446a0ef README SHA 86b0a19909f3; Kev-0.6B 4B 8B family; 4B new-source 0.790/0.806 *theirs*; 8B new-source 0.796/0.780 *theirs*; Jev hosted 0.857 *theirs*; Questions share the input text but cannot read each other; No Jev outputs were used for training; 8.2% ≥0.9 on wrong *theirs*; option order can change an answer; Qwen3 ≠ Archer; wire-compat ≠ logit-equiv; SHA move is not a replica; TheoOliveira/pi-jev 21★ fail-closed routing; JEV_THRESHOLD 0.65 still soft; routing ≠ permission; harshwasan/jev-sentinel fail closed never auto-allows; harshwasan/jev-sentinel ≠ leepokai/jev-guard; jackbarunz/jev-tool-router ≠ esinocchi/jev-tool-router; threshold 0.90 still soft; 76/81 vs 77/81 *theirs*; 0.419s vs 2.459s *theirs*; $0.00486 vs $0.03673 *theirs*; does not execute; not a security boundary; baronunread/leanest fail-open uncertainty means RUN; classifier.dev default Jev/Laya pluggable; openlayer-ai/jevals ≠ dayhaysoos/jevals; estimates not Harbor; classifier ≠ authorizer; MrJev/awesome-jev 118 entries catalog ≠ endorsement; MrJev/awesome-jev ≠ yibie/awesome-jev; Koushik890/jev-firewall fail closed ask_below 0.7 still soft; CompleteTech-LLC-AI-Research/jev-codex-approval experimental native not compiled; confidence is not a measured probability; rh-guard owns primary gates; hf:rAVEUK/open-jev-deberta-v3-large encoder class member not Jev replica; hf:p-yan/laya-quanto serving substrate ≠ calibrated replica; hf:Gtrkrsk/laya serving substrate ≠ calibrated replica; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48; notes.md §126 + +**Hourly 1542 HIGH (`notes.md` §126).** TypeLLM README densify 3k→12k B. Batch 5.8x *theirs*. Constrained AR ≠ calibrated Noul. kev family densify. 4B new-source 0.790/0.806 *theirs*. 8.2% ≥0.9 on wrong *theirs*. option order can change an answer. fail-closed routing vs fail-open test selection. classifier ≠ authorizer. estimates not Harbor. catalog ≠ endorsement. SHA move is not a replica. Do not reopen or amend PR #23–#48. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. +Hourly 1542 uniqueness lock: TypeLLM/TypeLLM densify HEAD 6a48f9f1e623 README SHA dbdc1f193537; README densify 3k→12k B; thinking=True/False per-field budget; type safety does not guarantee factual accuracy; Batch 5.8x *theirs*; Constrained AR ≠ calibrated Noul; Qwen/Qwen3.8-27B ≠ Archer; jaredpalmer/kev densify HEAD b339f446a0ef README SHA 86b0a19909f3; Kev-0.6B 4B 8B family; 4B new-source 0.790/0.806 *theirs*; 8B new-source 0.796/0.780 *theirs*; Jev hosted 0.857 *theirs*; Questions share the input text but cannot read each other; No Jev outputs were used for training; 8.2% ≥0.9 on wrong *theirs*; option order can change an answer; Qwen3 ≠ Archer; wire-compat ≠ logit-equiv; SHA move is not a replica; TheoOliveira/pi-jev 21★ fail-closed routing; JEV_THRESHOLD 0.65 still soft; routing ≠ permission; harshwasan/jev-sentinel fail closed never auto-allows; harshwasan/jev-sentinel ≠ leepokai/jev-guard; jackbarunz/jev-tool-router ≠ esinocchi/jev-tool-router; threshold 0.90 still soft; 76/81 vs 77/81 *theirs*; 0.419s vs 2.459s *theirs*; $0.00486 vs $0.03673 *theirs*; does not execute; not a security boundary; baronunread/leanest fail-open uncertainty means RUN; classifier.dev default Jev/Laya pluggable; openlayer-ai/jevals ≠ dayhaysoos/jevals; estimates not Harbor; classifier ≠ authorizer; MrJev/awesome-jev 118 entries catalog ≠ endorsement; MrJev/awesome-jev ≠ yibie/awesome-jev; Koushik890/jev-firewall fail closed ask_below 0.7 still soft; CompleteTech-LLC-AI-Research/jev-codex-approval experimental native not compiled; confidence is not a measured probability; rh-guard owns primary gates; hf:rAVEUK/open-jev-deberta-v3-large encoder class member not Jev replica; hf:p-yan/laya-quanto serving substrate ≠ calibrated replica; hf:Gtrkrsk/laya serving substrate ≠ calibrated replica; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48; notes.md §126 diff --git a/research/refresh-log.md b/research/refresh-log.md index 342bdef..c2bd009 100644 --- a/research/refresh-log.md +++ b/research/refresh-log.md @@ -3291,3 +3291,7 @@ Hourly 0947 uniqueness lock: Fast and cheap agent evals. jev as judge.; 18,041 s **Hourly 1248 HIGH (`notes.md` §123).** decide is not generate. tryDecide returns typed calibrated judgments not a token stream. GLiNER/GLiClass ports are class members not Jev replicas. 93.5% *theirs* not Harbor. 74.9 *theirs* not Harbor. 8.7x *theirs* not Harbor. wire-compat ≠ logit-equiv. SHA move is not a replica. catalog ≠ endorsement. Do not reopen or amend PR #23–#45. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. Hourly 1248 uniqueness lock: decide is not generate; tryDecide returns typed calibrated judgments not a token stream; juspay/neurolink 133★ MIT HEAD 268b0fe83130 README SHA e709cadfa6b6 tag v12.19.0; GLiNER/GLiClass ports are class members not Jev replicas; MacPaw/Gliner2Swift ≠ Knowledgator/GLiClass.c ≠ fbilhaut/gliclass-rs ≠ Knowledgator/GLiClass.js ≠ gravitee-io/GLiNER4j ≠ apiplant/gliner-rs ≠ codesoda/gliner2-rs; 8.7x faster 4.4x fewer prompts *theirs*; 153 was a reporting error; corrected 156-case 9.8x faster 4.2x fewer prompts *theirs*; independent v0.2.1 1.24x vs Mini *theirs*; Approvals only; anpicasso/hermes-jev-approvals ≠ hermes-switchyard; scx-router GLiClass ranks candidate LLMs in one non-generative pass; typesafeai-dotnet-sdk Not affiliated with TypeSafe AI; hyperspaceai/jevcache ≠ kushals256/jevcache; ST-jeved measures each reply; 400 plain-text for unaskable question; razorback16/openjev:0.2.1 Docker densify HEAD 794a81b87131; wire-compat ≠ logit-equiv; Option-Marker joint attention 93.5% macro *theirs*; 93.6% micro *theirs*; n=78; T = 1.0367 vs T = 1.1692 two temperatures; guaranteeing is soundness theater; wfzyx/von densify HEAD bed7e7337791; Benchmark Heaven leaderboard #2 74.9 *theirs*; NLL calibration assets; 77.10% still §71 claim-audit; do not re-fold as a beat; Heman10x-NGU/openJev-verdict-2.0 densify HEAD bff28567cff4; kev-family weight tarballs; PLAN_Qwen35 proposal for review; deadline 0.53→0.82 at 9B *theirs*; Qwen3.5-9B ≠ Archer; isolation would fail by construction on DeltaNet; jaredpalmer/kev densify; JevBench v1.2.2 jeff 66.9 (#9) jev 75.3 (#2) *theirs*; logan-markewich/jeff densify HEAD 34b32f99a727; thinking=True/False per-field budget; type safety does not guarantee factual accuracy; TypeLLM/TypeLLM densify HEAD c4b03ba9e792; us/jev-local stub until hf; Eran-BA/Jev_from_GLiNER2 spec ≠ replica; lsu-ub-uu/systemone ≠ TypeSafe System One; Layan/Laya HF spaces name-match; catalog ≠ endorsement; decide ≠ generate ≠ stream; 93.5% *theirs* not Harbor; 74.9 *theirs* not Harbor; 8.7x *theirs* not Harbor; SHA move is not a replica; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45; notes.md §123 + + +**Hourly 1542 HIGH (`notes.md` §126).** TypeLLM README densify 3k→12k B. Batch 5.8x *theirs*. Constrained AR ≠ calibrated Noul. kev family densify. 4B new-source 0.790/0.806 *theirs*. 8.2% ≥0.9 on wrong *theirs*. option order can change an answer. fail-closed routing vs fail-open test selection. classifier ≠ authorizer. estimates not Harbor. catalog ≠ endorsement. SHA move is not a replica. Do not reopen or amend PR #23–#48. Does not bump 0.5.0. Skip Archer. `invented_signal: false`. +Hourly 1542 uniqueness lock: TypeLLM/TypeLLM densify HEAD 6a48f9f1e623 README SHA dbdc1f193537; README densify 3k→12k B; thinking=True/False per-field budget; type safety does not guarantee factual accuracy; Batch 5.8x *theirs*; Constrained AR ≠ calibrated Noul; Qwen/Qwen3.8-27B ≠ Archer; jaredpalmer/kev densify HEAD b339f446a0ef README SHA 86b0a19909f3; Kev-0.6B 4B 8B family; 4B new-source 0.790/0.806 *theirs*; 8B new-source 0.796/0.780 *theirs*; Jev hosted 0.857 *theirs*; Questions share the input text but cannot read each other; No Jev outputs were used for training; 8.2% ≥0.9 on wrong *theirs*; option order can change an answer; Qwen3 ≠ Archer; wire-compat ≠ logit-equiv; SHA move is not a replica; TheoOliveira/pi-jev 21★ fail-closed routing; JEV_THRESHOLD 0.65 still soft; routing ≠ permission; harshwasan/jev-sentinel fail closed never auto-allows; harshwasan/jev-sentinel ≠ leepokai/jev-guard; jackbarunz/jev-tool-router ≠ esinocchi/jev-tool-router; threshold 0.90 still soft; 76/81 vs 77/81 *theirs*; 0.419s vs 2.459s *theirs*; $0.00486 vs $0.03673 *theirs*; does not execute; not a security boundary; baronunread/leanest fail-open uncertainty means RUN; classifier.dev default Jev/Laya pluggable; openlayer-ai/jevals ≠ dayhaysoos/jevals; estimates not Harbor; classifier ≠ authorizer; MrJev/awesome-jev 118 entries catalog ≠ endorsement; MrJev/awesome-jev ≠ yibie/awesome-jev; Koushik890/jev-firewall fail closed ask_below 0.7 still soft; CompleteTech-LLC-AI-Research/jev-codex-approval experimental native not compiled; confidence is not a measured probability; rh-guard owns primary gates; hf:rAVEUK/open-jev-deberta-v3-large encoder class member not Jev replica; hf:p-yan/laya-quanto serving substrate ≠ calibrated replica; hf:Gtrkrsk/laya serving substrate ≠ calibrated replica; Archer still promised_not_landed; Hub archerhume/4rcherhume HTTP 401; do not reopen or amend PR #23/#24/#25/#26/#27/#28/#29/#30/#31/#32/#33/#34/#35/#36/#37/#38/#39/#40/#41/#42/#43/#44/#45/#46/#47/#48; notes.md §126 diff --git a/research/revisit_fingerprints.json b/research/revisit_fingerprints.json index b36ffbd..964dbb3 100644 --- a/research/revisit_fingerprints.json +++ b/research/revisit_fingerprints.json @@ -20,7 +20,7 @@ "forks_count", "likes" ], - "note": "Last-look snapshots for already-catalogued sources. Hourly diffs these four fingerprints. Star-noise is not a fold. SHA move is not a replica. Seeded from merged notes (SemIf \u00a7117, NanoJev \u00a7115) plus hourly 1248 densify (openjev \u00a775, von \u00a749, verdict \u00a771, kev \u00a745, jeff \u00a760, TypeLLM \u00a7113) plus hourly 1340 densify (openjev sdk 0.7 + MLX 400, kev PLAN_Qwen35) plus hourly 1441 densify (openjev STE backends+Codiv dual serving) and 1441 first sightings (jev-visual, jev-mcp, litjev, Open-Jev, jeq, laya-coreai). default_sha is the full HEAD. pushed_at is the GitHub push clock. description_hash is sha256[:12] of the GitHub/Space description, or null when notes do not quote it. README SHA is an optional extra, not a substitute for description_hash.", + "note": "Last-look snapshots for already-catalogued sources. Hourly diffs these four fingerprints. Star-noise is not a fold. SHA move is not a replica. Seeded from merged notes (SemIf §117, NanoJev §115) plus hourly 1248 densify (openjev §75, von §49, verdict §71, kev §45, jeff §60, TypeLLM §113) plus hourly 1340 densify (openjev sdk 0.7 + MLX 400, kev PLAN_Qwen35) plus hourly 1441 densify (openjev STE backends+Codiv dual serving) and 1441 first sightings plus hourly 1542 densify (TypeLLM README 3k→12k B, kev family new-source) and 1542 first sightings (pi-jev densify, jev-sentinel, tool-routers, leanest, jevals, MrJev, jev-firewall, jev-codex-approval, HF encoder/quanto). default_sha is the full HEAD. pushed_at is the GitHub push clock. description_hash is sha256[:12] of the GitHub/Space description, or null when notes do not quote it. README SHA is an optional extra, not a substitute for description_hash.", "looks": [ { "id": "github:TheoLeeCJ/SemIf", @@ -85,14 +85,14 @@ { "id": "github:jaredpalmer/kev", "notes_section": "45", - "last_look": "2026-09-20T19:40Z", + "last_look": "2026-09-20T21:42Z", "fingerprints": { - "default_sha": "75cc15ddb8e2d84bb85e68200c32d31da7f09f54", - "pushed_at": "2026-09-20T19:08:13Z", + "default_sha": "b339f446a0ef0d691d9541daa42fb5416dfba63b", + "pushed_at": "2026-09-20T20:54:46Z", "description_hash": "8f68dcc43f12", "release_tag": "kev-family" }, - "readme_sha": "bd8f04d0747e56e59e60cc8ba9a122f4e24a0707" + "readme_sha": "86b0a19909f30608fd0b404b30651f0812eeacdf" }, { "id": "github:logan-markewich/jeff", @@ -109,14 +109,14 @@ { "id": "github:TypeLLM/TypeLLM", "notes_section": "113", - "last_look": "2026-09-20T18:48Z", + "last_look": "2026-09-20T21:42Z", "fingerprints": { - "default_sha": "c4b03ba9e792ef80095bb66f239b08835fbe155d", - "pushed_at": "2026-09-20T13:11:30Z", + "default_sha": "6a48f9f1e623500b936d978140d239d9c11ce82a", + "pushed_at": "2026-09-20T21:02:54Z", "description_hash": "4e7869c96688", "release_tag": null }, - "readme_sha": "116ca5f89352a8a2ffbe411bae5eeb5c8b342ee2" + "readme_sha": "dbdc1f193537d57b6721bd42b623c4cf83d5977b" }, { "id": "github:sseanliu/Jev-Vision", @@ -201,6 +201,150 @@ "release_tag": null }, "readme_sha": null + }, + { + "id": "github:TheoOliveira/pi-jev", + "notes_section": "42", + "last_look": "2026-09-20T21:42Z", + "fingerprints": { + "default_sha": "549c2bfa249269d0f5590658b5743801c47af369", + "pushed_at": "2026-09-20T21:09:11Z", + "description_hash": "2155badd27d8", + "release_tag": null + }, + "readme_sha": "a465d819f32b00c22060381000a52ce4ba726dfb" + }, + { + "id": "github:harshwasan/jev-sentinel", + "notes_section": "126", + "last_look": "2026-09-20T21:42Z", + "fingerprints": { + "default_sha": "4ae67df78c95a6c9c5b9d872c76ba5123c2028b0", + "pushed_at": "2026-09-20T21:42:49Z", + "description_hash": "3b4ee158cdc9", + "release_tag": null + }, + "readme_sha": "77d8a24dfdac128212d4d059bcb40c9648025b8e" + }, + { + "id": "github:jackbarunz/jev-tool-router", + "notes_section": "126", + "last_look": "2026-09-20T21:42Z", + "fingerprints": { + "default_sha": "e16dbd2ae684445738cc4e1aaf8a61f10a470422", + "pushed_at": "2026-09-20T20:44:17Z", + "description_hash": "78fe8f9c1c0d", + "release_tag": null + }, + "readme_sha": "0fb6eb070c88fbc16de4924a935be2d382065995" + }, + { + "id": "github:esinocchi/jev-tool-router", + "notes_section": "126", + "last_look": "2026-09-20T21:42Z", + "fingerprints": { + "default_sha": "67e8a48305ae05ced9fcdf6a666531793309c51e", + "pushed_at": "2026-09-20T21:21:18Z", + "description_hash": null, + "release_tag": null + }, + "readme_sha": "fa6b023405c61ad46ca7381371c866e93d3966fa" + }, + { + "id": "github:baronunread/leanest", + "notes_section": "126", + "last_look": "2026-09-20T21:42Z", + "fingerprints": { + "default_sha": "b5529b59e76dc083df012a3abdbab7df8fc0bd5b", + "pushed_at": "2026-09-20T21:33:34Z", + "description_hash": "8cb06d40a24c", + "release_tag": null + }, + "readme_sha": "aeb946d2c70b00cd626f4df549a98a4a76a4a4d3" + }, + { + "id": "github:openlayer-ai/jevals", + "notes_section": "126", + "last_look": "2026-09-20T21:42Z", + "fingerprints": { + "default_sha": "a38a971095c48748cb49b31530ca6d20ecc830ae", + "pushed_at": "2026-09-20T21:55:09Z", + "description_hash": "6d6d99246390", + "release_tag": null + }, + "readme_sha": "af6f2244641651ce47bbc18e760527cd2e72bd75" + }, + { + "id": "github:MrJev/awesome-jev", + "notes_section": "126", + "last_look": "2026-09-20T21:42Z", + "fingerprints": { + "default_sha": "e9cac99e4919d6cec6ce1596e9d974515fe0a00a", + "pushed_at": "2026-09-20T18:55:53Z", + "description_hash": "4b8fc61080e1", + "release_tag": null + }, + "readme_sha": "159151e96121bb1a425f41fc12b69a839314d3d2" + }, + { + "id": "github:Koushik890/jev-firewall", + "notes_section": "126", + "last_look": "2026-09-20T21:42Z", + "fingerprints": { + "default_sha": "7cab712c5ee2bba1d5d5b63644f9acdccb2c58c1", + "pushed_at": "2026-09-20T21:41:26Z", + "description_hash": "f98968cbc88e", + "release_tag": null + }, + "readme_sha": "15fdfd9b9a507cbb06a8b6a1317a71e28eeb2b6e" + }, + { + "id": "github:CompleteTech-LLC-AI-Research/jev-codex-approval", + "notes_section": "126", + "last_look": "2026-09-20T21:42Z", + "fingerprints": { + "default_sha": "0b931ee24a907c9dc47dc1828a94a6afcd7775b6", + "pushed_at": "2026-09-20T20:55:26Z", + "description_hash": "35ec86ed6238", + "release_tag": null + }, + "readme_sha": "7575914c62c7f5bf70fdf2ce48528846ed32f8fc" + }, + { + "id": "hf:rAVEUK/open-jev-deberta-v3-large", + "notes_section": "126", + "last_look": "2026-09-20T21:42Z", + "fingerprints": { + "default_sha": "7beac18983e6df5a12e1038c6ce44762cc937dcc", + "pushed_at": "2026-09-20T21:04:09.000Z", + "description_hash": null, + "release_tag": null + }, + "readme_sha": null + }, + { + "id": "hf:p-yan/laya-quanto", + "notes_section": "126", + "last_look": "2026-09-20T21:42Z", + "fingerprints": { + "default_sha": "070a6e6097632683a455c1b9180953f0ef8d6d1d", + "pushed_at": "2026-09-20T21:57:38.000Z", + "description_hash": null, + "release_tag": null + }, + "readme_sha": null + }, + { + "id": "hf:Gtrkrsk/laya", + "notes_section": "126", + "last_look": "2026-09-20T21:42Z", + "fingerprints": { + "default_sha": "f6029af5253c735c891f0678ba49879469ba4cf8", + "pushed_at": "2026-09-20T21:16:06.000Z", + "description_hash": null, + "release_tag": null + }, + "readme_sha": null } ] } diff --git a/research/revisit_fingerprints.py b/research/revisit_fingerprints.py index f639166..960e719 100644 --- a/research/revisit_fingerprints.py +++ b/research/revisit_fingerprints.py @@ -285,6 +285,7 @@ def self_test() -> None: "github:jaredpalmer/kev": "45", "github:logan-markewich/jeff": "60", "github:TypeLLM/TypeLLM": "113", + "github:TheoOliveira/pi-jev": "42", } for look_id, section in densify_original_ids.items(): assert look_id in by_id, look_id diff --git a/research/sources.json b/research/sources.json index d10ca94..6d076b3 100644 --- a/research/sources.json +++ b/research/sources.json @@ -1,6 +1,6 @@ { "refresh_cadence": "hourly", - "retrieved": "2026-09-20T20:41Z", + "retrieved": "2026-09-20T21:42Z", "sources": [ { "kind": "docs",