From 3ee75be44fbba05fb9008a0dc44e24d7a8eb919b Mon Sep 17 00:00:00 2001 From: mrchatam <287639636+mrchatam@users.noreply.github.com> Date: Sun, 27 Sep 2026 00:50:14 +0330 Subject: [PATCH 1/7] feat: token savings for supervisors (v0.3.0) - wait_task long-poll (single/many ids, any/all, capped at 55 s) and delegate_tasks batches - compact MCP JSON, task_result view brief|full, handoff context deduplicated in the full view - presets and size routing; auto_fix_rounds and escalate_to chains with hard caps and a trail - optional advisory auto_review on a cheap profile (transient status reviewing) - usage_report / workhorse stats with per-run tokens and a labelled supervisor ESTIMATE - opt-in worker token_savers: terse and minimal_code fragments, RTK rewrite in the guard plugin - approvals.require_operator with a separate operator token confirmed via the CLI - per-profile stall_minutes, audit.jsonl rotation - TEST-ONLY stub backend, registered only with WH_ENABLE_STUB_BACKEND=1 --- adapters/index.mjs | 4 + .../kilo/config/plugin/workhorse-guard.js | 26 + adapters/stub/adapter.mjs | 47 ++ adapters/stub/stub-cli.mjs | 121 ++++ bin/workhorse | 74 ++- bin/workhorse-mcp | 72 ++- config/examples/profiles.tiered.json | 72 +++ lib/admin.mjs | 71 +++ lib/audit.mjs | 43 +- lib/config.mjs | 82 +++ lib/daemon.mjs | 17 +- lib/handoff.mjs | 19 +- lib/operator.mjs | 47 ++ lib/savers.mjs | 72 +++ lib/tasks.mjs | 550 ++++++++++++++++-- lib/usage.mjs | 140 +++++ lib/views.mjs | 67 +++ 17 files changed, 1464 insertions(+), 60 deletions(-) create mode 100644 adapters/stub/adapter.mjs create mode 100755 adapters/stub/stub-cli.mjs create mode 100644 config/examples/profiles.tiered.json create mode 100644 lib/operator.mjs create mode 100644 lib/savers.mjs create mode 100644 lib/usage.mjs create mode 100644 lib/views.mjs diff --git a/adapters/index.mjs b/adapters/index.mjs index 79ba414..f675004 100644 --- a/adapters/index.mjs +++ b/adapters/index.mjs @@ -32,8 +32,12 @@ import claudeCode from "./claude-code/adapter.mjs" import codex from "./codex/adapter.mjs" import gemini from "./gemini/adapter.mjs" import aider from "./aider/adapter.mjs" +import stub, { stubEnabled } from "./stub/adapter.mjs" export const BACKENDS = { kilo, opencode, "claude-code": claudeCode, codex, gemini, aider } +// Test-only scripted backend: registered only when the daemon process has WH_ENABLE_STUB_BACKEND=1 +// (see adapters/stub/adapter.mjs for why production can never select it). +if (stubEnabled()) BACKENDS.stub = stub export function backendNames() { return Object.keys(BACKENDS) diff --git a/adapters/kilo/config/plugin/workhorse-guard.js b/adapters/kilo/config/plugin/workhorse-guard.js index 9ab6f3a..b604b6d 100644 --- a/adapters/kilo/config/plugin/workhorse-guard.js +++ b/adapters/kilo/config/plugin/workhorse-guard.js @@ -18,8 +18,14 @@ // tool-output dir). // Used by the kilo and opencode backends as a plugin, and by the claude-code backend through // adapters/claude-code/pretooluse-guard.mjs (which maps Claude tool names onto these). +// 3. Optional token saver (daemon.json token_savers.rtk): when WH_RTK_BIN is set, a bash command that +// passed the checks above is replaced by RTK's compact equivalent (`rtk rewrite`, e.g. `git status` +// -> `rtk git status`; https://github.com/rtk-ai/rtk, Apache-2.0, run as an external binary). The +// rewritten command is checked again. Multi-line commands are left alone. Only the worker's own +// shell output is compacted; the daemon's test run never goes through this. // NOTE: Kilo/OpenCode call every export of this module as a plugin, so helpers must stay unexported. import path from "node:path" +import { execFileSync } from "node:child_process" const STATIC_SECRET_VARS = [ "NVIDIA_API_KEY", "OPENAI_API_KEY", "ANTHROPIC_API_KEY", "GITHUB_TOKEN", "GH_TOKEN", @@ -101,6 +107,21 @@ function checkBash(cmd, deny) { for (const p of denyPaths()) if (cmd.includes(p)) deny("access to the secret store or daemon token is not allowed") } +// `rtk rewrite ` prints the rewritten command and exits 0 (allowed) or 3 (rewritten; the host +// decides permissions, which the checks here do); 1 = no RTK equivalent, 2 = deny rule. +function rtkRewrite(bin, cmd) { + if (!bin || !path.isAbsolute(bin) || !cmd || cmd.length > 2000 || /[\n\r]/.test(cmd)) return null + let out + try { + out = execFileSync(bin, ["rewrite", cmd], { encoding: "utf8", timeout: 3000, stdio: ["ignore", "pipe", "ignore"] }) + } catch (e) { + if (e && e.status === 3 && typeof e.stdout === "string") out = e.stdout + else return null + } + const r = String(out || "").trim() + return r && r !== cmd && !/[\n\r]/.test(r) ? r : null +} + const FILE_TOOLS = new Set(["read", "write", "edit", "multiedit", "apply_patch", "patch", "glob", "grep", "list", "lsp"]) const PATH_KEYS = ["filePath", "path", "file_path"] @@ -123,6 +144,11 @@ export const WorkhorseGuard = async ({ directory, worktree }) => { if (input.tool === "bash") { checkBash(String(args.command ?? ""), deny) if (args.workdir && !inside(String(args.workdir))) deny("workdir outside the task worktree") + const rw = rtkRewrite(process.env.WH_RTK_BIN, typeof args.command === "string" ? args.command : "") + if (rw) { + checkBash(rw, deny) + args.command = rw + } return } if (FILE_TOOLS.has(input.tool)) { diff --git a/adapters/stub/adapter.mjs b/adapters/stub/adapter.mjs new file mode 100644 index 0000000..5d67996 --- /dev/null +++ b/adapters/stub/adapter.mjs @@ -0,0 +1,47 @@ +// TEST-ONLY scripted fake worker ("stub" backend). It lets the integration tests exercise the real +// daemon paths (scheduler -> spawn -> event parsing -> daemon-run tests -> finalize -> handoff -> +// approve/continue/retry/fallback/restart/auto-fix/escalation/review) without a coding-agent CLI or a +// model, e.g. in GitHub Actions. +// +// It can never be selected in production: +// - adapters/index.mjs registers it only when the DAEMON process has WH_ENABLE_STUB_BACKEND=1. The +// supervisor/MCP auto-start passes the daemon a minimal env (lib/client.mjs startSupervisor), so a +// daemon started by the MCP shim or `workhorse start` never has it unless an operator exports it. +// - validate() refuses again at task start when the flag is missing, and refuses any bin other than +// the bundled adapters/stub/stub-cli.mjs. +// - It is not in the config defaults, the installer or `workhorse validate`'s known backends. +// Scenarios: JSON files in daemon.json backends.stub.scenarios_dir (see stub-cli.mjs for the format). +import path from "node:path" +import { fileURLToPath } from "node:url" + +export const STUB_CLI = path.join(path.dirname(fileURLToPath(import.meta.url)), "stub-cli.mjs") +export const stubEnabled = () => process.env.WH_ENABLE_STUB_BACKEND === "1" + +export default { + name: "stub", + status: "test-only", + summary: "Scripted fake worker for integration tests (daemon env WH_ENABLE_STUB_BACKEND=1 only)", + notes: ["test-only; never available in production"], + capabilities: { resume: true, review_agent: true, inner_sandbox: false, guard: "none", providers: "none" }, + + validate({ bc }) { + const out = [] + if (!stubEnabled()) out.push("backend 'stub' is test-only (the daemon was not started with WH_ENABLE_STUB_BACKEND=1)") + if (bc?.bin_real && path.resolve(bc.bin_real) !== path.resolve(STUB_CLI)) out.push("backend 'stub' must use the bundled adapters/stub/stub-cli.mjs") + return out + }, + prepare() {}, + command(ctx) { + return ["--model", ctx.profile.model, ...(ctx.resumeSession ? ["--session", ctx.resumeSession] : []), "--", ctx.message] + }, + env(ctx) { + return { WH_STUB_STATE: ctx.dataDir, WH_STUB_SCENARIOS: ctx.bc.scenarios_dir || "" } + }, + sandbox(ctx) { + return { roBinds: [path.dirname(STUB_CLI), ctx.bc.scenarios_dir].filter(Boolean), rwBinds: [ctx.dataDir] } + }, + // stub-cli already prints normalized events. + parse(ev) { + return ev && typeof ev.type === "string" ? [ev] : [] + }, +} diff --git a/adapters/stub/stub-cli.mjs b/adapters/stub/stub-cli.mjs new file mode 100755 index 0000000..538cb55 --- /dev/null +++ b/adapters/stub/stub-cli.mjs @@ -0,0 +1,121 @@ +#!/usr/bin/env node +// TEST-ONLY fake coding-agent CLI for the "stub" backend (see adapter.mjs). Prints normalized workhorse +// events (JSON lines) on stdout, following a scripted scenario. +// +// stub-cli.mjs --model [--session ] -- +// env: WH_STUB_STATE (per-task state dir), WH_STUB_SCENARIOS (scenario dir) +// +// The scenario is named by `MOCK_SCENARIO=` in a message (remembered for later runs of the task): +// { "runs": [ [action, ...], [action, ...], ... ] } +// Run i of the task (counted across sessions, failed attempts excluded) plays runs[i] (the last entry +// repeats). Actions: +// {"write": {"path": "rel/path", "content": "..."}} write a file in the worktree (cwd) +// {"bash": "command"} run it with /bin/sh in the worktree +// {"text": "..."} assistant message (end a run with the RESULT block) +// {"tokens": {"input": N, "output": N, ...}} token usage (default per run: 1000 in / 200 out) +// {"error": {"status": 429, "message": "...", "times": 1, "models": ["p/m"]}} +// fail this attempt (exit 1) the first `times` times +// {"sleep": seconds} stay silent (stall/timeout/restart tests) +// {"if_model": "p/m", "then": [...], "else": [...]} branch on --model +// {"exit": code} +// "count_at_start": true (top level) counts a run when it starts, so a run killed midway (restart tests) +// is not replayed by the next one. +// Every received message is appended to /messages.jsonl (tests inspect it). +import fs from "node:fs" +import path from "node:path" +import crypto from "node:crypto" +import { spawnSync } from "node:child_process" + +const argv = process.argv.slice(2) +let model = "stub/model" +let session = null +let message = "" +for (let i = 0; i < argv.length; i++) { + if (argv[i] === "--model") model = argv[++i] + else if (argv[i] === "--session") session = argv[++i] + else if (argv[i] === "--") { + message = argv.slice(i + 1).join(" ") + break + } +} +const stateDir = process.env.WH_STUB_STATE || process.cwd() +const scenDir = process.env.WH_STUB_SCENARIOS || "" +fs.mkdirSync(stateDir, { recursive: true }) +const stateFile = path.join(stateDir, "stub-state.json") +let st = { scenario: null, runs_done: 0, errors: {}, sessions: [] } +try { + st = { ...st, ...JSON.parse(fs.readFileSync(stateFile, "utf8")) } +} catch {} +const save = () => fs.writeFileSync(stateFile, JSON.stringify(st)) +const emit = (ev) => process.stdout.write(JSON.stringify(ev) + "\n") + +const m = /MOCK_SCENARIO=([A-Za-z0-9_-]+)/.exec(message) +if (m) st.scenario = m[1] +if (!session || !st.sessions.includes(session)) { + session = `stub-${crypto.randomBytes(4).toString("hex")}` + st.sessions.push(session) +} +fs.appendFileSync(path.join(stateDir, "messages.jsonl"), JSON.stringify({ run: st.runs_done, model, session, message }) + "\n") +save() +emit({ type: "session", id: session }) + +let scenario = { runs: [[{ text: "Nothing to do.\n\n## RESULT\nstatus: done\nsummary: stub default run, no changes\nfiles_changed: none\ntests: none\nconcerns: none" }]] } +if (st.scenario && scenDir) { + try { + scenario = JSON.parse(fs.readFileSync(path.join(scenDir, `${st.scenario}.json`), "utf8")) + } catch (e) { + emit({ type: "error", message: `stub: cannot read scenario ${st.scenario}: ${e.message}`, statusCode: null, retryable: false }) + process.exit(1) + } +} +const runs = scenario.runs || [] +const actions = runs.length ? runs[Math.min(st.runs_done, runs.length - 1)] : [] +let tokensSent = false +if (scenario.count_at_start) { + st.runs_done++ + save() +} + +async function play(list, path0) { + for (const [i, a] of list.entries()) { + const key = `${st.runs_done}:${path0}${i}` + if (a.error) { + const e = a.error + if (e.models && !e.models.includes(model)) continue + const n = st.errors[key] || 0 + if (n >= (e.times ?? 1)) continue + st.errors[key] = n + 1 + save() + const status = e.status ?? 500 + emit({ type: "error", message: e.message || `stub provider error ${status}`, statusCode: status, retryable: [408, 429, 500, 502, 503, 504].includes(status) }) + process.exit(1) + } else if (a.if_model !== undefined) { + await play(a.if_model === model ? a.then || [] : a.else || [], `${path0}${i}.`) + } else if (a.write) { + const p = path.resolve(process.cwd(), a.write.path) + fs.mkdirSync(path.dirname(p), { recursive: true }) + fs.writeFileSync(p, a.write.content ?? "") + emit({ type: "tool", tool: "write", shell: false, input: a.write.path, ok: true, output: "", exit: null }) + } else if (a.bash !== undefined) { + const r = spawnSync("/bin/sh", ["-c", a.bash], { cwd: process.cwd(), encoding: "utf8" }) + emit({ type: "tool", tool: "bash", shell: true, input: a.bash, ok: true, output: String(r.stdout || "") + String(r.stderr || ""), exit: r.status }) + } else if (a.text !== undefined) { + emit({ type: "text", text: a.text }) + } else if (a.tokens) { + tokensSent = true + emit({ type: "step", turns: 1, tokens: a.tokens, cost: a.cost || 0 }) + } else if (a.sleep !== undefined) { + await new Promise((r) => setTimeout(r, a.sleep * 1000)) + } else if (a.exit !== undefined) { + process.exit(a.exit) + } + } +} + +await play(actions, "") +if (!tokensSent) emit({ type: "step", turns: 1, tokens: { input: 1000, output: 200, reasoning: 0, cache_read: 0, cache_write: 0 }, cost: 0 }) +if (!scenario.count_at_start) { + st.runs_done++ + save() +} +process.exit(0) diff --git a/bin/workhorse b/bin/workhorse index 85d72c4..15ae055 100755 --- a/bin/workhorse +++ b/bin/workhorse @@ -16,6 +16,10 @@ Service: tasks [N] [--status S] [--owner O] recent tasks (S: a status, active, terminal, parked, needs_attention) attention [N] tasks that need someone: parked (needs_approval) or handoff not done/closed + wait ... [--all] [--max S] [--full] + long-poll until the task(s) finish (brief result; --all waits for every id) + stats [--days N] [--profile P] [--repo R] [--json] + worker tokens/cost by profile and day + ESTIMATE of supervisor tokens avoided cleanup-old [--days N] [--task-days N] [--parked-days N] [--dry-run] remove finished worktrees older than N days (default: daemon.json retention; parked tasks are kept unless --parked-days / retention.parked_days is set) @@ -26,6 +30,9 @@ Handoff and approval: approve a parked/blocked task and resume it in the same session + worktree reject [--instructions ""] [--note ""] [--by NAME] deny it: close the task, or resume with --instructions (do something else) + operator-token init [--force] [--enable] + create the operator token (approvals.require_operator); prints only its sha256. + With require_operator on, \`sudo workhorse approve\` sends the token. Config (the config dir is root-owned once locked: use sudo for these): repos show the allowlist add-repo [--name N] [--test ""] [--allow-test ""]... [--base BRANCH] @@ -33,6 +40,8 @@ Config (the config dir is root-owned once locked: use sudo for these): remove-repo validate static config check backends [--json] worker backends (coding-agent CLIs): status, installed, capabilities + token-savers [--terse off|lite|full] [--minimal-code off|lite|full] [--rtk on|off] [--rtk-bin PATH] | token-savers off + show or set the opt-in worker token savers (docs/token-savings.md) Checks: check-provider [profile...] one tool-calling request per profile (default + fallbacks); prints status only hello [--repo NAME] [--profile P] [--keep] @@ -161,13 +170,69 @@ try { const f = flags(rest, { instructions: "str", note: "str", timeout: "str", profile: "str", by: "str" }) const id = f._[0] if (!id) throw new Error(`usage: workhorse ${cmd} [--instructions ""] [--note ""]`) + // approvals.require_operator: send the operator token (readable only by the operator, e.g. root). + let opTok = null + if (cmd === "approve") { + const { daemonConfig } = await import("../lib/config.mjs") + const { readOperatorToken, requireOperator, operatorTokenPath } = await import("../lib/operator.mjs") + const cfg = daemonConfig() + if (requireOperator(cfg)) { + opTok = readOperatorToken(cfg) + if (!opTok) console.error(`note: approvals.require_operator is on but ${operatorTokenPath(cfg)} is not readable; this only records an approval request (use sudo)`) + } + } out(await rpc("approve_task", { - task_id: id, decision: cmd, + task_id: id, decision: cmd, ...(opTok ? { operator_token: opTok } : {}), ...(f.instructions ? { instructions: f.instructions } : {}), ...(f.note ? { note: f.note } : {}), ...(f.timeout ? { timeout_minutes: Number(f.timeout) } : {}), ...(f.profile ? { profile: f.profile } : {}), ...(f.by ? { by: f.by } : {}), }, { caller: { client: "workhorse" } })) break } + case "wait": { + const f = flags(rest, { all: "bool", max: "str", full: "bool" }) + if (!f._.length) throw new Error("usage: workhorse wait ... [--all] [--max S] [--full]") + const params = { ...(f._.length === 1 ? { task_id: f._[0] } : { task_ids: f._, mode: f.all ? "all" : "any" }), view: f.full ? "full" : "brief" } + const deadline = f.max ? Date.now() + Number(f.max) * 1000 : Infinity + let r + do { + r = await rpc("wait_task", { ...params, max_wait_s: Math.max(0, Math.min(55, (deadline - Date.now()) / 1000)) }, { caller: { client: "workhorse" } }) + } while (!r.done && Date.now() < deadline) + out(r) + process.exitCode = r.done ? 0 : 3 + break + } + case "stats": { + const f = flags(rest, { days: "str", profile: "str", repo: "str", json: "bool" }) + const r = await rpc("usage_report", { ...(f.days ? { days: Number(f.days) } : {}), ...(f.profile ? { profile: f.profile } : {}), ...(f.repo ? { repo: f.repo } : {}) }, { caller: { client: "workhorse" } }) + if (f.json) { out(r); break } + const tok = (t) => t.input + t.output + t.reasoning + const k = (n) => (n >= 1e6 ? `${(n / 1e6).toFixed(2)}M` : n >= 1e3 ? `${(n / 1e3).toFixed(1)}k` : String(n)) + const usd = (x) => (x ? `$${x.toFixed(4)}` : "-") + console.log(`Worker usage, last ${r.window_days} day(s): ${r.tasks} task(s)`) + console.log(`\n${"profile".padEnd(18)} ${"tasks".padStart(5)} ${"runs".padStart(5)} ${"tokens".padStart(9)} ${"cache_rd".padStart(9)} ${"est.cost".padStart(10)} verdicts`) + for (const p of r.by_profile) console.log(`${p.profile.padEnd(18)} ${String(p.tasks).padStart(5)} ${String(p.runs).padStart(5)} ${k(tok(p.tokens)).padStart(9)} ${k(p.tokens.cache_read).padStart(9)} ${usd(p.est_list_cost_usd).padStart(10)} ${Object.entries(p.verdicts).map(([v, n]) => `${v}:${n}`).join(" ")}`) + console.log(`\n${"day".padEnd(11)} ${"profile".padEnd(18)} ${"tasks".padStart(5)} ${"runs".padStart(5)} ${"tokens".padStart(9)} ${"est.cost".padStart(10)}`) + for (const d of r.by_day) console.log(`${d.day.padEnd(11)} ${d.profile.padEnd(18)} ${String(d.tasks).padStart(5)} ${String(d.runs).padStart(5)} ${k(tok(d.tokens)).padStart(9)} ${usd(d.est_list_cost_usd).padStart(10)}`) + const se = r.supervisor_estimate + console.log(`\nSupervisor tokens avoided (ESTIMATE): ~${k(se.est_supervisor_tokens_avoided)} over ${se.successful_tasks} successful task(s)` + (se.est_supervisor_cost_avoided_usd !== null ? `, ~$${se.est_supervisor_cost_avoided_usd} net` : "")) + console.log(` supervisor I/O with workhorse: ${se.supervisor_io.calls} call(s), ~${k(se.supervisor_io.est_tokens)} tokens; worker work tokens: ${k(se.worker_work_tokens)}`) + console.log(` formula: ${se.formula}`) + console.log(` ${se.caveat}`) + break + } + case "operator-token": { + const f = flags(rest, { force: "bool", enable: "bool" }) + if (f._[0] !== "init") throw new Error("usage: workhorse operator-token init [--force] [--enable]") + const { operatorTokenInit } = await import("../lib/admin.mjs") + out(operatorTokenInit({ force: !!f.force, enable: !!f.enable })) + break + } + case "token-savers": { + const f = flags(rest, { terse: "str", "minimal-code": "str", rtk: "str", "rtk-bin": "str" }) + const { tokenSavers } = await import("../lib/admin.mjs") + out(tokenSavers({ terse: f.terse, minimal: f["minimal-code"], rtk: f.rtk, rtkBin: f["rtk-bin"], off: f._[0] === "off" })) + break + } case "cleanup-old": { const f = flags(rest, { days: "str", "task-days": "str", "parked-days": "str", "dry-run": "bool" }) out(await rpc("cleanup_old", { @@ -227,14 +292,13 @@ try { const repo = f.repo || "hello-world" const r = await rpc("delegate_task", { repo, task: HELLO_TASK, ...(f.profile ? { profile: f.profile } : {}), timeout_minutes: Number(f.timeout) || 15 }, { caller: { client: "workhorse hello" } }) console.error(`delegated ${r.task_id} to profile ${r.profile} (${r.backend}, ${r.model}); waiting...`) - let st const t0 = Date.now() let lastPhase = "" while (true) { - st = await rpc("task_status", { task_id: r.task_id }) + const w = await rpc("wait_task", { task_id: r.task_id, max_wait_s: 10, view: "status" }, { caller: { client: "workhorse hello" } }) + const st = w.task if (st.phase !== lastPhase) { console.error(` [${Math.round((Date.now() - t0) / 1000)}s] ${st.status}: ${st.phase}`); lastPhase = st.phase } - if (st.terminal) break - await wait(5000) + if (w.done) break } const res = await rpc("task_result", { task_id: r.task_id }) out({ task_id: r.task_id, verdict: res.verdict, profile: res.profile, backend: res.backend, worker_model: res.worker_model, files_changed: res.files_changed.map((x) => x.path), tests_passed: res.test_results?.passed ?? null, seconds: res.timings?.total_s, concerns: res.remaining_concerns, errors: res.errors }) diff --git a/bin/workhorse-mcp b/bin/workhorse-mcp index f1c0e2f..63bd323 100755 --- a/bin/workhorse-mcp +++ b/bin/workhorse-mcp @@ -10,7 +10,7 @@ import { daemonConfig, secretEnvNames, VERSION } from "../lib/config.mjs" const caller = { shim_pid: process.pid, ppid: process.ppid, client: process.env.WH_CALLER || null } const server = new McpServer({ name: "grok-workhorse", version: VERSION }, { instructions: - "Delegate coding tasks to a sandboxed coding-agent worker (Kilo CLI, OpenCode, ... per profile) running on the model profiles the owner configured (see list_models). Workflow: list_repos / list_models once -> delegate_task -> poll task_status every 30-60 s until terminal -> task_result (small structured summary; its handoff.next_action says exactly what to do next and who must do it) -> optionally task_details for the diff or logs -> review/merge the branch yourself -> cleanup_task. Status needs_approval means the task is parked for a human decision: relay handoff.next_action and answer with approve_task. Record your own handoffs with update_handoff. Workers never commit, merge or push; changes stay uncommitted in the task's worktree.", + "Delegate coding tasks to a sandboxed coding-agent worker (Kilo CLI, OpenCode, ... per profile) running on cheaper model profiles the owner configured (see list_models), to save your own tokens. Workflow: list_repos / list_models once -> delegate_task (or delegate_tasks for several; use a preset or size when the owner configured them) -> wait_task (long-poll; returns the brief result when done; repeat while done=false; do not poll task_status in a loop) -> act on next.action (who must do what) -> task_details only if you need the diff or logs -> review/merge the branch yourself -> cleanup_task. Status needs_approval means the task is parked for a human decision: relay next.action and answer with approve_task. Record your own handoffs with update_handoff. Workers never commit, merge or push; changes stay uncommitted in the task's worktree.", }) let ready = null @@ -48,7 +48,8 @@ function tool(name, description, shape, method = name) { server.registerTool(name, { description, inputSchema: shape }, async (args) => { try { const result = await call(method, args) - return { content: [{ type: "text", text: JSON.stringify(result, null, 2) }] } + // Compact JSON: every byte here is a token the supervisor model reads. + return { content: [{ type: "text", text: JSON.stringify(result) }] } } catch (e) { return { isError: true, content: [{ type: "text", text: `workhorse error: ${e.message}` }] } } @@ -57,22 +58,65 @@ function tool(name, description, shape, method = name) { const taskId = z.string().regex(/^wh-[0-9]{8}-[0-9]{6}-[0-9a-f]{4}$/).describe("Task id returned by delegate_task, e.g. wh-20260926-171500-ab12") +// Extra delegate_task parameters (v0.3.0): presets/size routing and automatic follow-ups. +const delegateExtras = { + preset: z.string().max(64).optional().describe("Named preset from list_models (profile, size, timeout, test command and standing instructions chosen by the owner)"), + size: z.enum(["small", "medium", "large"]).optional().describe("Rough task size; picks the profile from the owner's routing table when no profile/preset profile is given"), + auto_fix_rounds: z.number().int().min(0).max(3).optional().describe("Automatic fix rounds (same session) when tests fail or the result is incomplete; default from config (usually 0)"), + escalate: z.boolean().optional().describe("On failure, retry automatically on the profile's escalate_to (cheap -> mid -> strong), within the owner's caps"), + auto_review: z.union([z.boolean(), z.string().max(32)]).optional().describe("Run a cheap advisory review of a successful diff (true = configured review profile, or a profile name)"), +} + tool("list_repos", "List the repositories a worker may be delegated to (allowlist), with default base branch and the default/allowed test commands. Only these names are accepted by delegate_task.", {}) -tool("list_models", "List the worker model profiles configured by the owner, with backend (coding-agent CLI), model, fallback chain, availability (credentials present, backend installed), the default profile, modes, backends, and limits (max concurrent tasks, timeout range, stall timeout).", {}) +tool("list_models", "List the worker model profiles configured by the owner, with backend (coding-agent CLI), model, fallback chain, availability (credentials present, backend installed), the default profile, presets and size routing, automatic follow-up defaults, token savers, modes, backends, and limits (max concurrent tasks, timeout range, stall timeout).", {}) + +const delegateShape = { + repo: z.string().max(64).describe("Repository name from list_repos (not a path or URL)"), + task: z.string().min(1).max(20000).describe("Self-contained instructions: goal, relevant files/functions, constraints, acceptance criteria. The worker sees nothing else from your conversation."), + profile: z.string().max(32).optional().describe("Model profile name from list_models. Omit to use the configured default profile."), + test_command: z.string().max(500).optional().describe("Test command to verify the work. Must equal the repo's default test command or match its allowed patterns (see list_repos). Omit to use the repo default."), + base_ref: z.string().max(200).optional().describe("Branch, tag or commit to start from (default: repo default branch). In review mode without review_task_id: the branch to review."), + timeout_minutes: z.number().min(0).max(1440).optional().describe("Wall-clock limit for the worker run (default and allowed range: see list_models)."), + mode: z.enum(["implement", "review"]).optional().describe("'implement' (default) edits code. 'review' runs a read-only reviewer on a copy of another task's diff (review_task_id) or on base_ref's commits."), + review_task_id: taskId.optional().describe("With mode='review': the finished task whose diff should be reviewed."), + ...delegateExtras, +} tool( "delegate_task", - "Start a coding task in the background. The daemon creates a fresh git worktree on branch workhorse/ from base_ref, runs the profile's coding-agent backend there (sandboxed: edits and shell only inside the worktree, no commits/pushes; see list_models for per-backend limits), then runs the tests itself and builds a structured result. Returns immediately with task_id; poll task_status, then call task_result. Tasks beyond the concurrency limit wait in a queue.", + "Start a coding task in the background. The daemon creates a fresh git worktree on branch workhorse/ from base_ref, runs the profile's coding-agent backend there (sandboxed: edits and shell only inside the worktree, no commits/pushes; see list_models for per-backend limits), then runs the tests itself and builds a structured result. Returns immediately with task_id; then call wait_task (long-poll, returns the brief result). Tasks beyond the concurrency limit wait in a queue. Optional: preset/size (owner-configured routing), auto_fix_rounds/escalate/auto_review (automatic follow-ups within the owner's caps).", + delegateShape, +) + +tool( + "delegate_tasks", + "Start up to 10 independent tasks in one call (each entry takes the delegate_task parameters; `defaults` are merged into every entry). Partial success: each result says ok + task_id or the error. Then wait_task with task_ids.", + { + tasks: z.array(z.object(delegateShape).partial()).min(1).max(10).describe("delegate_task parameter objects"), + defaults: z.object(delegateShape).partial().optional().describe("Parameters applied to every entry unless the entry sets them (e.g. repo, profile, preset)"), + }, +) + +tool( + "wait_task", + "Long-poll: block until the task (task_id) or tasks (task_ids, mode any|all) finish or are parked, or max_wait_s passes (default 45, max 55). Finished tasks come back with their result (view brief by default: verdict, summary, files, tests, concerns, next action + resume call, usage), so no separate task_status/task_result calls are needed. If done=false, call it again.", { - repo: z.string().max(64).describe("Repository name from list_repos (not a path or URL)"), - task: z.string().min(1).max(20000).describe("Self-contained instructions: goal, relevant files/functions, constraints, acceptance criteria. The worker sees nothing else from your conversation."), - profile: z.string().max(32).optional().describe("Model profile name from list_models. Omit to use the configured default profile."), - test_command: z.string().max(500).optional().describe("Test command to verify the work. Must equal the repo's default test command or match its allowed patterns (see list_repos). Omit to use the repo default."), - base_ref: z.string().max(200).optional().describe("Branch, tag or commit to start from (default: repo default branch). In review mode without review_task_id: the branch to review."), - timeout_minutes: z.number().min(0).max(1440).optional().describe("Wall-clock limit for the worker run (default and allowed range: see list_models)."), - mode: z.enum(["implement", "review"]).optional().describe("'implement' (default) edits code. 'review' runs a read-only reviewer on a copy of another task's diff (review_task_id) or on base_ref's commits."), - review_task_id: taskId.optional().describe("With mode='review': the finished task whose diff should be reviewed."), + task_id: taskId.optional(), + task_ids: z.array(taskId).min(1).max(20).optional().describe("Several tasks; see mode"), + mode: z.enum(["any", "all"]).optional().describe("With task_ids: return when any (default) or all are settled"), + max_wait_s: z.number().min(0).max(55).optional().describe("Default 45"), + view: z.enum(["brief", "full", "status"]).optional().describe("Result shape for finished tasks (default brief)"), + }, +) + +tool( + "usage_report", + "Worker token use and estimated cost by profile and by day, verdict counts, and an ESTIMATE of supervisor tokens avoided by delegating (formula included in the output).", + { + days: z.number().min(1).max(3650).optional().describe("Window in days (default 30)"), + profile: z.string().max(32).optional(), + repo: z.string().max(64).optional(), }, ) @@ -80,8 +124,8 @@ tool("task_status", "Cheap progress check for a task: status (queued, running, r tool( "task_result", - "Structured result of a finished task (small): verdict (success, tests_failed, no_changes, success_untested, worker_error, timeout, stalled, cancelled, interrupted, blocked, integrity_violation), worker summary, files_changed + diffstat, daemon-run test results, errors, remaining_concerns, branch/worktree/diff_path, timings, token usage/cost, activity counts, and the handoff record: state (done, needs_fix, needs_review, needs_approval, needs_input, blocked, retryable, closed), failed_checks, owner (who acts next), next_action (one exact instruction), resume (tool + suggested args, session/worktree/branch) and context. Use task_details for the full diff or logs.", - { task_id: taskId }, + "Structured result of a finished task. view='brief' returns only what you need to decide the next step (recommended; wait_task already returns it); view='full' (default) returns everything: verdict (success, tests_failed, no_changes, success_untested, worker_error, timeout, stalled, cancelled, interrupted, blocked, integrity_violation), worker summary, files_changed + diffstat, daemon-run test results, errors, remaining_concerns, branch/worktree/diff_path, timings, token usage/cost, activity counts, and the handoff record: state (done, needs_fix, needs_review, needs_approval, needs_input, blocked, retryable, closed), failed_checks, owner (who acts next), next_action (one exact instruction), resume (tool + suggested args, session/worktree/branch) and context. Use task_details for the full diff or logs.", + { task_id: taskId, view: z.enum(["full", "brief"]).optional().describe("'brief' (recommended) or 'full' (default)") }, ) tool( diff --git a/config/examples/profiles.tiered.json b/config/examples/profiles.tiered.json new file mode 100644 index 0000000..7cc7343 --- /dev/null +++ b/config/examples/profiles.tiered.json @@ -0,0 +1,72 @@ +{ + "_comment": "Tiered example for v0.3 token savings: cheap -> mid -> strong escalation, size routing, presets and opt-in automatic follow-ups. Model ids are examples; check https://openrouter.ai/models (pick models with reliable tool calling) and set price_per_mtok from your provider's current pricing.", + "default_profile": "cheap", + "profiles": { + "cheap": { + "description": "Small coder model for small, well-specified tasks. Escalates to mid when automatic escalation is on.", + "model": "openrouter/qwen3-coder-small", + "fallback": [], + "escalate_to": "mid", + "stall_minutes": 10, + "token_savers": { "terse": "lite", "minimal_code": "lite" } + }, + "mid": { + "description": "Mid-size coder model for normal implement/fix/test work.", + "model": "openrouter/qwen3-coder", + "fallback": [], + "escalate_to": "strong", + "token_savers": { "terse": "lite" } + }, + "strong": { + "description": "Large reasoning model for hard tasks and the end of the escalation chain.", + "model": "openrouter/nemotron-ultra", + "fallback": [] + } + }, + "routing": { "small": "cheap", "medium": "mid", "large": "strong" }, + "presets": { + "quick-fix": { + "description": "Small bounded fix with tests: cheap model, one automatic fix round, escalation on failure.", + "size": "small", + "timeout_minutes": 15, + "auto_fix_rounds": 1, + "escalate": true, + "instructions": "Keep the diff minimal. Do not change public APIs." + }, + "feature": { + "description": "Normal feature work on the mid model with a cheap advisory review of the diff.", + "size": "medium", + "timeout_minutes": 30, + "auto_fix_rounds": 1, + "auto_review": "cheap" + }, + "explore": { + "description": "Read-only investigation or code review; edits nothing.", + "mode": "review", + "size": "small", + "timeout_minutes": 15 + } + }, + "auto": { + "fix_rounds": 0, + "escalate": false, + "max_auto_runs": 3, + "max_tokens": 2000000, + "max_cost_usd": null, + "review": { "enabled": false, "profile": "cheap" } + }, + "providers": { + "openrouter": { + "name": "OpenRouter", + "base_url": "https://openrouter.ai/api/v1", + "api_key_env": "OPENROUTER_API_KEY", + "max_concurrent": 3, + "models": { + "qwen3-coder-small": { "id": "qwen/qwen3-coder-30b-a3b-instruct", "context": 131072, "output": 16384 }, + "qwen3-coder": { "id": "qwen/qwen3-coder", "context": 131072, "output": 16384 }, + "nemotron-ultra": { "id": "nvidia/nemotron-3-ultra-550b-a55b", "context": 131072, "output": 16384, "reasoning": true } + } + } + }, + "kilo_overlay": {} +} diff --git a/lib/admin.mjs b/lib/admin.mjs index 57eb130..ebd74df 100644 --- a/lib/admin.mjs +++ b/lib/admin.mjs @@ -11,6 +11,8 @@ import { detectTestCommand } from "./tasks.mjs" import { loadFromStore } from "./credentials.mjs" import { redact, registerSecret } from "./util.mjs" import { BACKENDS, profileBackend } from "../adapters/index.mjs" +import { hashToken, newOperatorToken, operatorTokenPath } from "./operator.mjs" +import { effectiveSavers, rtkBin, LEVELS } from "./savers.mjs" const REPOS_FILE = () => path.join(CONFIG_DIR, "repos.json") export const escapeRegex = (s) => s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&") @@ -205,6 +207,7 @@ export function healthChecks(report, daemonError) { return out } out.push(check("daemon", "ok", `pid ${report.pid}, v${report.version}, up ${report.uptime_s}s, ${report.running} running`)) + if (report.test_stub_backend_enabled) out.push(check("stub_backend", "warn", "the daemon runs with WH_ENABLE_STUB_BACKEND=1 (test-only scripted backend); restart it without that variable")) const def = report.profiles.find((p) => p.name === report.default_profile) if (!def) out.push(check("credentials", "fail", `default profile '${report.default_profile}' not found`)) else if (!def.available) out.push(check("credentials", "fail", `default profile '${def.name}' unavailable: ${def.missing_credentials?.length ? `missing ${def.missing_credentials.join(", ")}` : (def.backend_problems || []).join("; ") || "(disabled)"}`)) @@ -333,3 +336,71 @@ export function createHelloRepo(dir) { } export { DATA_DIR, APP_DIR, secretEnvNames } + +// ---------- daemon.json edits (operator CLI; needs sudo once the config is locked) ---------- +const DAEMON_FILE = () => path.join(CONFIG_DIR, "daemon.json") +function editDaemonJson(mutate) { + let doc = {} + try { + doc = JSON.parse(fs.readFileSync(DAEMON_FILE(), "utf8")) + } catch (e) { + if (e.code !== "ENOENT") throw new Error(`cannot read ${DAEMON_FILE()}: ${e.message}`) + } + mutate(doc) + try { + writeJsonKeepingOwner(DAEMON_FILE(), doc) + } catch (e) { + if (e.code === "EACCES" || e.code === "EPERM") throw new Error(`cannot write ${DAEMON_FILE()} (${e.code}): the config is root-owned (locked). Re-run with sudo.`) + throw e + } + return doc +} + +// `workhorse operator-token init [--force] [--enable]`: create the operator token file (0600, owned by +// the caller, normally root) and print its sha256; --enable also writes approvals.require_operator=true +// and the hash into daemon.json. The token itself is never printed. +export function operatorTokenInit({ force = false, enable = false } = {}) { + const cfg = daemonConfig() + const file = operatorTokenPath(cfg) + if (fs.existsSync(file) && !force) throw new Error(`${file} exists; pass --force to replace it (the old token stops working once the hash is updated)`) + const tok = newOperatorToken() + fs.mkdirSync(path.dirname(file), { recursive: true }) + fs.writeFileSync(file, tok + "\n", { mode: 0o600 }) + fs.chmodSync(file, 0o600) + const sha = hashToken(tok) + if (enable) editDaemonJson((d) => { d.approvals = { ...(d.approvals || {}), require_operator: true, operator_token_sha256: sha } }) + return { + token_file: file, operator_token_sha256: sha, require_operator: enable ? true : cfg.approvals?.require_operator === true, + next: enable ? "Restart the daemon (workhorse restart). Approvals from the supervisor now need `sudo workhorse approve `." : `Set daemon.json approvals: {"require_operator": true, "operator_token_sha256": "${sha}"} (or re-run with --enable), then restart the daemon.`, + note: "Keep the token file readable only by the operator (root); the daemon and the supervisor never need it.", + } +} + +// `workhorse token-savers [--terse L] [--minimal-code L] [--rtk on|off] [--rtk-bin PATH] | off` +export function tokenSavers(opts = {}) { + const set = opts.terse !== undefined || opts.minimal !== undefined || opts.rtk !== undefined || opts.rtkBin !== undefined || opts.off + if (set) { + for (const [k, v] of [["terse", opts.terse], ["minimal-code", opts.minimal]]) if (v !== undefined && !LEVELS.includes(v)) throw new Error(`--${k} must be one of ${LEVELS.join(", ")}`) + if (opts.rtk !== undefined && !["on", "off"].includes(opts.rtk)) throw new Error("--rtk must be on or off") + if (opts.rtkBin !== undefined && !path.isAbsolute(opts.rtkBin)) throw new Error("--rtk-bin must be an absolute path") + editDaemonJson((d) => { + const ts = opts.off ? {} : { ...(d.token_savers || {}) } + if (opts.off) Object.assign(ts, { terse: "off", minimal_code: "off", rtk: { enabled: false, bin: d.token_savers?.rtk?.bin ?? null } }) + if (opts.terse !== undefined) ts.terse = opts.terse + if (opts.minimal !== undefined) ts.minimal_code = opts.minimal + if (opts.rtk !== undefined || opts.rtkBin !== undefined) { + const cur = typeof ts.rtk === "object" && ts.rtk ? ts.rtk : { enabled: ts.rtk === true, bin: null } + ts.rtk = { ...cur, ...(opts.rtk !== undefined ? { enabled: opts.rtk === "on" } : {}), ...(opts.rtkBin !== undefined ? { bin: opts.rtkBin } : {}) } + } + d.token_savers = ts + }) + } + const cfg = daemonConfig() + const s = effectiveSavers(cfg) + const bin = rtkBin(s, cfg.env_path) || (s.rtk.bin || null) + return { + token_savers: s, rtk_binary: s.rtk.enabled ? (rtkBin(s, cfg.env_path) ? bin : `NOT FOUND (${bin || "rtk not on PATH"}); runs continue without it`) : null, + ...(set ? { next: "Applies to new worker runs (no restart needed). Per-profile overrides: profiles.json profiles..token_savers." } : {}), + docs: "docs/token-savings.md", + } +} diff --git a/lib/audit.mjs b/lib/audit.mjs index c175174..3c6fea2 100644 --- a/lib/audit.mjs +++ b/lib/audit.mjs @@ -2,7 +2,7 @@ // task text is truncated and hashed, everything passes through redact(). import fs from "node:fs" import path from "node:path" -import { dirs } from "./config.mjs" +import { dirs, daemonConfig } from "./config.mjs" import { now, redact, sha256, head } from "./util.mjs" const AUDIT = () => path.join(dirs.logs, "audit.jsonl") @@ -23,7 +23,46 @@ export function auditRecord(kind, data) { return redact(withReserved({ ts: now(), kind }, data)) } +// Size-based rotation: audit.jsonl -> audit.jsonl.1 -> ... -> audit.jsonl. (the oldest is dropped). +// daemon.json audit: { max_mb (default 20; 0 = never rotate), keep (default 5) }. Checked on the first +// write and then every 200 writes, so the cost is one stat per 200 records. +let writes = 0 +export function rotateAudit({ file = AUDIT(), maxBytes, keep } = {}) { + let lim = maxBytes + let k = keep + if (lim === undefined || k === undefined) { + let a = {} + try { + a = daemonConfig().audit || {} + } catch {} + if (lim === undefined) lim = (a.max_mb ?? 20) * 1024 * 1024 + if (k === undefined) k = a.keep ?? 5 + } + k = Math.max(1, Math.min(50, Math.floor(Number(k) || 5))) + if (!(lim > 0)) return false + let size = 0 + try { + size = fs.statSync(file).size + } catch { + return false + } + if (size < lim) return false + try { + fs.rmSync(`${file}.${k}`, { force: true }) + } catch {} + for (let i = k - 1; i >= 1; i--) { + try { + fs.renameSync(`${file}.${i}`, `${file}.${i + 1}`) + } catch {} + } + fs.renameSync(file, `${file}.1`) + return true +} + export function audit(kind, data) { + try { + if (writes++ % 200 === 0) rotateAudit() + } catch {} try { fs.appendFileSync(AUDIT(), JSON.stringify(auditRecord(kind, data)) + "\n", { mode: 0o600 }) } catch { @@ -41,5 +80,7 @@ export function sanitizeParams(method, params) { p[k] = head(p[k], 200) } } + if (p.operator_token !== undefined) p.operator_token = "[given]" + if (Array.isArray(p.tasks)) p.tasks = p.tasks.slice(0, 20).map((t) => (t && typeof t === "object" ? sanitizeParams(method, t) : t)) return p } diff --git a/lib/config.mjs b/lib/config.mjs index a050031..6091887 100644 --- a/lib/config.mjs +++ b/lib/config.mjs @@ -90,8 +90,34 @@ const DEFAULTS = { ro_binds: [], auto_bind_toolchain: true, }, + // Opt-in token savers for worker runs (lib/savers.mjs, docs/token-savings.md). A profile may override + // with its own token_savers object in profiles.json. + token_savers: { terse: "off", minimal_code: "off", rtk: { enabled: false, bin: null } }, + // approvals.require_operator: approve_task from the supervisor only records a request; a human confirms + // with `sudo workhorse approve ` (operator token; the daemon stores only its sha256). + approvals: { require_operator: false, operator_token_sha256: null, operator_token_path: null }, + // audit.jsonl size-based rotation (max_mb 0 = never rotate). + audit: { max_mb: 20, keep: 5 }, + // usage_report: supervisor price (USD per million tokens {input, output}) for the ESTIMATE of supervisor + // cost avoided, and the chars-per-token ratio used to convert supervisor I/O. + supervisor: { price_per_mtok: null, chars_per_token: 4 }, } +export const SIZES = ["small", "medium", "large"] +// Hard ceilings for automatic follow-ups, whatever the config says. +export const AUTO_HARD = { fix_rounds: 3, max_auto_runs: 6 } +export const AUTO_DEFAULTS = { + fix_rounds: 0, // automatic fix rounds per profile (same session) after a failed check + fix_on: ["tests_failed", "no_result_block", "partial"], + escalate: false, // on failure, move up the profile's escalate_to chain (fresh session, same worktree) + escalate_on: ["tests_failed", "worker_error", "partial"], + max_auto_runs: 3, // automatic follow-up runs per task (fix + escalate), hard cap AUTO_HARD.max_auto_runs + max_tokens: null, // stop automatic follow-ups once the task used this many (input+output+reasoning) tokens + max_cost_usd: null, // ... or this estimated list cost (needs price_per_mtok on the profiles) + review: { enabled: false, profile: null, on: ["success", "success_untested"], timeout_minutes: 10 }, +} +const AUTO_TRIGGERS = ["tests_failed", "no_result_block", "partial", "no_changes", "worker_error"] + function readJson(file, fallback) { try { return JSON.parse(fs.readFileSync(file, "utf8")) @@ -250,10 +276,36 @@ export function profilesConfig() { } } const names = Object.keys(profiles) + const presets = {} + for (const [k, v] of Object.entries(raw.presets || {})) if (v && typeof v === "object" && !Array.isArray(v) && !k.startsWith("_")) presets[k] = v + const routing = {} + for (const sz of SIZES) if (typeof raw.routing?.[sz] === "string" && raw.routing[sz]) routing[sz] = raw.routing[sz] + const a = raw.auto && typeof raw.auto === "object" ? raw.auto : {} + const num = (v, d) => (typeof v === "number" && Number.isFinite(v) && v >= 0 ? v : d) + const list = (v, d) => (Array.isArray(v) ? v.filter((x) => AUTO_TRIGGERS.includes(x) || ["success", "success_untested"].includes(x)) : d) + const rv = a.review && typeof a.review === "object" ? a.review : {} + const auto = { + fix_rounds: Math.min(AUTO_HARD.fix_rounds, Math.floor(num(a.fix_rounds, AUTO_DEFAULTS.fix_rounds))), + fix_on: list(a.fix_on, AUTO_DEFAULTS.fix_on), + escalate: a.escalate === true, + escalate_on: list(a.escalate_on, AUTO_DEFAULTS.escalate_on), + max_auto_runs: Math.min(AUTO_HARD.max_auto_runs, Math.floor(num(a.max_auto_runs, AUTO_DEFAULTS.max_auto_runs))), + max_tokens: num(a.max_tokens, null) || null, + max_cost_usd: num(a.max_cost_usd, null) || null, + review: { + enabled: rv.enabled === true, + profile: typeof rv.profile === "string" && rv.profile ? rv.profile : null, + on: list(rv.on, AUTO_DEFAULTS.review.on), + timeout_minutes: num(rv.timeout_minutes, AUTO_DEFAULTS.review.timeout_minutes) || AUTO_DEFAULTS.review.timeout_minutes, + }, + } return { default_profile: raw.default_profile || names[0] || "default", profiles, providers, + presets, + routing, + auto, kilo_overlay: raw.kilo_overlay || {}, opencode_overlay: raw.opencode_overlay || {}, } @@ -308,6 +360,20 @@ export function kiloProviderOverlay(pc) { return Object.keys(provider).length ? { enabled_providers: Object.keys(provider), provider } : {} } +export const PRESET_KEYS = ["description", "profile", "size", "mode", "timeout_minutes", "test_command", "instructions", "auto_fix_rounds", "escalate", "auto_review"] + +// token_savers shape check (kept here so config.mjs has no import cycle with savers.mjs). +function saverProblemsLite(obj, where) { + const out = [] + if (obj === undefined || obj === null) return out + if (typeof obj !== "object" || Array.isArray(obj)) return [`${where} must be an object`] + const levels = ["off", "lite", "full"] + for (const k of Object.keys(obj)) if (!["terse", "minimal_code", "rtk"].includes(k)) out.push(`${where}: unknown key '${k}'`) + for (const k of ["terse", "minimal_code"]) if (obj[k] !== undefined && typeof obj[k] !== "boolean" && !levels.includes(obj[k])) out.push(`${where}.${k} must be one of ${levels.join(", ")}`) + if (obj.rtk !== undefined && typeof obj.rtk !== "boolean" && (typeof obj.rtk !== "object" || Array.isArray(obj.rtk))) out.push(`${where}.rtk must be an object {enabled, bin}`) + return out +} + // Static checks used by `workhorse health` / `workhorse validate`. Returns a list of problems (strings). export function validateConfig() { const problems = [] @@ -332,7 +398,23 @@ export function validateConfig() { else if (def?.base_url && !def.models?.[mk]) problems.push(`profile '${name}': model '${mk}' is not defined under providers.${prov}.models`) for (const f of p.fallback) if (!pc.profiles[f]) problems.push(`profile '${name}': fallback profile '${f}' is not defined`) if (p.backend && !Object.hasOwn(DEFAULTS.backends, p.backend)) problems.push(`profile '${name}': unknown backend '${p.backend}' (known: ${Object.keys(DEFAULTS.backends).join(", ")})`) + if (p.escalate_to !== undefined && (typeof p.escalate_to !== "string" || !pc.profiles[p.escalate_to] || p.escalate_to === name)) problems.push(`profile '${name}': escalate_to must name another defined profile`) + if (p.stall_minutes !== undefined && !(typeof p.stall_minutes === "number" && p.stall_minutes > 0)) problems.push(`profile '${name}': stall_minutes must be a positive number`) + problems.push(...saverProblemsLite(p.token_savers, `profile '${name}': token_savers`)) } + for (const [name, x] of Object.entries(pc.presets)) { + if (x.profile !== undefined && !pc.profiles[x.profile]) problems.push(`preset '${name}': profile '${x.profile}' is not defined`) + if (x.size !== undefined && !SIZES.includes(x.size)) problems.push(`preset '${name}': size must be one of ${SIZES.join(", ")}`) + if (x.mode !== undefined && !["implement", "review"].includes(x.mode)) problems.push(`preset '${name}': mode must be implement or review`) + for (const k of Object.keys(x)) if (!PRESET_KEYS.includes(k)) problems.push(`preset '${name}': unknown key '${k}' (known: ${PRESET_KEYS.join(", ")})`) + } + for (const [sz, prof] of Object.entries(pc.routing)) if (!pc.profiles[prof]) problems.push(`routing.${sz}: profile '${prof}' is not defined`) + if (pc.auto.review.profile && !pc.profiles[pc.auto.review.profile]) problems.push(`auto.review.profile '${pc.auto.review.profile}' is not defined`) + try { + const d = daemonConfig() + problems.push(...saverProblemsLite(d.token_savers, "daemon.json token_savers")) + if (d.approvals?.require_operator === true && !/^[0-9a-f]{64}$/i.test(String(d.approvals.operator_token_sha256 || ""))) problems.push("approvals.require_operator is on but approvals.operator_token_sha256 is not a sha256 hex digest (run `workhorse operator-token init`)") + } catch {} try { reposConfig() } catch (e) { diff --git a/lib/daemon.mjs b/lib/daemon.mjs index ad57902..707c019 100644 --- a/lib/daemon.mjs +++ b/lib/daemon.mjs @@ -55,6 +55,8 @@ async function waitForPreviousDaemon(maxMs = 60000) { } } +const QUIET_METHODS = new Set(["health", "health_report", "offer_env", "cleanup_old", "usage_report"]) + export async function startDaemon() { ensureDir(dirs.run, 0o700) fs.chmodSync(dirs.run, 0o700) @@ -76,14 +78,17 @@ export async function startDaemon() { const methods = { health: () => ({ ok: true, version: VERSION, pid: process.pid, running: mgr.live.size, tasks: mgr.tasks.size, credentials: Object.keys(mgr.secretEnv()), credential_sources: mgr.credentialSources() }), - health_report: () => ({ ok: true, ...mgr.healthReport(), credential_sources: mgr.credentialSources() }), + health_report: () => ({ ok: true, ...mgr.healthReport(), credential_sources: mgr.credentialSources(), ...(process.env.WH_ENABLE_STUB_BACKEND === "1" ? { test_stub_backend_enabled: true } : {}) }), cleanup_old: (p) => mgr.sweep({ worktree_days: p.worktree_days, task_days: p.task_days, parked_days: p.parked_days, dry_run: p.dry_run === true, trigger: "manual" }), offer_env: (p) => mgr.offerEnv(p.env), delegate_task: (p) => mgr.delegate(p), + delegate_tasks: (p) => mgr.delegateMany(p), + wait_task: (p) => mgr.wait(p), + usage_report: (p) => mgr.usage(p), task_status: (p) => mgr.status(p.task_id), - task_result: (p) => mgr.result(p.task_id), + task_result: (p) => mgr.result(p.task_id, p.view), task_details: (p) => mgr.details(p.task_id, p.kind, p.offset, p.max_bytes, p.run), - continue_task: (p) => mgr.continueTask(p.task_id, p.instructions, p.timeout_minutes, p.profile), + continue_task: (p) => mgr.continueTask(p.task_id, p.instructions, p.timeout_minutes, p.profile, { operatorToken: p.operator_token }), cancel_task: (p) => mgr.cancel(p.task_id), list_tasks: (p) => mgr.list(p), get_handoff: (p) => mgr.handoff(p.task_id), @@ -128,6 +133,12 @@ export async function startDaemon() { try { const result = await fn(params || {}, caller) if (method !== "health" && method !== "health_report" && !(method === "offer_env" && !result?.changed)) audit("rpc", { method, caller, params: sanitizeParams(method, params), ok: true, ms: Date.now() - started, task_id: params?.task_id || result?.task_id }) + // Supervisor (MCP) I/O per task for usage_report's estimate; the operator CLI is not counted. + if (caller?.client !== "workhorse" && !QUIET_METHODS.has(method)) { + try { + mgr.recordSupervisorIo(params, result, JSON.stringify(params || {}).length, JSON.stringify(result ?? null).length) + } catch {} + } send(200, { result }) } catch (e) { const user = e instanceof UserError diff --git a/lib/handoff.mjs b/lib/handoff.mjs index d4a7ba8..36d1a24 100644 --- a/lib/handoff.mjs +++ b/lib/handoff.mjs @@ -48,7 +48,7 @@ function hasText(s) { return typeof s === "string" && s.trim() && !/^(none|n\/a|-)\.?$/i.test(s.trim()) } -const SETUP_ERROR_RE = /missing credentials|cannot run:|is disabled|unknown profile|failed to start the .* worker|worktree is missing|not installed/i +export const SETUP_ERROR_RE = /missing credentials|cannot run:|is disabled|unknown profile|failed to start the .* worker|worktree is missing|not installed/i // Build the handoff record for a finished task. `t` is the persisted task (t.result set by finalize). export function deriveHandoff(t, { by = "daemon", at = new Date().toISOString() } = {}) { @@ -189,6 +189,23 @@ export function deriveHandoff(t, { by = "daemon", at = new Date().toISOString() tool = null } + // Advisory auto-review by a cheap profile (auto_review): it can only make a "done" task need a look. + const rv = r.review + if (rv && state === "done") { + if (rv.verdict === "request_changes") { + const f = (rv.findings || []).slice(0, 2).join("; ") || head(rv.summary || "", 200) + failed.push({ check: "auto_review", verdict: rv.verdict, profile: rv.profile, findings: (rv.findings || []).slice(0, 5), task_id: rv.task_id }) + state = "needs_review" + next = `The cheap auto-reviewer (${rv.profile}, advisory) requested changes: ${head(f, 300)}. Check these against the diff (task_details kind=diff); if valid run continue_task with them, otherwise merge ${t.branch} and cleanup_task.` + tool = "continue_task" + args = cont(`An automated reviewer reported: ${head((rv.findings || []).join("; ") || rv.summary || "", 1200)}. Fix the findings that are valid (ignore ones that are wrong, and say why under concerns), rerun the tests and end with the ## RESULT block.`) + } else if (rv.verdict === "approve") { + next = `${next} (The cheap auto-reviewer ${rv.profile} approved; advisory only.)` + } + } + const auto = r.auto + if (auto?.runs && state !== "done") next = `${next} Automatic follow-ups already ran ${auto.runs}x (${auto.trail.map((e) => `${e.profile}:${e.verdict}`).join(" -> ")}${auto.stopped_reason ? `; stopped: ${auto.stopped_reason}` : ""}).` + return { state, owner, next_action: head(next, 1000), failed_checks: failed, diff --git a/lib/operator.mjs b/lib/operator.mjs new file mode 100644 index 0000000..72ce9c8 --- /dev/null +++ b/lib/operator.mjs @@ -0,0 +1,47 @@ +// Operator confirmation for approvals (daemon.json approvals.require_operator). +// +// With require_operator on, approve_task from the MCP supervisor only RECORDS an approval request; a +// human confirms it with `sudo workhorse approve `, which reads the operator token file and +// sends the token along. The daemon stores only the token's SHA-256 (approvals.operator_token_sha256), +// so neither the supervisor (which never sees the file) nor the daemon config can produce it. +import crypto from "node:crypto" +import fs from "node:fs" +import path from "node:path" +import { CONFIG_DIR } from "./config.mjs" + +export const TOKEN_RE = /^[0-9a-f]{64}$/ + +export function operatorTokenPath(cfg) { + return cfg?.approvals?.operator_token_path || path.join(CONFIG_DIR, "operator.token") +} + +export function hashToken(tok) { + return crypto.createHash("sha256").update(String(tok).trim()).digest("hex") +} + +export function newOperatorToken() { + return crypto.randomBytes(32).toString("hex") +} + +// True when `given` hashes to the configured sha256 (constant-time compare). +export function checkOperatorToken(cfg, given) { + const want = String(cfg?.approvals?.operator_token_sha256 || "").toLowerCase() + if (!TOKEN_RE.test(want) || typeof given !== "string" || !given) return false + const a = Buffer.from(hashToken(given), "hex") + const b = Buffer.from(want, "hex") + return a.length === b.length && crypto.timingSafeEqual(a, b) +} + +export function requireOperator(cfg) { + return cfg?.approvals?.require_operator === true +} + +// Read the operator token file (CLI side). Returns null when absent/unreadable. +export function readOperatorToken(cfg) { + try { + const v = fs.readFileSync(operatorTokenPath(cfg), "utf8").trim() + return TOKEN_RE.test(v) ? v : null + } catch { + return null + } +} diff --git a/lib/savers.mjs b/lib/savers.mjs new file mode 100644 index 0000000..a3ff4ad --- /dev/null +++ b/lib/savers.mjs @@ -0,0 +1,72 @@ +// Opt-in token savers for worker runs (see docs/token-savings.md). All off by default. +// +// daemon.json (defaults for every profile; a profile in profiles.json may override with its own +// `token_savers` object): +// "token_savers": { +// "terse": "off" | "lite" | "full", // worker prose style: fewer output tokens +// "minimal_code": "off" | "lite" | "full", // smallest-diff bias: fewer output tokens, smaller diffs +// "rtk": { "enabled": false, "bin": null } // rewrite worker shell commands through RTK +// } +// +// terse and minimal_code are short instruction fragments appended to the worker's task message. They +// are our own wording, inspired by Caveman (github.com/JuliusBrussee/caveman, MIT outside its BSL engine dirs) and +// Ponytail (github.com/DietrichGebert/ponytail, MIT); no code or text is copied from either project +// (see NOTICE). rtk runs RTK (github.com/rtk-ai/rtk, Apache-2.0) as an external binary: the guard +// plugin asks `rtk rewrite` for a compact equivalent of each bash command the worker runs (e.g. +// `git status` -> `rtk git status`). The daemon's own test run never goes through any of this. +import fs from "node:fs" +import path from "node:path" +import { which } from "./config.mjs" + +export const LEVELS = ["off", "lite", "full"] +export const DEFAULT_SAVERS = { terse: "off", minimal_code: "off", rtk: { enabled: false, bin: null } } + +const level = (v) => (v === true ? "lite" : v === false || v == null ? "off" : LEVELS.includes(v) ? v : "off") + +// Effective savers for a run: daemon.json token_savers, then the profile's token_savers. +export function effectiveSavers(cfg, profile = {}) { + const d = cfg?.token_savers || {} + const p = profile?.token_savers || {} + const rtk = { ...DEFAULT_SAVERS.rtk, ...(typeof d.rtk === "object" ? d.rtk : { enabled: d.rtk === true }), ...(typeof p.rtk === "object" ? p.rtk : p.rtk === undefined ? {} : { enabled: p.rtk === true }) } + return { + terse: level(p.terse ?? d.terse), + minimal_code: level(p.minimal_code ?? d.minimal_code), + rtk: { enabled: rtk.enabled === true, bin: typeof rtk.bin === "string" && rtk.bin ? rtk.bin : null }, + } +} + +// Absolute path of the rtk binary, or null. +export function rtkBin(s, envPath) { + if (!s?.rtk?.enabled) return null + const b = s.rtk.bin || which("rtk", envPath || process.env.PATH || "") + if (!b || !path.isAbsolute(b)) return null + try { + fs.accessSync(b, fs.constants.X_OK) + return b + } catch { + return null + } +} + +const TERSE = { + lite: "Output style (token saver): keep prose short. Do not narrate what you are about to do, do not restate the task or echo tool output back, no pleasantries. Short sentences are fine. Never shorten code, commands, paths, numbers, error messages or negations. The final ## RESULT block keeps its exact format and all fields; its summary stays clear and complete.", + full: "Output style (token saver): minimal prose. No narration between tool calls, no restating the task, no recap of tool output, no pleasantries; use terse fragments. Never shorten code, commands, paths, numbers, error messages or negations. The final ## RESULT block keeps its exact format and all fields; summary: at most two short, precise sentences.", +} + +const MINIMAL = { + lite: "Code style (token saver): make the smallest change that fully solves the task. Before writing code, check in order: is it needed at all; does the standard library, the platform or an already-installed dependency do it; can it be a one-line change. Reuse existing helpers. Do not add dependencies, abstractions with a single user, configuration for fixed values or scaffolding for later. Prefer deleting code to adding it when that solves the task. Never drop input validation at trust boundaries, error handling that prevents data loss, security checks, tests the task asks for, or anything explicitly requested. The repository's conventions win over these rules.", + full: "Code style (token saver): shortest correct diff wins. Before writing code, check in order: is it needed at all; does the standard library, the platform or an already-installed dependency do it; can it be a one-line change. Reuse existing helpers; no new dependencies, no single-use abstractions, no configuration for fixed values, no scaffolding for later; delete rather than add where that solves the task. Never drop input validation at trust boundaries, error handling that prevents data loss, security checks, tests the task asks for, or anything explicitly requested. The repository's conventions win over these rules. List anything you deliberately left out under concerns as 'skipped: , add when '.", +} + +// Instruction fragments appended to a worker message (empty string when all prompt savers are off). +// minimal_code does not apply to read-only review tasks. +export function saverFragments(s, mode = "implement") { + const parts = [] + if (s?.terse && s.terse !== "off") parts.push(TERSE[s.terse]) + if (mode !== "review" && s?.minimal_code && s.minimal_code !== "off") parts.push(MINIMAL[s.minimal_code]) + return parts.length ? `\n\n${parts.join("\n")}` : "" +} + +export function saverSummary(s) { + return { terse: s.terse, minimal_code: s.minimal_code, rtk: s.rtk.enabled } +} diff --git a/lib/tasks.mjs b/lib/tasks.mjs index 0f67ebf..18c12f5 100644 --- a/lib/tasks.mjs +++ b/lib/tasks.mjs @@ -5,17 +5,23 @@ import path from "node:path" import crypto from "node:crypto" import { spawn } from "node:child_process" import { dirs, daemonConfig, reposConfig, profilesConfig, secretEnvNames, REPO_NAME_RE, DATA_DIR, VERSION } from "./config.mjs" -import { now, writeJsonAtomic, readJsonSafe, redact, registerSecret, tail, head, readRange, ensureDir } from "./util.mjs" +import { now, writeJsonAtomic, readJsonSafe, redact, registerSecret, tail, head, readRange, ensureDir, sleep } from "./util.mjs" import * as G from "./git.mjs" import { alive, procStart, killTree } from "./procs.mjs" import { audit, withReserved } from "./audit.mjs" import { loadFromStore } from "./credentials.mjs" import { outerSandboxArgs } from "./sandbox.mjs" import { getBackend, profileBackend, backendProblems, backendSummary } from "../adapters/index.mjs" -import { deriveHandoff, handoffSummary, normalizeWorkerStatus, extractFailingTests, HANDOFF_STATES, PARK_STATES, QUIET_STATES, APPROVABLE_STATES, OWNER_RE } from "./handoff.mjs" +import { deriveHandoff, handoffSummary, normalizeWorkerStatus, extractFailingTests, HANDOFF_STATES, PARK_STATES, QUIET_STATES, APPROVABLE_STATES, OWNER_RE, SETUP_ERROR_RE } from "./handoff.mjs" +import { briefResult, dedupeHandoff, VIEWS } from "./views.mjs" +import { effectiveSavers, saverFragments, saverSummary, rtkBin } from "./savers.mjs" +import { usageReport, SUCCESS } from "./usage.mjs" +import { requireOperator, checkOperatorToken } from "./operator.mjs" +import { AUTO_HARD, SIZES } from "./config.mjs" const TERMINAL = new Set(["completed", "failed", "timeout", "stalled", "cancelled", "interrupted"]) -const ACTIVE = new Set(["queued", "running", "testing", "finalizing", "retry_wait"]) +// reviewing: the worker finished and an automatic advisory review task (auto_review) is running on its diff. +const ACTIVE = new Set(["queued", "running", "testing", "finalizing", "retry_wait", "reviewing"]) // Parked: the worker has finished (no process, result ready) but the task waits for a human decision // (approve_task / continue_task). Not closed, so retention never removes its worktree automatically // (unless retention.parked_days is set). Pollers treat it like a terminal status (task_status terminal: true). @@ -24,6 +30,10 @@ const settled = (s) => TERMINAL.has(s) || PARKED.has(s) export { TERMINAL, PARKED, ACTIVE } const hasText = (s) => typeof s === "string" && !!s.trim() && !/^(none|n\/a|-)\.?$/i.test(s.trim()) const TASK_ID_RE = /^wh-[0-9]{8}-[0-9]{6}-[0-9a-f]{4}$/ +const BATCH_MAX = 10 // delegate_tasks +const WAIT_MAX_IDS = 20 +export const WAIT_CAP_S = 55 // wait_task: stays under the common 60 s MCP client request timeout +export const WAIT_DEFAULT_S = 45 class UserError extends Error {} export { UserError } @@ -116,6 +126,7 @@ export class Manager { this.storeMissing = {} // name -> reason the store could not provide it (no values) this.cooldown = {} this.shuttingDown = false + this.unattributedIo = { calls: 0, request_chars: 0, response_chars: 0 } // supervisor calls not tied to a task } // ---------- persistence ---------- @@ -197,6 +208,14 @@ export class Manager { await this.finalize(t, { tests: false, status: "interrupted" }) } } + // Tasks waiting for their automatic review: finish them if the review task is gone or already done + // (a review task that was running was finalized above, which already applied its verdict). + for (const t of this.tasks.values()) { + if (t.status !== "reviewing") continue + const child = t.review_pending?.task_id ? this.tasks.get(t.review_pending.task_id) : null + if (!child) await this.finishReview(t, null, "review task missing after restart") + else if (settled(child.status) && child.result) await this.finishReview(t, child) + } } // ---------- config helpers ---------- @@ -306,26 +325,75 @@ export class Manager { } // ---------- public API ---------- - async delegate(params) { - const allowedKeys = new Set(["repo", "task", "profile", "test_command", "base_ref", "timeout_minutes", "mode", "review_task_id"]) + // Which profile a new task uses: explicit profile > preset.profile > routing[size] (size from the call + // or the preset) > default_profile. + resolveChoice(params, pc = profilesConfig()) { + let preset = null + if (params.preset !== undefined && params.preset !== null) { + if (typeof params.preset !== "string" || !Object.hasOwn(pc.presets, params.preset)) throw new UserError(`unknown preset '${String(params.preset).slice(0, 40)}'; call list_models for presets`) + preset = { name: params.preset, ...pc.presets[params.preset] } + } + const size = params.size ?? preset?.size + if (size !== undefined && size !== null && !SIZES.includes(size)) throw new UserError(`size must be one of ${SIZES.join(", ")}`) + let profileName = params.profile ?? preset?.profile + let routed = null + if (!profileName && size && pc.routing[size]) { + profileName = pc.routing[size] + routed = size + } + return { preset, size: size || null, profileName, routed } + } + + // Automatic follow-ups for a new implement task (profiles.json `auto`, overridable per call/preset). + autoSettings(params, preset, mode, pc = profilesConfig()) { + const a = pc.auto + const pick = (k, d) => params[k] ?? preset?.[k] ?? d + const fix = pick("auto_fix_rounds", a.fix_rounds) + if (!Number.isInteger(fix) || fix < 0 || fix > AUTO_HARD.fix_rounds) throw new UserError(`auto_fix_rounds must be an integer 0..${AUTO_HARD.fix_rounds}`) + const escalate = pick("escalate", a.escalate) + if (typeof escalate !== "boolean") throw new UserError("escalate must be true or false") + let rv = pick("auto_review", a.review.enabled ? a.review.profile || true : false) + if (rv !== true && rv !== false && typeof rv !== "string") throw new UserError("auto_review must be true, false or a profile name") + let reviewProfile = null + if (rv) { + reviewProfile = rv === true ? a.review.profile || pc.default_profile : rv + const rp = this.profile(reviewProfile) + this.checkBackend(rp) + } + if (mode !== "implement") return null + if (!fix && !escalate && !reviewProfile) return null + return { + fix_rounds: fix, fix_on: a.fix_on, escalate, escalate_on: a.escalate_on, max_auto_runs: a.max_auto_runs, + max_tokens: a.max_tokens, max_cost_usd: a.max_cost_usd, review_profile: reviewProfile, review_on: a.review.on, review_timeout_min: a.review.timeout_minutes, + } + } + + async delegate(params, internal = {}) { + const allowedKeys = new Set(["repo", "task", "profile", "test_command", "base_ref", "timeout_minutes", "mode", "review_task_id", "preset", "size", "auto_fix_rounds", "escalate", "auto_review"]) for (const k of Object.keys(params || {})) if (!allowedKeys.has(k)) throw new UserError(`unknown parameter '${k}'`) const cfg = daemonConfig() + const pc = profilesConfig() const repo = this.repo(params.repo) - const mode = params.mode || "implement" + const choice = this.resolveChoice(params, pc) + const preset = choice.preset + const mode = params.mode || preset?.mode || "implement" if (!["implement", "review"].includes(mode)) throw new UserError("mode must be 'implement' or 'review'") - const text = params.task + let text = params.task if (typeof text !== "string" || !text.trim()) throw new UserError("task (description) is required") + if (preset?.instructions) text = `${text}\n\nStanding instructions (preset ${preset.name}):\n${preset.instructions}` if (text.length > cfg.limits.task_max_chars) throw new UserError(`task description exceeds ${cfg.limits.task_max_chars} chars`) - const profile = this.profile(params.profile) + const profile = this.profile(choice.profileName) this.checkBackend(profile) - const testCommand = this.validateTestCommand(repo, params.test_command) - const timeout = this.timeoutMin(params.timeout_minutes) + const testCommand = this.validateTestCommand(repo, params.test_command ?? preset?.test_command) + const timeout = this.timeoutMin(params.timeout_minutes ?? preset?.timeout_minutes) + const auto = this.autoSettings(params, preset, mode, pc) let review = null if (params.review_task_id !== undefined) { if (mode !== "review") throw new UserError("review_task_id requires mode='review'") const target = this.get(params.review_task_id) if (target.repo !== repo.name) throw new UserError("review_task_id belongs to a different repo") - if (!settled(target.status)) throw new UserError("the task to review is still active; wait until it finishes") + const ownReview = internal.autoReviewOf === target.id && target.status === "reviewing" + if (!settled(target.status) && !ownReview) throw new UserError("the task to review is still active; wait until it finishes") const patchFile = path.join(this.taskDir(target.id), "diff.patch") if (!fs.existsSync(patchFile) || fs.statSync(patchFile).size === 0) throw new UserError("the task to review has no diff") review = { task_id: target.id, base_commit: target.base_commit, patch: patchFile } @@ -356,12 +424,40 @@ export class Manager { stats: { turns: 0, tool_calls: 0, tool_errors: 0, failed_commands: 0, blocked_calls: 0, tokens: { input: 0, output: 0, reasoning: 0, cache_read: 0, cache_write: 0 }, reported_cost: 0, by_tool: {} }, blocked: [], main_fingerprint: fp, result: null, previous_results: [], worktree_removed: false, pending_run: { kind: "initial", profile: profile.name, timeout_min: timeout }, + preset: preset?.name || null, size: choice.size, routed_by_size: choice.routed, auto, auto_trail: [], + auto_review_of: internal.autoReviewOf || null, supervisor_io: { calls: 0, request_chars: 0, response_chars: 0 }, } this.tasks.set(id, t) this.save(t) - this.event(t, "created", { repo: repo.name, profile: profile.name, backend: profile.backend, mode, base_commit: baseCommit }) + this.event(t, "created", { repo: repo.name, profile: profile.name, backend: profile.backend, mode, base_commit: baseCommit, ...(preset ? { preset: preset.name } : {}), ...(choice.routed ? { routed_by_size: choice.routed } : {}), ...(auto ? { auto_fix_rounds: auto.fix_rounds, escalate: auto.escalate, auto_review: auto.review_profile } : {}), ...(internal.autoReviewOf ? { auto_review_of: internal.autoReviewOf } : {}) }) await this.schedule() - return { task_id: id, status: t.status, repo: repo.name, mode, profile: profile.name, backend: profile.backend, model: profile.model, branch, worktree_path: wt, base_ref: t.base_ref, base_commit: baseCommit, queue_position: this.queuePosition(t), next_step: "Poll task_status every 30-60s; when status is terminal call task_result." } + return { + task_id: id, status: t.status, repo: repo.name, mode, profile: profile.name, backend: profile.backend, model: profile.model, branch, worktree_path: wt, base_ref: t.base_ref, base_commit: baseCommit, + queue_position: this.queuePosition(t), ...(preset ? { preset: preset.name } : {}), ...(choice.routed ? { routed_by_size: choice.routed } : {}), + ...(auto ? { auto: { fix_rounds: auto.fix_rounds, escalate: auto.escalate, review_profile: auto.review_profile, max_auto_runs: auto.max_auto_runs } } : {}), + next_step: "Call wait_task with this task_id (long-poll, returns the brief result when done); repeat while done=false.", + } + } + + // Several delegate_task calls in one request (partial success: each entry reports ok or error). + async delegateMany(p = {}) { + const allowed = new Set(["tasks", "defaults"]) + for (const k of Object.keys(p)) if (!allowed.has(k)) throw new UserError(`unknown parameter '${k}'`) + if (!Array.isArray(p.tasks) || p.tasks.length < 1 || p.tasks.length > BATCH_MAX) throw new UserError(`tasks must be an array of 1..${BATCH_MAX} delegate_task parameter objects`) + const defaults = p.defaults ?? {} + if (typeof defaults !== "object" || Array.isArray(defaults)) throw new UserError("defaults must be an object") + const results = [] + for (const [i, one] of p.tasks.entries()) { + try { + if (!one || typeof one !== "object" || Array.isArray(one)) throw new UserError("each task must be an object") + const r = await this.delegate({ ...defaults, ...one }) + results.push({ index: i, ok: true, task_id: r.task_id, status: r.status, profile: r.profile, queue_position: r.queue_position }) + } catch (e) { + results.push({ index: i, ok: false, error: head(e.message, 300) }) + } + } + const ids = results.filter((r) => r.ok).map((r) => r.task_id) + return { created: ids.length, failed: results.length - ids.length, results, task_ids: ids, next_step: ids.length ? "Call wait_task with task_ids (mode any or all); repeat while done=false." : "Nothing was created; fix the errors." } } queuePosition(t) { @@ -383,14 +479,110 @@ export class Manager { handoff: settled(t.status) ? handoffSummary(this.handoffOf(t)) : null, hint: PARKED.has(t.status) ? "Parked: waiting for a human decision. Read handoff.next_action; answer with approve_task (or continue_task)." - : TERMINAL.has(t.status) ? "Call task_result for the structured summary; handoff.next_action says what to do next." : "Still working; poll again later.", + : TERMINAL.has(t.status) ? "Call task_result for the structured summary; handoff.next_action says what to do next." : "Still working; call wait_task (long-poll) instead of polling.", } } - result(id) { + checkView(v, dflt = "full") { + if (v === undefined || v === null || v === "") return dflt + if (!VIEWS.includes(v)) throw new UserError(`view must be one of ${VIEWS.join(", ")}`) + return v + } + + // view "full" (default): the whole result; the handoff context omits fields already top-level. + // view "brief": the small decision summary (lib/views.mjs). + result(id, view) { const t = this.get(id) - if (!t.result) return { task_id: t.id, status: t.status, terminal: settled(t.status), message: "No result yet; the task is still active. Poll task_status." } - return { ...t.result, status: t.status, handoff: this.handoffOf(t) } + const v = this.checkView(view) + if (!t.result || !settled(t.status)) return { task_id: t.id, status: t.status, terminal: false, phase: t.phase, message: "No result yet; the task is still active. Call wait_task." } + const h = this.handoffOf(t) + if (v === "brief") return briefResult({ ...t.result, status: t.status, approval_request: t.approval_request ? { at: t.approval_request.at, by: t.approval_request.by, waiting_for: "operator" } : undefined }, h) + return { ...t.result, status: t.status, handoff: dedupeHandoff(h), ...(t.approval_request ? { approval_request: t.approval_request } : {}) } + } + + // Compact progress record for wait_task. + progress(t) { + const live = this.live.get(t.id) + return { + task_id: t.id, status: t.status, phase: t.phase, runs: t.runs.length, + elapsed_s: t.started_at ? Math.round((Date.now() - Date.parse(t.started_at)) / 1000) : 0, + idle_s: live ? Math.round((Date.now() - live.lastActivity) / 1000) : null, turns: t.stats.turns, tool_calls: t.stats.tool_calls, + } + } + + // Long-poll until the task(s) settle (finished or parked) or max_wait_s passes (cap WAIT_CAP_S). + // Replaces task_status polling + a separate task_result call: settled tasks come back with their + // result in the requested view (default brief). + async wait(p = {}) { + const allowed = new Set(["task_id", "task_ids", "mode", "max_wait_s", "view"]) + for (const k of Object.keys(p)) if (!allowed.has(k)) throw new UserError(`unknown parameter '${k}'`) + if ((p.task_id === undefined) === (p.task_ids === undefined)) throw new UserError("pass task_id or task_ids (not both)") + const multi = p.task_ids !== undefined + const ids = multi ? p.task_ids : [p.task_id] + if (!Array.isArray(ids) || ids.length < 1 || ids.length > WAIT_MAX_IDS) throw new UserError(`task_ids must be an array of 1..${WAIT_MAX_IDS} task ids`) + const uniq = [...new Set(ids)] + for (const id of uniq) this.get(id) + const mode = p.mode ?? "any" + if (!["any", "all"].includes(mode)) throw new UserError("mode must be 'any' or 'all'") + const view = p.view ?? "brief" + if (!["brief", "full", "status"].includes(view)) throw new UserError("view must be brief, full or status") + let maxWait = p.max_wait_s ?? WAIT_DEFAULT_S + if (typeof maxWait !== "number" || !Number.isFinite(maxWait) || maxWait < 0) throw new UserError(`max_wait_s must be a number 0..${WAIT_CAP_S}`) + maxWait = Math.min(maxWait, WAIT_CAP_S) + const t0 = Date.now() + const isDone = () => { + const n = uniq.filter((id) => { const t = this.tasks.get(id); return !t || settled(t.status) }).length + return mode === "all" ? n === uniq.length : n > 0 + } + while (!isDone() && !this.shuttingDown && Date.now() - t0 < maxWait * 1000) await sleep(Math.min(500, maxWait * 1000 - (Date.now() - t0))) + const done = isDone() + const one = (id) => { + const t = this.tasks.get(id) + if (!t) return { task_id: id, status: "deleted" } + if (!settled(t.status)) return this.progress(t) + if (view === "status") return this.status(id) + return this.result(id, view) + } + const waited = Math.round((Date.now() - t0) / 100) / 10 + const hint = done ? undefined : "Not finished yet; call wait_task again." + if (!multi) return { done, waited_s: waited, task: one(uniq[0]), ...(hint ? { hint } : {}) } + const tasks = uniq.map(one) + const n = uniq.filter((id) => settled(this.tasks.get(id)?.status)).length + return { done, mode, waited_s: waited, settled: n, pending: uniq.length - n, tasks, ...(hint ? { hint } : {}) } + } + + // Characters the supervisor sent (request) and read (response) per task, for usage_report. + recordSupervisorIo(params, result, reqChars, resChars) { + const ids = new Set() + const add = (x) => { if (typeof x === "string" && this.tasks.has(x)) ids.add(x) } + add(params?.task_id) + if (Array.isArray(params?.task_ids)) params.task_ids.forEach(add) + add(result?.task_id) + add(result?.task?.task_id) + for (const r of [...(Array.isArray(result?.tasks) ? result.tasks : []), ...(Array.isArray(result?.results) ? result.results : [])]) add(r?.task_id) + if (!ids.size) { + this.unattributedIo.calls++ + this.unattributedIo.request_chars += reqChars + this.unattributedIo.response_chars += resChars + return + } + for (const id of ids) { + const t = this.tasks.get(id) + const io = (t.supervisor_io ||= { calls: 0, request_chars: 0, response_chars: 0 }) + io.calls++ + io.request_chars += Math.round(reqChars / ids.size) + io.response_chars += Math.round(resChars / ids.size) + this.save(t) + } + } + + usage(p = {}) { + const allowed = new Set(["days", "profile", "repo"]) + for (const k of Object.keys(p)) if (!allowed.has(k)) throw new UserError(`unknown parameter '${k}'`) + if (p.days !== undefined && (typeof p.days !== "number" || !(p.days > 0))) throw new UserError("days must be a positive number") + const pc = profilesConfig() + const cfg = daemonConfig() + return usageReport([...this.tasks.values()], { days: p.days, profile: p.profile || null, repo: p.repo || null, profiles: pc.profiles, supervisor: cfg.supervisor || {}, unattributed: this.unattributedIo }) } // The task's handoff record: the persisted one, or (tasks finished before v0.2.0) derived on the fly. @@ -449,10 +641,13 @@ export class Manager { } } - async continueTask(id, instructions, timeoutMinutes, profileName, { approval = null } = {}) { + async continueTask(id, instructions, timeoutMinutes, profileName, { approval = null, operatorToken, operatorOk = false } = {}) { const t = this.get(id) if (!settled(t.status)) throw new UserError(`task is ${t.status}; continue_task works only on finished or parked tasks (cancel it first if needed)`) this.assertNotCleaning(t) + const cfg = daemonConfig() + if (PARKED.has(t.status) && requireOperator(cfg) && !operatorOk && !checkOperatorToken(cfg, operatorToken)) + throw new UserError("task is parked for approval and approvals.require_operator is on: use approve_task (records the request; a human confirms with `sudo workhorse approve `), or approve_task decision=reject") if (t.worktree_removed) throw new UserError("task worktree was cleaned up; delegate a new task instead") if (typeof instructions !== "string" || !instructions.trim()) throw new UserError("instructions are required") if (instructions.length > daemonConfig().limits.task_max_chars) throw new UserError("instructions too long") @@ -460,8 +655,10 @@ export class Manager { this.checkBackend(profile) const timeout = this.timeoutMin(timeoutMinutes) const h = this.handoffOf(t) - if (t.result) t.previous_results.push({ at: now(), status: t.status, verdict: t.result.verdict, summary: t.result.summary, handoff: h ? { state: h.state, owner: h.owner, next_action: h.next_action } : undefined, approval: approval || undefined }) + if (t.result) t.previous_results.push({ at: now(), status: t.status, verdict: t.result.verdict, summary: t.result.summary, handoff: h ? { state: h.state, owner: h.owner, next_action: h.next_action } : undefined, approval: approval || undefined, ...(t.result.auto ? { auto: t.result.auto } : {}), ...(t.result.review ? { review: { verdict: t.result.review.verdict, task_id: t.result.review.task_id } } : {}) }) if (approval) t.approvals = [...(t.approvals || []), approval].slice(-20) + t.auto_trail = [] + delete t.approval_request t.result = null t.handoff = null delete t.parked_from @@ -496,6 +693,13 @@ export class Manager { return { task_id: id, status: "cancelled" } } if (PARKED.has(t.status)) return this.closeTask(t, "supervisor", "cancelled via cancel_task") + if (t.status === "reviewing") { + // Stop only the advisory review; the task finishes with its own result (review: unavailable). + const child = this.tasks.get(t.review_pending?.task_id) + if (child && !settled(child.status)) await this.cancel(child.id) + else await this.finishReview(t, null, "review cancelled") + return { task_id: id, status: this.tasks.get(id)?.status, message: "automatic review cancelled; the task keeps its own result" } + } throw new UserError(`task is ${t.status}; nothing to cancel`) } @@ -531,11 +735,14 @@ export class Manager { // Supervisor/human edits of the handoff. Setting state needs_approval/needs_input parks a finished task; // setting any other state on a parked task unparks it (back to the status it had, else completed). updateHandoff(p = {}, caller = null) { - const allowed = new Set(["task_id", "owner", "next_action", "note", "state", "by"]) + const allowed = new Set(["task_id", "owner", "next_action", "note", "state", "by", "operator_token"]) for (const k of Object.keys(p)) if (!allowed.has(k)) throw new UserError(`unknown parameter '${k}'`) const t = this.get(p.task_id) if (!settled(t.status)) throw new UserError(`task is ${t.status}; the handoff exists once the task has finished`) this.assertNotCleaning(t) + const cfg = daemonConfig() + if (PARKED.has(t.status) && p.state !== undefined && !PARK_STATES.has(p.state) && p.state !== "closed" && requireOperator(cfg) && !checkOperatorToken(cfg, p.operator_token)) + throw new UserError("task is parked for approval and approvals.require_operator is on: only the operator can unpark it (`sudo workhorse approve `); state=closed is allowed") if (p.state !== undefined && !HANDOFF_STATES.includes(p.state)) throw new UserError(`state must be one of ${HANDOFF_STATES.join(", ")}`) if (p.owner !== undefined && (typeof p.owner !== "string" || !OWNER_RE.test(p.owner))) throw new UserError("owner must be 'supervisor', 'human' or a short name (letters, digits, space, _.:@/+-; max 80 chars)") const nextAction = this.checkText("next_action", p.next_action) @@ -591,7 +798,7 @@ export class Manager { // follow-up message. reject: with instructions, resume and tell the worker not to do it; without, // close the task (status cancelled, handoff closed; the worktree is kept). async approve(p = {}, caller = null) { - const allowed = new Set(["task_id", "decision", "instructions", "note", "by", "timeout_minutes", "profile"]) + const allowed = new Set(["task_id", "decision", "instructions", "note", "by", "timeout_minutes", "profile", "operator_token"]) for (const k of Object.keys(p)) if (!allowed.has(k)) throw new UserError(`unknown parameter '${k}'`) const t = this.get(p.task_id) if (p.decision !== "approve" && p.decision !== "reject") throw new UserError("decision must be 'approve' or 'reject'") @@ -599,9 +806,24 @@ export class Manager { this.assertNotCleaning(t) const h = this.handoffOf(t) if (!PARKED.has(t.status) && !APPROVABLE_STATES.has(h?.state)) throw new UserError(`task is not waiting for approval (status ${t.status}, handoff state ${h?.state || "none"}); use continue_task or update_handoff`) - const instructions = this.checkText("instructions", p.instructions, daemonConfig().limits.task_max_chars) + const cfg = daemonConfig() + let instructions = this.checkText("instructions", p.instructions, cfg.limits.task_max_chars) const note = this.checkText("note", p.note) - const who = this.actor(p.by, caller) + let operatorOk = false + if (requireOperator(cfg) && p.decision === "approve") { + operatorOk = checkOperatorToken(cfg, p.operator_token) + if (!operatorOk && p.operator_token !== undefined) { + this.event(t, "operator_token_rejected", {}) + throw new UserError("operator token rejected (check approvals.operator_token_sha256 and the token file)") + } + if (!operatorOk) return this.recordApprovalRequest(t, h, { who: this.actor(p.by, caller), instructions, note, profile: p.profile, timeout_minutes: p.timeout_minutes }) + const req = t.approval_request + if (req) { + instructions ??= req.instructions + p = { ...p, profile: p.profile ?? req.profile, timeout_minutes: p.timeout_minutes ?? req.timeout_minutes } + } + } + const who = operatorOk && (p.by === undefined || p.by === null || p.by === "") ? `operator${t.approval_request ? ` (requested by ${t.approval_request.by})` : ""}` : this.actor(p.by, caller) const source = this.sourceOf(caller) const request = h?.context?.worker_request || h?.next_action || "the pending request" const rec = { at: now(), by: who, source, decision: p.decision, request: head(request, 300), ...(note ? { note } : {}), ...(instructions ? { instructions: head(instructions, 300) } : {}) } @@ -613,11 +835,28 @@ export class Manager { const msg = p.decision === "approve" ? `APPROVED by ${who}: ${request}${note ? `\nNote: ${note}` : ""}\n\n${instructions || "Proceed with the approved action and finish the task."}\n\n(Approval does not change the sandbox: actions it blocks stay blocked. If the approved action needs network or an install, the operator has done it outside, or you must report that under concerns.)` : `DENIED by ${who}: ${request}. Do not do that.${note ? `\nNote: ${note}` : ""}\n\n${instructions}` - const r = await this.continueTask(t.id, msg, p.timeout_minutes, p.profile, { approval: rec }) + const r = await this.continueTask(t.id, msg, p.timeout_minutes, p.profile, { approval: rec, operatorOk: operatorOk || p.decision === "reject" }) logApproval() return { ...r, decision: p.decision, approval: rec } } + // approvals.require_operator: the supervisor's approve only records the request; the operator confirms. + recordApprovalRequest(t, h, { who, instructions, note, profile, timeout_minutes }) { + const at = now() + t.approval_request = redact({ at, by: who, ...(instructions ? { instructions } : {}), ...(note ? { note } : {}), ...(profile ? { profile } : {}), ...(timeout_minutes ? { timeout_minutes } : {}) }) + const nh = { ...h } + nh.owner = "human (operator)" + nh.next_action = `Approval requested by ${who}; waiting for the operator. Operator: review the request, then run \`sudo workhorse approve ${t.id}\` (uses the operator token) or \`workhorse reject ${t.id}\`.` + nh.history = [...(h?.history || []), { at, by: who, event: "approval_requested" }].slice(-20) + nh.derived = false + nh.updated_at = at + nh.updated_by = who + t.handoff = redact(nh) + this.save(t) + this.event(t, "approval_requested", { by: who }) + return { task_id: t.id, status: t.status, approval_requested: true, waiting_for: "operator", message: "Approval recorded as a request. approvals.require_operator is on: a human must confirm it on the host with `sudo workhorse approve `. Nothing runs until then.", handoff: handoffSummary(this.handoffOf(t)) } + } + closeTask(t, who, reason, rec = null, source = null) { const from = t.status const h = { ...this.handoffOf(t) } @@ -861,10 +1100,15 @@ export class Manager { const bp = backendProblems({ name, ...p }, pc, cfg) return { name, description: p.description, backend: profileBackend(p, cfg), model: p.model, model_id: modelId, provider: p.provider, fallback: p.fallback, + ...(p.escalate_to ? { escalate_to: p.escalate_to } : {}), ...(p.stall_minutes ? { stall_minutes: p.stall_minutes } : {}), available: p.enabled !== false && missing.length === 0 && bp.length === 0, unavailable_reason: p.enabled === false ? p.disabled_reason || "disabled" : bp.length ? bp.join("; ") : missing.length ? this.missingCredsMessage(missing) : null, } }), + presets: Object.entries(pc.presets).map(([name, x]) => ({ name, description: x.description || "", profile: x.profile || null, size: x.size || null, mode: x.mode || undefined })), + routing: pc.routing, + auto_defaults: { fix_rounds: pc.auto.fix_rounds, escalate: pc.auto.escalate, auto_review: pc.auto.review.enabled ? pc.auto.review.profile || pc.default_profile : false, max_auto_runs: pc.auto.max_auto_runs, max_tokens: pc.auto.max_tokens, max_cost_usd: pc.auto.max_cost_usd }, + token_savers: saverSummary(effectiveSavers(cfg)), modes: { implement: "worker agent edits code in a fresh worktree", review: "read-only review agent; pass review_task_id to review another task's diff, or base_ref to review a branch" }, backends: backendSummary(cfg).map((b) => ({ name: b.name, status: b.status, installed: b.installed, summary: b.summary })), limits: { max_concurrent: cfg.max_concurrent, timeout_minutes: { default: cfg.timeouts.default_min, min: cfg.timeouts.min_min, max: cfg.timeouts.max_min }, stall_minutes: cfg.timeouts.stall_min, provider_max_concurrent: Object.fromEntries(Object.entries(pc.providers).map(([k, v]) => [k, v.max_concurrent || 1])) }, @@ -937,6 +1181,12 @@ export class Manager { buildMessage(t, spec) { const short = t.base_commit.slice(0, 10) const testLine = t.test_command || this.repoSafe(t)?.test_command + if (spec.kind === "auto_fix") { + return `Automatic follow-up from the workhorse daemon for task ${t.id} (no human or supervisor involved):\n\n${spec.message}\n\nContinue in the same worktree (check \`git status\` / \`git diff\`).${testLine ? ` Test command: ${testLine}` : ""}\nEnd with an updated ## RESULT block.` + } + if (spec.kind === "escalate") { + return `${spec.message}\n\nContext:\n- Repository: ${t.repo}; dedicated git worktree on branch ${t.branch}, based on ${t.base_ref} @ ${short}. The previous attempt's changes are uncommitted in this worktree.\n- ${testLine ? `Test command: ${testLine} (run it before finishing).` : "Discover and run the project's tests if any."}\n- Follow the workhorse contract from your instructions and finish with the ## RESULT block.` + } if (spec.kind === "continue") { return `Follow-up instructions from the supervisor for task ${t.id}:\n\n${spec.message}\n\nContinue in the same worktree (check \`git status\` / \`git diff\` for what is already done).${testLine ? ` Test command: ${testLine}` : ""}\nEnd with an updated ## RESULT block.` } @@ -1033,11 +1283,18 @@ export class Manager { const backend = getBackend(profile.backend) const bc = cfg.backends[profile.backend] const sandboxed = cfg.worker_sandbox?.enabled !== false - const message = this.buildMessage(t, { ...spec, backend: profile.backend }) const agent = t.agent const { env: baseEnv, provSecrets } = this.baseWorkerEnv(profile, bc) - // Sessions are backend-specific: resume only a session created by the same backend. - const resumeSession = t.session_id && spec.kind !== "initial" && this.sessionBackend(t) === profile.backend ? t.session_id : null + // Sessions are backend-specific: resume only a session created by the same backend. Escalation starts + // a fresh session on purpose: the stronger model should not pay for the weaker one's whole history. + const resumeSession = t.session_id && spec.kind !== "initial" && spec.kind !== "escalate" && this.sessionBackend(t) === profile.backend ? t.session_id : null + // Token savers (opt-in): prompt fragments go into the first message of a fresh session only. + const savers = effectiveSavers(cfg, profile) + const message = this.buildMessage(t, { ...spec, backend: profile.backend }) + (resumeSession ? "" : saverFragments(savers, t.mode)) + // RTK is applied by the guard plugin, i.e. only on the Kilo/OpenCode backends. + const rtkCapable = profile.backend === "kilo" || profile.backend === "opencode" + const rtk = rtkCapable ? rtkBin(savers, cfg.env_path) : null + if (savers.rtk.enabled && rtkCapable && !rtk) this.activity(t, "token_savers.rtk is enabled but the rtk binary was not found (set token_savers.rtk.bin); running without it") const ctx = { t, profile, pc, cfg, bc, repo: this.repoSafe(t), agent, message, resumeSession, dataDir: this.backendDataDir(t.id), home: bc.home, sandboxed, secrets: provSecrets, @@ -1046,7 +1303,7 @@ export class Manager { if (bc.home) ensureDir(bc.home, 0o755) backend.prepare(ctx) const args = backend.command(ctx) - const env = { ...baseEnv, ...backend.env(ctx) } + const env = { ...baseEnv, ...backend.env(ctx), ...(rtk ? { WH_RTK_BIN: rtk } : {}) } const outFile = path.join(dir, `run-${n}.events.jsonl`) const errFile = path.join(dir, `run-${n}.stderr.log`) const outFd = fs.openSync(outFile, "a", 0o600) @@ -1056,7 +1313,9 @@ export class Manager { let bin = bc.bin let argv = args if (sandboxed) { - argv = [...(await this.workerSandboxArgs(t, cfg, backend.sandbox(ctx))), "--", bc.bin_real || bc.bin, ...args] + const extra = backend.sandbox(ctx) || {} + if (rtk) extra.roBinds = [...(extra.roBinds || []), path.dirname(fs.realpathSync(rtk))] + argv = [...(await this.workerSandboxArgs(t, cfg, extra)), "--", bc.bin_real || bc.bin, ...args] bin = cfg.bwrap } child = spawn(bin, argv, { cwd: t.worktree_path, env, stdio: ["ignore", outFd, errFd], detached: true }) @@ -1064,7 +1323,12 @@ export class Manager { fs.closeSync(outFd) fs.closeSync(errFd) } - const run = { n, kind: spec.kind, profile: profile.name, backend: profile.backend, model: profile.model, agent, resumed_session: !!resumeSession, started_at: now(), finished_at: null, pid: child.pid, pid_start: null, exit_code: null, signal: null, reason: null, events: 0, errors: [], timeout_min: spec.timeout_min || t.timeout_min } + const stallMin = Number(profile.stall_minutes) > 0 ? Number(profile.stall_minutes) : cfg.timeouts.stall_min + const run = { + n, kind: spec.kind, profile: profile.name, backend: profile.backend, model: profile.model, agent, resumed_session: !!resumeSession, started_at: now(), finished_at: null, pid: child.pid, pid_start: null, exit_code: null, signal: null, reason: null, events: 0, errors: [], + timeout_min: spec.timeout_min || t.timeout_min, stall_min: stallMin, tokens: { input: 0, output: 0, reasoning: 0, cache_read: 0, cache_write: 0 }, cost: 0, + token_savers: { ...saverSummary(savers), rtk: !!rtk }, + } run.pid_start = procStart(child.pid) t.runs.push(run) t.status = "running" @@ -1105,9 +1369,10 @@ export class Manager { await killTree(live.child.pid, cfg.timeouts.kill_grace_sec * 1000) return } - if ((Date.now() - live.lastActivity) / 60000 > cfg.timeouts.stall_min) { + const stallMin = live.run.stall_min || cfg.timeouts.stall_min + if ((Date.now() - live.lastActivity) / 60000 > stallMin) { live.killedFor = "stalled" - this.activity(t, `no activity for ${cfg.timeouts.stall_min} min; stopping the worker`) + this.activity(t, `no activity for ${stallMin} min; stopping the worker`) await killTree(live.child.pid, cfg.timeouts.kill_grace_sec * 1000) return } @@ -1162,8 +1427,13 @@ export class Manager { case "step": { s.turns += ev.turns ?? 1 const tk = ev.tokens || {} - for (const k of ["input", "output", "reasoning", "cache_read", "cache_write"]) s.tokens[k] += tk[k] || 0 + run.tokens ||= { input: 0, output: 0, reasoning: 0, cache_read: 0, cache_write: 0 } + for (const k of ["input", "output", "reasoning", "cache_read", "cache_write"]) { + s.tokens[k] += tk[k] || 0 + run.tokens[k] += tk[k] || 0 + } s.reported_cost = (s.reported_cost ?? s.kilo_cost ?? 0) + (ev.cost || 0) + run.cost = (run.cost || 0) + (ev.cost || 0) break } case "tool": { @@ -1228,7 +1498,7 @@ export class Manager { const cfg = daemonConfig() if (live.killedFor === "cancel") return this.finalize(t, { tests: false, status: "cancelled" }) if (live.killedFor === "timeout" || live.killedFor === "stalled") { - t.errors.push(live.killedFor === "timeout" ? `wall-clock timeout of ${run.timeout_min} min reached` : `no activity for ${cfg.timeouts.stall_min} min (stalled)`) + t.errors.push(live.killedFor === "timeout" ? `wall-clock timeout of ${run.timeout_min} min reached` : `no activity for ${run.stall_min || cfg.timeouts.stall_min} min (stalled)`) return this.finalize(t, { tests: true, status: live.killedFor }) } if (code === 0) return this.finalize(t, { tests: true, status: "completed" }) @@ -1408,8 +1678,7 @@ export class Manager { return {} } })() - const price = pc.price_per_mtok || null - const est = price ? ((t.stats.tokens.input + t.stats.tokens.cache_read) * (price.input || 0) + (t.stats.tokens.output + t.stats.tokens.reasoning) * (price.output || 0)) / 1e6 : null + const est = this.estCost(t) // Park the task for a human when the worker asks for approval/input, or reports blocked with a concrete // request or after policy/sandbox-blocked tool calls (an action it considers necessary). let finalStatus = status @@ -1443,17 +1712,216 @@ export class Manager { usage: { tokens: t.stats.tokens, backend_reported_cost_usd: Math.round((t.stats.reported_cost ?? t.stats.kilo_cost ?? 0) * 1e6) / 1e6, estimated_list_cost_usd: est === null ? null : Math.round(est * 1e6) / 1e6, pricing_note: pc.price_note || null }, activity: { turns: t.stats.turns, tool_calls: t.stats.tool_calls, tool_errors: t.stats.tool_errors, failed_commands: t.stats.failed_commands, blocked_calls: t.stats.blocked_calls, by_tool: t.stats.by_tool, runs: t.runs.length, retries: t.retries, fallback_used: t.fallback_used, fallbacks_tried: t.fallbacks_tried || [] }, blocked_examples: t.blocked.slice(0, 5), + ...(last?.token_savers && (last.token_savers.terse !== "off" || last.token_savers.minimal_code !== "off" || last.token_savers.rtk) ? { token_savers: last.token_savers } : {}), + ...(t.preset ? { preset: t.preset } : {}), details_hint: "task_details kinds: activity, diff, test_log, final_message, raw_events, stderr", }) t.errors = [] delete t.parked_from + // Automatic follow-ups (auto_fix_rounds / escalate chain) and the advisory auto-review. + const plan = this.planAuto(t) + if (plan.follow) return this.queueAuto(t, plan.follow) + if (t.auto_trail?.length || plan.stopped) { + t.auto_trail = [...(t.auto_trail || []), this.trailEntry(t, null)] + t.result.auto = { trail: t.auto_trail, runs: t.auto_trail.filter((e) => e.next).length, stopped_reason: plan.stopped || null } + } + if (plan.review) return this.startAutoReview(t, plan.review) t.handoff = redact(deriveHandoff(t, { at: finishedAt })) this.save(t) this.event(t, "finished", { verdict, files: diff.files.length, tests_passed: testRes.passed ?? null, handoff_state: t.handoff.state, owner: t.handoff.owner }) if (PARKED.has(t.status)) this.event(t, "parked", { handoff_state: t.handoff.state, request: t.handoff.context?.worker_request || null }) + if (t.auto_review_of) { + const parent = this.tasks.get(t.auto_review_of) + if (parent?.status === "reviewing" && (!parent.review_pending?.task_id || parent.review_pending.task_id === t.id)) await this.finishReview(parent, t) + } + this.schedule().catch(() => {}) + } + + // Estimated list cost: each run's tokens at its own profile's price_per_mtok (runs before v0.3.0 have + // no per-run tokens: the task total at the last run's profile price). null when nothing is priced. + estCost(t) { + const price = (name) => { + try { + return this.profile(name).price_per_mtok || null + } catch { + return null + } + } + const cost = (tk, pr) => ((tk.input + tk.cache_read) * (pr.input || 0) + (tk.output + tk.reasoning) * (pr.output || 0)) / 1e6 + const runs = t.runs.filter((r) => r.tokens) + if (!runs.length) { + const pr = price(t.runs[t.runs.length - 1]?.profile || t.profile) + return pr ? cost(t.stats.tokens, pr) : null + } + let total = null + for (const r of runs) { + const pr = price(r.profile) + if (pr) total = (total || 0) + cost(r.tokens, pr) + } + return total + } + + // ---------- automatic follow-ups ---------- + // What went wrong in a finished implement run, as an auto trigger name (null: nothing to fix). + autoTrigger(t) { + const r = t.result || {} + const v = r.verdict + if (v === "tests_failed") return "tests_failed" + if (v === "no_changes") return "no_changes" + if (v === "worker_error") return (r.errors || []).some((e) => SETUP_ERROR_RE.test(e)) ? null : "worker_error" + if (SUCCESS.has(v)) { + if (!r.worker_reported) return "no_result_block" + if (r.worker_reported.status === "partial") return "partial" + } + return null + } + + // Decide the next automatic step after a run: {follow: {kind, profile, trigger}} | {review: profile} | {stopped} | {}. + planAuto(t) { + const a = t.auto + if (!a || t.mode !== "implement" || !["completed", "failed"].includes(t.status)) return {} + const trail = t.auto_trail || [] + const tk = t.stats.tokens + const used = tk.input + tk.output + tk.reasoning + const cost = this.estCost(t) + const autoRuns = trail.filter((e) => e.next).length + const budget = a.max_tokens && used >= a.max_tokens ? "max_tokens" : a.max_cost_usd && cost !== null && cost >= a.max_cost_usd ? "max_cost_usd" : null + const trigger = this.autoTrigger(t) + const cur = t.runs[t.runs.length - 1]?.profile || t.profile + if (trigger) { + const wantFix = a.fix_on.includes(trigger) && trail.filter((e) => e.next === "auto_fix" && e.profile === cur).length < a.fix_rounds + const wantEsc = a.escalate && a.escalate_on.includes(trigger) + if (!wantFix && !wantEsc) return a.fix_rounds && a.fix_on.includes(trigger) ? { stopped: "fix_rounds_exhausted" } : {} + if (autoRuns >= Math.min(a.max_auto_runs, AUTO_HARD.max_auto_runs)) return { stopped: "max_auto_runs" } + if (budget) return { stopped: budget } + if (wantFix) return { follow: { kind: "auto_fix", profile: cur, trigger } } + const next = this.escalationTarget(t, cur) + return next ? { follow: { kind: "escalate", profile: next, from: cur, trigger } } : { stopped: "no_escalation_target" } + } + if (a.review_profile && a.review_on.includes(t.result.verdict) && t.status === "completed") return budget ? { stopped: `${budget} (auto-review skipped)` } : { review: a.review_profile } + return {} + } + + // Next runnable profile up the escalate_to chain of `cur`, never one this task already used. + escalationTarget(t, cur) { + const used = new Set([t.profile, ...t.runs.map((r) => r.profile)]) + let name = cur + for (let i = 0; i < 6; i++) { + let p + try { + p = this.profile(name) + } catch { + return null + } + const next = p.escalate_to + if (typeof next !== "string" || !next || used.has(next)) return null + try { + const np = this.profile(next) + if (!backendProblems(np, profilesConfig(), daemonConfig()).length && !this.missingCreds(np).length) return next + } catch {} + used.add(next) + name = next + } + return null + } + + trailEntry(t, f) { + const trail = t.auto_trail || [] + const since = trail.length ? trail[trail.length - 1].run : 0 + let tokens = 0 + for (const r of t.runs) if (r.n > since && r.tokens) tokens += r.tokens.input + r.tokens.output + r.tokens.reasoning + const last = t.runs[t.runs.length - 1] + return { run: last?.n || 0, profile: last?.profile || t.profile, verdict: t.result?.verdict || null, tokens, ...(f ? { trigger: f.trigger, next: f.kind, next_profile: f.profile } : {}) } + } + + autoMessage(t, h, f) { + const r = t.result + const tests = (h.failed_checks || []).find((c) => c.check === "tests") + const testPart = tests ? `\nFailing test output (tail):\n${head(tests.excerpt || "", 900)}` : "" + if (f.kind === "auto_fix") { + const base = f.trigger === "no_result_block" + ? "You stopped without the ## RESULT block. Finish anything that is left, run the tests and end with the ## RESULT block." + : f.trigger === "partial" + ? "You reported status partial. Finish the remaining work, rerun the tests and end with the ## RESULT block." + : h.resume?.args?.instructions || "The previous run did not finish the task. Check `git status` / `git diff`, finish it and end with the ## RESULT block." + return `${base}${testPart}` + } + const why = f.trigger === "tests_failed" ? `the daemon-run tests failed (\`${r.test_results?.command}\` exit ${r.test_results?.exit_code}${tests?.failing_tests?.length ? `; failing: ${tests.failing_tests.slice(0, 5).join(", ")}` : ""})` + : f.trigger === "worker_error" ? `the worker failed: ${head((r.errors || []).slice(-1)[0] || "error", 300)}` + : f.trigger === "no_changes" ? "it made no changes" + : f.trigger === "partial" ? "it reported the work as partial" + : "it did not finish with a result" + return `ESCALATION for task ${t.id}: a previous attempt by a smaller model (profile ${f.from}) did not finish the task: ${why}.\n\nOriginal task:\n${t.task}\n\nThe previous attempt reported: ${head(r.summary || "(nothing)", 600)}${testPart}\n\nIts changes are uncommitted in this worktree: check \`git status\` / \`git diff\`, keep what is correct, fix or replace the rest. Do not weaken or delete tests.` + } + + async queueAuto(t, f) { + const r = t.result + const h = deriveHandoff(t, { at: now() }) + t.auto_trail = [...(t.auto_trail || []), this.trailEntry(t, f)] + t.previous_results.push({ at: now(), status: t.status, verdict: r.verdict, summary: r.summary, handoff: { state: h.state, owner: h.owner, next_action: h.next_action }, auto: f.kind, trigger: f.trigger }) + const message = redact(this.autoMessage(t, h, f)) + t.result = null + t.handoff = null + t.status = "queued" + t.phase = f.kind === "auto_fix" ? `automatic fix round (${f.trigger}) with ${f.profile}` : `escalating to profile ${f.profile} (${f.trigger})` + t.finished_at = null + t.retries = 0 + t.fallback_used = false + t.fallbacks_tried = [] + t.pending_profile_origin = f.profile + t.pending_run = { kind: f.kind, message, profile: f.profile, timeout_min: t.timeout_min } + this.save(t) + this.event(t, "auto_followup", { auto_kind: f.kind, trigger: f.trigger, profile: f.profile, ...(f.from ? { from_profile: f.from } : {}) }) + this.activity(t, `automatic ${f.kind} (${f.trigger}) queued with profile ${f.profile}`) this.schedule().catch(() => {}) } + // ---------- automatic advisory review ---------- + async startAutoReview(t, profileName) { + const cfg = daemonConfig() + t.review_pending = { profile: profileName, final_status: t.status, started_at: now() } + t.status = "reviewing" + t.phase = `automatic advisory review by profile ${profileName}` + this.save(t) + const timeout = Math.min(cfg.timeouts.max_min, Math.max(cfg.timeouts.min_min, Number(t.auto?.review_timeout_min) || 10)) + const task = `Automatic advisory review of task ${t.id} (cheap first-pass reviewer; the supervisor decides). The task was:\n\n${head(t.task, 4000)}\n\nCheck the diff for correctness bugs, missing tests for new behaviour and changes outside the task's scope. Be brief: at most 5 findings, most severe first. Start summary with "approve" or "request changes".` + try { + const r = await this.delegate({ repo: t.repo, mode: "review", review_task_id: t.id, profile: profileName, task, timeout_minutes: timeout }, { autoReviewOf: t.id }) + if (t.status === "reviewing" && t.review_pending) { + t.review_pending.task_id = r.task_id + this.save(t) + } + this.event(t, "auto_review_started", { review_task_id: r.task_id, profile: profileName }) + } catch (e) { + await this.finishReview(t, null, `could not start the review: ${head(e.message, 300)}`) + } + } + + async finishReview(t, child, reason = null) { + if (t.status !== "reviewing") return + const pend = t.review_pending || {} + const cr = child?.result + let review + if (!cr || child.status !== "completed") { + review = { advisory: true, verdict: "unavailable", profile: pend.profile || null, task_id: child?.id || pend.task_id || null, reason: reason || (child ? `review task ended ${child.status}` : "no review task") } + } else { + const tk = cr.usage?.tokens || {} + review = { + advisory: true, verdict: reviewVerdict(cr.summary), profile: child.profile, task_id: child.id, summary: head(cr.summary || "", 300), + findings: (cr.remaining_concerns || []).filter((c) => !DAEMON_CONCERN_RE.test(c)).slice(0, 5).map((c) => head(c, 300)), + tokens: (tk.input || 0) + (tk.output || 0) + (tk.reasoning || 0), est_cost_usd: cr.usage?.estimated_list_cost_usd ?? null, + } + } + t.result = redact({ ...t.result, review }) + t.status = pend.final_status || "completed" + delete t.review_pending + t.phase = "finished" + t.handoff = redact(deriveHandoff(t, { at: now() })) + this.save(t) + this.event(t, "auto_review_done", { review_verdict: review.verdict, review_task_id: review.task_id }) + this.event(t, "finished", { verdict: t.result.verdict, files: (t.result.files_changed || []).length, tests_passed: t.result.test_results?.passed ?? null, handoff_state: t.handoff.state, owner: t.handoff.owner }) + } + async shutdown() { this.shuttingDown = true clearInterval(this.timer) @@ -1476,6 +1944,16 @@ export class Manager { } } +const DAEMON_CONCERN_RE = /^(daemon-run tests failed|INTEGRITY|worker reported status|\d+ tool call\(s\) were blocked|review agent modified files|worker did not end with|fallback profile was used)/ + +// Advisory review verdict from the reviewer's summary. +export function reviewVerdict(summary) { + const s = String(summary || "").toLowerCase() + if (/request(?:s|ed|ing)?[\s_-]+changes|changes[\s_-]+(?:requested|required|needed)|\breject/.test(s)) return "request_changes" + if (/\bapprove[sd]?\b|\blgtm\b/.test(s)) return "approve" + return "unclear" +} + export function detectTestCommand(wt) { const has = (f) => fs.existsSync(path.join(wt, f)) if (has("package.json")) { diff --git a/lib/usage.mjs b/lib/usage.mjs new file mode 100644 index 0000000..c657126 --- /dev/null +++ b/lib/usage.mjs @@ -0,0 +1,140 @@ +// usage_report / `workhorse stats`: worker token use and cost by profile and day, plus an ESTIMATE of +// the supervisor (e.g. Grok) tokens the delegation avoided. See docs/token-savings.md#usage-report. +// +// ESTIMATE, per task whose final verdict is success or success_untested (work the supervisor did not +// have to redo): +// worker_work_tokens = worker input + output + reasoning tokens (cache reads excluded) +// supervisor_overhead = (chars the supervisor sent to workhorse for this task +// + chars workhorse returned to it) / chars_per_token +// supervisor_tokens_avoided = max(0, worker_work_tokens - supervisor_overhead) +// The assumption is that the supervisor would have needed about as many tokens as the worker to do the +// same work itself; a stronger model may need fewer turns (so this over-estimates) and delegation also +// saves the supervisor's own context growth (not counted, so it under-estimates). Tasks that did not +// succeed count only as overhead. With daemon.json supervisor.price_per_mtok {input, output} the +// estimate is also converted to USD (net of the workers' estimated list cost). +const KEYS = ["input", "output", "reasoning", "cache_read", "cache_write"] +const zero = () => Object.fromEntries(KEYS.map((k) => [k, 0])) +const addTo = (a, b) => { + for (const k of KEYS) a[k] += Number(b?.[k]) || 0 + return a +} +const round = (x, n = 6) => (x === null || x === undefined ? null : Math.round(x * 10 ** n) / 10 ** n) +export const SUCCESS = new Set(["success", "success_untested"]) + +// Local calendar day of an ISO timestamp. +export function dayOf(ts) { + const d = new Date(ts) + if (Number.isNaN(d.getTime())) return "unknown" + return `${d.getFullYear()}-${String(d.getMonth() + 1).padStart(2, "0")}-${String(d.getDate()).padStart(2, "0")}` +} + +function listCost(tokens, price) { + if (!price) return null + return ((tokens.input + tokens.cache_read) * (price.input || 0) + (tokens.output + tokens.reasoning) * (price.output || 0)) / 1e6 +} + +// Per-run token records of a task. Runs recorded before v0.3.0 have no per-run tokens: the task total is +// attributed to its last run. +export function runUsage(t) { + const runs = t.runs || [] + if (runs.some((r) => r.tokens)) return runs.map((r) => ({ profile: r.profile || t.profile, started_at: r.started_at || t.created_at, tokens: addTo(zero(), r.tokens), cost: Number(r.cost) || 0 })) + const last = runs[runs.length - 1] + return [{ profile: last?.profile || t.profile, started_at: last?.started_at || t.created_at, tokens: addTo(zero(), t.stats?.tokens), cost: Number(t.stats?.reported_cost ?? t.stats?.kilo_cost ?? 0) || 0, runs: runs.length }] +} + +export function usageReport(tasks, { days = 30, profile = null, repo = null, profiles = {}, supervisor = {}, unattributed = null, nowMs = Date.now() } = {}) { + const d = Math.max(1, Math.min(3650, Number(days) || 30)) + const cutoff = nowMs - d * 86400000 + const cpt = Number(supervisor.chars_per_token) > 0 ? Number(supervisor.chars_per_token) : 4 + const sp = supervisor.price_per_mtok || null + const byKey = new Map() + const byProfile = new Map() + const sup = { calls: 0, request_chars: 0, response_chars: 0 } + let workTokens = 0 + let workIn = 0 + let workOut = 0 + let succeeded = 0 + let overheadTokens = 0 + let overheadIn = 0 + let overheadOut = 0 + let workerCost = 0 + let tasksCounted = 0 + const prof = (name) => { + if (!byProfile.has(name)) byProfile.set(name, { profile: name, tasks: 0, runs: 0, tokens: zero(), backend_reported_cost_usd: 0, est_list_cost_usd: 0, priced: !!profiles[name]?.price_per_mtok, verdicts: {} }) + return byProfile.get(name) + } + for (const t of tasks) { + if (Date.parse(t.created_at) < cutoff) continue + if (repo && t.repo !== repo) continue + const runs = runUsage(t) + if (profile && !runs.some((r) => r.profile === profile) && t.profile !== profile) continue + tasksCounted++ + const seen = new Set() + const taskTok = zero() + for (const r of runs) { + if (profile && r.profile !== profile) continue + const day = dayOf(r.started_at) + const key = `${day}\u0000${r.profile}` + if (!byKey.has(key)) byKey.set(key, { day, profile: r.profile, tasks: 0, runs: 0, tokens: zero(), backend_reported_cost_usd: 0, est_list_cost_usd: 0 }) + const row = byKey.get(key) + const price = profiles[r.profile]?.price_per_mtok || null + const est = listCost(r.tokens, price) || 0 + row.runs += r.runs || 1 + addTo(row.tokens, r.tokens) + row.backend_reported_cost_usd += r.cost + row.est_list_cost_usd += est + if (!seen.has(key)) { + seen.add(key) + row.tasks++ + } + const p = prof(r.profile) + p.runs += r.runs || 1 + addTo(p.tokens, r.tokens) + p.backend_reported_cost_usd += r.cost + p.est_list_cost_usd += est + addTo(taskTok, r.tokens) + workerCost += est + } + const finalProfile = t.result?.profile || t.profile + const fp = prof(finalProfile) + fp.tasks++ + const v = t.result?.verdict || t.status + fp.verdicts[v] = (fp.verdicts[v] || 0) + 1 + const io = t.supervisor_io || {} + sup.calls += io.calls || 0 + sup.request_chars += io.request_chars || 0 + sup.response_chars += io.response_chars || 0 + const ovIn = (io.response_chars || 0) / cpt // what the supervisor read + const ovOut = (io.request_chars || 0) / cpt // what the supervisor wrote + overheadIn += ovIn + overheadOut += ovOut + overheadTokens += ovIn + ovOut + if (SUCCESS.has(t.result?.verdict)) { + succeeded++ + workIn += taskTok.input + workOut += taskTok.output + taskTok.reasoning + workTokens += taskTok.input + taskTok.output + taskTok.reasoning + } + } + const fin = (o) => ({ ...o, backend_reported_cost_usd: round(o.backend_reported_cost_usd), est_list_cost_usd: round(o.est_list_cost_usd) }) + const avoided = Math.max(0, Math.round(workTokens - overheadTokens)) + const avoidedUsd = sp ? round(((workIn - overheadIn) * (sp.input || 0) + (workOut - overheadOut) * (sp.output || 0)) / 1e6 - workerCost, 4) : null + return { + window_days: d, + tasks: tasksCounted, + by_profile: [...byProfile.values()].map(fin).sort((a, b) => b.tokens.input + b.tokens.output - (a.tokens.input + a.tokens.output)), + by_day: [...byKey.values()].map(fin).sort((a, b) => (a.day === b.day ? a.profile.localeCompare(b.profile) : a.day < b.day ? 1 : -1)), + supervisor_estimate: { + label: "ESTIMATE", + successful_tasks: succeeded, + worker_work_tokens: workTokens, + supervisor_io: { ...sup, est_tokens: Math.round(overheadTokens), chars_per_token: cpt }, + est_supervisor_tokens_avoided: avoided, + est_supervisor_cost_avoided_usd: avoidedUsd, + est_worker_list_cost_usd: round(workerCost, 4), + ...(unattributed ? { unattributed_supervisor_io_since_daemon_start: unattributed } : {}), + formula: "per successful task: worker (input+output+reasoning) tokens - (supervisor request chars + response chars)/chars_per_token; failed tasks count only as overhead; cache reads excluded" + (sp ? "; USD = avoided input*price.input + avoided output*price.output - worker est. list cost" : "; set daemon.json supervisor.price_per_mtok {input, output} for a USD estimate"), + caveat: "Assumes the supervisor would need about as many tokens as the worker for the same work. Not a measurement.", + }, + } +} diff --git a/lib/views.mjs b/lib/views.mjs new file mode 100644 index 0000000..a946ad1 --- /dev/null +++ b/lib/views.mjs @@ -0,0 +1,67 @@ +// Compact views of task results, to keep what the supervisor model reads (and pays for) small. +// +// full (default, backward compatible): the whole result; the handoff's `context` no longer repeats +// fields that are already top-level in the same response (summary, concerns, diffstat, ...). +// brief: only what the supervisor needs to decide the next step (~0.5-1 KB): verdict, a short +// summary, changed files, test outcome, top concerns, the handoff's next action + resume +// call, auto-review verdict, automatic follow-up trail and token totals. +import { head } from "./util.mjs" + +export const VIEWS = ["full", "brief"] + +// Handoff context keys that duplicate top-level task_result fields. +const DUP_CONTEXT = ["verdict", "status", "summary", "remaining_concerns", "diffstat", "files_changed", "diff_path", "diff"] + +export function dedupeHandoff(h) { + if (!h || !h.context) return h + const context = { ...h.context } + for (const k of DUP_CONTEXT) delete context[k] + for (const [k, v] of Object.entries(context)) if (v === null || v === undefined) delete context[k] + return { ...h, context } +} + +const tokensOf = (u) => { + const t = u?.tokens || {} + return { input: t.input || 0, output: t.output || 0, reasoning: t.reasoning || 0, cache_read: t.cache_read || 0 } +} + +// "path" for modified files, "A path" / "D path" / "R100 path" otherwise. +const fileLabel = (f) => (f && typeof f === "object" ? (f.status && f.status !== "M" ? `${f.status} ${f.path}` : f.path) : String(f)) + +export function briefResult(r, h, { maxFiles = 20, maxConcerns = 5, summaryChars = 300 } = {}) { + if (!r) return null + const files = Array.isArray(r.files_changed) ? r.files_changed : [] + const tr = r.test_results || {} + const tests = tr.executed + ? { passed: !!tr.passed, command: tr.command, counts: tr.counts && Object.keys(tr.counts).length ? tr.counts : undefined, ...(tr.passed ? {} : { failing: (tr.failing_tests || []).slice(0, 5), exit_code: tr.exit_code }) } + : { executed: false, reason: tr.reason || null } + const concerns = (r.remaining_concerns || []).slice(0, maxConcerns).map((c) => head(c, 200)) + const tk = tokensOf(r.usage) + const out = { + task_id: r.task_id, + status: r.status, + verdict: r.verdict, + mode: r.mode, + profile: r.profile, + summary: head(r.summary || "", summaryChars), + files: files.slice(0, maxFiles).map(fileLabel).concat(files.length > maxFiles ? [`… ${files.length - maxFiles} more`] : []), + diffstat: r.diffstat, + tests, + ...(concerns.length ? { concerns } : {}), + ...(r.errors?.length && !["success", "success_untested"].includes(r.verdict) ? { errors: r.errors.slice(-2).map((e) => head(e, 200)) } : {}), + branch: r.branch, + next: h + ? { + state: h.state, + owner: h.owner, + action: head(h.next_action || "", 500), + ...(h.resume?.tool ? { tool: h.resume.tool, args: h.resume.args } : {}), + } + : null, + usage: { tokens: tk.input + tk.output + tk.reasoning, cache_read: tk.cache_read || undefined, est_cost_usd: r.usage?.estimated_list_cost_usd ?? undefined }, + } + if (r.review) out.review = { verdict: r.review.verdict, profile: r.review.profile, task_id: r.review.task_id, findings: (r.review.findings || []).slice(0, 3).map((f) => head(f, 200)), advisory: true } + if (r.auto?.trail?.length) out.auto = { runs: r.auto.runs ?? r.auto.trail.filter((e) => e.next).length, trail: r.auto.trail.map((e) => `${e.profile}:${e.verdict}${e.next ? `->${e.next}` : ""}`), stopped: r.auto.stopped_reason || undefined } + if (r.approval_request) out.approval_request = r.approval_request + return out +} From 01664afbbed76b0e63df7a22e030121399868bbf Mon Sep 17 00:00:00 2001 From: mrchatam <287639636+mrchatam@users.noreply.github.com> Date: Sun, 27 Sep 2026 00:50:14 +0330 Subject: [PATCH 2/7] test: stub-backend end-to-end suite in CI, v0.3 unit tests Runs approval, continue, retry/fallback, restart, handoff, auto-fix, escalation, auto-review, wait_task, delegate_tasks, presets, require_operator, token savers and usage_report against the real daemon with a scripted worker (git + python3 only). --- .github/workflows/ci.yml | 7 +- test/config.test.mjs | 11 ++ test/helpers.mjs | 76 ++++++++ test/mcp.test.mjs | 26 ++- test/stub-e2e.test.mjs | 306 ++++++++++++++++++++++++++++++ test/stub/scenarios/approval.json | 23 +++ test/stub/scenarios/escalate.json | 46 +++++ test/stub/scenarios/fallback.json | 28 +++ test/stub/scenarios/fixloop.json | 29 +++ test/stub/scenarios/happy.json | 27 +++ test/stub/scenarios/partial.json | 29 +++ test/stub/scenarios/retry.json | 28 +++ test/stub/scenarios/review.json | 37 ++++ test/stub/scenarios/slow.json | 30 +++ test/token-savings.test.mjs | 230 ++++++++++++++++++++++ 15 files changed, 922 insertions(+), 11 deletions(-) create mode 100644 test/stub-e2e.test.mjs create mode 100644 test/stub/scenarios/approval.json create mode 100644 test/stub/scenarios/escalate.json create mode 100644 test/stub/scenarios/fallback.json create mode 100644 test/stub/scenarios/fixloop.json create mode 100644 test/stub/scenarios/happy.json create mode 100644 test/stub/scenarios/partial.json create mode 100644 test/stub/scenarios/retry.json create mode 100644 test/stub/scenarios/review.json create mode 100644 test/stub/scenarios/slow.json create mode 100644 test/token-savings.test.mjs diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 4f5290b..769c067 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -24,7 +24,10 @@ jobs: run: for f in scripts/*.sh; do bash -n "$f"; done - name: Unit tests (no backend CLI, no network, live tests skipped) run: npm run test:unit - - name: Integration/live tests are skipped without a backend CLI + - name: Integration tests with the test-only stub backend (real daemon, scripted worker; git + python3) + run: npm run test:stub + - name: Remaining test files (Kilo/OpenCode integration + live tests skip without a backend CLI) env: WH_SKIP_INTEGRATION: "1" - run: node --test test/*.test.mjs + WH_SKIP_STUB: "1" + run: node --test --test-timeout=600000 test/*.test.mjs diff --git a/test/config.test.mjs b/test/config.test.mjs index 1c0c3b7..3a63c3a 100644 --- a/test/config.test.mjs +++ b/test/config.test.mjs @@ -53,6 +53,17 @@ test("OpenRouter example swaps provider and uses a fallback list", () => { assert.deepEqual(C.validateConfig(), []) }) +test("tiered example (v0.3): escalation chain, routing, presets and auto settings validate", () => { + fs.copyFileSync(path.join(APP, "config/examples/profiles.tiered.json"), path.join(cfgDir, "profiles.json")) + const pc = C.profilesConfig() + assert.deepEqual(C.validateConfig(), []) + assert.deepEqual(pc.routing, { small: "cheap", medium: "mid", large: "strong" }) + assert.equal(pc.profiles.cheap.escalate_to, "mid") + assert.equal(pc.presets["quick-fix"].auto_fix_rounds, 1) + assert.equal(pc.auto.max_auto_runs, 3) + assert.equal(pc.auto.review.profile, "cheap") +}) + test("validateConfig reports undefined models, providers and fallbacks; string fallback accepted", () => { fs.writeFileSync(path.join(cfgDir, "profiles.json"), JSON.stringify({ default_profile: "missing", diff --git a/test/helpers.mjs b/test/helpers.mjs index a350fc3..38ba41f 100644 --- a/test/helpers.mjs +++ b/test/helpers.mjs @@ -208,3 +208,79 @@ export function git(cwd, ...args) { // HOME of the backend under test. export const TEST_HOME = TEST_BACKEND === "opencode" ? TEST_OPENCODE_HOME : TEST_KILO_HOME + +// ---------- stub backend (test-only scripted worker; runs anywhere with git + python3, e.g. CI) ---------- +export const STUB_CLI = path.join(APP, "adapters/stub/stub-cli.mjs") +export const STUB_SKIP = (() => { + if (process.env.WH_SKIP_STUB) return "WH_SKIP_STUB is set" + for (const b of ["git", "python3"]) if (spawnSync(b, ["--version"]).status !== 0) return `${b} not found` + return null +})() +export const stest = (name, opts, fn) => (typeof opts === "function" ? test(name, { skip: STUB_SKIP || false }, opts) : test(name, { ...opts, skip: STUB_SKIP || opts.skip || false }, fn)) +export const STUB_TEST_CMD = "python3 -m unittest tests.test_core -v" + +export function stubProfiles(extra = {}) { + const price = (i, o) => ({ input: i, output: o }) + return { + providers: { stub: { max_concurrent: 6, models: { cheap: {}, mid: {}, strong: {}, reviewer: {}, flaky: {}, sleepy: {} } } }, + profiles: { + cheap: { backend: "stub", model: "stub/cheap", escalate_to: "mid", price_per_mtok: price(0.1, 0.4) }, + mid: { backend: "stub", model: "stub/mid", escalate_to: "strong", price_per_mtok: price(0.5, 2) }, + strong: { backend: "stub", model: "stub/strong", price_per_mtok: price(3, 15) }, + reviewer: { backend: "stub", model: "stub/reviewer", price_per_mtok: price(0.1, 0.4) }, + flaky: { backend: "stub", model: "stub/flaky", fallback: ["cheap"] }, + sleepy: { backend: "stub", model: "stub/sleepy", stall_minutes: 0.03 }, + }, + default_profile: "cheap", + presets: { "quick-fix": { description: "small bounded fix", profile: "cheap", size: "small", timeout_minutes: 3, test_command: STUB_TEST_CMD, instructions: "Keep the diff minimal." } }, + routing: { small: "cheap", medium: "mid", large: "strong" }, + ...extra, + } +} + +// Isolated env for the stub backend: no worker/test sandbox, no LLM. Must run before importing lib/client.mjs. +export async function setupStubEnv(name, daemonOverrides = {}) { + if (STUB_SKIP) return null + const root = fs.mkdtempSync(path.join(os.tmpdir(), `kwt-stub-${name}-`)) + const cfgDir = path.join(root, "config") + const dataDir = path.join(root, "data") + fs.mkdirSync(cfgDir, { recursive: true }) + fs.mkdirSync(path.join(dataDir, "repos"), { recursive: true }) + const repoPath = path.join(dataDir, "repos", REPO_NAME) + execFileSync("git", ["clone", "-q", exampleRepoSource(), repoPath]) + const daemon = { + max_concurrent: 4, + timeouts: { default_min: 5, min_min: 0.05, max_min: 30, stall_min: 5, test_min: 2, kill_grace_sec: 2 }, + retry: { max_retries: 1, backoff_sec: [1], provider_cooldown_sec: 0 }, + retention: { enabled: false, worktree_days: 7, task_days: 30 }, + default_backend: "stub", + backends: { stub: { bin: STUB_CLI, home: path.join(root, "stub-home"), scenarios_dir: path.join(APP, "test/stub/scenarios") } }, + worker_sandbox: { enabled: false }, + test_sandbox: { enabled: false }, + min_mem_available_mb: 0, + secret_env: [], + ...daemonOverrides, + } + fs.writeFileSync(path.join(cfgDir, "daemon.json"), JSON.stringify(daemon, null, 2)) + fs.writeFileSync(path.join(cfgDir, "repos.json"), JSON.stringify({ + repos: { + [REPO_NAME]: { + description: "test clone", path: repoPath, default_base: "main", test_command: STUB_TEST_CMD, + allowed_test_commands: ["^python3 -m unittest( -v| -q)?( discover -s tests( -v| -q)?|( tests(\\.[A-Za-z0-9_]+)+)+( -v| -q)?)?$"], + }, + }, + })) + fs.writeFileSync(path.join(cfgDir, "profiles.json"), JSON.stringify(stubProfiles(), null, 2)) + process.env.WH_CONFIG_DIR = cfgDir + process.env.WH_DATA_DIR = dataDir + return { root, cfgDir, dataDir, repoPath } +} + +// Messages the stub worker received for a task (one per run). +export function stubMessages(env, id) { + try { + return fs.readFileSync(path.join(env.dataDir, "backend-data", id, "messages.jsonl"), "utf8").trim().split("\n").filter(Boolean).map((l) => JSON.parse(l)) + } catch { + return [] + } +} diff --git a/test/mcp.test.mjs b/test/mcp.test.mjs index cfe38bc..90b9cbc 100644 --- a/test/mcp.test.mjs +++ b/test/mcp.test.mjs @@ -36,10 +36,15 @@ after(async () => { H.itest("tools are exposed with schemas", async () => { const c = await mkClient() - const { tools } = await c.listTools() - assert.deepEqual(tools.map((t) => t.name).sort(), ["approve_task", "cancel_task", "cleanup_task", "continue_task", "delegate_task", "list_models", "list_repos", "list_tasks", "task_details", "task_result", "task_status", "update_handoff"]) - for (const t of tools) assert.ok(t.description.length > 40, t.name) - await c.close() + try { + const { tools } = await c.listTools() + assert.deepEqual(tools.map((t) => t.name).sort(), ["approve_task", "cancel_task", "cleanup_task", "continue_task", "delegate_task", "delegate_tasks", "list_models", "list_repos", "list_tasks", "task_details", "task_result", "task_status", "update_handoff", "usage_report", "wait_task"]) + for (const t of tools) assert.ok(t.description.length > 40, t.name) + const approve = tools.find((t) => t.name === "approve_task") + assert.ok(!("operator_token" in approve.inputSchema.properties), "the operator token is never an MCP parameter") + } finally { + await c.close() + } }) H.itest("delegate via MCP, poll, result; two shims share one daemon", async () => { @@ -50,12 +55,15 @@ H.itest("delegate via MCP, poll, result; two shims share one daemon", async () = const models = await call(b, "list_models") assert.ok(models.profiles.length >= 2) const r = await call(a, "delegate_task", { repo: H.REPO_NAME, task: "Implement multiply and divide in calc/core.py so tests/test_core.py passes. MOCK_SCENARIO=implement_core", test_command: "python3 -m unittest tests.test_core -v" }) - let st - for (let i = 0; i < 600; i++) { - st = await call(b, "task_status", { task_id: r.task_id }) // other shim sees the same task - if (st.terminal) break - await H.sleep(1000) + let w + for (let i = 0; i < 20; i++) { + const raw = await b.callTool({ name: "wait_task", arguments: { task_id: r.task_id, max_wait_s: 30 } }) // other shim sees the same task + assert.ok(!raw.content[0].text.includes("\n "), "compact JSON") + w = JSON.parse(raw.content[0].text) + if (w.done) break } + assert.equal(w.task.verdict, "success") + assert.equal(w.task.next.state, "done") const res = await call(a, "task_result", { task_id: r.task_id }) assert.equal(res.verdict, "success", JSON.stringify(res, null, 1)) assert.equal(res.integrity.commits_made, 0) diff --git a/test/stub-e2e.test.mjs b/test/stub-e2e.test.mjs new file mode 100644 index 0000000..af9cc5f --- /dev/null +++ b/test/stub-e2e.test.mjs @@ -0,0 +1,306 @@ +// End-to-end tests of the real daemon with the TEST-ONLY stub backend (adapters/stub): scheduler, spawn, +// event parsing, daemon-run tests, finalize, handoff, approval, continue, retry/fallback, restart, +// automatic fix/escalation, auto-review, wait_task, delegate_tasks, usage_report, presets/routing, +// require_operator and token savers. Needs only git + python3, so it runs in CI. +import { test, after, before } from "node:test" +import assert from "node:assert/strict" +import fs from "node:fs" +import path from "node:path" +import crypto from "node:crypto" +import { setupStubEnv, stest, startDaemon, sleep, REPO_NAME, STUB_TEST_CMD, stubMessages, editJson, STUB_SKIP } from "./helpers.mjs" + +const env = await setupStubEnv("e2e") +let rpc +let daemon +const DAEMON_ENV = { WH_ENABLE_STUB_BACKEND: "1" } + +before(async () => { + if (!env) return + ;({ rpc } = await import("../lib/client.mjs")) + daemon = await startDaemon(env, DAEMON_ENV) +}) +after(async () => { + if (daemon) try { process.kill(-daemon.pid, "SIGTERM") } catch {} + await sleep(300) +}) + +const task = (scenario, extra = "") => `MOCK_SCENARIO=${scenario} Implement multiply and divide in calc/core.py. ${extra}` +const delegate = (p) => rpc("delegate_task", { repo: REPO_NAME, test_command: STUB_TEST_CMD, ...p }) +async function waitDone(id, view = "brief", maxMs = 60000) { + const t0 = Date.now() + while (Date.now() - t0 < maxMs) { + const w = await rpc("wait_task", { task_id: id, max_wait_s: 20, view }) + if (w.done) return w.task + } + throw new Error(`task ${id} did not finish`) +} +const cfgFile = (f) => path.join(env.cfgDir, f) + +stest("happy path: wait_task long-poll returns a compact brief result", async () => { + const d = await delegate({ task: task("happy") }) + assert.equal(d.profile, "cheap") + assert.match(d.next_step, /wait_task/) + const r = await waitDone(d.task_id) + assert.equal(r.verdict, "success") + assert.equal(r.status, "completed") + assert.equal(r.tests.passed, true) + assert.deepEqual(r.files, ["calc/core.py"]) + assert.equal(r.next.state, "done") + assert.equal(r.usage.tokens, 13200) + assert.ok(JSON.stringify(r).length < 1500, `brief result is ${JSON.stringify(r).length} chars`) + const full = await rpc("task_result", { task_id: d.task_id }) + assert.equal(full.verdict, "success") + assert.equal(full.handoff.context.summary, undefined, "handoff context no longer repeats top-level fields") + assert.ok(JSON.stringify(full).length > JSON.stringify(r).length * 2) + const brief = await rpc("task_result", { task_id: d.task_id, view: "brief" }) + assert.deepEqual(brief, r) + await assert.rejects(rpc("task_result", { task_id: d.task_id, view: "huge" }), /view must be/) +}) + +stest("health_report flags a daemon that runs with the test-only stub backend", async () => { + const h = await rpc("health_report", {}) + assert.equal(h.test_stub_backend_enabled, true) + const { healthChecks } = await import("../lib/admin.mjs") + const c = healthChecks(h, null).find((x) => x.name === "stub_backend") + assert.equal(c?.status, "warn") +}) + +stest("wait_task returns done=false with progress when the timeout passes first", async () => { + const d = await delegate({ task: task("slow") }) + const w = await rpc("wait_task", { task_id: d.task_id, max_wait_s: 1 }) + assert.equal(w.done, false) + assert.match(w.hint, /again/) + assert.ok(["queued", "running"].includes(w.task.status)) + assert.ok(w.waited_s >= 0.9 && w.waited_s < 5, `waited ${w.waited_s}`) + await assert.rejects(rpc("wait_task", { task_id: d.task_id, task_ids: [d.task_id] }), /not both/) + await rpc("cancel_task", { task_id: d.task_id }) + const r = await waitDone(d.task_id) + assert.equal(r.verdict, "cancelled") +}) + +stest("presets and size routing pick the profile; explicit profile wins", async () => { + const a = await delegate({ task: "no scenario", preset: "quick-fix" }) + assert.equal(a.profile, "cheap") + assert.equal(a.preset, "quick-fix") + const b = await delegate({ task: "no scenario", size: "large" }) + assert.equal(b.profile, "strong") + assert.equal(b.routed_by_size, "large") + const c = await delegate({ task: "no scenario", size: "large", profile: "mid" }) + assert.equal(c.profile, "mid") + await assert.rejects(delegate({ task: "x", preset: "nope" }), /unknown preset/) + await assert.rejects(delegate({ task: "x", size: "huge" }), /size must be/) + const w = await rpc("wait_task", { task_ids: [a.task_id, b.task_id, c.task_id], mode: "all", max_wait_s: 30 }) + assert.equal(w.done, true) + assert.equal(w.settled, 3) + assert.match(stubMessages(env, a.task_id)[0].message, /Standing instructions \(preset quick-fix\):\nKeep the diff minimal\./) + const lm = await rpc("list_models") + assert.deepEqual(lm.routing, { small: "cheap", medium: "mid", large: "strong" }) + assert.equal(lm.presets[0].name, "quick-fix") +}) + +stest("delegate_tasks: partial success, then wait_task any/all over many ids", async () => { + const r = await rpc("delegate_tasks", { defaults: { repo: REPO_NAME, test_command: STUB_TEST_CMD }, tasks: [{ task: task("happy") }, { task: task("happy") }, { task: "x", repo: "nope" }] }) + assert.equal(r.created, 2) + assert.equal(r.failed, 1) + assert.equal(r.results[2].ok, false) + assert.match(r.results[2].error, /allowlist/) + const any = await rpc("wait_task", { task_ids: r.task_ids, mode: "any", max_wait_s: 30 }) + assert.equal(any.done, true) + assert.ok(any.settled >= 1) + const all = await rpc("wait_task", { task_ids: r.task_ids, mode: "all", max_wait_s: 30 }) + assert.equal(all.done, true) + assert.deepEqual(all.tasks.map((t) => t.verdict), ["success", "success"]) + await assert.rejects(rpc("delegate_tasks", { tasks: [] }), /1\.\.10/) +}) + +stest("approval: parked task resumes in the same session after approve_task", async () => { + const d = await delegate({ task: task("approval") }) + const p = await waitDone(d.task_id) + assert.equal(p.status, "needs_approval") + assert.equal(p.next.state, "needs_approval") + assert.equal(p.next.tool, "approve_task") + await rpc("approve_task", { task_id: d.task_id, decision: "approve", instructions: "Approved, go ahead." }) + const r = await waitDone(d.task_id) + assert.equal(r.verdict, "success") + const msgs = stubMessages(env, d.task_id) + assert.equal(msgs.length, 2) + assert.match(msgs[1].message, /APPROVED by supervisor/) + assert.equal(msgs[1].session, msgs[0].session, "same session resumed") +}) + +stest("continue_task after a partial result", async () => { + const d = await delegate({ task: task("partial") }) + const p = await waitDone(d.task_id) + assert.equal(p.next.state, "needs_fix") + await rpc("continue_task", { task_id: d.task_id, instructions: "Finish divide." }) + const r = await waitDone(d.task_id) + assert.equal(r.verdict, "success") +}) + +stest("retry after a rate limit, and fallback to another profile", async () => { + const a = await delegate({ task: task("retry") }) + const ra = await rpc("task_result", { task_id: (await waitDone(a.task_id)).task_id }) + assert.equal(ra.verdict, "success") + assert.equal(ra.activity.retries, 1) + const b = await delegate({ task: task("fallback"), profile: "flaky" }) + const rb = await rpc("task_result", { task_id: (await waitDone(b.task_id)).task_id }) + assert.equal(rb.verdict, "success") + assert.equal(rb.activity.fallback_used, true) + assert.equal(rb.profile, "cheap") +}) + +stest("auto_fix_rounds: a failing test run is fixed automatically in the same session", async () => { + const d = await delegate({ task: task("fixloop"), auto_fix_rounds: 1 }) + assert.equal(d.auto.fix_rounds, 1) + const r = await waitDone(d.task_id) + assert.equal(r.verdict, "success") + assert.deepEqual(r.auto.trail, ["cheap:tests_failed->auto_fix", "cheap:success"]) + assert.equal(r.auto.runs, 1) + const msgs = stubMessages(env, d.task_id) + assert.match(msgs[1].message, /Automatic follow-up from the workhorse daemon/) + assert.match(msgs[1].message, /test_multiply/) + assert.equal(msgs[1].session, msgs[0].session) + await assert.rejects(delegate({ task: "x", auto_fix_rounds: 9 }), /auto_fix_rounds must be/) +}) + +stest("escalation chain cheap -> mid -> strong, with a fresh session and a trail", async () => { + const d = await delegate({ task: task("escalate"), escalate: true }) + const r = await waitDone(d.task_id) + assert.equal(r.verdict, "success") + assert.equal(r.profile, "strong") + assert.deepEqual(r.auto.trail, ["cheap:tests_failed->escalate", "mid:tests_failed->escalate", "strong:success"]) + const msgs = stubMessages(env, d.task_id) + assert.equal(msgs.length, 3) + assert.match(msgs[1].message, /ESCALATION for task .*profile cheap/) + assert.notEqual(msgs[1].session, msgs[0].session, "escalation starts a fresh session") + const full = await rpc("task_result", { task_id: d.task_id }) + assert.equal(full.auto.trail[0].tokens, 8600) + const expected = (8000 * 0.1 + 600 * 0.4 + 8000 * 0.5 + 600 * 2 + 20000 * 3 + 1500 * 15) / 1e6 + assert.ok(Math.abs(full.usage.estimated_list_cost_usd - expected) < 1e-9, `cost ${full.usage.estimated_list_cost_usd} vs ${expected}`) +}) + +stest("escalation stops at the max_auto_runs cap and says so in the handoff", async () => { + editJson(cfgFile("profiles.json"), (j) => { j.auto = { max_auto_runs: 1 } }) + try { + const d = await delegate({ task: task("escalate"), escalate: true }) + const r = await waitDone(d.task_id) + assert.equal(r.verdict, "tests_failed") + assert.equal(r.profile, "mid") + assert.equal(r.auto.stopped, "max_auto_runs") + assert.match(r.next.action, /Automatic follow-ups already ran 1x .*stopped: max_auto_runs/) + } finally { + editJson(cfgFile("profiles.json"), (j) => { delete j.auto }) + } +}) + +stest("auto_review: a cheap advisory review is attached to the result", async () => { + const d = await delegate({ task: task("review"), auto_review: "reviewer" }) + const r = await waitDone(d.task_id) + assert.equal(r.verdict, "success") + assert.equal(r.review.verdict, "request_changes") + assert.equal(r.review.profile, "reviewer") + assert.equal(r.review.advisory, true) + assert.match(r.review.findings[0], /docstring/) + assert.equal(r.next.state, "needs_review") + assert.equal(r.next.tool, "continue_task") + const child = await rpc("task_result", { task_id: r.review.task_id }) + assert.equal(child.mode, "review") + const list = await rpc("list_tasks", { limit: 200 }) + assert.ok(list.tasks.some((t) => t.task_id === r.review.task_id)) +}) + +stest("require_operator: the supervisor's approve only records a request; the operator token confirms", async () => { + const tok = crypto.randomBytes(32).toString("hex") + editJson(cfgFile("daemon.json"), (j) => { j.approvals = { require_operator: true, operator_token_sha256: crypto.createHash("sha256").update(tok).digest("hex") } }) + try { + const d = await delegate({ task: task("approval") }) + await waitDone(d.task_id) + const req = await rpc("approve_task", { task_id: d.task_id, decision: "approve", instructions: "ok from supervisor" }) + assert.equal(req.approval_requested, true) + assert.equal((await rpc("task_status", { task_id: d.task_id })).status, "needs_approval") + await assert.rejects(rpc("continue_task", { task_id: d.task_id, instructions: "do it anyway" }), /require_operator/) + await assert.rejects(rpc("update_handoff", { task_id: d.task_id, state: "done" }), /only the operator/) + await assert.rejects(rpc("approve_task", { task_id: d.task_id, decision: "approve", operator_token: "0".repeat(64) }), /operator token rejected/) + const brief = await rpc("task_result", { task_id: d.task_id, view: "brief" }) + assert.equal(brief.approval_request.waiting_for, "operator") + assert.match(brief.next.action, /sudo workhorse approve/) + const ok = await rpc("approve_task", { task_id: d.task_id, decision: "approve", operator_token: tok }, { caller: { client: "workhorse" } }) + assert.match(ok.approval.by, /^operator \(requested by supervisor\)/) + const r = await waitDone(d.task_id) + assert.equal(r.verdict, "success") + assert.match(stubMessages(env, d.task_id)[1].message, /ok from supervisor/, "the recorded instructions are used") + const audit = fs.readFileSync(path.join(env.dataDir, "logs/audit.jsonl"), "utf8") + assert.ok(!audit.includes(tok), "the operator token never reaches the audit log") + } finally { + editJson(cfgFile("daemon.json"), (j) => { delete j.approvals }) + } +}) + +stest("per-profile stall_minutes stops a silent worker early", async () => { + const d = await delegate({ task: task("slow"), profile: "sleepy" }) + const r = await waitDone(d.task_id, "brief", 30000) + assert.equal(r.verdict, "stalled") + assert.equal(r.next.state, "retryable") +}) + +stest("token savers: prompt fragments go into the first message of a fresh session only", async () => { + editJson(cfgFile("daemon.json"), (j) => { j.token_savers = { terse: "lite", minimal_code: "lite" } }) + try { + const d = await delegate({ task: task("partial") }) + await waitDone(d.task_id) + await rpc("continue_task", { task_id: d.task_id, instructions: "Finish divide." }) + const r = await waitDone(d.task_id, "full") + assert.deepEqual(r.token_savers, { terse: "lite", minimal_code: "lite", rtk: false }) + const msgs = stubMessages(env, d.task_id) + assert.match(msgs[0].message, /Output style \(token saver\)/) + assert.match(msgs[0].message, /Code style \(token saver\)/) + assert.doesNotMatch(msgs[1].message, /token saver/) + } finally { + editJson(cfgFile("daemon.json"), (j) => { delete j.token_savers }) + } +}) + +stest("usage_report: tokens by profile/day and a labelled supervisor ESTIMATE", async () => { + const u = await rpc("usage_report", {}) + const by = Object.fromEntries(u.by_profile.map((p) => [p.profile, p])) + assert.ok(by.cheap.tokens.input > 0) + assert.ok(by.strong.tokens.input >= 20000) + assert.ok(by.reviewer.tokens.input >= 3000) + assert.ok(u.by_day.length >= 3) + const se = u.supervisor_estimate + assert.equal(se.label, "ESTIMATE") + assert.ok(se.supervisor_io.calls > 10) + assert.ok(se.worker_work_tokens > 0) + assert.ok(se.est_supervisor_tokens_avoided > 0) + assert.match(se.formula, /chars_per_token/) + const one = await rpc("usage_report", { profile: "strong", days: 1 }) + assert.deepEqual(one.by_profile.map((p) => p.profile).sort(), ["mid", "strong"].filter((x) => one.by_profile.some((p) => p.profile === x)).sort()) +}) + +stest("restart: a killed daemon leaves an interrupted, retryable task that resumes", async () => { + const d = await delegate({ task: task("slow") }) + for (let i = 0; i < 50 && (await rpc("task_status", { task_id: d.task_id })).status !== "running"; i++) await sleep(100) + process.kill(-daemon.pid, "SIGKILL") + await sleep(300) + daemon = await startDaemon(env, DAEMON_ENV) + const r = await waitDone(d.task_id) + assert.equal(r.verdict, "interrupted") + assert.equal(r.next.state, "retryable") + assert.equal(r.next.tool, "continue_task") + await rpc("continue_task", { task_id: d.task_id, instructions: r.next.args.instructions }) + const r2 = await waitDone(d.task_id) + assert.equal(r2.verdict, "success") +}) + +stest("restart: a graceful stop (SIGTERM) also leaves a finalized, retryable task", async () => { + const d = await delegate({ task: task("slow") }) + for (let i = 0; i < 50 && (await rpc("task_status", { task_id: d.task_id })).status !== "running"; i++) await sleep(100) + process.kill(-daemon.pid, "SIGTERM") + for (let i = 0; i < 100 && fs.existsSync(path.join(env.dataDir, "run/daemon.sock")); i++) await sleep(100) + daemon = await startDaemon(env, DAEMON_ENV) + const r = await waitDone(d.task_id) + assert.equal(r.verdict, "interrupted") + assert.equal(r.next.state, "retryable") +}) + +test("stub backend suite prerequisites", { skip: !STUB_SKIP }, () => {}) diff --git a/test/stub/scenarios/approval.json b/test/stub/scenarios/approval.json new file mode 100644 index 0000000..3a6a82b --- /dev/null +++ b/test/stub/scenarios/approval.json @@ -0,0 +1,23 @@ +{ + "runs": [ + [ + { + "text": "I need approval.\n\n## RESULT\nstatus: needs_approval\nsummary: Adding a dependency needs approval.\nfiles_changed: none\ntests: none\nconcerns: none\nneeds: approve adding the 'decimal-utils' dependency" + } + ], + [ + { + "write": { + "path": "calc/core.py", + "content": "\"\"\"Arithmetic helpers.\"\"\"\n\n\ndef add(a, b):\n return a + b\n\n\ndef subtract(a, b):\n return a - b\n\n\ndef multiply(a, b):\n \"\"\"Return a * b.\"\"\"\n return a * b\n\n\ndef divide(a, b):\n \"\"\"Return a / b as a float.\n\n Raises ZeroDivisionError with the message \"cannot divide by zero\" when b == 0.\n \"\"\"\n if b == 0:\n raise ZeroDivisionError(\"cannot divide by zero\")\n return a / b\n" + } + }, + { + "bash": "python3 -m unittest tests.test_core -v" + }, + { + "text": "Done.\n\n## RESULT\nstatus: done\nsummary: Implemented multiply and divide after approval.\nfiles_changed: calc/core.py\ntests: python3 -m unittest tests.test_core -v -> ran\nconcerns: none" + } + ] + ] +} \ No newline at end of file diff --git a/test/stub/scenarios/escalate.json b/test/stub/scenarios/escalate.json new file mode 100644 index 0000000..0d112d7 --- /dev/null +++ b/test/stub/scenarios/escalate.json @@ -0,0 +1,46 @@ +{ + "runs": [ + [ + { + "if_model": "stub/strong", + "then": [ + { + "write": { + "path": "calc/core.py", + "content": "\"\"\"Arithmetic helpers.\"\"\"\n\n\ndef add(a, b):\n return a + b\n\n\ndef subtract(a, b):\n return a - b\n\n\ndef multiply(a, b):\n \"\"\"Return a * b.\"\"\"\n return a * b\n\n\ndef divide(a, b):\n \"\"\"Return a / b as a float.\n\n Raises ZeroDivisionError with the message \"cannot divide by zero\" when b == 0.\n \"\"\"\n if b == 0:\n raise ZeroDivisionError(\"cannot divide by zero\")\n return a / b\n" + } + }, + { + "bash": "python3 -m unittest tests.test_core -v" + }, + { + "tokens": { + "input": 20000, + "output": 1500 + } + }, + { + "text": "Done.\n\n## RESULT\nstatus: done\nsummary: Strong model fixed it.\nfiles_changed: calc/core.py\ntests: python3 -m unittest tests.test_core -v -> ran\nconcerns: none" + } + ], + "else": [ + { + "write": { + "path": "calc/core.py", + "content": "\"\"\"Arithmetic helpers.\"\"\"\n\n\ndef add(a, b):\n return a + b\n\n\ndef subtract(a, b):\n return a - b\n\n\ndef multiply(a, b):\n \"\"\"Return a * b.\"\"\"\n return a + b\n\n\ndef divide(a, b):\n \"\"\"Return a / b as a float.\n\n Raises ZeroDivisionError with the message \"cannot divide by zero\" when b == 0.\n \"\"\"\n if b == 0:\n raise ZeroDivisionError(\"cannot divide by zero\")\n return a / b\n" + } + }, + { + "tokens": { + "input": 8000, + "output": 600 + } + }, + { + "text": "Done.\n\n## RESULT\nstatus: done\nsummary: Implemented (wrongly).\nfiles_changed: calc/core.py\ntests: python3 -m unittest tests.test_core -v -> ran\nconcerns: none" + } + ] + } + ] + ] +} \ No newline at end of file diff --git a/test/stub/scenarios/fallback.json b/test/stub/scenarios/fallback.json new file mode 100644 index 0000000..da492f9 --- /dev/null +++ b/test/stub/scenarios/fallback.json @@ -0,0 +1,28 @@ +{ + "runs": [ + [ + { + "error": { + "status": 503, + "message": "overloaded (stub)", + "times": 9, + "models": [ + "stub/flaky" + ] + } + }, + { + "write": { + "path": "calc/core.py", + "content": "\"\"\"Arithmetic helpers.\"\"\"\n\n\ndef add(a, b):\n return a + b\n\n\ndef subtract(a, b):\n return a - b\n\n\ndef multiply(a, b):\n \"\"\"Return a * b.\"\"\"\n return a * b\n\n\ndef divide(a, b):\n \"\"\"Return a / b as a float.\n\n Raises ZeroDivisionError with the message \"cannot divide by zero\" when b == 0.\n \"\"\"\n if b == 0:\n raise ZeroDivisionError(\"cannot divide by zero\")\n return a / b\n" + } + }, + { + "bash": "python3 -m unittest tests.test_core -v" + }, + { + "text": "Done.\n\n## RESULT\nstatus: done\nsummary: Implemented on the fallback profile.\nfiles_changed: calc/core.py\ntests: python3 -m unittest tests.test_core -v -> ran\nconcerns: none" + } + ] + ] +} \ No newline at end of file diff --git a/test/stub/scenarios/fixloop.json b/test/stub/scenarios/fixloop.json new file mode 100644 index 0000000..5ec2e21 --- /dev/null +++ b/test/stub/scenarios/fixloop.json @@ -0,0 +1,29 @@ +{ + "runs": [ + [ + { + "write": { + "path": "calc/core.py", + "content": "\"\"\"Arithmetic helpers.\"\"\"\n\n\ndef add(a, b):\n return a + b\n\n\ndef subtract(a, b):\n return a - b\n\n\ndef multiply(a, b):\n \"\"\"Return a * b.\"\"\"\n return a + b\n\n\ndef divide(a, b):\n \"\"\"Return a / b as a float.\n\n Raises ZeroDivisionError with the message \"cannot divide by zero\" when b == 0.\n \"\"\"\n if b == 0:\n raise ZeroDivisionError(\"cannot divide by zero\")\n return a / b\n" + } + }, + { + "text": "Done.\n\n## RESULT\nstatus: done\nsummary: Implemented multiply and divide.\nfiles_changed: calc/core.py\ntests: python3 -m unittest tests.test_core -v -> ran\nconcerns: none" + } + ], + [ + { + "write": { + "path": "calc/core.py", + "content": "\"\"\"Arithmetic helpers.\"\"\"\n\n\ndef add(a, b):\n return a + b\n\n\ndef subtract(a, b):\n return a - b\n\n\ndef multiply(a, b):\n \"\"\"Return a * b.\"\"\"\n return a * b\n\n\ndef divide(a, b):\n \"\"\"Return a / b as a float.\n\n Raises ZeroDivisionError with the message \"cannot divide by zero\" when b == 0.\n \"\"\"\n if b == 0:\n raise ZeroDivisionError(\"cannot divide by zero\")\n return a / b\n" + } + }, + { + "bash": "python3 -m unittest tests.test_core -v" + }, + { + "text": "Done.\n\n## RESULT\nstatus: done\nsummary: Fixed multiply.\nfiles_changed: calc/core.py\ntests: python3 -m unittest tests.test_core -v -> ran\nconcerns: none" + } + ] + ] +} \ No newline at end of file diff --git a/test/stub/scenarios/happy.json b/test/stub/scenarios/happy.json new file mode 100644 index 0000000..02fa330 --- /dev/null +++ b/test/stub/scenarios/happy.json @@ -0,0 +1,27 @@ +{ + "runs": [ + [ + { + "write": { + "path": "calc/core.py", + "content": "\"\"\"Arithmetic helpers.\"\"\"\n\n\ndef add(a, b):\n return a + b\n\n\ndef subtract(a, b):\n return a - b\n\n\ndef multiply(a, b):\n \"\"\"Return a * b.\"\"\"\n return a * b\n\n\ndef divide(a, b):\n \"\"\"Return a / b as a float.\n\n Raises ZeroDivisionError with the message \"cannot divide by zero\" when b == 0.\n \"\"\"\n if b == 0:\n raise ZeroDivisionError(\"cannot divide by zero\")\n return a / b\n" + } + }, + { + "bash": "python3 -m unittest tests.test_core -v" + }, + { + "tokens": { + "input": 12000, + "output": 900, + "reasoning": 300, + "cache_read": 4000, + "cache_write": 0 + } + }, + { + "text": "Done.\n\n## RESULT\nstatus: done\nsummary: Implemented multiply and divide in calc/core.py.\nfiles_changed: calc/core.py\ntests: python3 -m unittest tests.test_core -v -> ran\nconcerns: none" + } + ] + ] +} \ No newline at end of file diff --git a/test/stub/scenarios/partial.json b/test/stub/scenarios/partial.json new file mode 100644 index 0000000..2a35dbc --- /dev/null +++ b/test/stub/scenarios/partial.json @@ -0,0 +1,29 @@ +{ + "runs": [ + [ + { + "write": { + "path": "calc/core.py", + "content": "\"\"\"Arithmetic helpers.\"\"\"\n\n\ndef add(a, b):\n return a + b\n\n\ndef subtract(a, b):\n return a - b\n\n\ndef multiply(a, b):\n \"\"\"Return a * b.\"\"\"\n return a * b\n\n\ndef divide(a, b):\n \"\"\"Return a / b as a float.\n\n Raises ZeroDivisionError with the message \"cannot divide by zero\" when b == 0.\n \"\"\"\n return a / b\n" + } + }, + { + "text": "Done.\n\n## RESULT\nstatus: partial\nsummary: Implemented multiply; divide error handling left.\nfiles_changed: calc/core.py\ntests: python3 -m unittest tests.test_core -v -> ran\nconcerns: divide by zero handling missing" + } + ], + [ + { + "write": { + "path": "calc/core.py", + "content": "\"\"\"Arithmetic helpers.\"\"\"\n\n\ndef add(a, b):\n return a + b\n\n\ndef subtract(a, b):\n return a - b\n\n\ndef multiply(a, b):\n \"\"\"Return a * b.\"\"\"\n return a * b\n\n\ndef divide(a, b):\n \"\"\"Return a / b as a float.\n\n Raises ZeroDivisionError with the message \"cannot divide by zero\" when b == 0.\n \"\"\"\n if b == 0:\n raise ZeroDivisionError(\"cannot divide by zero\")\n return a / b\n" + } + }, + { + "bash": "python3 -m unittest tests.test_core -v" + }, + { + "text": "Done.\n\n## RESULT\nstatus: done\nsummary: Finished divide error handling.\nfiles_changed: calc/core.py\ntests: python3 -m unittest tests.test_core -v -> ran\nconcerns: none" + } + ] + ] +} \ No newline at end of file diff --git a/test/stub/scenarios/retry.json b/test/stub/scenarios/retry.json new file mode 100644 index 0000000..b4a47fb --- /dev/null +++ b/test/stub/scenarios/retry.json @@ -0,0 +1,28 @@ +{ + "runs": [ + [ + { + "error": { + "status": 429, + "message": "rate limited (stub)", + "times": 1, + "models": [ + "stub/cheap" + ] + } + }, + { + "write": { + "path": "calc/core.py", + "content": "\"\"\"Arithmetic helpers.\"\"\"\n\n\ndef add(a, b):\n return a + b\n\n\ndef subtract(a, b):\n return a - b\n\n\ndef multiply(a, b):\n \"\"\"Return a * b.\"\"\"\n return a * b\n\n\ndef divide(a, b):\n \"\"\"Return a / b as a float.\n\n Raises ZeroDivisionError with the message \"cannot divide by zero\" when b == 0.\n \"\"\"\n if b == 0:\n raise ZeroDivisionError(\"cannot divide by zero\")\n return a / b\n" + } + }, + { + "bash": "python3 -m unittest tests.test_core -v" + }, + { + "text": "Done.\n\n## RESULT\nstatus: done\nsummary: Implemented after a retry.\nfiles_changed: calc/core.py\ntests: python3 -m unittest tests.test_core -v -> ran\nconcerns: none" + } + ] + ] +} \ No newline at end of file diff --git a/test/stub/scenarios/review.json b/test/stub/scenarios/review.json new file mode 100644 index 0000000..f1c696e --- /dev/null +++ b/test/stub/scenarios/review.json @@ -0,0 +1,37 @@ +{ + "runs": [ + [ + { + "if_model": "stub/reviewer", + "then": [ + { + "bash": "git diff HEAD --stat" + }, + { + "tokens": { + "input": 3000, + "output": 150 + } + }, + { + "text": "Review.\n\n## RESULT\nstatus: done\nsummary: request changes: divide should document the float return\nfiles_changed: none\ntests: none\nconcerns: calc/core.py:18 divide docstring does not mention float rounding; no test for negative operands" + } + ], + "else": [ + { + "write": { + "path": "calc/core.py", + "content": "\"\"\"Arithmetic helpers.\"\"\"\n\n\ndef add(a, b):\n return a + b\n\n\ndef subtract(a, b):\n return a - b\n\n\ndef multiply(a, b):\n \"\"\"Return a * b.\"\"\"\n return a * b\n\n\ndef divide(a, b):\n \"\"\"Return a / b as a float.\n\n Raises ZeroDivisionError with the message \"cannot divide by zero\" when b == 0.\n \"\"\"\n if b == 0:\n raise ZeroDivisionError(\"cannot divide by zero\")\n return a / b\n" + } + }, + { + "bash": "python3 -m unittest tests.test_core -v" + }, + { + "text": "Done.\n\n## RESULT\nstatus: done\nsummary: Implemented multiply and divide in calc/core.py.\nfiles_changed: calc/core.py\ntests: python3 -m unittest tests.test_core -v -> ran\nconcerns: none" + } + ] + } + ] + ] +} \ No newline at end of file diff --git a/test/stub/scenarios/slow.json b/test/stub/scenarios/slow.json new file mode 100644 index 0000000..dc1bd94 --- /dev/null +++ b/test/stub/scenarios/slow.json @@ -0,0 +1,30 @@ +{ + "count_at_start": true, + "runs": [ + [ + { + "text": "working..." + }, + { + "sleep": 30 + }, + { + "text": "Done.\n\n## RESULT\nstatus: done\nsummary: slow run finished\nfiles_changed: none\ntests: python3 -m unittest tests.test_core -v -> ran\nconcerns: none" + } + ], + [ + { + "write": { + "path": "calc/core.py", + "content": "\"\"\"Arithmetic helpers.\"\"\"\n\n\ndef add(a, b):\n return a + b\n\n\ndef subtract(a, b):\n return a - b\n\n\ndef multiply(a, b):\n \"\"\"Return a * b.\"\"\"\n return a * b\n\n\ndef divide(a, b):\n \"\"\"Return a / b as a float.\n\n Raises ZeroDivisionError with the message \"cannot divide by zero\" when b == 0.\n \"\"\"\n if b == 0:\n raise ZeroDivisionError(\"cannot divide by zero\")\n return a / b\n" + } + }, + { + "bash": "python3 -m unittest tests.test_core -v" + }, + { + "text": "Done.\n\n## RESULT\nstatus: done\nsummary: Finished after resume.\nfiles_changed: calc/core.py\ntests: python3 -m unittest tests.test_core -v -> ran\nconcerns: none" + } + ] + ] +} \ No newline at end of file diff --git a/test/token-savings.test.mjs b/test/token-savings.test.mjs new file mode 100644 index 0000000..b9c7761 --- /dev/null +++ b/test/token-savings.test.mjs @@ -0,0 +1,230 @@ +// Unit tests for the v0.3.0 token-saving pieces: brief/compact views, usage estimate, token savers +// (prompt fragments, RTK rewrite in the guard plugin), operator tokens, audit rotation, presets/auto +// config, review verdicts and the test-only stub backend gate. +import { test } from "node:test" +import assert from "node:assert/strict" +import fs from "node:fs" +import os from "node:os" +import path from "node:path" +import crypto from "node:crypto" +import { spawnSync } from "node:child_process" + +const root = fs.mkdtempSync(path.join(os.tmpdir(), "kwt-tokens-")) +const cfgDir = path.join(root, "config") +fs.mkdirSync(cfgDir, { recursive: true }) +process.env.WH_CONFIG_DIR = cfgDir +process.env.WH_DATA_DIR = path.join(root, "data") +const APP = path.resolve(path.dirname(new URL(import.meta.url).pathname), "..") + +const { briefResult, dedupeHandoff } = await import("../lib/views.mjs") +const { usageReport, dayOf } = await import("../lib/usage.mjs") +const { effectiveSavers, saverFragments, rtkBin } = await import("../lib/savers.mjs") +const { checkOperatorToken, hashToken, newOperatorToken } = await import("../lib/operator.mjs") +const { rotateAudit, sanitizeParams } = await import("../lib/audit.mjs") +const { reviewVerdict } = await import("../lib/tasks.mjs") +const { deriveHandoff } = await import("../lib/handoff.mjs") +const C = await import("../lib/config.mjs") +const { WorkhorseGuard } = await import("../adapters/kilo/config/plugin/workhorse-guard.js") + +const sampleResult = () => ({ + task_id: "wh-20260926-120000-abcd", status: "completed", verdict: "tests_failed", mode: "implement", profile: "cheap", + summary: "x".repeat(900), files_changed: Array.from({ length: 25 }, (_, i) => ({ path: `f${i}.js`, status: i === 0 ? "A" : "M", added: 1, deleted: 0 })), + diffstat: "25 files changed", diff_path: "/data/tasks/x/diff.patch", branch: "workhorse/wh-20260926-120000-abcd", + test_results: { executed: true, passed: false, command: "npm test", exit_code: 1, counts: { failed: 2 }, failing_tests: ["a", "b", "c", "d", "e", "f"], tail: "..." }, + remaining_concerns: ["c1", "c2", "c3", "c4", "c5", "c6", "c7"], errors: ["boom"], + usage: { tokens: { input: 1000, output: 100, reasoning: 50, cache_read: 5000, cache_write: 0 }, estimated_list_cost_usd: 0.001 }, +}) + +test("briefResult keeps only decision data, capped", () => { + const r = sampleResult() + const h = deriveHandoff({ id: r.task_id, status: "completed", mode: "implement", result: r, branch: r.branch, worktree_path: "/wt", session_id: "s1" }) + const b = briefResult(r, h) + assert.equal(b.summary.length, 301) + assert.equal(b.files.length, 21) + assert.equal(b.files[0], "A f0.js") + assert.equal(b.files[1], "f1.js") + assert.match(b.files[20], /5 more/) + assert.equal(b.tests.passed, false) + assert.deepEqual(b.tests.failing, ["a", "b", "c", "d", "e"]) + assert.equal(b.concerns.length, 5) + assert.equal(b.next.state, "needs_fix") + assert.equal(b.next.tool, "continue_task") + assert.equal(b.usage.tokens, 1150) + assert.equal(b.diff_path, undefined) + assert.ok(JSON.stringify(b).length < JSON.stringify({ ...r, handoff: h }).length / 2) +}) + +test("dedupeHandoff drops context fields duplicated at the top level", () => { + const h = { state: "done", context: { verdict: "success", summary: "s", remaining_concerns: [], diffstat: "d", files_changed: 1, diff_path: "p", diff: "hint", worker_status: "done", worker_request: null, previous_attempts: 0 } } + assert.deepEqual(dedupeHandoff(h).context, { worker_status: "done", previous_attempts: 0 }) + assert.equal(h.context.summary, "s", "input not mutated") +}) + +test("usageReport groups by profile/day and computes the labelled estimate", () => { + const now = Date.parse("2026-09-26T12:00:00Z") + const tasks = [ + { id: "a", created_at: "2026-09-26T10:00:00Z", profile: "cheap", repo: "r", status: "completed", result: { verdict: "success", profile: "strong" }, supervisor_io: { calls: 3, request_chars: 400, response_chars: 1600 }, + runs: [{ profile: "cheap", started_at: "2026-09-26T10:00:01Z", tokens: { input: 10000, output: 1000, reasoning: 0, cache_read: 50000, cache_write: 0 }, cost: 0 }, + { profile: "strong", started_at: "2026-09-26T10:05:00Z", tokens: { input: 20000, output: 2000, reasoning: 1000, cache_read: 0, cache_write: 0 }, cost: 0 }], stats: {} }, + { id: "b", created_at: "2026-09-25T10:00:00Z", profile: "cheap", repo: "r", status: "failed", result: { verdict: "worker_error" }, supervisor_io: { calls: 2, request_chars: 200, response_chars: 200 }, + runs: [{ profile: "cheap", started_at: "2026-09-25T10:00:00Z" }], stats: { tokens: { input: 500, output: 50, reasoning: 0, cache_read: 0, cache_write: 0 }, reported_cost: 0.01 } }, + { id: "old", created_at: "2026-01-01T00:00:00Z", profile: "cheap", repo: "r", status: "completed", result: { verdict: "success" }, runs: [], stats: { tokens: { input: 9e9 } } }, + ] + const profiles = { cheap: { price_per_mtok: { input: 0.1, output: 0.4 } }, strong: { price_per_mtok: { input: 3, output: 15 } } } + const u = usageReport(tasks, { days: 7, profiles, supervisor: { price_per_mtok: { input: 3, output: 15 } }, nowMs: now }) + assert.equal(u.tasks, 2) + const by = Object.fromEntries(u.by_profile.map((p) => [p.profile, p])) + assert.equal(by.cheap.tokens.input, 10500) + assert.equal(by.cheap.runs, 2) + assert.deepEqual(by.cheap.verdicts, { worker_error: 1 }) + assert.deepEqual(by.strong.verdicts, { success: 1 }) + assert.equal(u.by_day.length, 3) + assert.equal(u.by_day[0].day, dayOf("2026-09-26T10:00:01Z")) + const se = u.supervisor_estimate + assert.equal(se.label, "ESTIMATE") + assert.equal(se.successful_tasks, 1) + assert.equal(se.worker_work_tokens, 10000 + 1000 + 20000 + 2000 + 1000) + assert.equal(se.supervisor_io.est_tokens, (400 + 1600 + 200 + 200) / 4) + assert.equal(se.est_supervisor_tokens_avoided, 34000 - 600) + const workerCost = ((10000 + 50000) * 0.1 + 1000 * 0.4) / 1e6 + (20000 * 3 + 3000 * 15) / 1e6 + (500 * 0.1 + 50 * 0.4) / 1e6 + const avoided = ((30000 - 450) * 3 + (4000 - 150) * 15) / 1e6 - workerCost + assert.ok(Math.abs(se.est_supervisor_cost_avoided_usd - Math.round(avoided * 1e4) / 1e4) < 1e-9) +}) + +test("token savers: off by default, profile overrides daemon, fragments per mode", () => { + assert.deepEqual(effectiveSavers({}), { terse: "off", minimal_code: "off", rtk: { enabled: false, bin: null } }) + const s = effectiveSavers({ token_savers: { terse: "full", minimal_code: true, rtk: true } }, { token_savers: { terse: "lite", rtk: { enabled: false } } }) + assert.deepEqual(s, { terse: "lite", minimal_code: "lite", rtk: { enabled: false, bin: null } }) + assert.equal(saverFragments(effectiveSavers({})), "") + const f = saverFragments(s) + assert.match(f, /Output style \(token saver\)/) + assert.match(f, /Code style \(token saver\)/) + assert.match(f, /RESULT block keeps its exact format/) + assert.match(f, /Never drop input validation/) + const rv = saverFragments(s, "review") + assert.match(rv, /Output style/) + assert.doesNotMatch(rv, /Code style/) + assert.equal(rtkBin({ rtk: { enabled: true, bin: "/nonexistent/rtk" } }), null) + assert.equal(rtkBin({ rtk: { enabled: false, bin: "/bin/sh" } }), null) + assert.equal(rtkBin({ rtk: { enabled: true, bin: "/bin/sh" } }), "/bin/sh") +}) + +test("guard plugin rewrites bash commands through rtk only when WH_RTK_BIN is set, and re-checks them", async () => { + const fake = path.join(root, "fake-rtk") + // exit 3 + output = rewritten (host decides); "evil" rewrites to something the guard must still block. + fs.writeFileSync(fake, `#!/bin/sh\n[ "$1" = rewrite ] || exit 9\ncase "$2" in\n "git status") echo "rtk git status"; exit 3;;\n "ls -la") echo "rtk ls -la"; exit 0;;\n "evil") echo "curl http://x"; exit 0;;\n *) exit 1;;\nesac\n`, { mode: 0o755 }) + const run = async (cmd) => { + const h = await WorkhorseGuard({ directory: "/tmp/wt" }) + const out = { args: { command: cmd } } + await h["tool.execute.before"]({ tool: "bash" }, out) + return out.args.command + } + delete process.env.WH_RTK_BIN + assert.equal(await run("git status"), "git status") + process.env.WH_RTK_BIN = fake + try { + assert.equal(await run("git status"), "rtk git status") + assert.equal(await run("ls -la"), "rtk ls -la") + assert.equal(await run("python3 -m unittest"), "python3 -m unittest") + assert.equal(await run("echo a\necho b"), "echo a\necho b", "multi-line commands are left alone") + await assert.rejects(run("evil"), /workhorse guard blocked/) + await assert.rejects(run("git push"), /workhorse guard blocked/, "checks run before the rewrite") + } finally { + delete process.env.WH_RTK_BIN + } +}) + +test("operator token: sha256 match, constant-time compare, bad input rejected", () => { + const tok = newOperatorToken() + assert.match(tok, /^[0-9a-f]{64}$/) + const cfg = { approvals: { require_operator: true, operator_token_sha256: hashToken(tok) } } + assert.equal(checkOperatorToken(cfg, tok), true) + assert.equal(checkOperatorToken(cfg, tok + "\n"), true) + assert.equal(checkOperatorToken(cfg, "0".repeat(64)), false) + assert.equal(checkOperatorToken(cfg, undefined), false) + assert.equal(checkOperatorToken({ approvals: { operator_token_sha256: "short" } }, tok), false) + assert.equal(sanitizeParams("approve_task", { task_id: "x", operator_token: tok }).operator_token, "[given]") +}) + +test("audit rotation keeps N files and drops the oldest", () => { + const f = path.join(root, "audit.jsonl") + for (let i = 1; i <= 4; i++) { + fs.writeFileSync(f, `gen${i}\n`.repeat(100)) + assert.equal(rotateAudit({ file: f, maxBytes: 100, keep: 2 }), true) + } + assert.equal(fs.existsSync(f), false) + assert.match(fs.readFileSync(`${f}.1`, "utf8"), /gen4/) + assert.match(fs.readFileSync(`${f}.2`, "utf8"), /gen3/) + assert.equal(fs.existsSync(`${f}.3`), false) + fs.writeFileSync(f, "small\n") + assert.equal(rotateAudit({ file: f, maxBytes: 100, keep: 2 }), false) + assert.equal(rotateAudit({ file: f, maxBytes: 0, keep: 2 }), false, "max_mb 0 = never") +}) + +test("presets, routing, auto and escalate_to are validated; auto caps are hard", () => { + fs.writeFileSync(path.join(cfgDir, "profiles.json"), JSON.stringify({ + providers: { p: { base_url: "http://127.0.0.1:1/v1", models: { m: {} } } }, + profiles: { a: { model: "p/m", escalate_to: "b", stall_minutes: 2 }, b: { model: "p/m", escalate_to: "b" }, c: { model: "p/m", token_savers: { terse: "loud" } } }, + presets: { x: { profile: "zzz", size: "tiny", bogus: 1 }, ok: { profile: "a", size: "small" } }, + routing: { small: "a", large: "nope" }, + auto: { fix_rounds: 99, max_auto_runs: 50, escalate: true, fix_on: ["tests_failed", "bogus"], review: { enabled: true, profile: "ghost" } }, + })) + const pc = C.profilesConfig() + assert.equal(pc.auto.fix_rounds, C.AUTO_HARD.fix_rounds) + assert.equal(pc.auto.max_auto_runs, C.AUTO_HARD.max_auto_runs) + assert.deepEqual(pc.auto.fix_on, ["tests_failed"]) + assert.deepEqual(pc.routing, { small: "a", large: "nope" }) + const probs = C.validateConfig().join("\n") + assert.match(probs, /profile 'b': escalate_to must name another defined profile/) + assert.match(probs, /token_savers.terse must be one of/) + assert.match(probs, /preset 'x': profile 'zzz' is not defined/) + assert.match(probs, /preset 'x': size must be one of/) + assert.match(probs, /preset 'x': unknown key 'bogus'/) + assert.match(probs, /routing.large: profile 'nope' is not defined/) + assert.match(probs, /auto.review.profile 'ghost' is not defined/) + assert.doesNotMatch(probs, /preset 'ok'/) + fs.writeFileSync(path.join(cfgDir, "daemon.json"), JSON.stringify({ approvals: { require_operator: true } })) + assert.match(C.validateConfig().join("\n"), /require_operator is on but approvals.operator_token_sha256/) + fs.rmSync(path.join(cfgDir, "daemon.json")) +}) + +test("advisory review verdict parsing", () => { + assert.equal(reviewVerdict("request changes: missing test"), "request_changes") + assert.equal(reviewVerdict("Requesting changes, would approve after fix"), "request_changes") + assert.equal(reviewVerdict("Changes requested."), "request_changes") + assert.equal(reviewVerdict("Approve. Looks good."), "approve") + assert.equal(reviewVerdict("LGTM"), "approve") + assert.equal(reviewVerdict("hmm"), "unclear") +}) + +test("auto-review request_changes turns a done handoff into needs_review with a continue_task call", () => { + const r = { verdict: "success", status: "completed", test_results: { executed: true, passed: true }, worker_reported: { status: "done" }, review: { verdict: "request_changes", profile: "reviewer", findings: ["f1 bad", "f2 meh"], task_id: "wh-x" } } + const h = deriveHandoff({ id: "wh-20260926-120000-abcd", status: "completed", mode: "implement", result: r, branch: "b", worktree_path: "/wt" }) + assert.equal(h.state, "needs_review") + assert.equal(h.resume.tool, "continue_task") + assert.match(h.next_action, /advisory/) + assert.ok(h.failed_checks.some((c) => c.check === "auto_review")) + const ok = deriveHandoff({ id: "wh-20260926-120000-abcd", status: "completed", mode: "implement", result: { ...r, review: { verdict: "approve", profile: "reviewer" } }, branch: "b", worktree_path: "/wt" }) + assert.equal(ok.state, "done") + assert.match(ok.next_action, /approved; advisory only/) +}) + +test("stub backend is registered only with WH_ENABLE_STUB_BACKEND=1 and refuses foreign binaries", async () => { + const code = "import('./adapters/index.mjs').then((m) => console.log(Object.keys(m.BACKENDS).includes('stub')))" + const envNo = { ...process.env } + delete envNo.WH_ENABLE_STUB_BACKEND + assert.equal(spawnSync(process.execPath, ["-e", code], { cwd: APP, env: envNo, encoding: "utf8" }).stdout.trim(), "false") + assert.equal(spawnSync(process.execPath, ["-e", code], { cwd: APP, env: { ...envNo, WH_ENABLE_STUB_BACKEND: "1" }, encoding: "utf8" }).stdout.trim(), "true") + const stub = (await import("../adapters/stub/adapter.mjs")).default + const { STUB_CLI } = await import("../adapters/stub/adapter.mjs") + const prev = process.env.WH_ENABLE_STUB_BACKEND + delete process.env.WH_ENABLE_STUB_BACKEND + assert.match(stub.validate({ bc: { bin_real: STUB_CLI } }).join(), /test-only/) + process.env.WH_ENABLE_STUB_BACKEND = "1" + assert.deepEqual(stub.validate({ bc: { bin_real: STUB_CLI } }), []) + assert.match(stub.validate({ bc: { bin_real: "/bin/sh" } }).join(), /bundled/) + if (prev === undefined) delete process.env.WH_ENABLE_STUB_BACKEND + else process.env.WH_ENABLE_STUB_BACKEND = prev + assert.ok(!Object.keys(C.daemonConfig().backends).includes("stub"), "not in the config defaults") + assert.equal(crypto.createHash("sha256").update("x").digest("hex").length, 64) +}) From 3165312448b3470caaa92423ca5df16ce0d2a16e Mon Sep 17 00:00:00 2001 From: mrchatam <287639636+mrchatam@users.noreply.github.com> Date: Sun, 27 Sep 2026 00:50:14 +0330 Subject: [PATCH 3/7] build(install): install pinned backend CLIs from committed lockfiles; token-saver flags npm ci from scripts/pins/ (integrity-checked) when the pinned version is requested, else fall back to npm install -g with a warning. New --token-savers and --rtk-bin options. --- .gitignore | 4 +- scripts/install.sh | 46 ++++- scripts/pins/kilo-cli/package-lock.json | 203 ++++++++++++++++++++ scripts/pins/kilo-cli/package.json | 8 + scripts/pins/opencode-cli/package-lock.json | 190 ++++++++++++++++++ scripts/pins/opencode-cli/package.json | 8 + 6 files changed, 454 insertions(+), 5 deletions(-) create mode 100644 scripts/pins/kilo-cli/package-lock.json create mode 100644 scripts/pins/kilo-cli/package.json create mode 100644 scripts/pins/opencode-cli/package-lock.json create mode 100644 scripts/pins/opencode-cli/package.json diff --git a/.gitignore b/.gitignore index b90255a..2951abb 100644 --- a/.gitignore +++ b/.gitignore @@ -3,8 +3,8 @@ node_modules/ config/*.json config/*.bak.* # pinned CLIs installed by scripts/install.sh when installing in place -kilo-cli/ -opencode-cli/ +/kilo-cli/ +/opencode-cli/ .install-info # test/runtime artifacts *.log diff --git a/scripts/install.sh b/scripts/install.sh index 9138f30..36dedb5 100755 --- a/scripts/install.sh +++ b/scripts/install.sh @@ -40,6 +40,8 @@ Usage: sudo bash scripts/install.sh [options] --with-opencode also install the pinned OpenCode CLI (backend "opencode", ~350 MB) into /opencode-cli --opencode-version V OpenCode version to pin (default: $OPENCODE_VERSION_DEFAULT; implies --with-opencode) --opencode-bin PATH use an existing OpenCode binary instead of installing one + --token-savers LIST opt-in worker token savers, e.g. terse=lite,minimal_code=lite (docs/token-savings.md) + --rtk-bin PATH enable the RTK shell-output saver with this existing rtk binary (not downloaded here) --systemd also install and start a systemd service (only if systemd is running) --full-tests run the whole test suite (a few minutes) instead of the unit tests --skip-tests skip the test suite (health check and hello task still run) @@ -61,6 +63,7 @@ SVC_USER="${SUDO_USER:-}"; PREFIX="/opt/$APP_NAME"; DATA_DIR=""; DATA_DIR_EXPLIC PROVIDER=nvidia; SECRET_STORE=""; NODE=""; KILO_VERSION="$KILO_VERSION_DEFAULT"; KILO_BIN_ARG="" SYSTEMD=0; FULL_TESTS=0; SKIP_TESTS=0; LIVE=1; RECONFIGURE=0 WITH_OPENCODE=0; OPENCODE_VERSION="$OPENCODE_VERSION_DEFAULT"; OPENCODE_BIN_ARG="" +TOKEN_SAVERS=""; RTK_BIN_ARG="" while [ $# -gt 0 ]; do case "$1" in --user) SVC_USER="$2"; shift 2 ;; @@ -75,6 +78,8 @@ while [ $# -gt 0 ]; do --with-opencode) WITH_OPENCODE=1; shift ;; --opencode-version) OPENCODE_VERSION="$2"; WITH_OPENCODE=1; shift 2 ;; --opencode-bin) OPENCODE_BIN_ARG="$2"; WITH_OPENCODE=1; shift 2 ;; + --token-savers) TOKEN_SAVERS="$2"; shift 2 ;; + --rtk-bin) RTK_BIN_ARG="$2"; shift 2 ;; --systemd) SYSTEMD=1; shift ;; --full-tests) FULL_TESTS=1; shift ;; --skip-tests) SKIP_TESTS=1; shift ;; @@ -140,6 +145,23 @@ done ok "node deps installed (MCP SDK, zod; Kilo/OpenCode plugin SDKs for the guard plugin)" # --------------------------------------------------------------------------------------------- +# Backend CLIs: when the requested version is the one pinned in scripts/pins/ (exact versions and +# sha512 integrity of every package, verified by `npm ci`), install from that lockfile; otherwise fall +# back to `npm install -g` of the exact top-level version (transitive packages then not integrity-pinned). +pinned_install() { # + local pin="$SRC/scripts/pins/$1" pkg="$2" ver="$3" dir="$4" exe="$5" + if [ -f "$pin/package-lock.json" ] && [ "$("$NODE" -p "require('$pin/package.json').dependencies['$pkg']")" = "$ver" ]; then + rm -rf "$dir"; mkdir -p "$dir/bin" + cp "$pin/package.json" "$pin/package-lock.json" "$dir/" + (cd "$dir" && "$NPM" ci --no-audit --no-fund --loglevel=error >/dev/null) || return 1 + ln -sfn "../node_modules/$pkg/bin/$exe" "$dir/bin/$exe" + ok "installed $pkg@$ver into $dir from the committed lockfile (integrity-checked)" + else + "$NPM" install -g --prefix "$dir" "$pkg@$ver" --no-audit --no-fund --loglevel=error >/dev/null || return 1 + warn "installed $pkg@$ver into $dir (no committed lockfile for this version: only the top-level version is pinned)" + fi +} + step "3/8 Kilo CLI $KILO_VERSION" kilo_version() { HOME="$(mktemp -d)" PATH="$NODE_DIR:/usr/bin:/bin" "$1" --version 2>/dev/null | tail -n1 | awk '{print $NF}'; } if [ -n "$KILO_BIN_ARG" ]; then @@ -151,8 +173,7 @@ else if [ -x "$KILO_BIN" ] && [ "$(kilo_version "$KILO_BIN")" = "$KILO_VERSION" ]; then ok "already installed at $KILO_BIN" else - "$NPM" install -g --prefix "$PREFIX/kilo-cli" "@kilocode/cli@$KILO_VERSION" --no-audit --no-fund --loglevel=error >/dev/null || die "installing @kilocode/cli@$KILO_VERSION failed" - ok "installed @kilocode/cli@$KILO_VERSION into $PREFIX/kilo-cli" + pinned_install kilo-cli @kilocode/cli "$KILO_VERSION" "$PREFIX/kilo-cli" kilo || die "installing @kilocode/cli@$KILO_VERSION failed" fi fi GOT="$(kilo_version "$KILO_BIN")" @@ -165,7 +186,7 @@ if [ "$WITH_OPENCODE" = 1 ]; then else OPENCODE_BIN="$PREFIX/opencode-cli/bin/opencode" if ! { [ -x "$OPENCODE_BIN" ] && [ "$(kilo_version "$OPENCODE_BIN")" = "$OPENCODE_VERSION" ]; }; then - "$NPM" install -g --prefix "$PREFIX/opencode-cli" "opencode-ai@$OPENCODE_VERSION" --no-audit --no-fund --loglevel=error >/dev/null || die "installing opencode-ai@$OPENCODE_VERSION failed" + pinned_install opencode-cli opencode-ai "$OPENCODE_VERSION" "$PREFIX/opencode-cli" opencode || die "installing opencode-ai@$OPENCODE_VERSION failed" fi fi OGOT="$(kilo_version "$OPENCODE_BIN")" @@ -215,6 +236,25 @@ if [ ! -f "$CFG/repos.json" ]; then else ok "repos.json kept (existing)" fi +if [ -n "$TOKEN_SAVERS$RTK_BIN_ARG" ]; then + TS_ARGS=() + IFS=',' read -r -a TS_PAIRS <<< "$TOKEN_SAVERS" + for kv in "${TS_PAIRS[@]}"; do + [ -n "$kv" ] || continue + case "$kv" in + terse=*) TS_ARGS+=(--terse "${kv#terse=}") ;; + minimal_code=*|minimal-code=*) TS_ARGS+=(--minimal-code "${kv#*=}") ;; + rtk=*) TS_ARGS+=(--rtk "${kv#rtk=}") ;; + *) die "--token-savers: unknown entry '$kv' (use terse=, minimal_code=, rtk=)" ;; + esac + done + if [ -n "$RTK_BIN_ARG" ]; then + RTK_BIN_ARG="$(readlink -f "$RTK_BIN_ARG")"; [ -x "$RTK_BIN_ARG" ] || die "--rtk-bin $RTK_BIN_ARG is not executable" + TS_ARGS+=(--rtk on --rtk-bin "$RTK_BIN_ARG") + fi + WH_CONFIG_DIR="$CFG" "$NODE" "$PREFIX/bin/workhorse" token-savers "${TS_ARGS[@]}" >/dev/null || die "setting token savers failed" + ok "token savers: ${TOKEN_SAVERS:-} ${RTK_BIN_ARG:+rtk=$RTK_BIN_ARG}" +fi DATA_DIR="$("$NODE" "$PREFIX/scripts/render-config.mjs" get "$CFG/daemon.json" data_dir)" cat > "$PREFIX/.install-info" < Date: Sun, 27 Sep 2026 00:50:14 +0330 Subject: [PATCH 4/7] docs: token savings, v0.3 configuration, handoff, security; skill prefers wait_task docs/token-savings.md (features, RTK/Headroom/Caveman/Ponytail evaluation, benchmarks, usage estimate formula), configuration/handoff/architecture/security/troubleshooting updates, README, delegation skill draft, NOTICE attributions, CHANGELOG 0.3.0 (Unreleased), version 0.3.0. --- CHANGELOG.md | 53 +++++ NOTICE | 13 ++ README.md | 45 +++- docs/architecture.md | 31 ++- docs/configuration.md | 51 ++++- docs/handoff.md | 41 +++- docs/security.md | 17 +- docs/token-savings.md | 237 ++++++++++++++++++++ docs/troubleshooting.md | 26 ++- grok-template/workhorse-delegation/SKILL.md | 63 ++++-- package-lock.json | 4 +- package.json | 7 +- 12 files changed, 539 insertions(+), 49 deletions(-) create mode 100644 docs/token-savings.md diff --git a/CHANGELOG.md b/CHANGELOG.md index da2ae15..dc3df5e 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,59 @@ All notable changes to this project are documented here. The format follows ## [Unreleased] +## [0.3.0] - Unreleased + +Token savings. Workhorse exists to reduce supervisor (for example Grok) usage by letting smaller, +user-chosen models do bounded work; this release cuts what the supervisor reads and how often it has to +act, and adds opt-in savers for the workers. Builds on 0.2.0. See [docs/token-savings.md](docs/token-savings.md). + +### Added +- **`wait_task`** (MCP, RPC, `workhorse wait`): long-poll one task (`task_id`) or several (`task_ids`, + `mode: "any" | "all"`) until they finish or park, up to `max_wait_s` (default 45, cap 55 s to stay + under common 60 s MCP request timeouts). Returns the brief result. +- **`view: "brief"`** for `task_result` and `wait_task` (~0.5-1 KB: verdict, short summary, files, + tests, concerns, `next` with the suggested call, review, automatic trail, tokens). `full` stays the + default for `task_result`. +- **`delegate_tasks`**: up to 10 tasks in one call, with per-entry errors. +- **Presets and size routing** in `profiles.json`: `presets` (profile, size, mode, timeout, test + command, standing instructions, follow-up flags) and `routing` (`small`/`medium`/`large` -> profile). + `list_models` shows them. +- **Automatic follow-ups** (off by default; per call, per preset or `profiles.json` `auto`): + `auto_fix_rounds` (same-session fix rounds after failing tests or a missing/partial RESULT, max 3), + `escalate` along each profile's new `escalate_to` chain (fresh session, same worktree), hard caps + `max_auto_runs` (default 3, cap 6), `max_tokens`, `max_cost_usd`. The trail is in `result.auto`. +- **`auto_review`**: an advisory read-only review on a (cheap) profile after a successful run; new + transient status `reviewing`. `request_changes` makes the handoff `needs_review`. +- **`usage_report`** (MCP, RPC) and **`workhorse stats`**: worker tokens and estimated list cost by + profile and day, plus a clearly labelled ESTIMATE of supervisor tokens avoided (formula documented; + optional `daemon.json supervisor.price_per_mtok` for USD). The daemon now records per-run tokens and + cost, and the characters each supervisor call sent and received. +- **Worker token savers** (`token_savers` in daemon.json, per-profile override; all off by default): + `terse` and `minimal_code` instruction fragments (`lite`/`full`, our own wording inspired by Caveman + and Ponytail) and `rtk` (Kilo/OpenCode: the guard plugin rewrites worker bash commands through + `rtk rewrite`). `workhorse token-savers`, installer `--token-savers` / `--rtk-bin`. The daemon's own + test run never goes through them. +- **`approvals.require_operator`**: MCP `approve_task` only records an approval request; a human + confirms with `sudo workhorse approve `, which sends a separate operator token (the daemon stores + its SHA-256). `workhorse operator-token init [--enable]`. +- **Per-profile `stall_minutes`.** +- **Audit log rotation** (`audit.max_mb`, `audit.keep`). +- **Test-only stub backend** (`adapters/stub`), registered only when the daemon runs with + `WH_ENABLE_STUB_BACKEND=1` and limited to its bundled script, so the approval, continue, retry, + fallback, restart, handoff, auto-fix, escalation and review flows run end to end in GitHub Actions + (`npm run test:stub`). `workhorse health` warns if a daemon runs with the flag. + +### Changed +- The MCP shim returns compact JSON (9-18% fewer characters on typical responses). +- The full `task_result` no longer repeats top-level fields inside `handoff.context` (`get_handoff` + still returns the complete record). +- `delegate_task`'s `next_step` and the MCP instructions point to `wait_task`; the delegation skill draft + prefers `wait_task`, the brief view, presets/size, batches and automatic follow-ups. +- The installer installs Kilo CLI and OpenCode from committed lockfiles (`scripts/pins/`, `npm ci`, + integrity-checked) when the pinned version is requested, and falls back to `npm install -g` with a + warning otherwise. +- CI runs the unit tests, then the stub end-to-end suite. + ## [0.2.0] - Unreleased Task handoff records and a human-approval state, based on community feedback. A supervisor should keep a diff --git a/NOTICE b/NOTICE index 4025379..b82b14d 100644 --- a/NOTICE +++ b/NOTICE @@ -13,7 +13,20 @@ This product includes third-party material: adapters/kilo/config/skills/SUPERPOWERS-LICENSE. Files are unmodified except that skill-authoring fixtures were removed (see adapters/kilo/config/skills/SOURCES.md). +Ideas credited (no code or text copied; our own wording, see docs/token-savings.md): + +2. The `terse` worker-instruction fragment (lib/savers.mjs) is inspired by Caveman, + https://github.com/JuliusBrussee/caveman (MIT License, (c) 2026 Julius Brussee, except its + engine-linked directories such as engine/ and proxy/, which are under BSL-1.1 and are not used). +3. The `minimal_code` worker-instruction fragment (lib/savers.mjs) is inspired by Ponytail, + https://github.com/DietrichGebert/ponytail (MIT License, (c) 2026 DietrichGebert), and by its + adaptation in 9router, https://github.com/decolua/9router (MIT License, (c) 2024-2026 decolua and contributors). +4. The optional RTK integration calls the RTK binary, https://github.com/rtk-ai/rtk (Apache License + 2.0), which is not bundled and must be installed separately. The guard-plugin code that calls + `rtk rewrite` is our own. + Not included, installed separately at install time (each under its own license): - Kilo CLI (@kilocode/cli) and @kilocode/plugin, by Kilo Code - OpenCode (opencode-ai) and @opencode-ai/plugin, by the OpenCode authors (optional) - @modelcontextprotocol/sdk and zod (npm dependencies) + - RTK (rtk), optional, only when token_savers.rtk is enabled diff --git a/README.md b/README.md index f41e6e2..763a1ee 100644 --- a/README.md +++ b/README.md @@ -4,7 +4,7 @@ **Grok Workhorse lets a supervising AI agent (for example Grok Bot) hand coding tasks to sandboxed coding-agent workers on your own Linux machine.** The supervisor calls a small MCP interface -(`delegate_task`, `task_status`, `task_result`, ...). A local daemon creates a fresh git worktree for +(`delegate_task`, `wait_task`, `task_result`, ...). A local daemon creates a fresh git worktree for each task and runs a coding-agent CLI (Kilo CLI or OpenCode today, with more adapters on the way) inside a bubblewrap sandbox, using any OpenAI-compatible model you configure. The daemon then runs your tests itself and returns a short structured result plus the diff. Workers never commit, merge or push: @@ -12,7 +12,12 @@ you (or your supervisor) review the branch and decide. - Repo: https://github.com/mrchatam/Grok-workhorse - License: MIT (vendored skills: MIT, see [NOTICE](NOTICE)) -- Status: v0.2.0 in development (latest release v0.1.0), Linux only +- Status: v0.3.0 in development (latest release v0.1.0), Linux only + +**Why:** to reduce supervisor (for example Grok) usage. The expensive model plans and reviews; smaller +models you choose do the bounded coding work, and v0.3 keeps what the supervisor reads and does per task +small (one `wait_task` call, a ~0.5-1 KB brief result, automatic fix rounds and reviews on cheap +models). See [docs/token-savings.md](docs/token-savings.md). ## Features @@ -24,7 +29,14 @@ you (or your supervisor) review the branch and decide. - **Structured results**: verdict (`success`, `tests_failed`, `no_changes`, `blocked`, `integrity_violation`, ...), diffstat, daemon-run test results, the worker's self-report, concerns, token usage and timings. Raw logs are paged on demand. - **Follow-ups and reviews**: `continue_task` resumes the same session and worktree. `mode: "review"` runs a read-only reviewer on another task's diff. - **Handoff records and human approval**: every finished task says who acts next (`owner`), the one exact `next_action`, which checks failed, and how to resume. A worker that needs a human decision parks the task as `needs_approval` (worktree kept) until someone answers with `approve_task`. Supervisors record their own handoffs with `update_handoff`. See [docs/handoff.md](docs/handoff.md). -- **Operations**: stall detection (15 min by default), wall-clock timeouts, cancel, recovery after a daemon restart, retention sweeps, an append-only JSONL audit log, and `workhorse health` for daily checks. +- **Token savings (v0.3)**: `wait_task` long-poll (one call instead of a polling loop), compact JSON + and a `brief` result view, `delegate_tasks` batches, presets and size routing, automatic fix rounds and + cheap-to-strong escalation with hard caps, an optional cheap advisory review, `usage_report` / + `workhorse stats` with a labelled estimate of supervisor tokens avoided, and opt-in worker savers + (terse output, minimal-code bias, RTK for shell output). See [docs/token-savings.md](docs/token-savings.md). +- **Operator-confirmed approvals** (optional): with `approvals.require_operator`, the supervisor's + approval only records a request and a human confirms it on the host with a separate operator token. +- **Operations**: stall detection (15 min by default, per profile with `stall_minutes`), wall-clock timeouts, cancel, recovery after a daemon restart, retention sweeps, an append-only JSONL audit log (rotated by size), and `workhorse health` for daily checks. - **Credentials from the environment first** (daemon env, or the MCP connector env passed through the shim), with an optional secret-store fallback. Keys never appear in argv, logs or results. ## Architecture @@ -86,7 +98,7 @@ The installer is idempotent, so you can re-run it to upgrade. It copies the app (root-owned), pins the Kilo CLI, checks bubblewrap, writes and locks the config, creates a hello-world repo, installs the `workhorse` and `workhorse-mcp` commands, runs the tests, and finishes with a live hello task if the key is available. Useful options: `--with-opencode`, `--systemd`, -`--secret-store PATH`, `--prefix`, `--data-dir`, `--full-tests`. Run `bash scripts/install.sh --help` +`--secret-store PATH`, `--prefix`, `--data-dir`, `--full-tests`, `--token-savers LIST` / `--rtk-bin PATH` (opt-in worker token savers). Run `bash scripts/install.sh --help` for the full list. Then: @@ -114,9 +126,9 @@ Config lives in `/config/` and is root-owned once locked. To edit it, ru | File | What it holds | |---|---| -| `profiles.json` | providers (OpenAI-compatible `base_url` plus the *name* of the env var holding the key), models, profiles (`backend`, `model`, `fallback` list), `default_profile` | +| `profiles.json` | providers (OpenAI-compatible `base_url` plus the *name* of the env var holding the key), models, profiles (`backend`, `model`, `fallback` list, `escalate_to`, `stall_minutes`, `token_savers`), `default_profile`, `presets`, size `routing`, `auto` follow-ups | | `repos.json` | repo allowlist: local `path` or clone `url`, default branch, default/allowed test commands, `test_network`, `trust_project_config` | -| `daemon.json` | concurrency, timeouts (stall 15 min), retries, retention, backends (`bin`, pinned version), sandbox binds/env, `secret_store_path`, allowed clone hosts | +| `daemon.json` | concurrency, timeouts (stall 15 min), retries, retention, backends (`bin`, pinned version), sandbox binds/env, `secret_store_path`, allowed clone hosts, `token_savers`, `approvals`, `audit` rotation, `supervisor` estimate inputs | A profile that runs Kilo on NVIDIA and falls back to OpenCode on OpenRouter: @@ -137,7 +149,8 @@ A profile that runs Kilo on NVIDIA and falls back to OpenCode on OpenRouter: ``` The complete reference is in [docs/configuration.md](docs/configuration.md). Ready-made examples are in -[config/examples/](config/examples/). +[config/examples/](config/examples/); `profiles.tiered.json` shows a cheap -> mid -> strong setup with +presets, size routing and automatic follow-ups. **Credentials.** The daemon looks up each provider's `api_key_env` in this order: its own environment, the env the MCP shim was started with (offered to the daemon in memory only), and finally the optional @@ -166,9 +179,10 @@ tests only. See [docs/adapters.md](docs/adapters.md) for the adapter interface a - **getting-started**: walks a new user through choosing a provider and model, adding repos, storing the key securely, running the installer, registering the stdio connector and running the first task. -- **delegation**: how a supervisor should delegate. It reads `list_models` and `list_repos`, writes - self-contained task descriptions, polls, reviews the result and diff, follows `handoff.next_action`, - uses `continue_task` for fixes, relays approvals, and merges or cleans up. +- **delegation**: how a supervisor should delegate cheaply. It reads `list_models` and `list_repos`, + picks a preset or size, writes self-contained task descriptions, waits with `wait_task`, reviews the + brief result (and the diff when needed), follows `next` / `handoff.next_action`, uses automatic fix + rounds or `continue_task` for fixes, relays approvals, and merges or cleans up. Copy them into your Grok Bot skills if you want them. Nothing in this repo installs them automatically. @@ -181,7 +195,11 @@ workhorse backends adapters: status, installed, capabilitie workhorse tasks | logs | audit recent tasks, daemon log, audit log workhorse attention tasks that need someone (parked or handoff not done) workhorse handoff [--owner O --next "..." --note "..." --state S] show or update a handoff -workhorse approve | reject answer a parked (needs_approval) task +workhorse approve | reject answer a parked (needs_approval) task (sends the operator token if required) +workhorse wait ... [--all] [--max S] [--full] long-poll until tasks finish or park +workhorse stats [--days N] [--profile P] [--json] worker usage by profile/day + supervisor ESTIMATE +workhorse token-savers [...] | off show or set opt-in worker token savers +workhorse operator-token init [--enable] create the operator token for require_operator workhorse cleanup-old [--days N] retention sweep now workhorse repos | add-repo | remove-repo | validate | check-provider | hello ``` @@ -191,6 +209,7 @@ workhorse repos | add-repo | remove-repo | validate | check-provider | hello ```bash npm run setup # npm ci for the app and the backend config dirs npm run test:unit # fast, no CLI or network needed (runs in CI) +npm run test:stub # end-to-end flows with the test-only stub backend: git + python3 only (runs in CI) npm test # full suite: needs the Kilo CLI, bwrap, git, python3 (mock LLM), ~10 min WH_TEST_BACKEND=opencode WH_OPENCODE_BIN=$(command -v opencode) npm test # same suite on OpenCode WH_LIVE_TEST=1 NVIDIA_API_KEY=... npm run test:live # one real task @@ -228,4 +247,6 @@ or host. The worker skills in `adapters/kilo/config/skills/` are vendored (unmodified, some files removed) from [obra/superpowers](https://github.com/obra/superpowers) (MIT, Jesse Vincent). Kilo CLI and OpenCode are -projects of their respective authors. See [NOTICE](NOTICE). +projects of their respective authors. The `terse` and `minimal_code` token-saver fragments are our own +wording, inspired by Caveman and Ponytail (both MIT); the optional RTK integration calls the separately +installed RTK binary (Apache-2.0). See [NOTICE](NOTICE). diff --git a/docs/architecture.md b/docs/architecture.md index e3aa95b..3d91494 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -16,9 +16,9 @@ sequenceDiagram D->>D: normalize events: stats, activity log, audit W-->>D: exit D->>D: run tests in test sandbox, collect diff, integrity checks, verdict - S->>M: task_status / task_result / task_details - M->>D: RPC - D-->>S: structured result + S->>M: wait_task (long-poll, ≤55 s per call) + M->>D: RPC (held until the task settles or the wait ends) + D-->>S: brief result (task_result / task_details for more) ``` ## Components @@ -34,13 +34,19 @@ sequenceDiagram | `lib/git.mjs` | Clones (allowlisted hosts only), worktrees, diffs, main-clone fingerprint. | | `lib/config.mjs` | Config loading, defaults, auto-detection of backend binaries, bwrap and toolchain dirs. | | `lib/credentials.mjs` | Optional JSON secret-store fallback. | -| `lib/audit.mjs` | Append-only JSONL audit log (values redacted). | +| `lib/audit.mjs` | Append-only JSONL audit log (values redacted, rotated by size). | +| `lib/views.mjs` | `brief` result view and handoff deduplication for the full view. | +| `lib/usage.mjs` | `usage_report` / `workhorse stats`: per-run tokens by profile and day, supervisor ESTIMATE. | +| `lib/savers.mjs` | Opt-in worker token savers (instruction fragments, RTK settings). | +| `lib/operator.mjs` | Operator token for `approvals.require_operator` (hash check, token file). | +| `adapters/stub/` | TEST-ONLY scripted backend for CI, registered only with `WH_ENABLE_STUB_BACKEND=1`. | | `adapters/` | One adapter per coding-agent CLI plus the shared helpers; see [adapters.md](adapters.md). | | `adapters/kilo/config/` | Kilo config shipped with the app: permissions, agents (`worker`, `review`, `explore`), guard plugin, worker contract (`AGENTS.md`), vendored skills. | ## Task lifecycle -`queued` → `running` → (`retry_wait` → `queued` …) → `testing` → `finalizing` → one terminal state: +`queued` → `running` → (`retry_wait` → `queued` …) → `testing` → `finalizing` → (automatic fix or +escalation run → `queued` …) → (`reviewing` while an auto-review child runs) → one terminal state: `completed`, `failed`, `timeout`, `stalled`, `cancelled` or `interrupted`, or the parked state `needs_approval` when the worker asked for a human decision (see below). @@ -75,6 +81,17 @@ completed/failed/... ──update_handoff state=needs_approval──▶ needs_ap Details, field reference and the verdict-to-handoff table: [handoff.md](handoff.md). +## Automatic follow-ups (v0.3) + +At the end of `finalize`, `planAuto` decides whether the daemon itself queues another run instead of +settling: a **fix** round (same session) when the verdict is in `auto.fix_on` and fix rounds remain, +otherwise an **escalation** run on the profile's `escalate_to` (fresh session, same worktree) when the +verdict is in `auto.escalate_on`. Caps: `max_auto_runs`, `max_tokens`, `max_cost_usd`. Each decision is +appended to `t.auto_trail` and copied to `result.auto`. After a successful final run, `auto_review` +starts a read-only review child task (`auto_review_of` = parent); the parent stays `reviewing` until +the child finishes, then gets `result.review` and an updated handoff. Recovery after a restart finishes +a parent whose review child already settled. + ## Data layout (`data_dir`) ``` @@ -83,7 +100,7 @@ worktrees/// one worktree per task (branch workhorse/) tasks// task.json (incl. result + handoff), activity.log, run-N.events.jsonl, run-N.stderr.log, diff.patch, test.log backend-data// per-task agent data (session DB, snapshots, per-task settings) kilo-home/, opencode-home/ backend HOMEs (config/code dirs root-owned after lock-config) -logs/ daemon.log, audit.jsonl (one JSON object per line: ts, kind, then event fields) +logs/ daemon.log, audit.jsonl (+ audit.jsonl.1 … after rotation; one JSON object per line: ts, kind, then event fields) run/ socket, token, pid files (0700) ``` @@ -95,7 +112,7 @@ carry `task_id`, `event` (`created`, `run_started`, `run_exited`, `finished`, `p `handoff_updated`, `approval`, `closed`, `cleanup`, …) and the task's `status`. Those fields are reserved: if event data uses one of them, the reserved value wins and the data value is kept as `data_`. `run_started` records the run kind (`initial`, `continue`, `retry`, `fallback`) as -`run_kind`. +`run_kind` (also `auto_fix` and `escalate` since v0.3). ## Retries and fallback diff --git a/docs/configuration.md b/docs/configuration.md index 91289fe..2bdfbdd 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -24,9 +24,24 @@ sudo bash /scripts/lock-config.sh "explore_model": "/", // optional cheaper model for Kilo/OpenCode's explore subagent "fallback": ["", "..."], // tried in order after repeated 429/5xx; [] = none "enabled": true, "disabled_reason": "...", - "price_per_mtok": { "input": 0.5, "output": 2.5 }, "price_note": "estimate only" + "price_per_mtok": { "input": 0.5, "output": 2.5 }, "price_note": "estimate only", + "escalate_to": "", // next step of the automatic escalation chain (v0.3) + "stall_minutes": 10, // per-profile stall timeout; else daemon.json timeouts.stall_min + "token_savers": { "terse": "lite" } // per-profile override of daemon.json token_savers } }, + "presets": { // named task templates shown in list_models (v0.3) + "quick-fix": { "description": "small bounded fix", "profile": "cheap", "size": "small", + "timeout_minutes": 10, "test_command": "npm test", "instructions": "Keep the diff minimal.", + "auto_fix_rounds": 1, "escalate": true, "auto_review": false } + }, + "routing": { "small": "cheap", "medium": "mid", "large": "strong" }, // size -> profile + "auto": { // automatic follow-ups, off unless enabled here or per call + "fix_rounds": 0, "fix_on": ["tests_failed", "no_result_block", "partial"], + "escalate": false, "escalate_on": ["tests_failed", "worker_error", "partial"], + "max_auto_runs": 3, "max_tokens": null, "max_cost_usd": null, + "review": { "enabled": false, "profile": null, "on": ["success", "success_untested"], "timeout_minutes": 10 } + }, "providers": { "": { "name": "display name", @@ -47,6 +62,30 @@ sudo bash /scripts/lock-config.sh } ``` +### Presets, size routing and automatic follow-ups (v0.3) + +These exist to keep supervisor usage low: the supervisor names a preset or a size, and the daemon does +the routine follow-up work on cheap models. + +- **Profile choice** for a new task: explicit `profile` > `preset.profile` > `routing[size]` (size from + the call or the preset) > `default_profile`. A preset may set any of `description`, `profile`, `size`, + `mode`, `timeout_minutes`, `test_command`, `instructions` (appended to the task as standing + instructions), `auto_fix_rounds`, `escalate`, `auto_review`. Call parameters win over the preset. +- **`auto_fix_rounds`** (0..3, hard cap 3): when an implement run ends with a verdict in `auto.fix_on`, + the daemon sends the same session a fix message with the failing tests/concerns, before settling. +- **`escalate`** (bool): when fix rounds are used up (or none were set) and the verdict is in + `auto.escalate_on`, the task moves to the profile's `escalate_to` (fresh session, same worktree, + existing changes kept). Chains are followed one step per run, cycles are refused by `workhorse validate`. +- **Hard caps**: `max_auto_runs` (default 3, cap 6) automatic runs per task, plus optional + `max_tokens` (input+output+reasoning of all runs) and `max_cost_usd` (needs `price_per_mtok`). The + first cap reached stops the chain; the result records `auto.stopped_reason`. +- **Escalation trail**: every automatic step is in `result.auto.trail` (`profile`, `verdict`, `next`, + tokens) and in the brief view as `"cheap:tests_failed->fix"`, `"cheap:tests_failed->mid"`, ... +- **`auto_review`** (`true` or a profile name; default profile `auto.review.profile`, else the default + profile): after a verdict in `auto.review.on`, a read-only review task runs on the diff. Its verdict is + **advisory**: `result.review` and, for `request_changes`, handoff state `needs_review` with a suggested + `continue_task`. It never changes the daemon's own verdict and never triggers a fix round by itself. + Examples: [`config/examples/profiles.nvidia.json`](../config/examples/profiles.nvidia.json), [`profiles.openrouter.json`](../config/examples/profiles.openrouter.json) and [`profiles.custom.json`](../config/examples/profiles.custom.json) (vLLM, LiteLLM, Ollama, …). @@ -74,7 +113,7 @@ Every key is optional. Defaults are in `lib/config.mjs`. |---|---|---| | `data_dir` | `~/.local/share/grok-workhorse` | Clones, worktrees, tasks, logs, socket (`WH_DATA_DIR` overrides) | | `max_concurrent` | 2 | Tasks running at once | -| `timeouts` | default 30, min 1, max 120, **stall 15**, test 10 (minutes), kill_grace_sec 10 | | +| `timeouts` | default 30, min 1, max 120, **stall 15**, test 10 (minutes), kill_grace_sec 10 | A profile's `stall_minutes` overrides the stall timeout for its runs | | `retry` | 2 retries, back-off [30, 120] s, provider cooldown 60 s | For retryable model API errors | | `retention` | enabled, worktree_days 7, task_days 30, parked_days null, sweep every 60 min | Automatic cleanup. Parked (`needs_approval`) tasks are skipped unless `parked_days` is set. When it is set, their worktree is removed that many days after the last handoff update and the task is closed ([handoff.md](handoff.md)) | | `default_backend` | `kilo` | Backend for profiles without `backend` | @@ -89,6 +128,11 @@ Every key is optional. Defaults are in `lib/config.mjs`. | `test_sandbox` | enabled, `ro_binds` [] , `env` {} | | | `worker_sandbox` | enabled, `hide` [...], `ro_binds` [], `auto_bind_toolchain` true, `env` {} | Outer sandbox around worker runs | +| `token_savers` | all off | Opt-in worker token savers: `terse` and `minimal_code` (`off`/`lite`/`full`), `rtk` (`{enabled, bin}`). Profiles may override. See [token-savings.md](token-savings.md) | +| `approvals` | `require_operator` false | `require_operator: true` makes MCP `approve_task` only record the request; a human confirms with `workhorse approve` and the operator token (`operator_token_sha256` holds its hash, `operator_token_path` the default token file location). Set up with `sudo workhorse operator-token init --enable` | +| `audit` | `max_mb` 20, `keep` 5 | Rotation of `logs/audit.jsonl` (`audit.jsonl.1` ... `.keep`) | +| `supervisor` | `price_per_mtok` null, `chars_per_token` 4 | Inputs of the `usage_report` / `workhorse stats` supervisor-token ESTIMATE ([token-savings.md](token-savings.md#usage-report-and-the-estimate)) | + The legacy keys `kilo` (maps to `backends.kilo` and `bwrap`) and `kilo_sandbox` (maps to `worker_sandbox`) are still accepted. @@ -123,4 +167,5 @@ Only the keys required by the chosen profile's provider are passed to that worke | `WH_CONFIG_DIR`, `WH_DATA_DIR`, `WH_APP_DIR` | all: override locations | | `WH_KILO_BIN`, `WH_OPENCODE_BIN`, `WH_CLAUDE_CODE_BIN`, `WH_CODEX_BIN` | backend binary auto-detection | | `WH_ALLOW_ROOT` | allow running the daemon as root (not recommended) | -| `WH_TEST_BACKEND`, `WH_LIVE_TEST`, `WH_LIVE_PROFILES`, `WH_LIVE_PROFILE`, `WH_SKIP_INTEGRATION` | tests | +| `WH_TEST_BACKEND`, `WH_LIVE_TEST`, `WH_LIVE_PROFILES`, `WH_LIVE_PROFILE`, `WH_SKIP_INTEGRATION`, `WH_SKIP_STUB` | tests | +| `WH_ENABLE_STUB_BACKEND` | **tests only**: registers the scripted `stub` backend in the daemon. Never set it in production (see [security.md](security.md)) | diff --git a/docs/handoff.md b/docs/handoff.md index c1a92cd..a4f6278 100644 --- a/docs/handoff.md +++ b/docs/handoff.md @@ -15,7 +15,9 @@ asked to write it. When the worker reports `status: blocked`, `needs_approval` o | Where | What | |---|---| -| `task_result` | the full record under `handoff` | +| `task_result` (full view, default) | the record under `handoff`; since v0.3 its `context` no longer repeats fields that are already top-level in the same response (verdict, status, summary, remaining_concerns, diffstat, files_changed, diff_path, diff) | +| `task_result` with `view: "brief"`, `wait_task` | `next`: state, owner, action (first 500 chars) and the suggested `tool` + `args` | +| `get_handoff` (RPC) / `workhorse handoff ` | the complete record, including the full `context` | | `task_status` | `handoff`: state, owner, next_action (first 300 chars), names of failed checks, updated_at/by | | `list_tasks` | `handoff`: state, owner, next_action per task; filters `status: "needs_attention"`, `status: "parked"`, `owner` | | `task.json` | persisted with the task, so it survives daemon restarts | @@ -145,9 +147,44 @@ grant. The guard and bwrap rules still apply, so a worker still cannot reach the packages. If the approved action needs that (for example adding a dependency to an offline cache), the human does it outside the sandbox first and then approves. The `next_action` text says so. +## Operator confirmation (`approvals.require_operator`, v0.3) + +By default the supervisor's `approve_task` is final. An owner who wants a human in the loop for every +approval sets `daemon.json` `approvals.require_operator: true` (easiest: `sudo workhorse operator-token +init --enable`, which writes a random token to a root-only file and stores only its SHA-256 in +daemon.json). Then: + +1. MCP `approve_task` (from the supervisor) does not resume anything. It records an + **approval request** (`decision`, `instructions`, `by`, time) on the task, returns + `approval_requested: true` (handoff owner `human (operator)`), and the brief/full result shows + `approval_request`. Rejections do not need the operator: they only stop or redirect work. +2. A human runs `sudo workhorse approve ` (or `reject`). The CLI reads the token file and sends + the token; the daemon checks it against the hash (constant-time) and applies the recorded request's + instructions unless the CLI passes its own. +3. `continue_task` and `update_handoff` cannot unpark a parked task without the token either, so the + supervisor cannot route around the gate. The token is never an MCP parameter, is masked in the + audit log, and never appears in results. + +## Automatic follow-ups, reviews and handoff (v0.3) + +With `auto_fix_rounds`, `escalate` or `auto_review` (per call, per preset or in `profiles.json` +`auto`), the daemon runs routine follow-ups itself before it settles the handoff: + +- A fix round or escalation step happens instead of a `needs_fix` / `retryable` handoff to the + supervisor. Only the final run's handoff is reported; `result.auto.trail` lists every step + (`profile`, `verdict`, `next`) and the handoff's `next_action` gets a one-line note of the trail. If a + cap stops the chain, `auto.stopped_reason` says which one. +- An auto-review runs as a separate read-only task after a successful run; the parent shows status + `reviewing` meanwhile (not terminal; `wait_task` keeps waiting). Its verdict is advisory: + `approve` leaves the handoff `done` (with a note); `request_changes` turns it into `needs_review` + with a failed check `auto_review`, the findings, and a suggested `continue_task`. The daemon's own + verdict is never changed by a review. +- `continue_task` on a task resets its automatic trail for the new round, and clears a pending + approval request. + ## Supervisor loop (short) -1. After `task_result`, read `handoff.state`, `owner` and `next_action`. +1. After `wait_task` (or `task_result`), read `next` (brief) or `handoff.state`, `owner` and `next_action`. 2. If the owner is `supervisor`, do the next action (often the `resume` suggestion). If it is `human`, relay `next_action` to the user verbatim and wait for their decision, then call `approve_task`. 3. When you stop working on a task that is not done, record where it stands with `update_handoff` diff --git a/docs/security.md b/docs/security.md index ea72a9e..54bb2c4 100644 --- a/docs/security.md +++ b/docs/security.md @@ -52,7 +52,18 @@ 9. **Root-owned config.** `lock-config.sh` makes the config, `adapters/` and the code-bearing dirs of each backend HOME root-owned, so a process running as the service user cannot widen its own permissions. -10. **Audit.** Every RPC, task event and tool call is written as JSONL with values redacted. +10. **Audit.** Every RPC, task event and tool call is written as JSONL with values redacted. The log is + rotated by size (`audit.max_mb`, `audit.keep`); the operator token is masked as `[given]`. +11. **Test-only stub backend is gated.** `adapters/stub` (a scripted fake worker used by CI) is + registered only when the daemon's own environment has `WH_ENABLE_STUB_BACKEND=1`. Without it the + backend name is unknown, so a profile cannot select it, and even with it the adapter refuses any + binary other than the bundled `stub-cli.mjs`. The installer, systemd unit and supervisor never set + the flag, and `workhorse health` warns if a running daemon has it. +12. **Token savers stay outside the trust boundary.** `terse` / `minimal_code` are only extra text in + the worker's message. The optional RTK integration runs a local binary (read-only bind) that + rewrites the worker's own shell commands inside the sandbox; the guard checks the command before + and after rewriting, and the daemon's test run never goes through it. No proxy sees prompts or + keys (Headroom-style proxies are deliberately not integrated; see [token-savings.md](token-savings.md)). ## Known limitations @@ -73,8 +84,8 @@ - `approve_task` is a coordination signal, not a permission grant. It never widens the sandbox or the guard. The approver name (`by`) is recorded as given, and any client holding the daemon socket and token (the supervisor included) can approve. The recorded `source.channel` (`mcp` / `cli`) is - self-declared by the client for the same reason; `source.auth` says what was verified. Keep a human in the loop at the supervisor level when - approvals matter. + self-declared by the client for the same reason; `source.auth` says what was verified. See + `approvals.require_operator` below for putting a human in the loop. ## Recommendations diff --git a/docs/token-savings.md b/docs/token-savings.md new file mode 100644 index 0000000..038c646 --- /dev/null +++ b/docs/token-savings.md @@ -0,0 +1,237 @@ +# Token savings + +Grok Workhorse exists to **reduce supervisor usage**: a strong, expensive model (for example Grok as +the supervisor) plans and reviews, and smaller models you choose do the bounded coding work. v0.3 +adds features that cut the supervisor's share further and keep the total spend low. All of them are +backward compatible, and everything that changes worker behaviour is **opt-in**. + +| Where tokens go | Feature | Default | +|---|---|---| +| Supervisor: polling round trips | `wait_task` long-poll (single or many ids, `any`/`all`, up to 55 s per call) | available; the skill prefers it | +| Supervisor: reading results | compact JSON from the MCP shim, `view: "brief"` (~0.5-1 KB), deduplicated handoff in the full view | compact: on; brief: opt-in per call (the skill uses it), `wait_task` returns brief | +| Supervisor: fix/escalate/review turns | `auto_fix_rounds`, `escalate` (cheap -> mid -> strong via `escalate_to`), `auto_review` on a cheap profile | off | +| Supervisor: choosing and batching | presets, size routing, `delegate_tasks` (up to 10 per call) | available | +| Worker output | `token_savers.terse` (short prose), `token_savers.minimal_code` (smallest-diff bias) | off | +| Worker input (shell output) | `token_savers.rtk` (RTK rewrites the worker's shell commands) | off | +| Visibility | `usage_report` MCP tool, `workhorse stats` | available | + +## Supervisor-side savings + +### `wait_task` instead of polling + +`wait_task {task_id}` (or `{task_ids: [...], mode: "any" | "all"}`) blocks until the task settles +(finished or parked for approval) or `max_wait_s` passes (default 45, capped at **55 s**), then returns +`{done, waited_s, task}` with the brief result. If `done` is false, call it again. The cap keeps each +MCP call under the 60 s request timeout that many MCP clients use by default (the TypeScript SDK's +`DEFAULT_REQUEST_TIMEOUT_MSEC` is 60000), with margin for the round trip. A waiting call holds no +daemon resources except a timer. + +### Compact JSON and the brief view + +The MCP shim now returns compact JSON (no indentation; 9-18% fewer characters on typical responses, measured below). `task_result` +takes `view: "full"` (default, unchanged fields) or `view: "brief"`: verdict, a 300-char summary, up to +20 changed files, test outcome (failing test names only on failure), up to 5 concerns, `next` (handoff +state, owner, action and the suggested tool call), the advisory review, the automatic trail and token +totals. In the full view the handoff's `context` no longer repeats fields that are already top-level. +`get_handoff` (RPC) and `workhorse handoff ` still show the complete record. + +### Automatic follow-ups on cheap models + +Instead of the supervisor reading a failed result, writing a fix request and waiting again, the daemon +can do it (see [configuration.md](configuration.md#presets-size-routing-and-automatic-follow-ups-v03)): + +- `auto_fix_rounds` 1..3: fix rounds in the same session after failing tests or a missing/partial + RESULT block; +- `escalate: true`: when that is not enough, the next profile of the `escalate_to` chain (fresh + session, same worktree); +- `auto_review`: a read-only review on a cheap profile; its verdict is advisory. + +Hard caps: at most 3 fix rounds, `auto.max_auto_runs` (default 3, never more than 6) automatic runs +per task, optional `auto.max_tokens` and `auto.max_cost_usd`. The result's `auto.trail` shows every +step, for example `["cheap:tests_failed->fix", "cheap:tests_failed->mid", "mid:success"]`. + +### Presets, size routing, batches + +`presets` bundle profile, size, timeout, test command, standing instructions and follow-up flags under +a name, so a supervisor call can be as short as `{repo, task, preset: "quick-fix"}`. `routing` maps +`small | medium | large` to profiles, so the cheapest adequate model is picked by size. +`delegate_tasks` creates up to 10 tasks in one call and `wait_task` with `task_ids` waits for them. + +## Worker-side token savers (`token_savers`) + +Configured in `daemon.json` (defaults for all profiles) and overridable per profile in +`profiles.json`. Set them with the CLI (config must be unlocked): + +```bash +workhorse token-savers # show the effective settings +workhorse token-savers --terse lite --minimal-code lite # prompt savers +workhorse token-savers --rtk on --rtk-bin /usr/local/bin/rtk +workhorse token-savers off +# installer: --token-savers terse=lite,minimal_code=lite[,rtk] [--rtk-bin PATH] +``` + +```jsonc +// daemon.json +"token_savers": { "terse": "off", "minimal_code": "off", "rtk": { "enabled": false, "bin": null } } +// profiles.json, per profile: +"cheap": { "backend": "kilo", "model": "...", "token_savers": { "terse": "lite", "minimal_code": "lite" } } +``` + +- **`terse`** (`lite` | `full`): a short style instruction appended to the worker's first message of + a fresh session: no narration, no restating the task, no echoing tool output. Code, commands, paths, + numbers, error messages and negations are never shortened, and the `## RESULT` block keeps its exact + format (the daemon parses it). +- **`minimal_code`** (`lite` | `full`): a "smallest correct diff" bias: check whether the code is + needed, whether the standard library or an existing dependency already does it, reuse helpers, no + new dependencies or single-use abstractions, deletion over addition. It never drops validation at + trust boundaries, error handling, security checks, requested tests or requested features, and the + repository's conventions win. `full` also asks the worker to list what it deliberately skipped under + concerns. Not applied to review tasks. +- **`rtk`**: Kilo and OpenCode only. The guard plugin asks `rtk rewrite ` for a compact + equivalent of each bash command the worker runs (`git status` -> `rtk git status`) and uses it only + if the rewritten command passes the same guard checks. Multi-line commands are left alone. The rtk + binary's directory is bound read-only into the sandbox. Claude Code and Codex runs ignore it. + +Fragments go only into the first message of a fresh session (not into follow-ups, which already have +them in context). **The daemon's own test run is never routed through RTK or any other compression**: +its output feeds the verdict and the failing-test parser. + +### Recommended defaults + +- Leave everything off on your strongest profile. +- On cheap/mid profiles: `terse: "lite"` and `minimal_code: "lite"`. Use `full` only after checking + results on your own repos. +- `rtk`: optional; helps mostly when workers run many `git`, `ls`, `grep` or `find` commands through + bash. Most Kilo/OpenCode reads go through the built-in read/grep/list tools, which RTK does not touch. + +## The tools we evaluated + +Community feedback suggested four tools used by the 9router project +([decolua/9router](https://github.com/decolua/9router), MIT). We checked each project (September 2026): + +| Tool | Project, license | What it is | Decision | +|---|---|---|---| +| RTK | [rtk-ai/rtk](https://github.com/rtk-ai/rtk), Apache-2.0, very active (v0.50.0, ~80k stars) | Rust CLI that runs common commands (git, ls, grep, find, test runners, ...) and prints a compressed version of their output. `rtk rewrite ` returns an equivalent command (exit 0 allow, 1 no equivalent, 2 deny, 3 rewritten). Ships hooks for Claude Code and a plugin for OpenCode; telemetry is off unless you consent | **Kept, opt-in**, through our own guard-plugin code on Kilo/OpenCode (their plugin is not vendored). Not for Claude Code/Codex yet | +| Headroom | [headroomlabs-ai/headroom](https://github.com/headroomlabs-ai/headroom), Apache-2.0, active | Python library/proxy (`headroom proxy`, `/v1/compress`) that compresses prompts and tool output with heuristics and an ML model, with retrieval of the originals through an MCP tool | **Discarded** as a built-in: a proxy would see all code and the provider key; heavy Python/ML dependency; lossy compression of tool output; retrieval needs a tool our workers don't have. Advanced users can point a provider's `base_url` at their own proxy (unsupported) | +| Caveman | [JuliusBrussee/caveman](https://github.com/JuliusBrussee/caveman); MIT except engine-linked dirs (engine/, proxy/, ...) under BSL-1.1 | A prompt/skill that makes the model answer in terse "caveman" style to cut output tokens | **Kept as our own `terse` fragment**, written from scratch, with safeguards (code, paths, errors, negations and the RESULT block are never shortened). No code or text vendored; the BSL parts are never used | +| Ponytail | [DietrichGebert/ponytail](https://github.com/DietrichGebert/ponytail), MIT (9router's adaptation: `open-sse/rtk/ponytailPrompt.js`, MIT) | A "lazy senior dev" prompt: YAGNI, reuse the standard library, deletion over addition, minimal code | **Kept as our own `minimal_code` fragment**, written from scratch. We dropped its code-comment markers and "one runnable self-check" rule and added explicit never-drop rules (validation, error handling, security, requested tests/features) | + +Attribution is in [NOTICE](../NOTICE). + +Risks we designed around: + +- **Tool-output parsing.** RTK changes what the *worker* sees, never what the daemon parses: the daemon + runs tests itself, without RTK, and reads the diff from git. `rtk git diff` condenses diffs, so a + worker that needs the exact patch should read the files directly; this is one reason RTK stays off by + default. +- **Quality loss.** Prompt savers can make a model skip context or cut corners. That is why they are + off by default, the RESULT format is protected, and `minimal_code` has never-drop rules. The daemon's + verdict still comes from the tests. +- **Secrets.** No saver sends data anywhere. RTK runs inside the worker sandbox (no network for + model-run commands on Kilo) with telemetry off by default. + +## Benchmarks + +All numbers are small samples on this repository and its test fixtures. Token counts use the +`o200k_base` tokenizer as a proxy (the real supervisor and worker models use their own tokenizers), so +treat them as **estimates**. + +### Supervisor side (stub backend, v0.2 flow vs v0.3 flow) + +Measured with the test-only stub backend on the calc fixture. "v0.2" = `delegate_task`, `task_status` +polls, `task_result` (full view, pretty JSON, handoff with duplicated context). "v0.3" = +`delegate_task` + one `wait_task` (brief, compact JSON). + +| Response the supervisor reads | chars | tokens (o200k) | +|---|---|---| +| v0.2 `delegate_task` | 563 | 203 | +| v0.2 `task_status` while running / final | 634 / 935 | 224 / 314 | +| v0.2 `task_result` full, success | 4,318 | 1,427 | +| v0.2 `task_result` full, tests_failed | 7,150 | 2,182 | +| v0.3 `task_result` full (compact, deduped), success | 3,179 | 976 | +| v0.3 `delegate_task` (compact) | 510 | 162 | +| v0.3 `wait_task` with brief result, success | 671 | 198 | + +Per task (the number of polls is an assumption: 5 "still running" polls, i.e. about 5-10 minutes at +the old skill's 1-2 minute cadence): + +| Scenario | v0.2 | v0.3 | Change | +|---|---|---|---| +| Task succeeds first time | 8 tool calls, ~3,060 tokens read | 2 calls, ~360 tokens | about -88% tokens, -6 turns | +| Same with no polling at all (best case for v0.2) | 3 calls, ~1,940 tokens | 2 calls, ~360 tokens | about -81% | +| Tests fail once, then fixed (v0.2: supervisor sends `continue_task`; v0.3: `auto_fix_rounds: 1`) | ~16 calls, ~6,900 tokens (the `continue_task` response is estimated at ~200) | 2 calls, ~400 tokens | about -94%; the fix round costs worker tokens on the cheap profile instead | + +Tasks longer than 55 s need one extra `wait_task` call per 55 s (about 150 tokens each for a +`done: false` response), which is still cheaper than a status poll at the same cadence. Every avoided +tool call also avoids one supervisor turn, which re-reads the whole conversation (usually cached, but +not free); that effect is not included above. + +### RTK (worker shell output) + +RTK v0.50.0 on this repository, output tokens before and after the rewrite: + +| Command | before | after | change | +|---|---|---|---| +| `git status` | 125 | 30 | -76% | +| `git diff` | 208 | 180 | -13% | +| `git log -n 20` | 2,127 | 838 | -61% | +| `ls -la` | 359 | 146 | -59% | +| `grep -rn` | 1,569 | 781 | -50% | +| `find` | 287 | 192 | -33% | +| `python3 -m unittest -v` | not rewritten | | 0% | +| `cat file` (-> `rtk read`) | unchanged | | 0% | +| **total** | **7,127** | **4,619** | **-35%** | + +The project's "60-90%" claim holds for some commands (status, log) but not across this mix. The effect +on a real worker bill is smaller still, because Kilo/OpenCode workers read and search files mostly with +built-in tools, not bash. Hence: opt-in, not recommended by default. + +### Prompt savers (real model, small sample) + +Three A/B pairs on NVIDIA Nemotron 3 Ultra 550B-A55B (NIM) through Kilo, on the calc fixture. +"base" = no savers, "saver" = `terse: "lite"` + `minimal_code: "lite"`; each pair ran the same task +at the same time. Token counts are as reported by the provider. + +| Task | profile | verdict | input | output | diff | +|---|---|---|---|---|---| +| implement multiply/divide (run 1) | base / saver | tests_failed / tests_failed * | 54,670 / 55,736 | 1,159 / 996 | +4/-2 / +4/-2 | +| implement multiply/divide (run 2) | base / saver | tests_failed / tests_failed * | 36,688 / 50,481 | 958 / 860 | +4/-2 / +4/-2 | +| make the whole suite pass (several fixes) | base / saver | success / success | 116,987 / 84,809 | 2,968 / 2,704 | +30/-18 / +30/-18 | +| **total** | | | **208,345 / 191,026** | **5,085 / 4,560 (-10%)** | same | + +\* The fixture's default test command runs every exercise, including ones the task did not ask for; +the requested `test_core` tests passed in all four runs and the diffs are correct. + +Reading: output tokens went down about 10% (the "~65%" claimed for Caveman-style prompts applies to +chatty assistants; a coding worker's output is mostly tool calls and code, which the fragment +deliberately does not shorten). Input tokens vary far more with the number of turns than with the +fragments (which themselves add about 210 tokens to every request of the session: terse lite ~80, minimal_code lite ~130), so the input difference here is noise. +Diff sizes were identical on these small tasks; `minimal_code` should matter more on open-ended work, +which this sample did not measure. Three pairs are not statistically meaningful: treat these as +indicative only, and measure on your own repos with `workhorse stats` before enabling `full`. + +## Usage report and the estimate + +`usage_report` (MCP) and `workhorse stats [--days N] [--profile P] [--repo R] [--json]` group worker +tokens and estimated list cost by profile and by day (per run, so escalated tasks are split across +profiles). Tasks from before v0.3 count their totals on their last run. + +The report also contains `supervisor_estimate`, **labelled ESTIMATE**. Formula, per task whose final +verdict is `success` or `success_untested`: + +``` +worker_work_tokens = worker input + output + reasoning tokens (cache reads excluded) +supervisor_overhead = (chars the supervisor sent for this task + chars it received) / chars_per_token +est_supervisor_tokens_avoided = max(0, sum(worker_work_tokens) - sum(supervisor_overhead of all tasks)) +``` + +Failed tasks count only as overhead. Supervisor characters are recorded by the daemon for every MCP +call made through `workhorse-mcp` (the operator CLI is excluded). `chars_per_token` defaults to 4 +(`daemon.json supervisor.chars_per_token`). With `supervisor.price_per_mtok: {input, output}` it also +gives a USD figure: avoided input tokens x input price + avoided output tokens x output price - the +workers' estimated list cost. + +The assumption behind it: the supervisor would have needed about as many tokens as the worker to do +the same work itself. A stronger model may need fewer turns (the estimate is then too high); doing +the work itself would also grow the supervisor's context for the rest of the session (not counted, so +too low). It is a planning aid, not a measurement. diff --git a/docs/troubleshooting.md b/docs/troubleshooting.md index 9de0ada..b89e3aa 100644 --- a/docs/troubleshooting.md +++ b/docs/troubleshooting.md @@ -39,7 +39,31 @@ store. Put it in the connector env (recommended) or restart the daemon from an e `stalled` means no output from the agent for `timeouts.stall_min` (15 min by default), which is often a hung provider or a model stuck in a long reasoning step. `timeout` means the wall-clock limit was hit. Check `task_details kind=activity` and `stderr`, then use `continue_task` (it resumes the session) or -choose a faster profile. +choose a faster profile. A profile's `stall_minutes` overrides the stall timeout for that profile only +(for example shorter for a fast cheap model, longer for a slow reasoning model). + +## `wait_task` returns `done: false` + +Normal for tasks longer than `max_wait_s` (cap 55 s): call it again. If your MCP client uses a request +timeout shorter than 60 s, pass a smaller `max_wait_s`. + +## Automatic follow-ups did not run, or stopped early + +Check `result.auto` (`trail`, `stopped_reason`): `max_auto_runs`, `max_tokens` or `max_cost_usd` was +reached, the verdict was not in `auto.fix_on` / `auto.escalate_on`, or the profile has no `escalate_to`. +`max_cost_usd` needs `price_per_mtok` on the profiles. Automatic follow-ups apply to implement tasks only. + +## `approve_task` says "approval recorded as a request" + +`approvals.require_operator` is on. A human must run `sudo workhorse approve ` on the host +(it reads the operator token file). `operator token rejected` means the file does not match +`approvals.operator_token_sha256`: re-run `sudo workhorse operator-token init --force --enable` and restart. + +## RTK is enabled but commands are not rewritten + +`workhorse token-savers` shows whether the binary was found. RTK applies to Kilo and OpenCode only, +needs an absolute `rtk.bin` or `rtk` on the daemon's `env_path`, and leaves commands alone when it has +no equivalent (exit 1) or when the command spans several lines. Tests run by the daemon never use it. ## Verdict `blocked` or many blocked calls diff --git a/grok-template/workhorse-delegation/SKILL.md b/grok-template/workhorse-delegation/SKILL.md index cacc354..7a7cc4f 100644 --- a/grok-template/workhorse-delegation/SKILL.md +++ b/grok-template/workhorse-delegation/SKILL.md @@ -2,8 +2,8 @@ name: workhorse-delegation description: >- Use when a coding task could be delegated to Grok Workhorse (MCP tools such as delegate_task, - task_status, task_result): deciding whether to delegate, picking a profile, writing the task, - reviewing and accepting or rejecting the result, and fixing setup gaps. + wait_task, task_result): deciding whether to delegate, picking a preset or profile, writing the + task, waiting cheaply, reviewing and accepting or rejecting the result, and fixing setup gaps. --- # Delegating to Grok Workhorse (supervisor playbook) @@ -11,6 +11,10 @@ You are the supervisor. A Workhorse worker runs one bounded coding task in an is and returns a structured result. It never commits, merges or pushes. Planning, review and the final decision are yours. +**Why it exists: to save your (supervisor) tokens.** Every tool call you make and every byte you read +costs more than the worker's cheap model. So: delegate bounded work, wait with one `wait_task` call +instead of polling, read the `brief` view, and let the daemon do automatic fix rounds and cheap reviews. + ## 1. Delegate or not Delegate bounded, checkable work: a clearly specified function or feature, mechanical refactors, writing or fixing tests, a focused bug, or code exploration (`mode: "review"`, which edits nothing). @@ -18,10 +22,16 @@ Do it yourself, or ask the user first, when requirements are ambiguous, when the or security design (auth, crypto, secrets), when it is latency-sensitive (tasks take minutes), or when the repo is not in `list_repos`. Adding a repo is the owner's decision. -## 2. Profile -Call `list_models` once per session. Use `default_profile` unless the user asks for another profile -or a profile's description clearly fits better. Skip profiles with `available: false` and tell the user -their `unavailable_reason`. Don't invent model names. +## 2. Preset, size or profile +Call `list_models` once per session. Prefer, in this order: +- a **preset** whose description fits (`preset: "quick-fix"`); presets bundle profile, size, timeout, + test command, standing instructions and automatic follow-ups chosen by the owner; +- a **size** (`size: "small" | "medium" | "large"`) when `routing` is configured: the daemon picks the + cheapest profile the owner routed for that size; +- an explicit `profile` only when the user asks for one or its description clearly fits better; +- otherwise `default_profile`. +Skip profiles with `available: false` and tell the user their `unavailable_reason`. Don't invent model +names. ## 3. Write the task The worker sees nothing from this conversation, so the task text must stand on its own: @@ -30,22 +40,39 @@ The worker sees nothing from this conversation, so the task text must stand on i - `test_command` only if the repo default is wrong (it must match the allowed patterns in `list_repos`) - `timeout_minutes` sized to the task (default 30) -Split large work into independent tasks, and never put secrets in the task text. +Split large work into independent tasks, and never put secrets in the task text. Independent tasks +can go in **one `delegate_tasks` call** (up to 10). + +**Automatic follow-ups (let the daemon spend cheap tokens instead of yours):** +- `auto_fix_rounds: 1..3`: after failing tests or a missing/partial RESULT block, the daemon sends the + worker a fix round itself (same session) before reporting back. +- `escalate: true`: if the profile still fails, retry on the next profile of its `escalate_to` chain + (for example cheap -> mid -> strong). Hard caps on runs, tokens and cost apply; the result carries the + trail (`auto.trail`, e.g. `cheap:tests_failed->fix`, `cheap:tests_failed->mid`). +- `auto_review: true` (or a profile name): a cheap reviewer reads a successful diff and adds an + advisory `review` verdict. `request_changes` turns the handoff into `needs_review`; you decide. +Use them when the owner configured them or for well-tested repos; skip them for exploratory work. -## 4. Wait -`delegate_task` returns a `task_id` immediately. Check `task_status` every 1-2 minutes, not in a tight -loop. A few minutes with no turns usually means the provider is queueing. The daemon detects stalls -on its own. Stop polling when `terminal` is true. That includes `needs_approval`, which means the task -is parked and waiting for a human decision. +## 4. Wait (one call, not a polling loop) +`delegate_task` returns a `task_id` immediately. Then call **`wait_task`** with `task_id` (or +`task_ids` plus `mode: "any" | "all"`). It blocks up to `max_wait_s` (max 55 s, default 45) and returns +as soon as the task is finished or parked, with the brief result included. If `done` is false, call it +again. Don't poll `task_status` in a loop; use it only to look at progress. A few minutes with no turns +usually means the provider is queueing, and the daemon detects stalls on its own. A parked task +(`needs_approval`) also counts as done: it waits for a human decision. -## 5. Review (`task_result`, then `task_details` as needed) -- `integrity.commits_made` must be 0 and `main_clone_unchanged` must be true. If not, reject and tell the user. +## 5. Review (the brief view first, the full view or `task_details` only when needed) +`wait_task` and `task_result` with `view: "brief"` return about 0.5-1 KB: `verdict`, `summary`, +`files`, `tests`, `concerns`, `next` (state, owner, action and the suggested tool call), `review` and +`auto`. That is enough to accept a green, small change. Ask for `task_result` (full, the default) or +`task_details` only when you need integrity details, the whole handoff, logs or the diff. +- In the full view, `integrity.commits_made` must be 0 and `main_clone_unchanged` must be true. If not, reject and tell the user. - Trust the daemon-run `test_results` over the worker's self-reported tests. - For non-trivial changes, read the diff (`task_details` kind `diff`). Look for scope creep, deleted or weakened tests, hard-coded outputs, new dependencies and stray tool files. - Verdicts: `success`, `tests_failed`, `no_changes`, `success_untested`, `blocked`, `worker_error`, `timeout`, `stalled`, `cancelled`, `interrupted`, `integrity_violation`. -- **Read `handoff` first.** `handoff.state` (done, needs_fix, needs_review, needs_approval, needs_input, +- **Read `next` (brief) or `handoff` (full) first.** `handoff.state` (done, needs_fix, needs_review, needs_approval, needs_input, blocked, retryable, closed), `handoff.owner` (who acts next) and `handoff.next_action` (one exact instruction) tell you what to do. `handoff.failed_checks` lists the facts (failing tests with names, blocked calls, integrity, timeout or stall). `handoff.resume` has a suggested tool call and says whether @@ -58,7 +85,9 @@ is parked and waiting for a human decision. `interrupted`, `timeout`, `stalled` and `needs_approval` tasks. `handoff.resume.args` is a good starting point. - **Human approval** (`status: needs_approval`, `handoff.owner: "human"`): relay `handoff.next_action` - to the user word for word and wait for their answer. Don't approve on their behalf. Then call + to the user word for word and wait for their answer. Don't approve on their behalf. If the owner + enabled `approvals.require_operator`, your `approve_task` only records the request; the human then + confirms it with `workhorse approve ` and the operator token. Tell the user that. Then call `approve_task` with `decision: "approve"` (put their answer or conditions in `instructions`) or `decision: "reject"` (with `instructions` to redirect the worker, or without them to close the task). Approval does not lift sandbox rules. If the request needs network or an install, the user must do that @@ -70,6 +99,8 @@ is parked and waiting for a human decision. `state: "closed"` when it needs nobody. - **Find open work**: `list_tasks` with `status: "needs_attention"` (optionally `owner: "human"`). - **Second opinion**: `delegate_task` with `mode: "review"` and `review_task_id`. +- **Spend**: `usage_report` (or `workhorse stats` for the owner) shows worker tokens by profile and day + and an ESTIMATE of supervisor tokens avoided. - **Stop**: `cancel_task`. Use `cleanup_task` only once the user no longer needs the work (it archives the patch and refuses unmerged changes unless told to discard them). diff --git a/package-lock.json b/package-lock.json index 7e6bb92..0e3d546 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1,12 +1,12 @@ { "name": "grok-workhorse", - "version": "0.2.0", + "version": "0.3.0", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "grok-workhorse", - "version": "0.2.0", + "version": "0.3.0", "dependencies": { "@modelcontextprotocol/sdk": "^1.30.1", "zod": "^3.25.76" diff --git a/package.json b/package.json index f427a1e..bc3813a 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "grok-workhorse", - "version": "0.2.0", + "version": "0.3.0", "private": true, "type": "module", "description": "Sandboxed coding-agent workers (Kilo CLI, OpenCode, ...) that a supervising AI agent drives over MCP, with any OpenAI-compatible model provider", @@ -26,8 +26,9 @@ "scripts": { "setup": "npm ci && npm ci --prefix adapters/kilo/config && npm ci --prefix adapters/opencode/config", "test": "node --test --test-timeout=600000 test/*.test.mjs", - "test:unit": "node --test test/unit.test.mjs test/config.test.mjs test/adapters.test.mjs test/handoff.test.mjs test/handoff-flow.test.mjs test/audit.test.mjs", - "test:live": "WH_LIVE_TEST=1 node --test --test-timeout=1800000 test/live.test.mjs" + "test:unit": "node --test test/unit.test.mjs test/config.test.mjs test/adapters.test.mjs test/handoff.test.mjs test/handoff-flow.test.mjs test/audit.test.mjs test/token-savings.test.mjs", + "test:live": "WH_LIVE_TEST=1 node --test --test-timeout=1800000 test/live.test.mjs", + "test:stub": "node --test --test-timeout=600000 test/stub-e2e.test.mjs" }, "engines": { "node": ">=22" From 1f2cbddfba836e59e51ec101964b01cab52d8942 Mon Sep 17 00:00:00 2001 From: mrchatam <287639636+mrchatam@users.noreply.github.com> Date: Sun, 27 Sep 2026 01:31:25 +0330 Subject: [PATCH 5/7] Fix review findings: operator gate, request ids, review verdicts, budgets, auto-review children, rtk env - approvals.require_operator: a task that parks keeps a persistent operator_gate until the operator answers with the token; continue_task is refused on gated tasks whatever their status, closing via update_handoff/cancel_task/reject keeps the gate, reject with instructions needs the operator token - approval requests get an id; the operator CLI prints the pending request and sends back the displayed id; the daemon refuses the decision if the request changed - reviewVerdict reads the first line only, handles negations, else unclear - max_tokens/max_cost_usd 0 is an explicit zero budget (invalid values fail validate); auto-review child tokens/cost count against the parent budget; continue_task restarts run counters only - auto-review children: recover orphans after a crash, hidden from list_tasks unless include_auto_reviews, never needs_attention - full view keeps the complete handoff (backcompat); brief stays small - rtk rewrite: minimal env, 1 s timeout, falls back to the original command; bind only the rtk file - wait_task stops when the client disconnects; usage_report estimate counts worker output only - docs: operator gate is a parked-task flow gate, not a capability boundary; caps, trail, formula --- .../kilo/config/plugin/workhorse-guard.js | 25 ++- bin/workhorse | 70 +++++-- bin/workhorse-mcp | 3 +- docs/configuration.md | 24 ++- docs/handoff.md | 58 ++++-- docs/security.md | 18 +- docs/token-savings.md | 89 +++++--- docs/troubleshooting.md | 16 +- lib/config.mjs | 13 +- lib/daemon.mjs | 9 +- lib/handoff.mjs | 2 + lib/tasks.mjs | 194 ++++++++++++++---- lib/usage.mjs | 37 ++-- lib/views.mjs | 14 +- test/stub-e2e.test.mjs | 89 ++++++-- test/token-savings.test.mjs | 51 +++-- 16 files changed, 514 insertions(+), 198 deletions(-) diff --git a/adapters/kilo/config/plugin/workhorse-guard.js b/adapters/kilo/config/plugin/workhorse-guard.js index b604b6d..c3465e7 100644 --- a/adapters/kilo/config/plugin/workhorse-guard.js +++ b/adapters/kilo/config/plugin/workhorse-guard.js @@ -21,8 +21,11 @@ // 3. Optional token saver (daemon.json token_savers.rtk): when WH_RTK_BIN is set, a bash command that // passed the checks above is replaced by RTK's compact equivalent (`rtk rewrite`, e.g. `git status` // -> `rtk git status`; https://github.com/rtk-ai/rtk, Apache-2.0, run as an external binary). The -// rewritten command is checked again. Multi-line commands are left alone. Only the worker's own -// shell output is compacted; the daemon's test run never goes through this. +// rewrite call gets a minimal environment (PATH, HOME, RTK_TELEMETRY_DISABLED=1; no API keys) and a +// 1 s timeout; if it fails, times out, or its output does not pass the checks above, the original +// (already checked) command runs unchanged. It still runs inside the worker's outer sandbox, which +// has network access. Multi-line commands are left alone. Only the worker's own shell output is +// compacted; the daemon's test run never goes through this. // NOTE: Kilo/OpenCode call every export of this module as a plugin, so helpers must stay unexported. import path from "node:path" import { execFileSync } from "node:child_process" @@ -107,13 +110,25 @@ function checkBash(cmd, deny) { for (const p of denyPaths()) if (cmd.includes(p)) deny("access to the secret store or daemon token is not allowed") } +function passesBashChecks(cmd) { + try { + checkBash(cmd, (why) => { throw new Error(why) }) + return true + } catch { + return false + } +} + // `rtk rewrite ` prints the rewritten command and exits 0 (allowed) or 3 (rewritten; the host // decides permissions, which the checks here do); 1 = no RTK equivalent, 2 = deny rule. +// The plugin hook has to return the final command, so this call is synchronous; it is bounded by a +// short timeout and gets no secrets (minimal env). function rtkRewrite(bin, cmd) { if (!bin || !path.isAbsolute(bin) || !cmd || cmd.length > 2000 || /[\n\r]/.test(cmd)) return null + const env = { PATH: process.env.PATH || "/usr/local/bin:/usr/bin:/bin", HOME: process.env.HOME || "/nonexistent", RTK_TELEMETRY_DISABLED: "1" } let out try { - out = execFileSync(bin, ["rewrite", cmd], { encoding: "utf8", timeout: 3000, stdio: ["ignore", "pipe", "ignore"] }) + out = execFileSync(bin, ["rewrite", cmd], { encoding: "utf8", timeout: 1000, env, stdio: ["ignore", "pipe", "ignore"] }) } catch (e) { if (e && e.status === 3 && typeof e.stdout === "string") out = e.stdout else return null @@ -146,8 +161,8 @@ export const WorkhorseGuard = async ({ directory, worktree }) => { if (args.workdir && !inside(String(args.workdir))) deny("workdir outside the task worktree") const rw = rtkRewrite(process.env.WH_RTK_BIN, typeof args.command === "string" ? args.command : "") if (rw) { - checkBash(rw, deny) - args.command = rw + // Use the rewrite only if it passes the same checks; otherwise keep the original command. + if (passesBashChecks(rw)) args.command = rw } return } diff --git a/bin/workhorse b/bin/workhorse index 15ae055..fa5e339 100755 --- a/bin/workhorse +++ b/bin/workhorse @@ -13,8 +13,9 @@ Service: status quick daemon status (exit 1 if not running) health [--json] full check for a daily cron/agent run (exit 1 on any FAIL) logs [N] | audit [N] tail daemon.log / audit.jsonl - tasks [N] [--status S] [--owner O] - recent tasks (S: a status, active, terminal, parked, needs_attention) + tasks [N] [--status S] [--owner O] [--include-auto-reviews] + recent tasks (S: a status, active, terminal, parked, needs_attention); + automatic review tasks are hidden unless --include-auto-reviews attention [N] tasks that need someone: parked (needs_approval) or handoff not done/closed wait ... [--all] [--max S] [--full] long-poll until the task(s) finish (brief result; --all waits for every id) @@ -26,13 +27,15 @@ Service: Handoff and approval: handoff [--state S] [--owner O] [--next ""] [--note ""] [--by NAME] show the task's handoff record, or update it when options are given - approve [--instructions ""] [--note ""] [--timeout MIN] [--profile P] [--by NAME] + approve [--instructions ""] [--note ""] [--timeout MIN] [--profile P] [--by NAME] [--yes] approve a parked/blocked task and resume it in the same session + worktree - reject [--instructions ""] [--note ""] [--by NAME] + reject [--instructions ""] [--note ""] [--by NAME] [--yes] deny it: close the task, or resume with --instructions (do something else) operator-token init [--force] [--enable] create the operator token (approvals.require_operator); prints only its sha256. - With require_operator on, \`sudo workhorse approve\` sends the token. + With require_operator on, \`sudo workhorse approve|reject\` prints the pending + request, asks for confirmation on a terminal (--yes skips it) and sends the + token plus the displayed request id (refused if the request changed). Config (the config dir is root-owned once locked: use sudo for these): repos show the allowlist add-repo [--name N] [--test ""] [--allow-test ""]... [--base BRANCH] @@ -151,9 +154,9 @@ try { break case "tasks": case "attention": { - const f = flags(rest, { status: "str", owner: "str" }) + const f = flags(rest, { status: "str", owner: "str", "include-auto-reviews": "bool" }) const status = cmd === "attention" ? "needs_attention" : f.status - out(await rpc("list_tasks", { limit: Number(f._[0]) || 20, ...(status ? { status } : {}), ...(f.owner ? { owner: f.owner } : {}) })) + out(await rpc("list_tasks", { limit: Number(f._[0]) || 20, ...(status ? { status } : {}), ...(f.owner ? { owner: f.owner } : {}), ...(f["include-auto-reviews"] ? { include_auto_reviews: true } : {}) })) break } case "handoff": { @@ -167,25 +170,50 @@ try { } case "approve": case "reject": { - const f = flags(rest, { instructions: "str", note: "str", timeout: "str", profile: "str", by: "str" }) + const f = flags(rest, { instructions: "str", note: "str", timeout: "str", profile: "str", by: "str", yes: "bool", "request-id": "str" }) const id = f._[0] - if (!id) throw new Error(`usage: workhorse ${cmd} [--instructions ""] [--note ""]`) - // approvals.require_operator: send the operator token (readable only by the operator, e.g. root). + if (!id) throw new Error(`usage: workhorse ${cmd} [--instructions ""] [--note ""] [--yes] [--request-id ID]`) + const caller = { caller: { client: "workhorse" } } + // approvals.require_operator: send the operator token (readable only by the operator, e.g. root), + // after showing what is being answered; the displayed approval request id goes back to the daemon, + // which refuses the decision if the request changed in the meantime. let opTok = null - if (cmd === "approve") { - const { daemonConfig } = await import("../lib/config.mjs") - const { readOperatorToken, requireOperator, operatorTokenPath } = await import("../lib/operator.mjs") - const cfg = daemonConfig() - if (requireOperator(cfg)) { - opTok = readOperatorToken(cfg) - if (!opTok) console.error(`note: approvals.require_operator is on but ${operatorTokenPath(cfg)} is not readable; this only records an approval request (use sudo)`) + let requestId + const { daemonConfig } = await import("../lib/config.mjs") + const { readOperatorToken, requireOperator, operatorTokenPath } = await import("../lib/operator.mjs") + const cfg = daemonConfig() + if (requireOperator(cfg)) { + opTok = readOperatorToken(cfg) + if (!opTok) { + console.error(`note: approvals.require_operator is on but ${operatorTokenPath(cfg)} is not readable (use sudo): ${cmd === "approve" ? "this only records an approval request" : "reject can only close the task"}`) + } else { + const hf = await rpc("get_handoff", { task_id: id }, caller) + const h = hf.handoff || {} + const req = hf.approval_request + const lines = [`Task ${id}: status ${hf.status}, handoff ${h.state || "none"}`] + lines.push(` worker request: ${h.context?.worker_request || h.next_action || "(none recorded)"}`) + if (req) { + lines.push(` pending approval request ${req.id} by ${req.by} (${req.source?.channel || "?"}) at ${req.at}`) + if (req.instructions) lines.push(` instructions for the worker: ${req.instructions}`) + if (req.note) lines.push(` note: ${req.note}`) + if (req.profile || req.timeout_minutes) lines.push(` profile: ${req.profile || "(same)"}, timeout: ${req.timeout_minutes || "(default)"} min`) + } else lines.push(" no supervisor approval request pending") + if (f.instructions) lines.push(` your instructions (sent instead): ${f.instructions}`) + console.error(lines.join("\n")) + requestId = f["request-id"] || req?.id + if (process.stdin.isTTY && !f.yes) { + const rl = (await import("node:readline/promises")).createInterface({ input: process.stdin, output: process.stderr }) + const ans = await rl.question(`${cmd === "approve" ? "Approve" : "Reject"} this${requestId ? ` (request ${requestId})` : ""}? [y/N] `) + rl.close() + if (!/^y(es)?$/i.test(ans.trim())) throw new Error("aborted; nothing was sent") + } } } out(await rpc("approve_task", { - task_id: id, decision: cmd, ...(opTok ? { operator_token: opTok } : {}), + task_id: id, decision: cmd, ...(opTok ? { operator_token: opTok } : {}), ...(opTok && requestId ? { request_id: requestId } : {}), ...(f.instructions ? { instructions: f.instructions } : {}), ...(f.note ? { note: f.note } : {}), ...(f.timeout ? { timeout_minutes: Number(f.timeout) } : {}), ...(f.profile ? { profile: f.profile } : {}), ...(f.by ? { by: f.by } : {}), - }, { caller: { client: "workhorse" } })) + }, caller)) break } case "wait": { @@ -214,8 +242,8 @@ try { console.log(`\n${"day".padEnd(11)} ${"profile".padEnd(18)} ${"tasks".padStart(5)} ${"runs".padStart(5)} ${"tokens".padStart(9)} ${"est.cost".padStart(10)}`) for (const d of r.by_day) console.log(`${d.day.padEnd(11)} ${d.profile.padEnd(18)} ${String(d.tasks).padStart(5)} ${String(d.runs).padStart(5)} ${k(tok(d.tokens)).padStart(9)} ${usd(d.est_list_cost_usd).padStart(10)}`) const se = r.supervisor_estimate - console.log(`\nSupervisor tokens avoided (ESTIMATE): ~${k(se.est_supervisor_tokens_avoided)} over ${se.successful_tasks} successful task(s)` + (se.est_supervisor_cost_avoided_usd !== null ? `, ~$${se.est_supervisor_cost_avoided_usd} net` : "")) - console.log(` supervisor I/O with workhorse: ${se.supervisor_io.calls} call(s), ~${k(se.supervisor_io.est_tokens)} tokens; worker work tokens: ${k(se.worker_work_tokens)}`) + console.log(`\nSupervisor tokens avoided (ESTIMATE): ~${k(se.est_supervisor_tokens_avoided)} over ${se.successful_tasks} successful task(s)` + (se.est_net_usd !== null ? `, ~$${se.est_net_usd} net of worker cost` : "")) + console.log(` supervisor I/O with workhorse: ${se.supervisor_io.calls} call(s), ~${k(se.supervisor_io.est_tokens)} tokens; worker output tokens counted: ${k(se.worker_output_tokens)} (worker input not counted)`) console.log(` formula: ${se.formula}`) console.log(` ${se.caveat}`) break diff --git a/bin/workhorse-mcp b/bin/workhorse-mcp index 63bd323..ca520ce 100755 --- a/bin/workhorse-mcp +++ b/bin/workhorse-mcp @@ -160,6 +160,7 @@ tool( status: z.string().max(20).optional().describe("Filter: a status name, 'active', 'terminal', 'parked' (needs_approval) or 'needs_attention' (parked tasks plus finished tasks whose handoff is not done/closed)"), repo: z.string().max(64).optional(), owner: z.string().max(80).optional().describe("Filter by handoff owner, e.g. 'human' or 'supervisor'"), + include_auto_reviews: z.boolean().optional().describe("Also list the automatic review tasks (auto.review); hidden by default"), limit: z.number().int().min(1).max(200).optional().describe("Default 20"), }, ) @@ -190,7 +191,7 @@ tool( tool( "approve_task", - "Answer a task that is parked (needs_approval) or blocked. decision=approve resumes it in the same session and worktree with the approval (and optional instructions) sent to the worker. decision=reject with instructions resumes it telling the worker not to do it; without instructions it closes the task (status cancelled, worktree kept). Approval does not lift sandbox rules.", + "Answer a task that is parked (needs_approval) or blocked. decision=approve resumes it in the same session and worktree with the approval (and optional instructions) sent to the worker. decision=reject with instructions resumes it telling the worker not to do it; without instructions it closes the task (status cancelled, worktree kept). Approval does not lift sandbox rules. If the owner turned on approvals.require_operator, approve only records a request (a human confirms it on the host), reject with instructions is refused, and a closed task stays gated: continue_task on it is refused too.", { task_id: taskId, decision: z.enum(["approve", "reject"]), diff --git a/docs/configuration.md b/docs/configuration.md index 2bdfbdd..db6308d 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -71,20 +71,32 @@ the routine follow-up work on cheap models. the call or the preset) > `default_profile`. A preset may set any of `description`, `profile`, `size`, `mode`, `timeout_minutes`, `test_command`, `instructions` (appended to the task as standing instructions), `auto_fix_rounds`, `escalate`, `auto_review`. Call parameters win over the preset. -- **`auto_fix_rounds`** (0..3, hard cap 3): when an implement run ends with a verdict in `auto.fix_on`, - the daemon sends the same session a fix message with the failing tests/concerns, before settling. +- **`auto_fix_rounds`** (0..3, hard cap 3, counted **per profile**): when an implement run ends with a + verdict in `auto.fix_on`, the daemon sends the same session a fix message with the failing + tests/concerns, before settling. After an escalation the next profile gets its own rounds; the total + is still limited by `max_auto_runs`. - **`escalate`** (bool): when fix rounds are used up (or none were set) and the verdict is in `auto.escalate_on`, the task moves to the profile's `escalate_to` (fresh session, same worktree, existing changes kept). Chains are followed one step per run, cycles are refused by `workhorse validate`. -- **Hard caps**: `max_auto_runs` (default 3, cap 6) automatic runs per task, plus optional - `max_tokens` (input+output+reasoning of all runs) and `max_cost_usd` (needs `price_per_mtok`). The - first cap reached stops the chain; the result records `auto.stopped_reason`. +- **Hard caps**: `max_auto_runs` (default 3, cap 6) automatic runs per task in total (fix rounds and + escalations together), plus optional `max_tokens` (input+output+reasoning of all runs, including the + task's automatic review tasks) and `max_cost_usd` (needs `price_per_mtok`; also includes the reviews). + The first cap reached stops the chain; the result records `auto.stopped_reason`. Budgets are checked + between runs, so one run can overshoot them. `0` is an explicit zero budget (no automatic follow-ups); + omit the key or use `null` for no budget; negative or non-numeric values fail `workhorse validate`. + `continue_task` restarts the trail and the run counters for its new round, but not the token/cost + budgets. - **Escalation trail**: every automatic step is in `result.auto.trail` (`profile`, `verdict`, `next`, - tokens) and in the brief view as `"cheap:tests_failed->fix"`, `"cheap:tests_failed->mid"`, ... + tokens) and in the brief view as `"cheap:tests_failed->auto_fix"`, `"cheap:tests_failed->escalate"`, + `"mid:success"`. - **`auto_review`** (`true` or a profile name; default profile `auto.review.profile`, else the default profile): after a verdict in `auto.review.on`, a read-only review task runs on the diff. Its verdict is **advisory**: `result.review` and, for `request_changes`, handoff state `needs_review` with a suggested `continue_task`. It never changes the daemon's own verdict and never triggers a fix round by itself. + The reviewer's verdict is read from the first line of its answer only (`approve`, `request_changes`, + negations such as "not approved" count as request_changes); anything else is `unclear`. Review tasks + are hidden from `list_tasks` / `workhorse tasks` unless `include_auto_reviews` / + `--include-auto-reviews` is set, and never show up as needing attention themselves. Examples: [`config/examples/profiles.nvidia.json`](../config/examples/profiles.nvidia.json), [`profiles.openrouter.json`](../config/examples/profiles.openrouter.json) and diff --git a/docs/handoff.md b/docs/handoff.md index a4f6278..2d9384f 100644 --- a/docs/handoff.md +++ b/docs/handoff.md @@ -15,7 +15,7 @@ asked to write it. When the worker reports `status: blocked`, `needs_approval` o | Where | What | |---|---| -| `task_result` (full view, default) | the record under `handoff`; since v0.3 its `context` no longer repeats fields that are already top-level in the same response (verdict, status, summary, remaining_concerns, diffstat, files_changed, diff_path, diff) | +| `task_result` (full view, default) | the complete record under `handoff` (unchanged in v0.3; its `context` repeats a few top-level fields, kept for compatibility) | | `task_result` with `view: "brief"`, `wait_task` | `next`: state, owner, action (first 500 chars) and the suggested `tool` + `args` | | `get_handoff` (RPC) / `workhorse handoff ` | the complete record, including the full `context` | | `task_status` | `handoff`: state, owner, next_action (first 300 chars), names of failed checks, updated_at/by | @@ -149,21 +149,47 @@ human does it outside the sandbox first and then approves. The `next_action` tex ## Operator confirmation (`approvals.require_operator`, v0.3) -By default the supervisor's `approve_task` is final. An owner who wants a human in the loop for every -approval sets `daemon.json` `approvals.require_operator: true` (easiest: `sudo workhorse operator-token -init --enable`, which writes a random token to a root-only file and stores only its SHA-256 in -daemon.json). Then: - -1. MCP `approve_task` (from the supervisor) does not resume anything. It records an - **approval request** (`decision`, `instructions`, `by`, time) on the task, returns - `approval_requested: true` (handoff owner `human (operator)`), and the brief/full result shows - `approval_request`. Rejections do not need the operator: they only stop or redirect work. -2. A human runs `sudo workhorse approve ` (or `reject`). The CLI reads the token file and sends - the token; the daemon checks it against the hash (constant-time) and applies the recorded request's - instructions unless the CLI passes its own. -3. `continue_task` and `update_handoff` cannot unpark a parked task without the token either, so the - supervisor cannot route around the gate. The token is never an MCP parameter, is masked in the - audit log, and never appears in results. +By default the supervisor's `approve_task` is final. An owner who wants a human to confirm approvals +sets `daemon.json` `approvals.require_operator: true` (easiest: `sudo workhorse operator-token init +--enable`, which writes a random token to a root-only file and stores only its SHA-256 in daemon.json). + +**What it covers, and what it does not.** The option gates the *parked-task flow*: a task that parked +for approval cannot be resumed until the operator answers with the operator token. It is **not a +capability boundary**. The supervisor can still delegate a new task that asks the worker for the same +thing (that is visible in `workhorse tasks` and the audit log, but nothing blocks it), and approval +never widens the sandbox in any case. Use it to keep a human in the loop for decisions the worker +flagged, not as a way to stop a supervisor that is determined to do something. + +How it works: + +1. When a task parks (the worker asked for approval or input, or someone set a parking state with + `update_handoff`) the daemon sets a persistent **operator gate** on it (`operator_gate` in + `get_handoff`). The gate stays until the operator answers with the token, whatever happens to the + task's status meanwhile: closing it (`update_handoff state=closed`, `cancel_task`, + `approve_task decision=reject` without instructions) keeps the gate, so a later `continue_task` is + refused. Tasks parked before the option was turned on are gated while parked, and get the gate + when they are closed. +2. Without the token, MCP `approve_task decision=approve` does not resume anything. It records an + **approval request** with an id (`request_id`, a short hash; the request holds `instructions`, + `note`, `profile`, `timeout_minutes`, `by`, `source`, time), returns `approval_requested: true` + (handoff owner `human (operator)`), and the brief/full result shows `approval_request`. Recording a + new request replaces the old one and gets a new id. +3. Without the token, `decision=reject` may only **close** the task. Reject *with instructions* would + resume the worker with supervisor text, so it is refused. `continue_task` and `update_handoff` + (any state that would unpark it) are refused too; `update_handoff state=closed` is allowed and + closes the task the same way (status `cancelled`, gate kept). +4. A human runs `sudo workhorse approve ` (or `reject`). The CLI fetches the handoff, prints the + worker's request and the pending approval request (id, who, instructions, note), asks for + confirmation on a terminal (`--yes` skips the prompt), and sends the operator token plus the + **displayed request id**. The daemon checks the token against the hash (constant-time) and refuses + the decision if the pending request changed after it was displayed (or if a request is pending and + no id was sent), so the operator never approves text it did not see. On approval the recorded + request's instructions are used unless the operator passes `--instructions`. The operator can also + resume a gated task that was closed meanwhile, and can reject with instructions. The operator's + answer clears the gate (audit event `operator_gate_cleared`). +5. The token is never an MCP parameter (the MCP `approve_task` schema has no token or request id), is + masked in the audit log, and never appears in results. Approvals made with it record + `source.auth: "operator_token"`; everything else records `daemon_token`. ## Automatic follow-ups, reviews and handoff (v0.3) diff --git a/docs/security.md b/docs/security.md index 54bb2c4..4d16238 100644 --- a/docs/security.md +++ b/docs/security.md @@ -60,9 +60,12 @@ binary other than the bundled `stub-cli.mjs`. The installer, systemd unit and supervisor never set the flag, and `workhorse health` warns if a running daemon has it. 12. **Token savers stay outside the trust boundary.** `terse` / `minimal_code` are only extra text in - the worker's message. The optional RTK integration runs a local binary (read-only bind) that - rewrites the worker's own shell commands inside the sandbox; the guard checks the command before - and after rewriting, and the daemon's test run never goes through it. No proxy sees prompts or + the worker's message. The optional RTK integration runs a local binary (only that file is bound + read-only) that rewrites the worker's own shell commands. The `rtk rewrite` call runs in the + worker's outer sandbox, which has network access, but with a minimal environment (`PATH`, `HOME`, + `RTK_TELEMETRY_DISABLED=1`, no provider keys) and a 1 s timeout. The guard checks the command + before rewriting and checks the rewrite again; a rewrite that fails the check is dropped and the + original command runs. The daemon's test run never goes through it. No proxy sees prompts or keys (Headroom-style proxies are deliberately not integrated; see [token-savings.md](token-savings.md)). ## Known limitations @@ -84,8 +87,13 @@ - `approve_task` is a coordination signal, not a permission grant. It never widens the sandbox or the guard. The approver name (`by`) is recorded as given, and any client holding the daemon socket and token (the supervisor included) can approve. The recorded `source.channel` (`mcp` / `cli`) is - self-declared by the client for the same reason; `source.auth` says what was verified. See - `approvals.require_operator` below for putting a human in the loop. + self-declared by the client for the same reason; `source.auth` says what was verified + (`daemon_token`, or `operator_token` for a decision made with the operator token). +- `approvals.require_operator` gates the parked-task flow only: a parked task (and the same task after + it is closed or cancelled) cannot be resumed without the operator token, and the operator confirms a + specific approval request by id. It is **not a capability boundary**: the supervisor can still + delegate a new task that asks for the same thing (visible in the task list and audit log, not + blocked). See [handoff.md](handoff.md#operator-confirmation-approvalsrequire_operator-v03). ## Recommendations diff --git a/docs/token-savings.md b/docs/token-savings.md index 038c646..0ef70f6 100644 --- a/docs/token-savings.md +++ b/docs/token-savings.md @@ -8,7 +8,7 @@ backward compatible, and everything that changes worker behaviour is **opt-in**. | Where tokens go | Feature | Default | |---|---|---| | Supervisor: polling round trips | `wait_task` long-poll (single or many ids, `any`/`all`, up to 55 s per call) | available; the skill prefers it | -| Supervisor: reading results | compact JSON from the MCP shim, `view: "brief"` (~0.5-1 KB), deduplicated handoff in the full view | compact: on; brief: opt-in per call (the skill uses it), `wait_task` returns brief | +| Supervisor: reading results | compact JSON from the MCP shim, `view: "brief"` (~0.5-1 KB) | compact: on; brief: opt-in per call (the skill uses it), `wait_task` returns brief | | Supervisor: fix/escalate/review turns | `auto_fix_rounds`, `escalate` (cheap -> mid -> strong via `escalate_to`), `auto_review` on a cheap profile | off | | Supervisor: choosing and batching | presets, size routing, `delegate_tasks` (up to 10 per call) | available | | Worker output | `token_savers.terse` (short prose), `token_savers.minimal_code` (smallest-diff bias) | off | @@ -23,8 +23,9 @@ backward compatible, and everything that changes worker behaviour is **opt-in**. (finished or parked for approval) or `max_wait_s` passes (default 45, capped at **55 s**), then returns `{done, waited_s, task}` with the brief result. If `done` is false, call it again. The cap keeps each MCP call under the 60 s request timeout that many MCP clients use by default (the TypeScript SDK's -`DEFAULT_REQUEST_TIMEOUT_MSEC` is 60000), with margin for the round trip. A waiting call holds no -daemon resources except a timer. +`DEFAULT_REQUEST_TIMEOUT_MSEC` is 60000), with margin for the round trip. While it waits, a call +holds one open socket connection to the daemon and a light in-process check every 500 ms (no worker +or model tokens); the loop stops as soon as the client disconnects or the daemon shuts down. ### Compact JSON and the brief view @@ -32,23 +33,36 @@ The MCP shim now returns compact JSON (no indentation; 9-18% fewer characters on takes `view: "full"` (default, unchanged fields) or `view: "brief"`: verdict, a 300-char summary, up to 20 changed files, test outcome (failing test names only on failure), up to 5 concerns, `next` (handoff state, owner, action and the suggested tool call), the advisory review, the automatic trail and token -totals. In the full view the handoff's `context` no longer repeats fields that are already top-level. -`get_handoff` (RPC) and `workhorse handoff ` still show the complete record. +totals. The full view keeps every field it had in v0.2, including the complete handoff (its +`context` repeats some top-level fields; they are kept for compatibility). `get_handoff` (RPC) and +`workhorse handoff ` show the complete record too. ### Automatic follow-ups on cheap models Instead of the supervisor reading a failed result, writing a fix request and waiting again, the daemon can do it (see [configuration.md](configuration.md#presets-size-routing-and-automatic-follow-ups-v03)): -- `auto_fix_rounds` 1..3: fix rounds in the same session after failing tests or a missing/partial - RESULT block; +- `auto_fix_rounds` 1..3: fix rounds **per profile** in the same session after failing tests or a + missing/partial RESULT block (after an escalation the next profile gets its own rounds); - `escalate: true`: when that is not enough, the next profile of the `escalate_to` chain (fresh session, same worktree); - `auto_review`: a read-only review on a cheap profile; its verdict is advisory. -Hard caps: at most 3 fix rounds, `auto.max_auto_runs` (default 3, never more than 6) automatic runs -per task, optional `auto.max_tokens` and `auto.max_cost_usd`. The result's `auto.trail` shows every -step, for example `["cheap:tests_failed->fix", "cheap:tests_failed->mid", "mid:success"]`. +Hard caps: at most 3 fix rounds per profile, and **in total** at most `auto.max_auto_runs` automatic +runs per task (default 3, never more than 6), whatever mix of fix rounds and escalations that is; +optional `auto.max_tokens` and `auto.max_cost_usd` budgets. The result's `auto.trail` shows every +step, for example `["cheap:tests_failed->auto_fix", "cheap:tests_failed->escalate", "mid:success"]`. + +Budget details: + +- The budgets count every run of the task (including the first) plus the tokens and estimated cost of + its automatic review tasks. +- They are checked between runs, so a single run can overshoot them; the next automatic run is then + not started. +- `0` means an explicit zero budget (no automatic follow-ups at all), not "unlimited". Leave the key + out (or `null`) for no budget. +- `continue_task` starts a new automatic round: the trail and the fix-round / `max_auto_runs` + counters restart, but the token and cost budgets keep counting the whole task. ### Presets, size routing, batches @@ -89,8 +103,13 @@ workhorse token-savers off concerns. Not applied to review tasks. - **`rtk`**: Kilo and OpenCode only. The guard plugin asks `rtk rewrite ` for a compact equivalent of each bash command the worker runs (`git status` -> `rtk git status`) and uses it only - if the rewritten command passes the same guard checks. Multi-line commands are left alone. The rtk - binary's directory is bound read-only into the sandbox. Claude Code and Codex runs ignore it. + if the rewritten command passes the same guard checks; if it does not (or `rtk rewrite` fails or + takes longer than 1 s), the original command runs unchanged. The rewrite call gets a minimal + environment (`PATH`, `HOME`, `RTK_TELEMETRY_DISABLED=1`; no provider keys). It runs synchronously + in the plugin hook (the hook must return the final command), inside the worker's outer sandbox, + which has network access; the model-run command itself runs in Kilo's inner sandbox. Multi-line + commands are left alone. Only the rtk binary itself (resolved path) is bound read-only into the + sandbox, not its directory. Claude Code and Codex runs ignore it. Fragments go only into the first message of a fresh session (not into follow-ups, which already have them in context). **The daemon's own test run is never routed through RTK or any other compression**: @@ -127,8 +146,10 @@ Risks we designed around: - **Quality loss.** Prompt savers can make a model skip context or cut corners. That is why they are off by default, the RESULT format is protected, and `minimal_code` has never-drop rules. The daemon's verdict still comes from the tests. -- **Secrets.** No saver sends data anywhere. RTK runs inside the worker sandbox (no network for - model-run commands on Kilo) with telemetry off by default. +- **Secrets.** No saver sends data anywhere. The `rtk rewrite` call runs with a minimal environment + (no API keys) and `RTK_TELEMETRY_DISABLED=1`. It runs in the worker's outer sandbox, which does have + network access; the rewritten command then runs like any model-run command (on Kilo: inner + sandbox, no network). ## Benchmarks @@ -139,8 +160,8 @@ treat them as **estimates**. ### Supervisor side (stub backend, v0.2 flow vs v0.3 flow) Measured with the test-only stub backend on the calc fixture. "v0.2" = `delegate_task`, `task_status` -polls, `task_result` (full view, pretty JSON, handoff with duplicated context). "v0.3" = -`delegate_task` + one `wait_task` (brief, compact JSON). +polls, `task_result` (full view, pretty JSON). "v0.3" = `delegate_task` + one `wait_task` (brief, +compact JSON). The v0.3 full view has the same fields as v0.2, only compact JSON. | Response the supervisor reads | chars | tokens (o200k) | |---|---|---| @@ -148,7 +169,6 @@ polls, `task_result` (full view, pretty JSON, handoff with duplicated context). | v0.2 `task_status` while running / final | 634 / 935 | 224 / 314 | | v0.2 `task_result` full, success | 4,318 | 1,427 | | v0.2 `task_result` full, tests_failed | 7,150 | 2,182 | -| v0.3 `task_result` full (compact, deduped), success | 3,179 | 976 | | v0.3 `delegate_task` (compact) | 510 | 162 | | v0.3 `wait_task` with brief result, success | 671 | 198 | @@ -216,22 +236,29 @@ indicative only, and measure on your own repos with `workhorse stats` before ena tokens and estimated list cost by profile and by day (per run, so escalated tasks are split across profiles). Tasks from before v0.3 count their totals on their last run. -The report also contains `supervisor_estimate`, **labelled ESTIMATE**. Formula, per task whose final -verdict is `success` or `success_untested`: +The report also contains `supervisor_estimate`, **labelled ESTIMATE**. It is deliberately +conservative: ``` -worker_work_tokens = worker input + output + reasoning tokens (cache reads excluded) -supervisor_overhead = (chars the supervisor sent for this task + chars it received) / chars_per_token -est_supervisor_tokens_avoided = max(0, sum(worker_work_tokens) - sum(supervisor_overhead of all tasks)) +worker_output_tokens = output + reasoning tokens of tasks whose final verdict is success or success_untested +supervisor_overhead = (chars the supervisor sent to workhorse + chars it read back, all tasks) / chars_per_token +est_supervisor_tokens_avoided = max(0, worker_output_tokens - supervisor_overhead) ``` -Failed tasks count only as overhead. Supervisor characters are recorded by the daemon for every MCP -call made through `workhorse-mcp` (the operator CLI is excluded). `chars_per_token` defaults to 4 -(`daemon.json supervisor.chars_per_token`). With `supervisor.price_per_mtok: {input, output}` it also -gives a USD figure: avoided input tokens x input price + avoided output tokens x output price - the -workers' estimated list cost. +Worker **input** tokens are not counted at all (`worker_input_tokens_not_counted` shows them): most of +them are the same context re-read on every turn, and a supervisor doing the work itself would read a +different amount. Failed tasks count only as overhead. Supervisor characters are recorded by the +daemon for every MCP call made through `workhorse-mcp` (the operator CLI is excluded). +`chars_per_token` defaults to 4 (`daemon.json supervisor.chars_per_token`). With +`supervisor.price_per_mtok: {input, output}` it also gives + +``` +est_net_usd = worker_output_tokens x price.output + - supervisor tokens read x price.input - supervisor tokens written x price.output + - the workers' estimated list cost (can be negative) +``` -The assumption behind it: the supervisor would have needed about as many tokens as the worker to do -the same work itself. A stronger model may need fewer turns (the estimate is then too high); doing -the work itself would also grow the supervisor's context for the rest of the session (not counted, so -too low). It is a planning aid, not a measurement. +The assumption behind it: for work that succeeded, the supervisor would have had to generate about +as many output tokens as the workers did. A stronger model may write less (too high); the input it +would have read and the growth of its own context are ignored (too low). With small tasks the result +is often 0. It is a planning aid, not a measurement. diff --git a/docs/troubleshooting.md b/docs/troubleshooting.md index b89e3aa..6fe1ad2 100644 --- a/docs/troubleshooting.md +++ b/docs/troubleshooting.md @@ -56,14 +56,24 @@ reached, the verdict was not in `auto.fix_on` / `auto.escalate_on`, or the profi ## `approve_task` says "approval recorded as a request" `approvals.require_operator` is on. A human must run `sudo workhorse approve ` on the host -(it reads the operator token file). `operator token rejected` means the file does not match -`approvals.operator_token_sha256`: re-run `sudo workhorse operator-token init --force --enable` and restart. +(it reads the operator token file, prints the pending request and confirms its id). `operator token +rejected` means the file does not match `approvals.operator_token_sha256`: re-run `sudo workhorse +operator-token init --force --enable` and restart. `the pending approval request changed` means the +supervisor recorded a new request after the CLI displayed it: run the command again and review it. + +## `continue_task` says "this task waits for the operator" + +The task parked for approval while `approvals.require_operator` was on, and the operator has not +answered yet. Closing or cancelling the task does not remove that gate. The operator resumes it with +`sudo workhorse approve ` (or `reject --instructions ...`). ## RTK is enabled but commands are not rewritten `workhorse token-savers` shows whether the binary was found. RTK applies to Kilo and OpenCode only, needs an absolute `rtk.bin` or `rtk` on the daemon's `env_path`, and leaves commands alone when it has -no equivalent (exit 1) or when the command spans several lines. Tests run by the daemon never use it. +no equivalent (exit 1), when the command spans several lines, when `rtk rewrite` takes longer than +1 s, or when the rewritten command would be blocked by the guard (the original runs instead). Tests run +by the daemon never use it. ## Verdict `blocked` or many blocked calls diff --git a/lib/config.mjs b/lib/config.mjs index 6091887..6a7c434 100644 --- a/lib/config.mjs +++ b/lib/config.mjs @@ -112,8 +112,8 @@ export const AUTO_DEFAULTS = { escalate: false, // on failure, move up the profile's escalate_to chain (fresh session, same worktree) escalate_on: ["tests_failed", "worker_error", "partial"], max_auto_runs: 3, // automatic follow-up runs per task (fix + escalate), hard cap AUTO_HARD.max_auto_runs - max_tokens: null, // stop automatic follow-ups once the task used this many (input+output+reasoning) tokens - max_cost_usd: null, // ... or this estimated list cost (needs price_per_mtok on the profiles) + max_tokens: null, // stop automatic follow-ups once the task used this many (input+output+reasoning) tokens, auto-reviews included; 0 = none at all + max_cost_usd: null, // ... or this estimated list cost (needs price_per_mtok on the profiles); 0 = none at all review: { enabled: false, profile: null, on: ["success", "success_untested"], timeout_minutes: 10 }, } const AUTO_TRIGGERS = ["tests_failed", "no_result_block", "partial", "no_changes", "worker_error"] @@ -284,14 +284,17 @@ export function profilesConfig() { const num = (v, d) => (typeof v === "number" && Number.isFinite(v) && v >= 0 ? v : d) const list = (v, d) => (Array.isArray(v) ? v.filter((x) => AUTO_TRIGGERS.includes(x) || ["success", "success_untested"].includes(x)) : d) const rv = a.review && typeof a.review === "object" ? a.review : {} + const autoProblems = [] + for (const k of ["max_tokens", "max_cost_usd"]) if (a[k] !== undefined && a[k] !== null && !(typeof a[k] === "number" && Number.isFinite(a[k]) && a[k] >= 0)) autoProblems.push(`auto.${k} must be a number >= 0, or null for no budget (0 = no automatic follow-ups)`) const auto = { fix_rounds: Math.min(AUTO_HARD.fix_rounds, Math.floor(num(a.fix_rounds, AUTO_DEFAULTS.fix_rounds))), fix_on: list(a.fix_on, AUTO_DEFAULTS.fix_on), escalate: a.escalate === true, escalate_on: list(a.escalate_on, AUTO_DEFAULTS.escalate_on), max_auto_runs: Math.min(AUTO_HARD.max_auto_runs, Math.floor(num(a.max_auto_runs, AUTO_DEFAULTS.max_auto_runs))), - max_tokens: num(a.max_tokens, null) || null, - max_cost_usd: num(a.max_cost_usd, null) || null, + // null = no budget; 0 is an explicit zero budget (no automatic follow-ups or reviews at all). + max_tokens: num(a.max_tokens, null), + max_cost_usd: num(a.max_cost_usd, null), review: { enabled: rv.enabled === true, profile: typeof rv.profile === "string" && rv.profile ? rv.profile : null, @@ -306,6 +309,7 @@ export function profilesConfig() { presets, routing, auto, + auto_problems: autoProblems, kilo_overlay: raw.kilo_overlay || {}, opencode_overlay: raw.opencode_overlay || {}, } @@ -410,6 +414,7 @@ export function validateConfig() { } for (const [sz, prof] of Object.entries(pc.routing)) if (!pc.profiles[prof]) problems.push(`routing.${sz}: profile '${prof}' is not defined`) if (pc.auto.review.profile && !pc.profiles[pc.auto.review.profile]) problems.push(`auto.review.profile '${pc.auto.review.profile}' is not defined`) + problems.push(...(pc.auto_problems || [])) try { const d = daemonConfig() problems.push(...saverProblemsLite(d.token_savers, "daemon.json token_savers")) diff --git a/lib/daemon.mjs b/lib/daemon.mjs index 707c019..e0d2089 100644 --- a/lib/daemon.mjs +++ b/lib/daemon.mjs @@ -83,12 +83,12 @@ export async function startDaemon() { offer_env: (p) => mgr.offerEnv(p.env), delegate_task: (p) => mgr.delegate(p), delegate_tasks: (p) => mgr.delegateMany(p), - wait_task: (p) => mgr.wait(p), + wait_task: (p, caller, ctx) => mgr.wait(p, ctx), usage_report: (p) => mgr.usage(p), task_status: (p) => mgr.status(p.task_id), task_result: (p) => mgr.result(p.task_id, p.view), task_details: (p) => mgr.details(p.task_id, p.kind, p.offset, p.max_bytes, p.run), - continue_task: (p) => mgr.continueTask(p.task_id, p.instructions, p.timeout_minutes, p.profile, { operatorToken: p.operator_token }), + continue_task: (p, caller) => mgr.continueTask(p.task_id, p.instructions, p.timeout_minutes, p.profile, { operatorToken: p.operator_token, caller }), cancel_task: (p) => mgr.cancel(p.task_id), list_tasks: (p) => mgr.list(p), get_handoff: (p) => mgr.handoff(p.task_id), @@ -130,8 +130,11 @@ export async function startDaemon() { const fn = typeof method === "string" && Object.hasOwn(methods, method) ? methods[method] : null if (!fn) return send(400, { error: `unknown method ${String(method).slice(0, 40)}` }) const started = Date.now() + // Lets long-polls (wait_task) stop early when the client goes away. + const ac = new AbortController() + res.on("close", () => { if (!res.writableEnded) ac.abort() }) try { - const result = await fn(params || {}, caller) + const result = await fn(params || {}, caller, { signal: ac.signal }) if (method !== "health" && method !== "health_report" && !(method === "offer_env" && !result?.changed)) audit("rpc", { method, caller, params: sanitizeParams(method, params), ok: true, ms: Date.now() - started, task_id: params?.task_id || result?.task_id }) // Supervisor (MCP) I/O per task for usage_report's estimate; the operator CLI is not counted. if (caller?.client !== "workhorse" && !QUIET_METHODS.has(method)) { diff --git a/lib/handoff.mjs b/lib/handoff.mjs index 36d1a24..d8fa4a1 100644 --- a/lib/handoff.mjs +++ b/lib/handoff.mjs @@ -201,6 +201,8 @@ export function deriveHandoff(t, { by = "daemon", at = new Date().toISOString() args = cont(`An automated reviewer reported: ${head((rv.findings || []).join("; ") || rv.summary || "", 1200)}. Fix the findings that are valid (ignore ones that are wrong, and say why under concerns), rerun the tests and end with the ## RESULT block.`) } else if (rv.verdict === "approve") { next = `${next} (The cheap auto-reviewer ${rv.profile} approved; advisory only.)` + } else if (rv.verdict === "unclear") { + next = `${next} (The cheap auto-reviewer ${rv.profile} gave no clear verdict; read result.review if you want its notes.)` } } const auto = r.auto diff --git a/lib/tasks.mjs b/lib/tasks.mjs index 18c12f5..99d5091 100644 --- a/lib/tasks.mjs +++ b/lib/tasks.mjs @@ -13,7 +13,7 @@ import { loadFromStore } from "./credentials.mjs" import { outerSandboxArgs } from "./sandbox.mjs" import { getBackend, profileBackend, backendProblems, backendSummary } from "../adapters/index.mjs" import { deriveHandoff, handoffSummary, normalizeWorkerStatus, extractFailingTests, HANDOFF_STATES, PARK_STATES, QUIET_STATES, APPROVABLE_STATES, OWNER_RE, SETUP_ERROR_RE } from "./handoff.mjs" -import { briefResult, dedupeHandoff, VIEWS } from "./views.mjs" +import { briefResult, VIEWS } from "./views.mjs" import { effectiveSavers, saverFragments, saverSummary, rtkBin } from "./savers.mjs" import { usageReport, SUCCESS } from "./usage.mjs" import { requireOperator, checkOperatorToken } from "./operator.mjs" @@ -212,10 +212,24 @@ export class Manager { // (a review task that was running was finalized above, which already applied its verdict). for (const t of this.tasks.values()) { if (t.status !== "reviewing") continue - const child = t.review_pending?.task_id ? this.tasks.get(t.review_pending.task_id) : null + const pend = t.review_pending || {} + const child = (pend.task_id && this.tasks.get(pend.task_id)) || [...this.tasks.values()].filter((c) => c.auto_review_of === t.id && (!pend.started_at || c.created_at >= pend.started_at)).sort((a, b) => (a.created_at < b.created_at ? 1 : -1))[0] || null + if (child && t.review_pending && !pend.task_id) { + t.review_pending.task_id = child.id + this.save(t) + } if (!child) await this.finishReview(t, null, "review task missing after restart") else if (settled(child.status) && child.result) await this.finishReview(t, child) } + // Review tasks whose parent no longer waits for them (crash between the two saves, or the parent was + // finished meanwhile): stop them instead of spending tokens on a verdict nobody reads. + for (const c of this.tasks.values()) { + if (!c.auto_review_of || settled(c.status)) continue + const p = this.tasks.get(c.auto_review_of) + if (p?.status === "reviewing" && p.review_pending?.task_id === c.id) continue + this.event(c, "orphan_auto_review_cancelled", { parent: c.auto_review_of }) + await this.cancel(c.id).catch(() => {}) + } } // ---------- config helpers ---------- @@ -427,6 +441,13 @@ export class Manager { preset: preset?.name || null, size: choice.size, routed_by_size: choice.routed, auto, auto_trail: [], auto_review_of: internal.autoReviewOf || null, supervisor_io: { calls: 0, request_chars: 0, response_chars: 0 }, } + // An auto-review child is linked from its parent before the child is saved, so a crash in between + // leaves a parent pointing at a missing task (recover() finishes it) rather than an orphan review. + const parent = internal.autoReviewOf ? this.tasks.get(internal.autoReviewOf) : null + if (parent?.status === "reviewing" && parent.review_pending) { + parent.review_pending.task_id = id + this.save(parent) + } this.tasks.set(id, t) this.save(t) this.event(t, "created", { repo: repo.name, profile: profile.name, backend: profile.backend, mode, base_commit: baseCommit, ...(preset ? { preset: preset.name } : {}), ...(choice.routed ? { routed_by_size: choice.routed } : {}), ...(auto ? { auto_fix_rounds: auto.fix_rounds, escalate: auto.escalate, auto_review: auto.review_profile } : {}), ...(internal.autoReviewOf ? { auto_review_of: internal.autoReviewOf } : {}) }) @@ -489,7 +510,7 @@ export class Manager { return v } - // view "full" (default): the whole result; the handoff context omits fields already top-level. + // view "full" (default): the whole result with the complete handoff record (unchanged from v0.2). // view "brief": the small decision summary (lib/views.mjs). result(id, view) { const t = this.get(id) @@ -497,7 +518,7 @@ export class Manager { if (!t.result || !settled(t.status)) return { task_id: t.id, status: t.status, terminal: false, phase: t.phase, message: "No result yet; the task is still active. Call wait_task." } const h = this.handoffOf(t) if (v === "brief") return briefResult({ ...t.result, status: t.status, approval_request: t.approval_request ? { at: t.approval_request.at, by: t.approval_request.by, waiting_for: "operator" } : undefined }, h) - return { ...t.result, status: t.status, handoff: dedupeHandoff(h), ...(t.approval_request ? { approval_request: t.approval_request } : {}) } + return { ...t.result, status: t.status, handoff: h, ...(t.approval_request ? { approval_request: t.approval_request } : {}) } } // Compact progress record for wait_task. @@ -513,7 +534,7 @@ export class Manager { // Long-poll until the task(s) settle (finished or parked) or max_wait_s passes (cap WAIT_CAP_S). // Replaces task_status polling + a separate task_result call: settled tasks come back with their // result in the requested view (default brief). - async wait(p = {}) { + async wait(p = {}, { signal } = {}) { const allowed = new Set(["task_id", "task_ids", "mode", "max_wait_s", "view"]) for (const k of Object.keys(p)) if (!allowed.has(k)) throw new UserError(`unknown parameter '${k}'`) if ((p.task_id === undefined) === (p.task_ids === undefined)) throw new UserError("pass task_id or task_ids (not both)") @@ -534,7 +555,7 @@ export class Manager { const n = uniq.filter((id) => { const t = this.tasks.get(id); return !t || settled(t.status) }).length return mode === "all" ? n === uniq.length : n > 0 } - while (!isDone() && !this.shuttingDown && Date.now() - t0 < maxWait * 1000) await sleep(Math.min(500, maxWait * 1000 - (Date.now() - t0))) + while (!isDone() && !this.shuttingDown && !signal?.aborted && Date.now() - t0 < maxWait * 1000) await sleep(Math.min(500, maxWait * 1000 - (Date.now() - t0))) const done = isDone() const one = (id) => { const t = this.tasks.get(id) @@ -596,6 +617,7 @@ export class Manager { needsAttention(t) { if (!settled(t.status)) return false + if (t.auto_review_of) return false // advisory review children: their parent carries the outcome if (PARKED.has(t.status)) return true const h = this.handoffOf(t) return !!h && !QUIET_STATES.has(h.state) @@ -623,6 +645,8 @@ export class Manager { list(filter = {}) { let arr = [...this.tasks.values()] + // Automatic review tasks belong to their parent (result.review); shown only on request. + if (filter.include_auto_reviews !== true) arr = arr.filter((t) => !t.auto_review_of) const f = filter.status // 'terminal' = no worker running (includes parked tasks); 'parked' = waiting for approve_task; // 'needs_attention' = parked tasks plus finished tasks whose handoff is not done/closed. @@ -636,18 +660,38 @@ export class Manager { tasks: arr.slice(0, limit).map((t) => ({ task_id: t.id, status: t.status, verdict: t.result?.verdict || null, repo: t.repo, mode: t.mode, profile: t.profile, created_at: t.created_at, finished_at: t.finished_at, task: head(t.task, 120), branch: t.branch, worktree_removed: t.worktree_removed, + ...(t.auto_review_of ? { auto_review_of: t.auto_review_of } : {}), handoff: settled(t.status) ? (({ state, owner, next_action }) => ({ state, owner, next_action }))(handoffSummary(this.handoffOf(t)) || {}) : null, })), } } - async continueTask(id, instructions, timeoutMinutes, profileName, { approval = null, operatorToken, operatorOk = false } = {}) { + // approvals.require_operator: a task that parked for approval stays gated (t.operator_gate, persisted) + // until the operator answers with the operator token, whatever happens to its status meanwhile + // (closed via update_handoff, cancel_task, unparked, ...). Tasks parked before the option was turned + // on are gated while they are parked. + operatorGated(t, cfg = daemonConfig()) { + return requireOperator(cfg) && (!!t.operator_gate || PARKED.has(t.status)) + } + + setOperatorGate(t, cfg, reason) { + if (requireOperator(cfg) && !t.operator_gate) t.operator_gate = { at: now(), reason } + } + + async continueTask(id, instructions, timeoutMinutes, profileName, { approval = null, operatorToken, operatorOk = false, caller = null } = {}) { const t = this.get(id) if (!settled(t.status)) throw new UserError(`task is ${t.status}; continue_task works only on finished or parked tasks (cancel it first if needed)`) this.assertNotCleaning(t) const cfg = daemonConfig() - if (PARKED.has(t.status) && requireOperator(cfg) && !operatorOk && !checkOperatorToken(cfg, operatorToken)) - throw new UserError("task is parked for approval and approvals.require_operator is on: use approve_task (records the request; a human confirms with `sudo workhorse approve `), or approve_task decision=reject") + if (this.operatorGated(t, cfg) && !operatorOk) { + if (operatorToken !== undefined && !checkOperatorToken(cfg, operatorToken)) { + this.event(t, "operator_token_rejected", { method: "continue_task" }) + throw new UserError("operator token rejected (check approvals.operator_token_sha256 and the token file)") + } + if (operatorToken === undefined) + throw new UserError("this task waits for the operator (approvals.require_operator): the supervisor can only request approval with approve_task (a human confirms with `sudo workhorse approve `) or close it (approve_task decision=reject without instructions). Delegating the same work as a new task is visible to the operator in the audit log.") + operatorOk = true + } if (t.worktree_removed) throw new UserError("task worktree was cleaned up; delegate a new task instead") if (typeof instructions !== "string" || !instructions.trim()) throw new UserError("instructions are required") if (instructions.length > daemonConfig().limits.task_max_chars) throw new UserError("instructions too long") @@ -657,6 +701,12 @@ export class Manager { const h = this.handoffOf(t) if (t.result) t.previous_results.push({ at: now(), status: t.status, verdict: t.result.verdict, summary: t.result.summary, handoff: h ? { state: h.state, owner: h.owner, next_action: h.next_action } : undefined, approval: approval || undefined, ...(t.result.auto ? { auto: t.result.auto } : {}), ...(t.result.review ? { review: { verdict: t.result.review.verdict, task_id: t.result.review.task_id } } : {}) }) if (approval) t.approvals = [...(t.approvals || []), approval].slice(-20) + if (operatorOk && t.operator_gate) { + this.event(t, "operator_gate_cleared", { by: approval?.by || "operator", source: approval?.source || { channel: caller?.client === "workhorse" ? "cli" : "mcp", auth: "operator_token" } }) + delete t.operator_gate + } + // A continue starts a new automatic round: the fix-round and max_auto_runs counters restart; the + // max_tokens / max_cost_usd budgets keep counting the whole task (all runs + auto-reviews). t.auto_trail = [] delete t.approval_request t.result = null @@ -729,7 +779,7 @@ export class Manager { handoff(id) { const t = this.get(id) if (!settled(t.status)) return { task_id: t.id, status: t.status, handoff: null, message: "The task is still active; the handoff is written when it finishes." } - return { task_id: t.id, status: t.status, verdict: t.result?.verdict || null, handoff: this.handoffOf(t), approvals: t.approvals || [] } + return { task_id: t.id, status: t.status, verdict: t.result?.verdict || null, handoff: this.handoffOf(t), approvals: t.approvals || [], ...(t.approval_request ? { approval_request: t.approval_request } : {}), ...(t.operator_gate ? { operator_gate: t.operator_gate } : {}) } } // Supervisor/human edits of the handoff. Setting state needs_approval/needs_input parks a finished task; @@ -741,8 +791,11 @@ export class Manager { if (!settled(t.status)) throw new UserError(`task is ${t.status}; the handoff exists once the task has finished`) this.assertNotCleaning(t) const cfg = daemonConfig() - if (PARKED.has(t.status) && p.state !== undefined && !PARK_STATES.has(p.state) && p.state !== "closed" && requireOperator(cfg) && !checkOperatorToken(cfg, p.operator_token)) - throw new UserError("task is parked for approval and approvals.require_operator is on: only the operator can unpark it (`sudo workhorse approve `); state=closed is allowed") + let operatorOk = false + if (PARKED.has(t.status) && p.state !== undefined && !PARK_STATES.has(p.state) && p.state !== "closed" && this.operatorGated(t, cfg)) { + if (!checkOperatorToken(cfg, p.operator_token)) throw new UserError("task is parked for approval and approvals.require_operator is on: only the operator can unpark it (`sudo workhorse approve `); state=closed is allowed (the task stays gated)") + operatorOk = true + } if (p.state !== undefined && !HANDOFF_STATES.includes(p.state)) throw new UserError(`state must be one of ${HANDOFF_STATES.join(", ")}`) if (p.owner !== undefined && (typeof p.owner !== "string" || !OWNER_RE.test(p.owner))) throw new UserError("owner must be 'supervisor', 'human' or a short name (letters, digits, space, _.:@/+-; max 80 chars)") const nextAction = this.checkText("next_action", p.next_action) @@ -778,10 +831,12 @@ export class Manager { if (t.worktree_removed) throw new UserError("task worktree was cleaned up; it cannot be parked for approval") t.parked_from = t.status t.status = "needs_approval" + this.setOperatorGate(t, cfg, "parked via update_handoff") if (p.owner === undefined && h.owner !== "human") { h.owner = "human"; changed.push("owner") } } else if (!PARK_STATES.has(p.state) && PARKED.has(t.status)) { t.status = t.parked_from || "completed" delete t.parked_from + if (operatorOk) delete t.operator_gate } } h.history = [...(h.history || []), { at, by: who, source, event: "update", changed, state: h.state, owner: h.owner, ...(from !== t.status ? { from_status: from, to_status: t.status } : {}) }].slice(-20) @@ -798,68 +853,96 @@ export class Manager { // follow-up message. reject: with instructions, resume and tell the worker not to do it; without, // close the task (status cancelled, handoff closed; the worktree is kept). async approve(p = {}, caller = null) { - const allowed = new Set(["task_id", "decision", "instructions", "note", "by", "timeout_minutes", "profile", "operator_token"]) + const allowed = new Set(["task_id", "decision", "instructions", "note", "by", "timeout_minutes", "profile", "operator_token", "request_id"]) for (const k of Object.keys(p)) if (!allowed.has(k)) throw new UserError(`unknown parameter '${k}'`) const t = this.get(p.task_id) if (p.decision !== "approve" && p.decision !== "reject") throw new UserError("decision must be 'approve' or 'reject'") if (!settled(t.status)) throw new UserError(`task is ${t.status}; nothing to approve yet`) this.assertNotCleaning(t) - const h = this.handoffOf(t) - if (!PARKED.has(t.status) && !APPROVABLE_STATES.has(h?.state)) throw new UserError(`task is not waiting for approval (status ${t.status}, handoff state ${h?.state || "none"}); use continue_task or update_handoff`) const cfg = daemonConfig() let instructions = this.checkText("instructions", p.instructions, cfg.limits.task_max_chars) const note = this.checkText("note", p.note) + const opMode = requireOperator(cfg) + const gated = this.operatorGated(t, cfg) let operatorOk = false - if (requireOperator(cfg) && p.decision === "approve") { + if (opMode && p.operator_token !== undefined) { operatorOk = checkOperatorToken(cfg, p.operator_token) - if (!operatorOk && p.operator_token !== undefined) { - this.event(t, "operator_token_rejected", {}) + if (!operatorOk) { + this.event(t, "operator_token_rejected", { method: "approve_task" }) throw new UserError("operator token rejected (check approvals.operator_token_sha256 and the token file)") } - if (!operatorOk) return this.recordApprovalRequest(t, h, { who: this.actor(p.by, caller), instructions, note, profile: p.profile, timeout_minutes: p.timeout_minutes }) - const req = t.approval_request - if (req) { + } + const h = this.handoffOf(t) + // The operator may also resume a gated task that was closed meanwhile (its handoff is no longer approvable). + if (!PARKED.has(t.status) && !APPROVABLE_STATES.has(h?.state) && !(operatorOk && t.operator_gate)) throw new UserError(`task is not waiting for approval (status ${t.status}, handoff state ${h?.state || "none"}); use continue_task or update_handoff`) + if (opMode && !operatorOk) { + // Without the operator token: approve only records a request; reject may only close the task + // (reject with instructions would resume the worker with supervisor text). + if (p.decision === "approve") return this.recordApprovalRequest(t, h, { who: this.actor(p.by, caller), source: this.sourceOf(caller), instructions, note, profile: p.profile, timeout_minutes: p.timeout_minutes }) + if (instructions && gated) throw new UserError("approvals.require_operator is on: reject with instructions would resume the worker, so only the operator can do it (`sudo workhorse reject --instructions ...`). Reject without instructions to close the task.") + } + const req = t.approval_request + if (operatorOk) { + // The operator confirms exactly the request it was shown: the CLI prints it and sends its id back. + // (Closing without instructions does not act on the request, so it needs no id.) + if (req && p.request_id !== req.id && (p.decision === "approve" || instructions || p.request_id)) throw new UserError(p.request_id ? `the pending approval request changed (now ${req.id}); run the command again to review it` : `an approval request (${req.id}) is pending: review it first (\`workhorse approve ${t.id}\` prints it and sends its id)`) + if (!req && p.request_id) throw new UserError("no approval request is pending for this task") + if (req && p.decision === "approve") { instructions ??= req.instructions p = { ...p, profile: p.profile ?? req.profile, timeout_minutes: p.timeout_minutes ?? req.timeout_minutes } } } - const who = operatorOk && (p.by === undefined || p.by === null || p.by === "") ? `operator${t.approval_request ? ` (requested by ${t.approval_request.by})` : ""}` : this.actor(p.by, caller) - const source = this.sourceOf(caller) + const who = operatorOk && (p.by === undefined || p.by === null || p.by === "") ? `operator${req && p.decision === "approve" ? ` (requested by ${req.by})` : ""}` : this.actor(p.by, caller) + const source = operatorOk ? { channel: caller?.client === "workhorse" ? "cli" : "mcp", auth: "operator_token" } : this.sourceOf(caller) const request = h?.context?.worker_request || h?.next_action || "the pending request" - const rec = { at: now(), by: who, source, decision: p.decision, request: head(request, 300), ...(note ? { note } : {}), ...(instructions ? { instructions: head(instructions, 300) } : {}) } + const rec = { at: now(), by: who, source, decision: p.decision, request: head(request, 300), ...(note ? { note } : {}), ...(instructions ? { instructions: head(instructions, 300) } : {}), ...(operatorOk && req ? { request_id: req.id } : {}) } const logApproval = () => this.event(t, "approval", { decision: p.decision, by: who, source, handoff_state: h?.state || null, resumed: !(p.decision === "reject" && !instructions) }) if (p.decision === "reject" && !instructions) { logApproval() - return this.closeTask(t, who, `approval rejected by ${who}${note ? `: ${note}` : ""}`, rec, source) + const res = this.closeTask(t, who, `approval rejected by ${who}${note ? `: ${note}` : ""}`, rec, source) + if (operatorOk && t.operator_gate) { + // The operator's own answer: the task is closed and no longer waits for the operator. + delete t.operator_gate + this.save(t) + this.event(t, "operator_gate_cleared", { by: who, source }) + } + return res } const msg = p.decision === "approve" ? `APPROVED by ${who}: ${request}${note ? `\nNote: ${note}` : ""}\n\n${instructions || "Proceed with the approved action and finish the task."}\n\n(Approval does not change the sandbox: actions it blocks stay blocked. If the approved action needs network or an install, the operator has done it outside, or you must report that under concerns.)` : `DENIED by ${who}: ${request}. Do not do that.${note ? `\nNote: ${note}` : ""}\n\n${instructions}` - const r = await this.continueTask(t.id, msg, p.timeout_minutes, p.profile, { approval: rec, operatorOk: operatorOk || p.decision === "reject" }) + const r = await this.continueTask(t.id, msg, p.timeout_minutes, p.profile, { approval: rec, operatorOk, caller }) logApproval() return { ...r, decision: p.decision, approval: rec } } - // approvals.require_operator: the supervisor's approve only records the request; the operator confirms. - recordApprovalRequest(t, h, { who, instructions, note, profile, timeout_minutes }) { + // approvals.require_operator: the supervisor's approve only records the request; the operator confirms + // it by id (a changed request gets a new id, so the operator never approves text it did not see). + recordApprovalRequest(t, h, { who, source, instructions, note, profile, timeout_minutes }) { const at = now() - t.approval_request = redact({ at, by: who, ...(instructions ? { instructions } : {}), ...(note ? { note } : {}), ...(profile ? { profile } : {}), ...(timeout_minutes ? { timeout_minutes } : {}) }) + const body = { decision: "approve", by: who, ...(instructions ? { instructions } : {}), ...(note ? { note } : {}), ...(profile ? { profile } : {}), ...(timeout_minutes ? { timeout_minutes } : {}) } + const id = crypto.createHash("sha256").update(JSON.stringify(body) + at + crypto.randomBytes(8).toString("hex")).digest("hex").slice(0, 12) + t.approval_request = redact({ id, at, source, ...body }) + this.setOperatorGate(t, daemonConfig(), "approval requested") const nh = { ...h } nh.owner = "human (operator)" - nh.next_action = `Approval requested by ${who}; waiting for the operator. Operator: review the request, then run \`sudo workhorse approve ${t.id}\` (uses the operator token) or \`workhorse reject ${t.id}\`.` - nh.history = [...(h?.history || []), { at, by: who, event: "approval_requested" }].slice(-20) + nh.next_action = `Approval requested by ${who} (request ${id}); waiting for the operator. Operator: run \`sudo workhorse approve ${t.id}\` (shows the request, then confirms it with the operator token) or \`workhorse reject ${t.id}\`.` + nh.history = [...(h?.history || []), { at, by: who, source, event: "approval_requested", request_id: id }].slice(-20) nh.derived = false nh.updated_at = at nh.updated_by = who t.handoff = redact(nh) this.save(t) - this.event(t, "approval_requested", { by: who }) - return { task_id: t.id, status: t.status, approval_requested: true, waiting_for: "operator", message: "Approval recorded as a request. approvals.require_operator is on: a human must confirm it on the host with `sudo workhorse approve `. Nothing runs until then.", handoff: handoffSummary(this.handoffOf(t)) } + this.event(t, "approval_requested", { by: who, source, request_id: id }) + return { task_id: t.id, status: t.status, approval_requested: true, request_id: id, waiting_for: "operator", message: "Approval recorded as a request. approvals.require_operator is on: a human must confirm it on the host with `sudo workhorse approve `. Nothing runs until then.", handoff: handoffSummary(this.handoffOf(t)) } } closeTask(t, who, reason, rec = null, source = null) { const from = t.status const h = { ...this.handoffOf(t) } + // A parked task keeps its operator gate when it is closed (also one parked before + // require_operator was turned on), so a later continue_task still needs the operator. + if (PARKED.has(from)) this.setOperatorGate(t, daemonConfig(), "closed while parked") t.status = "cancelled" delete t.parked_from if (t.result) t.result = { ...t.result, status: "cancelled" } @@ -869,7 +952,9 @@ export class Manager { h.owner = "supervisor" h.next_action = t.worktree_removed ? `Closed (${reason}). The worktree is gone; the diff is archived at ${path.join(this.taskDir(t.id), "diff.patch")}. Delegate a new task if the work is still needed.` - : `Closed (${reason}). The worktree is kept: merge any useful partial diff, continue_task to resume, or cleanup_task (discard_unmerged_changes=true) to drop it.` + : t.operator_gate + ? `Closed (${reason}). The worktree is kept. Only the operator can resume it (approvals.require_operator: \`sudo workhorse approve ${t.id}\`); otherwise merge any useful partial diff or cleanup_task (discard_unmerged_changes=true).` + : `Closed (${reason}). The worktree is kept: merge any useful partial diff, continue_task to resume, or cleanup_task (discard_unmerged_changes=true) to drop it.` h.resume = { ...h.resume, tool: null, args: null } h.history = [...(h.history || []), { at, by: who, ...(source ? { source } : {}), event: "closed", reason, from_status: from }].slice(-20) h.derived = false @@ -1293,7 +1378,9 @@ export class Manager { const message = this.buildMessage(t, { ...spec, backend: profile.backend }) + (resumeSession ? "" : saverFragments(savers, t.mode)) // RTK is applied by the guard plugin, i.e. only on the Kilo/OpenCode backends. const rtkCapable = profile.backend === "kilo" || profile.backend === "opencode" - const rtk = rtkCapable ? rtkBin(savers, cfg.env_path) : null + let rtk = rtkCapable ? rtkBin(savers, cfg.env_path) : null + // Resolve symlinks so the sandbox can bind back exactly this one file (not its directory). + if (rtk) { try { rtk = fs.realpathSync(rtk) } catch { rtk = null } } if (savers.rtk.enabled && rtkCapable && !rtk) this.activity(t, "token_savers.rtk is enabled but the rtk binary was not found (set token_savers.rtk.bin); running without it") const ctx = { t, profile, pc, cfg, bc, repo: this.repoSafe(t), agent, message, resumeSession, @@ -1314,7 +1401,7 @@ export class Manager { let argv = args if (sandboxed) { const extra = backend.sandbox(ctx) || {} - if (rtk) extra.roBinds = [...(extra.roBinds || []), path.dirname(fs.realpathSync(rtk))] + if (rtk) extra.roBinds = [...(extra.roBinds || []), rtk] // the rtk binary only argv = [...(await this.workerSandboxArgs(t, cfg, extra)), "--", bc.bin_real || bc.bin, ...args] bin = cfg.bwrap } @@ -1688,6 +1775,7 @@ export class Manager { const finishedAt = now() status = finalStatus t.status = status + if (PARKED.has(status)) this.setOperatorGate(t, cfg, "worker asked for approval") t.finished_at = finishedAt t.phase = "finished" t.result = redact({ @@ -1781,11 +1869,15 @@ export class Manager { const a = t.auto if (!a || t.mode !== "implement" || !["completed", "failed"].includes(t.status)) return {} const trail = t.auto_trail || [] + // Budgets cover the whole task: every run (also before a continue_task) plus its auto-review tasks. + // They are checked between runs, so one run can overshoot them. const tk = t.stats.tokens - const used = tk.input + tk.output + tk.reasoning - const cost = this.estCost(t) + const rvu = t.auto_review_usage || { tokens: 0, cost_usd: 0 } + const used = tk.input + tk.output + tk.reasoning + (rvu.tokens || 0) + const own = this.estCost(t) + const cost = own === null && !rvu.cost_usd ? null : (own || 0) + (rvu.cost_usd || 0) const autoRuns = trail.filter((e) => e.next).length - const budget = a.max_tokens && used >= a.max_tokens ? "max_tokens" : a.max_cost_usd && cost !== null && cost >= a.max_cost_usd ? "max_cost_usd" : null + const budget = a.max_tokens != null && used >= a.max_tokens ? "max_tokens" : a.max_cost_usd != null && (a.max_cost_usd === 0 || (cost !== null && cost >= a.max_cost_usd)) ? "max_cost_usd" : null const trigger = this.autoTrigger(t) const cur = t.runs[t.runs.length - 1]?.profile || t.profile if (trigger) { @@ -1884,7 +1976,7 @@ export class Manager { t.phase = `automatic advisory review by profile ${profileName}` this.save(t) const timeout = Math.min(cfg.timeouts.max_min, Math.max(cfg.timeouts.min_min, Number(t.auto?.review_timeout_min) || 10)) - const task = `Automatic advisory review of task ${t.id} (cheap first-pass reviewer; the supervisor decides). The task was:\n\n${head(t.task, 4000)}\n\nCheck the diff for correctness bugs, missing tests for new behaviour and changes outside the task's scope. Be brief: at most 5 findings, most severe first. Start summary with "approve" or "request changes".` + const task = `Automatic advisory review of task ${t.id} (cheap first-pass reviewer; the supervisor decides). The task was:\n\n${head(t.task, 4000)}\n\nCheck the diff for correctness bugs, missing tests for new behaviour and changes outside the task's scope. Be brief: at most 5 findings, most severe first. Start the summary's first line with exactly "approve" or "request changes" (nothing before it).` try { const r = await this.delegate({ repo: t.repo, mode: "review", review_task_id: t.id, profile: profileName, task, timeout_minutes: timeout }, { autoReviewOf: t.id }) if (t.status === "reviewing" && t.review_pending) { @@ -1912,6 +2004,13 @@ export class Manager { tokens: (tk.input || 0) + (tk.output || 0) + (tk.reasoning || 0), est_cost_usd: cr.usage?.estimated_list_cost_usd ?? null, } } + if (child?.stats?.tokens) { + const ck = child.stats.tokens + const u = (t.auto_review_usage ||= { tokens: 0, cost_usd: 0, reviews: 0 }) + u.tokens += (ck.input || 0) + (ck.output || 0) + (ck.reasoning || 0) + u.cost_usd = Math.round((u.cost_usd + (this.estCost(child) || 0)) * 1e6) / 1e6 + u.reviews++ + } t.result = redact({ ...t.result, review }) t.status = pend.final_status || "completed" delete t.review_pending @@ -1946,11 +2045,18 @@ export class Manager { const DAEMON_CONCERN_RE = /^(daemon-run tests failed|INTEGRITY|worker reported status|\d+ tool call\(s\) were blocked|review agent modified files|worker did not end with|fallback profile was used)/ -// Advisory review verdict from the reviewer's summary. +// Advisory review verdict from the reviewer's summary. The reviewer is told to start with "approve" or +// "request changes", so only the start of the first line counts (words later in the text, such as +// "nothing to reject", must not flip it). Negated approvals ("not approved", "cannot approve", "do not +// approve") are request_changes; anything else is "unclear". export function reviewVerdict(summary) { - const s = String(summary || "").toLowerCase() - if (/request(?:s|ed|ing)?[\s_-]+changes|changes[\s_-]+(?:requested|required|needed)|\breject/.test(s)) return "request_changes" - if (/\bapprove[sd]?\b|\blgtm\b/.test(s)) return "approve" + let s = String(summary || "").trim().split(/\r?\n/).find((l) => l.trim()) || "" + s = s.toLowerCase().replace(/[*_`#>"']/g, "").trim() + s = s.replace(/^(?:(?:review|verdict|decision|summary|result|recommendation)\s*[:=-]\s*)+/, "").trim() + const NEG = /^(?:not|no|never|cannot|can ?not|can't|cant|do not|don't|dont|does not|doesn't|will not|won't|would not|wouldn't|unable to|refuse to)\b/ + if (NEG.test(s)) return /^[^.;:!?]{0,40}\b(?:approv\w*|lgtm|accept\w*|merge\w*)/.test(s) ? "request_changes" : "unclear" + if (/^(?:request(?:s|ed|ing)?[\s_-]+changes|changes[\s_-]+(?:requested|required|needed)|needs?[\s_-]+(?:changes|work)|reject(?:s|ed)?)\b/.test(s)) return "request_changes" + if (/^(?:approve[sd]?|approval|lgtm|looks good(?: to me)?)\b/.test(s)) return "approve" return "unclear" } diff --git a/lib/usage.mjs b/lib/usage.mjs index c657126..4d9c34a 100644 --- a/lib/usage.mjs +++ b/lib/usage.mjs @@ -1,17 +1,17 @@ // usage_report / `workhorse stats`: worker token use and cost by profile and day, plus an ESTIMATE of // the supervisor (e.g. Grok) tokens the delegation avoided. See docs/token-savings.md#usage-report. // -// ESTIMATE, per task whose final verdict is success or success_untested (work the supervisor did not -// have to redo): -// worker_work_tokens = worker input + output + reasoning tokens (cache reads excluded) -// supervisor_overhead = (chars the supervisor sent to workhorse for this task -// + chars workhorse returned to it) / chars_per_token -// supervisor_tokens_avoided = max(0, worker_work_tokens - supervisor_overhead) -// The assumption is that the supervisor would have needed about as many tokens as the worker to do the -// same work itself; a stronger model may need fewer turns (so this over-estimates) and delegation also -// saves the supervisor's own context growth (not counted, so it under-estimates). Tasks that did not -// succeed count only as overhead. With daemon.json supervisor.price_per_mtok {input, output} the -// estimate is also converted to USD (net of the workers' estimated list cost). +// Conservative ESTIMATE: +// worker_output_tokens = output + reasoning tokens of tasks whose final verdict is success or +// success_untested (what the supervisor would have had to generate itself) +// supervisor_overhead = (chars the supervisor sent to workhorse + chars it read back, for every +// task in the window) / chars_per_token +// est_supervisor_tokens_avoided = max(0, worker_output_tokens - supervisor_overhead) +// Worker INPUT tokens are not counted at all: most of them are the same context re-read on every turn, +// and a supervisor doing the work would re-read differently. Failed tasks count only as overhead. +// With daemon.json supervisor.price_per_mtok {input, output}: +// est_net_usd = worker_output_tokens * price.output - overhead_read * price.input +// - overhead_written * price.output - workers' estimated list cost (may be negative) const KEYS = ["input", "output", "reasoning", "cache_read", "cache_write"] const zero = () => Object.fromEntries(KEYS.map((k) => [k, 0])) const addTo = (a, b) => { @@ -50,7 +50,6 @@ export function usageReport(tasks, { days = 30, profile = null, repo = null, pro const byKey = new Map() const byProfile = new Map() const sup = { calls: 0, request_chars: 0, response_chars: 0 } - let workTokens = 0 let workIn = 0 let workOut = 0 let succeeded = 0 @@ -113,12 +112,11 @@ export function usageReport(tasks, { days = 30, profile = null, repo = null, pro succeeded++ workIn += taskTok.input workOut += taskTok.output + taskTok.reasoning - workTokens += taskTok.input + taskTok.output + taskTok.reasoning } } const fin = (o) => ({ ...o, backend_reported_cost_usd: round(o.backend_reported_cost_usd), est_list_cost_usd: round(o.est_list_cost_usd) }) - const avoided = Math.max(0, Math.round(workTokens - overheadTokens)) - const avoidedUsd = sp ? round(((workIn - overheadIn) * (sp.input || 0) + (workOut - overheadOut) * (sp.output || 0)) / 1e6 - workerCost, 4) : null + const avoided = Math.max(0, Math.round(workOut - overheadTokens)) + const avoidedUsd = sp ? round((workOut * (sp.output || 0) - overheadIn * (sp.input || 0) - overheadOut * (sp.output || 0)) / 1e6 - workerCost, 4) : null return { window_days: d, tasks: tasksCounted, @@ -127,14 +125,15 @@ export function usageReport(tasks, { days = 30, profile = null, repo = null, pro supervisor_estimate: { label: "ESTIMATE", successful_tasks: succeeded, - worker_work_tokens: workTokens, + worker_output_tokens: workOut, + worker_input_tokens_not_counted: workIn, supervisor_io: { ...sup, est_tokens: Math.round(overheadTokens), chars_per_token: cpt }, est_supervisor_tokens_avoided: avoided, - est_supervisor_cost_avoided_usd: avoidedUsd, + est_net_usd: avoidedUsd, est_worker_list_cost_usd: round(workerCost, 4), ...(unattributed ? { unattributed_supervisor_io_since_daemon_start: unattributed } : {}), - formula: "per successful task: worker (input+output+reasoning) tokens - (supervisor request chars + response chars)/chars_per_token; failed tasks count only as overhead; cache reads excluded" + (sp ? "; USD = avoided input*price.input + avoided output*price.output - worker est. list cost" : "; set daemon.json supervisor.price_per_mtok {input, output} for a USD estimate"), - caveat: "Assumes the supervisor would need about as many tokens as the worker for the same work. Not a measurement.", + formula: "max(0, worker output+reasoning tokens of successful tasks - (supervisor request chars + response chars of all tasks)/chars_per_token); worker input tokens are not counted" + (sp ? "; est_net_usd = worker output tokens*price.output - supervisor read tokens*price.input - supervisor written tokens*price.output - workers' est. list cost (can be negative)" : "; set daemon.json supervisor.price_per_mtok {input, output} for a USD estimate"), + caveat: "Conservative planning aid, not a measurement: assumes the supervisor would have generated about as many output tokens as the workers did for work that succeeded, and ignores the input tokens it would have read.", }, } } diff --git a/lib/views.mjs b/lib/views.mjs index a946ad1..cf1eee9 100644 --- a/lib/views.mjs +++ b/lib/views.mjs @@ -1,7 +1,6 @@ // Compact views of task results, to keep what the supervisor model reads (and pays for) small. // -// full (default, backward compatible): the whole result; the handoff's `context` no longer repeats -// fields that are already top-level in the same response (summary, concerns, diffstat, ...). +// full (default, backward compatible): the whole result with the complete handoff record. // brief: only what the supervisor needs to decide the next step (~0.5-1 KB): verdict, a short // summary, changed files, test outcome, top concerns, the handoff's next action + resume // call, auto-review verdict, automatic follow-up trail and token totals. @@ -9,17 +8,6 @@ import { head } from "./util.mjs" export const VIEWS = ["full", "brief"] -// Handoff context keys that duplicate top-level task_result fields. -const DUP_CONTEXT = ["verdict", "status", "summary", "remaining_concerns", "diffstat", "files_changed", "diff_path", "diff"] - -export function dedupeHandoff(h) { - if (!h || !h.context) return h - const context = { ...h.context } - for (const k of DUP_CONTEXT) delete context[k] - for (const [k, v] of Object.entries(context)) if (v === null || v === undefined) delete context[k] - return { ...h, context } -} - const tokensOf = (u) => { const t = u?.tokens || {} return { input: t.input || 0, output: t.output || 0, reasoning: t.reasoning || 0, cache_read: t.cache_read || 0 } diff --git a/test/stub-e2e.test.mjs b/test/stub-e2e.test.mjs index af9cc5f..898161c 100644 --- a/test/stub-e2e.test.mjs +++ b/test/stub-e2e.test.mjs @@ -7,7 +7,7 @@ import assert from "node:assert/strict" import fs from "node:fs" import path from "node:path" import crypto from "node:crypto" -import { setupStubEnv, stest, startDaemon, sleep, REPO_NAME, STUB_TEST_CMD, stubMessages, editJson, STUB_SKIP } from "./helpers.mjs" +import { setupStubEnv, stest, startDaemon, stopDaemon, sleep, REPO_NAME, STUB_TEST_CMD, stubMessages, editJson, STUB_SKIP } from "./helpers.mjs" const env = await setupStubEnv("e2e") let rpc @@ -20,8 +20,7 @@ before(async () => { daemon = await startDaemon(env, DAEMON_ENV) }) after(async () => { - if (daemon) try { process.kill(-daemon.pid, "SIGTERM") } catch {} - await sleep(300) + await stopDaemon(daemon) }) const task = (scenario, extra = "") => `MOCK_SCENARIO=${scenario} Implement multiply and divide in calc/core.py. ${extra}` @@ -50,7 +49,7 @@ stest("happy path: wait_task long-poll returns a compact brief result", async () assert.ok(JSON.stringify(r).length < 1500, `brief result is ${JSON.stringify(r).length} chars`) const full = await rpc("task_result", { task_id: d.task_id }) assert.equal(full.verdict, "success") - assert.equal(full.handoff.context.summary, undefined, "handoff context no longer repeats top-level fields") + assert.equal(full.handoff.context.summary, full.summary, "the full view keeps the complete handoff (compatibility)") assert.ok(JSON.stringify(full).length > JSON.stringify(r).length * 2) const brief = await rpc("task_result", { task_id: d.task_id, view: "brief" }) assert.deepEqual(brief, r) @@ -206,7 +205,11 @@ stest("auto_review: a cheap advisory review is attached to the result", async () const child = await rpc("task_result", { task_id: r.review.task_id }) assert.equal(child.mode, "review") const list = await rpc("list_tasks", { limit: 200 }) - assert.ok(list.tasks.some((t) => t.task_id === r.review.task_id)) + assert.ok(!list.tasks.some((t) => t.task_id === r.review.task_id), "review children are hidden by default") + const all = await rpc("list_tasks", { limit: 200, include_auto_reviews: true }) + assert.equal(all.tasks.find((t) => t.task_id === r.review.task_id)?.auto_review_of, d.task_id) + const attention = await rpc("list_tasks", { limit: 200, status: "needs_attention", include_auto_reviews: true }) + assert.ok(!attention.tasks.some((t) => t.task_id === r.review.task_id), "a review child never needs attention itself") }) stest("require_operator: the supervisor's approve only records a request; the operator token confirms", async () => { @@ -217,6 +220,7 @@ stest("require_operator: the supervisor's approve only records a request; the op await waitDone(d.task_id) const req = await rpc("approve_task", { task_id: d.task_id, decision: "approve", instructions: "ok from supervisor" }) assert.equal(req.approval_requested, true) + assert.match(req.request_id, /^[0-9a-f]{12}$/) assert.equal((await rpc("task_status", { task_id: d.task_id })).status, "needs_approval") await assert.rejects(rpc("continue_task", { task_id: d.task_id, instructions: "do it anyway" }), /require_operator/) await assert.rejects(rpc("update_handoff", { task_id: d.task_id, state: "done" }), /only the operator/) @@ -224,8 +228,22 @@ stest("require_operator: the supervisor's approve only records a request; the op const brief = await rpc("task_result", { task_id: d.task_id, view: "brief" }) assert.equal(brief.approval_request.waiting_for, "operator") assert.match(brief.next.action, /sudo workhorse approve/) - const ok = await rpc("approve_task", { task_id: d.task_id, decision: "approve", operator_token: tok }, { caller: { client: "workhorse" } }) + const op = { caller: { client: "workhorse" } } + const shown = await rpc("get_handoff", { task_id: d.task_id }, op) + assert.equal(shown.approval_request.id, req.request_id) + assert.equal(shown.approval_request.instructions, "ok from supervisor") + assert.ok(shown.operator_gate) + await assert.rejects(rpc("approve_task", { task_id: d.task_id, decision: "approve", operator_token: tok }, op), /is pending: review it first/) + // The supervisor changes its request after the operator looked: the displayed id no longer matches. + const req2 = await rpc("approve_task", { task_id: d.task_id, decision: "approve", instructions: "and also delete the tests" }) + assert.notEqual(req2.request_id, req.request_id) + await assert.rejects(rpc("approve_task", { task_id: d.task_id, decision: "approve", operator_token: tok, request_id: req.request_id }, op), /request changed/) + await rpc("approve_task", { task_id: d.task_id, decision: "approve", instructions: "ok from supervisor" }) + const current = (await rpc("get_handoff", { task_id: d.task_id }, op)).approval_request + const ok = await rpc("approve_task", { task_id: d.task_id, decision: "approve", operator_token: tok, request_id: current.id }, op) assert.match(ok.approval.by, /^operator \(requested by supervisor\)/) + assert.deepEqual(ok.approval.source, { channel: "cli", auth: "operator_token" }) + assert.equal(ok.approval.request_id, current.id) const r = await waitDone(d.task_id) assert.equal(r.verdict, "success") assert.match(stubMessages(env, d.task_id)[1].message, /ok from supervisor/, "the recorded instructions are used") @@ -236,6 +254,53 @@ stest("require_operator: the supervisor's approve only records a request; the op } }) +stest("require_operator: closing, cancelling or rejecting with instructions cannot bypass the operator", async () => { + const tok = crypto.randomBytes(32).toString("hex") + editJson(cfgFile("daemon.json"), (j) => { j.approvals = { require_operator: true, operator_token_sha256: crypto.createHash("sha256").update(tok).digest("hex") } }) + const op = { caller: { client: "workhorse" } } + const parked = async () => { + const d = await delegate({ task: task("approval") }) + assert.equal((await waitDone(d.task_id)).status, "needs_approval") + return d.task_id + } + try { + // (a) update_handoff state=closed on the parked task, then continue_task + const a = await parked() + const closed = await rpc("update_handoff", { task_id: a, state: "closed", note: "closing it" }) + assert.equal(closed.status, "cancelled") + await assert.rejects(rpc("continue_task", { task_id: a, instructions: "do the approved thing" }), /waits for the operator/) + assert.ok((await rpc("get_handoff", { task_id: a })).operator_gate, "the gate survives closing") + // (b) cancel_task on the parked task, then continue_task + const b = await parked() + assert.equal((await rpc("cancel_task", { task_id: b })).status, "cancelled") + await assert.rejects(rpc("continue_task", { task_id: b, instructions: "do the approved thing" }), /waits for the operator/) + await assert.rejects(rpc("continue_task", { task_id: b, instructions: "x", operator_token: "0".repeat(64) }), /operator token rejected|unknown parameter/) + // (c) approve_task reject WITH instructions resumes the worker, so it needs the operator + const c = await parked() + await assert.rejects(rpc("approve_task", { task_id: c, decision: "reject", instructions: "instead, do this other thing" }), /only the operator/) + assert.equal((await rpc("task_status", { task_id: c })).status, "needs_approval") + // reject without instructions only closes it, and it stays gated + assert.equal((await rpc("approve_task", { task_id: c, decision: "reject" })).status, "cancelled") + await assert.rejects(rpc("continue_task", { task_id: c, instructions: "x" }), /waits for the operator/) + for (const id of [a, b, c]) assert.equal(stubMessages(env, id).length, 1, `task ${id}: the worker was never resumed`) + // The operator can still resume a closed gated task; that clears the gate. + const r = await rpc("approve_task", { task_id: a, decision: "approve", operator_token: tok, instructions: "operator says go" }, op) + assert.ok(["queued", "running"].includes(r.status)) + assert.equal((await waitDone(a)).verdict, "success") + assert.match(stubMessages(env, a)[1].message, /APPROVED by operator[\s\S]*operator says go/) + assert.equal((await rpc("get_handoff", { task_id: a })).operator_gate, undefined) + // The operator's reject with instructions is allowed. + const rr = await rpc("approve_task", { task_id: b, decision: "reject", operator_token: tok, instructions: "do not do it; just summarise" }, op) + assert.ok(["queued", "running"].includes(rr.status)) + await waitDone(b) + assert.match(stubMessages(env, b)[1].message, /DENIED by operator/) + const audit = fs.readFileSync(path.join(env.dataDir, "logs/audit.jsonl"), "utf8") + assert.ok(!audit.includes(tok)) + } finally { + editJson(cfgFile("daemon.json"), (j) => { delete j.approvals }) + } +}) + stest("per-profile stall_minutes stops a silent worker early", async () => { const d = await delegate({ task: task("slow"), profile: "sleepy" }) const r = await waitDone(d.task_id, "brief", 30000) @@ -270,8 +335,10 @@ stest("usage_report: tokens by profile/day and a labelled supervisor ESTIMATE", const se = u.supervisor_estimate assert.equal(se.label, "ESTIMATE") assert.ok(se.supervisor_io.calls > 10) - assert.ok(se.worker_work_tokens > 0) - assert.ok(se.est_supervisor_tokens_avoided > 0) + assert.ok(se.worker_output_tokens > 0) + // Conservative: only successful workers' output tokens count, minus everything the supervisor read and + // wrote; with the tiny stub outputs that is usually 0, never negative. + assert.equal(se.est_supervisor_tokens_avoided, Math.max(0, Math.round(se.worker_output_tokens - se.supervisor_io.est_tokens))) assert.match(se.formula, /chars_per_token/) const one = await rpc("usage_report", { profile: "strong", days: 1 }) assert.deepEqual(one.by_profile.map((p) => p.profile).sort(), ["mid", "strong"].filter((x) => one.by_profile.some((p) => p.profile === x)).sort()) @@ -280,8 +347,7 @@ stest("usage_report: tokens by profile/day and a labelled supervisor ESTIMATE", stest("restart: a killed daemon leaves an interrupted, retryable task that resumes", async () => { const d = await delegate({ task: task("slow") }) for (let i = 0; i < 50 && (await rpc("task_status", { task_id: d.task_id })).status !== "running"; i++) await sleep(100) - process.kill(-daemon.pid, "SIGKILL") - await sleep(300) + await stopDaemon(daemon, "SIGKILL") daemon = await startDaemon(env, DAEMON_ENV) const r = await waitDone(d.task_id) assert.equal(r.verdict, "interrupted") @@ -295,8 +361,7 @@ stest("restart: a killed daemon leaves an interrupted, retryable task that resum stest("restart: a graceful stop (SIGTERM) also leaves a finalized, retryable task", async () => { const d = await delegate({ task: task("slow") }) for (let i = 0; i < 50 && (await rpc("task_status", { task_id: d.task_id })).status !== "running"; i++) await sleep(100) - process.kill(-daemon.pid, "SIGTERM") - for (let i = 0; i < 100 && fs.existsSync(path.join(env.dataDir, "run/daemon.sock")); i++) await sleep(100) + await stopDaemon(daemon, "SIGTERM") daemon = await startDaemon(env, DAEMON_ENV) const r = await waitDone(d.task_id) assert.equal(r.verdict, "interrupted") diff --git a/test/token-savings.test.mjs b/test/token-savings.test.mjs index b9c7761..a9ae73e 100644 --- a/test/token-savings.test.mjs +++ b/test/token-savings.test.mjs @@ -16,7 +16,7 @@ process.env.WH_CONFIG_DIR = cfgDir process.env.WH_DATA_DIR = path.join(root, "data") const APP = path.resolve(path.dirname(new URL(import.meta.url).pathname), "..") -const { briefResult, dedupeHandoff } = await import("../lib/views.mjs") +const { briefResult } = await import("../lib/views.mjs") const { usageReport, dayOf } = await import("../lib/usage.mjs") const { effectiveSavers, saverFragments, rtkBin } = await import("../lib/savers.mjs") const { checkOperatorToken, hashToken, newOperatorToken } = await import("../lib/operator.mjs") @@ -54,12 +54,6 @@ test("briefResult keeps only decision data, capped", () => { assert.ok(JSON.stringify(b).length < JSON.stringify({ ...r, handoff: h }).length / 2) }) -test("dedupeHandoff drops context fields duplicated at the top level", () => { - const h = { state: "done", context: { verdict: "success", summary: "s", remaining_concerns: [], diffstat: "d", files_changed: 1, diff_path: "p", diff: "hint", worker_status: "done", worker_request: null, previous_attempts: 0 } } - assert.deepEqual(dedupeHandoff(h).context, { worker_status: "done", previous_attempts: 0 }) - assert.equal(h.context.summary, "s", "input not mutated") -}) - test("usageReport groups by profile/day and computes the labelled estimate", () => { const now = Date.parse("2026-09-26T12:00:00Z") const tasks = [ @@ -83,12 +77,14 @@ test("usageReport groups by profile/day and computes the labelled estimate", () const se = u.supervisor_estimate assert.equal(se.label, "ESTIMATE") assert.equal(se.successful_tasks, 1) - assert.equal(se.worker_work_tokens, 10000 + 1000 + 20000 + 2000 + 1000) + // conservative: only the successful workers' output+reasoning tokens count; input is not counted + assert.equal(se.worker_output_tokens, 1000 + 2000 + 1000) + assert.equal(se.worker_input_tokens_not_counted, 10000 + 20000) assert.equal(se.supervisor_io.est_tokens, (400 + 1600 + 200 + 200) / 4) - assert.equal(se.est_supervisor_tokens_avoided, 34000 - 600) + assert.equal(se.est_supervisor_tokens_avoided, 4000 - 600) const workerCost = ((10000 + 50000) * 0.1 + 1000 * 0.4) / 1e6 + (20000 * 3 + 3000 * 15) / 1e6 + (500 * 0.1 + 50 * 0.4) / 1e6 - const avoided = ((30000 - 450) * 3 + (4000 - 150) * 15) / 1e6 - workerCost - assert.ok(Math.abs(se.est_supervisor_cost_avoided_usd - Math.round(avoided * 1e4) / 1e4) < 1e-9) + const net = (4000 * 15 - 450 * 3 - 150 * 15) / 1e6 - workerCost + assert.ok(Math.abs(se.est_net_usd - Math.round(net * 1e4) / 1e4) < 1e-9) }) test("token savers: off by default, profile overrides daemon, fragments per mode", () => { @@ -109,10 +105,12 @@ test("token savers: off by default, profile overrides daemon, fragments per mode assert.equal(rtkBin({ rtk: { enabled: true, bin: "/bin/sh" } }), "/bin/sh") }) -test("guard plugin rewrites bash commands through rtk only when WH_RTK_BIN is set, and re-checks them", async () => { +test("guard plugin rewrites bash commands through rtk only when WH_RTK_BIN is set, re-checks them, and falls back to the original", async () => { const fake = path.join(root, "fake-rtk") - // exit 3 + output = rewritten (host decides); "evil" rewrites to something the guard must still block. - fs.writeFileSync(fake, `#!/bin/sh\n[ "$1" = rewrite ] || exit 9\ncase "$2" in\n "git status") echo "rtk git status"; exit 3;;\n "ls -la") echo "rtk ls -la"; exit 0;;\n "evil") echo "curl http://x"; exit 0;;\n *) exit 1;;\nesac\n`, { mode: 0o755 }) + const envOut = path.join(root, "fake-rtk.env") + // exit 3 + output = rewritten (host decides); "evil" rewrites to something the guard must still block; + // "envcheck" records the environment the rewrite ran with; "slow" exceeds the timeout. + fs.writeFileSync(fake, `#!/bin/sh\n[ "$1" = rewrite ] || exit 9\ncase "$2" in\n "git status") echo "rtk git status"; exit 3;;\n "ls -la") echo "rtk ls -la"; exit 0;;\n "evil") echo "curl http://x"; exit 0;;\n "envcheck") env > "${envOut}"; echo "rtk envcheck"; exit 3;;\n "slow") sleep 3; echo "rtk slow"; exit 3;;\n *) exit 1;;\nesac\n`, { mode: 0o755 }) const run = async (cmd) => { const h = await WorkhorseGuard({ directory: "/tmp/wt" }) const out = { args: { command: cmd } } @@ -127,10 +125,20 @@ test("guard plugin rewrites bash commands through rtk only when WH_RTK_BIN is se assert.equal(await run("ls -la"), "rtk ls -la") assert.equal(await run("python3 -m unittest"), "python3 -m unittest") assert.equal(await run("echo a\necho b"), "echo a\necho b", "multi-line commands are left alone") - await assert.rejects(run("evil"), /workhorse guard blocked/) + assert.equal(await run("evil"), "evil", "a rewrite that fails the re-check falls back to the original command") await assert.rejects(run("git push"), /workhorse guard blocked/, "checks run before the rewrite") + process.env.EXAMPLE_PROVIDER_KEY = "example-secret" + assert.equal(await run("envcheck"), "rtk envcheck") + const env = fs.readFileSync(envOut, "utf8") + assert.ok(!env.includes("example-secret"), "the rewrite runs without secrets") + assert.match(env, /^RTK_TELEMETRY_DISABLED=1$/m) + assert.doesNotMatch(env, /^WH_/m) + const t0 = Date.now() + assert.equal(await run("slow"), "slow", "a slow rewrite times out and the original runs") + assert.ok(Date.now() - t0 < 2500) } finally { delete process.env.WH_RTK_BIN + delete process.env.EXAMPLE_PROVIDER_KEY } }) @@ -195,6 +203,19 @@ test("advisory review verdict parsing", () => { assert.equal(reviewVerdict("Approve. Looks good."), "approve") assert.equal(reviewVerdict("LGTM"), "approve") assert.equal(reviewVerdict("hmm"), "unclear") + // only the start of the first line counts; negations are handled + assert.equal(reviewVerdict("Not approved: the divide change has no test"), "request_changes") + assert.equal(reviewVerdict("cannot approve this yet"), "request_changes") + assert.equal(reviewVerdict("Do not approve - breaks the API"), "request_changes") + assert.equal(reviewVerdict("Don't merge: missing validation"), "request_changes") + assert.equal(reviewVerdict("approve - nothing to reject"), "approve") + assert.equal(reviewVerdict("Approved with minor nits; no need to request changes"), "approve") + assert.equal(reviewVerdict("**Verdict: approve**\nDetails follow"), "approve") + assert.equal(reviewVerdict("Review: request changes"), "request_changes") + assert.equal(reviewVerdict("\n\nRejected: scope creep"), "request_changes") + assert.equal(reviewVerdict("The change looks fine, I would approve"), "unclear") + assert.equal(reviewVerdict("No issues found"), "unclear") + assert.equal(reviewVerdict(""), "unclear") }) test("auto-review request_changes turns a done handoff into needs_review with a continue_task call", () => { From f112b1cfaf6c25bd3595cf83823e3b9f57a710fa Mon Sep 17 00:00:00 2001 From: mrchatam <287639636+mrchatam@users.noreply.github.com> Date: Sun, 27 Sep 2026 01:31:25 +0330 Subject: [PATCH 6/7] Fix install.sh --with-opencode dangling link: resolve the target from package.json bin opencode-ai@1.18.32 maps opencode to ./bin/opencode.exe; scripts/pin-bin.mjs reads the bin field and install.sh refuses a dangling link. CI: npm ci --ignore-scripts on each pin and assert the target exists. --- .github/workflows/ci.yml | 26 ++++++++++++++++++++++++++ scripts/install.sh | 5 ++++- scripts/pin-bin.mjs | 21 +++++++++++++++++++++ 3 files changed, 51 insertions(+), 1 deletion(-) create mode 100644 scripts/pin-bin.mjs diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 769c067..411e1c6 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -31,3 +31,29 @@ jobs: WH_SKIP_INTEGRATION: "1" WH_SKIP_STUB: "1" run: node --test --test-timeout=600000 test/*.test.mjs + + pins: + name: Installer pins resolve to an existing binary + runs-on: ubuntu-latest + timeout-minutes: 15 + steps: + - uses: actions/checkout@v7 + - uses: actions/setup-node@v7 + with: + node-version: 22 + - name: npm ci --ignore-scripts on each scripts/pins/* lockfile and check the install.sh link target + # scripts/pins/-cli pins one package whose executable is (kilo-cli -> kilo, + # opencode-cli -> opencode); the target comes from the package's package.json "bin" field, + # exactly as scripts/install.sh resolves it. + run: | + set -euo pipefail + for pin in scripts/pins/*/; do + name="$(basename "$pin")"; exe="${name%-cli}" + pkg="$(node -p "Object.keys(require('./${pin}package.json').dependencies)[0]")" + tmp="$(mktemp -d)"; cp "$pin/package.json" "$pin/package-lock.json" "$tmp/" + (cd "$tmp" && npm ci --ignore-scripts --no-audit --no-fund --loglevel=error) + rel="$(node scripts/pin-bin.mjs "$tmp" "$pkg" "$exe")" + mkdir -p "$tmp/bin" && ln -sfn "../node_modules/$pkg/$rel" "$tmp/bin/$exe" + test -e "$tmp/bin/$exe" || { echo "::error::$name: bin/$exe -> $rel is dangling"; exit 1; } + echo "$name: $pkg bin $exe -> node_modules/$pkg/$rel (exists)" + done diff --git a/scripts/install.sh b/scripts/install.sh index 36dedb5..ed000b7 100755 --- a/scripts/install.sh +++ b/scripts/install.sh @@ -154,7 +154,10 @@ pinned_install() { # rm -rf "$dir"; mkdir -p "$dir/bin" cp "$pin/package.json" "$pin/package-lock.json" "$dir/" (cd "$dir" && "$NPM" ci --no-audit --no-fund --loglevel=error >/dev/null) || return 1 - ln -sfn "../node_modules/$pkg/bin/$exe" "$dir/bin/$exe" + # The link target comes from the package's own package.json "bin" field (not bin/). + local rel; rel="$("$NODE" "$SRC/scripts/pin-bin.mjs" "$dir" "$pkg" "$exe")" || return 1 + ln -sfn "../node_modules/$pkg/$rel" "$dir/bin/$exe" + [ -e "$dir/bin/$exe" ] || { warn "$dir/bin/$exe -> ../node_modules/$pkg/$rel is a dangling link"; return 1; } ok "installed $pkg@$ver into $dir from the committed lockfile (integrity-checked)" else "$NPM" install -g --prefix "$dir" "$pkg@$ver" --no-audit --no-fund --loglevel=error >/dev/null || return 1 diff --git a/scripts/pin-bin.mjs b/scripts/pin-bin.mjs new file mode 100644 index 0000000..1ae5531 --- /dev/null +++ b/scripts/pin-bin.mjs @@ -0,0 +1,21 @@ +#!/usr/bin/env node +// Usage: node scripts/pin-bin.mjs +// Prints the path of relative to /node_modules/, taken from the package's +// own package.json "bin" field (e.g. opencode-ai@1.18.32 maps "opencode" to ./bin/opencode.exe), so +// install.sh links to the file npm actually installed instead of guessing bin/. +// Exits 1 if the package has no such bin entry, the entry escapes the package dir, or the file is missing. +import fs from "node:fs" +import path from "node:path" + +const [dir, pkg, exe] = process.argv.slice(2) +const fail = (msg) => { process.stderr.write(`pin-bin: ${msg}\n`); process.exit(1) } +if (!dir || !pkg || !exe) fail("usage: pin-bin.mjs ") +const pkgDir = path.resolve(dir, "node_modules", pkg) +let meta +try { meta = JSON.parse(fs.readFileSync(path.join(pkgDir, "package.json"), "utf8")) } catch (e) { fail(`cannot read ${pkg}/package.json: ${e.message}`) } +const bin = typeof meta.bin === "string" ? (exe === path.basename(meta.name || "") ? meta.bin : null) : meta.bin?.[exe] +if (typeof bin !== "string" || !bin) fail(`${pkg} has no "${exe}" entry in its package.json bin field`) +const rel = path.posix.normalize(bin.replace(/\\/g, "/")).replace(/^\.\//, "") +if (rel.startsWith("../") || path.isAbsolute(rel)) fail(`${pkg} bin "${bin}" points outside the package`) +if (!fs.existsSync(path.join(pkgDir, rel))) fail(`${pkg} bin target ${rel} does not exist`) +process.stdout.write(rel + "\n") From b18f9e01e7e9f70335b54cd2089d5a99e9c93ad5 Mon Sep 17 00:00:00 2001 From: mrchatam <287639636+mrchatam@users.noreply.github.com> Date: Sun, 27 Sep 2026 01:33:09 +0330 Subject: [PATCH 7/7] Review follow-ups: README CI badge and measured-savings section, CHANGELOG, recovery and budget tests - README: CI status badge; 'Measured savings' table, every figure labelled an estimate with how it was measured, linking docs/token-savings.md - CHANGELOG 0.3.0: operator gate and request ids, review verdict parsing, budgets (0 = explicit zero, reviews counted, per-run overshoot, continue semantics), rtk hardening, installer bin resolution, auto-review list/recovery, full view unchanged, conservative usage estimate - stub e2e: restart recovery re-links a review child and cancels an orphan review - unit: auto.max_tokens / max_cost_usd 0 kept as explicit zero budgets, invalid values reported --- CHANGELOG.md | 44 +++++++++++++++++++++++++++---------- README.md | 21 +++++++++++++++++- docs/architecture.md | 2 +- docs/configuration.md | 2 +- docs/token-savings.md | 2 +- test/stub-e2e.test.mjs | 34 ++++++++++++++++++++++++++++ test/token-savings.test.mjs | 11 ++++++++++ 7 files changed, 100 insertions(+), 16 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index dc3df5e..d8871d0 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -26,21 +26,35 @@ act, and adds opt-in savers for the workers. Builds on 0.2.0. See [docs/token-sa - **Automatic follow-ups** (off by default; per call, per preset or `profiles.json` `auto`): `auto_fix_rounds` (same-session fix rounds after failing tests or a missing/partial RESULT, max 3), `escalate` along each profile's new `escalate_to` chain (fresh session, same worktree), hard caps - `max_auto_runs` (default 3, cap 6), `max_tokens`, `max_cost_usd`. The trail is in `result.auto`. + `max_auto_runs` (default 3, cap 6, total per round), `max_tokens`, `max_cost_usd` (both include the + task's auto-review tasks; `0` = explicit zero budget (no follow-ups or reviews), not unlimited; checked between runs, so one run + can overshoot). `continue_task` restarts the run counters but not the token/cost budgets. The trail + is in `result.auto`. - **`auto_review`**: an advisory read-only review on a (cheap) profile after a successful run; new - transient status `reviewing`. `request_changes` makes the handoff `needs_review`. + transient status `reviewing`. `request_changes` makes the handoff `needs_review`. The verdict is read + from the first line of the reviewer's summary (negations such as "not approved" count as + `request_changes`; anything else is `unclear`). Review tasks are hidden from `list_tasks` unless + `include_auto_reviews` is set and never need attention themselves; the daemon recovers or cancels an + orphaned review after a crash. - **`usage_report`** (MCP, RPC) and **`workhorse stats`**: worker tokens and estimated list cost by - profile and day, plus a clearly labelled ESTIMATE of supervisor tokens avoided (formula documented; - optional `daemon.json supervisor.price_per_mtok` for USD). The daemon now records per-run tokens and + profile and day, plus a clearly labelled, conservative ESTIMATE of supervisor tokens avoided (only + successful workers' output tokens minus the supervisor's own I/O; worker input is not counted; + formula documented; optional `daemon.json supervisor.price_per_mtok` for a net USD figure). The daemon now records per-run tokens and cost, and the characters each supervisor call sent and received. - **Worker token savers** (`token_savers` in daemon.json, per-profile override; all off by default): `terse` and `minimal_code` instruction fragments (`lite`/`full`, our own wording inspired by Caveman and Ponytail) and `rtk` (Kilo/OpenCode: the guard plugin rewrites worker bash commands through - `rtk rewrite`). `workhorse token-savers`, installer `--token-savers` / `--rtk-bin`. The daemon's own - test run never goes through them. -- **`approvals.require_operator`**: MCP `approve_task` only records an approval request; a human - confirms with `sudo workhorse approve `, which sends a separate operator token (the daemon stores - its SHA-256). `workhorse operator-token init [--enable]`. + `rtk rewrite`, with a minimal environment and a 1 s timeout, falling back to the original command + if the rewrite fails or does not pass the guard; only the rtk binary itself is bound into the + sandbox). `workhorse token-savers`, installer `--token-savers` / `--rtk-bin`. The daemon's own test + run never goes through them. +- **`approvals.require_operator`**: MCP `approve_task` only records an approval request (with a + request id); a human confirms with `sudo workhorse approve `, which prints the pending request + and sends a separate operator token (the daemon stores its SHA-256) plus the displayed request id + (refused if the request changed). A task that parked stays gated until the operator answers, even if + it is closed or cancelled meanwhile; without the token, reject may only close the task. This gates + the parked-task flow; it is not a capability boundary (a supervisor can still delegate a new task + asking for the same thing). `workhorse operator-token init [--enable]`. - **Per-profile `stall_minutes`.** - **Audit log rotation** (`audit.max_mb`, `audit.keep`). - **Test-only stub backend** (`adapters/stub`), registered only when the daemon runs with @@ -50,14 +64,20 @@ act, and adds opt-in savers for the workers. Builds on 0.2.0. See [docs/token-sa ### Changed - The MCP shim returns compact JSON (9-18% fewer characters on typical responses). -- The full `task_result` no longer repeats top-level fields inside `handoff.context` (`get_handoff` - still returns the complete record). +- The full `task_result` view is unchanged from 0.2.0 apart from compact JSON (the complete handoff, + including `handoff.context`, is kept for compatibility; use `view: "brief"` for the short form). - `delegate_task`'s `next_step` and the MCP instructions point to `wait_task`; the delegation skill draft prefers `wait_task`, the brief view, presets/size, batches and automatic follow-ups. - The installer installs Kilo CLI and OpenCode from committed lockfiles (`scripts/pins/`, `npm ci`, integrity-checked) when the pinned version is requested, and falls back to `npm install -g` with a warning otherwise. -- CI runs the unit tests, then the stub end-to-end suite. +- The installer resolves the pinned CLI's executable from the package's own `package.json` `bin` field + (opencode-ai 1.18.32 ships `bin/opencode.exe`, which left a dangling link before) and fails if the + link target is missing. +- CI runs the unit tests, then the stub end-to-end suite, and a job that installs every + `scripts/pins/*` lockfile with `npm ci --ignore-scripts` and checks the link target exists. +- This branch is rebased onto the 0.2.0 review fixes (restart socket race, closed-on-parked semantics, + approval source). ## [0.2.0] - Unreleased diff --git a/README.md b/README.md index 763a1ee..2cae71d 100644 --- a/README.md +++ b/README.md @@ -1,5 +1,7 @@ # Grok Workhorse +[![CI](https://github.com/mrchatam/Grok-workhorse/actions/workflows/ci.yml/badge.svg)](https://github.com/mrchatam/Grok-workhorse/actions/workflows/ci.yml) + > Unofficial community project. Not affiliated with, endorsed by or sponsored by xAI. **Grok Workhorse lets a supervising AI agent (for example Grok Bot) hand coding tasks to sandboxed @@ -35,10 +37,27 @@ models). See [docs/token-savings.md](docs/token-savings.md). `workhorse stats` with a labelled estimate of supervisor tokens avoided, and opt-in worker savers (terse output, minimal-code bias, RTK for shell output). See [docs/token-savings.md](docs/token-savings.md). - **Operator-confirmed approvals** (optional): with `approvals.require_operator`, the supervisor's - approval only records a request and a human confirms it on the host with a separate operator token. + approval only records a request and a human confirms that exact request on the host with a separate + operator token. It gates the parked-task flow; it is not a capability boundary (see + [docs/handoff.md](docs/handoff.md#operator-confirmation-approvalsrequire_operator-v03)). - **Operations**: stall detection (15 min by default, per profile with `stall_minutes`), wall-clock timeouts, cancel, recovery after a daemon restart, retention sweeps, an append-only JSONL audit log (rotated by size), and `workhorse health` for daily checks. - **Credentials from the environment first** (daemon env, or the MCP connector env passed through the shim), with an optional secret-store fallback. Keys never appear in argv, logs or results. +## Measured savings + +All figures below are **estimates** from small samples on this repository; details, method and caveats +are in [docs/token-savings.md](docs/token-savings.md#benchmarks). + +| What | Estimate | How it was measured | +|---|---|---| +| Supervisor tokens read per task, succeeds first time (v0.2 polling flow vs v0.3 `wait_task` + brief) | ~3,060 -> ~360 (about -88%) | test-only stub backend on the calc fixture; responses counted with the `o200k_base` tokenizer as a proxy; assumes 5 status polls in the v0.2 flow | +| Same, tests fail once then fixed (v0.2 manual `continue_task` vs v0.3 `auto_fix_rounds: 1`) | ~6,900 -> ~400 (about -94%) | same method; the fix round costs worker tokens on the cheap profile instead | +| RTK on worker shell output (8 common commands) | about -35% overall (0% to -76% per command) | RTK v0.50.0 on this repository, `o200k_base` token counts of each command's output before/after the rewrite | +| Worker output with `terse` + `minimal_code` (`lite`) | about -10% output tokens | 3 A/B pairs on one real model through Kilo, provider-reported tokens; not statistically meaningful | + +`workhorse stats` / `usage_report` give a conservative running ESTIMATE for your own tasks (formula in +the same doc). + ## Architecture ```mermaid diff --git a/docs/architecture.md b/docs/architecture.md index 3d91494..a82848c 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -35,7 +35,7 @@ sequenceDiagram | `lib/config.mjs` | Config loading, defaults, auto-detection of backend binaries, bwrap and toolchain dirs. | | `lib/credentials.mjs` | Optional JSON secret-store fallback. | | `lib/audit.mjs` | Append-only JSONL audit log (values redacted, rotated by size). | -| `lib/views.mjs` | `brief` result view and handoff deduplication for the full view. | +| `lib/views.mjs` | `brief` result view. | | `lib/usage.mjs` | `usage_report` / `workhorse stats`: per-run tokens by profile and day, supervisor ESTIMATE. | | `lib/savers.mjs` | Opt-in worker token savers (instruction fragments, RTK settings). | | `lib/operator.mjs` | Operator token for `approvals.require_operator` (hash check, token file). | diff --git a/docs/configuration.md b/docs/configuration.md index db6308d..a9b69da 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -82,7 +82,7 @@ the routine follow-up work on cheap models. escalations together), plus optional `max_tokens` (input+output+reasoning of all runs, including the task's automatic review tasks) and `max_cost_usd` (needs `price_per_mtok`; also includes the reviews). The first cap reached stops the chain; the result records `auto.stopped_reason`. Budgets are checked - between runs, so one run can overshoot them. `0` is an explicit zero budget (no automatic follow-ups); + between runs, so one run can overshoot them. `0` is an explicit zero budget (no automatic follow-ups or reviews); omit the key or use `null` for no budget; negative or non-numeric values fail `workhorse validate`. `continue_task` restarts the trail and the run counters for its new round, but not the token/cost budgets. diff --git a/docs/token-savings.md b/docs/token-savings.md index 0ef70f6..2a821e9 100644 --- a/docs/token-savings.md +++ b/docs/token-savings.md @@ -59,7 +59,7 @@ Budget details: its automatic review tasks. - They are checked between runs, so a single run can overshoot them; the next automatic run is then not started. -- `0` means an explicit zero budget (no automatic follow-ups at all), not "unlimited". Leave the key +- `0` means an explicit zero budget (no automatic follow-ups or reviews at all), not "unlimited". Leave the key out (or `null`) for no budget. - `continue_task` starts a new automatic round: the trail and the fix-round / `max_auto_runs` counters restart, but the token and cost budgets keep counting the whole task. diff --git a/test/stub-e2e.test.mjs b/test/stub-e2e.test.mjs index 898161c..39e7c2d 100644 --- a/test/stub-e2e.test.mjs +++ b/test/stub-e2e.test.mjs @@ -344,6 +344,40 @@ stest("usage_report: tokens by profile/day and a labelled supervisor ESTIMATE", assert.deepEqual(one.by_profile.map((p) => p.profile).sort(), ["mid", "strong"].filter((x) => one.by_profile.some((p) => p.profile === x)).sort()) }) +stest("restart recovery: a review whose parent link was lost is re-linked; an orphan review is cancelled", async () => { + const d1 = await delegate({ task: task("review"), auto_review: "reviewer" }) + const d2 = await delegate({ task: task("review"), auto_review: "reviewer" }) + const r1 = await waitDone(d1.task_id) + const r2 = await waitDone(d2.task_id) + const c1 = r1.review.task_id + const c2 = r2.review.task_id + await stopDaemon(daemon) + const file = (id) => path.join(env.dataDir, "tasks", id, "task.json") + const edit = (id, fn) => { const j = JSON.parse(fs.readFileSync(file(id), "utf8")); fn(j); fs.writeFileSync(file(id), JSON.stringify(j)) } + let childCreated + edit(c1, (j) => { childCreated = j.created_at }) + // Crash after the review child was saved but before the parent recorded its id. + edit(d1.task_id, (j) => { + j.status = "reviewing" + j.review_pending = { profile: "reviewer", final_status: "completed", started_at: new Date(Date.parse(childCreated) - 1000).toISOString() } + delete j.result.review + delete j.auto_review_usage + }) + // A review child that never ran while its parent already finished. + edit(c2, (j) => { j.status = "queued"; j.result = null; j.handoff = null; j.finished_at = null }) + daemon = await startDaemon(env, DAEMON_ENV) + const p1 = await rpc("task_result", { task_id: d1.task_id }) + assert.equal(p1.status, "completed") + assert.equal(p1.review.task_id, c1) + assert.equal(p1.review.verdict, "request_changes") + let st + for (let i = 0; i < 50 && (st = await rpc("task_status", { task_id: c2 })).status !== "cancelled"; i++) await sleep(100) + assert.equal(st.status, "cancelled") + assert.equal((await rpc("task_status", { task_id: d2.task_id })).status, "completed") + const audit = fs.readFileSync(path.join(env.dataDir, "logs/audit.jsonl"), "utf8") + assert.match(audit, /orphan_auto_review_cancelled/) +}) + stest("restart: a killed daemon leaves an interrupted, retryable task that resumes", async () => { const d = await delegate({ task: task("slow") }) for (let i = 0; i < 50 && (await rpc("task_status", { task_id: d.task_id })).status !== "running"; i++) await sleep(100) diff --git a/test/token-savings.test.mjs b/test/token-savings.test.mjs index a9ae73e..a9e15e6 100644 --- a/test/token-savings.test.mjs +++ b/test/token-savings.test.mjs @@ -191,6 +191,17 @@ test("presets, routing, auto and escalate_to are validated; auto caps are hard", assert.match(probs, /routing.large: profile 'nope' is not defined/) assert.match(probs, /auto.review.profile 'ghost' is not defined/) assert.doesNotMatch(probs, /preset 'ok'/) + // Budgets: 0 is an explicit zero budget (kept, not "unlimited"); negative values are reported. + const prof = JSON.parse(fs.readFileSync(path.join(cfgDir, "profiles.json"), "utf8")) + fs.writeFileSync(path.join(cfgDir, "profiles.json"), JSON.stringify({ ...prof, auto: { max_tokens: 0, max_cost_usd: 0 } })) + assert.equal(C.profilesConfig().auto.max_tokens, 0) + assert.equal(C.profilesConfig().auto.max_cost_usd, 0) + assert.doesNotMatch(C.validateConfig().join("\n"), /auto.max_/) + fs.writeFileSync(path.join(cfgDir, "profiles.json"), JSON.stringify({ ...prof, auto: { max_tokens: -1, max_cost_usd: "5" } })) + assert.equal(C.profilesConfig().auto.max_tokens, null) + assert.match(C.validateConfig().join("\n"), /auto.max_tokens must be a number >= 0/) + assert.match(C.validateConfig().join("\n"), /auto.max_cost_usd must be a number >= 0/) + fs.writeFileSync(path.join(cfgDir, "profiles.json"), JSON.stringify(prof)) fs.writeFileSync(path.join(cfgDir, "daemon.json"), JSON.stringify({ approvals: { require_operator: true } })) assert.match(C.validateConfig().join("\n"), /require_operator is on but approvals.operator_token_sha256/) fs.rmSync(path.join(cfgDir, "daemon.json"))